From 9db7dcbac42675b6731eb34ff922629825a20e91 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:37:46 -0500 Subject: [PATCH 001/111] Integrate reviewed validator performance changes for combined testing --- .github/workflows/go_build.yml | 28 + cmd/mithril/configcmd/configcmd.go | 7 +- cmd/mithril/configcmd/configcmd_test.go | 24 + cmd/mithril/node/config_bool.go | 15 + cmd/mithril/node/config_bool_test.go | 38 + cmd/mithril/node/node.go | 222 +++--- cmd/mithril/node/sigverify_reporter.go | 3 +- cmd/mithril/node/vote_startup.go | 40 + cmd/mithril/node/vote_startup_test.go | 73 ++ cmd/repair-sim/main.go | 137 ++++ config.example.toml | 53 +- docs/alpenglow_branch_engine.md | 25 + docs/certificate-processing-evidence.md | 14 + docs/certpool-offlock.md | 78 ++ docs/certpool-point-reuse.md | 18 + docs/erasure_recovery_experiments.md | 130 ++++ docs/fec-producer-evidence.md | 18 + docs/leader-packing-evidence.md | 14 + docs/leader_block_packing.md | 98 +++ docs/out-of-order-entry-prefetch.md | 28 + docs/repair_sim.md | 153 ++++ docs/reserved-vote-history.md | 162 ++++ docs/rewards-unwind-retirement.md | 58 ++ docs/shred-spool-completion.md | 85 +++ docs/shred_retention_performance.md | 44 ++ docs/spool-completion-journal-evidence.md | 14 + docs/status-checkpoint-capture.md | 98 +++ docs/status-checkpoint-expiry-evidence.md | 19 + docs/streaming-preparation-evidence.md | 14 + docs/streaming_message_identities.md | 106 +++ docs/transaction-status-expiry.md | 61 ++ docs/transaction-status-publication.md | 73 ++ docs/transaction_sigverify_streaming.md | 159 ++++ docs/turbine-relay-buffers.md | 60 ++ docs/vote-delivery-persistence-evidence.md | 14 + docs/votor-peer-isolation.md | 99 +++ go.mod | 2 +- go.sum | 4 +- pkg/accounts/mem_accounts.go | 15 +- pkg/accounts/mem_accounts_test.go | 15 + pkg/accounts/overlay_test.go | 39 + pkg/accounts/working_set.go | 32 +- pkg/alpenglow/broadcaster.go | 113 ++- pkg/alpenglow/certpool.go | 542 +++++++++---- pkg/alpenglow/certpool_bench_test.go | 160 ++++ pkg/alpenglow/certpool_concurrency_test.go | 252 +++++++ pkg/alpenglow/certpool_points_test.go | 102 +++ pkg/alpenglow/certpool_progress_test.go | 112 +++ pkg/alpenglow/certpool_test.go | 48 ++ pkg/alpenglow/observer.go | 87 ++- pkg/alpenglow/observer_bench_test.go | 57 ++ pkg/alpenglow/observer_stats_test.go | 127 ++++ pkg/alpenglow/peer_sender.go | 181 +++++ pkg/alpenglow/peer_sender_recovery_test.go | 136 ++++ pkg/alpenglow/peer_sender_test.go | 245 ++++++ pkg/alpenglow/testdata/README.md | 9 + .../testdata/agave_votor_certificate.der | Bin 0 -> 249 bytes pkg/alpenglow/tls_identity.go | 51 +- pkg/alpenglow/tls_identity_test.go | 46 ++ pkg/alpenglow/vote_history.go | 113 ++- pkg/alpenglow/vote_history_snapshot_test.go | 49 ++ pkg/alpenglow/vote_reservation.go | 104 +++ pkg/alpenglow/vote_reservation_test.go | 71 ++ pkg/block/block.go | 17 +- pkg/block/block_test.go | 7 +- pkg/block/verified_message_identity.go | 37 + pkg/block/verified_message_identity_test.go | 42 ++ pkg/blockprod/bank.go | 52 +- pkg/blockprod/bank_test.go | 13 + pkg/blockprod/entry.go | 73 +- pkg/blockprod/entry_bench_test.go | 75 ++ pkg/blockprod/entry_test.go | 100 +++ pkg/blockprod/leader.go | 43 +- .../leader_completion_reserve_test.go | 32 + pkg/blockprod/leader_processing_test.go | 81 ++ .../leader_signing_reservation_test.go | 17 + pkg/blockprod/leader_throughput_bench_test.go | 117 +++ pkg/blockprod/prepared_transaction_test.go | 142 ++++ pkg/blockprod/producer_block_bench_test.go | 66 ++ pkg/blockprod/producer_branch_bench_test.go | 253 +++++++ pkg/blockprod/producer_maxsize_bench_test.go | 153 ++++ pkg/blockprod/readonly_block_bench_test.go | 189 +++++ .../readonly_block_prepared_bench_test.go | 45 ++ pkg/blockprod/readonly_load_bench_test.go | 95 +++ pkg/blockprod/scheduler/buffer.go | 127 ++-- pkg/blockprod/scheduler/buffer_bench_test.go | 45 ++ pkg/blockprod/scheduler/buffer_order_test.go | 106 +++ .../scheduler/buffer_retention_test.go | 158 ++++ pkg/blockprod/scheduler/config_test.go | 28 + pkg/blockprod/scheduler/scheduler.go | 56 +- pkg/blockprod/scheduler/scheduler_test.go | 20 +- pkg/config/config.go | 4 + pkg/consensus/engine.go | 40 +- pkg/consensus/vote_history_writer.go | 125 +++ pkg/consensus/vote_history_writer_test.go | 201 +++++ pkg/consensus/vote_reservation.go | 296 ++++++++ pkg/consensus/vote_reservation_test.go | 330 ++++++++ pkg/consensus/voter.go | 331 ++++++-- pkg/consensus/voter_finality_ordering_test.go | 192 +++++ pkg/consensus/voter_wait_slot_test.go | 91 +++ pkg/costmodel/entry_bytes.go | 7 +- pkg/costmodel/limits.go | 54 +- pkg/costmodel/limits_test.go | 13 + pkg/costmodel/slot_limits_test.go | 88 +++ pkg/costmodel/transaction_cost.go | 15 +- pkg/merkletree/merkletree.go | 44 +- pkg/merkletree/root_test.go | 60 ++ pkg/metrics/metrics.go | 49 +- pkg/replay/async_checkpoint_capture_test.go | 172 +++++ pkg/replay/async_promotion_test.go | 41 +- pkg/replay/block.go | 34 +- pkg/replay/chain_state.go | 11 + pkg/replay/promotion.go | 105 ++- pkg/replay/rewards_retirement.go | 49 ++ pkg/replay/rewards_retirement_test.go | 119 +++ pkg/replay/transaction_preparation.go | 133 ++++ pkg/replay/transaction_processing_pure.go | 134 ++-- pkg/replay/transaction_status_cache.go | 330 ++++++-- .../transaction_status_capture_bench_test.go | 113 +++ pkg/replay/transaction_status_capture_test.go | 299 ++++++++ pkg/replay/transaction_status_expiry_test.go | 129 ++++ ...ansaction_status_overlap_benchmark_test.go | 110 +++ .../transaction_status_plan_binding_test.go | 3 +- .../transaction_status_prepared_test.go | 18 +- pkg/replay/transaction_status_publication.go | 83 ++ ...ction_status_publication_benchmark_test.go | 166 ++++ .../transaction_status_publication_test.go | 133 ++++ pkg/replay/transaction_status_validation.go | 37 + .../transaction_status_validation_test.go | 151 ++++ pkg/sbpf/pooling_test.go | 75 ++ pkg/sealevel/execution_ctx.go | 16 +- pkg/sealevel/vote_deque_ownership_test.go | 29 + pkg/sealevel/vote_program.go | 9 +- pkg/sigverify/config_policy_test.go | 63 ++ pkg/sigverify/sigverify.go | 70 +- pkg/statsd/statsd.go | 48 +- pkg/statsd/statsd_test.go | 9 + pkg/tpu/txfixture/readonly_pair.go | 42 ++ pkg/tpu/txfixture/readonly_pair_test.go | 68 ++ pkg/turbine/assembler.go | 268 +++++-- pkg/turbine/broadcast.go | 48 +- pkg/turbine/broadcast_test.go | 64 ++ pkg/turbine/cancellation_regression_test.go | 67 +- pkg/turbine/cluster_nodes.go | 12 +- pkg/turbine/completion_order_root_test.go | 228 ++++++ pkg/turbine/component_shredder.go | 61 +- pkg/turbine/component_test.go | 30 + pkg/turbine/entries.go | 174 ++++- pkg/turbine/entries_direct_compare_test.go | 100 +++ pkg/turbine/entries_prefetch_test.go | 146 ++++ pkg/turbine/entry_batch_index.go | 162 ++++ pkg/turbine/entry_batch_index_test.go | 106 +++ pkg/turbine/entry_batch_transactions_test.go | 37 + pkg/turbine/entry_hash.go | 9 +- pkg/turbine/entry_hash_bench_test.go | 24 + pkg/turbine/entry_identity_recovery_test.go | 131 ++++ pkg/turbine/entry_pipeline_trace.go | 212 ++++++ pkg/turbine/entry_pipeline_trace_test.go | 68 ++ pkg/turbine/entry_prefetch.go | 428 +++++++++++ pkg/turbine/entry_prefetch_benchmark_test.go | 288 +++++++ pkg/turbine/entry_prefetch_bounds_test.go | 50 ++ pkg/turbine/entry_prefetch_test.go | 528 +++++++++++++ pkg/turbine/fec_root_cache.go | 56 ++ pkg/turbine/generate.go | 229 ++++-- pkg/turbine/generate_bench_test.go | 242 ++++++ pkg/turbine/generate_test.go | 49 +- .../generated_component_boundary_test.go | 47 ++ pkg/turbine/internal/rsrecover/doc.go | 5 + pkg/turbine/internal/rsrecover/recover.go | 599 +++++++++++++++ .../internal/rsrecover/recover_bench_test.go | 222 ++++++ .../internal/rsrecover/recover_test.go | 375 +++++++++ pkg/turbine/receiver.go | 37 +- .../receiver_prefetch_lifecycle_test.go | 116 +++ pkg/turbine/recover_one_data_test.go | 113 +++ pkg/turbine/repairsim/ledger.go | 257 +++++++ pkg/turbine/repairsim/sim.go | 713 ++++++++++++++++++ pkg/turbine/repairsim/sim_test.go | 254 +++++++ pkg/turbine/retention_sweep_test.go | 137 ++++ pkg/turbine/retransmit.go | 91 ++- .../retransmit_allocation_bench_test.go | 89 +++ pkg/turbine/retransmit_allocation_test.go | 267 +++++++ pkg/turbine/shred.go | 5 +- pkg/turbine/shredspool.go | 89 ++- pkg/turbine/shredspool_benchmark_test.go | 37 + pkg/turbine/shredspool_journal.go | 102 +++ pkg/turbine/shredspool_journal_test.go | 205 +++++ pkg/turbine/sigcache.go | 46 +- .../streaming_message_identity_test.go | 104 +++ pkg/turbine/transaction_job_groups_test.go | 118 +++ pkg/turbine/transaction_verifier.go | 450 ++++++++--- pkg/turbine/transaction_verifier_admission.go | 76 ++ ...ction_verifier_admission_benchmark_test.go | 118 +++ .../transaction_verifier_admission_test.go | 172 +++++ ...ransaction_verifier_flow_benchmark_test.go | 479 ++++++++++++ pkg/turbine/transaction_verifier_test.go | 304 +++++++- pkg/txstatus/message_identity.go | 9 +- pkg/txverify/message_identity.go | 25 + pkg/txverify/message_identity_test.go | 70 ++ pkg/txverify/txverify.go | 32 + 199 files changed, 20117 insertions(+), 1385 deletions(-) create mode 100644 cmd/mithril/configcmd/configcmd_test.go create mode 100644 cmd/mithril/node/config_bool.go create mode 100644 cmd/mithril/node/config_bool_test.go create mode 100644 cmd/mithril/node/vote_startup.go create mode 100644 cmd/mithril/node/vote_startup_test.go create mode 100644 cmd/repair-sim/main.go create mode 100644 docs/certificate-processing-evidence.md create mode 100644 docs/certpool-offlock.md create mode 100644 docs/certpool-point-reuse.md create mode 100644 docs/erasure_recovery_experiments.md create mode 100644 docs/fec-producer-evidence.md create mode 100644 docs/leader-packing-evidence.md create mode 100644 docs/leader_block_packing.md create mode 100644 docs/out-of-order-entry-prefetch.md create mode 100644 docs/repair_sim.md create mode 100644 docs/reserved-vote-history.md create mode 100644 docs/rewards-unwind-retirement.md create mode 100644 docs/shred-spool-completion.md create mode 100644 docs/shred_retention_performance.md create mode 100644 docs/spool-completion-journal-evidence.md create mode 100644 docs/status-checkpoint-capture.md create mode 100644 docs/status-checkpoint-expiry-evidence.md create mode 100644 docs/streaming-preparation-evidence.md create mode 100644 docs/streaming_message_identities.md create mode 100644 docs/transaction-status-expiry.md create mode 100644 docs/transaction-status-publication.md create mode 100644 docs/transaction_sigverify_streaming.md create mode 100644 docs/turbine-relay-buffers.md create mode 100644 docs/vote-delivery-persistence-evidence.md create mode 100644 docs/votor-peer-isolation.md create mode 100644 pkg/alpenglow/certpool_bench_test.go create mode 100644 pkg/alpenglow/certpool_concurrency_test.go create mode 100644 pkg/alpenglow/certpool_points_test.go create mode 100644 pkg/alpenglow/certpool_progress_test.go create mode 100644 pkg/alpenglow/observer_bench_test.go create mode 100644 pkg/alpenglow/observer_stats_test.go create mode 100644 pkg/alpenglow/peer_sender.go create mode 100644 pkg/alpenglow/peer_sender_recovery_test.go create mode 100644 pkg/alpenglow/peer_sender_test.go create mode 100644 pkg/alpenglow/testdata/agave_votor_certificate.der create mode 100644 pkg/alpenglow/vote_history_snapshot_test.go create mode 100644 pkg/alpenglow/vote_reservation.go create mode 100644 pkg/alpenglow/vote_reservation_test.go create mode 100644 pkg/block/verified_message_identity.go create mode 100644 pkg/block/verified_message_identity_test.go create mode 100644 pkg/blockprod/entry_bench_test.go create mode 100644 pkg/blockprod/leader_completion_reserve_test.go create mode 100644 pkg/blockprod/leader_processing_test.go create mode 100644 pkg/blockprod/leader_signing_reservation_test.go create mode 100644 pkg/blockprod/leader_throughput_bench_test.go create mode 100644 pkg/blockprod/prepared_transaction_test.go create mode 100644 pkg/blockprod/producer_block_bench_test.go create mode 100644 pkg/blockprod/producer_branch_bench_test.go create mode 100644 pkg/blockprod/producer_maxsize_bench_test.go create mode 100644 pkg/blockprod/readonly_block_bench_test.go create mode 100644 pkg/blockprod/readonly_block_prepared_bench_test.go create mode 100644 pkg/blockprod/readonly_load_bench_test.go create mode 100644 pkg/blockprod/scheduler/buffer_bench_test.go create mode 100644 pkg/blockprod/scheduler/buffer_order_test.go create mode 100644 pkg/blockprod/scheduler/buffer_retention_test.go create mode 100644 pkg/blockprod/scheduler/config_test.go create mode 100644 pkg/consensus/vote_history_writer.go create mode 100644 pkg/consensus/vote_history_writer_test.go create mode 100644 pkg/consensus/vote_reservation.go create mode 100644 pkg/consensus/vote_reservation_test.go create mode 100644 pkg/consensus/voter_finality_ordering_test.go create mode 100644 pkg/consensus/voter_wait_slot_test.go create mode 100644 pkg/costmodel/limits_test.go create mode 100644 pkg/costmodel/slot_limits_test.go create mode 100644 pkg/merkletree/root_test.go create mode 100644 pkg/replay/async_checkpoint_capture_test.go create mode 100644 pkg/replay/rewards_retirement.go create mode 100644 pkg/replay/rewards_retirement_test.go create mode 100644 pkg/replay/transaction_preparation.go create mode 100644 pkg/replay/transaction_status_capture_bench_test.go create mode 100644 pkg/replay/transaction_status_capture_test.go create mode 100644 pkg/replay/transaction_status_expiry_test.go create mode 100644 pkg/replay/transaction_status_overlap_benchmark_test.go create mode 100644 pkg/replay/transaction_status_publication.go create mode 100644 pkg/replay/transaction_status_publication_benchmark_test.go create mode 100644 pkg/replay/transaction_status_publication_test.go create mode 100644 pkg/replay/transaction_status_validation.go create mode 100644 pkg/replay/transaction_status_validation_test.go create mode 100644 pkg/sbpf/pooling_test.go create mode 100644 pkg/sealevel/vote_deque_ownership_test.go create mode 100644 pkg/sigverify/config_policy_test.go create mode 100644 pkg/tpu/txfixture/readonly_pair.go create mode 100644 pkg/tpu/txfixture/readonly_pair_test.go create mode 100644 pkg/turbine/completion_order_root_test.go create mode 100644 pkg/turbine/entries_direct_compare_test.go create mode 100644 pkg/turbine/entries_prefetch_test.go create mode 100644 pkg/turbine/entry_batch_index.go create mode 100644 pkg/turbine/entry_batch_index_test.go create mode 100644 pkg/turbine/entry_batch_transactions_test.go create mode 100644 pkg/turbine/entry_hash_bench_test.go create mode 100644 pkg/turbine/entry_identity_recovery_test.go create mode 100644 pkg/turbine/entry_pipeline_trace.go create mode 100644 pkg/turbine/entry_pipeline_trace_test.go create mode 100644 pkg/turbine/entry_prefetch.go create mode 100644 pkg/turbine/entry_prefetch_benchmark_test.go create mode 100644 pkg/turbine/entry_prefetch_bounds_test.go create mode 100644 pkg/turbine/entry_prefetch_test.go create mode 100644 pkg/turbine/fec_root_cache.go create mode 100644 pkg/turbine/generate_bench_test.go create mode 100644 pkg/turbine/generated_component_boundary_test.go create mode 100644 pkg/turbine/internal/rsrecover/doc.go create mode 100644 pkg/turbine/internal/rsrecover/recover.go create mode 100644 pkg/turbine/internal/rsrecover/recover_bench_test.go create mode 100644 pkg/turbine/internal/rsrecover/recover_test.go create mode 100644 pkg/turbine/receiver_prefetch_lifecycle_test.go create mode 100644 pkg/turbine/recover_one_data_test.go create mode 100644 pkg/turbine/repairsim/ledger.go create mode 100644 pkg/turbine/repairsim/sim.go create mode 100644 pkg/turbine/repairsim/sim_test.go create mode 100644 pkg/turbine/retention_sweep_test.go create mode 100644 pkg/turbine/retransmit_allocation_bench_test.go create mode 100644 pkg/turbine/retransmit_allocation_test.go create mode 100644 pkg/turbine/shredspool_benchmark_test.go create mode 100644 pkg/turbine/shredspool_journal.go create mode 100644 pkg/turbine/shredspool_journal_test.go create mode 100644 pkg/turbine/streaming_message_identity_test.go create mode 100644 pkg/turbine/transaction_job_groups_test.go create mode 100644 pkg/turbine/transaction_verifier_admission.go create mode 100644 pkg/turbine/transaction_verifier_admission_benchmark_test.go create mode 100644 pkg/turbine/transaction_verifier_admission_test.go create mode 100644 pkg/turbine/transaction_verifier_flow_benchmark_test.go create mode 100644 pkg/txverify/message_identity.go create mode 100644 pkg/txverify/message_identity_test.go diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index 50383467f..4fa7306c9 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -17,3 +17,31 @@ jobs: - name: Build run: go build -v ./cmd/mithril + + regression-tests: + runs-on: ubuntu-latest + timeout-minutes: 20 + permissions: + contents: read + env: + GOMAXPROCS: "2" + steps: + - uses: actions/checkout@v3 + + - name: Setup Go + uses: actions/setup-go@v4 + with: + go-version: 1.26.4 + + - name: Voting, checkpoint, streaming and scheduler race regressions + # Run the complete affected package suites, including subprocess crash + # recovery and cancellation tests. The independent sealevel suite has + # known base-branch failures documented in the validation report. + run: >- + go test -race -p 2 -count=1 + ./pkg/alpenglow ./pkg/consensus ./pkg/replay + ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... + ./cmd/mithril/node ./cmd/mithril/configcmd + + - name: Vote-program deque ownership race regression + run: go test -race -count=1 ./pkg/sealevel -run '^TestProcessNewVoteStateOwnsRetainedDeque$' diff --git a/cmd/mithril/configcmd/configcmd.go b/cmd/mithril/configcmd/configcmd.go index 903ff68ca..3236b9fcc 100644 --- a/cmd/mithril/configcmd/configcmd.go +++ b/cmd/mithril/configcmd/configcmd.go @@ -261,7 +261,12 @@ max_rps = 8 # Verifier's own RPC budget (never shares the block-fe # ── Replay tuning ──────────────────────────────────────────────────────── [tuning] txpar = 24 # Validator auto-defaults to 2x CPU cores only when unset; explicit 0 = sequential -sigverify_backend = "auto" # auto|r51|generic|stdlib; stdlib uses Go's crypto/ed25519 impl after strict checks. + +[sigverify] +backend = "auto" # auto|r51|generic|stdlib +workers = 0 # 0 = min(2, GOMAXPROCS); explicit value overrides the shared transaction pool +batch_target = 8 # 4 or 8 signature lanes; available work runs immediately +disable_shred_overlap = false # Diagnostic fallback: verify after complete block assembly # ── Mithril's RPC server ───────────────────────────────────────────────── [rpc] diff --git a/cmd/mithril/configcmd/configcmd_test.go b/cmd/mithril/configcmd/configcmd_test.go new file mode 100644 index 000000000..2d62bb8b5 --- /dev/null +++ b/cmd/mithril/configcmd/configcmd_test.go @@ -0,0 +1,24 @@ +package configcmd + +import ( + "strings" + "testing" + + "github.com/spf13/viper" + "github.com/stretchr/testify/require" +) + +func TestStarterConfigSignatureVerification(t *testing.T) { + for _, validator := range []bool{false, true} { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(generateStarterConfig(validator)))) + require.Equal(t, "auto", v.GetString("sigverify.backend")) + require.True(t, v.IsSet("sigverify.workers")) + require.Zero(t, v.GetInt("sigverify.workers")) + require.Equal(t, 8, v.GetInt("sigverify.batch_target")) + require.True(t, v.IsSet("sigverify.disable_shred_overlap")) + require.False(t, v.GetBool("sigverify.disable_shred_overlap")) + require.False(t, v.IsSet("tuning.sigverify_backend")) + } +} diff --git a/cmd/mithril/node/config_bool.go b/cmd/mithril/node/config_bool.go new file mode 100644 index 000000000..e779efb61 --- /dev/null +++ b/cmd/mithril/node/config_bool.go @@ -0,0 +1,15 @@ +package node + +import "github.com/spf13/pflag" + +// resolveBoolOption preserves explicit false at either precedence level. +// Defaults use DefValue, not a flag value potentially left by a previous run. +func resolveBoolOption(flag *pflag.Flag, configured bool, configuredValue bool) bool { + if flag != nil && flag.Changed { + return flag.Value.String() == "true" + } + if configured { + return configuredValue + } + return flag != nil && flag.DefValue == "true" +} diff --git a/cmd/mithril/node/config_bool_test.go b/cmd/mithril/node/config_bool_test.go new file mode 100644 index 000000000..ab23313d0 --- /dev/null +++ b/cmd/mithril/node/config_bool_test.go @@ -0,0 +1,38 @@ +package node + +import ( + "github.com/spf13/pflag" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" + "strings" + "testing" +) + +func TestResolveBoolOptionPrecedence(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + defaultValue, want bool + }{ + {"omitted true default", "", "", true, true}, + {"omitted false default", "", "", false, false}, + {"TOML false", "enabled=false", "", true, false}, + {"TOML true", "enabled=true", "", false, true}, + {"CLI false beats TOML true", "enabled=true", "false", true, false}, + {"CLI true beats TOML false", "enabled=false", "true", false, true}, + {"CLI false without TOML", "", "false", true, false}, + } { + t.Run(tc.name, func(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(tc.toml))) + flags := pflag.NewFlagSet("test", pflag.ContinueOnError) + flags.Bool("enabled", tc.defaultValue, "") + if tc.cli != "" { + require.NoError(t, flags.Set("enabled", tc.cli)) + } + require.Equal(t, tc.want, resolveBoolOption(flags.Lookup("enabled"), v.IsSet("enabled"), v.GetBool("enabled"))) + }) + } + require.False(t, resolveBoolOption(nil, false, false)) + require.True(t, resolveBoolOption(nil, true, true)) +} diff --git a/cmd/mithril/node/node.go b/cmd/mithril/node/node.go index c452d494e..57c8f76a0 100644 --- a/cmd/mithril/node/node.go +++ b/cmd/mithril/node/node.go @@ -74,35 +74,40 @@ var ( }, } - bootstrapMode string // "auto", "snapshot", "new-snapshot", "new-incremental", or "accountsdb" - snapshotArchivePath string - incrementalSnapshotFilename string - accountsPath string - scratchDirectory string - rpcEndpoints []string - cluster string // "alpenglow", "mainnet-beta", "testnet", or "devnet" - legacyGenesisHash string // explicit lineage for pre-binding AccountsDB/ledger artifacts - blockSource string // "turbine", "rpc", or "lightbringer" - lightbringerEndpoint string - repairCatchupMaxGapSlots int // Resume gaps up to this fill via turbine repair instead of RPC (0 = off) - repairMaxRequestsPerSecond int // Repair request-rate ceiling override (0 = adaptive default) - blockRPCFallback bool // Allow RPC block fetch when > repairCatchupMaxGapSlots behind (default false: shreds only) - blockMaxRPS int // Rate limit for block fetching - blockMaxInflight int // Max concurrent block fetch workers - blockTipPollIntervalMs int // Tip poll interval in milliseconds - blockTipSafetyMargin int // Don't fetch within N slots of tip - consensusModeFlag string // raw --consensus-mode value (cobra binding) - consensusMode string // resolved: "verifying" (default) or "validator" - alpenglowObserverBindAddr string - alpenglowMaxMessageBytes int64 - alpenglowBLSDST string - validatorIdentityKeypair string - validatorVoteAccountKeypair string - validatorAuthorizedVoterKeypair string - validatorWithdrawerKeypair string - validatorTPUQUICBind string - validatorAdvertisedIP string - validatorSigverifyWorkers int + bootstrapMode string // "auto", "snapshot", "new-snapshot", "new-incremental", or "accountsdb" + snapshotArchivePath string + incrementalSnapshotFilename string + accountsPath string + scratchDirectory string + rpcEndpoints []string + cluster string // "alpenglow", "mainnet-beta", "testnet", or "devnet" + legacyGenesisHash string // explicit lineage for pre-binding AccountsDB/ledger artifacts + blockSource string // "turbine", "rpc", or "lightbringer" + lightbringerEndpoint string + repairCatchupMaxGapSlots int // Resume gaps up to this fill via turbine repair instead of RPC (0 = off) + repairMaxRequestsPerSecond int // Repair request-rate ceiling override (0 = adaptive default) + blockRPCFallback bool // Allow RPC block fetch when > repairCatchupMaxGapSlots behind (default false: shreds only) + blockMaxRPS int // Rate limit for block fetching + blockMaxInflight int // Max concurrent block fetch workers + blockTipPollIntervalMs int // Tip poll interval in milliseconds + blockTipSafetyMargin int // Don't fetch within N slots of tip + consensusModeFlag string // raw --consensus-mode value (cobra binding) + consensusMode string // resolved: "verifying" (default) or "validator" + alpenglowObserverBindAddr string + alpenglowMaxMessageBytes int64 + alpenglowBLSDST string + validatorIdentityKeypair string + validatorVoteAccountKeypair string + validatorAuthorizedVoterKeypair string + validatorWithdrawerKeypair string + validatorTPUQUICBind string + validatorAdvertisedIP string + validatorSigverifyWorkers int + validatorWaitToVoteSlot uint64 + validatorReservedHistory bool + validatorInitializeReservation bool + validatorCompletionReserveMs int + validatorMaxBufferedTransactions int // Mode thresholds blockNearTipThreshold int // Enter near-tip when gap <= this @@ -547,6 +552,11 @@ func init() { Run.Flags().StringVar(&validatorTPUQUICBind, "tpu-quic-bind-addr", "", "Validator TPU QUIC listen address (default 0.0.0.0:8004)") Run.Flags().StringVar(&validatorAdvertisedIP, "validator-advertised-ip", "", "Public IP advertised for validator TPU QUIC") Run.Flags().IntVar(&validatorSigverifyWorkers, "tpu-sigverify-workers", 0, "TPU signature verification workers (0 = GOMAXPROCS)") + Run.Flags().BoolVar(&validatorReservedHistory, "reserved-vote-history", false, "Use durable signing reservations with unsynchronized per-vote history writes") + Run.Flags().BoolVar(&validatorInitializeReservation, "initialize-vote-reservation", false, "Enroll complete synchronous vote history in reserved mode (one-time migration)") + Run.Flags().Uint64Var(&validatorWaitToVoteSlot, "wait-to-vote-slot", 0, "Do not cast new votes below this slot; the automatic startup cutoff still applies (0 = automatic only)") + Run.Flags().IntVar(&validatorCompletionReserveMs, "leader-completion-reserve-ms", 0, "Time reserved for leader finalization and broadcast (0 = 75ms default; tune from measured completion times)") + Run.Flags().IntVar(&validatorMaxBufferedTransactions, "tpu-max-buffered-transactions", 0, "Maximum queued TPU transactions (0 = 131072 default)") // [tuning] section flags Run.Flags().Uint64Var(¶mArenaSizeMB, "param-arena-size-mb", 512, "Size in MB for serialized parameter arena (0 to disable)") @@ -560,6 +570,12 @@ func init() { Run.Flags().StringVar(&snapshot.SnapshotIndexTempDir, "snapshot-index-temp-dir", "", "Optional directory for snapshot index shard logs/SST staging") Run.Flags().StringVar(&sigverify.Cfg.Backend, "sigverify-backend", sigverify.Defaults().Backend, "ed25519 verification backend: auto|r51|generic|stdlib") + Run.Flags().IntVar(&sigverify.Cfg.Workers, "sigverify-workers", 0, + "Turbine transaction signature verification workers (0 = min(2, GOMAXPROCS))") + Run.Flags().IntVar(&sigverify.Cfg.BatchTarget, "sigverify-batch-target", sigverify.Defaults().BatchTarget, + "Turbine transaction signature batch target: 4 or 8 (available short batches run immediately)") + Run.Flags().BoolVar(&sigverify.Cfg.DisableShredOverlap, "sigverify-disable-shred-overlap", false, + "Defer Turbine transaction decoding and signature verification until all block shreds arrive") Run.Flags().BoolVar(&sbpf.UsePool, "use-pool", true, "Disable to allocate fresh slices") Run.Flags().IntVar(&accountsdb.StoreAccountsWorkers, "store-accounts-workers", 128, "Number of workers to write account updates") Run.Flags().IntVar(&accountsdb.ProgramCacheMaxMB, "program-cache-max-mb", accountsdb.DefaultProgramCacheMaxMB, "Maximum approximate SBPF program cache size in MiB") @@ -639,6 +655,11 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { if err := config.InitConfig(); err != nil { return err } + if slot, err := configuredWaitToVoteSlot(cmd); err != nil { + return err + } else { + validatorWaitToVoteSlot = slot + } // Check if a CLI flag was explicitly set by the user flagChanged := func(name string) bool { @@ -721,14 +742,9 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { return 0 } - // Helper to get bool: CLI flag if explicitly set, otherwise TOML config + // Match numeric options: explicit CLI, configured value, then flag default. getBool := func(cliKey, tomlKey string) bool { - if flagChanged(cliKey) { - if f := cmd.Flags().Lookup(cliKey); f != nil { - return f.Value.String() == "true" - } - } - return config.GetBool(tomlKey) + return resolveBoolOption(cmd.Flags().Lookup(cliKey), config.IsSet(tomlKey), config.GetBool(tomlKey)) } // Helper to get string slice: CLI flag if explicitly set, otherwise TOML config @@ -863,6 +879,14 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { } validatorAdvertisedIP = getString("validator-advertised-ip", "validator.advertised_ip") validatorSigverifyWorkers = getInt("tpu-sigverify-workers", "validator.tpu_sigverify_workers") + validatorCompletionReserveMs = getInt("leader-completion-reserve-ms", "validator.block_completion_reserve_ms") + validatorMaxBufferedTransactions = getInt("tpu-max-buffered-transactions", "validator.tpu_max_buffered_transactions") + if validatorMaxBufferedTransactions < 0 { + return fmt.Errorf("TPU maximum buffered transactions must be nonnegative") + } + if validatorCompletionReserveMs < 0 || validatorCompletionReserveMs >= int(blockprod.AlpenglowSlotDuration/time.Millisecond) { + return fmt.Errorf("leader completion reserve must be 0 (default) or between 1 and 199 milliseconds") + } // [block] section blockSource = getString("block-source", "block.source") @@ -1127,10 +1151,17 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { // later: narya pins its backend on first use, and selecting it explicitly // doubles as a startup health check, so a machine that cannot run the // requested backend fails now instead of at the first block. - sigverify.Cfg.Backend = getString("sigverify-backend", "tuning.sigverify_backend") + backendKey := "sigverify.backend" + if !config.IsSet(backendKey) { + backendKey = "tuning.sigverify_backend" // older configuration files + } + sigverify.Cfg.Backend = getString("sigverify-backend", backendKey) + sigverify.Cfg.Workers = getInt("sigverify-workers", "sigverify.workers") + sigverify.Cfg.BatchTarget = getInt("sigverify-batch-target", "sigverify.batch_target") + sigverify.Cfg.DisableShredOverlap = getBool("sigverify-disable-shred-overlap", "sigverify.disable_shred_overlap") resolved, err := sigverify.Configure(sigverify.Cfg) if err != nil { - return fmt.Errorf("tuning.sigverify_backend: %w", err) + return fmt.Errorf("signature verification configuration: %w", err) } resolvedSigverifyBackend = resolved sbpf.UsePool = getBool("use-pool", "tuning.use_pool") @@ -2642,52 +2673,9 @@ postBootstrap: } global.SeedWallClockSlot(wallClockSeed) startupWallSlot := global.WallClockSlot() - waitToVoteSlot := startupWallSlot - startupWallSlot%alpenglow.LeaderWindowSlots - if waitToVoteSlot <= math.MaxUint64-2*alpenglow.LeaderWindowSlots { - waitToVoteSlot += 2 * alpenglow.LeaderWindowSlots - } else { - waitToVoteSlot = math.MaxUint64 - } - mlog.Log.Infof("ALPENGLOW voting startup watermark: wall_clock=%d wait_to_vote=%d", startupWallSlot, waitToVoteSlot) + waitToVoteSlot := effectiveWaitToVoteSlot(startupWallSlot, validatorWaitToVoteSlot) + mlog.Log.Infof("ALPENGLOW voting startup watermark: wall_clock=%d configured_wait_to_vote=%d wait_to_vote=%d", startupWallSlot, validatorWaitToVoteSlot, waitToVoteSlot) - identityPubkey := solana.PrivateKey(validatorIdentity).PublicKey() - if err := consensusEngine.EnableVoting(consensusengine.VotingConfig{ - Identity: validatorIdentity, - AuthorizedVoter: validatorAuthorizedVoter, - VoteAccount: validatorVoteAccount, - HistoryDir: blockstorePath, - EpochForSlot: epochSchedule.GetEpoch, - SlotDuration: blockprod.AlpenglowSlotDuration, - WaitToVoteSlot: waitToVoteSlot, - ReadyToVote: func(slot uint64) bool { - wallSlot := global.WallClockSlot() - if liveSlot, ok := consensusEngine.AlpenglowLiveSlot(); ok { - wallSlot = liveSlot - } - return slot >= wallSlot || wallSlot-slot <= alpenglow.LeaderWindowSlots - }, - Peers: func(validators []alpenglow.ValidatorStake) []alpenglow.VotorPeer { - peers := make([]alpenglow.VotorPeer, 0, len(validators)) - seen := make(map[solana.PublicKey]struct{}, len(validators)) - for _, validator := range validators { - if validator.Stake == 0 || validator.NodePubkey == identityPubkey { - continue - } - addr, ok := sharedGossip.LookupAlpenglow(validator.NodePubkey) - if !ok { - continue - } - if _, duplicate := seen[validator.NodePubkey]; duplicate { - continue - } - seen[validator.NodePubkey] = struct{}{} - peers = append(peers, alpenglow.VotorPeer{Identity: validator.NodePubkey, Addr: addr}) - } - return peers - }, - }); err != nil { - klog.Fatalf("enable Alpenglow voting: %v", err) - } broadcaster, err := turbine.NewTurbineBroadcaster(turbine.TurbineBroadcasterConfig{ Self: solana.PrivateKey(validatorIdentity).PublicKey(), Peers: sharedGossip, @@ -2705,7 +2693,9 @@ postBootstrap: defer broadcaster.Close() controller := blockprod.NewController() - topicSink := scheduler.New(controller) + topicSink := scheduler.NewWithConfig(controller, scheduler.Config{ + FeatureSource: replay.ChainTipFeatures, MaxBufferedTransactions: validatorMaxBufferedTransactions, + }) topicSink.Start(ctx) defer topicSink.Stop() tpuCfg := tpu.DefaultConfig() @@ -2744,6 +2734,50 @@ postBootstrap: mlog.Log.Warnf("validator gossip TPU advertisement: %v", err) } + // Bind and validate local transports before consuming the durable clean + // voting marker. Startup configuration failures must not force recovery. + identityPubkey := solana.PrivateKey(validatorIdentity).PublicKey() + if err := consensusEngine.EnableVoting(consensusengine.VotingConfig{ + Identity: validatorIdentity, + AuthorizedVoter: validatorAuthorizedVoter, + VoteAccount: validatorVoteAccount, + HistoryDir: blockstorePath, + ReservedHistory: validatorReservedHistory, + InitializeVoteReservation: validatorInitializeReservation, + Genesis: solana.MustHashFromBase58(networkGenesisHash), + EpochForSlot: epochSchedule.GetEpoch, + SlotDuration: blockprod.AlpenglowSlotDuration, + WaitToVoteSlot: waitToVoteSlot, + ReadyToVote: func(slot uint64) bool { + wallSlot := global.WallClockSlot() + if liveSlot, ok := consensusEngine.AlpenglowLiveSlot(); ok { + wallSlot = liveSlot + } + return slot >= wallSlot || wallSlot-slot <= alpenglow.LeaderWindowSlots + }, + Peers: func(validators []alpenglow.ValidatorStake) []alpenglow.VotorPeer { + peers := make([]alpenglow.VotorPeer, 0, len(validators)) + seen := make(map[solana.PublicKey]struct{}, len(validators)) + for _, validator := range validators { + if validator.Stake == 0 || validator.NodePubkey == identityPubkey { + continue + } + addr, ok := sharedGossip.LookupAlpenglow(validator.NodePubkey) + if !ok { + continue + } + if _, duplicate := seen[validator.NodePubkey]; duplicate { + continue + } + seen[validator.NodePubkey] = struct{}{} + peers = append(peers, alpenglow.VotorPeer{Identity: validator.NodePubkey, Addr: addr}) + } + return peers + }, + }); err != nil { + klog.Fatalf("enable Alpenglow voting: %v", err) + } + rewardBuilder := rewardcerts.NewBuilder(rewardcerts.BuilderConfig{ RootSlot: global.Slot, BeforeBuild: consensusEngine.FlushAlpenglowRewardVotes, @@ -2752,14 +2786,15 @@ postBootstrap: leaderStop := make(chan struct{}) leaderDone := make(chan struct{}) leaderLoop := blockprod.NewLeaderLoop(blockprod.LeaderLoopConfig{ - Controller: controller, - Identity: solana.PrivateKey(validatorIdentity), - AccountsDb: accountsDb, - Broadcaster: broadcaster, - ShredVersion: uint16(turbineShredVersion), - EpochSchedule: epochSchedule, - AlpenglowClock: true, - SlotDuration: blockprod.AlpenglowSlotDuration, + Controller: controller, + Identity: solana.PrivateKey(validatorIdentity), + AccountsDb: accountsDb, + Broadcaster: broadcaster, + ShredVersion: uint16(turbineShredVersion), + EpochSchedule: epochSchedule, + AlpenglowClock: true, + SlotDuration: blockprod.AlpenglowSlotDuration, + CompletionReserve: time.Duration(validatorCompletionReserveMs) * time.Millisecond, ParentContext: func(slot uint64) blockprod.ParentContext { tip := replay.ChainTipParentContext() // Blockprod owns the replay-readiness rule. In particular, the first @@ -2796,6 +2831,7 @@ postBootstrap: } }, ProductionParent: consensusEngine.AlpenglowBlockProductionParent, + CanSignSlot: consensusEngine.AlpenglowCanSignLeaderSlot, CurrentSlot: func() uint64 { if slot, ok := consensusEngine.AlpenglowLiveSlot(); ok { return slot @@ -3214,6 +3250,8 @@ func printStartupInfo(commandName string) { } fmt.Printf(" Sigverify: %s%s%s %s(%s)%s\n", green, resolvedSigverifyBackend, reset, dim, sigverifyDesc, reset) + fmt.Printf(" workers=%d batch_target=%d shred_overlap=%t\n", + sigverify.TransactionWorkers(), sigverify.TransactionBatchTarget(), !sigverify.Cfg.DisableShredOverlap) } // Load state file for detailed info (only show for modes that use existing AccountsDB) diff --git a/cmd/mithril/node/sigverify_reporter.go b/cmd/mithril/node/sigverify_reporter.go index d374d93b9..597a3c116 100644 --- a/cmd/mithril/node/sigverify_reporter.go +++ b/cmd/mithril/node/sigverify_reporter.go @@ -42,7 +42,8 @@ func startSigverifyReporter(ctx context.Context) { // against, and so the resolved backend is recorded even on a node that // exits before the first tick. previous := sigverify.Stats() - mlog.NamedFilef("sigverify", "startup: %s", previous) + mlog.NamedFilef("sigverify", "startup: %s workers=%d batch_target=%d shred_overlap=%t", previous, + sigverify.TransactionWorkers(), sigverify.TransactionBatchTarget(), !sigverify.Cfg.DisableShredOverlap) for { select { diff --git a/cmd/mithril/node/vote_startup.go b/cmd/mithril/node/vote_startup.go new file mode 100644 index 000000000..547e55194 --- /dev/null +++ b/cmd/mithril/node/vote_startup.go @@ -0,0 +1,40 @@ +package node + +import ( + "fmt" + "math" + "strconv" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/config" + "github.com/spf13/cobra" +) + +func configuredWaitToVoteSlot(cmd *cobra.Command) (uint64, error) { + if flag := cmd.Flags().Lookup("wait-to-vote-slot"); flag != nil && flag.Changed { + return cmd.Flags().GetUint64("wait-to-vote-slot") + } + const key = "validator.wait_to_vote_slot" + if !config.IsSet(key) { + return 0, nil + } + // Unlike GetUint64, parsing explicitly must not turn an invalid operator + // cutoff into zero and silently remove the requested voting restriction. + slot, err := strconv.ParseUint(config.GetString(key), 10, 64) + if err != nil { + return 0, fmt.Errorf("%s must be an unsigned 64-bit slot: %w", key, err) + } + return slot, nil +} + +// The operator cutoff can postpone voting but cannot weaken the existing +// startup guard. Equality permits voting, subject to all other Votor checks. +func effectiveWaitToVoteSlot(startupWallSlot, configured uint64) uint64 { + automatic := startupWallSlot - startupWallSlot%alpenglow.LeaderWindowSlots + if automatic <= math.MaxUint64-2*alpenglow.LeaderWindowSlots { + automatic += 2 * alpenglow.LeaderWindowSlots + } else { + automatic = math.MaxUint64 + } + return max(automatic, configured) +} diff --git a/cmd/mithril/node/vote_startup_test.go b/cmd/mithril/node/vote_startup_test.go new file mode 100644 index 000000000..d86a02528 --- /dev/null +++ b/cmd/mithril/node/vote_startup_test.go @@ -0,0 +1,73 @@ +package node + +import ( + "math" + "strings" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/config" + "github.com/spf13/cobra" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" +) + +func TestConfiguredWaitToVoteSlot(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + want uint64 + invalid bool + }{ + {name: "default"}, + {name: "toml", toml: "wait_to_vote_slot = 1234", want: 1234}, + {name: "cli wins", toml: "wait_to_vote_slot = 1234", cli: "5678", want: 5678}, + {name: "explicit zero wins", toml: "wait_to_vote_slot = 1234", cli: "0"}, + {name: "maximum CLI", cli: "18446744073709551615", want: math.MaxUint64}, + {name: "negative TOML", toml: "wait_to_vote_slot = -1", invalid: true}, + {name: "fractional TOML", toml: "wait_to_vote_slot = 1.5", invalid: true}, + {name: "malformed TOML value", toml: `wait_to_vote_slot = "oops"`, invalid: true}, + {name: "empty TOML value", toml: `wait_to_vote_slot = ""`, invalid: true}, + {name: "overflow TOML value", toml: `wait_to_vote_slot = "18446744073709551616"`, invalid: true}, + {name: "negative CLI", cli: "-1", invalid: true}, + {name: "overflow CLI", cli: "18446744073709551616", invalid: true}, + } { + t.Run(tc.name, func(t *testing.T) { + viper.Reset() + t.Cleanup(viper.Reset) + config.ApplyDefaults(viper.GetViper()) + viper.SetConfigType("toml") + require.NoError(t, viper.ReadConfig(strings.NewReader("[validator]\n"+tc.toml))) + cmd := &cobra.Command{} + cmd.Flags().Uint64("wait-to-vote-slot", 0, "") + var err error + if tc.cli != "" { + err = cmd.Flags().Set("wait-to-vote-slot", tc.cli) + } + var got uint64 + if err == nil { + got, err = configuredWaitToVoteSlot(cmd) + } + if tc.invalid { + require.Error(t, err) + return + } + require.NoError(t, err) + require.Equal(t, tc.want, got) + }) + } + require.NotNil(t, Run.Flags().Lookup("wait-to-vote-slot")) +} + +func TestEffectiveWaitToVoteSlot(t *testing.T) { + for _, tc := range []struct{ startup, configured, want uint64 }{ + {100, 0, 108}, + {103, 0, 108}, + {103, 104, 108}, + {103, 108, 108}, + {103, 123, 123}, // Operator cutoff need not align with a leader window. + {103, math.MaxUint64, math.MaxUint64}, + {math.MaxUint64 - 7, 0, math.MaxUint64}, + {math.MaxUint64, 0, math.MaxUint64}, + } { + require.Equal(t, tc.want, effectiveWaitToVoteSlot(tc.startup, tc.configured), "%+v", tc) + } +} diff --git a/cmd/repair-sim/main.go b/cmd/repair-sim/main.go new file mode 100644 index 000000000..5eebf05e9 --- /dev/null +++ b/cmd/repair-sim/main.go @@ -0,0 +1,137 @@ +// repair-sim runs deterministic, single-node Turbine repair scenarios. +package main + +import ( + "encoding/json" + "flag" + "fmt" + "os" + "os/exec" + "runtime" + "strings" + "time" + + "github.com/Overclock-Validator/mithril/pkg/turbine/repairsim" +) + +type environment struct { + GoVersion string `json:"go_version"` + GOOS string `json:"goos"` + GOARCH string `json:"goarch"` + CPU string `json:"cpu"` +} + +type report struct { + Environment environment `json:"environment"` + Ledger repairsim.LedgerConfig `json:"ledger"` + Network repairsim.Config `json:"network"` + LedgerGenerationWall time.Duration `json:"ledger_generation_wall_ns"` + Result repairsim.Result `json:"result"` +} + +func main() { + var ( + scenarioFlag = flag.String("scenario", string(repairsim.ScenarioNearTip), "near-tip or deep-catchup") + slots = flag.Int("slots", 200, "number of deterministic slots") + fecSets = flag.Int("fec-sets", 4, "FEC sets generated per slot") + entries = flag.Int("entries", 0, "entries per slot (0 derives an exact FEC count)") + seed = flag.Int64("seed", 1, "deterministic content and network seed") + availability = flag.String("availability", "", "complete, near-loss, sparse, or mixed") + repair = flag.Bool("repair", true, "enable repair requests") + latency = flag.Duration("repair-latency", 20*time.Millisecond, "synthetic one-way response latency") + jitter = flag.Duration("repair-jitter", 2*time.Millisecond, "deterministic +/- response jitter") + loss = flag.Float64("packet-loss", 0, "repair response loss probability [0,1]") + duplicates = flag.Float64("duplicates", 0.02, "duplicate response probability [0,1]") + bandwidth = flag.Int64("repair-bandwidth", 100*1024*1024, "synthetic repair bytes/sec (0 is unlimited)") + concurrent = flag.Int("max-concurrent", 256, "maximum outstanding repair shreds") + corrupt = flag.Int("corrupt-responses", 0, "corrupt the first N repair responses") + naturalLate = flag.Bool("natural-late", true, "schedule selected late live shreds during repair") + spoolDir = flag.String("spool-dir", "", "persistent shred-spool directory (empty uses a temporary directory)") + cpuLabel = flag.String("cpu-label", "", "explicit CPU label when platform discovery is unavailable") + output = flag.String("output", "", "write JSON to this file instead of stdout") + includeTrace = flag.Bool("trace", true, "include the logical event trace in JSON") + ) + flag.Parse() + + scenario := repairsim.Scenario(*scenarioFlag) + network := repairsim.DefaultConfig(scenario) + network.Availability = repairsim.Availability(*availability) + network.RepairEnabled = *repair + network.RepairLatency = *latency + network.RepairJitter = *jitter + network.PacketLoss = *loss + network.DuplicateProbability = *duplicates + network.BandwidthBytesPerSec = *bandwidth + network.MaxConcurrent = *concurrent + network.CorruptResponses = *corrupt + network.NaturalLateShreds = *naturalLate + network.CollectTrace = *includeTrace + network.Seed = *seed + network.SpoolDir = *spoolDir + if network.Availability == "" { + network.Availability = repairsim.DefaultConfig(scenario).Availability + } + + ledgerCfg := repairsim.LedgerConfig{ + StartSlot: 10_000, + Slots: *slots, + FECsPerSlot: *fecSets, + EntriesPerSlot: *entries, + Seed: *seed, + ShredVersion: 1, + ReferenceTick: 63, + } + started := time.Now() + ledger, err := repairsim.GenerateLedger(ledgerCfg) + if err != nil { + fatalf("generate ledger: %v", err) + } + generationWall := time.Since(started) + result, err := repairsim.Run(ledger, network) + if err != nil { + fatalf("run simulation: %v", err) + } + cpu := *cpuLabel + if cpu == "" { + cpu = cpuModel() + } + report := report{ + Environment: environment{GoVersion: runtime.Version(), GOOS: runtime.GOOS, GOARCH: runtime.GOARCH, CPU: cpu}, + Ledger: ledger.Config, Network: network, LedgerGenerationWall: generationWall, Result: result, + } + encoded, err := json.MarshalIndent(report, "", " ") + if err != nil { + fatalf("marshal report: %v", err) + } + encoded = append(encoded, '\n') + if *output == "" { + _, _ = os.Stdout.Write(encoded) + return + } + if err := os.WriteFile(*output, encoded, 0o644); err != nil { + fatalf("write %s: %v", *output, err) + } +} + +func cpuModel() string { + if runtime.GOOS == "linux" { + if data, err := os.ReadFile("/proc/cpuinfo"); err == nil { + for _, line := range strings.Split(string(data), "\n") { + if key, value, ok := strings.Cut(line, ":"); ok && strings.TrimSpace(key) == "model name" { + return strings.TrimSpace(value) + } + } + } + } + if runtime.GOOS == "darwin" { + if out, err := exec.Command("sysctl", "-n", "machdep.cpu.brand_string").Output(); err == nil { + return strings.TrimSpace(string(out)) + } + } + return "unknown" +} + +func fatalf(format string, args ...any) { + _, _ = fmt.Fprintf(os.Stderr, format+"\n", args...) + os.Exit(1) +} diff --git a/config.example.toml b/config.example.toml index 7e3833087..7106f737a 100644 --- a/config.example.toml +++ b/config.example.toml @@ -328,6 +328,24 @@ name = "mithril" # Signature-verification workers (0 = GOMAXPROCS). tpu_sigverify_workers = 0 + # Optional minimum slot for NEW votes (inclusive), also --wait-to-vote-slot. + # Useful when rejoining after recovery. CLI overrides this setting. + # Zero adds no operator cutoff; the automatic startup cutoff and normal + # consensus checks still apply. A lower value cannot bypass those checks. + # Replay/repair continue while waiting. Previously recorded, authenticated + # votes can still be restored/rebroadcast under the existing recovery rules. + # This does not coordinate a cluster restart or wait for supermajority, + # and does not allow resetting a corrupt vote-history file. + wait_to_vote_slot = 0 + # Bounded cross-slot TPU queue. Zero keeps the 131,072-transaction default. + # Larger queues can prefill four busy leader slots, using additional memory. + tpu_max_buffered_transactions = 0 + + # Milliseconds reserved for local finalization and broadcast, not consensus + # finality. Zero keeps the conservative 75ms default. Tune from measured + # completion margins; this does not change the protocol slot deadline. + block_completion_reserve_ms = 0 + # ============================================================================ # [consensus] - Alpenglow Consensus # ============================================================================ @@ -524,14 +542,6 @@ name = "mithril" # Number of borrowed accounts to preallocate in arena (0 to disable) borrowed_account_arena_size = 1024 - # ed25519 signature verification backend. - # auto - use the AVX-512 accelerated backend when the CPU has - # AVX512-IFMA (Zen 4/5, Ice Lake and newer), else portable - # r51 - force the accelerated backend; startup fails without AVX512-IFMA - # generic - force the portable pure-Go backend - # stdlib - use Go’s crypto/ed25519 implementation after the mandatory strict rejection checks. - sigverify_backend = "auto" - # Enable/disable pool allocator for slices use_pool = true @@ -574,6 +584,33 @@ name = "mithril" # Filename to write CPU profile (for offline analysis with go tool pprof) # cpu_profile_path = "/mnt/mithril-data/profiling/cpu.pprof" +# ============================================================================ +# [sigverify] - Transaction Signature Verification +# ============================================================================ + +[sigverify] + # ed25519 backend: auto selects AVX-512 IFMA when available, else portable. + # r51 requires AVX-512 IFMA; generic and stdlib force portable backends. + # Strict signature checks are always enabled. The older + # tuning.sigverify_backend key remains supported when this key is absent. + backend = "auto" + + # Shared Turbine transaction verification workers; 0 = min(2, GOMAXPROCS). + # Leaves execution and shred/consensus processing room to run concurrently. + # This does not change validator.tpu_sigverify_workers or replay's fallback pool. + workers = 0 + + # Signature lanes per group: 4 or 8 (0 also means 8). Transactions stay + # indivisible, so a multisignature transaction may exceed this target. + # Ready short groups run immediately; no timer waits for more shreds. + batch_target = 8 + + # Decode complete entry batches and verify while later shreds arrive. + # Set true to compare against completion-only verification. + # Early work reserves at most 8 slots and 64 MiB of encoded component bytes; + # decoded transactions and Go bookkeeping use additional heap memory. + disable_shred_overlap = false + # ============================================================================ # [debug] - Debug Logging # ============================================================================ diff --git a/docs/alpenglow_branch_engine.md b/docs/alpenglow_branch_engine.md index 6ca91a1f4..33bc12d7f 100644 --- a/docs/alpenglow_branch_engine.md +++ b/docs/alpenglow_branch_engine.md @@ -154,6 +154,31 @@ mixed (heterogeneous-client) or Mithril-only cluster identically: timeouts, vote signing/transmission, durable vote-history persistence, and standstill participation. +## Replay-observer diagnostics + +The observer retains certificate history for deduplication and match/mismatch +reporting. A separate bounded index contains only retained, block-bearing +certificates that have not yet been reconciled against replay. Reconciliation +removes an entry after either a match or mismatch; eviction removes it together +with the historical certificate. Hashless/skipped replay cannot reconcile a +block-bearing certificate. Pending counts and age/window statistics retain the +same semantics, but scan unresolved entries rather than completed history. + +This index is disposable, process-local diagnostic state. It neither authorizes +votes nor substitutes for verified certificates, the chain tracker's finality +checks, durable signing bounds, vote history, or checkpoint recovery. Those +checks and persistence contracts are unchanged. + +`BenchmarkObserverEmptyReplay` measures observer work for an empty block, with +or without four preceding skipped slots, against 4,096 retained certificates. +It covers 0, 32, and 4,096 unresolved entries. On Ryzen 9700X (GOMAXPROCS=8, +three 300 ms runs), median time for the four-skips-plus-empty case with 32 +unresolved entries was 686.4 µs before the index and 1.87 µs afterward. With +all 4,096 entries unresolved it was 380.5 → 159.3 µs. These are component +benchmarks; they exclude execution, certificate cryptography, network delivery, +and end-to-end FAST inclusion. Live comparisons must account for observer +history warming after a restart and different leader/skip patterns. + ## What this proves — and does not The certificate layer proves which block *data* the cluster settled on. In diff --git a/docs/certificate-processing-evidence.md b/docs/certificate-processing-evidence.md new file mode 100644 index 000000000..c7b520914 --- /dev/null +++ b/docs/certificate-processing-evidence.md @@ -0,0 +1,14 @@ +# Certificate Processing: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/72514abc5a2a98d2a2823fe92f0bfbbeedbcc9bc) +(tag `review-evidence-20260916-certificate-processing`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/72514abc5a2a98d2a2823fe92f0bfbbeedbcc9bc/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/certpool-offlock.md b/docs/certpool-offlock.md new file mode 100644 index 000000000..646d3e70b --- /dev/null +++ b/docs/certpool-offlock.md @@ -0,0 +1,78 @@ +# Incoming vote verification outside the pool lock + +The incoming vote pool previously held one mutex while checking BLS batches and folding signatures into aggregates. A 45-second trace on the live Zen 5 validator observed a 9.49 ms p95 acquisition wait and a 57.68 ms maximum across 34,580 normally completed AddVote calls. This blocked other incoming votes and readers; durable pruning could also wait for this lock while holding the consensus output mutex. + +The change retains one expensive BLS batch at a time using a separate verification mutex, while releasing the pool-state mutex around cryptography. One caller owns processing for each slot. Ordinary arrivals may join that slot's pending maps while verification runs, and the owner revisits state after those arrivals. Admission that needs authentication to resolve quota pressure or competing signatures waits instead of weakening bounds or first-packet-poisoning checks. + +Candidates remain pending until verification finishes, so in-flight votes count against admission limits and exact duplicates consume no additional space. Only selected candidates are removed on completion. Pruning and eviction may remove a slot during verification; pointer identity checks prevent stale work from resurrecting it or releasing accounting twice. If the epoch lookup, installed validator-set identity or shred version changes, the old slot state is retired instead of mixing bindings. Normal engine validator sets are immutable within an epoch. + +Each completed verified batch publishes promptly, outside both mutexes. New arrivals do not defer an authenticated quorum until the slot stops receiving votes. Reward-footer flushing waits for the active owner, drains relevant pending votes, and honors the publication barrier. Verified stake reads preserve freshness by waiting for that slot's owner. Snapshot counters and durable pruning do not wait for cryptography. + +Point reuse is included: aggregation uses already-verified signature points, failed batches subdivide parsed members, and the unused tally public-key aggregate is removed. Randomized coefficients, signature checks, stake thresholds, equivocation budgets and paired-vote disjointness remain. + +## Lock ownership + +`verifyAndFoldTallyWithLockReleased` requires the pool lock on entry and returns +with it held, including early exits. It releases the lock during crypto and +revalidates slot and validator bindings after reacquiring it. The caller owns +that slot's processing marker. `finishSlotAndUnlock` consumes lock ownership: +it releases the lock, emits certificates and returns unlocked. Callers must not +pair it with a deferred unlock. + +## Bounded multi-scalar aggregation + +With more than one Go execution thread, batches of at least 16 parsed votes +use gnark `MultiExp` for each weighted +public-key/signature sum. G1 and G2 run sequentially with `NbTasks: 1`; the +existing pool verification mutex still admits one expensive batch at a time. +Smaller batches keep the scalar loop because bucket setup costs more than it +saves, particularly during failed-batch subdivision and two-candidate checks. +Single-thread configurations also retain the scalar path: native contention +tests found that MultiExp task handoffs could increase certificate wall time +there, despite reducing arithmetic. + +Each member still receives a fresh, independent, nonzero coefficient sampled +from the full scalar field. Its public key and signature use the same +coefficient. Entropy/aggregation errors retain the individual-verification +fallback. Parsing, subgroup/infinity checks, failed-batch subdivision, +publication and stake accounting are unchanged. + +### Native staging validation + +AMD Ryzen 7 9700X, Go 1.26.4, GOMAXPROCS=2, Nice 15, two-core CPU quota, with +the normal validator workload still running. The baseline restores the scalar +implementation from the preceding certificate split head (`395e4566`) with +identical fixtures. Baseline/candidate/candidate/baseline runs were followed by +a final check after adding the single-thread safeguard. Full-fold ranges: + +| Votes in fold | Scalar baseline | Final implementation | +| --- | ---: | ---: | +| 32 | 7.385–7.386 ms | 4.238–4.344 ms | +| 64 | 14.217–14.280 ms | 6.800–7.135 ms | + +A controlled scheduling experiment runs 64 valid votes every 200 ms alongside +4,096 repeated transfer executions. With two Go execution threads, the final +alternating native comparison reduced certificate processing from +16.25–16.39 ms to 8.43–10.22 ms. Execution results were noisy: 13.80–14.38 ms +baseline versus 13.84–17.56 ms candidate. The earlier native comparison and the +local comparison showed no execution regression, but the slower final sample +is retained; the shared host does not establish absence of interference. +This workload excludes account commits, network delivery and vote persistence. + +The initial prototype's single-thread certificate latency could worsen while +execution improved slightly. The final implementation therefore keeps the +original scalar arithmetic whenever GOMAXPROCS is one. No worker-count or +verification-admission changes accompany this optimization. + +## Reproduce and validate + +Run `go test -race ./pkg/alpenglow` for concurrency, pending-budget, stale-binding, +invalid-share, equivocation and publication-barrier coverage. +Run `go test ./pkg/alpenglow -run '^$' -bench '^BenchmarkCertPool(WeightedPairing|FoldVerifiedBatch)$' -benchmem -benchtime=500ms -count=3` with the same fixture on each revision. + +The earlier concurrent 64-vote diagnostic reduced Snapshot p95 from +8.57–8.72 ms with point reuse alone to 0.008–0.038 ms with verification outside +the lock. This measures reader latency, not vote throughput. The diagnostic's +ns/op includes intentional sampling pauses. Historical component baselines, +live observations and full qualifications are preserved in the +[evidence archive](certificate-processing-evidence.md). diff --git a/docs/certpool-point-reuse.md b/docs/certpool-point-reuse.md new file mode 100644 index 000000000..cc2ac3a84 --- /dev/null +++ b/docs/certpool-point-reuse.md @@ -0,0 +1,18 @@ +# Reuse verified BLS signature points + +Incoming verification returns parsed, verified members and reuses their signature +points when folding the tally. Failed aggregate checks subdivide parsed members +instead of reparsing each subset. Individual verification returns the same member +representation. Installed validator sets retain their parsed public keys. + +The unused tally public-key sum is removed. Randomized verification still builds +its required weighted public-key sum with independent full-field coefficients. +Subgroup/infinity checks, invalid-share rejection, duplicate/equivocation checks, +stake accounting and paired-vote disjointness remain enforced. + +Point ownership and message association are covered by differential tests against +individual verification, including malformed signatures, invalid ranks and wrong +payloads. See [pool concurrency and aggregation](certpool-offlock.md) for the +current lock contract, combined implementation and benchmark method. +The isolated point-reuse experiment is preserved in the +[historical evidence](certificate-processing-evidence.md). diff --git a/docs/erasure_recovery_experiments.md b/docs/erasure_recovery_experiments.md new file mode 100644 index 000000000..5c47fe2fe --- /dev/null +++ b/docs/erasure_recovery_experiments.md @@ -0,0 +1,130 @@ +# Fixed-shape FEC recovery + +Production uses direct recovery only for exactly one missing data shard in a +32-data/32-coding FEC set with an available coding shard. All other availability +patterns keep the general decoder. Recovery still passes ordinary packet/root +validation before assembler admission. + +The deterministic production-path harness is described in [repair_sim.md](repair_sim.md). +Reduced-subset and all-coding plans remain reference benchmarks, not runtime +policies. Historical investigation notes are in the [evidence archive](fec-producer-evidence.md). + +## Matrix contract + +The experiment uses the systematic generator over `GF(256)/0x11d`: + +```text +V[x,j] = x^j +A = V[0:32,0:32] +G = V * A^-1 +G = [I_32; C] +``` + +For the fixed 32+32 shape, exhaustive scalar tests confirm all 1,024 entries: + +```text +C[r,c] = 0xa5 / (0x20 xor r xor c) +``` + +They also confirm `C*C=I`. Recovery tests remain differential against +`github.com/klauspost/reedsolomon`; the closed form is not the sole oracle. + +## Candidates + +### Near tip: direct one-data recovery + +For missing data position `m` and available coding position `r`: + +```text +D_m = C[r,m]^-1 * P_r + + sum(i != m, C[r,m]^-1 * C[r,i] * D_i) +``` + +This prepares one 32-source coefficient row and writes one destination. It does +not construct or invert a general 32x32 matrix. + +Production uses a process-wide table containing every missing-data and +coding-row combination. This removes per-call plan construction while keeping +the same equation. The table is exhaustively differential-tested against the +general decoder across all 32 x 32 combinations. + +### Catch up: reduced missing-data system + +For missing data columns `M` and selected coding rows `R`, substitute every +known data shard and solve: + +```text +B[i,j] = C[R[i],M[j]] +B * D_M = adjusted_coding_rows +``` + +The experiment uses the Cauchy closed form to construct `B^-1` in `O(m^2)`, +expands direct rows over 32 selected sources, and proves each row satisfies the +requested systematic generator row before byte processing. An independently +implemented Gauss-Jordan inverse is the setup fallback. + +The byte kernel is intentionally portable and uses `reedsolomon.LowLevel`. +This isolates algorithm and plan costs; it is not evidence that a portable +kernel will beat the dependency's generated AVX2/GFNI kernels on amd64. + +### Catch up edge: all coding rows + +When all data rows are missing and every coding row is present, `C*C=I` means +the existing optimized encoder can apply `C` to the coding rows and recover the +data directly. This is kept as a separate synthetic arm. It is simpler than a +general decoder, but a cached reduced-system plan may still have a faster byte +kernel; hardware decides between them. + +## Synthetic coverage + +The tests cover: + +- every missing-data position with every coding-row choice for the direct path; +- every pair of missing data positions; +- deterministic mixed patterns at 2, 4, 8, 16, 24, and 32 missing data shreds; +- exactly-threshold and one-below-threshold availability; +- changed availability between setup and execution; +- destination failure atomicity; +- coefficient mutation detection; +- Cauchy inverses against independent Gauss-Jordan inversion; +- recovered bytes against the existing general decoder. + +Run: + +```bash +go test ./pkg/turbine/internal/rsrecover +go test -run '^$' \ + -bench '^(BenchmarkRecoverOneData|BenchmarkRecoverDataSubset)$' \ + -benchmem -benchtime=2s -count=6 \ + ./pkg/turbine/internal/rsrecover +``` + +Benchmark result interpretation must keep these cases separate: + +- `prepare`: cold or changing erasure pattern; +- `execute`: prepared/repeated pattern; +- `prepare-and-execute`: first useful output for a new pattern; +- general cache off: existing decoder with changing pattern cost; +- general cache on: existing decoder after the inversion is cached. + +No production dispatch threshold should be chosen from an Apple benchmark. +Final crossover decisions require the pinned amd64 target and synthetic arrival +traces for progressing, stalled, and bursty slots. + +## Zen 5 production gate + +The direct one-data path was measured on a Ryzen 7 9700X with Go 1.26.4, +`GOMAXPROCS=1`, and one pinned physical core. Medians below are from seven +sequential one-second samples unless otherwise noted. + +| Benchmark | General path | Direct one-data path | Change | +| --- | ---: | ---: | ---: | +| one-missing `SlotAssembler` boundary | 10.73 us/FEC | 2.82 us/FEC | -73.7% (3.8x) | +| near-tip repair simulation | 3.0295 ms/op | 2.8659 ms/op | -5.40% | +| deep-mixed repair simulation | 4.0343 ms/op | 4.0240 ms/op | -0.26% | +| deep-sparse repair simulation | 10.8489 ms/op | 10.8327 ms/op | -0.15% | + +The production dispatch is intentionally narrow. The deep scenarios do not +enter it and remain effectively neutral, while the near-tip workload benefits +from repeated exactly-one-missing recoveries. The one-missing boundary also +dropped from 144 to 5 allocations per operation. diff --git a/docs/fec-producer-evidence.md b/docs/fec-producer-evidence.md new file mode 100644 index 000000000..68c57c7d2 --- /dev/null +++ b/docs/fec-producer-evidence.md @@ -0,0 +1,18 @@ +# FEC producer and recovery evidence + +The original #259 source, standalone producer benchmarks, raw samples and +recovery derivation are preserved at +[the historical snapshot](https://github.com/Overclock-Validator/mithril/tree/a3b16ebaaf803807ad04a7975f3eccf1c15649ea) +(tag `review-evidence-20260916-fec-producer`). + +The September 6 comparison used development head `7e4e8af1`, not today's #278 +base. Fifty thousand 1,232-byte legacy transactions across three slots took +734.052 → 282.296 ms on one pinned Zen 5 CPU. With both versions using the same +30,816-byte batch target, the result was 734.052 → 288.378 ms. This measures +serial producer work, excluding transaction execution, admission verification, +worker queue overlap, routing and network delivery. It is not a whole-validator +speedup or a benchmark of the rebased combined Turbine review. + +The rebase preserves newer slot-byte reservations and the asynchronous shred +worker. Fixtures explicitly use the legacy 1,232-byte limit rather than the +newer 4,096-byte transport maximum. Reusable benchmark code remains in source. diff --git a/docs/leader-packing-evidence.md b/docs/leader-packing-evidence.md new file mode 100644 index 000000000..8a439d1eb --- /dev/null +++ b/docs/leader-packing-evidence.md @@ -0,0 +1,14 @@ +# Leader Packing: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/06ef067798c99947e8cc527450ad28430a9a7333) +(tag `review-evidence-20260916-leader-packing`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/06ef067798c99947e8cc527450ad28430a9a7333/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/leader_block_packing.md b/docs/leader_block_packing.md new file mode 100644 index 000000000..fd76fe466 --- /dev/null +++ b/docs/leader_block_packing.md @@ -0,0 +1,98 @@ +# Leader block packing and synthetic load tests + +This change reduces work performed during a leader's available packing window. +The scheduler owns and decodes packet bytes once, and prepares static message +validation, instructions/account metadata, compute limits, message hash and cost +while transactions are queued. Each bank checks the immutable feature snapshot +before reuse. Account state, age, duplicates, strict fee-payer eligibility, rent, +execution and all resource budgets remain bank-dependent checks. + +Leader execution reuses borrowed-account scratch and skips detailed replay +stage timers. Missing-current-bank lookup errors defer base58 formatting until +used, avoiding wasted work before parent lookup. Entry Merkle hashing retains +only the root-building scratch, and max-heap removal avoids heap interface dispatch +while preserving priority/FIFO order. Both priority heaps now track entry +indexes so consumption, eviction and expiry remove every buffer reference. +Repeated rebuffering reuses the caller's intact transaction without accumulating +duplicate heap references. The existing slot-local skip scanning policy remains +unchanged, including selection of newly arrived higher-priority transactions. + +## Capacity and protocol limits + +Live banks use slot-dependent budgets with slot-time feature gates taking effect +in the epoch after activation. For the September 13 cluster's 200ms regime with +RaiseBlockLimitsTo100m, the budget was 50M block-cost units, 20M writable-account +cost units, 50MB allocated-data growth and 10MiB entry bytes. The entry packer +reserves 48 bytes for the ending tick. A 100M per-slot budget would be incorrect +in this regime. Active limits are logged when a leader bank opens. + +The readonly-pair fixture has one signature, one writable fee payer, two existing +readonly accounts, no instructions, and 198 wire bytes. Its observed actual cost +is 1,028 units: 720 signature, 300 write lock and eight loaded-account units. Its +program execution cost is zero. The theoretical cost-only ceiling is 48,638, +but upfront admission must fit the larger estimated loaded-data reservation: +the offline bank test includes 48,622 before rejecting the next transaction. +This workload is designed for signature/packing load, not application execution. + +## Reproduce locally + +All commands below are offline. They use deterministic test keys and in-memory +accounts; no RPC, faucet, funding or transaction submission occurs. + +```sh +# The 200,000-message, eight-payer, two-blockhash workload used to prefill four slots. +go test ./pkg/tpu/txfixture -run '^TestReadonlyPair200KDistinctMessages$' -count=1 + +# Fill a 50M-cost bank, reject the next tx, check fees, and round-trip all entries +# through actual shred generation and decoding, preserving transaction order/hash. +go test ./pkg/blockprod -run '^TestReadonlyPairBlockCapacityAndShredRoundTrip$' -count=1 + +# Whole-bank construction: one serial caller; pre-signed unique transactions. +GOMAXPROCS=8 go test ./pkg/blockprod -run '^$' \ + -bench '^BenchmarkReadonlyPair(FullBlock|PreparedFullBlock)$' -benchtime=3x -count=3 + +# More representative instruction workloads and smaller component microbenchmarks. +GOMAXPROCS=8 go test ./pkg/blockprod -run '^$' \ + -bench '^BenchmarkWorkingBank(Decoded|Prepared)?HotAccounts$' -benchtime=2s -count=3 +``` + +The whole-bank benchmark admits 48,622 transactions and includes execution, +account publication, entry batching/hash work and final entry flush. Signing and +bank setup are excluded. The prepared variant additionally does static +preparation before timing, modeling a queue ready before leadership. That work +is moved, not eliminated. Actual AccountsDB, signature verification, network, +consensus and the protocol deadline are outside this benchmark. The correctness +test's shred round-trip is also outside the timed benchmark. + +`txfixture.ReadonlyPairWire` provides the same ordered-pair construction as the +live experiment: 128×127 unique messages per payer/blockhash. Repeated ordinals +need a different payer or blockhash. The 200k test verifies uniqueness across +all messages and decodes/verifies representative signatures and phase boundaries. + +## Queue supply and completion reserve + +```toml +[validator] +tpu_max_buffered_transactions = 0 # default 131072 +block_completion_reserve_ms = 0 # default 75ms +``` + +The corresponding flags are `--tpu-max-buffered-transactions` and +`--leader-completion-reserve-ms`. The measured full-prefill trial used 262,144 +queue entries and a 60ms reserve. Those are opt-in tuning values; defaults stay +unchanged. A larger queue uses additional memory for owned wire, decoded and +prepared objects. A shorter reserve needs measured local finalization/broadcast +margin and does not change the protocol deadline. Shifting completion also +shifts later bank start times, so it does not add the same packing time to all +four blocks. + +The live results showed why total supply and timing matter: 120k transactions +cannot fill four approximately 48.6k blocks. An 80k refill competed with ongoing +work. Preloading 200k into a larger queue improved the observed four-block total, +while the first block still had less usable time and later banks awaited local +replay/adoption. These observations do not isolate a single CPU bottleneck. + +[Measured results, exact baselines and evidence](https://github.com/Overclock-Validator/mithril/blob/06ef067798c99947e8cc527450ad28430a9a7333/docs/results/leader-block-packing/2026-09-13/README.md) +include both the successful near-limit block and the still-underfilled four-slot +window. The archived live helper is historical experiment source with explicit +cluster/identity/path constants; the offline fixture is the portable reproduction. diff --git a/docs/out-of-order-entry-prefetch.md b/docs/out-of-order-entry-prefetch.md new file mode 100644 index 000000000..9645b6d3d --- /dev/null +++ b/docs/out-of-order-entry-prefetch.md @@ -0,0 +1,28 @@ +# Prepare complete entry batches despite earlier shred gaps + +Live tracing found five large external blocks whose final assembly-to-ready time was 20–39 ms. Most fallback verification work was already available more than 20 ms before full assembly. For one 33,760-transaction block, 6,506 transactions in later complete batches waited 44–116 ms for discovery behind an earlier missing shred. + +The old discovery cursor stopped at the first missing data index. The new bounded bitmap index discovers each complete DATA_COMPLETE range independently. Receiving a data shred or recovering one through FEC can release its containing batch and, when it supplies a boundary, the next batch. A batch still needs every data shred and its preceding boundary, unless it begins at index zero. Results are queued in discovery order and final assembly restores wire order using the existing exact range and byte-identity checks. + +The index uses about 24 KiB per retained slot and is allocated only with streaming preparation enabled. Successor/predecessor queries have bounded cost even with reverse or adversarial arrival order. Existing worker counts, verifier batching, retained-byte limits, cancellation, generation ownership, final block checks and signature validation remain unchanged. + +Regression coverage includes a delayed earlier shred, delayed preceding boundary, FEC-recovered boundary, disabled preparation, randomized arrival against a reference oracle, duplicate discovery, index word/group/slot boundaries, and the existing final-validation and cancellation tests. Native Turbine/replay race tests and Turbine vet passed. + +## Controlled Zen 5 benchmark + +AMD Ryzen 7 9700X, Go 1.26.4, Narya r51, GOMAXPROCS=8, two transaction verification workers, target batch size eight. 33,760 generated signed transactions arrive as complete component bursts across 200 ms. One data shred in a component three quarters through the block is withheld until after the footer. Both versions run with identical fixtures in baseline/candidate/candidate/baseline order, three iterations per case in each run, while the validator remains running. Ranges below are the two run medians, not a confidence interval. + +| Wire transaction size | Prior assembly → ready | New assembly → ready | +| --- | ---: | ---: | +| 228 bytes, delayed shred | 21.43–22.77 ms | 2.82–2.83 ms | +| 1,232 bytes, delayed shred | 35.46–36.34 ms | 7.04–8.08 ms | +| 228 bytes, ordered | 2.50–2.56 ms | 2.40–2.52 ms | +| 1,232 bytes, ordered | 6.91–7.67 ms | 6.87–7.89 ms | + +The delayed-shred case now verifies roughly 33.5–33.7k transactions before assembly, compared with 25.3–25.6k previously. The benchmark asserts every retained signature is verified exactly once and block metadata remains correct. It includes assembly, decoding, final validation and transaction verification; it excludes network I/O, replay execution and actual FAST-certificate inclusion. This modeled case is not an end-to-end validator speedup or a prediction of overall FAST percentage. + +Reproduce with `MITHRIL_SIGVERIFY_FLOW_BACKEND=r51 GOMAXPROCS=8 go test ./pkg/turbine -run '^$' -bench '^BenchmarkEntryPrefetchGapArrival$' -benchtime=3x` on each implementation, copying the same benchmark file to the baseline. + +Historical live trials and their limitations are in the +[archived evidence](streaming-preparation-evidence.md). Component results do not establish +a sustained FAST-inclusion improvement. diff --git a/docs/repair_sim.md b/docs/repair_sim.md new file mode 100644 index 000000000..b0240f015 --- /dev/null +++ b/docs/repair_sim.md @@ -0,0 +1,153 @@ +# Deterministic repair simulation + +`cmd/repair-sim` is a single-process harness for measuring the local path from +an incomplete slot to a block that is available to replay. It exists because +near-tip repair and deep catch-up optimize different outcomes: + +- near the tip, latency of the first replay-blocking slot matters; +- during catch-up, sustained useful data and completed slots per second matter. + +The first implementation deliberately stops before UDP and transaction +execution. It establishes a deterministic, correctness-checked baseline before +network realism or alternative scheduling policies are introduced. + +## What is real and what is simulated + +| Stage | Implementation | +| --- | --- | +| block-component serialization | production `turbine.MarshalBlockComponent` | +| 32+32 FEC generation and Merkle signing | production `turbine.Shredder` | +| missing-shred selection | production `SlotAssembler.RepairRequests` | +| packet parsing and Merkle/signature validation | production `ParseShred` and `ShredSignatureVerifier` (the receiver's per-root cache) | +| verified-shred insertion | production `ShredSpool` | +| threshold detection and Reed-Solomon recovery | production `SlotAssembler.AddShredFrom` | +| component decode and transaction-signature gate | production slot completion path | +| remote peer, latency, jitter, loss, duplication, bandwidth | deterministic in-process simulator | +| replay notification | block emission is recorded as “offered to replay” | +| transaction execution | not run in this version | + +Synthetic entries are valid Alpenglow entry-batch components but contain no +transactions. This isolates shred/FEC/repair/storage costs; it is not a replay +execution benchmark. + +## Scenarios + +### Near tip + +`near-loss` alternates two useful patterns across FEC sets: + +- 30 data + 1 coding shred: one repair response crosses the threshold and + reconstructs the other missing data shred; +- 31 data shreds: the one missing data shred must be fetched directly. + +Selected omitted shreds can also arrive through the simulated live path while +a repair response is outstanding. This measures cancellation/late-response +behavior without changing production scheduling. + +Disabling repair stops request scheduling but still delivers natural late live +shreds. The run ends when all slots complete or those live arrivals are +exhausted; incomplete slots are reported without a repair-stall error. + +### Deep catch-up + +`mixed` begins every FEC set with 16 data + 15 coding shreds. One fetched data +shred crosses the threshold and reconstructs the remaining 15. + +`sparse` begins every FEC set with two data shreds and no coding layout. The +ordinary repair interface serves data shreds only, so nearly all missing data +must arrive over the simulated network. Comparing `mixed` with `sparse` +quantifies the network work avoided by already-held coding shreds. + +## Commands + +```sh +go test ./pkg/turbine/repairsim + +go run ./cmd/repair-sim \ + -scenario=near-tip \ + -slots=200 \ + -fec-sets=4 \ + -seed=1 \ + -cpu-label='Ryzen 7 9700X' \ + -output=/tmp/repair-near.json + +go run ./cmd/repair-sim \ + -scenario=deep-catchup \ + -availability=mixed \ + -slots=1000 \ + -fec-sets=4 \ + -seed=1 \ + -repair-latency=20ms \ + -repair-bandwidth=104857600 \ + -output=/tmp/repair-deep-mixed.json + +go run ./cmd/repair-sim \ + -scenario=deep-catchup \ + -availability=sparse \ + -slots=1000 \ + -fec-sets=4 \ + -seed=1 \ + -repair-latency=20ms \ + -repair-bandwidth=104857600 \ + -output=/tmp/repair-deep-sparse.json + +go test ./pkg/turbine/repairsim \ + -run '^$' -bench '^BenchmarkScenarios$' -benchmem -count=5 +``` + +Logical trace timestamps are deterministic. `wall_elapsed_ns`, allocations, +and `stage_cpu_ns` are actual local measurements and therefore are not expected +to be byte-identical across runs. + +## Correctness gates + +The current tests require: + +- exact requested FEC-set counts from authentic generated shreds; +- Merkle/signature validity for every canonical packet; +- no completed slot when loss is present and repair is disabled; +- complete, canonical entry streams after threshold recovery; +- a complete `ShredSpool` journal record before replay admission; +- rejection and retry of a corrupted repair response; +- deterministic logical traces for identical seeds; +- late repair responses to leave a completed block unchanged. + +The Turbine package also retains focused byte-for-byte Reed-Solomon recovery +tests. The simulator verifies the stronger end-to-end consequence: recovered +shreds must decode to the exact canonical entry sequence and pass the normal +completion gates. + +## Generator compatibility finding + +Building this harness exposed a multi-FEC component bug: the local generator +set `DATA_COMPLETE_SHRED` at every FEC boundary. The decoder correctly treats +that flag as the end of one serialized component, so a component spanning more +than one FEC set was truncated and failed to decode. The generator now follows +Agave's ordering: construct every FEC set, then mark only the final data shred +of the component complete (or last-in-slot). A regression test round-trips one +1,300-entry component across multiple FEC sets. + +## Future mode selection + +No production mode switch is added here. A later policy experiment should use +both replay distance and observed Turbine usefulness, with hysteresis: + +- stay in near-tip mode while the replay gap is small and Turbine supplies a + high fraction of useful shreds before repair deadlines; +- enter catch-up mode only when the replay-blocking gap is sustained and live + Turbine delivery is insufficient to approach FEC thresholds; +- return to near-tip mode only after both the gap and repair backlog fall below + lower thresholds. + +That signal is preferable to slot distance alone: a node may be numerically +close to the tip while receiving too few live shreds, or far behind while its +local spool already holds most FEC thresholds. + +## Next steps + +1. Add shallow catch-up and whole-block-pressure configurations. +2. Add a loopback-UDP transport without replacing the deterministic mode. +3. Expose internal FEC start/finish timing through opt-in instrumentation. +4. Run replay execution against a reusable synthetic bank fixture. +5. Compare ordinary requests with explicit test-only threshold-acquisition and + earliest-blocked-slot policies. diff --git a/docs/reserved-vote-history.md b/docs/reserved-vote-history.md new file mode 100644 index 000000000..bd94e2e57 --- /dev/null +++ b/docs/reserved-vote-history.md @@ -0,0 +1,162 @@ +# Voting persistence and crash recovery + +## Intended guarantee + +A crash must not lead to conflicting externally published votes or reserved-mode +leader actions because the validator forgot its earlier local decisions. The design permits +loss of recent detailed history and sacrifices voting availability when its +completeness is uncertain. It does not promise immediate restart voting, that +every signed vote reaches disk, or recovery from rolled-back safety files. +Normal anti-equivocation, execution, parent and validator-binding checks remain +necessary; this is a persistence contract, not a proof of the entire protocol. + +The failure model includes process termination and host/power failure, provided +successful file and directory syncs survive, the current safety files are +preserved, and only one fenced owner uses the signing identity. Software tests +exercise the recovery decisions; they do not qualify actual storage against +power loss. Valid signatures on saved files establish integrity, not freshness. + +## Two different publication guarantees + +| Mode | Required before a vote can escape to the pool/network | What a restart may trust | +| --- | --- | --- | +| Default synchronous history | Exact validated history is written, file-synced, renamed and directory-synced before local pool admission or network enqueue. | Retained exact decisions and their rooted boundary, subject to normal restoration checks. | +| Opt-in reserved history | The vote's slot is covered by an acknowledged durable reservation before signing. Its exact history snapshot is prepared and queued before publication, without waiting for per-vote I/O. | The startup reservation bound, unless a separately validated clean-history seal proves exact retained history. | + +In synchronous mode, BLS bytes may be computed privately in RAM before history +is synced. The guarantee is **persist before publication**, including local +pool admission because it can publish a certificate. In reserved mode, a +successful queue submission, background rename, or `written` counter is **not** +a durable acknowledgement of that vote. Replacing a whole file atomically is +not the same as making its bytes/directory entry survive power loss. + +Both modes retain complete decisions in memory while running and prune them +only through the normal verified-root rules. The synchronous history guarantee +covers votes; it is not a complete journal of produced leader blocks. The +additional leader reservation barrier applies only in reserved mode. + +## Reserved-mode restart rule + +Let **H** be the reservation's `Through` value loaded at startup, **F** the +verified finality/checkpoint floor, and **S** a proposed signing slot. H bounds +what the previous process *might* have signed; it is not its last actual vote. + +Without a validated clean seal, every vote type and historical local-vote +restoration must obey all of these conditions: + +- **S > H**: never re-sign in the uncertain range during this run, even when + an older history file contains that particular vote. +- **F >= H**: do not sign above the range until verified finality/checkpoint + state has reached its end. F equal to H is sufficient; S equal to H is not. +- **S <= the current acknowledged reservation**, plus all ordinary protocol, + live-joining and configured minimum-slot checks. + +The startup recovery bound stays fixed during the run. Background renewal may +raise the current signing allowance; it does not move the recovery target. +Newly received blocks, elapsed wall time, an RPC tip, replay progress alone, and +`--wait-to-vote-slot` cannot substitute for verified finality/checkpoint state. +The recovery wait can be indefinite if the cluster halts below H. + +For example, suppose votes through 1,015 escaped, detailed history survives only +through 1,012, and the durable reservation is 1,032. After an unclean restart, +slots <= 1,032 remain forbidden. Slot 1,033 is also forbidden while F < 1,032. +Once F >= 1,032, it can pass the recovery gate only after an acknowledged grant +covers 1,033 and all normal voting checks pass. Seeing a new block after restart +does not by itself meet these conditions. + +## Clean shutdown is a durable protocol + +1. Stop/join the voter and leader producer. Halt/join the reservation worker; + drain/join the history writer so an older rename cannot overwrite the seal. +2. Require no latched safety fault or history-write failure, no unresolved + reservation-write uncertainty, and verified recovery through the startup H. +3. Sync exact retained history, including the verified rooted boundary. +4. Sync a reservation record containing the digest of that exact history. + +A successful process exit, a signal handler running, or an attempted final save +is not sufficient. Startup must load and validate the history and reservation, +match the clean digest, and **durably consume the clean marker with a dirty +successor before new vote/leader signing or new history decisions**. A later crash uses +the reservation again, even if the old history still looks valid. + +| Restart state | Vote recovery | Leader recovery | +| --- | --- | --- | +| Valid history and reservation, no matching clean seal | Enforce S > H and F >= H, including restoration. | Enforce S > H and F >= H. | +| Valid matching clean seal, successfully consumed | Resume using exact retained voting decisions and ordinary checks; no additional vote quarantine through H. | Still enforce S > H and F >= H: vote history does not enumerate every block that may have been signed. | +| Missing, corrupt, unreadable or incompatible enrolled state | Refuse automatic reset/startup; do not infer safety from an RPC tip. | Same refusal. | + +Errors during sealing do not authorize treating the session as clean; startup +must validate whichever durable record survived. A failed reservation write +may have reached storage despite its error. Running signers retain only the +previous acknowledged allowance; a later monotonic successful write can resolve +that uncertainty. An unresolved error prevents deliberately sealing clean. + +## Lifetime, storage and enrollment + +The reservation grants up to 32 slots beyond the requested slot and renews when +16 or fewer remain. Only successful file-and-directory sync acknowledgement +publishes new permission. Exhaustion pauses signing while replay/verification +continue. Repeated restarts without new grants do not advance the bound. + +Detailed history uses one ordered writer with one in-flight and at most one +newer pending complete snapshot. New complete snapshots may supersede unwritten +ones. Validation/encoding/signing of the snapshot remain on the voter goroutine; +only file I/O is asynchronous. Writer errors latch a safety fault and stop +voting. An already in-flight vote is still covered by its durable reservation. + +Preserve both `vote_history-.mithril.json` and +`vote_reservation-.mithril.json` independently of AccountsDB snapshots. +The checkpoint encoding cache is only an encoding optimization; it is not the +vote journal or a replacement for the reservation. Rolling back AccountsDB or +an application binary must not roll back either signing-safety file. + +The directory lock only excludes concurrent owners of the same history path. +It cannot fence copies of the identity on other hosts/paths. Signed records and +generation numbers cannot detect an operator restoring an old valid pair of +safety files. Media loss, stale safety-file restoration, dishonest sync behavior +and compromised/copied signing keys are outside this automatic recovery contract. + +Use `--reserved-vote-history` to opt in; default persistence remains synchronous. +First enrollment additionally requires `--initialize-vote-reservation`, exclusive +identity ownership and complete synchronous history from the stopped previous +writer. An empty directory is appropriate only for a previously unused identity. +The software cannot distinguish that case from deleting both files for an old +identity; initialization is not a safe disaster-recovery reset. Remove the +initialization flag afterward. These are CLI flags, not TOML settings. + +Enrolled history uses version 2, rejected by older binaries that do not enforce +the reservation. Missing/corrupt enrolled state or a changed identity, vote +account, authorized voter, genesis or shred version must not be automatically +re-enrolled. Disabling the flag or deleting files is not a supported downgrade. +Transport binding/validation runs before the clean marker is consumed. + +## Live admission is separate from restart recovery + +The live vote admission floor uses retained consensus-pool state and the history +root, not the newest finalization certificate. Replay can still contribute a +notarization within the bounded 16-slot retained tail before later certificates +are collected; finalized retained slots can also receive skips. Durable-root +pruning is ordered behind completed replay events on the voter. These admission +changes also apply to synchronous mode. They do not weaken reserved-mode F >= H. +`--wait-to-vote-slot N` adds an inclusive minimum; it cannot bypass recovery, +execution, retained-root or parent checks. + +## Evidence and limits + +Existing tests map the recovery contract to these cases: + +| Contract | Tests | +| --- | --- | +| Lost valid history suffix; fixed bound across repeated crashes | `TestReservedVotingLostHistorySuffixAndRepeatedCrash` | +| All five vote types, restoration and leader gates at H/H+1 | `TestReservedVotingEverySignatureTypeAtBound` | +| Clean-marker consumption and stricter leader restart | `TestReservedVotingCleanMarkerConsumedBeforeSigning` | +| Digest mismatch, missing/corrupt/domain-mismatched state | `TestReservedCleanDigestMismatchUsesCrashRecovery`, `TestReservedVotingRejectsMissingCorruptOrWrongDomain` | +| Pending/uncertain sync cannot authorize signing | `TestSigningReservationUnacknowledgedSyncCannotAuthorize`, `TestSigningReservationUncertainWriteSurvivesRestart` | +| Writer drain, failure and lost pending snapshots | `TestAsyncHistoryBlockedWriteDoesNotDelayVotesAndCleanCloseDrains`, `TestAsyncHistoryFailureStopsVoterAndPreventsCleanMarker`, `TestAsyncHistoryProcessCrashLosesPendingSnapshots` | + +The subprocess-kill test exercises lost in-flight/pending application snapshots; +it does not power-cycle a host or prove filesystem durability. Fault-injection +and older-history fixtures check the algorithm's decisions under the stated +storage contract. This is not a formal consensus proof or mainnet storage +qualification. Persistence benchmarks likewise do not establish replay or FAST +performance. diff --git a/docs/rewards-unwind-retirement.md b/docs/rewards-unwind-retirement.md new file mode 100644 index 000000000..edf207be8 --- /dev/null +++ b/docs/rewards-unwind-retirement.md @@ -0,0 +1,58 @@ +# Retiring durable rewards bookkeeping + +A completed partitioned-rewards distribution used to leave its in-memory +descriptor alive for the rest of the replay attempt. The fork-switch guard +rejects any such descriptor because account-overlay unwind cannot restore the +consumed spool or its distribution counters. This is necessary while completion +is speculative, but unnecessarily forces checkpoint replay after completion +has become durable. + +Replay now observes the inactive EpochRewards sysvar in a successfully executed +bank's immutable snapshot, with zero partitions remaining. It remembers that +bank's slot and the exact distribution descriptor. Only applying a successful +durable fold through that slot retires the descriptor. Later bank observations +do not move the completion slot forward. A new descriptor/epoch invalidates the +old evidence; missing sysvars or unknown completion retain the old fallback. + +## Safety and recovery contract + +- Completion in memory, certificate finality, and submitting a fold do not + authorize retirement. Failed folds leave the durable watermark unchanged. +- Active distribution and completed-but-not-durable distribution retain the + existing rewards guard. No spool reconstruction or rewards rollback is added. +- After retirement, in-memory switches still require the existing epoch, + vote/stake-cache, parent-context, sysvar and transaction-status checks. + Switches at/below the durable watermark still require durable recovery. +- Completion evidence is replay-thread-owned and process-local. It does not + change checkpoint formats, signing reservations, persisted vote history, + clean-shutdown rules or restart authorization. Restart retains the existing + persisted EpochRewards validation. No extra file or disk sync is introduced. + +## Incident motivating the change + +On Zen 5, distribution completed at slot 3,942,001. At a later parent-linked +switch, the durable checkpoint was already 3,944,067; child 3,944,076 selected +parent 3,944,073, abandoning the suffix from 3,944,074. The remaining descriptor +forced the rewards-window fallback even though completion was below the root. +Checkpoint recovery re-fetched previously received blocks, with logged waits +of 2.739 seconds and 0.967 seconds. A buffered 665-transaction block waited +3,613.510 ms for replay admission and then executed in 7.520 ms. + +These are incident observations, not a before/after benchmark or a measurement +of checkpoint encoding/fsync time. Thirteen observed FAST aggregates omitted +our vote during the recovery interval; that does not prove absence from every +FAST aggregate or a single cause for all thirteen omissions. No live latency +improvement is established until a comparable switch exercises the new path. + +## Validation + +`rewards_retirement_test.go` covers active/missing bank state, unknown completion, +the exact durable boundary, later-bank observations, generation changes, failed +and successful folds, and an exact-parent unwind after retirement (including +account values, resume state and immutable rewards sysvars). Existing unwind +tests still require fallback for zero-remaining bookkeeping without retirement, +cross-epoch switches, dirty vote/stake caches and invalid parent snapshots. + +Full replay/rewards race suites passed locally and in the combined native +build; native node recovery/checkpoint race tests, vet and validator build also +passed. These are software tests, not mainnet power-loss qualification. diff --git a/docs/shred-spool-completion.md b/docs/shred-spool-completion.md new file mode 100644 index 000000000..b726265f3 --- /dev/null +++ b/docs/shred-spool-completion.md @@ -0,0 +1,85 @@ +# Shred-spool completion publication + +`MarkComplete` runs on the block-delivery path. Previously it wrote a 16-byte +`complete.idx` record while holding the spool mutex. This write uses no fsync, +but ordinary filesystem writes can still wait. Earlier live observations included +32–40 ms completion-to-delivery delays; one separate detailed trace localized a +37.8 ms pause to the journal write on an already-adopted own-leader block. That +own-leader example did not delay its vote. Do not equate all completion-to-delivery +stalls with journal I/O without the finer trace. + +Completion publication now updates the in-memory map and makes a nonblocking +submission to one writer with a 256-record queue. The writer never takes the +spool mutex. A full queue drops only the persistent completion hint; in-memory +completeness remains available. This bounds pending memory and prevents storage +backpressure from directly reaching completion publication. No worker-count or +validator configuration changes are required. + +## Recovery and ownership contract + +The spool is a disposable verified-shred cache, not vote history, account state, +or a durable checkpoint. Completion hints can be lost on an unexpected stop or +queue overflow. Recovery then reassembles/re-repairs; a hint never replaces the +assembler's coverage and block validation. The packet checksums and journal/file +formats are unchanged. There is no new fsync or power-loss durability guarantee. + +Deletion and corrupt-tail truncation are different: they must not race an older +queued completion. Their tombstone goes through the same ordered writer, and the +caller waits for it **before changing the slot file**. Thus an older queued hint +cannot be written after the tombstone and resurrect completeness for a replacement +partial file. These uncommon operations can still wait for storage while holding +the spool mutex; this change does not remove every source of spool contention. + +After a short/error write, the worker stops appending hints and truncates the +journal to zero. This avoids appending behind a partial record and removes old +completion hints before an invalidation is acknowledged. If truncation also fails, +the invalidation fails and the slot-file mutation is refused; a later attempt can +retry. Losing all cached completion hints is an acceptable repair-cost fallback. + +`Close` excludes further mutations, flushes packet buffers, retries current +completion hints if the queue overflowed, and drains/closes the journal worker +before returning. The next opener therefore preserves the existing clean-handoff +contract when storage succeeds. A stuck disk can still delay shutdown. The worker +must not outlive ownership of the spool directory. No voting-resume, signing-bound, +checkpoint-coverage or consensus-safety rule changes. + +## Validation and measurements + +Tests block the writer and overflow the queue while asserting that completions +remain available, then verify clean-close recovery. A replacement-file test keeps +the old completion write blocked and verifies that replacement cannot proceed +before its tombstone. Fault tests inject partial writes and failed truncation, +verify refusal to mutate the slot, then retry and check that no stale hint returns +on restart. Existing checksum/torn-tail, retention, handoff, receiver shutdown and +invalid-block tests remain covered. + +Native full turbine/blockstream race suites, vet and the combined validator build +passed on the Ryzen 9700X (Zen 5), Go 1.26.4. The test process used GOMAXPROCS=2, +Nice=19 and a 200% CPU quota on the active validator host. + +Run the identical `BenchmarkShredSpoolMarkComplete` file on both source revisions. +Each sample opens a fresh spool, appends a packet outside the timer, times one +completion, then closes/drains outside the timer. Thus no already-complete dedupe +or queue-overflow drop is measured. Three runs of 300 iterations: + +| Component | Before | Candidate | +|---|---|---| +| Median run p50 | 2.805 µs | 0.170 µs | +| Median run p99 | 6.201 µs | 0.581 µs | + +This measures ordinary storage, not injected tail latency, total CPU work, replay +or FAST inclusion. Disk work moves to the worker; it does not disappear. The +baseline is the exact previously deployed combined validator, SHA256 +`a58366704680154628ff0a4c6b4027a3e5b79f1909eb39bffeec326c008f0b68`, not the full +branch versus alpenglow-dev. + +## Limits + +The blocked-writer regression establishes isolation of completion publication. +It does not eliminate all spool I/O: ordered invalidations, slot-file operations +and shutdown can still wait for storage. Historical live trials did not establish +an overall large-block p99 improvement and included remaining verification tails. +Keep those limits separate from the component benchmark above. + +[Historical measurements and source](spool-completion-journal-evidence.md) retain +the original deployment comparison and its trace/probe qualifications. diff --git a/docs/shred_retention_performance.md b/docs/shred_retention_performance.md new file mode 100644 index 000000000..ebcb9af82 --- /dev/null +++ b/docs/shred_retention_performance.md @@ -0,0 +1,44 @@ +# Shred retention sweep scheduling + +`SlotAssembler` previously scanned its incomplete/completed slots, block-ID hints, +rejected IDs, partial observations, and priority repair state on every incoming +shred, including duplicates and completed-slot packets. The new schedule skips +age sweeps when neither the observed edge nor the retention floor nor relevant +retained state has changed. The original sweep algorithm and age limits remain. + +Mutations invalidate the cached sweep after adding old identity hints or partial +observations, resetting a generation, or terminating completion. Completion +success, error, cancellation and abort all release protected parent-ID state. +Repair-floor advancement, reduction and clearing are checked on the next packet. +The hard incomplete-slot capacity check remains unconditional on every packet, +including catch-up insertion at an unchanged edge. Eviction preference and +protection for completing generations and the repair head are unchanged. + +Regression coverage includes changing the repair floor without advancing the +edge, every terminal completion outcome, newly added old identity hints, and +capacity overflow at a fixed edge. Existing completion, cancellation, generation, +FEC and repair tests also run as part of the full Turbine suite. + +## Isolated benchmark + +`BenchmarkRetentionRepeatedCompletedShred` drives the public `AddShred` path +with a completed-slot packet and 513 retained entries in each of four metadata +maps. It measures the repeated scan/rejection case, not full packet decoding, +authentication, FEC, replay, or a whole-validator speedup. Five 500 ms runs: + +| Host | Baseline median | Candidate median | Allocation | +| --- | ---: | ---: | ---: | +| Apple M4 Pro | 11,617 ns/packet | 7.718 ns/packet | 0 on both | +| Ryzen 7 9700X, GOMAXPROCS=2 | 11,443 ns/packet | 12.99 ns/packet | 0 on both | + +Baseline and candidate use the same fixture; Go's source overlay selects the old +assembler for baseline runs without changing other source. This fixture shows the +avoided work, not a prediction for a validator's actual map population. + +Validation passed locally and natively: race suites for Turbine, replay, +consensus and node; Turbine vet; complete validator build. The live integration +applies this patch over the exact deployed FEC/peer-isolation/status-expiry source. + +Historical live trials and their limitations are in the +[archived evidence](streaming-preparation-evidence.md). Component results do not establish +a sustained FAST-inclusion improvement. diff --git a/docs/spool-completion-journal-evidence.md b/docs/spool-completion-journal-evidence.md new file mode 100644 index 000000000..d2b40d986 --- /dev/null +++ b/docs/spool-completion-journal-evidence.md @@ -0,0 +1,14 @@ +# Spool Completion Journal: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/f0b72ab239efbbd4811498d73b36ba73eb6192e1) +(tag `review-evidence-20260916-spool-completion-journal`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/f0b72ab239efbbd4811498d73b36ba73eb6192e1/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md new file mode 100644 index 000000000..8ba391cb5 --- /dev/null +++ b/docs/status-checkpoint-capture.md @@ -0,0 +1,98 @@ +# Transaction-status checkpoint capture and encoding + +Replay captures immutable lineage and coverage metadata before submitting a +checkpoint to the promotion worker. Sorting, encoding and writing happen on +that worker. Capture does not retain parent links outside the selected window. +Publication and durable-root ordering are unchanged. + +Each node memoizes its canonical encoded body on first serialization. Capture +and pruning share the same cache object when copying a node header; they never +copy a used synchronization primitive. Encoding depends on the immutable slot, +block-ID presence/value and status delta, not its parent link. Concurrent +encoders synchronize through `sync.Once` without taking the live cache lock. +Each snapshot still constructs its own coverage header and returns an owned +output buffer. The MTS2 format and restore validation are unchanged. + +The cache retains roughly one extra encoded window (30 MB for 1.5 million +keys), plus any nodes pinned by older views. There is no global encoding map: +caches become collectible with their last node/view. A completely new window +still pays for all sorting. Output copying and checkpoint I/O remain necessary. + +## Encoding benchmark + +`BenchmarkTransactionStatusCheckpointEncoding` uses a 300-root window with +5,000 keys per root (1.5 million keys, roughly 30 MB encoded). Each iteration +replaces the specified number of roots. Fixture creation and initial warming +are excluded; new node headers, sorting and output allocations are included. +The baseline is the original uncached wire encoder retained in tests. + +Apple M4 Pro, Go 1.26.4, one caller, GOMAXPROCS=12; medians of three runs: + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 158.05 ms | 1.35 ms | +| 8 | 157.14 ms | 5.09 ms | +| 32 | 155.62 ms | 17.51 ms | +| 128 (default fold cadence) | 153.74 ms | 67.11 ms | +| 300 (entirely new) | 155.90 ms | 156.29 ms | + +At the default cadence, allocated bytes per encoding fell from 99.12 MB to +57.33 MB; this excludes retained heap. These are encoding measurements, not +end-to-end fold/replay timings or live FAST improvements. Data distribution +matters: newly rooted large blocks can account for most keys in the window. + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusCheckpointEncoding$' -benchmem -benchtime=1s -count=3`. + +Tests compare exact bytes with the original encoder across coverage flags, +block IDs and sorted groups; check concurrent encoding during pruning/unwind; +verify cache sharing before and after warming; and restore checkpoints after +callers mutate their own output buffers. The replay race suite and vet pass. + +Related behavior: [status expiry](transaction-status-expiry.md) and +[status publication](transaction-status-publication.md). + +## Native Zen 5 validation + +AMD Ryzen 7 9700X, Go 1.26.4, GOMAXPROCS=2, Nice 15 and a two-core CPU quota, +while the validator continued its normal workload. Same moving-window fixture; +three samples per case, medians below. This compares the original uncached +encoder with memoization, not the whole status-publication change against dev. + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 195.34 ms | 3.73 ms | +| 8 | 195.15 ms | 8.56 ms | +| 32 | 194.77 ms | 23.54 ms | +| 128 (default fold cadence) | 194.52 ms | 85.02 ms | +| 300 (entirely new) | 201.88 ms | 198.55 ms | + +The default-cadence result is approximately 2.3x, with the same 99.12 → 57.33 MB +allocation reduction. Cold/all-new windows remain roughly unchanged. Native +combined race suites, vet and the validator build passed. These are historical staging measurements. They do not establish an isolated +live reduction in durable-root lag or missed FAST votes. + +## Fold admission before collecting account writes + +Replay checks for a checkpoint batch on every iteration, including skipped +slots. `WorkingSet.PromotionChunk` first counts eligible held slots under its +read lock. If fewer than the configured batch size are available, ordinary +admission returns nil without allocating account-pointer lists. When ready, +it collects only the oldest batch, not the entire eligible suffix. Forced +partial folds still collect the available prefix. + +This preflight is not a finality shortcut or a new recovery policy. Replay's +existing finality/verification gates supply the upper bound. Selection and +collection hold the same lock; account pointers retain their existing ownership +contract. Preparation does not prune the suffix or advance the durable root. +The worker's write/commit order, required resume context, checkpoint reference +validation, completion bookkeeping, and forced shutdown/epoch-boundary paths +are unchanged. + +`BenchmarkBuildFoldJobWaitingForBatch` holds 127 slots with 512 account writes +each while waiting for the default 128-slot batch. On Ryzen 9700X, +GOMAXPROCS=8, three 300 ms runs, median admission-check time fell from 426 µs +to 31.8 ns; 627,008 bytes and 134 allocations per rejected preparation became +zero. This measures an ineligible batch check, not encoding, disk I/O, or a +ready checkpoint. Boundary tests cover gaps, the finality upper bound, a full +batch, forced partial batches, and selection after promotion; existing replay +checkpoint/recovery tests cover the unchanged durable path. diff --git a/docs/status-checkpoint-expiry-evidence.md b/docs/status-checkpoint-expiry-evidence.md new file mode 100644 index 000000000..978180120 --- /dev/null +++ b/docs/status-checkpoint-expiry-evidence.md @@ -0,0 +1,19 @@ +# Status Checkpoint Expiry: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf) +(tag `review-evidence-20260916-status-checkpoint-expiry`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. + +The tagged snapshot also preserves the later 64-partition visible-status-map +experiment. That experiment is deliberately excluded from this review: it added +preparation work and did not demonstrate an overall large-block p99 benefit. +Earlier publication preparation and immutable-node encoding reuse remain. diff --git a/docs/streaming-preparation-evidence.md b/docs/streaming-preparation-evidence.md new file mode 100644 index 000000000..53e444472 --- /dev/null +++ b/docs/streaming-preparation-evidence.md @@ -0,0 +1,14 @@ +# Streaming Preparation: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/1c1171d3661d0404b013a9bf9391e23eb660706e) +(tag `review-evidence-20260916-streaming-preparation`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/streaming_message_identities.md b/docs/streaming_message_identities.md new file mode 100644 index 000000000..d852b22d4 --- /dev/null +++ b/docs/streaming_message_identities.md @@ -0,0 +1,106 @@ +# Prepare transaction message identities during shred arrival + +The live Zen 5 admission trace found ~12.45 ms of message serialization, +hashing and identity-cache preparation after assembly on large blocks, followed +by ~1.66 ms of same-block duplicate-plan work. Most blocks reached this stage +while replay was already waiting. + +Turbine now asks signature-verification workers to retain a message identity +derived from the exact canonical bytes they already serialize for verification. +The existing two workers process available groups without waiting for more +transactions. TPU callers of the ordinary verifier do not compute these extra +identities. + +Each opaque result binds a successful signature verdict to the transaction +pointer, message version and recent blockhash. Completion joins requests and +imports only identities covering the final ordered transaction slice. Canceled +requests are reverified; discarded UpdateParent prefixes are excluded. Results +are caller-owned and cannot refer to reusable verifier scratch. The block cache +owns its imported storage and remains nonserialized. The existing requirement +that signed message contents remain immutable still applies; arbitrary in-place +message edits require invalidation, as before. + +Same-block duplicate rejection and mutable ancestor/status checks remain in +place. This change does not cache the duplicate-check plan or alter epoch +processing, vote persistence, worker counts, scheduling deadlines, or packing. + +## Zen 5 comparison + +Baseline: the currently deployed shred-retention source, including its existing +FEC and peer-isolation integrations. Candidate: that same source plus streaming +identities. Both binaries use the identical new benchmark harness. + +`BenchmarkEntryMessageIdentityArrival` feeds real generated data shreds through +the assembler, entry prefetch and signature-verification pipeline, then includes +the admission-time identity lookup. Each block has 33,760 single-signature +transactions, either 228 or 1,232 bytes on the wire. The tip model schedules +component arrivals across 200 ms; catchup offers every shred immediately. +Fixture construction, packet parsing and shred-signature authentication are +outside the timer. There is no network loss, transaction execution, PoH/reward +processing or final whole-block duplicate map in this benchmark. + +AMD Ryzen 7 9700X, Narya `r51` AVX-512 backend, two signature workers, target eight +signature lanes, GOMAXPROCS=16. Five iterations per scenario per run, two runs +per variant in baseline/candidate/candidate/baseline order. Values below are the +median of the two run medians, not a percentile computed over all ten samples. +The live validator and load services continued running, so host contention can +affect the results. + +| Arrival model | Transaction size | Last shred → identities available, baseline | Candidate | +| --- | ---: | ---: | ---: | +| Over 200 ms | 228 bytes | 13.62 ms | 2.66 ms | +| Over 200 ms | 1,232 bytes | 73.72 ms | 7.81 ms | +| Catchup | 228 bytes | 92.90 ms | 88.40 ms | +| Catchup | 1,232 bytes | 154.75 ms | 118.20 ms | + +The table includes completion work; it does not merely move that work out of +the admission timer. Admission's identity lookup alone falls from ~12.1 to +0.24 ms for 228-byte tip transactions and ~66.9 to 0.26 ms for maximum-size tip +transactions. The block-wide duplicate check remains additional work. + +CPU time per tip block was 189.75→197.20 ms for the small fixture (~4% higher) +and 316.40→304.80 ms for the maximum-size fixture (~4% lower). The intended gain +is less work after arrival, not a blanket claim of lower CPU. Maximum-size +fixtures allocate roughly 37 MB fewer bytes per block by avoiding a second +message serialization; retaining opaque identities also has a memory cost. + +The initial runs without an explicit backend used the library's unconfigured +default and are excluded from these deployment-relevant results. + +## Validation and current limits + +Local affected-package tests, race tests for txverify/block/turbine/replay, Go +vet, and a full validator build passed. Tests cover legacy/v0/v1 canonical +identities, failed signatures, scratch reuse, pointer/order/blockhash mismatch, +storage ownership, JSON round-trips, early preparation and mixed canceled-batch +fallback. Existing turbine tests cover invalid and discarded prefixes and +completion cancellation. + +Historical deployment observations are retained in the +[archived evidence](streaming-preparation-evidence.md). Their unequal workloads +and epoch boundary do not establish an isolated FAST-score improvement. + +## Recovery from inconsistent prefetch metadata + +Missing/oversized retained ranges, partial identities and identity-binding +mismatches now log a warning and fall back to full signature verification of the +final block. Bounds are checked before slicing the final transaction array. Old +readers are joined first, including on cancellation. Valid final transactions +can therefore recover from an optimization bookkeeping fault; a successful +cached verdict for different bytes cannot authorize the final transaction. +Normal cached signature failures remain errors, as do failed re-verification, +cancellation and a closed verifier. No signatures or duplicate checks are skipped. + +Regression tests cover valid and invalid final blocks for each metadata fault, +cache identities matching the final transaction order, canceled-reader ownership +and verifier failure. The ordinary path retains exact-range/byte checks and +verified identity reuse. Full re-verification is exceptional and costs additional +work; this change is a correctness/availability fix, not a throughput claim. + +The `[sigverify]` starter configuration and its test moved here from the voting +branch, because this branch reads those keys and supports the legacy backend key. +The unrelated explicit `tuning.use_pool` template setting is omitted: enabling +pooling by default and fixing retained vote ownership belong to the runtime +branch. In the combined build, its configuration defaults still enable pooling. +The default remains two Turbine verification workers (bounded by GOMAXPROCS). +Many-core catch-up tuning remains a separate measurement question. diff --git a/docs/transaction-status-expiry.md b/docs/transaction-status-expiry.md new file mode 100644 index 000000000..538acb0f8 --- /dev/null +++ b/docs/transaction-status-expiry.md @@ -0,0 +1,61 @@ +# Batched transaction-status expiry + +Applying an asynchronous checkpoint still calls `TransactionStatusCache.Root` +on replay. Live Zen 5 instruction probes measured 43–101 ms inside that function. +The previous expiry path visited every key in every retired bank, even when an +entire recent-blockhash group could be discarded. + +Expiry now examines the expired and retained bank deltas by blockhash. It drops +fully expired groups directly. For a group spanning the cutoff, it either +subtracts the expired keys or rebuilds the visible reference counts from the +retained deltas, whichever requires fewer key visits. Retained unrooted banks +are included. Physical map reclamation is still Go GC work; this is not a claim +that memory reclamation costs disappear. + +The 300-root retention rule, immediate logical expiry, duplicate-key reference +counts, selected-parent validation, checkpoint format and immutable producer +views are unchanged. All index changes remain under the existing cache lock. +This does not move unsafe mutable state to another goroutine or delay expiry. +A long-lived blockhash with many transactions on both sides of the cutoff can +still require substantial per-key work. This patch reduces that work to the +smaller side; it does not give a constant-time worst-case bound. + +## Validation + +The replay race suite, replay vet and validator production build pass. New tests +compare exact visible indexes against the original per-key removal for 100 +random lineages with shared hashes, collisions and empty groups, then unwind +surviving banks. A Root integration test checks pinned producer views, +checkpoint bytes, restored duplicate detection and rooted-unwind rejection. + +M4 Pro, Go benchmark, single caller, two iterations per case. Each iteration +expires 128 banks of 33,760 unique keys (4,321,280 entries) and retains another +33,760 entries. Setup is outside the timer. The baseline invokes the original +per-key removal; the new path invokes batched expiry. These are **expiry-path** +measurements, not end-to-end Root/replay or a prediction of live FAST scores. + +| Recent-blockhash grouping | Old expiry | Batched expiry | +|---|---:|---:| +| Groups shared by four expired banks | 185–189 ms | 0.037–0.080 ms | +| One fully expired group | 604–614 ms | 0.025–0.026 ms | +| One group shared by expired and retained banks | 590 ms | 1.63–2.36 ms | + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusBatchExpiry$' -benchtime=1x -count=2`. + +## Native benchmark + +Ryzen 7 9700X, Go 1.26.4, original per-key expiry versus batched expiry. +Benchmarks ran with GOMAXPROCS=2, nice=15, one caller and three iterations per +case, while the validator and loader remained active. Setup and later GC are +excluded from the expiry timer. Each case expires 4,321,280 entries (128 banks +of 33,760) and retains 33,760 entries. These synthetic batches exceed the earlier +live stall samples and are not an end-to-end replay or FAST-score comparison. + +| Shape | Original expiry | New expiry | +|---|---:|---:| +| Four-bank blockhash groups | 306–311 ms | 0.049–0.057 ms | +| One fully expired blockhash group | 700–718 ms | 0.024–0.031 ms | +| Group crossing the retention boundary | 717–735 ms | 1.85–2.05 ms | + +[Historical evidence](status-checkpoint-expiry-evidence.md) preserves the original +source revisions, raw measurements and validation. diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md new file mode 100644 index 000000000..c74be8335 --- /dev/null +++ b/docs/transaction-status-publication.md @@ -0,0 +1,73 @@ +# Preparing transaction-status publication during execution + +Replay previously built the immutable per-bank transaction-status delta and grew the visible duplicate index only after execution and bank-state publication. In a prior live sample of 25 large blocks, TransactionStatusCommit took 7.704 ms median and 10.206 ms maximum. Those live timings motivate this change; they are not the controlled benchmark baseline below. + +Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. + +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. + +TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. + +## Native benchmark + +AMD Ryzen 7 9700X (Zen 5), Go 1.26.4, GOMAXPROCS=2. Tests ran in a separate process on the validator host with Nice=15 and a 200% CPU quota; the validator and loader continued running. This is a shared-host microbenchmark, with observable timing variation. Five samples per case, ten iterations per sample; values below are medians of sample means, not per-block percentiles. + +Each block has 33,760 unique prepared message identities spread across one or four recent blockhashes. Existing-group cases seed 33,760 different ancestor transactions. Fixture creation, hashing, seeding and unwind are untimed. Existing maps retain capacity after unwind: the first timed commit's growth is amortized across the ten iterations. This does not model an index growing indefinitely across live blocks. + +The frozen baseline functions exactly match alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions use the same prepared identities, parent/duplicate checks and fixtures. + +| Recent blockhash groups | Parent has keys in these groups | Baseline commit | Sized maps, inline | Preparation + commit, no overlap | Commit after preparation | +|---|---|---:|---:|---:|---:| +| 1 | No | 4.990 ms | 3.332 ms | 3.393 ms | 1.587 ms | +| 1 | Yes | 4.991 ms | 4.657 ms | 4.434 ms | 2.757 ms | +| 4 | No | 4.625 ms | 3.744 ms | 5.916 ms | 2.205 ms | +| 4 | Yes | 5.224 ms | 4.644 ms | 4.949 ms | 2.858 ms | + +The last column deliberately excludes delta preparation: it measures the work remaining if execution hides preparation completely. It is not total replay or CPU work. Total publication allocations with new groups fell from approximately 6.30 MB to 3.15 MB per block. Existing-group allocation figures include the amortized first growth described above. + +The four-new-group total-work sample was slower. Preserve that result rather than claiming improvement in every sample. A subsequent baseline/candidate/candidate/baseline comparison of that same case, with 50 iterations per sample, measured baseline **4.400 and 4.565 ms**, candidate **2.985 and 3.131 ms**. This supports a reduction in work but does not isolate the cause of the earlier timing variation. + +## Execution contention and small blocks + +A separate controlled benchmark performs 4,096 load-and-execute calls using the existing transfer fixture while preparing 33,760 independent status keys. It does not commit transfer accounts, and its status fixture differs from the repeated transfer fixture. It tests scheduling/allocation contention, not whole-block replay or a valid block workload. + +With two Go execution threads, the final implementation measured **20.678 ms baseline**, **18.574 ms with sizing alone**, and **17.091 ms with overlap**. Execution itself measured 15.070, 14.369 and 14.967 ms respectively. Thus preparation competed with execution relative to sizing alone, but the shorter final stage outweighed that cost in this controlled workload. These are separate medians and need not add exactly. + +The initial unrestricted version showed no additional total-time benefit from overlap with GOMAXPROCS=1. Tiny-block measurements also showed roughly a microsecond of avoidable scheduling overhead. The final implementation therefore does no background preparation with one Go execution thread or at most 32 transactions. Empty and one-transaction cases retain the baseline allocation counts. The 32-transaction case benefits from sizing without launching a worker. Threshold and single-thread behavior have regression coverage. + +## Validation and limits + +Full replay and block race suites passed on both Zen 5 and M4 Pro. Metrics has no tests. Native vet for replay/metrics and the validator build passed. Tests cover fork replacement introducing a duplicate after preparation, concurrent sibling publication, stale identity binding, changed snapshot slice offsets, rejected/incomplete banks, mismatched preparation, pinned views, snapshot restore, unwind, empty banks and scheduling boundaries. + +Raw logs, source hashes, summaries and the alternating recheck are in [results/status-publication/2026-09-15](https://github.com/Overclock-Validator/mithril/blob/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results/status-publication/2026-09-15). The baseline comparison covers only status publication. No live replay or FAST improvement is claimed. The staging binary was not deployed; the existing validator remained active and voting throughout the tests. + +Reproduce from this branch: + +```sh +GOMAXPROCS=2 go test -race -p 2 ./pkg/replay ./pkg/block ./pkg/metrics -count=1 +GOMAXPROCS=2 go vet -p 2 ./pkg/replay ./pkg/metrics +GOMAXPROCS=2 go build -p 2 ./cmd/mithril +GOMAXPROCS=2 go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusPublication$' -benchtime=10x -count=5 +go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatus(ExecutionOverlap|SmallPublication)$' -benchtime=100ms -count=5 -cpu=1,2 +``` + +## Reusing pre-execution ancestor validation + +`ProcessBlock` now carries a private validation receipt from its successful ancestor scan to status publication. Under the commit lock, an unchanged receipt avoids scanning all transaction messages again. Publication still checks block binding, complete coverage and parent lineage every time; a missing, foreign or stale receipt performs the full ancestor scan. Direct `CommitBlock` callers retain the full scan. + +The receipt is bound to the cache instance and exact immutable prepared-identity pointer. Visible-index insertion/removal, tip binding, root/prune and restore invalidate the version, including empty commits. Committing and then unwinding back to an identical parent cannot revive a receipt. Version saturation disables reuse permanently rather than wrapping. Snapshot/Agave recovery creates a new cache instance. Receipts are never persisted, and no checkpoint format, durability, voting-resume or crash-recovery guarantee changes. + +The publication benchmark adds `validated_commit` and `invalidated_commit` alongside `prepared_commit`. All three exclude delta preparation and the pre-execution scan. The first reuses that scan; the second calls `Root` between validation and publication, forcing revalidation. Each iteration unwinds and obtains a fresh receipt outside the timer. These are incremental publication comparisons, not the full PR against alpenglow-dev or per-block tail latency. Tests exercise fork replacement introducing duplicates, concurrent sibling commits, cross-cache and cross-identity misuse, snapshot replacement, pruning/root invalidation, binding changes, transaction replacement and version saturation. + +Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five samples × 20 iterations, Nice=19 / 200% CPU quota on the running validator host): + +| Recent blockhash groups | Existing ancestor groups | Full recheck | Reused validation | Invalidated validation | +|---|---|---|---|---| +| 1 | yes | 2.510 ms | 1.364 ms | 2.536 ms | +| 4 | yes | 2.412 ms | 1.283 ms | 2.440 ms | +| 1 | no | 1.461 ms | 1.472 ms | 1.769 ms | +| 4 | no | 1.323 ms | 1.395 ms | 1.303 ms | + +Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. + +Original run artifacts are retained in the [evidence archive](status-checkpoint-expiry-evidence.md). diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md new file mode 100644 index 000000000..42e4bf311 --- /dev/null +++ b/docs/transaction_sigverify_streaming.md @@ -0,0 +1,159 @@ +# Transaction signature verification during shred arrival + +Turbine verifies transaction signatures as complete entry batches arrive. The +default shared transaction pool has `min(2, GOMAXPROCS)` workers and targets eight signature +lanes. A ready batch containing four transactions runs immediately; there is no +timer or minimum occupancy requirement. The same policy handles live reception +and repair catch-up without a mode transition or a 200 ms batching delay. + +## Order of work + +1. The receiver authenticates the shred and performs existing assembly/FEC + recovery. The packet reader advances a contiguous data-shred frontier and + attempts a nonblocking background enqueue. Assembly retains an authenticated + root and a snapshot of its source shred for each FEC set when available. +2. Two background preparation workers copy and decode complete DATA_COMPLETE + entry batches. A batch may span several FEC sets. They submit its immutable + transactions to the shared signature pool while later shreds arrive. +3. Signature groups contain available transactions up to the configured lane + target. Transactions with multiple signatures remain indivisible. Large + already-decoded requests bundle four vector groups per dispatch job (normally + 32 one-signature transactions). Requests smaller than + `2 * workers * batch_target * 4` transactions keep one group per job; the + default threshold is 128 ready transactions. A rolling per-request window + refills when any job finishes, without a wave barrier or batching timer. +4. Once the full slot is assembled, completion compares shred slices directly + against the cached component bytes, including padding. Cache hits avoid a + second component buffer; misses allocate a fresh buffer at its exact size. + Complete contiguous slots supply shred order directly, avoiding two sorts. + Completion processes all Alpenglow markers and FEC roots and constructs the + final ordered block. It submits any transactions not already covered and joins + signature work for the retained batches before marking the block verified. +5. The existing replay pipeline receives the verified block. This change does + not execute transactions before full-slot admission, alter execution batching, + or move parent-dependent validation ahead of its required state. + +The 200 ms slot interval provides an opportunity to overlap work. It is not a +mandatory local wait: replay can continue as soon as the block is available and +its checks complete. If all shreds arrive in a burst during catch-up, full-slot +completion uses the same efficient groups with whatever early work was able to +start. Four workers can improve catch-up latency but occupy more cores at once. + +## Bounds and correctness + +- At most eight slot generations hold early preparation reservations, with a + combined 64 MiB budget for raw component bytes and a 1 MiB per-component limit. + Decoded transaction objects add heap overhead. Saturation skips optional early + work; normal full-slot verification still covers every retained transaction. +- One queued/active preparation token per generation coalesces packet arrivals. + Verifier requests and each request's outstanding jobs are also bounded. + Request admission can wait behind existing requests; this is not a strict + replay-head priority scheduler. +- No transaction decoding or verifier admission occurs under the assembler mutex or on the + packet reader. A gap prevents early component decoding until recovery or + arrival closes it. Duplicate shreds do not create duplicate requests. +- Cached results belong to one generation, shred range and exact byte sequence. + Reset/eviction cancels that generation. Reservations remain charged until + admitted readers have relinquished their transaction buffers. +- A FEC root cache retains at most one source snapshot per FEC state, in addition + to the entry-prefetch budget. Completion preserves the deterministic choice of + the lowest-index non-recovered data proof, then lowest coding position. Cached + roots require the same source, parsed root inputs, and exact payload bytes; + mismatches, unauthenticated callers, and spool hydration recompute the root. +- `UpdateParent` can discard an optimistic prefix. Parse and marker checks still + cover that prefix, while its transaction signature verdict is discarded along + with its transactions. Retained signatures must all pass before replay. +- Cancellation is not an invalid-signature verdict. A retry on the same slot + generation verifies transactions again if an earlier request was canceled. + An admitted job finishes its first vector group; cancellation can skip later + groups in that job. The request joins all admitted jobs before releasing input. + +## Configuration and observability + +```toml +[sigverify] +backend = "auto" +workers = 0 # min(2, GOMAXPROCS) +batch_target = 8 # 4 or 8; short groups never wait to fill +disable_shred_overlap = false +``` + +Equivalent CLI flags are `--sigverify-workers`, `--sigverify-batch-target` and +`--sigverify-disable-shred-overlap`. These settings apply to Turbine transaction +signatures; TPU, shred signatures, consensus BLS and replay's fallback verifier +retain their existing configuration. + +Use `TurbineFullToReady` to measure residual wall time after the slot becomes +complete. `TurbineEarlyVerifiedTransactions` counts retained transactions whose +verification finished before full assembly. `TurbineTransactionSigverify` now +measures completion's outstanding-signature join/fallback work. The remaining +wait for an already claimed background preparation job, including +any outstanding admission delay, appears in `TurbineEarlyPreparationWait`, +separately from active completion decoding. It does not sum every background +admission wait. Early parse and signature durations are component sums observed +at completion; +a discarded prefix still being verified is not included. These durations +overlap reception and each other and must not be added as sequential stages +or interpreted as CPU time. + +## Benchmark scope + +`BenchmarkTransactionVerificationFlow` compares two/four workers and targets +four/eight using catch-up, synthetic 200 ms component arrivals, and sparse +four/seven/eight-transaction components. It supports captured public transaction +fixtures and generated, distinct valid transactions of exactly 228 or 1,232 wire +bytes. Decoding and fixture construction are outside those pool measurements. + +The execution contention probe runs the real transfer load/execute benchmark +alongside signature work on the same eight physical cores. Its separate +processes measure hardware/OS contention, excluding shared Go scheduler/heap +effects, full-block dependency planning, commit and live network timing. Pool +throughput alone is insufficient evidence of end-to-end replay improvement. + +Measured Zen 5 results, raw logs, and validation details are in the +[September 12 benchmark report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/sigverify-streaming/2026-09-12-zen5/README.md). +The subsequent [direct cache-comparison report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/sigverify-direct-cache/2026-09-12-zen5/README.md) +isolates the removal of redundant component-buffer construction at completion. +The [completion follow-up report](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/completion-followup/2026-09-12-zen5/README.md) +measures direct ordering, authenticated-root reuse, and the four-vector job policy. + +The [standalone PR review](https://github.com/Overclock-Validator/mithril/blob/1c1171d3661d0404b013a9bf9391e23eb660706e/docs/results/streaming-pr-review/2026-09-13/README.md) +records extraction onto current `alpenglow-dev`, the small shared component-boundary +prerequisite, final allocation improvement, and the scope of the live trial. + +## Worker default compatibility + +The automatic transaction-verifier default changes from `(GOMAXPROCS + 1) / 2` workers to +`min(2, GOMAXPROCS)`, including when shred overlap is disabled. This favors spare +CPU capacity for execution and other verification at the tip. It is not a claim +of maximum catch-up throughput on every core count or backend. Set +`--sigverify-workers N` or `[sigverify] workers = N` explicitly when tuning a +larger machine; disabling overlap alone does not restore the previous worker +count. Existing two/four-worker contention measurements are in the September 12 +report above. No 32-core comparison was performed. + +## Reserved admission for completion + +Decoded prefetch components now use a separate admission class. The existing total request limit remains `2 * workers`; at most `2 * workers - 1` requests may prefetch. Thus the default two-worker pool keeps four total permits, with at most three occupied by prefetch. Waiting completion/full-block recovery requests win the next free permit over prefetch. Their admission, cancellation and close registration share the verifier mutex; notification channels are allocated only when callers must wait. + +No worker or job queue is added, and verification, vector width, job grouping and per-request rolling windows are unchanged. Accepted jobs finish normally and are joined before transaction memory can be reused. No signature checks are skipped. Prefetch may wait while completion callers remain queued and resumes when that backlog drains. This is completion-class priority, not exact replay-head priority: future-slot completions also qualify, and already admitted prefetch work is not promoted or preempted. Checkpoint, persistence and voting-recovery contracts are unchanged. + +### Saturation benchmark + +`BenchmarkVerifierCompletionReservation` verifies four already-ready prefetch components (256 or 4,096 signed 228-byte transactions each) plus a 32-transaction completion request. All requests are joined; total-work timing includes all four components. Two verifier workers, eight signature lanes, four vector groups, GOMAXPROCS=8, Narya r51, Ryzen 9700X / Go1.26.4. Three runs of 100 iterations each, Nice19 and a 200% CPU quota on the shared validator host. Test intervals are excluded from live FAST comparisons. + +The before comparison uses the previously deployed source (status-validation combined build, SHA256 `2c81fc403e8a6eca73b87ede51890041347e1af0e3c63df005fcfeb26448e872`) with only the benchmark added via a Go test overlay. The candidate also measures the shared-class control to distinguish policy from incidental overhead. These are incremental admission results, not the whole PR versus alpenglow-dev. + +For four 4,096-transaction components, medians of the three per-run statistics were: + +| Measurement | Previous deployment | Reserved admission | +|---|---:|---:| +| Completion admission p50 | 37.19 ms | 0.000742 ms | +| Completion admission p99 | 45.78 ms | 0.004599 ms | +| Completion finished p99 | 47.31 ms | 1.488 ms | +| All work finished p50 | 39.43 ms | 39.11 ms | +| All work finished p99 | 49.09 ms | 50.65 ms | + +Completion-finished p99 ranged 46.43–54.52 ms before and 1.477–1.719 ms after. Admission p99 ranged 44.27–53.76 ms before and 0.003206–0.06401 ms after. Every iteration reached its intended request occupancy. For 256-transaction components, completion-finished p99 medians were 3.215→1.165 ms. Shared-host scheduling introduces variation; reserving admission does not remove queued-job or CPU delays, and these 100-sample tails are not a live p99/FAST claim. Total-work throughput was roughly unchanged; no total-work tail improvement is claimed. + +Original run artifacts are retained in the [evidence archive](streaming-preparation-evidence.md). diff --git a/docs/turbine-relay-buffers.md b/docs/turbine-relay-buffers.md new file mode 100644 index 000000000..3ee75f3f0 --- /dev/null +++ b/docs/turbine-relay-buffers.md @@ -0,0 +1,60 @@ +# Turbine relay buffer ownership + +Each retransmit worker reuses a private peer-result slice. Weighted shuffle, +tree placement, address filtering and fanout are unchanged. The exported +`RetransmitPeers` API still returns independently allocated result storage. +Workers clear their entire scratch slice after each send so it cannot retain +addresses from an old cluster snapshot. + +`SubmitFrom` copies a packet into exclusively owned storage before returning +to its caller. Canonical packets use a fixed-size, GC-reclaimable `sync.Pool`; +oversized inputs retain the independent allocation path. Queue admission +transfers ownership to the worker. The worker returns storage only after all +synchronous sends and retries finish, including error and no-peer paths. +Queue rejection returns storage immediately. Shutdown closes admission under +a short lock, joins workers, then drains remaining copies. The channel stays +open for concurrent submitters, which cannot enqueue after admission closes. + +`packetBatchSender.Send` borrows both packet bytes and peer addresses only +until return, even on partial sends or errors. Implementations that retain +either must copy them. A pooled packet must never escape this lifetime. +The admission lock is not held during authentication, routing or socket I/O. + +## Measurement + +`BenchmarkRetransmitPipeline` measures deduplication, packet copying, queue +handoff, weighted routing and dispatch to a mock sender. Run with: + +```sh +GOMAXPROCS=2 go test ./pkg/turbine -run '^$' \ + -bench '^BenchmarkRetransmitPipeline$' -benchmem -benchtime=20000x -count=3 +``` + +On a Ryzen 7 9700X (Zen 5), Go 1.26.4, one producer and one relay worker, +the incremental comparison against the relay implementation at +`c40ac9e8ca8aa09a9f2a8759a231199b0f90346e` was: + +| Gossip contacts | Before ns/shred | After ns/shred | Before B/shred | After B/shred | Allocations before → after | +|---|---:|---:|---:|---:|---:| +| 90 | 3,906 | 3,761 | 3,784 | ~729 | 8 → 6 | +| 512 | 23,789 | 22,056 | 3,784 | ~729 | 8 → 6 | + +Times are medians of six samples per version, in baseline/candidate/candidate/ +baseline order, three samples per round. Both versions use the same benchmark +and combined validator source; only the two relay production files change. +Each iteration submits a distinct 1,203-byte data shred with cached topology; +the benchmark checks that none are dropped and waits for workers to finish. +The fixture includes staked and zero-stake peers. Contact count does not mean +that every shred has that many recipients: tree position determines forwarding. + +Allocation fell about 81%. The 4–7% median timing improvement is modest and +noisy on the shared validator host; one candidate sample was slower than every +baseline sample in its case. These are amortized in-memory pipeline times, +not network delivery latency or live FAST-score gains. Socket syscalls, parent +authentication and retransmitter signing are excluded from this fixture. + +Routing tests compare against a full weighted permutation, including both +shuffle modes and missing/unroutable contacts. Ownership tests exercise caller +buffer reuse, queue pressure, send retries, concurrent routing and shutdown. +The full Turbine race suite passed locally and natively; native vet and the +combined validator build passed. Historical run artifacts are linked from [the evidence archive](streaming-preparation-evidence.md). diff --git a/docs/vote-delivery-persistence-evidence.md b/docs/vote-delivery-persistence-evidence.md new file mode 100644 index 000000000..1ae1f5cfa --- /dev/null +++ b/docs/vote-delivery-persistence-evidence.md @@ -0,0 +1,14 @@ +# Vote Delivery Persistence: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/54b233ff0e27fb929644f7d53bf4a699cb590cd8) +(tag `review-evidence-20260916-vote-delivery-persistence`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/54b233ff0e27fb929644f7d53bf4a699cb590cd8/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. diff --git a/docs/votor-peer-isolation.md b/docs/votor-peer-isolation.md new file mode 100644 index 000000000..061b0f80a --- /dev/null +++ b/docs/votor-peer-isolation.md @@ -0,0 +1,99 @@ +# Votor outbound peer isolation + +A stalled QUIC peer could previously occupy all 32 shared send/connect workers. +quic-go's `SendDatagram` blocks when its 32-frame connection queue is full. +Repeated jobs for that peer could therefore prevent votes reaching healthy +peers, even while the broadcaster reported zero queue drops. + +Each authenticated connection now owns one sender and a FIFO queue of at most +256 encoded messages. The encoded payload is immutable and shared across peer +queues. The existing bounded worker pool handles connection attempts only; +`VotorBroadcasterConfig.Workers` controls that pool. A blocked connection cannot +consume another peer's sender or a connection worker. + +A watchdog checks every 100 ms whether the active datagram's peer-queue wait +plus its current `SendDatagram` duration has reached one second. Occasional QUIC +PTO probes can free queue entries without proving delivery; they no longer +restart this budget for an old backlog. Dequeue also retires a connection before +feeding an already-one-second-old entry into QUIC, even if sends keep completing +between watchdog ticks. The effective active-send bound is one second from +fanout enqueue, plus up to one watchdog interval and runtime scheduling delay. + +Retirement closes that connection, wakes a blocked sender, discards its queued +copies and requests a bounded reconnect. This is an operational limit on local +queueing, not a consensus validity deadline or a remote-delivery guarantee. It +adds no per-send timer or goroutine. An idle connection is not expired merely +because its previous send had a long queue delay. + +## Failure and ordering semantics + +- The global `Enqueue` contract is unchanged: rejecting a newly signed message + from the global queue returns an error to the voting engine. +- Per-peer fanout remains best effort. A full peer queue rejects that peer's + newest copy, increments both `PeerQueueDrops` and the existing + `MessagesDropped` counter, and continues sending to other peers. +- Messages queued on a failed, removed or replaced connection are discarded and + counted in `PeerQueueDiscarded`. They are not transferred to a new connection + or address. A datagram expired at dequeue is counted in `PeerQueueDiscarded`; + a failed `SendDatagram` call is counted in `PeerSendErrors`. The watchdog and + dequeue path claim retirement under the sender mutex, counting one timeout. +- Every sender exit requests a bounded, deduplicated reconnect. This covers + both a remote-close notification and closure detected while dequeuing. Peer + departure, shutdown and a healthy replacement suppress obsolete requests; + normal remote closure no longer relies on the periodic reconciliation tick. +- Messages for disconnected peers increment `PeerSendsSkipped`, as before. + This change adds no automatic application-level retransmission. +- A single sender preserves local enqueue order within its connection. QUIC + datagrams themselves still do not guarantee arrival or ordering. +- Shutdown cancels connection attempts, closes the connections outside the + broadcaster mutex, drains their queues, and waits for sender exit. + +Signing decisions, persisted vote history, durable slot reservations, and +certificate validation are unchanged. + +## Observability + +The voting log adds `peer_queue_drops`, `peer_queue_discarded`, +`peer_send_timeouts`, and `peer_queue_max_delay`. These counters/high-water marks +survive reconnects. `PeerQueueMaxDelay` measures time from fanout enqueue to +sender dequeue, including entries retired before reaching QUIC; it does not +include time blocked inside `SendDatagram`. + +Voting snapshots also expose `broadcast_peer_queues` with each current peer's +identity, address, queue depth, time in the active send, last/max queue delay, +and queue drops. Durations in the JSON snapshot use nanoseconds. Per-connection +statistics disappear when that connection is replaced; totals remain available. +Successful `PeerSends` only means quic-go accepted the datagram, not that a +remote validator received it. + +## Regression coverage + +`TestVotorBroadcasterIsolatesBlockedPeer` establishes two real loopback QUIC +connections, then blackholes one UDP path. It observes the blocked +`datagramQueue.Add` stack and requires a healthy-peer vote to arrive within +250 ms while the failed connection is still open. It also covers peer queue +overflow, watchdog reconnection, fresh traffic after reconnect, shutdown with a +blocked sender, peer departure, and replacement of a blocked peer's address. +Its connection-retirement deadline is measured from the blackhole, with explicit +scheduler slack; the healthy-peer latency assertion remains 250 ms. + +Deterministic regressions reproduce a fresh send following an old queue wait, +remote closure while idle and while dequeuing, and an already-aged queue entry. +The reconnect fixture has no reconciliation loop, so a timer cannot hide a missed +reconnect trigger. The pre-fix failures and fresh validation are retained under +[historical evidence](https://github.com/Overclock-Validator/mithril/blob/54b233ff0e27fb929644f7d53bf4a699cb590cd8/docs/results/review-fixes/2026-09-15). + +The signature-verification config template and its tests now belong to the +streaming branch, which reads those settings. This standalone voting branch +retains its base branch's supported `tuning.sigverify_backend` template. + +```sh +go test -race ./pkg/alpenglow ./pkg/consensus -count=1 -timeout=180s +go vet ./pkg/alpenglow ./pkg/consensus +go build ./cmd/mithril +``` + +This is a transport regression, not a prediction of FAST score improvement. +The live validator needs a separately validated integrated build and matching +probe addresses before deployment; the PR checkout does not include every +change in the currently enrolled validator binary. diff --git a/go.mod b/go.mod index 4607b7603..2463c0b17 100644 --- a/go.mod +++ b/go.mod @@ -5,7 +5,7 @@ go 1.26.4 replace github.com/gagliardetto/binary => github.com/palmerlao/binary v0.0.0-20250617062159-3054b4d33aed require ( - github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d + github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713 github.com/cespare/xxhash/v2 v2.3.0 github.com/charmbracelet/bubbles v0.21.1-0.20250623103423-23b8fd6302d7 github.com/charmbracelet/bubbletea v1.3.10 diff --git a/go.sum b/go.sum index 069571ee5..086dc43c2 100644 --- a/go.sum +++ b/go.sum @@ -12,8 +12,8 @@ github.com/Overclock-Validator/crypto v0.0.0-20250307094320-aaf52fac5261 h1:Y715 github.com/Overclock-Validator/crypto v0.0.0-20250307094320-aaf52fac5261/go.mod h1:ZhRHOaVg8I1gg0VK4wmqOQPnlgPgKFT9McZ+TCW/hBA= github.com/Overclock-Validator/gnark-crypto v0.0.0-20250309203346-2a67ed08a105 h1:mP6FWHZ8ddcmbE8UTrVVI2Mi2c24aqX/8p12Vn6zokQ= github.com/Overclock-Validator/gnark-crypto v0.0.0-20250309203346-2a67ed08a105/go.mod h1:Poczuq3dbt+CwyTKgOjGaEwJOMP7YxQobF7QhgNcguk= -github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d h1:ipaL+9MHKeI8QIfWneId0VLa+STLRM1e6MnvZhQyhPU= -github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260726222623-da0d045dae9d/go.mod h1:B7/xqV/5NtGJa8OlZAa9TRMHgeIE+VEJNiPzMP4FrIg= +github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713 h1:nRAD+snanlR/sX4W5rkrxr6+sYBZ3Hl2xZQ6HCpM0Hc= +github.com/Overclock-Validator/narya-ed25519 v0.0.0-20260730051143-c265ee966713/go.mod h1:B7/xqV/5NtGJa8OlZAa9TRMHgeIE+VEJNiPzMP4FrIg= github.com/Overclock-Validator/solana-snapshot-finder-go v0.0.0-20260223201452-d8363b514fc0 h1:elgavEQb8l7Zn3gS3Y+2/98PlUylOWdlM3V1VumQ7mA= github.com/Overclock-Validator/solana-snapshot-finder-go v0.0.0-20260223201452-d8363b514fc0/go.mod h1:XbqbvMA2NKeosY0w3WdBOpCg2eYJesBjfE4cNt9HSE8= github.com/Overclock-Validator/wide v0.0.0-20250221123529-f80959d02044 h1:ph9gnWIY116AWT/iCfXoPe9/cn2aWx2uJBuLdf/LyEE= diff --git a/pkg/accounts/mem_accounts.go b/pkg/accounts/mem_accounts.go index 4158be553..630ffb420 100644 --- a/pkg/accounts/mem_accounts.go +++ b/pkg/accounts/mem_accounts.go @@ -1,7 +1,6 @@ package accounts import ( - "fmt" "sync" "github.com/Overclock-Validator/mithril/pkg/base58" @@ -13,6 +12,16 @@ type MemAccounts struct { mu *sync.RWMutex } +// A miss is an ordinary step when falling back to parent accounts. Defer the +// diagnostic encoding until it is needed, and copy the key so callers can reuse it. +type missingMemAccountError struct { + key [32]byte +} + +func (e *missingMemAccountError) Error() string { + return "no such account " + base58.Encode(e.key[:]) + " found" +} + func NewMemAccounts() MemAccounts { return MemAccounts{ Map: make(map[[32]byte]*Account), @@ -32,7 +41,7 @@ func (m MemAccounts) GetAccount(pubkey *[32]byte) (*Account, error) { defer m.mu.RUnlock() acct, ok := m.Map[*pubkey] if !ok { - return nil, fmt.Errorf("no such account %s found", base58.Encode(pubkey[:])) + return nil, &missingMemAccountError{key: *pubkey} } return acct, nil } @@ -40,7 +49,7 @@ func (m MemAccounts) GetAccount(pubkey *[32]byte) (*Account, error) { func (m MemAccounts) GetAccountWithoutLock(pubkey solana.PublicKey) (*Account, error) { acct, ok := m.Map[pubkey] if !ok { - return nil, fmt.Errorf("no such account %s found", base58.Encode(pubkey[:])) + return nil, &missingMemAccountError{key: pubkey} } return acct, nil } diff --git a/pkg/accounts/mem_accounts_test.go b/pkg/accounts/mem_accounts_test.go index bd55cb218..4a614dd9f 100644 --- a/pkg/accounts/mem_accounts_test.go +++ b/pkg/accounts/mem_accounts_test.go @@ -3,8 +3,23 @@ package accounts import ( "testing" "time" + + "github.com/gagliardetto/solana-go" ) +func TestMemAccountMissingErrorRetainsLookupKey(t *testing.T) { + mem := NewMemAccounts() + key := [32]byte{} + _, lockedErr := mem.GetAccount(&key) + _, unlockedErr := mem.GetAccountWithoutLock(solana.PublicKey(key)) + key[0] = 99 // Lookup callers may reuse their key storage before reporting an error. + for _, err := range []error{lockedErr, unlockedErr} { + if err == nil || err.Error() != "no such account 11111111111111111111111111111111 found" { + t.Fatalf("missing error lost its original key: %v", err) + } + } +} + func TestMemAccountsReadsAreConcurrent(t *testing.T) { mem := NewMemAccounts() var key [32]byte diff --git a/pkg/accounts/overlay_test.go b/pkg/accounts/overlay_test.go index f40a26dfc..682e47217 100644 --- a/pkg/accounts/overlay_test.go +++ b/pkg/accounts/overlay_test.go @@ -411,3 +411,42 @@ func TestOverlayDeltaAccountsIncludesOverride(t *testing.T) { assert.Equal(t, pk(1), delta[0].Key) assert.Equal(t, uint64(99), delta[0].Lamports) } + +func TestWorkingSetPromotionChunkBoundaries(t *testing.T) { + w := NewWorkingSet() + for _, slot := range []uint64{5, 7, 9, 11} { + w.Add(slot, []*Account{uoAcct(1, slot), uoAcct(2, slot+100)}) + } + for _, tc := range []struct { + through uint64 + limit int + partial bool + slots []uint64 + }{ + {4, 2, true, nil}, {5, 2, false, nil}, {7, 2, false, []uint64{5, 7}}, + {11, 2, false, []uint64{5, 7}}, {9, 4, true, []uint64{5, 7, 9}}, + {9, 4, false, nil}, {11, 0, true, nil}, {11, -1, false, nil}, + } { + got := w.PromotionChunk(tc.through, tc.limit, tc.partial) + var slots []uint64 + for _, sd := range got { + slots = append(slots, sd.Slot) + require.Len(t, sd.Delta, 2) + for _, acct := range sd.Delta { + require.True(t, acct.Lamports == sd.Slot || acct.Lamports == sd.Slot+100) + } + } + require.Equal(t, tc.slots, slots) + } + // Preparing a job leaves the live suffix intact. Once the caller commits + // and promotes a prefix, the next chunk must start at the surviving slot. + require.Equal(t, 4, w.HeldSlots()) + w.PromotePrefix(7) + chunk := w.PromotionChunk(11, 2, false) + require.Equal(t, []uint64{9, 11}, []uint64{chunk[0].Slot, chunk[1].Slot}) + require.Zero(t, testing.AllocsPerRun(100, func() { + if w.PromotionChunk(9, 2, false) != nil { + panic("partial chunk escaped") + } + })) +} diff --git a/pkg/accounts/working_set.go b/pkg/accounts/working_set.go index d88725f24..083a4405d 100644 --- a/pkg/accounts/working_set.go +++ b/pkg/accounts/working_set.go @@ -134,17 +134,37 @@ func (w *WorkingSet) PromotionPrefix(through uint64) []SlotDelta { w.mu.RLock() defer w.mu.RUnlock() - var batch []SlotDelta - for _, slot := range w.order { // ascending - if slot > through { - break - } + return w.promotionChunkLocked(through, len(w.order), true) +} + +// PromotionChunk returns at most maxSlots oldest held slots through the caller's +// verified promotion bound. Unless allowPartial is set, an incomplete chunk +// returns nil before allocating or collecting account writes. Selection and +// collection share one read lock, so pruning cannot change the selected prefix. +// This only prepares borrowed account pointers; it does not commit, prune, or +// advance durability. Callers still own finality checks and durable commit order. +func (w *WorkingSet) PromotionChunk(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + w.mu.RLock() + defer w.mu.RUnlock() + return w.promotionChunkLocked(through, maxSlots, allowPartial) +} + +func (w *WorkingSet) promotionChunkLocked(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + count := 0 + for count < len(w.order) && count < maxSlots && w.order[count] <= through { + count++ + } + if count == 0 || (!allowPartial && count < maxSlots) { + return nil + } + batch := make([]SlotDelta, count) + for i, slot := range w.order[:count] { layer := w.bySlot[slot] delta := make([]*Account, 0, len(layer.writes)) for _, a := range layer.writes { delta = append(delta, a) } - batch = append(batch, SlotDelta{Slot: slot, Delta: delta}) + batch[i] = SlotDelta{Slot: slot, Delta: delta} } return batch } diff --git a/pkg/alpenglow/broadcaster.go b/pkg/alpenglow/broadcaster.go index bd8e3ee7c..557ea997d 100644 --- a/pkg/alpenglow/broadcaster.go +++ b/pkg/alpenglow/broadcaster.go @@ -18,7 +18,7 @@ import ( const ( defaultVotorBroadcastQueue = 1024 - defaultVotorSendWorkers = 32 + defaultVotorConnectWorkers = 32 defaultVotorPeerJobQueue = 16384 defaultVotorPeerRefreshInterval = time.Second ) @@ -35,7 +35,8 @@ type VotorBroadcasterConfig struct { ShredVersion uint16 Peers VotorPeerSource QueueSize int - Workers int + // Workers bounds concurrent connection attempts. Sends are isolated per connection. + Workers int } type VotorBroadcasterStats struct { @@ -50,28 +51,25 @@ type VotorBroadcasterStats struct { ConnectionAttempts uint64 ConnectionErrors uint64 ConnectionJobsDropped uint64 + PeerQueueDrops uint64 + PeerQueueDiscarded uint64 + PeerSendTimeouts uint64 + PeerQueueMaxDelay time.Duration + PeerQueues []VotorPeerQueueStats LastPeerSendError string LastPeerSendErrorAt time.Time LastConnectionError string LastConnectionErrorAt time.Time } -type votorPeerJobKind uint8 - -const ( - votorPeerJobConnect votorPeerJobKind = iota - votorPeerJobSend -) - type votorPeerJob struct { - kind votorPeerJobKind - peer VotorPeer - payload []byte + peer VotorPeer } type votorConnection struct { - addr string - conn *quic.Conn + addr string + conn *quic.Conn + sender *votorPeerSender } type votorDial struct { @@ -109,6 +107,10 @@ type VotorBroadcaster struct { connectionAttempts atomic.Uint64 connectionErrors atomic.Uint64 connectionJobsDropped atomic.Uint64 + peerQueueDrops atomic.Uint64 + peerQueueDiscarded atomic.Uint64 + peerSendTimeouts atomic.Uint64 + peerQueueMaxDelay atomic.Int64 } func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) { @@ -122,7 +124,7 @@ func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) cfg.QueueSize = defaultVotorBroadcastQueue } if cfg.Workers <= 0 { - cfg.Workers = defaultVotorSendWorkers + cfg.Workers = defaultVotorConnectWorkers } certificate, err := newVotorQUICCertificate(cfg.Identity) if err != nil { @@ -151,7 +153,7 @@ func NewVotorBroadcaster(cfg VotorBroadcasterConfig) (*VotorBroadcaster, error) } b.wg.Add(2 + cfg.Workers) for range cfg.Workers { - go b.sendLoop() + go b.connectLoop() } go b.broadcastLoop() // Populate the desired set and queue the first bounded preconnects before @@ -199,71 +201,43 @@ func (b *VotorBroadcaster) broadcastLoop() { b.recordSendError(VotorPeer{}, fmt.Errorf("encode Votor message: %w", err)) continue } - peers, skipped := b.connectedPeers() + senders, skipped := b.connectedSenders() b.sendsSkipped.Add(uint64(skipped)) - for _, peer := range peers { - job := votorPeerJob{kind: votorPeerJobSend, peer: peer, payload: payload} - select { - case b.jobs <- job: - case <-b.done: - return - default: - b.dropped.Add(1) - } + job := votorDatagram{payload: payload, queuedAt: time.Now()} + for _, sender := range senders { + sender.enqueue(job) } } } } -func (b *VotorBroadcaster) sendLoop() { +// Connection attempts never occupy a peer's sender or delay connected peers. +func (b *VotorBroadcaster) connectLoop() { defer b.wg.Done() for { select { case <-b.done: return case job := <-b.jobs: - switch job.kind { - case votorPeerJobConnect: - b.connectPeer(job.peer.Identity) - case votorPeerJobSend: - if err := b.send(job.peer, job.payload); err != nil { - b.recordSendError(job.peer, err) - } else { - b.sends.Add(1) - } - } + b.connectPeer(job.peer.Identity) } } } -func (b *VotorBroadcaster) send(peer VotorPeer, payload []byte) error { - conn, ok := b.establishedConnection(peer) - if !ok { - b.queueConnect(peer.Identity) - return fmt.Errorf("send Votor datagram to %s (%s): no established connection", peer.Identity, peer.Addr) - } - err := conn.SendDatagram(payload) - if err == nil { - return nil - } - var tooLarge *quic.DatagramTooLargeError - if !errors.As(err, &tooLarge) { - b.dropConnection(peer.Identity, conn) - b.queueConnect(peer.Identity) - } - return fmt.Errorf("send Votor datagram to %s (%s): %w", peer.Identity, peer.Addr, err) -} - func (b *VotorBroadcaster) peerReconcileLoop() { defer b.wg.Done() ticker := time.NewTicker(defaultVotorPeerRefreshInterval) defer ticker.Stop() + watchdog := time.NewTicker(votorSendWatchInterval) + defer watchdog.Stop() for { select { case <-b.done: return case <-ticker.C: b.reconcilePeers() + case now := <-watchdog.C: + b.expirePeerSends(now) } } } @@ -305,9 +279,9 @@ func (b *VotorBroadcaster) reconcilePeers() { } } -func (b *VotorBroadcaster) connectedPeers() ([]VotorPeer, int) { +func (b *VotorBroadcaster) connectedSenders() ([]*votorPeerSender, int) { b.connMu.Lock() - peers := make([]VotorPeer, 0, len(b.desired)) + peers := make([]*votorPeerSender, 0, len(b.desired)) skipped := 0 for identity, peer := range b.desired { existing, connected := b.conns[identity] @@ -315,8 +289,7 @@ func (b *VotorBroadcaster) connectedPeers() ([]VotorPeer, int) { skipped++ continue } - peer.Addr = cloneUDPAddr(peer.Addr) - peers = append(peers, peer) + peers = append(peers, existing.sender) } b.connMu.Unlock() return peers, skipped @@ -349,7 +322,7 @@ func (b *VotorBroadcaster) queueConnectLocked(identity solana.PublicKey) { if _, queued := b.connectQueued[identity]; queued || b.dialing[identity] != nil { return } - job := votorPeerJob{kind: votorPeerJobConnect, peer: peer} + job := votorPeerJob{peer: peer} select { case b.jobs <- job: b.connectQueued[identity] = struct{}{} @@ -476,7 +449,12 @@ func (b *VotorBroadcaster) connection(peer VotorPeer) (*quic.Conn, error) { return existing.conn, nil } stale := b.conns[peer.Identity].conn - b.conns[peer.Identity] = votorConnection{addr: addr, conn: conn} + sender := &votorPeerSender{b: b, peer: peer, conn: conn, queue: make(chan votorDatagram, defaultVotorPeerSendQueue), done: make(chan struct{})} + b.conns[peer.Identity] = votorConnection{addr: addr, conn: conn, sender: sender} + // Close takes connMu before waiting, so no sender can be added after it + // observes the closed flag and drains the connection set. + b.wg.Add(1) + go sender.run() b.connMu.Unlock() if stale != nil { _ = stale.CloseWithError(0, "Votor peer address changed") @@ -499,12 +477,14 @@ func (b *VotorBroadcaster) Stats() VotorBroadcasterStats { } b.connMu.Lock() connections := 0 + peerQueues := make([]VotorPeerQueueStats, 0, len(b.conns)) for identity, existing := range b.conns { if existing.conn.Context().Err() != nil { delete(b.conns, identity) continue } connections++ + peerQueues = append(peerQueues, existing.sender.stats()) } desiredPeers := len(b.desired) pendingConnections := len(b.connectQueued) @@ -525,6 +505,11 @@ func (b *VotorBroadcaster) Stats() VotorBroadcasterStats { ConnectionAttempts: b.connectionAttempts.Load(), ConnectionErrors: b.connectionErrors.Load(), ConnectionJobsDropped: b.connectionJobsDropped.Load(), + PeerQueueDrops: b.peerQueueDrops.Load(), + PeerQueueDiscarded: b.peerQueueDiscarded.Load(), + PeerSendTimeouts: b.peerSendTimeouts.Load(), + PeerQueueMaxDelay: time.Duration(b.peerQueueMaxDelay.Load()), + PeerQueues: peerQueues, LastPeerSendError: lastSendError, LastPeerSendErrorAt: lastSendErrorAt, LastConnectionError: lastConnectionError, @@ -542,11 +527,15 @@ func (b *VotorBroadcaster) Close() error { close(b.done) b.connMu.Lock() clear(b.desired) + stale := make([]*quic.Conn, 0, len(b.conns)) for identity, existing := range b.conns { - _ = existing.conn.CloseWithError(0, "Votor broadcaster closed") + stale = append(stale, existing.conn) delete(b.conns, identity) } b.connMu.Unlock() + for _, conn := range stale { + _ = conn.CloseWithError(0, "Votor broadcaster closed") + } b.wg.Wait() }) return nil diff --git a/pkg/alpenglow/certpool.go b/pkg/alpenglow/certpool.go index 60b1882e3..33483cc10 100644 --- a/pkg/alpenglow/certpool.go +++ b/pkg/alpenglow/certpool.go @@ -5,8 +5,11 @@ import ( "crypto/sha256" "fmt" "math/big" + "runtime" "sync" + "sync/atomic" + "github.com/Overclock-Validator/gnark-crypto/ecc" bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" blsfr "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381/fr" "github.com/Overclock-Validator/mithril/pkg/mlog" @@ -31,7 +34,7 @@ const maxUnverifiedCandidatesPerRank = 2 // facts. Ingest (AddVote) only shape-checks, bounds buffering, and parks the // vote — it NEVER mutates dedupe, equivocation, disjointness, or tally state // that assembly or fork choice depends on. Those mutate only AFTER the BLS -// signature verifies (foldTallyLocked). This prevents a bogus vote for a +// signature verifies (verifyAndFoldTallyWithLockReleased). This prevents a bogus vote for a // victim rank from suppressing that validator's real vote (dedupe poisoning) // or forging equivocation evidence against an honest validator. // @@ -109,14 +112,12 @@ type tally struct { pending map[uint16]map[[sha256.Size]byte]VoteMessage // unverified candidates, by rank and signature verified map[uint16]struct{} // ranks folded into the aggregate aggSig bls12381.G2Affine // sum of verified signatures - aggPub bls12381.G1Affine // sum of verified pubkeys stake uint64 // verified stake } func newTally() *tally { t := &tally{pending: make(map[uint16]map[[sha256.Size]byte]VoteMessage), verified: make(map[uint16]struct{})} t.aggSig.SetInfinity() - t.aggPub.SetInfinity() return t } @@ -136,7 +137,14 @@ type tallyKey struct { } type poolSlot struct { - tallies map[tallyKey]*tally + // One caller owns folding for a slot while other callers may buffer votes. + // All fields, like the maps below, are protected by CertPool.mu. + processing bool + // Queued and active reward flushes take precedence over new arrivals/owners. + // A count preserves priority when multiple footer builders overlap. + flushWaiting int + dirty bool + tallies map[tallyKey]*tally // verifiedHash tracks the block hashes a rank has cast VERIFIED votes for, // per (rank, type), for equivocation/vote-budget enforcement. Populated only // after signature verification — never from raw ingest — so a bogus vote can @@ -166,16 +174,20 @@ type CertPool struct { // end and must not make downstream consensus decisions itself. verifiedVoteSink func(VerifiedVote) - mu sync.Mutex - epochForSlot func(slot uint64) uint64 - slots map[uint64]*poolSlot - emitted map[CertificateKey]struct{} - floor uint64 - liveSlot uint64 // trusted replay/observed watermark (NOT advanced by raw votes) - highestSlot uint64 // observability only: highest vote slot seen - totalPending int - equivocation []EquivocationEvidence - snap CertPoolSnapshot + mu sync.Mutex + workCond *sync.Cond + verificationMu sync.Mutex // Keep expensive batch concurrency bounded at one. + batchesVerified atomic.Uint64 + epochGeneration uint64 + epochForSlot func(slot uint64) uint64 + slots map[uint64]*poolSlot + emitted map[CertificateKey]struct{} + floor uint64 + liveSlot atomic.Uint64 // trusted replay/observed watermark (NOT advanced by raw votes) + highestSlot uint64 // observability only: highest vote slot seen + totalPending int + equivocation []EquivocationEvidence + snap CertPoolSnapshot publicationMu sync.Mutex publicationCond *sync.Cond @@ -214,6 +226,7 @@ func NewCertPool(cfg CertPoolConfig, verifier *CertificateVerifier, emit func(Ce publicationCompleted: make(map[uint64]struct{}), } p.publicationCond = sync.NewCond(&p.publicationMu) + p.workCond = sync.NewCond(&p.mu) return p } @@ -229,6 +242,7 @@ func (p *CertPool) SetVerifiedVoteSink(sink func(VerifiedVote)) { func (p *CertPool) SetEpochLookup(fn func(slot uint64) uint64) { p.mu.Lock() p.epochForSlot = fn + p.epochGeneration++ p.mu.Unlock() } @@ -236,21 +250,24 @@ func (p *CertPool) SetEpochLookup(fn func(slot uint64) uint64) { // (from replay progress / observed finality — never from raw votes, so an // attacker cannot slide the window forward). Monotonic. func (p *CertPool) NoteLiveSlot(slot uint64) { - p.mu.Lock() - if slot > p.liveSlot { - p.liveSlot = slot + // Replay must not wait for BLS verification under p.mu merely to announce + // progress. Concurrent trusted updates may arrive out of order. + for current := p.liveSlot.Load(); slot > current; current = p.liveSlot.Load() { + if p.liveSlot.CompareAndSwap(current, slot) { + return + } } - p.mu.Unlock() } // windowAnchorLocked is the trusted upper anchor of the live vote window: the // higher of the finalized floor and the replay-observed live slot. It is NOT // derived from raw votes, so ingest cannot advance it. func (p *CertPool) windowAnchorLocked() uint64 { - if p.floor > p.liveSlot { + liveSlot := p.liveSlot.Load() + if p.floor > liveSlot { return p.floor } - return p.liveSlot + return liveSlot } // setForSlotLocked resolves the validator set covering slot. Returns nil (votes @@ -271,45 +288,60 @@ func (p *CertPool) setForSlotLocked(slot uint64) *ValidatorSet { // AddVote ingests one raw (unverified) votor vote. It ONLY shape-checks, bounds // buffering, and parks the vote in its (type, hash) tally. It does NOT touch // dedupe/equivocation/disjointness state — those mutate only after the vote's -// signature verifies (foldTallyLocked). A malformed vote, a vote outside the +// signature verifies (verifyAndFoldTallyWithLockReleased). A malformed vote, a vote outside the // trusted slot window, or a vote past a memory bound is dropped. func (p *CertPool) AddVote(msg VoteMessage) { if msg.Vote.ValidateBasic() != nil || len(msg.Signature) != BLSSignatureSize { return } slot := msg.Vote.Slot + sigKey := sha256.Sum256(msg.Signature) - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() - if slot <= p.floor { - p.snap.VotesRejected++ - p.mu.Unlock() - return - } - // Window anchored to the TRUSTED watermark (floor / replay-observed), not to - // the highest vote slot seen — otherwise an attacker could slide it forward - // vote by vote and retain arbitrarily many future slots. - if anchor := p.windowAnchorLocked(); anchor > 0 && slot > anchor+p.cfg.MaxSlotsAhead { - p.snap.VotesRejected++ - p.mu.Unlock() - return - } - - ps := p.slots[slot] - if ps == nil { - // Hard global bound on retained slots (independent of the window). - if len(p.slots) >= p.cfg.MaxLiveSlots && !p.evictFartherFutureSlotLocked(slot) { + var ps *poolSlot + for { + if slot <= p.floor { + p.snap.VotesRejected++ + p.mu.Unlock() + return + } + // Window anchored to the TRUSTED watermark (floor / replay-observed), not to + // the highest vote slot seen — otherwise an attacker could slide it forward + // vote by vote and retain arbitrarily many future slots. + if anchor := p.windowAnchorLocked(); anchor > 0 && slot > anchor+p.cfg.MaxSlotsAhead { p.snap.VotesRejected++ p.mu.Unlock() return } - ps = &poolSlot{ - tallies: make(map[tallyKey]*tally), - verifiedHash: make(map[voteDedupKey][]solana.Hash), - pendingByRank: make(map[uint16]int), + + ps = p.slots[slot] + if ps == nil { + // Hard global bound on retained slots (independent of the window). + if len(p.slots) >= p.cfg.MaxLiveSlots && !p.evictFartherFutureSlotLocked(slot) { + p.snap.VotesRejected++ + p.mu.Unlock() + return + } + ps = &poolSlot{ + tallies: make(map[tallyKey]*tally), + verifiedHash: make(map[voteDedupKey][]solana.Hash), + pendingByRank: make(map[uint16]int), + } + p.slots[slot] = ps } - p.slots[slot] = ps + // Ordinary arrivals can join the bounded pending maps during BLS work. + // Quota pressure and competing signatures still wait for authentication + // before admission, preserving the first-packet-poisoning protections. + // A queued reward flush also stops new arrivals from extending its drain. + if ps.flushWaiting > 0 || (ps.processing && p.admissionNeedsFoldLocked(ps, msg, sigKey)) { + p.workCond.Wait() + continue + } + break + } + owner := !ps.processing + if owner { + ps.processing = true } tk := tallyKey{Type: msg.Vote.Type, Hash: msg.Vote.BlockHash} @@ -324,7 +356,6 @@ func (p *CertPool) AddVote(msg VoteMessage) { // one. Different block hashes remain separate tallies. forceFold := false if _, done := tl.verified[msg.Rank]; !done { - sigKey := sha256.Sum256(msg.Signature) candidates := tl.pending[msg.Rank] _, duplicate := candidates[sigKey] if !duplicate && ps.pendingByRank[msg.Rank] >= p.cfg.MaxPendingVotesPerRankSlot { @@ -332,7 +363,11 @@ func (p *CertPool) AddVote(msg VoteMessage) { // the rank quota, authenticate that rank's parked candidates and free // the invalid ones before deciding whether the real vote has room. if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldPendingRankLocked(slot, ps, msg.Rank, set)...) + p.foldPendingRankLocked(slot, ps, msg.Rank, set) + } + if p.slots[slot] != ps { + p.finishSlotAndUnlock(slot, ps, nil) + return } candidates = tl.pending[msg.Rank] } @@ -342,7 +377,11 @@ func (p *CertPool) AddVote(msg VoteMessage) { atCap := ps.pendingCount >= p.cfg.MaxPendingVotesPerSlot || p.totalPending >= p.cfg.MaxPendingVotesTotal if !rejectIncoming && atCap { if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldAllPendingLocked(slot, ps, set)...) + p.foldAllPendingLocked(slot, ps, set) + } + if p.slots[slot] != ps { + p.finishSlotAndUnlock(slot, ps, nil) + return } if p.totalPending >= p.cfg.MaxPendingVotesTotal { p.evictFartherFutureSlotLocked(slot) @@ -382,90 +421,160 @@ func (p *CertPool) AddVote(msg VoteMessage) { if slot > p.highestSlot { p.highestSlot = slot } + if !owner { + ps.dirty = true + p.mu.Unlock() + return + } if forceFold { if set := p.setForSlotLocked(slot); set != nil { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) + } + } + emits := p.drainSlotLocked(slot, ps, false) + p.finishSlotAndUnlock(slot, ps, emits) +} + +func (p *CertPool) admissionNeedsFoldLocked(ps *poolSlot, msg VoteMessage, sigKey [sha256.Size]byte) bool { + tl := ps.tallies[tallyKey{Type: msg.Vote.Type, Hash: msg.Vote.BlockHash}] + if tl != nil { + if _, done := tl.verified[msg.Rank]; done { + return false + } + candidates := tl.pending[msg.Rank] + if _, duplicate := candidates[sigKey]; duplicate { + return false + } + if len(candidates) > 0 { + return true } } - // Keep verified stake fresh for the Votor fallback triggers. Only plain - // notarize/skip arrivals can change the trigger predicates. - if msg.Vote.Type == VoteTypeNotarize || msg.Vote.Type == VoteTypeSkip { - verified = append(verified, p.maybeFoldTriggersLocked(slot, ps)...) + return ps.pendingByRank[msg.Rank] >= p.cfg.MaxPendingVotesPerRankSlot || + ps.pendingCount >= p.cfg.MaxPendingVotesPerSlot || p.totalPending >= p.cfg.MaxPendingVotesTotal +} + +// waitForSlotLocked releases mu while the current owner finishes. Pruning can +// remove or replace the slot, so callers always use the returned current state. +func (p *CertPool) waitForSlotLocked(slot uint64) *poolSlot { + for { + ps := p.slots[slot] + if ps == nil || (!ps.processing && ps.flushWaiting == 0) { + return ps + } + p.workCond.Wait() } - var assembledVerified []VerifiedVote - emits, assembledVerified = p.maybeAssembleLocked(slot, ps) - verified = append(verified, assembledVerified...) - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) - p.mu.Unlock() +} - p.publishVerifiedVotes(sink, verified, publication) +// drainSlotLocked catches arrivals buffered while the owner was outside mu. +// Sub-threshold votes remain lazy except when preparing a reward footer. +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) drainSlotLocked(slot uint64, ps *poolSlot, flush bool) []Certificate { + var emits []Certificate + for p.slots[slot] == ps { + ps.dirty = false + if flush { + if set := p.setForSlotLocked(slot); set != nil { + for key, tl := range ps.tallies { + if key.Type == VoteTypeSkip || key.Type == VoteTypeNotarize { + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) + } + } + } + } + p.maybeFoldTriggersLocked(slot, ps) + certs := p.maybeAssembleLocked(slot, ps) + emits = append(emits, certs...) + if !ps.dirty { + break + } + } + return emits +} + +// finishSlotAndUnlock requires p.mu held and returns with p.mu released. +// It takes the publication barrier before making the slot available to a +// flushing caller, then unlocks and emits certificates. Callers must not defer +// an unlock across this call. +func (p *CertPool) finishSlotAndUnlock(slot uint64, ps *poolSlot, emits []Certificate) uint64 { + if p.slots[slot] != ps { + emits = nil + } + target := p.publicationTargetLocked() + ps.processing = false + p.workCond.Broadcast() + p.mu.Unlock() for _, cert := range emits { p.emitCert(cert) } + return target } // OnValidatorSetInstalled retries assembly for buffered slots that resolve to // the newly-installed epoch. Requires a real slot→epoch lookup; without one no // slot can be safely attributed to the epoch, so nothing is retried. func (p *CertPool) OnValidatorSetInstalled(epoch uint64) { - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() + var slots []uint64 if p.epochForSlot != nil { - for slot, ps := range p.slots { - if p.epochForSlot(slot) != epoch { - continue + for slot := range p.slots { + if p.epochForSlot(slot) == epoch { + slots = append(slots, slot) } - verified = append(verified, p.maybeFoldTriggersLocked(slot, ps)...) - certs, newlyVerified := p.maybeAssembleLocked(slot, ps) - emits = append(emits, certs...) - verified = append(verified, newlyVerified...) } } - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) p.mu.Unlock() - p.publishVerifiedVotes(sink, verified, publication) - for _, cert := range emits { - p.emitCert(cert) + for _, slot := range slots { + p.mu.Lock() + ps := p.waitForSlotLocked(slot) + if ps == nil || p.epochForSlot == nil || p.epochForSlot(slot) != epoch { + p.mu.Unlock() + continue + } + ps.processing = true + emits := p.drainSlotLocked(slot, ps, false) + p.finishSlotAndUnlock(slot, ps, emits) } } // FlushRewardVotes batch-verifies every pending plain skip/notarize vote for a // reward slot. Normal consensus verification stays lazy, but block production // calls this just before building the slot+8 footer so valid below-threshold -// votes are not omitted from reward certificates. +// votes are not omitted from reward certificates. Queued flushes take precedence +// over new slot owners and arrivals; an existing owner finishes first. Pruning +// can invalidate that generation, in which case the current slot is rechecked. func (p *CertPool) FlushRewardVotes(slot uint64) { - var emits []Certificate - var verified []VerifiedVote p.mu.Lock() - ps := p.slots[slot] - set := p.setForSlotLocked(slot) - if ps != nil && set != nil { - for key, tl := range ps.tallies { - if key.Type == VoteTypeSkip || key.Type == VoteTypeNotarize { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) - } + var ps *poolSlot + for { + ps = p.slots[slot] + if ps == nil { + break + } + ps.flushWaiting++ + for p.slots[slot] == ps && ps.processing { + p.workCond.Wait() } - var assembled []VerifiedVote - emits, assembled = p.maybeAssembleLocked(slot, ps) - verified = append(verified, assembled...) + if p.slots[slot] == ps { + break + } + // Pruning or a binding change replaced this generation while waiting. + ps.flushWaiting-- + p.workCond.Broadcast() } - sink := p.verifiedVoteSink - publication := p.reservePublicationLocked(verified) - targetPublication := p.publicationTargetLocked() - p.mu.Unlock() - - p.publishVerifiedVotes(sink, verified, publication) - for _, cert := range emits { - p.emitCert(cert) + if ps == nil { + target := p.publicationTargetLocked() + p.mu.Unlock() + p.waitForPublication(target) + return } - // A vote can finish BLS verification in AddVote just before this flush takes - // the pool lock, then still be in flight to the reward builder. Wait through - // that publication sequence so a footer cannot omit an already-verified vote. - p.waitForPublication(targetPublication) + ps.processing = true + emits := p.drainSlotLocked(slot, ps, true) + ps.flushWaiting-- + target := p.finishSlotAndUnlock(slot, ps, emits) + // Includes publication by an owner that was verifying when flush arrived. + p.waitForPublication(target) } // reservePublicationLocked assigns ordering while p.mu is held. Flush can then @@ -541,6 +650,7 @@ func (p *CertPool) ObserveFloor(finalizedSlot uint64) { delete(p.emitted, key) } } + p.workCond.Broadcast() } p.mu.Unlock() } @@ -557,10 +667,11 @@ func (p *CertPool) Snapshot() CertPoolSnapshot { p.mu.Lock() defer p.mu.Unlock() snap := p.snap + snap.BatchesVerified = p.batchesVerified.Load() snap.Slots = len(p.slots) snap.Floor = p.floor snap.HighestSlot = p.highestSlot - snap.LiveSlot = p.liveSlot + snap.LiveSlot = p.liveSlot.Load() snap.PendingTotal = p.totalPending return snap } @@ -648,10 +759,15 @@ func meets(f Fraction, stake, total uint64) bool { // implementation (Agave) would. This is the ONLY fold policy: one fork-choice // behavior for observer and voting nodes alike; sub-trigger tallies still // cost nothing. -func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) []VerifiedVote { +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) { + if p.slots[slot] != ps { + return + } set := p.setForSlotLocked(slot) if set == nil { - return nil // votes stay buffered; retried on OnValidatorSetInstalled + return // votes stay buffered; retried on OnValidatorSetInstalled } total := set.TotalStake v := buildTriggerViewLocked(ps, set) @@ -686,22 +802,21 @@ func (p *CertPool) maybeFoldTriggersLocked(slot uint64, ps *poolSlot) []Verified } if !foldSkip && !foldAllNotar && len(foldNotar) == 0 { - return nil + return } - var verified []VerifiedVote for tk, tl := range ps.tallies { switch tk.Type { case VoteTypeNotarize: if foldAllNotar || foldNotar[tk.Hash] { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } case VoteTypeSkip: if foldSkip { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } } } - return verified + return } // VotorStakes is the verified-stake observation Votor's fallback-trigger @@ -723,12 +838,12 @@ type VotorStakes struct { func (p *CertPool) VerifiedVotorStakes(slot uint64) (VotorStakes, bool) { p.mu.Lock() defer p.mu.Unlock() + ps := p.waitForSlotLocked(slot) set := p.setForSlotLocked(slot) if set == nil { return VotorStakes{}, false } out := VotorStakes{Notarize: make(map[solana.Hash]uint64), TotalStake: set.TotalStake} - ps := p.slots[slot] if ps == nil { return out, true } @@ -804,14 +919,18 @@ func targetsForSlot(ps *poolSlot) []certTarget { // maybeAssembleLocked checks every assemblable target for the slot: folds // pending votes (batch verification) once candidate stake crosses the // threshold, and returns any newly assembled certificates for emission. -func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate, []VerifiedVote) { +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) []Certificate { + if p.slots[slot] != ps { + return nil + } set := p.setForSlotLocked(slot) if set == nil { - return nil, nil // validator set / epoch not resolvable yet; votes stay buffered + return nil // validator set / epoch not resolvable yet; votes stay buffered } var emits []Certificate - var verified []VerifiedVote for _, target := range targetsForSlot(ps) { key := CertificateKey{Type: target.certType, Slot: slot} if target.certType.HasBlock() { @@ -846,8 +965,11 @@ func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate } // Candidate stake crossed: fold pending votes (one pairing per tally). - verified = append(verified, p.foldTallyLocked(slot, ps, base, set)...) - verified = append(verified, p.foldTallyLocked(slot, ps, fb, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, base, set) + p.verifyAndFoldTallyWithLockReleased(slot, ps, fb, set) + if p.slots[slot] != ps { + return nil + } verifiedStake := uint64(0) if base != nil { @@ -869,26 +991,39 @@ func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) ([]Certificate p.snap.CertsEmitted++ emits = append(emits, cert) } - return emits, verified + return emits } -// foldTallyLocked batch-verifies a tally's pending votes. All sign the same +// verifyAndFoldTallyWithLockReleased requires p.mu held on entry and returns +// with p.mu held on every path. The caller must own ps.processing. Verification +// temporarily releases p.mu; retained slot and validator bindings are rechecked +// after reacquiring it before any verified results are installed. +// +// It batch-verifies a tally's pending votes. All sign the same // payload, so randomized weighted pubkey/signature sums need one pairing; // failures bisect to isolate the bad votes (dropped + counted). Only AFTER a // vote verifies does it update the durable per-slot state — the vote-budget / // equivocation ledger (verifiedHash) and base↔fallback disjointness — so raw // votes can never poison those. -func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *ValidatorSet) []VerifiedVote { - if tl == nil || len(tl.pending) == 0 { - return nil +func (p *CertPool) verifyAndFoldTallyWithLockReleased(slot uint64, ps *poolSlot, tl *tally, set *ValidatorSet) { + if p.slots[slot] != ps || tl == nil || len(tl.pending) == 0 { + return } batch := make([]VoteMessage, 0, pendingCandidateCount(tl)) + // Reuse admission keys without hashing each signature again on removal. + var keyStorage [64][sha256.Size]byte + keys := keyStorage[:0] + if cap(batch) > len(keyStorage) { + keys = make([][sha256.Size]byte, 0, cap(batch)) + } for rank, candidates := range tl.pending { count := len(candidates) if int(rank) < len(set.Validators) { - for _, msg := range candidates { + for key, msg := range candidates { batch = append(batch, msg) + keys = append(keys, key) } + continue } else { p.snap.VotesRejected += uint64(count) } @@ -900,10 +1035,50 @@ func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *Va } p.totalPending -= count } + // Keep candidates in the pending maps while verifying. Concurrent arrivals + // can deduplicate against them, and every in-flight byte still consumes the + // normal per-rank, per-slot and global admission budget. + generation := p.epochGeneration + shredVersion := p.verifier.ShredVersion() + p.mu.Unlock() + p.verificationMu.Lock() good := p.verifyBatch(batch, set) + p.verificationMu.Unlock() + p.mu.Lock() + if p.slots[slot] != ps { + return // pruning/eviction already released the pending accounting + } + current := p.setForSlotLocked(slot) + if generation != p.epochGeneration || shredVersion != p.verifier.ShredVersion() || !sameInstalledValidatorSet(current, set) { + // Do not mix an old aggregate with a new epoch/key/stake binding. Retire + // this slot's old state; a later arrival starts afresh with the current set. + p.totalPending -= ps.pendingCount + delete(p.slots, slot) + for key := range p.emitted { + if key.Slot == slot { + delete(p.emitted, key) + } + } + p.workCond.Broadcast() + return + } + for i, msg := range batch { + candidates := tl.pending[msg.Rank] + delete(candidates, keys[i]) + if len(candidates) == 0 { + delete(tl.pending, msg.Rank) + } + ps.pendingCount-- + ps.pendingByRank[msg.Rank]-- + if ps.pendingByRank[msg.Rank] == 0 { + delete(ps.pendingByRank, msg.Rank) + } + p.totalPending-- + } verified := make([]VerifiedVote, 0, len(good)) - for _, msg := range good { + for i := range good { + msg := good[i].message verified = append(verified, VerifiedVote{ Message: msg, Result: VoteVerifyResult{ @@ -951,40 +1126,47 @@ func (p *CertPool) foldTallyLocked(slot uint64, ps *poolSlot, tl *tally, set *Va } } - pub, err := validatorBLSPubkey(*set, int(msg.Rank)) - if err != nil { - continue - } - var sig bls12381.G2Affine - if _, err := sig.SetBytes(msg.Signature); err != nil { - continue - } - tl.aggPub.Add(&tl.aggPub, &pub) - tl.aggSig.Add(&tl.aggSig, &sig) + // Reuse the exact signature point authenticated by verifyBatch. + tl.aggSig.Add(&tl.aggSig, &good[i].sig) tl.verified[msg.Rank] = struct{}{} tl.stake += set.Validators[msg.Rank].Stake ps.verifiedHash[dk] = append(seen, msg.Vote.BlockHash) } p.snap.BadSignatures += uint64(len(batch) - len(good)) - return verified + // Publish this batch before verifying any newly buffered work. In particular, + // a growing slot must not hold back an already-authenticated quorum. + sink := p.verifiedVoteSink + publication := p.reservePublicationLocked(verified) + p.mu.Unlock() + p.publishVerifiedVotes(sink, verified, publication) + p.mu.Lock() } -func (p *CertPool) foldPendingRankLocked(slot uint64, ps *poolSlot, rank uint16, set *ValidatorSet) []VerifiedVote { - var verified []VerifiedVote +// Installed sets own immutable parsed-key arrays. Reinstallation, even for the +// same epoch and keys, gets a new array and therefore invalidates in-flight work. +func sameInstalledValidatorSet(a, b *ValidatorSet) bool { + return a != nil && b != nil && a.Epoch == b.Epoch && len(a.parsedPubkeys) > 0 && + len(a.parsedPubkeys) == len(b.parsedPubkeys) && &a.parsedPubkeys[0] == &b.parsedPubkeys[0] +} + +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) foldPendingRankLocked(slot uint64, ps *poolSlot, rank uint16, set *ValidatorSet) { for _, tl := range ps.tallies { if len(tl.pending[rank]) != 0 { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } } - return verified + return } -func (p *CertPool) foldAllPendingLocked(slot uint64, ps *poolSlot, set *ValidatorSet) []VerifiedVote { - var verified []VerifiedVote +// Requires and returns with p.mu held, but may release it while folding; +// ps may have been removed or replaced on return. The caller owns ps.processing. +func (p *CertPool) foldAllPendingLocked(slot uint64, ps *poolSlot, set *ValidatorSet) { for _, tl := range ps.tallies { - verified = append(verified, p.foldTallyLocked(slot, ps, tl, set)...) + p.verifyAndFoldTallyWithLockReleased(slot, ps, tl, set) } - return verified + return } func (p *CertPool) evictFartherFutureSlotLocked(incoming uint64) bool { @@ -1017,6 +1199,7 @@ func (p *CertPool) evictFartherFutureSlotLocked(incoming uint64) bool { p.totalPending = 0 } delete(p.slots, victim) + p.workCond.Broadcast() return true } @@ -1045,13 +1228,13 @@ type parsedBatchVote struct { // check, bisecting on failure. Unweighted aggregation is insufficient here: // invalid shares can cancel while the downstream pool later keeps only a subset. // Independent random coefficients bind success to every individual member. -func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []VoteMessage { +func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []parsedBatchVote { if len(batch) == 0 { return nil } - p.snap.BatchesVerified++ payload, err := EncodeVotePayloadToSign(batch[0].Vote, p.verifier.ShredVersion()) if err != nil { + p.batchesVerified.Add(1) return nil } @@ -1068,31 +1251,80 @@ func (p *CertPool) verifyBatch(batch []VoteMessage, set *ValidatorSet) []VoteMes members = append(members, parsedBatchVote{message: msg, pubkey: pub, sig: sig}) } + return p.verifyParsedBatch(members, payload) +} + +// verifyParsedBatch keeps the owned parsed points through failed-batch +// subdivision and returns only individually bound, verified members. Each +// subdivision still uses fresh random coefficients; parsing is the only work +// reused. Verification does not read or mutate the pool's slot maps. +func (p *CertPool) verifyParsedBatch(members []parsedBatchVote, payload []byte) []parsedBatchVote { + p.batchesVerified.Add(1) if len(members) == 0 { return nil } if len(members) == 1 { if aggregatePairingOK(members[0].pubkey, payload, members[0].sig) { - return []VoteMessage{members[0].message} + return members } return nil } if ok, err := randomizedAggregatePairingOK(members, payload); err == nil && ok { - return batchMessages(members) + return members } else if err != nil { // Entropy failure must reduce performance, never verification strength. return individuallyVerifiedBatch(members, payload) } // Aggregate failed: bisect the structurally-valid subset. - messages := batchMessages(members) - mid := len(messages) / 2 - valid := p.verifyBatch(messages[:mid], set) - valid = append(valid, p.verifyBatch(messages[mid:], set)...) + mid := len(members) / 2 + valid := p.verifyParsedBatch(members[:mid:mid], payload) + valid = append(valid, p.verifyParsedBatch(members[mid:], payload)...) return valid } func randomizedAggregatePairingOK(members []parsedBatchVote, payload []byte) (bool, error) { + // Bucket setup outweighs MultiExp's savings on small batches, including + // the two-candidate collision path and failed-batch subdivisions. Its + // internal task handoffs can also delay verification behind replay when + // only one Go execution thread is available, despite doing less arithmetic. + if len(members) < 16 || runtime.GOMAXPROCS(0) == 1 { + return randomizedAggregatePairingScalarOK(members, payload) + } + return randomizedAggregatePairingMultiExpOK(members, payload) +} + +func randomizedAggregatePairingMultiExpOK(members []parsedBatchVote, payload []byte) (bool, error) { + pubkeys := make([]bls12381.G1Affine, len(members)) + signatures := make([]bls12381.G2Affine, len(members)) + coefficients := make([]blsfr.Element, len(members)) + for i := range members { + coefficient, err := randomNonzeroBatchCoefficient() + if err != nil { + return false, err + } + // Keep independent, nonzero, full-field coefficients and apply the + // same coefficient to each member's public key and signature. + coefficients[i].SetBigInt(coefficient) + pubkeys[i] = members[i].pubkey + signatures[i] = members[i].sig + } + // MultiExp defaults to using all CPUs. Keep its arithmetic concurrency at + // one, and run G1/G2 sequentially, to avoid competing with replay workers. + // Pool-level verification admission remains bounded by verificationMu. + config := ecc.MultiExpConfig{NbTasks: 1} + var aggPub bls12381.G1Affine + if _, err := aggPub.MultiExp(pubkeys, coefficients, config); err != nil { + return false, err + } + var aggSig bls12381.G2Affine + if _, err := aggSig.MultiExp(signatures, coefficients, config); err != nil { + return false, err + } + return aggregatePairingOK(aggPub, payload, aggSig), nil +} + +func randomizedAggregatePairingScalarOK(members []parsedBatchVote, payload []byte) (bool, error) { var aggPub bls12381.G1Affine var aggSig bls12381.G2Affine aggPub.SetInfinity() @@ -1124,24 +1356,16 @@ func randomNonzeroBatchCoefficient() (*big.Int, error) { } } -func individuallyVerifiedBatch(members []parsedBatchVote, payload []byte) []VoteMessage { - valid := make([]VoteMessage, 0, len(members)) +func individuallyVerifiedBatch(members []parsedBatchVote, payload []byte) []parsedBatchVote { + valid := make([]parsedBatchVote, 0, len(members)) for i := range members { if aggregatePairingOK(members[i].pubkey, payload, members[i].sig) { - valid = append(valid, members[i].message) + valid = append(valid, members[i]) } } return valid } -func batchMessages(members []parsedBatchVote) []VoteMessage { - messages := make([]VoteMessage, len(members)) - for i := range members { - messages[i] = members[i].message - } - return messages -} - func aggregatePairingOK(aggPub bls12381.G1Affine, payload []byte, aggSig bls12381.G2Affine) bool { if aggPub.IsInfinity() { return false diff --git a/pkg/alpenglow/certpool_bench_test.go b/pkg/alpenglow/certpool_bench_test.go new file mode 100644 index 000000000..a6864f0af --- /dev/null +++ b/pkg/alpenglow/certpool_bench_test.go @@ -0,0 +1,160 @@ +package alpenglow + +import ( + "crypto/sha256" + "fmt" + "slices" + "sync" + "sync/atomic" + "testing" + "time" + + bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" + "github.com/gagliardetto/solana-go" +) + +// Benchmark the full pending-vote fold, including verification, aggregation and +// accounting. Keys and signatures are prepared outside the timed region. +func BenchmarkCertPoolFoldVerifiedBatch(b *testing.B) { + for _, size := range []int{1, 8, 32, 64} { + b.Run(fmt.Sprintf("votes=%d", size), func(b *testing.B) { + verifier, installed, vote, batch := certPoolBenchmarkFixture(b, size) + b.ReportAllocs() + b.ResetTimer() + for n := 0; n < b.N; n++ { + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return installed.Epoch }) + tl := newTally() + ps := &poolSlot{processing: true, verifiedHash: make(map[voteDedupKey][]solana.Hash), pendingByRank: make(map[uint16]int)} + pool.slots[vote.Slot] = ps + for _, msg := range batch { + tl.pending[msg.Rank] = map[[sha256.Size]byte]VoteMessage{sha256.Sum256(msg.Signature): msg} + ps.pendingByRank[msg.Rank]++ + ps.pendingCount++ + pool.totalPending++ + } + pool.mu.Lock() + pool.verifyAndFoldTallyWithLockReleased(vote.Slot, ps, tl, &installed) + pool.mu.Unlock() + if len(tl.verified) != size || tl.stake != uint64(size) || pool.totalPending != 0 { + b.Fatal("incomplete verified fold") + } + } + }) + } +} + +func certPoolBenchmarkFixture(b testing.TB, size int) (*CertificateVerifier, ValidatorSet, Vote, []VoteMessage) { + stakes := make([]uint64, size) + for i := range stakes { + stakes[i] = 1 + } + set, keys := testBLSValidatorSet(uint64(size), stakes...) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + b.Fatal(err) + } + // Exercise the same cached public-key representation used in production. + installed, ok := verifier.ValidatorSetForEpoch(set.Epoch) + if !ok { + b.Fatal("missing installed validator set") + } + vote := NewSkipVote(500) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + b.Fatal(err) + } + point, err := bls12381.HashToG2(payload, []byte(blsHashToPointDST)) + if err != nil { + b.Fatal(err) + } + batch := make([]VoteMessage, size) + for i := range batch { + var sig bls12381.G2Affine + sig.ScalarMultiplication(&point, keys[i]) + raw := sig.RawBytes() + batch[i] = VoteMessage{Vote: vote, Rank: uint16(i), Signature: raw[:]} + } + return verifier, installed, vote, batch +} + +// Observe short pool reads while eight peer-like producers submit one slot. +// The latency metrics, not ns/op (which includes deliberate sampling delays), +// measure whether cryptographic work blocks unrelated pool readers. +func BenchmarkCertPoolConcurrentVotes(b *testing.B) { + verifier, installed, vote, batch := certPoolBenchmarkFixture(b, 64) + waits := make([]int64, 0, 16*b.N) + b.ResetTimer() + for n := 0; n < b.N; n++ { + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return installed.Epoch }) + pool.NoteLiveSlot(vote.Slot) + var verified atomic.Int64 + pool.SetVerifiedVoteSink(func(VerifiedVote) { verified.Add(1) }) + start := make(chan struct{}) + var producers sync.WaitGroup + for worker := 0; worker < 8; worker++ { + producers.Add(1) + go func(worker int) { + defer producers.Done() + <-start + for i := worker; i < len(batch); i += 8 { + pool.AddVote(batch[i]) + } + }(worker) + } + close(start) + for sample := 0; sample < 16; sample++ { + time.Sleep(100 * time.Microsecond) + t := time.Now() + pool.Snapshot() + waits = append(waits, time.Since(t).Nanoseconds()) + } + producers.Wait() + pool.FlushRewardVotes(vote.Slot) + if verified.Load() != 64 || pool.Snapshot().PendingTotal != 0 { + b.Fatal("lost or duplicated votes") + } + } + b.StopTimer() + slices.Sort(waits) + b.ReportMetric(float64(waits[len(waits)*95/100]), "snapshot-p95-ns") + b.ReportMetric(float64(waits[len(waits)-1]), "snapshot-max-ns") +} + +func BenchmarkCertPoolWeightedPairing(b *testing.B) { + for _, size := range []int{2, 4, 8, 16, 32, 64, 128} { + verifier, set, vote, batch := certPoolBenchmarkFixture(b, size) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + b.Fatal(err) + } + members := make([]parsedBatchVote, size) + for i, msg := range batch { + pub, err := validatorBLSPubkey(set, int(msg.Rank)) + if err != nil { + b.Fatal(err) + } + members[i] = parsedBatchVote{message: msg, pubkey: pub} + if _, err := members[i].sig.SetBytes(msg.Signature); err != nil { + b.Fatal(err) + } + } + for _, impl := range []struct { + name string + verify func([]parsedBatchVote, []byte) (bool, error) + }{ + {"Scalar", randomizedAggregatePairingScalarOK}, + {"MultiExp", randomizedAggregatePairingMultiExpOK}, + } { + b.Run(fmt.Sprintf("votes=%d/%s", size, impl.name), func(b *testing.B) { + b.ReportAllocs() + for n := 0; n < b.N; n++ { + if ok, err := impl.verify(members, payload); err != nil || !ok { + b.Fatalf("verification: %t %v", ok, err) + } + } + }) + } + } +} diff --git a/pkg/alpenglow/certpool_concurrency_test.go b/pkg/alpenglow/certpool_concurrency_test.go new file mode 100644 index 000000000..a97c9eea9 --- /dev/null +++ b/pkg/alpenglow/certpool_concurrency_test.go @@ -0,0 +1,252 @@ +package alpenglow + +import ( + "fmt" + "sync" + "testing" + "time" +) + +func waitCertPoolCall(t *testing.T, done <-chan struct{}) { + t.Helper() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("certificate-pool operation did not finish") + } +} + +func certPoolAsync(fn func()) <-chan struct{} { + done := make(chan struct{}) + go func() { defer close(done); fn() }() + return done +} + +// Park a real owner at the production verification gate. No cryptography is +// stubbed; releasing the gate runs the normal randomized verification path. +func parkCertPoolVerification(t *testing.T, pool *CertPool, msg VoteMessage) (func(), <-chan struct{}) { + t.Helper() + pool.verificationMu.Lock() + var once sync.Once + release := func() { once.Do(pool.verificationMu.Unlock) } + t.Cleanup(release) + done := certPoolAsync(func() { pool.AddVote(msg) }) + ready := certPoolAsync(func() { + for pool.Snapshot().PendingTotal == 0 { + time.Sleep(time.Millisecond) + } + }) + waitCertPoolCall(t, ready) // Also proves Snapshot can acquire mu during work. + return release, done +} + +func TestCertPoolOffLockAdmissionAndRewardFlush(t *testing.T) { + pool, set, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + first := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + second := VoteMessage{Vote: vote, Rank: 1, Signature: signTestVote(t, vote, keys[1])} + release, owner := parkCertPoolVerification(t, pool, first) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(second) })) + if got := pool.Snapshot().PendingTotal; got != 2 { + t.Fatalf("pending = %d; in-flight and newly buffered votes must both count", got) + } + flush := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + select { + case <-flush: + t.Fatal("flush escaped in-flight verification") + case <-time.After(20 * time.Millisecond): + } + if len(seen) != 0 { + t.Fatal("unverified vote reached sink") + } + release() + waitCertPoolCall(t, owner) + waitCertPoolCall(t, flush) + if len(seen) != 2 || pool.Snapshot().PendingTotal != 0 { + t.Fatalf("flush lost or duplicated work: published=%d snapshot=%+v", len(seen), pool.Snapshot()) + } + if len(*emitted) == 0 { + t.Fatal("buffered arrivals did not drive certificate assembly") + } + for _, cert := range *emitted { + if _, _, err := verifyCertificateWithSet(set, cert, true); err != nil { + t.Fatalf("concurrently assembled certificate failed verification: %v", err) + } + } +} + +func TestCertPoolOffLockPruningDiscardsInFlightResults(t *testing.T) { + pool, _, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + release, owner := parkCertPoolVerification(t, pool, VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])}) + waitCertPoolCall(t, certPoolAsync(func() { pool.ObserveFloor(vote.Slot) })) + if got := pool.Snapshot(); got.PendingTotal != 0 || got.Slots != 0 { + t.Fatalf("pruning retained in-flight accounting: %+v", got) + } + release() + waitCertPoolCall(t, owner) + if len(seen) != 0 || len(*emitted) != 0 || pool.Snapshot().PendingTotal != 0 { + t.Fatal("pruned work was published or decremented accounting twice") + } + addVote(t, pool, NewSkipVote(501), 0, keys[0]) + if len(seen) != 1 { + t.Fatal("subsequent live vote was lost") + } +} + +func TestCertPoolOffLockBindingChangeDiscardsResults(t *testing.T) { + for _, kind := range []string{"validator-set", "epoch-lookup"} { + t.Run(kind, func(t *testing.T) { + pool, set, keys, emitted := newTestPool(t) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + vote := NewSkipVote(500) + msg := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + release, owner := parkCertPoolVerification(t, pool, msg) + if kind == "validator-set" { + if err := pool.verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + } else { + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + } + release() + waitCertPoolCall(t, owner) + if got := pool.Snapshot(); len(seen) != 0 || len(*emitted) != 0 || got.PendingTotal != 0 || got.Slots != 0 { + t.Fatalf("stale binding published or retained state: %+v", got) + } + pool.AddVote(msg) + if len(seen) != 1 { + t.Fatal("vote could not be retried under the current binding") + } + }) + } +} + +func TestCertPoolOffLockPendingBoundAndDuplicates(t *testing.T) { + pool, _, keys, _ := newTestPool(t) + pool.cfg.MaxPendingVotesPerSlot = 2 + pool.cfg.MaxPendingVotesTotal = 2 + vote := NewSkipVote(500) + msgs := make([]VoteMessage, 3) + for i := range msgs { + msgs[i] = VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, vote, keys[i])} + } + release, owner := parkCertPoolVerification(t, pool, msgs[0]) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msgs[1]) })) + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msgs[0]) })) + third := certPoolAsync(func() { pool.AddVote(msgs[2]) }) + select { + case <-third: + t.Fatal("capacity-pressure admission did not wait for authentication") + case <-time.After(20 * time.Millisecond): + } + if got := pool.Snapshot().PendingTotal; got != 2 { + t.Fatalf("in-flight votes escaped bounds or duplicate consumed capacity: %d", got) + } + release() + waitCertPoolCall(t, owner) + waitCertPoolCall(t, third) + pool.FlushRewardVotes(vote.Slot) + if got := pool.Snapshot(); got.PendingTotal != 0 || got.VotesAccepted != 3 { + t.Fatalf("capacity wakeup lost or duplicated votes: %+v", got) + } +} + +func TestCertPoolOffLockEvictionDoesNotResurrectSlot(t *testing.T) { + set, keys := testBLSValidatorSet(100, 40, 30, 15, 10, 5) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + pool := NewCertPool(CertPoolConfig{MaxLiveSlots: 1}, verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + pool.NoteLiveSlot(100) + seen := make(chan VerifiedVote, 8) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { seen <- v }) + future := NewSkipVote(110) + release, owner := parkCertPoolVerification(t, pool, VoteMessage{Vote: future, Rank: 0, Signature: signTestVote(t, future, keys[0])}) + near := NewSkipVote(101) + msg := VoteMessage{Vote: near, Rank: 4, Signature: signTestVote(t, near, keys[4])} + waitCertPoolCall(t, certPoolAsync(func() { pool.AddVote(msg) })) + if got := pool.Snapshot(); got.PendingTotal != 1 || got.Slots != 1 { + t.Fatalf("eviction accounting mismatch: %+v", got) + } + release() + waitCertPoolCall(t, owner) + pool.FlushRewardVotes(near.Slot) + if len(seen) != 1 || (<-seen).Message.Vote.Slot != near.Slot || pool.Snapshot().PendingTotal != 0 { + t.Fatal("evicted work displaced or corrupted the nearer slot") + } +} + +func TestCertPoolRewardFlushPriority(t *testing.T) { + for _, prune := range []bool{false, true} { + t.Run(fmt.Sprintf("prune=%t", prune), func(t *testing.T) { + pool, _, keys, _ := newTestPool(t) + vote := NewSkipVote(500) + // A below-threshold vote must be published by the reward flush. + pool.AddVote(VoteMessage{Vote: vote, Rank: 4, Signature: signTestVote(t, vote, keys[4])}) + pool.mu.Lock() + ps := pool.slots[vote.Slot] + ps.processing = true // Hold ownership until both flushes are queued. + pool.mu.Unlock() + flush1 := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + flush2 := certPoolAsync(func() { pool.FlushRewardVotes(vote.Slot) }) + deadline := time.Now().Add(2 * time.Second) + for { + pool.mu.Lock() + waiting := ps.flushWaiting + pool.mu.Unlock() + if waiting == 2 { + break + } + if time.Now().After(deadline) { + t.Fatal("flushes did not register") + } + time.Sleep(time.Millisecond) + } + entered, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + t.Cleanup(func() { once.Do(func() { close(release) }) }) + pool.SetVerifiedVoteSink(func(v VerifiedVote) { + if v.Message.Rank == 4 { + close(entered) + <-release + } + }) + msg := VoteMessage{Vote: vote, Rank: 0, Signature: signTestVote(t, vote, keys[0])} + arrival := certPoolAsync(func() { pool.AddVote(msg) }) + if prune { + pool.ObserveFloor(vote.Slot) + } else { + pool.mu.Lock() + ps.processing = false + pool.workCond.Broadcast() + pool.mu.Unlock() + waitCertPoolCall(t, entered) + if got := pool.Snapshot().VotesAccepted; got != 1 { + t.Fatalf("new arrival overtook reward flush: accepted=%d", got) + } + select { + case <-arrival: + t.Fatal("arrival escaped active flush") + default: + } + } + once.Do(func() { close(release) }) + waitCertPoolCall(t, flush1) + waitCertPoolCall(t, flush2) + waitCertPoolCall(t, arrival) + pool.mu.Lock() + defer pool.mu.Unlock() + if ps.flushWaiting != 0 { + t.Fatalf("leaked flush waiters: %d", ps.flushWaiting) + } + }) + } +} diff --git a/pkg/alpenglow/certpool_points_test.go b/pkg/alpenglow/certpool_points_test.go new file mode 100644 index 000000000..4f76ad01e --- /dev/null +++ b/pkg/alpenglow/certpool_points_test.go @@ -0,0 +1,102 @@ +package alpenglow + +import ( + "bytes" + "testing" + + bls12381 "github.com/Overclock-Validator/gnark-crypto/ecc/bls12-381" +) + +func TestCertPoolVerifiedPointsMatchIndividualVerification(t *testing.T) { + for _, mixed := range []bool{false, true} { + name := "valid" + if mixed { + name = "mixed-invalid" + } + t.Run(name, func(t *testing.T) { + pool, set, keys, _ := newTestPool(t) + vote := NewSkipVote(500) + var batch []VoteMessage + for i, key := range keys { + signedVote := vote + if mixed && i%2 == 1 { + signedVote = NewSkipVote(501) // Valid point, wrong payload, in both halves. + } + batch = append(batch, VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, signedVote, key)}) + } + if mixed { + var infinity bls12381.G2Affine + infinity.SetInfinity() + raw := infinity.RawBytes() + batch = append(batch, + VoteMessage{Vote: vote, Rank: 0, Signature: []byte{0xff}}, + VoteMessage{Vote: vote, Rank: 1, Signature: raw[:]}, + VoteMessage{Vote: vote, Rank: uint16(len(keys)), Signature: batch[0].Signature}, + ) + } + var expected []VoteMessage + for _, msg := range batch { + if _, err := verifyVoteMessageWithSet(set, msg); err == nil { + expected = append(expected, msg) + } + } + verified := pool.verifyBatch(batch, &set) + if len(verified) != len(expected) { + t.Fatalf("verified %d members, want %d", len(verified), len(expected)) + } + for i, member := range verified { + if member.message.Rank != expected[i].Rank || member.message.Vote != expected[i].Vote || !bytes.Equal(member.message.Signature, expected[i].Signature) { + t.Fatalf("member %d does not match its individually verified message", i) + } + raw := member.sig.RawBytes() + if !bytes.Equal(raw[:], expected[i].Signature) { + t.Fatalf("member %d retained the wrong signature point", i) + } + pub, err := validatorBLSPubkey(set, int(expected[i].Rank)) + if err != nil || !member.pubkey.Equal(&pub) { + t.Fatalf("member %d retained the wrong public key", i) + } + } + if mixed && pool.Snapshot().BatchesVerified <= 1 { + t.Fatal("mixed batch did not exercise recursive verification") + } + }) + } +} + +// Entropy failure uses this individual-verification path instead of accepting +// an unweighted aggregate. Reusing points must preserve its filtering too. +func TestIndividuallyVerifiedBatchRetainsOnlyValidPoints(t *testing.T) { + _, set, keys, _ := newTestPool(t) + vote := NewSkipVote(502) + payload, err := EncodeVotePayloadToSign(vote, 0) + if err != nil { + t.Fatal(err) + } + var members []parsedBatchVote + for i, key := range keys { + signedVote := vote + if i%2 == 1 { + signedVote = NewSkipVote(503) + } + msg := VoteMessage{Vote: vote, Rank: uint16(i), Signature: signTestVote(t, signedVote, key)} + pub, err := validatorBLSPubkey(set, i) + if err != nil { + t.Fatal(err) + } + var sig bls12381.G2Affine + if _, err := sig.SetBytes(msg.Signature); err != nil { + t.Fatal(err) + } + members = append(members, parsedBatchVote{message: msg, pubkey: pub, sig: sig}) + } + verified := individuallyVerifiedBatch(members, payload) + if len(verified) != 3 { + t.Fatalf("verified %d, want 3", len(verified)) + } + for i, member := range verified { + if member.message.Rank != uint16(i*2) || !member.sig.Equal(&members[i*2].sig) || !member.pubkey.Equal(&members[i*2].pubkey) { + t.Fatalf("wrong verified member at %d", i) + } + } +} diff --git a/pkg/alpenglow/certpool_progress_test.go b/pkg/alpenglow/certpool_progress_test.go new file mode 100644 index 000000000..6732dce4b --- /dev/null +++ b/pkg/alpenglow/certpool_progress_test.go @@ -0,0 +1,112 @@ +package alpenglow + +import ( + "sync" + "testing" + "time" +) + +func TestCertPoolProgressDoesNotWaitForVerificationLock(t *testing.T) { + pool := NewCertPool(DefaultCertPoolConfig(), NewCertificateVerifier(), nil) + pool.mu.Lock() // The same lock held while verifying incoming BLS votes. + done := make(chan struct{}) + go func() { + pool.NoteLiveSlot(123) + close(done) + }() + completed := false + select { + case <-done: + completed = true + case <-time.After(time.Second): + } + pool.mu.Unlock() + <-done + if !completed { + t.Fatal("trusted replay progress waited for the verification lock") + } + if got := pool.Snapshot().LiveSlot; got != 123 { + t.Fatalf("live slot = %d, want 123", got) + } +} + +func TestCertPoolConcurrentProgressRemainsMonotonic(t *testing.T) { + pool := NewCertPool(DefaultCertPoolConfig(), NewCertificateVerifier(), nil) + const workers, updates = 16, 256 + start := make(chan struct{}) + var wg sync.WaitGroup + for worker := range workers { + wg.Go(func() { + <-start + for n := range updates { + slot := uint64(n*workers + worker + 1) + pool.NoteLiveSlot(slot) + pool.NoteLiveSlot(slot / 2) // Stale updates race newer progress. + } + }) + } + finished := make(chan struct{}) + go func() { + wg.Wait() + close(finished) + }() + close(start) + var previous uint64 + for { + live := pool.Snapshot().LiveSlot + pool.mu.Lock() + anchor := pool.windowAnchorLocked() + pool.mu.Unlock() + if live < previous || anchor < live { + t.Errorf("progress regressed: previous=%d snapshot=%d anchor=%d", previous, live, anchor) + } + previous = anchor + select { + case <-finished: + pool.NoteLiveSlot(0) + pool.NoteLiveSlot(1) + if got := pool.Snapshot().LiveSlot; got != workers*updates { + t.Fatalf("live slot = %d, want %d", got, workers*updates) + } + return + default: + } + } +} + +func TestCertPoolProgressPreservesTrustedVoteWindow(t *testing.T) { + set, keys := testBLSValidatorSet(100, 40, 30, 15, 10, 5) + verifier := NewCertificateVerifier() + if err := verifier.SetValidatorSet(set); err != nil { + t.Fatal(err) + } + pool := NewCertPool(CertPoolConfig{MaxSlotsAhead: 10}, verifier, nil) + pool.SetEpochLookup(func(uint64) uint64 { return set.Epoch }) + pool.NoteLiveSlot(100) + checkVote := func(slot uint64, rejected bool, live uint64) { + t.Helper() + before := pool.Snapshot() + addVote(t, pool, NewSkipVote(slot), 4, keys[4]) + after := pool.Snapshot() + wantRejected := before.VotesRejected + if rejected { + wantRejected++ + } + if after.VotesRejected != wantRejected { + t.Fatalf("slot %d: rejected=%d, want %d", slot, after.VotesRejected, wantRejected) + } + if after.LiveSlot != live { + t.Fatalf("raw vote at %d moved trusted progress to %d, want %d", slot, after.LiveSlot, live) + } + } + checkVote(110, false, 100) // Inclusive upper edge. + checkVote(111, true, 100) // Accepted raw votes cannot slide the window. + pool.NoteLiveSlot(90) + checkVote(111, true, 100) + pool.NoteLiveSlot(101) + checkVote(111, false, 101) + pool.ObserveFloor(120) + checkVote(120, true, 101) // Finalized floor still rejects old votes. + checkVote(130, false, 101) // Floor can anchor the window above replay. + checkVote(131, true, 101) +} diff --git a/pkg/alpenglow/certpool_test.go b/pkg/alpenglow/certpool_test.go index 8c7c36789..24970345c 100644 --- a/pkg/alpenglow/certpool_test.go +++ b/pkg/alpenglow/certpool_test.go @@ -1,6 +1,7 @@ package alpenglow import ( + "fmt" "math/big" "testing" "time" @@ -805,3 +806,50 @@ func TestCertPoolEmitsOnce(t *testing.T) { t.Fatalf("no new certs expected, went from %d to %d", n, len(*emitted)) } } + +// Exercise both aggregation paths, including a malformed member at every +// position, a different signed payload, and failed-batch subdivision. Valid +// votes must survive independently of which member causes the batch to fail. +func TestCertPoolBatchAggregationPaths(t *testing.T) { + for _, size := range []int{2, 8, 16, 64} { + t.Run(fmt.Sprintf("votes=%d", size), func(t *testing.T) { + verifier, set, vote, batch := certPoolBenchmarkFixture(t, size) + pool := NewCertPool(DefaultCertPoolConfig(), verifier, nil) + payload, err := EncodeVotePayloadToSign(vote, verifier.ShredVersion()) + if err != nil { + t.Fatal(err) + } + members := pool.verifyBatch(batch, &set) + if len(members) != size { + t.Fatal("valid batch rejected") + } + var tweak bls12381.G2Affine + tweak.ScalarMultiplicationBase(big.NewInt(1234567)) + for i := range members { + bad := append([]parsedBatchVote(nil), members...) + bad[i].sig.Add(&bad[i].sig, &tweak) + if ok, err := randomizedAggregatePairingOK(bad, payload); err != nil || ok { + t.Fatalf("bad member %d: accepted=%t err=%v", i, ok, err) + } + } + wrongPayload := append([]byte(nil), payload...) + wrongPayload[0] ^= 1 + if ok, err := randomizedAggregatePairingOK(members, wrongPayload); err != nil || ok { + t.Fatalf("wrong payload: accepted=%t err=%v", ok, err) + } + // Preserve the unweighted sum while corrupting two shares in a + // large batch; subdivision must keep exactly the honest members. + members[0].sig.Add(&members[0].sig, &tweak) + members[len(members)-1].sig.Sub(&members[len(members)-1].sig, &tweak) + valid := pool.verifyParsedBatch(members, payload) + if len(valid) != size-2 { + t.Fatalf("verified %d, want %d", len(valid), size-2) + } + for _, member := range valid { + if member.message.Rank == 0 || int(member.message.Rank) == size-1 { + t.Fatal("invalid share survived") + } + } + }) + } +} diff --git a/pkg/alpenglow/observer.go b/pkg/alpenglow/observer.go index 9b84e58cb..8f3470bec 100644 --- a/pkg/alpenglow/observer.go +++ b/pkg/alpenglow/observer.go @@ -95,6 +95,17 @@ type Observer struct { replayBlocks map[uint64]BlockID replayOrder []uint64 replayChecks map[CertificateKey]certificateReplayCheck + // Only retained, block-bearing certificates that have not yet been checked + // against replay belong here. Checked history stays in certificates and + // replayChecks for diagnostics/deduplication, but need not be scanned on + // every block (including skipped slots). This is an in-memory observer + // index, not voting authorization or durable crash-recovery state. + pendingReplayCertificates map[CertificateKey]BlockID + // Votes do not change certificate/replay reconciliation. Reuse its exact + // statistics until one of those inputs changes instead of scanning every + // retained certificate for each incoming vote. + pendingStats certificateReplayPendingStats + pendingStatsValid bool votesObserved uint64 certificatesObserved uint64 @@ -149,11 +160,12 @@ func NewObserverWithConfig(cfg ObserverConfig) *Observer { cfg.MaxTrackedReplayBlocks = DefaultMaxTrackedReplayBlocks } return &Observer{ - cfg: cfg, - votes: make(map[VoteMessageKey]VoteMessage), - certificates: make(map[CertificateKey]Certificate), - replayBlocks: make(map[uint64]BlockID), - replayChecks: make(map[CertificateKey]certificateReplayCheck), + cfg: cfg, + votes: make(map[VoteMessageKey]VoteMessage), + certificates: make(map[CertificateKey]Certificate), + replayBlocks: make(map[uint64]BlockID), + replayChecks: make(map[CertificateKey]certificateReplayCheck), + pendingReplayCertificates: make(map[CertificateKey]BlockID), } } @@ -211,6 +223,7 @@ func (o *Observer) ObserveCertificate(cert Certificate) (Observation, error) { key := cert.Key() _, exists := o.certificates[key] if !exists { + o.pendingStatsValid = false tracked := o.trackCertificateLocked(key, cert) o.certificatesObserved++ o.applyCertificateLocked(cert) @@ -225,6 +238,7 @@ func (o *Observer) ObserveCertificate(cert Certificate) (Observation, error) { func (o *Observer) ObserveReplayBlock(obs ReplayBlockObservation) Observation { o.mu.Lock() defer o.mu.Unlock() + o.pendingStatsValid = false if obs.At.IsZero() { obs.At = time.Now() @@ -260,8 +274,8 @@ func (o *Observer) ObserveReplayResult(obs ReplayResultObservation) Observation } func (o *Observer) Snapshot() Snapshot { - o.mu.RLock() - defer o.mu.RUnlock() + o.mu.Lock() + defer o.mu.Unlock() return o.snapshotLocked() } @@ -339,12 +353,16 @@ func (o *Observer) trackCertificateLocked(key CertificateKey, cert Certificate) return false } o.certificates[key] = cert + if block, ok := cert.Block(); ok && block.HasHash() { + o.pendingReplayCertificates[key] = block + } o.certOrder = append(o.certOrder, key) for len(o.certificates) > o.cfg.MaxTrackedCertificates { old := o.certOrder[0] o.certOrder = o.certOrder[1:] delete(o.certificates, old) delete(o.replayChecks, old) + delete(o.pendingReplayCertificates, old) } return true } @@ -369,12 +387,10 @@ func (o *Observer) checkReplayBlockCertificatesLocked(block BlockID) { if !block.HasHash() { return } - for key, cert := range o.certificates { - certBlock, ok := cert.Block() - if !ok || certBlock.Slot != block.Slot { - continue + for key, certBlock := range o.pendingReplayCertificates { + if certBlock.Slot == block.Slot { + o.checkCertificateReplayLocked(key, o.certificates[key]) } - o.checkCertificateReplayLocked(key, cert) } } @@ -410,6 +426,7 @@ func (o *Observer) checkCertificateReplayLocked(key CertificateKey, cert Certifi } } o.replayChecks[key] = check + delete(o.pendingReplayCertificates, key) } type certificateReplayPendingStats struct { @@ -423,30 +440,24 @@ type certificateReplayPendingStats struct { func (o *Observer) certificateReplayPendingStatsLocked() certificateReplayPendingStats { var stats certificateReplayPendingStats - for key, cert := range o.certificates { - if _, checked := o.replayChecks[key]; checked { - continue + for _, certBlock := range o.pendingReplayCertificates { + stats.count++ + if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { + stats.oldestSlot = certBlock.Slot } - certBlock, ok := cert.Block() - if ok && certBlock.HasHash() { - stats.count++ - if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { - stats.oldestSlot = certBlock.Slot - } - if certBlock.Slot > stats.newestSlot { - stats.newestSlot = certBlock.Slot - } - if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { - stats.preWindow++ - } - if o.oldestReplayBlockSlot != 0 && - o.latestReplayBlockSlot != 0 && - certBlock.Slot >= o.oldestReplayBlockSlot && - certBlock.Slot <= o.latestReplayBlockSlot { - stats.mature++ - if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { - stats.matureOldestSlot = certBlock.Slot - } + if certBlock.Slot > stats.newestSlot { + stats.newestSlot = certBlock.Slot + } + if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { + stats.preWindow++ + } + if o.oldestReplayBlockSlot != 0 && + o.latestReplayBlockSlot != 0 && + certBlock.Slot >= o.oldestReplayBlockSlot && + certBlock.Slot <= o.latestReplayBlockSlot { + stats.mature++ + if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { + stats.matureOldestSlot = certBlock.Slot } } } @@ -454,7 +465,11 @@ func (o *Observer) certificateReplayPendingStatsLocked() certificateReplayPendin } func (o *Observer) snapshotLocked() Snapshot { - pending := o.certificateReplayPendingStatsLocked() + if !o.pendingStatsValid { + o.pendingStats = o.certificateReplayPendingStatsLocked() + o.pendingStatsValid = true + } + pending := o.pendingStats return Snapshot{ VotesObserved: o.votesObserved, CertificatesObserved: o.certificatesObserved, diff --git a/pkg/alpenglow/observer_bench_test.go b/pkg/alpenglow/observer_bench_test.go new file mode 100644 index 000000000..a792015cd --- /dev/null +++ b/pkg/alpenglow/observer_bench_test.go @@ -0,0 +1,57 @@ +package alpenglow + +import ( + "fmt" + "testing" +) + +func BenchmarkObserverVoteWithRetainedCertificates(b *testing.B) { + o := NewObserver() + for slot := uint64(1); slot <= DefaultMaxTrackedCertificates; slot++ { + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + b.Fatal(err) + } + } + msg := VoteMessage{Vote: NewSkipVote(5000), Rank: 1} + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if _, err := o.ObserveVote(msg); err != nil { + b.Fatal(err) + } + } +} + +// Model a full diagnostic history, with a small unresolved frontier or an +// entirely unresolved history as a worst case. These are observer-only costs: +// no transactions, BLS verification, or network traffic are included. +func BenchmarkObserverEmptyReplay(b *testing.B) { + for _, pending := range []int{0, 32, DefaultMaxTrackedCertificates} { + for _, skips := range []int{0, 4} { + b.Run(fmt.Sprintf("pending=%d/skips=%d", pending, skips), func(b *testing.B) { + o := NewObserver() + for slot := uint64(1); slot <= DefaultMaxTrackedCertificates; slot++ { + if slot <= uint64(DefaultMaxTrackedCertificates-pending) { + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: slot, Hash: testHash(1)}}) + } + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + b.Fatal(err) + } + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + base := uint64(DefaultMaxTrackedCertificates + 1 + i*(skips+1)) + for n := 0; n < skips; n++ { + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: base + uint64(n)}}) + } + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: base + uint64(skips), Hash: testHash(1)}}) + } + }) + } + } +} diff --git a/pkg/alpenglow/observer_stats_test.go b/pkg/alpenglow/observer_stats_test.go new file mode 100644 index 000000000..bf81999de --- /dev/null +++ b/pkg/alpenglow/observer_stats_test.go @@ -0,0 +1,127 @@ +package alpenglow + +import ( + "fmt" + "math/rand" + "sync" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestObserverPendingStatsMatchFullScan(t *testing.T) { + // Exercise eviction, duplicate certificates, out-of-order and hashless + // replay, and retained matches/mismatches with and without retention. + for _, retention := range []int{0, 1, 8, 64} { + t.Run(fmt.Sprint(retention), func(t *testing.T) { + o := NewObserverWithConfig(ObserverConfig{MaxTrackedVotes: 8, + MaxTrackedCertificates: retention, MaxTrackedReplayBlocks: retention}) + rng := rand.New(rand.NewSource(37)) + for i := 0; i < 1000; i++ { + slot := uint64(rng.Intn(80) + 1) + switch rng.Intn(5) { + case 0, 1: + typ := CertificateNotarize + if i%3 == 0 { + typ = CertificateFinalizeFast + } + _, err := o.ObserveCertificate(Certificate{Type: typ, Slot: slot, + BlockHash: testHash(byte(slot%3 + 1)), IncludedStake: 80, TotalStake: 100}) + require.NoError(t, err) + case 2: + block := BlockID{Slot: slot} + if i%5 != 0 { + block.Hash = testHash(byte(slot%4 + 1)) + } + o.ObserveReplayBlock(ReplayBlockObservation{Block: block}) + case 3: + _, err := o.ObserveVote(VoteMessage{Vote: NewSkipVote(slot), Rank: 1}) + require.NoError(t, err) + case 4: + o.ObserveReplayResult(ReplayResultObservation{Slot: slot}) + } + o.Snapshot() + o.mu.RLock() + cached, scanned := o.pendingStats, observerPendingStatsFullScan(o) + pending := make(map[CertificateKey]BlockID) + for key, cert := range o.certificates { + if _, checked := o.replayChecks[key]; !checked { + if block, ok := cert.Block(); ok && block.HasHash() { + pending[key] = block + } + } + } + require.Equal(t, pending, o.pendingReplayCertificates, "operation %d", i) + o.mu.RUnlock() + require.Equal(t, scanned, cached, "operation %d", i) + } + }) + } +} + +func TestObserverConcurrentSnapshotsAndReconciliation(t *testing.T) { + o := NewObserver() + var wg sync.WaitGroup + for worker := 0; worker < 4; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for slot := uint64(1); slot <= 100; slot++ { + switch worker { + case 0: + o.ObserveReplayBlock(ReplayBlockObservation{Block: BlockID{Slot: slot, Hash: testHash(1)}}) + case 1: + _, err := o.ObserveCertificate(Certificate{Type: CertificateNotarize, Slot: slot, + BlockHash: testHash(1), IncludedStake: 80, TotalStake: 100}) + if err != nil { + t.Error(err) + } + case 2: + _, err := o.ObserveVote(VoteMessage{Vote: NewSkipVote(slot), Rank: 1}) + if err != nil { + t.Error(err) + } + case 3: + o.Snapshot() + } + } + }(worker) + } + wg.Wait() + snapshot := o.Snapshot() + require.Equal(t, uint64(100), snapshot.CertificateReplayMatches) + require.Zero(t, snapshot.CertificateReplayPending) +} + +// Independent reference retains the original scan over every retained certificate. +func observerPendingStatsFullScan(o *Observer) certificateReplayPendingStats { + var stats certificateReplayPendingStats + for key, cert := range o.certificates { + if _, checked := o.replayChecks[key]; checked { + continue + } + certBlock, ok := cert.Block() + if ok && certBlock.HasHash() { + stats.count++ + if stats.oldestSlot == 0 || certBlock.Slot < stats.oldestSlot { + stats.oldestSlot = certBlock.Slot + } + if certBlock.Slot > stats.newestSlot { + stats.newestSlot = certBlock.Slot + } + if o.oldestReplayBlockSlot != 0 && certBlock.Slot < o.oldestReplayBlockSlot { + stats.preWindow++ + } + if o.oldestReplayBlockSlot != 0 && + o.latestReplayBlockSlot != 0 && + certBlock.Slot >= o.oldestReplayBlockSlot && + certBlock.Slot <= o.latestReplayBlockSlot { + stats.mature++ + if stats.matureOldestSlot == 0 || certBlock.Slot < stats.matureOldestSlot { + stats.matureOldestSlot = certBlock.Slot + } + } + } + } + return stats +} diff --git a/pkg/alpenglow/peer_sender.go b/pkg/alpenglow/peer_sender.go new file mode 100644 index 000000000..73c9176f6 --- /dev/null +++ b/pkg/alpenglow/peer_sender.go @@ -0,0 +1,181 @@ +package alpenglow + +import ( + "errors" + "sync" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/quic-go/quic-go" +) + +const ( + defaultVotorPeerSendQueue = 256 + votorSendTimeout = time.Second + votorSendWatchInterval = 100 * time.Millisecond +) + +// VotorPeerQueueStats describes local queueing, not remote delivery. Counters +// here cover the current connection; broadcaster totals survive reconnects. +type VotorPeerQueueStats struct { + Identity solana.PublicKey `json:"identity"` + Address string `json:"address"` + Queued int `json:"queued"` + SendingFor time.Duration `json:"sending_for_ns"` + LastQueueDelay time.Duration `json:"last_queue_delay_ns"` + MaxQueueDelay time.Duration `json:"max_queue_delay_ns"` + QueueDrops uint64 `json:"queue_drops"` +} + +type votorDatagram struct { + payload []byte // immutable, shared across the peer queues + queuedAt time.Time +} + +// Each authenticated connection owns one sender. No mutex is held while +// SendDatagram blocks on quic-go's bounded datagram queue. +type votorPeerSender struct { + b *VotorBroadcaster + peer VotorPeer + conn *quic.Conn + queue chan votorDatagram + done chan struct{} + + mu sync.Mutex + closed bool + sendingSince time.Time + lastQueueDelay time.Duration + maxQueueDelay time.Duration + queueDrops uint64 +} + +func (s *votorPeerSender) enqueue(job votorDatagram) { + s.mu.Lock() + defer s.mu.Unlock() + if s.closed || s.conn.Context().Err() != nil { + s.b.sendsSkipped.Add(1) + return + } + select { + case s.queue <- job: + default: + // A full queue rejects only this peer's copy. As before, fanout is + // best effort; the global Enqueue error contract is unchanged. + s.queueDrops++ + s.b.peerQueueDrops.Add(1) + s.b.dropped.Add(1) + } +} + +func (s *votorPeerSender) run() { + defer s.b.wg.Done() + defer close(s.done) + // Cover both remote-close select and close detected after dequeuing a job. + // The queue is bounded/deduplicated; shutdown, departure and a healthy + // replacement connection suppress obsolete reconnect requests. + defer s.b.queueConnect(s.peer.Identity) + defer func() { + s.mu.Lock() + s.closed = true + // Old-connection work is not replayed onto a new address/connection. + // Account for every queued copy discarded on failure or shutdown. + for { + select { + case <-s.queue: + s.b.peerQueueDiscarded.Add(1) + default: + s.mu.Unlock() + return + } + } + }() + for { + select { + case <-s.b.done: + return + case <-s.conn.Context().Done(): + return + case job := <-s.queue: + s.mu.Lock() + if s.closed || s.conn.Context().Err() != nil || s.b.closed.Load() { + s.mu.Unlock() + s.b.peerQueueDiscarded.Add(1) + return + } + now := time.Now() + s.sendingSince = now + s.lastQueueDelay = now.Sub(job.queuedAt) + s.maxQueueDelay = max(s.maxQueueDelay, s.lastQueueDelay) + for old := s.b.peerQueueMaxDelay.Load(); int64(s.lastQueueDelay) > old; old = s.b.peerQueueMaxDelay.Load() { + if s.b.peerQueueMaxDelay.CompareAndSwap(old, int64(s.lastQueueDelay)) { + break + } + } + if s.lastQueueDelay >= votorSendTimeout { + // Do not feed an already-stale backlog into a briefly writable + // QUIC queue between watchdog ticks. Retire this connection. + s.closed = true + s.mu.Unlock() + s.b.peerQueueDiscarded.Add(1) + s.timeout() + return + } + s.mu.Unlock() + err := s.conn.SendDatagram(job.payload) + s.mu.Lock() + s.sendingSince = time.Time{} + s.mu.Unlock() + if err != nil { + s.b.recordSendError(s.peer, err) + var tooLarge *quic.DatagramTooLargeError + if errors.As(err, &tooLarge) { + continue + } + s.b.dropConnection(s.peer.Identity, s.conn) + s.b.queueConnect(s.peer.Identity) + return + } + s.b.sends.Add(1) + } + } +} + +func (s *votorPeerSender) stats() VotorPeerQueueStats { + s.mu.Lock() + defer s.mu.Unlock() + var sendingFor time.Duration + if !s.sendingSince.IsZero() { + sendingFor = time.Since(s.sendingSince) + } + return VotorPeerQueueStats{ + Identity: s.peer.Identity, Address: s.peer.Addr.String(), + Queued: len(s.queue), SendingFor: sendingFor, + LastQueueDelay: s.lastQueueDelay, MaxQueueDelay: s.maxQueueDelay, QueueDrops: s.queueDrops, + } +} + +func (b *VotorBroadcaster) expirePeerSends(now time.Time) { + senders, _ := b.connectedSenders() + for _, s := range senders { + s.mu.Lock() + expired := !s.closed && !s.sendingSince.IsZero() && s.lastQueueDelay+now.Sub(s.sendingSince) >= votorSendTimeout + if expired { + // Serialize with send completion so a late watchdog cannot close a + // later, unrelated send after the blocked operation has finished. + s.closed = true + } + s.mu.Unlock() + if expired { + s.timeout() + } + } +} + +// Caller must first claim the timeout by setting closed under s.mu. This makes +// the dequeue check and watchdog mutually exclusive and counts one timeout. +func (s *votorPeerSender) timeout() { + s.b.peerSendTimeouts.Add(1) + // Closing wakes SendDatagram without leaking a timeout goroutine. + s.b.dropConnection(s.peer.Identity, s.conn) + s.b.queueConnect(s.peer.Identity) +} diff --git a/pkg/alpenglow/peer_sender_recovery_test.go b/pkg/alpenglow/peer_sender_recovery_test.go new file mode 100644 index 000000000..3f88446ea --- /dev/null +++ b/pkg/alpenglow/peer_sender_recovery_test.go @@ -0,0 +1,136 @@ +package alpenglow + +import ( + "context" + "crypto/ed25519" + "crypto/tls" + "net" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Real authenticated connection with no reconciliation loop or connection +// workers. Any reconnect job must come from the sender, never the periodic tick. +func passiveVotorSender(t *testing.T) (*VotorBroadcaster, *votorPeerSender, *Receiver) { + t.Helper() + serverIdentity := ed25519.NewKeyFromSeed(bytesOf(181, ed25519.SeedSize)) + r, err := NewReceiver(ReceiverConfig{ + BindAddr: "127.0.0.1:0", Identity: serverIdentity, LogInterval: -1, + AdmitPeer: func(solana.PublicKey) bool { return true }, + }, NewObserver()) + require.NoError(t, err) + runVotorReceiver(t, r) + cert, err := newVotorQUICCertificate(ed25519.NewKeyFromSeed(bytesOf(182, ed25519.SeedSize))) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + peer := VotorPeer{Identity: testVotorPubkey(serverIdentity), Addr: r.Addr().(*net.UDPAddr)} + b := &VotorBroadcaster{ + ctx: ctx, cancel: cancel, done: make(chan struct{}), + tlsConfig: &tls.Config{Certificates: []tls.Certificate{cert}, NextProtos: []string{VotorQUICALPN}, MinVersion: tls.VersionTLS13, InsecureSkipVerify: true}, + quicConfig: newVotorQUICConfig(), + jobs: make(chan votorPeerJob, 2), desired: map[solana.PublicKey]VotorPeer{peer.Identity: peer}, + conns: make(map[solana.PublicKey]votorConnection), dialing: make(map[solana.PublicKey]*votorDial), connectQueued: make(map[solana.PublicKey]struct{}), + } + t.Cleanup(func() { require.NoError(t, b.Close()) }) + _, err = b.connection(peer) + require.NoError(t, err) + b.connMu.Lock() + sender := b.conns[peer.Identity].sender + b.connMu.Unlock() + return b, sender, r +} + +func TestVotorPeerReconnectOnRemoteCloseWithoutReconcile(t *testing.T) { + for _, duringDequeue := range []bool{false, true} { + name := "idle" + if duringDequeue { + name = "dequeue" + } + t.Run(name, func(t *testing.T) { + b, s, receiver := passiveVotorSender(t) + if duringDequeue { + func() { + s.mu.Lock() + defer s.mu.Unlock() + s.queue <- votorDatagram{payload: []byte{1}, queuedAt: time.Now()} + // Force the job branch to win select, then close remotely + // while the sender is waiting to check the connection state. + require.Eventually(t, func() bool { return len(s.queue) == 0 }, time.Second, time.Millisecond) + require.NoError(t, receiver.Close()) + select { + case <-s.conn.Context().Done(): + case <-time.After(time.Second): + t.Fatal("remote close not observed") + } + }() + } else { + require.NoError(t, receiver.Close()) + } + select { + case <-s.done: + case <-time.After(time.Second): + t.Fatal("sender did not exit") + } + select { + case job := <-b.jobs: + require.Equal(t, s.peer.Identity, job.peer.Identity) + default: + t.Fatal("sender exited without requesting reconnect") + } + require.Empty(t, b.jobs, "only one reconnect request per peer") + }) + } +} + +func TestVotorPeerDeadlineIncludesQueueAge(t *testing.T) { + for _, tc := range []struct { + name string + queued, sending time.Duration + active, expired bool + }{ + {"progress_does_not_reset_age", 950 * time.Millisecond, 100 * time.Millisecond, true, true}, + {"below_deadline", 800 * time.Millisecond, 100 * time.Millisecond, true, false}, + {"blocked_call", 0, time.Second, true, true}, + {"completed_send", 2 * time.Second, 0, false, false}, + } { + t.Run(tc.name, func(t *testing.T) { + b, s, _ := passiveVotorSender(t) + now := time.Now() + s.mu.Lock() + if tc.active { + s.sendingSince = now.Add(-tc.sending) + } + s.lastQueueDelay = tc.queued + s.mu.Unlock() + // Inject a deterministic clock/state boundary. There is no actual + // datagram in progress and no timer goroutine in this fixture. + b.expirePeerSends(now) + require.Equal(t, tc.expired, s.conn.Context().Err() != nil) + if tc.expired { + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + b.expirePeerSends(now.Add(time.Second)) + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + } else { + require.Zero(t, b.peerSendTimeouts.Load()) + } + }) + } +} + +func TestVotorPeerRejectsAgedQueueBeforeQUICEnqueue(t *testing.T) { + b, s, _ := passiveVotorSender(t) + s.enqueue(votorDatagram{payload: []byte{1}, queuedAt: time.Now().Add(-2 * time.Second)}) + select { + case <-s.done: + case <-time.After(time.Second): + t.Fatal("aged queue did not retire its connection") + } + require.Error(t, s.conn.Context().Err()) + require.Zero(t, b.sends.Load(), "stale backlog must not enter the QUIC queue") + require.EqualValues(t, 1, b.peerSendTimeouts.Load()) + require.EqualValues(t, 1, b.peerQueueDiscarded.Load()) + require.Len(t, b.jobs, 1) +} diff --git a/pkg/alpenglow/peer_sender_test.go b/pkg/alpenglow/peer_sender_test.go new file mode 100644 index 000000000..dda145c4a --- /dev/null +++ b/pkg/alpenglow/peer_sender_test.go @@ -0,0 +1,245 @@ +package alpenglow + +import ( + "crypto/ed25519" + "net" + "runtime" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Real QUIC traffic: one path is blackholed after authentication while the +// other stays healthy. This reproduces the original shared-worker failure. +func TestVotorBroadcasterIsolatesBlockedPeer(t *testing.T) { + for _, action := range []string{"reconnect", "close", "depart", "move"} { + t.Run(action, func(t *testing.T) { testVotorBlockedPeer(t, action) }) + } +} + +func testVotorBlockedPeer(t *testing.T, action string) { + const markerSlot = 999999 + marker := make(chan time.Time, 1) + badMarker := make(chan time.Time, 1) + makeReceiver := func(seed byte, record bool) (*Receiver, solana.PublicKey) { + identity := ed25519.NewKeyFromSeed(bytesOf(seed, ed25519.SeedSize)) + r, err := NewReceiver(ReceiverConfig{ + BindAddr: "127.0.0.1:0", Identity: identity, LogInterval: -1, + MaxDatagramsPerSecond: 100000, + AdmitPeer: func(solana.PublicKey) bool { return true }, + AdmitMessage: func(_ solana.PublicKey, m Message) (Message, bool) { + if m.Slot() == markerSlot { + target := badMarker + if record { + target = marker + } + select { + case target <- time.Now(): + default: + } + } + return Message{}, false + }, + }, NewObserver()) + require.NoError(t, err) + runVotorReceiver(t, r) + return r, testVotorPubkey(identity) + } + badReceiver, badID := makeReceiver(191, false) + goodReceiver, goodID := makeReceiver(192, true) + badAddr := badReceiver.Addr().(*net.UDPAddr) + proxy, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + var blackhole atomic.Bool + proxyDone := make(chan struct{}) + go func() { + defer close(proxyDone) + buf := make([]byte, 65536) + var client *net.UDPAddr + for { + n, from, err := proxy.ReadFromUDP(buf) + if err != nil { + return + } + if blackhole.Load() { + continue + } + if from.String() == badAddr.String() { + if client != nil { + _, _ = proxy.WriteToUDP(buf[:n], client) + } + } else { + client = from + _, _ = proxy.WriteToUDP(buf[:n], badAddr) + } + } + }() + t.Cleanup(func() { _ = proxy.Close(); <-proxyDone }) + badPeer := VotorPeer{Identity: badID, Addr: proxy.LocalAddr().(*net.UDPAddr)} + goodPeer := VotorPeer{Identity: goodID, Addr: goodReceiver.Addr().(*net.UDPAddr)} + peers := newMutableVotorPeers([]VotorPeer{badPeer, goodPeer}) + b, err := NewVotorBroadcaster(VotorBroadcasterConfig{ + Identity: ed25519.NewKeyFromSeed(bytesOf(193, ed25519.SeedSize)), + Peers: peers.Snapshot, + Workers: defaultVotorConnectWorkers, + }) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, b.Close()) }) + require.Eventually(t, func() bool { return b.Stats().Connections == 2 }, 3*time.Second, 5*time.Millisecond) + badConn, ok := b.establishedConnection(badPeer) + require.True(t, ok) + // Establish an actual healthy delivery before introducing the fault. + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x41), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy baseline failed") + } + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("proxied baseline failed") + } + blackholedAt := time.Now() + blackhole.Store(true) + for slot := uint64(1); slot <= 256; slot++ { + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(slot), testSignatureSeq(0x41), 3))) + } + blockedWorkers := func() int { + buf := make([]byte, 2<<20) + stack := string(buf[:runtime.Stack(buf, true)]) + count := 0 + for _, goroutine := range strings.Split(stack, "\n\n") { + if strings.Contains(goroutine, "(*datagramQueue).Add") && strings.Contains(goroutine, "(*votorPeerSender).run") { + count++ + } + } + return count + } + require.Eventually(t, func() bool { return blockedWorkers() == 1 }, 3*time.Second, 5*time.Millisecond) + before := b.Stats() + require.Zero(t, before.MessagesDropped) + goodConn, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + b.connMu.Lock() + badSender := b.conns[badID].sender + b.connMu.Unlock() + start := time.Now() + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x42), 3))) + select { + case received := <-marker: + t.Logf("Healthy marker received in %s while other peer's SendDatagram is blocked", received.Sub(start)) + case <-time.After(250 * time.Millisecond): + t.Fatal("stalled peer delayed healthy delivery") + } + require.NoError(t, badConn.Context().Err(), "marker must arrive before watchdog releases stalled peer") + if action != "reconnect" { + switch action { + case "close": + closed := make(chan struct{}) + go func() { _ = b.Close(); close(closed) }() + select { + case <-closed: + case <-time.After(500 * time.Millisecond): + t.Fatal("Close waited for stalled SendDatagram") + } + case "depart": + peers.Set([]VotorPeer{goodPeer}) + b.reconcilePeers() + case "move": + replacement, _ := makeReceiver(191, false) + badPeer.Addr = replacement.Addr().(*net.UDPAddr) + peers.Set([]VotorPeer{badPeer, goodPeer}) + b.reconcilePeers() + } + select { + case <-badSender.done: + case <-time.After(500 * time.Millisecond): + t.Fatal("old sender did not stop") + } + require.Error(t, badConn.Context().Err()) + require.Empty(t, badSender.queue) + require.Positive(t, b.Stats().PeerQueueDiscarded) + if action == "close" { + return + } + if action == "move" { + require.Eventually(t, func() bool { + conn, ok := b.establishedConnection(badPeer) + return ok && conn != badConn + }, 3*time.Second, 5*time.Millisecond) + } + currentGood, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + require.Same(t, goodConn, currentGood) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x44), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy delivery stopped after peer change") + } + if action == "move" { + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("replacement address did not receive new vote") + } + } + return + } + // Deliberately fill just the stalled peer's bounded queue. Healthy peers + // remain independent even when the failed peer's copies are rejected. + payload, err := EncodeMessage(NewVoteMessage(NewSkipVote(123), testSignatureSeq(0x42), 3)) + require.NoError(t, err) + for range 2 * defaultVotorPeerSendQueue { + badSender.enqueue(votorDatagram{payload: payload, queuedAt: time.Now()}) + } + require.Equal(t, defaultVotorPeerSendQueue, len(badSender.queue)) + require.Positive(t, b.Stats().PeerQueueDrops) + require.Equal(t, b.Stats().PeerQueueDrops, b.Stats().MessagesDropped) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x43), 3))) + select { + case <-marker: + case <-time.After(250 * time.Millisecond): + t.Fatal("full peer queue delayed healthy delivery") + } + // Queue age is measured from fanout, so PTO progress no longer restarts + // the one-second budget. Allow two seconds of test scheduling slack beyond + // one watchdog tick, measured from the fault rather than this assertion. + remaining := time.Until(blackholedAt.Add(votorSendTimeout + votorSendWatchInterval + 2*time.Second)) + require.Positive(t, remaining) + require.Eventually(t, func() bool { return badConn.Context().Err() != nil }, remaining, 5*time.Millisecond) + t.Logf("Blackholed peer retired after %s", time.Since(blackholedAt)) + select { + case <-badSender.done: + case <-time.After(time.Second): + t.Fatal("stalled sender leaked after watchdog closed its connection") + } + require.EqualValues(t, 1, b.Stats().PeerSendTimeouts) + require.Positive(t, b.Stats().PeerQueueDiscarded) + require.Empty(t, badSender.queue) + // The old queue must never be resurrected on a replacement connection. + blackhole.Store(false) + require.Eventually(t, func() bool { + conn, ok := b.establishedConnection(badPeer) + return ok && conn != badConn + }, 5*time.Second, 10*time.Millisecond) + currentGood, ok := b.establishedConnection(goodPeer) + require.True(t, ok) + require.Same(t, goodConn, currentGood) + require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x44), 3))) + select { + case <-marker: + case <-time.After(time.Second): + t.Fatal("healthy peer did not continue after reconnect") + } + select { + case <-badMarker: + case <-time.After(time.Second): + t.Fatal("reconnected peer did not receive new vote") + } +} diff --git a/pkg/alpenglow/testdata/README.md b/pkg/alpenglow/testdata/README.md index 92b75d4eb..10997bf71 100644 --- a/pkg/alpenglow/testdata/README.md +++ b/pkg/alpenglow/testdata/README.md @@ -12,3 +12,12 @@ The v4.3 vote wire message intentionally excludes validator rank and stake; the authenticated Votor transport identity supplies those values after decode. The shred version is the final little-endian `u16`. Certificate bitmap vectors use wincode's default bincode-compatible little-endian `u64` length. + +`agave_votor_certificate.der` is the 249-byte certificate constructed by +`anza-xyz/agave` commit `8fe3f1201abc5b0244540aed0c7bf8c6bcafb3f5`, +`tls-utils/src/tls_certificates.rs::new_dummy_x509_certificate`, with public +key bytes `00..1f` at offsets 100–131. It also matches Firedancer commit +`de039cd7fc9f4714782ec7e3d47db3728903abc6`, +`src/ballet/x509/fd_x509_mock.c::fd_x509_mock_pubkey_v2`. Its fixed dummy +X.509 signature is intentional; TLS CertificateVerify proves possession of +the identity key. This fixture contains no private key. diff --git a/pkg/alpenglow/testdata/agave_votor_certificate.der b/pkg/alpenglow/testdata/agave_votor_certificate.der new file mode 100644 index 0000000000000000000000000000000000000000..e23bc0bfbe3e299e5b8ed04f44aad740f841d225 GIT binary patch literal 249 zcmXqL{ASR&ase|FBNGz`BNQ00vN3C?78r;biWms7F^94+^Kb{}=OpGOD&*y-q#7uQ z^O_qN7y=;}L`m?Q7+9Ji2^cUKXh98OR%BpcWMXDvWn<^yMC+6cQE@6%&_` zl#-T_m6KnrX`pT(4zx#Bkdg5}3$Fop6K76-a$-(KesPHb4@g27B*6qU7UD8yM~43t F0syv*UupmV literal 0 HcmV?d00001 diff --git a/pkg/alpenglow/tls_identity.go b/pkg/alpenglow/tls_identity.go index 1e8b91d91..3c18e0cfd 100644 --- a/pkg/alpenglow/tls_identity.go +++ b/pkg/alpenglow/tls_identity.go @@ -5,9 +5,7 @@ import ( "crypto/rand" "crypto/tls" "crypto/x509" - "crypto/x509/pkix" "fmt" - "math/big" "time" "github.com/gagliardetto/solana-go" @@ -27,6 +25,36 @@ func newVotorQUICConfig() *quic.Config { } } +// votorCertificateTemplate is Agave's dummy X.509 certificate from +// tls-utils/src/tls_certificates.rs (8fe3f1201abc5b0244540aed0c7bf8c6bcafb3f5). +// Firedancer's fd_x509_mock_pubkey_v2 requires this exact encoding except for +// the 32-byte SubjectPublicKeyInfo at offset 100. The X.509 signature is +// deliberately invalid: TLS 1.3 CertificateVerify authenticates the identity. +// Keep the template unchanged and copy it before inserting a public key. +var votorCertificateTemplate = [...]byte{ + 0x30, 0x81, 0xf6, 0x30, 0x81, 0xa9, 0xa0, 0x03, 0x02, 0x01, 0x02, 0x02, + 0x08, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x30, 0x05, 0x06, + 0x03, 0x2b, 0x65, 0x70, 0x30, 0x16, 0x31, 0x14, 0x30, 0x12, 0x06, 0x03, + 0x55, 0x04, 0x03, 0x0c, 0x0b, 0x53, 0x6f, 0x6c, 0x61, 0x6e, 0x61, 0x20, + 0x6e, 0x6f, 0x64, 0x65, 0x30, 0x20, 0x17, 0x0d, 0x37, 0x30, 0x30, 0x31, + 0x30, 0x31, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x5a, 0x18, 0x0f, 0x34, + 0x30, 0x39, 0x36, 0x30, 0x31, 0x30, 0x31, 0x30, 0x30, 0x30, 0x30, 0x30, + 0x30, 0x5a, 0x30, 0x00, 0x30, 0x2a, 0x30, 0x05, 0x06, 0x03, 0x2b, 0x65, + 0x70, 0x03, 0x21, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, + 0xa3, 0x29, 0x30, 0x27, 0x30, 0x17, 0x06, 0x03, 0x55, 0x1d, 0x11, 0x01, + 0x01, 0xff, 0x04, 0x0d, 0x30, 0x0b, 0x82, 0x09, 0x6c, 0x6f, 0x63, 0x61, + 0x6c, 0x68, 0x6f, 0x73, 0x74, 0x30, 0x0c, 0x06, 0x03, 0x55, 0x1d, 0x13, + 0x01, 0x01, 0xff, 0x04, 0x02, 0x30, 0x00, 0x30, 0x05, 0x06, 0x03, 0x2b, + 0x65, 0x70, 0x03, 0x41, 0x00, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, +} + // newVotorQUICCertificate creates the single-certificate Ed25519 identity // chain used by the Agave Votor transport. Peers recover the validator identity // directly from the leaf certificate's SubjectPublicKeyInfo; TLS 1.3's @@ -47,23 +75,8 @@ func newVotorQUICCertificate(identity ed25519.PrivateKey) (tls.Certificate, erro pub = priv.Public().(ed25519.PublicKey) } - template := &x509.Certificate{ - SerialNumber: big.NewInt(1), - Subject: pkix.Name{ - CommonName: "Mithril Alpenglow observer", - }, - NotBefore: time.Unix(0, 0), - NotAfter: time.Date(4096, 1, 1, 0, 0, 0, 0, time.UTC), - KeyUsage: x509.KeyUsageDigitalSignature, - ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageServerAuth, x509.ExtKeyUsageClientAuth}, - DNSNames: []string{"localhost"}, - BasicConstraintsValid: true, - } - - certDER, err := x509.CreateCertificate(rand.Reader, template, template, pub, priv) - if err != nil { - return tls.Certificate{}, err - } + certDER := append([]byte(nil), votorCertificateTemplate[:]...) + copy(certDER[100:100+ed25519.PublicKeySize], pub) cert, err := x509.ParseCertificate(certDER) if err != nil { return tls.Certificate{}, err diff --git a/pkg/alpenglow/tls_identity_test.go b/pkg/alpenglow/tls_identity_test.go index 7efaa4125..cb591e0fa 100644 --- a/pkg/alpenglow/tls_identity_test.go +++ b/pkg/alpenglow/tls_identity_test.go @@ -1,15 +1,61 @@ package alpenglow import ( + "context" "crypto/ed25519" "crypto/tls" "crypto/x509" + "net" + "os" "testing" "time" "github.com/stretchr/testify/require" ) +func TestVotorCertificateMatchesAgaveAndFiredancerTemplate(t *testing.T) { + // Independent fixture from Agave's constructor, also accepted by + // Firedancer's fd_x509_mock_pubkey_v2 byte-pattern parser. + fixture, err := os.ReadFile("testdata/agave_votor_certificate.der") + require.NoError(t, err) + require.Len(t, fixture, 249) + for _, seed := range []byte{61, 62} { + identity := ed25519.NewKeyFromSeed(bytesOf(seed, ed25519.SeedSize)) + certificate, err := newVotorQUICCertificate(identity) + require.NoError(t, err) + expected := append([]byte(nil), fixture...) + copy(expected[100:132], identity.Public().(ed25519.PublicKey)) + require.Equal(t, expected, certificate.Certificate[0]) + require.Equal(t, identity.Public(), certificate.Leaf.PublicKey) + // Do not replace the dummy signature with a real one: Firedancer's + // parser matches it too. Authentication happens in CertificateVerify. + require.Error(t, certificate.Leaf.CheckSignature(certificate.Leaf.SignatureAlgorithm, + certificate.Leaf.RawTBSCertificate, certificate.Leaf.Signature)) + } +} + +func TestVotorCertificateStillRequiresIdentityKeyPossession(t *testing.T) { + identity := ed25519.NewKeyFromSeed(bytesOf(63, ed25519.SeedSize)) + certificate, err := newVotorQUICCertificate(identity) + require.NoError(t, err) + certificate.PrivateKey = ed25519.NewKeyFromSeed(bytesOf(64, ed25519.SeedSize)) + serverConn, clientConn := net.Pipe() + t.Cleanup(func() { _ = serverConn.Close(); _ = clientConn.Close() }) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + server := tls.Server(serverConn, &tls.Config{ + Certificates: []tls.Certificate{certificate}, MinVersion: tls.VersionTLS13, + }) + client := tls.Client(clientConn, &tls.Config{ + InsecureSkipVerify: true, MinVersion: tls.VersionTLS13, + }) + serverDone := make(chan error, 1) + go func() { serverDone <- server.HandshakeContext(ctx) }() + // Skipping the dummy X.509 signature does not bypass CertificateVerify. + require.ErrorContains(t, client.HandshakeContext(ctx), "invalid signature") + require.Error(t, <-serverDone) +} + func TestVotorPeerIdentityRequiresOneEd25519Certificate(t *testing.T) { identity := ed25519.NewKeyFromSeed(bytesOf(61, ed25519.SeedSize)) certificate, err := newVotorQUICCertificate(identity) diff --git a/pkg/alpenglow/vote_history.go b/pkg/alpenglow/vote_history.go index d227bbe12..448977fa3 100644 --- a/pkg/alpenglow/vote_history.go +++ b/pkg/alpenglow/vote_history.go @@ -15,6 +15,7 @@ import ( ) const voteHistoryVersion = 1 +const reservedVoteHistoryVersion = 2 var ErrVoteHistoryNotFound = errors.New("alpenglow vote history not found") @@ -23,6 +24,7 @@ var ErrVoteHistoryNotFound = errors.New("alpenglow vote history not found") // Agave when resuming (notarized blocks and ParentReady edges), in addition to // the anti-equivocation vote sets. type VoteHistory struct { + ReservationRequired bool `json:"reservation_required,omitempty"` Version uint32 `json:"version"` NodePubkey solana.PublicKey `json:"node_pubkey"` Root uint64 `json:"root"` @@ -420,50 +422,111 @@ func VoteHistoryFilename(dir string, node solana.PublicKey) string { return filepath.Join(dir, fmt.Sprintf("vote_history-%s.mithril.json", node)) } -// SaveVoteHistory signs the exact serialized history with the validator -// identity and atomically replaces the previous file before a vote can be -// admitted to consensus or sent to the network. +// SaveVoteHistory authenticates and durably replaces the exact history: write, +// file sync, rename, then directory sync. Synchronous-mode callers require +// success before pool admission (which can publish certificates) or network +// enqueue. The BLS signature may already have been computed privately in RAM; +// this is persist-before-publication, not persist-before-BLS-computation. func SaveVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey) error { + return saveVoteHistory(dir, h, identity, true) +} + +// SaveReservedVoteHistory writes and renames without per-vote sync. Success +// does not prove that this history survived a host/power failure. The caller +// must enforce an independently durable signing reservation and its restart +// quarantine; a valid-looking older history is not evidence of completeness. +func SaveReservedVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey) error { + if h == nil || !h.ReservationRequired { + return fmt.Errorf("reserved history requires durable reservation enrollment") + } + return saveVoteHistory(dir, h, identity, false) +} + +// VoteHistorySnapshot owns signed, immutable bytes. It retains no reference to +// the voter's mutable maps or signing key and can be saved by a worker. +type VoteHistorySnapshot struct { + node solana.PublicKey + encoded []byte +} + +// PrepareReservedVoteHistory validates and signs the complete current history +// without filesystem access. Only the owner of h may call this while mutating h. +func PrepareReservedVoteHistory(h *VoteHistory, identity ed25519.PrivateKey) (*VoteHistorySnapshot, error) { + if h == nil || !h.ReservationRequired { + return nil, fmt.Errorf("reserved history requires durable reservation enrollment") + } + encoded, err := encodeVoteHistory(h, identity) + if err != nil { + return nil, err + } + return &VoteHistorySnapshot{node: h.NodePubkey, encoded: encoded}, nil +} + +// SaveReservedVoteHistorySnapshot replaces history without an explicit sync. +// Success means replacement completed, not durable vote acknowledgement. Callers +// must serialize writes and enforce the independent durable reservation. On an +// unclean restart even an intact snapshot cannot bypass the startup bound. +func SaveReservedVoteHistorySnapshot(dir string, snapshot *VoteHistorySnapshot) error { + if snapshot == nil || len(snapshot.encoded) == 0 { + return fmt.Errorf("save reserved history: empty snapshot") + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return replaceVoteHistoryFile(dir, VoteHistoryFilename(dir, snapshot.node), snapshot.encoded, false) +} + +func saveVoteHistory(dir string, h *VoteHistory, identity ed25519.PrivateKey, durable bool) error { + encoded, err := encodeVoteHistory(h, identity) + if err != nil { + return err + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return replaceVoteHistoryFile(dir, VoteHistoryFilename(dir, h.NodePubkey), encoded, durable) +} + +func encodeVoteHistory(h *VoteHistory, identity ed25519.PrivateKey) ([]byte, error) { if h == nil { - return fmt.Errorf("save vote history: nil history") + return nil, fmt.Errorf("save vote history: nil history") } if len(identity) != ed25519.PrivateKeySize { - return fmt.Errorf("save vote history: invalid identity key size %d", len(identity)) + return nil, fmt.Errorf("save vote history: invalid identity key size %d", len(identity)) } node := solana.PublicKey(identity.Public().(ed25519.PublicKey)) if node != h.NodePubkey { - return fmt.Errorf("save vote history: identity %s does not match history %s", node, h.NodePubkey) + return nil, fmt.Errorf("save vote history: identity %s does not match history %s", node, h.NodePubkey) } h.Version = voteHistoryVersion + if h.ReservationRequired { + h.Version = reservedVoteHistoryVersion + } if err := h.preparePersistedViews(); err != nil { - return fmt.Errorf("save vote history: %w", err) + return nil, fmt.Errorf("save vote history: %w", err) } defer func() { h.PersistedNotarized = nil h.PersistedParentReady = nil }() if err := h.validatePersistedState(); err != nil { - return fmt.Errorf("save vote history: %w", err) + return nil, fmt.Errorf("save vote history: %w", err) } data, err := json.Marshal(h) if err != nil { - return fmt.Errorf("serialize vote history: %w", err) + return nil, fmt.Errorf("serialize vote history: %w", err) } envelope := savedVoteHistory{ - Version: voteHistoryVersion, + Version: h.Version, Node: node, Data: data, Signature: ed25519.Sign(identity, data), } encoded, err := json.Marshal(envelope) if err != nil { - return fmt.Errorf("serialize saved vote history: %w", err) + return nil, fmt.Errorf("serialize saved vote history: %w", err) } - if err := ensureDurableVoteHistoryDirectory(dir); err != nil { - return err - } - filename := VoteHistoryFilename(dir, node) - return persistVoteHistoryFile(dir, filename, encoded) + return encoded, nil } // ensureDurableVoteHistoryDirectory creates each missing path component and @@ -515,6 +578,10 @@ func ensureDurableVoteHistoryDirectory(dir string) error { // alone does not guarantee that either the bytes or the new directory entry // survives a crash. func persistVoteHistoryFile(dir, filename string, encoded []byte) error { + return replaceVoteHistoryFile(dir, filename, encoded, true) +} + +func replaceVoteHistoryFile(dir, filename string, encoded []byte, durable bool) error { temporary, err := os.CreateTemp(dir, "."+filepath.Base(filename)+".tmp-") if err != nil { return fmt.Errorf("create temporary vote history: %w", err) @@ -538,8 +605,10 @@ func persistVoteHistoryFile(dir, filename string, encoded []byte) error { if n != len(encoded) { return fmt.Errorf("write temporary vote history: %w", io.ErrShortWrite) } - if err := temporary.Sync(); err != nil { - return fmt.Errorf("sync temporary vote history: %w", err) + if durable { + if err := temporary.Sync(); err != nil { + return fmt.Errorf("sync temporary vote history: %w", err) + } } closeErr := temporary.Close() closed = true @@ -551,8 +620,8 @@ func persistVoteHistoryFile(dir, filename string, encoded []byte) error { } renamed = true - if err := syncVoteHistoryDirectory(dir); err != nil { - return err + if durable { + return syncVoteHistoryDirectory(dir) } return nil } @@ -586,7 +655,7 @@ func LoadVoteHistory(dir string, node solana.PublicKey) (*VoteHistory, error) { if err := json.Unmarshal(encoded, &envelope); err != nil { return nil, fmt.Errorf("decode saved vote history: %w", err) } - if envelope.Version != voteHistoryVersion || envelope.Node != node { + if (envelope.Version != voteHistoryVersion && envelope.Version != reservedVoteHistoryVersion) || envelope.Node != node { return nil, fmt.Errorf("saved vote history identity/version mismatch") } if !ed25519.Verify(ed25519.PublicKey(node[:]), envelope.Data, envelope.Signature) { @@ -596,7 +665,7 @@ func LoadVoteHistory(dir string, node solana.PublicKey) (*VoteHistory, error) { if err := json.Unmarshal(envelope.Data, &h); err != nil { return nil, fmt.Errorf("decode vote history: %w", err) } - if h.Version != voteHistoryVersion || h.NodePubkey != node { + if h.Version != envelope.Version || h.NodePubkey != node || h.ReservationRequired != (h.Version == reservedVoteHistoryVersion) { return nil, fmt.Errorf("vote history identity/version mismatch") } if err := h.validatePersistedState(); err != nil { diff --git a/pkg/alpenglow/vote_history_snapshot_test.go b/pkg/alpenglow/vote_history_snapshot_test.go new file mode 100644 index 000000000..540e6fa14 --- /dev/null +++ b/pkg/alpenglow/vote_history_snapshot_test.go @@ -0,0 +1,49 @@ +package alpenglow + +import ( + "crypto/ed25519" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestReservedHistorySnapshotIsImmutable(t *testing.T) { + identity := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + node := solana.PublicKey(identity.Public().(ed25519.PublicKey)) + h := NewVoteHistory(node, 10) + h.ReservationRequired = true + block := BlockID{Slot: 11, Hash: solana.Hash{11}} + require.NoError(t, h.AddVote(NewNotarizationVote(11, block.Hash))) + h.NotarizedBlocks[block] = true + h.AddParentReady(12, block) + snapshot, err := PrepareReservedVoteHistory(h, identity) + require.NoError(t, err) + // Mutate/prune every transport-backed collection after taking the snapshot. + h.SetRoot(20) + require.NoError(t, h.AddVote(NewSkipVote(21))) + for i := range identity { + identity[i] = 0 + } + dir := t.TempDir() + require.NoError(t, SaveReservedVoteHistorySnapshot(dir, snapshot)) + loaded, err := LoadVoteHistory(dir, node) + require.NoError(t, err) + require.Equal(t, uint64(10), loaded.Root) + require.True(t, loaded.VotedAt(11)) + require.True(t, loaded.IsBlockNotarized(block)) + require.True(t, loaded.IsParentReady(12, block)) + require.False(t, loaded.HasSkipped(21)) +} + +func TestReservedHistorySnapshotRequiresEnrollmentAndValidHistory(t *testing.T) { + identity := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + h := NewVoteHistory(solana.PublicKey(identity.Public().(ed25519.PublicKey)), 10) + _, err := PrepareReservedVoteHistory(h, identity) + require.Error(t, err) + h.ReservationRequired = true + h.Voted[11] = true // Inconsistent with canonical VotesCast. + _, err = PrepareReservedVoteHistory(h, identity) + require.Error(t, err) + require.Error(t, SaveReservedVoteHistorySnapshot(t.TempDir(), &VoteHistorySnapshot{})) +} diff --git a/pkg/alpenglow/vote_reservation.go b/pkg/alpenglow/vote_reservation.go new file mode 100644 index 000000000..b577da189 --- /dev/null +++ b/pkg/alpenglow/vote_reservation.go @@ -0,0 +1,104 @@ +package alpenglow + +import ( + "crypto/ed25519" + "crypto/sha256" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + + "github.com/gagliardetto/solana-go" + "golang.org/x/sys/unix" +) + +// VoteReservation is the durable upper bound on slots this identity may sign, +// not a record of slots actually signed. It must survive independently of +// AccountsDB checkpoints and must never be restored from an older snapshot. +// A signature authenticates this file; it does not prove freshness against +// rollback, nor does Generation provide an external monotonic counter. +// CleanHistoryDigest permits exact-history vote recovery only after validation +// and durable consumption by a dirty successor before signing. It does not +// certify complete leader-block history. See docs/reserved-vote-history.md. +type VoteReservation struct { + Version uint32 `json:"version"` + Node solana.PublicKey `json:"node"` + VoteAccount solana.PublicKey `json:"vote_account"` + AuthorizedVoter solana.PublicKey `json:"authorized_voter"` + Genesis solana.Hash `json:"genesis"` + ShredVersion uint16 `json:"shred_version"` + Generation uint64 `json:"generation"` + Through uint64 `json:"through"` + CleanHistoryDigest []byte `json:"clean_history_digest,omitempty"` +} + +func VoteReservationFilename(dir string, node solana.PublicKey) string { + return filepath.Join(dir, fmt.Sprintf("vote_reservation-%s.mithril.json", node)) +} + +// LockVoteHistory excludes concurrent owners of this directory. Operators must +// still fence copies of the same identity on other hosts or in other paths. +func LockVoteHistory(dir string, node solana.PublicKey) (*os.File, error) { + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return nil, err + } + f, err := os.OpenFile(filepath.Join(dir, ".vote_history-"+node.String()+".lock"), os.O_CREATE|os.O_RDWR, 0600) + if err != nil { + return nil, err + } + if err := unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB); err != nil { + f.Close() + return nil, fmt.Errorf("vote history already owned: %w", err) + } + return f, nil +} + +func LoadVoteReservation(dir string, node solana.PublicKey) (VoteReservation, error) { + var r VoteReservation + encoded, err := os.ReadFile(VoteReservationFilename(dir, node)) + if err != nil { + return r, err + } + var envelope savedVoteHistory + if err := json.Unmarshal(encoded, &envelope); err != nil { + return r, err + } + if envelope.Version != 1 || envelope.Node != node || !ed25519.Verify(ed25519.PublicKey(node[:]), envelope.Data, envelope.Signature) { + return r, errors.New("invalid vote reservation signature/version/identity") + } + if err := json.Unmarshal(envelope.Data, &r); err != nil { + return r, err + } + if r.Version != 1 || r.Node != node || r.Generation == 0 || (len(r.CleanHistoryDigest) != 0 && len(r.CleanHistoryDigest) != sha256.Size) { + return r, errors.New("invalid vote reservation record") + } + return r, nil +} + +func SaveVoteReservation(dir string, r VoteReservation, identity ed25519.PrivateKey) error { + if len(identity) != ed25519.PrivateKeySize || solana.PublicKey(identity.Public().(ed25519.PublicKey)) != r.Node || r.Version != 1 || r.Generation == 0 { + return errors.New("invalid vote reservation signer/record") + } + data, err := json.Marshal(r) + if err != nil { + return err + } + encoded, err := json.Marshal(savedVoteHistory{Version: 1, Node: r.Node, Data: data, Signature: ed25519.Sign(identity, data)}) + if err != nil { + return err + } + if err := ensureDurableVoteHistoryDirectory(dir); err != nil { + return err + } + return persistVoteHistoryFile(dir, VoteReservationFilename(dir, r.Node), encoded) +} + +func VoteHistoryDigest(dir string, node solana.PublicKey) ([]byte, error) { + data, err := os.ReadFile(VoteHistoryFilename(dir, node)) + if err != nil { + return nil, err + } + digest := sha256.Sum256(data) + return digest[:], nil +} diff --git a/pkg/alpenglow/vote_reservation_test.go b/pkg/alpenglow/vote_reservation_test.go new file mode 100644 index 000000000..9e1840ade --- /dev/null +++ b/pkg/alpenglow/vote_reservation_test.go @@ -0,0 +1,71 @@ +package alpenglow + +import ( + "crypto/ed25519" + "encoding/json" + "os" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestReservedHistoryFormatAndIntegrity(t *testing.T) { + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + node := solana.PublicKey(key.Public().(ed25519.PublicKey)) + dir := t.TempDir() + h := NewVoteHistory(node, 39) + require.Error(t, SaveReservedVoteHistory(dir, h, key)) + h.ReservationRequired = true + require.NoError(t, h.AddVote(NewSkipVote(44))) + require.NoError(t, SaveReservedVoteHistory(dir, h, key)) + raw, err := os.ReadFile(VoteHistoryFilename(dir, node)) + require.NoError(t, err) + var envelope savedVoteHistory + require.NoError(t, json.Unmarshal(raw, &envelope)) + require.Equal(t, uint32(2), envelope.Version, "legacy reader must reject hybrid history") + loaded, err := LoadVoteHistory(dir, node) + require.NoError(t, err) + require.True(t, loaded.HasSkipped(44)) + require.True(t, loaded.ReservationRequired) + envelope.Data[10] ^= 1 + raw, err = json.Marshal(envelope) + require.NoError(t, err) + require.NoError(t, os.WriteFile(VoteHistoryFilename(dir, node), raw, 0600)) + _, err = LoadVoteHistory(dir, node) + require.Error(t, err) +} + +func BenchmarkVoteHistoryPersistence(b *testing.B) { + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + node := solana.PublicKey(key.Public().(ed25519.PublicKey)) + for _, reserved := range []bool{false, true} { + name := "synchronous" + if reserved { + name = "reserved-write-rename" + } + b.Run(name, func(b *testing.B) { + dir := b.TempDir() + h := NewVoteHistory(node, 39) + h.ReservationRequired = reserved + for slot := uint64(40); slot < 72; slot++ { + if err := h.AddVote(NewNotarizationVote(slot, solana.Hash{byte(slot)})); err != nil { + b.Fatal(err) + } + } + save := SaveVoteHistory + if reserved { + save = SaveReservedVoteHistory + } + if err := SaveVoteHistory(dir, h, key); err != nil { + b.Fatal(err) + } + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := save(dir, h, key); err != nil { + b.Fatal(err) + } + } + }) + } +} diff --git a/pkg/block/block.go b/pkg/block/block.go index 23bfd1f03..459fd68d5 100644 --- a/pkg/block/block.go +++ b/pkg/block/block.go @@ -14,15 +14,30 @@ import ( "github.com/gagliardetto/solana-go/rpc" ) -// TurbineIngressTimings is the per-slot decomposition carried only by a +// TurbineIngressTimings carries per-slot observations only on a // trusted in-memory Turbine block. Durations never serialize with Block. type TurbineIngressTimings struct { ShredCollection time.Duration CompletionQueueDelay time.Duration BlockDecode time.Duration + // Completion-only parse and outstanding-signature join/verification time. TransactionParse time.Duration TransactionSigverify time.Duration ReplayAdmission time.Duration + // Early durations sum completed prefetched component work, including an + // optimistic prefix later discarded, and overlap reception and each other. + // EarlyTransactionSigverify includes queueing through future completion; + // neither early duration is CPU time or an additive pipeline wall stage. + EarlyTransactionParse time.Duration + EarlyTransactionSigverify time.Duration + // Completion wait for already-claimed background parsing/submission. + // Recorded separately from BlockDecode's active completion work. + EarlyPreparationWait time.Duration + // Only retained transactions whose verification finished by ShredFullNanos. + EarlyVerifiedTransactions uint64 + // FullToReady is wall time from full shred assembly to replay-ready completion. + // It contains completion queueing, decode and outstanding verification waits. + FullToReady time.Duration } var transactionDerivedStateInitMu sync.Mutex diff --git a/pkg/block/block_test.go b/pkg/block/block_test.go index 2ffcd3029..31ac89e61 100644 --- a/pkg/block/block_test.go +++ b/pkg/block/block_test.go @@ -11,7 +11,12 @@ func TestTransactionSignaturesVerifiedMarkerIsNotSerialized(t *testing.T) { original.MarkTransactionSignaturesVerified() admissionStart := time.Now() original.MarkTurbineReplayAdmissionStart(admissionStart) - ingress := TurbineIngressTimings{ShredCollection: 12 * time.Millisecond, TransactionSigverify: 34 * time.Millisecond} + ingress := TurbineIngressTimings{ + ShredCollection: 12 * time.Millisecond, TransactionSigverify: 34 * time.Millisecond, + EarlyTransactionParse: 2 * time.Millisecond, EarlyTransactionSigverify: 56 * time.Millisecond, + EarlyPreparationWait: 3 * time.Millisecond, + EarlyVerifiedTransactions: 80, FullToReady: 35 * time.Millisecond, + } original.MarkTurbineIngressTimings(ingress) if got, ok := original.TurbineIngressTimings(); !ok || got != ingress { t.Fatalf("ingress timings = %+v, %t; want %+v, true", got, ok, ingress) diff --git a/pkg/block/verified_message_identity.go b/pkg/block/verified_message_identity.go new file mode 100644 index 000000000..9eff1d264 --- /dev/null +++ b/pkg/block/verified_message_identity.go @@ -0,0 +1,37 @@ +package block + +import ( + "fmt" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// CacheVerifiedTransactionMessageIdentities publishes identities from joined +// signature-verification requests. Every result must cover the exact ordered +// transaction slice. Failed/partial requests and obsolete prefetch generations +// cannot seed the cache. It does not replace block-wide duplicate/status checks. +func (b *Block) CacheVerifiedTransactionMessageIdentities(identities []txverify.VerifiedMessageIdentity) error { + if b == nil || len(identities) != len(b.Transactions) { + return fmt.Errorf("verified message identities do not cover block transactions") + } + prepared := &PreparedTransactionMessageIdentities{ + transactions: append([]*solana.Transaction(nil), b.Transactions...), + versions: make([]solana.MessageVersion, len(identities)), + identities: make([]txstatus.TransactionMessageIdentity, len(identities)), + } + for i, tx := range b.Transactions { + identity, ok := identities[i].ForTransaction(tx) + if !ok { + return fmt.Errorf("verified message identity does not match transaction %d", i) + } + prepared.versions[i] = tx.Message.GetVersion() + prepared.identities[i] = identity + } + state := b.transactionState() + state.mu.Lock() + defer state.mu.Unlock() + state.messageIdentities = prepared + return nil +} diff --git a/pkg/block/verified_message_identity_test.go b/pkg/block/verified_message_identity_test.go new file mode 100644 index 000000000..c82d5a4d2 --- /dev/null +++ b/pkg/block/verified_message_identity_test.go @@ -0,0 +1,42 @@ +package block + +import ( + "crypto/ed25519" + "encoding/json" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestVerifiedIdentityCacheOwnsStorageAndDoesNotSerializeTrust(t *testing.T) { + tx := identityTestTransaction(1) + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + tx.Message.AccountKeys[0] = solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + msg, err := tx.Message.MarshalBinary() + require.NoError(t, err) + tx.Signatures[0] = solana.SignatureFromBytes(ed25519.Sign(key, msg)) + blk := &Block{Transactions: []*solana.Transaction{tx}} + ids := make([]txverify.VerifiedMessageIdentity, 1) + errs := make([]error, 1) + var verifier txverify.BatchVerifier + verifier.VerifyWithMessageIdentities(blk.Transactions, errs, ids) + require.NoError(t, errs[0]) + require.NoError(t, blk.CacheVerifiedTransactionMessageIdentities(ids)) + cached := blk.transactionDerivedState.messageIdentities + require.NotNil(t, cached, "adoption must populate the cache before the first admission lookup") + clear(ids) + got, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, cached, got) + copyBlock := *blk + got, err = copyBlock.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, cached, got) + wire, err := json.Marshal(blk) + require.NoError(t, err) + var decoded Block + require.NoError(t, json.Unmarshal(wire, &decoded)) + require.Nil(t, decoded.transactionDerivedState) +} diff --git a/pkg/blockprod/bank.go b/pkg/blockprod/bank.go index 26784fdf5..b93e07d48 100644 --- a/pkg/blockprod/bank.go +++ b/pkg/blockprod/bank.go @@ -3,6 +3,7 @@ package blockprod import ( "sync" + "github.com/Overclock-Validator/mithril/pkg/arena" "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/fees" @@ -44,6 +45,10 @@ type WorkingBank struct { // seenMessages is the bank-local AlreadyProcessed status set. The TPU's // signature LRU is only an ingress optimization and is not authoritative. seenMessages map[[32]byte]struct{} + // Execution and commit are serialized by mu. Borrowed accounts never escape + // either phase, so their storage can be reset for the next transaction. + borrowedAccounts *arena.Arena[sealevel.BorrowedAccount] + preparer *replay.TransactionPreparer } type BankConfig struct { @@ -71,7 +76,12 @@ func NewWorkingBank(cfg BankConfig) *WorkingBank { if sink == nil { sink = NopBatchSink{} } + var preparer *replay.TransactionPreparer + if cfg.SlotCtx != nil { + preparer = replay.NewTransactionPreparer(cfg.SlotCtx.Features) + } return &WorkingBank{ + preparer: preparer, slotCtx: cfg.SlotCtx, slot: cfg.Slot, leader: cfg.Leader, @@ -82,6 +92,7 @@ func NewWorkingBank(cfg BankConfig) *WorkingBank { accepting: true, ancestorStatuses: cfg.TransactionStatuses, seenMessages: make(map[[32]byte]struct{}), + borrowedAccounts: arena.New[sealevel.BorrowedAccount](64), } } @@ -175,11 +186,30 @@ func (b *WorkingBank) Forge(wire []byte) (ForgeResult, costmodel.ExceedReason) { // ForgeTransaction executes and commits a parsed transaction. func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (ForgeResult, costmodel.ExceedReason) { + return b.forgeTransaction(tx, wireSize, nil) +} + +// ForgePreparedTransaction reuses static work from owned, immutable TPU bytes. +// A different bank feature snapshot falls back to the ordinary execution path. +func (b *WorkingBank) ForgePreparedTransaction(tx *solana.Transaction, wireSize int, prepared *replay.PreparedTransaction) (ForgeResult, costmodel.ExceedReason) { + if b.slotCtx == nil || !b.preparer.Matches(prepared, tx, b.slotCtx.Features) { + prepared = nil + } + return b.forgeTransaction(tx, wireSize, prepared) +} + +func (b *WorkingBank) forgeTransaction(tx *solana.Transaction, wireSize int, prepared *replay.PreparedTransaction) (ForgeResult, costmodel.ExceedReason) { if tx == nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone } - messageHash, err := replay.TransactionMessageHash(tx) + var messageHash [32]byte + var err error + if prepared != nil { + messageHash = prepared.MessageHash() + } else { + messageHash, err = replay.TransactionMessageHash(tx) + } if err != nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone @@ -201,7 +231,11 @@ func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (Fo f := features.NewFeaturesDefault() feats = f } - cost, err = costmodel.EstimateTransactionCost(tx, feats) + if prepared != nil { + cost = prepared.Cost() + } else { + cost, err = costmodel.EstimateTransactionCost(tx, feats) + } if err != nil { b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone @@ -238,15 +272,17 @@ func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (Fo if reason := b.reserveEntryBytesLocked(wireSize); reason != costmodel.ExceedNone { return ForgeDroppedCost, reason } - if err := fees.PayerCanFund(b.slotCtx, tx); err != nil { + if err := b.preparer.PayerCanFund(b.slotCtx, tx, prepared); err != nil { return ForgeDroppedExecution, costmodel.ExceedNone } - output := replay.LoadAndExecuteTransaction(replay.LoadAndExecuteTransactionInput{ - SlotCtx: b.slotCtx, - Transaction: tx, - LeanResult: true, - }) + output := b.preparer.LoadAndExecute(replay.LoadAndExecuteTransactionInput{ + SlotCtx: b.slotCtx, + Transaction: tx, + LeanResult: true, + SkipTimingMetrics: true, + Arena: b.borrowedAccounts, + }, prepared) if output.ProcessingResult.TransactionError != nil { feeInfo, err := replay.ApplyFeesOnlyTransaction(b.slotCtx, tx, output) if err != nil { diff --git a/pkg/blockprod/bank_test.go b/pkg/blockprod/bank_test.go index 53db0d5d4..98b9f7d13 100644 --- a/pkg/blockprod/bank_test.go +++ b/pkg/blockprod/bank_test.go @@ -797,6 +797,19 @@ func TestEntryBuilderFlush(t *testing.T) { assert.Greater(t, batchBytes, 0) } +func TestEntryBuilderDefaultTargetCoalescesTransactions(t *testing.T) { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{0xcd}) + for seq := uint64(0); seq < 2; seq++ { + wire := txfixture.MustSignedTransferWire(seq) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + entries, _, flushed := builder.Append(*tx, len(wire)) + assert.False(t, flushed) + assert.Empty(t, entries) + } + assert.Equal(t, 2, builder.PendingCount()) +} + func TestControllerWorkingBank(t *testing.T) { controller := NewController() assert.Nil(t, controller.WorkingBank()) diff --git a/pkg/blockprod/entry.go b/pkg/blockprod/entry.go index 098e3170e..95e51d600 100644 --- a/pkg/blockprod/entry.go +++ b/pkg/blockprod/entry.go @@ -1,11 +1,10 @@ package blockprod import ( - "bytes" - "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/statsd" "github.com/Overclock-Validator/mithril/pkg/turbine" - bin "github.com/gagliardetto/binary" "github.com/gagliardetto/solana-go" ) @@ -17,11 +16,12 @@ const entryBatchOverheadBytes = 8 + 8 + 32 + 8 type EntryBuilder struct { limits costmodel.Limits - pendingTxns []solana.Transaction - pendingWire int - flushedBytes int - reservedBytes int - entryHash solana.Hash + pendingTxns []solana.Transaction + pendingSerializedBytes int + pendingWire int + flushedBytes int + reservedBytes int + entryHash solana.Hash } func NewEntryBuilder(limits costmodel.Limits, entryHash solana.Hash) *EntryBuilder { @@ -102,32 +102,38 @@ func (b *EntryBuilder) dropReservation() { } // Append adds a forged transaction. The pending entry is held until the next -// transaction would overflow one FEC set. A short leftover is only emitted by -// Flush (slot end / Freeze). +// transaction would overflow the configured batch target. A short leftover is only emitted by +// Flush (slot end / Freeze). Appended transactions must remain immutable. func (b *EntryBuilder) Append(tx solana.Transaction, wireSize int) ([]turbine.Entry, int, bool) { + // Canonical component bytes may differ from a transport-size hint. Measure + // once per transaction and reuse the count at flush, preserving slot budgets. + wire, err := tx.MarshalBinary() + if err != nil { + _ = statsd.Count(statsd.BlockProductionEntrySerializationErrors, 1, nil) + mlog.Log.Errorf("entry builder: cannot serialize applied transaction: %v", err) + return nil, 0, false + } if wireSize <= 0 { - wire, err := tx.MarshalBinary() - if err != nil { - return nil, 0, false - } wireSize = len(wire) } b.consumeReserved(wireSize) - if b.wouldOverflowBatch(wireSize) { + if b.wouldOverflowBatch(len(wire)) { flushed, batchBytes := b.flushLocked() b.pendingTxns = append(b.pendingTxns[:0], tx) + b.pendingSerializedBytes = len(wire) b.pendingWire = wireSize return flushed, batchBytes, true } b.pendingTxns = append(b.pendingTxns, tx) + b.pendingSerializedBytes += len(wire) b.pendingWire += wireSize return nil, 0, false } func (b *EntryBuilder) projectedBytes(nextWire int) int { - return entryBatchOverheadBytes + b.pendingWire + nextWire + return entryBatchOverheadBytes + b.pendingSerializedBytes + nextWire } // Flush emits the current pending transactions as a single PoH entry. @@ -151,37 +157,10 @@ func (b *EntryBuilder) flushLocked() ([]turbine.Entry, int) { Txns: txns, }} b.entryHash = entryHash - batchBytes, err := marshalEntryBatchBytes(entries) - if err != nil { - return nil, 0 - } - b.flushedBytes += len(batchBytes) + batchBytes := entryBatchOverheadBytes + b.pendingSerializedBytes + b.flushedBytes += batchBytes b.pendingTxns = b.pendingTxns[:0] b.pendingWire = 0 - return entries, len(batchBytes) -} - -func marshalEntryBatchBytes(entries []turbine.Entry) ([]byte, error) { - var buf bytes.Buffer - enc := bin.NewEncoderWithEncoding(&buf, bin.EncodingBin) - if err := enc.WriteUint64(uint64(len(entries)), bin.LE); err != nil { - return nil, err - } - for _, entry := range entries { - if err := enc.WriteUint64(entry.NumHashes, bin.LE); err != nil { - return nil, err - } - if err := enc.WriteBytes(entry.Hash[:], false); err != nil { - return nil, err - } - if err := enc.WriteUint64(uint64(len(entry.Txns)), bin.LE); err != nil { - return nil, err - } - for i := range entry.Txns { - if err := entry.Txns[i].MarshalWithEncoder(enc); err != nil { - return nil, err - } - } - } - return buf.Bytes(), nil + b.pendingSerializedBytes = 0 + return entries, batchBytes } diff --git a/pkg/blockprod/entry_bench_test.go b/pkg/blockprod/entry_bench_test.go new file mode 100644 index 000000000..08f9bc5e7 --- /dev/null +++ b/pkg/blockprod/entry_bench_test.go @@ -0,0 +1,75 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" +) + +// Measure the producer's batch accounting both alone and through component +// serialization, shred generation, and broadcast-session bookkeeping. +// Transaction execution, peer routing, and UDP are excluded. +func BenchmarkEntryBuilder(b *testing.B) { + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + txns[i] = *tx + } + leader := txfixture.PayerPrivateKey() + for _, tc := range []struct { + name string + shred bool + broadcast bool + }{ + {name: "append"}, + {name: "append-and-shred", shred: true}, + {name: "append-and-broadcast", broadcast: true}, + } { + b.Run(tc.name, func(b *testing.B) { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + shredder := turbine.Shredder{Slot: 100, ParentSlot: 99, Version: 1} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 1, + Broadcaster: &benchmarkPacketBroadcaster{}, + }) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + index := i % len(txns) + entries, _, flushed := builder.Append(txns[index], len(wires[index])) + if flushed && tc.broadcast { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + } + if flushed && tc.shred { + component, err := turbine.NewEntryBatch(entries) + if err != nil { + b.Fatal(err) + } + if _, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, false, solana.Hash{}, 0, 0, + ); err != nil { + b.Fatal(err) + } + } + } + }) + } +} + +type benchmarkPacketBroadcaster struct { + packets int +} + +func (b *benchmarkPacketBroadcaster) Broadcast(packets [][]byte) error { + b.packets += len(packets) + return nil +} diff --git a/pkg/blockprod/entry_test.go b/pkg/blockprod/entry_test.go index 2a0f88fd9..a22ec5e45 100644 --- a/pkg/blockprod/entry_test.go +++ b/pkg/blockprod/entry_test.go @@ -5,6 +5,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" ) @@ -76,3 +77,102 @@ func mustTransferTx(t *testing.T, seq uint64) *solana.Transaction { require.NoError(t, err) return tx } + +func TestEntryBuilderBatchSizingMatchesComponentEncoding(t *testing.T) { + for _, tc := range []struct { + name string + count int + v0 bool + }{ + {name: "legacy", count: 2}, + {name: "mixed-version", count: 2, v0: true}, + {name: "more-than-128-transactions", count: 129, v0: true}, + } { + t.Run(tc.name, func(t *testing.T) { + txns := make([]solana.Transaction, tc.count) + for i := range txns { + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(uint64(i))) + require.NoError(t, err) + if tc.v0 && i%2 != 0 { + tx.Message.SetVersion(solana.MessageVersionV0) + } + txns[i] = *tx + } + component, err := turbine.NewEntryBatch([]turbine.Entry{{NumHashes: 1, Txns: txns}}) + require.NoError(t, err) + encoded, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + limits := costmodel.DefaultLimits() + limits.MaxBatchBytes = uint64(len(encoded)) + builder := NewEntryBuilder(limits, solana.Hash{1}) + + totalFlushed := 0 + checkBatch := func(entries []turbine.Entry, batchBytes int, want []solana.Transaction) { + t.Helper() + require.Len(t, entries, 1) + require.Equal(t, want, entries[0].Txns) + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + encoded, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + require.Equal(t, len(encoded), batchBytes) + totalFlushed += batchBytes + require.Equal(t, totalFlushed, builder.FlushedBytes()) + } + + // Repeat after both an automatic and explicit flush to catch stale + // size accounting. The limit fits the complete batch exactly. + for round := 0; round < 2; round++ { + pendingWire := 0 + for i := range txns { + wire, err := txns[i].MarshalBinary() + require.NoError(t, err) + wireHint := len(wire) + switch i % 3 { + case 0: + wireHint += 100 // Transport bytes must not affect batch sizing. + case 1: + wireHint = 0 // An absent hint must use the serialized length. + } + if wireHint > 0 { + pendingWire += wireHint + } else { + pendingWire += len(wire) + } + entries, _, flushed := builder.Append(txns[i], wireHint) + require.False(t, flushed, "transaction %d", i) + require.Empty(t, entries) + } + require.Equal(t, tc.count, builder.PendingCount()) + require.Equal(t, pendingWire, builder.PendingWireBytes()) + require.Equal(t, len(encoded), builder.projectedBytes(0)) + entries, batchBytes, flushed := builder.Append(txns[0], 0) + require.True(t, flushed) + checkBatch(entries, batchBytes, txns) + require.Equal(t, int(limits.MaxBatchBytes), batchBytes) + require.Equal(t, 1, builder.PendingCount()) + entries, batchBytes = builder.Flush() + checkBatch(entries, batchBytes, txns[:1]) + require.Zero(t, builder.PendingCount()) + require.Zero(t, builder.PendingWireBytes()) + } + }) + } +} + +func TestEntryBuilderAllowsSingleTransactionOverBatchTarget(t *testing.T) { + wire := txfixture.MustSignedTransferWire(0) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + limits := costmodel.DefaultLimits() + limits.MaxBatchBytes = 1 + builder := NewEntryBuilder(limits, solana.Hash{}) + entries, _, flushed := builder.Append(*tx, len(wire)) + require.False(t, flushed) + require.Empty(t, entries) + entries, _, flushed = builder.Append(*tx, len(wire)) + require.True(t, flushed) + require.Len(t, entries, 1) + require.Len(t, entries[0].Txns, 1) + require.Equal(t, 1, builder.PendingCount()) +} diff --git a/pkg/blockprod/leader.go b/pkg/blockprod/leader.go index e304ffb8d..7dc150967 100644 --- a/pkg/blockprod/leader.go +++ b/pkg/blockprod/leader.go @@ -97,17 +97,19 @@ type LeaderLoop struct { alpenglowClock bool parentContext func(uint64) ParentContext productionParent func(uint64) alpenglow.BlockProductionParent + canSignSlot func(uint64) bool onBlock func(*b.Block) commitLeaderSlot func(replay.CommitLeaderInput) (*sealevel.SlotCtx, error) currentSlot func() uint64 leaderForSlot func(uint64) (solana.PublicKey, bool) - pollInterval time.Duration - slotDuration time.Duration - now func() time.Time - tickStartedAt time.Time - tickDeliveryLag time.Duration + pollInterval time.Duration + slotDuration time.Duration + completionReserve time.Duration + now func() time.Time + tickStartedAt time.Time + tickDeliveryLag time.Duration mu sync.Mutex activeSlot uint64 @@ -145,6 +147,7 @@ type LeaderLoopConfig struct { AlpenglowClock bool ParentContext func(uint64) ParentContext ProductionParent func(slot uint64) alpenglow.BlockProductionParent + CanSignSlot func(slot uint64) bool OnBlock func(*b.Block) CurrentSlot func() uint64 LeaderForSlot func(uint64) (solana.PublicKey, bool) @@ -154,7 +157,10 @@ type LeaderLoopConfig struct { RewardCerts RewardCertBuilder PollInterval time.Duration SlotDuration time.Duration - Now func() time.Time + // CompletionReserve is time retained for finalization and broadcast. Zero + // uses the conservative default; tune only from measured completion times. + CompletionReserve time.Duration + Now func() time.Time } func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { @@ -170,6 +176,9 @@ func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { if cfg.Now == nil { cfg.Now = time.Now } + if cfg.CompletionReserve <= 0 { + cfg.CompletionReserve = leaderBlockCompletionReserve + } return &LeaderLoop{ controller: cfg.Controller, identity: cfg.Identity, @@ -181,12 +190,14 @@ func NewLeaderLoop(cfg LeaderLoopConfig) *LeaderLoop { alpenglowClock: cfg.AlpenglowClock, parentContext: cfg.ParentContext, productionParent: cfg.ProductionParent, + canSignSlot: cfg.CanSignSlot, onBlock: cfg.OnBlock, currentSlot: cfg.CurrentSlot, leaderForSlot: cfg.LeaderForSlot, rewardCerts: cfg.RewardCerts, pollInterval: cfg.PollInterval, slotDuration: cfg.SlotDuration, + completionReserve: cfg.CompletionReserve, now: cfg.Now, finishedLeaderSlots: make(map[uint64]struct{}), pendingFailures: make(map[uint64]leaderSlotFailure), @@ -329,9 +340,11 @@ func (l *LeaderLoop) tickScheduled(scheduledAt time.Time) { delete(l.pendingFailures, targetSlot) openedAt := l.now() l.recordTickTimingLocked(openedAt, "opened") - mlog.Log.InfofPrecise("ALPENGLOW block production: opened local leader slot=%d parent_slot=%d replay_frontier=%d live_slot=%d start_slot_ms=%d%s", + limits := l.activeBank.CostTracker().Limits() + mlog.Log.InfofPrecise("ALPENGLOW block production: opened local leader slot=%d parent_slot=%d replay_frontier=%d live_slot=%d start_slot_ms=%d block_cost_limit=%d account_cost_limit=%d entry_bytes_limit=%d%s", targetSlot, l.parentCtx.ParentSlot, global.ReplayFrontier(), wallSlot, - startDuration.Milliseconds(), l.productionStartTimingDetailLocked(targetSlot, openedAt)) + startDuration.Milliseconds(), limits.BlockCost, limits.WritableAccountCost, limits.MaxEntryBytes, + l.productionStartTimingDetailLocked(targetSlot, openedAt)) return } @@ -586,7 +599,10 @@ func (l *LeaderLoop) productionWindowDeadlineLocked(slot uint64) time.Time { if deadline.IsZero() { return time.Time{} } - reserve := leaderBlockCompletionReserve + reserve := l.completionReserve + if reserve <= 0 { + reserve = leaderBlockCompletionReserve + } if reserve >= l.slotDuration { reserve = l.slotDuration / 4 } @@ -1101,6 +1117,9 @@ func (l *LeaderLoop) revalidateProductionParentForStartLocked(slot uint64, selec } func (l *LeaderLoop) startSlotLocked(slot uint64) error { + if l.canSignSlot != nil && !l.canSignSlot(slot) { + return fmt.Errorf("%w: waiting for durable signing reservation", errParentNotReady) + } selectedParent, parentReadyRequired, err := l.resolveProductionParent(slot) if err != nil { return err @@ -1205,12 +1224,16 @@ func (l *LeaderLoop) startSlotLocked(slot uint64) error { if startEntryHash == (solana.Hash{}) { startEntryHash = parentCtx.ParentBankhash } + limits, err := costmodel.LimitsForSlot(slotCtx.Features, epochSchedule, slot) + if err != nil { + return fmt.Errorf("leader slot limits: %w", err) + } sink := NewShredSink(session) bank := NewWorkingBank(BankConfig{ SlotCtx: slotCtx, Slot: slot, Leader: l.identity.PublicKey(), - Limits: costmodel.LimitsForFeatures(slotCtx.Features), + Limits: limits, EntryHash: startEntryHash, Sink: sink, TransactionStatuses: parentCtx.TransactionStatuses, diff --git a/pkg/blockprod/leader_completion_reserve_test.go b/pkg/blockprod/leader_completion_reserve_test.go new file mode 100644 index 000000000..062bf84d4 --- /dev/null +++ b/pkg/blockprod/leader_completion_reserve_test.go @@ -0,0 +1,32 @@ +package blockprod + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +func TestConfiguredCompletionReserveKeepsProtocolDeadlines(t *testing.T) { + ready := time.Unix(1_700_000_000, 0) + now := ready + loop := NewLeaderLoop(LeaderLoopConfig{ + SlotDuration: AlpenglowSlotDuration, CompletionReserve: 60 * time.Millisecond, + Now: func() time.Time { return now }, + }) + loop.productionWindow = leaderProductionWindow{ + active: true, startSlot: 212, endSlot: 215, nextSlot: 212, readyAt: ready, + } + for offset := uint64(0); offset < 4; offset++ { + slot := 212 + offset + deadline := ready.Add(time.Duration(offset+1) * 200 * time.Millisecond) + require.Equal(t, deadline, loop.productionWindowProtocolDeadlineLocked(slot)) + require.Equal(t, deadline.Add(-60*time.Millisecond), loop.productionWindowDeadlineLocked(slot)) + // The override admits starts during the additional 15ms, but still + // rejects at its own cutoff instead of extending the protocol deadline. + now = deadline.Add(-70 * time.Millisecond) + require.NoError(t, loop.productionStartCutoffErrorLocked(slot)) + now = deadline.Add(-60 * time.Millisecond) + require.ErrorIs(t, loop.productionStartCutoffErrorLocked(slot), errProductionStartCutoffElapsed) + } +} diff --git a/pkg/blockprod/leader_processing_test.go b/pkg/blockprod/leader_processing_test.go new file mode 100644 index 000000000..71303a2cf --- /dev/null +++ b/pkg/blockprod/leader_processing_test.go @@ -0,0 +1,81 @@ +package blockprod + +import ( + "sync/atomic" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/arena" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestLeaderExecutionOptionsPreserveResults(t *testing.T) { + for _, testCase := range []string{"success", "instruction_failure", "payer_failure"} { + t.Run(testCase, func(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(42)) + require.NoError(t, err) + if testCase == "instruction_failure" { + tx.Message.Instructions[0].Data[0] = 0xff + } + if testCase == "payer_failure" { + setPayerLamports(t, env, 1) + } + input := replay.LoadAndExecuteTransactionInput{SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true} + reference := replay.LoadAndExecuteTransaction(input) + input.SkipTimingMetrics = true + // A one-object arena also exercises the heap fallback. Repeated calls + // reset it, while the previous result's account state remains valid. + input.Arena = arena.New[sealevel.BorrowedAccount](1) + before := atomic.LoadUint64(&metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.Count) + beforeDispatch := atomic.LoadUint64(&metrics.GlobalBlockReplay.GetNextIxCtx.Count) + for i := 0; i < 3; i++ { + got := replay.LoadAndExecuteTransaction(input) + require.Equal(t, reference.ProcessingResult, got.ProcessingResult) + require.Equal(t, reference.FeeInfo, got.FeeInfo) + require.Equal(t, reference.LoadedAccountsDataSize, got.LoadedAccountsDataSize) + if reference.ExecCtx != nil { + require.NotNil(t, got.ExecCtx) + require.Equal(t, reference.ExecCtx.ComputeMeter.Used(), got.ExecCtx.ComputeMeter.Used()) + require.Equal(t, reference.ExecCtx.TransactionContext.Accounts.Accounts, got.ExecCtx.TransactionContext.Accounts.Accounts) + } + } + require.Equal(t, before, atomic.LoadUint64(&metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.Count)) + require.Equal(t, beforeDispatch, atomic.LoadUint64(&metrics.GlobalBlockReplay.GetNextIxCtx.Count)) + }) + } +} + +func TestWorkingBankReusesBorrowedAccountsWithoutChangingMessages(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + payerBefore, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + payerBalance := payerBefore.Lamports + destBefore, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + destBalance := destBefore.Lamports + const count = 130 + for i := 0; i < count; i++ { + wire := txfixture.MustSignedTransferWire(uint64(i)) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + result, _ := env.Bank.ForgeTransaction(tx, len(wire)) + require.Equal(t, ForgeAccepted, result) + after, err := tx.MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, after) + } + payerAfter, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + destAfter, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + const transferred = count * (count + 1) / 2 + require.Equal(t, payerBalance-transferred-count*5000, payerAfter.Lamports) + require.Equal(t, destBalance+transferred, destAfter.Lamports) +} diff --git a/pkg/blockprod/leader_signing_reservation_test.go b/pkg/blockprod/leader_signing_reservation_test.go new file mode 100644 index 000000000..4b7e09c10 --- /dev/null +++ b/pkg/blockprod/leader_signing_reservation_test.go @@ -0,0 +1,17 @@ +package blockprod + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestSigningReservationGatesEveryLeaderSlotBeforeBuild(t *testing.T) { + for slot := uint64(40); slot < 44; slot++ { + var checked uint64 + l := &LeaderLoop{canSignSlot: func(s uint64) bool { checked = s; return false }} + require.ErrorIs(t, l.startSlotLocked(slot), errParentNotReady) + require.Equal(t, slot, checked) + // All other builder dependencies are deliberately nil: rejection must happen + // before accessing a working bank, executing or signing any shreds. + } +} diff --git a/pkg/blockprod/leader_throughput_bench_test.go b/pkg/blockprod/leader_throughput_bench_test.go new file mode 100644 index 000000000..006633c93 --- /dev/null +++ b/pkg/blockprod/leader_throughput_bench_test.go @@ -0,0 +1,117 @@ +package blockprod + +import ( + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + computebudget "github.com/gagliardetto/solana-go/programs/compute-budget" +) + +// BenchmarkWorkingBankHotAccounts measures admission, execution, publication, +// and entry batching. Signing and bank setup are outside the measured region. +// All wires are unique within a bank, so this cannot benchmark dedup rejection. +func BenchmarkWorkingBankHotAccounts(b *testing.B) { benchmarkWorkingBankHotAccounts(b, "wire") } +func BenchmarkWorkingBankDecodedHotAccounts(b *testing.B) { + benchmarkWorkingBankHotAccounts(b, "decoded") +} +func BenchmarkWorkingBankPreparedHotAccounts(b *testing.B) { + benchmarkWorkingBankHotAccounts(b, "prepared") +} + +func benchmarkWorkingBankHotAccounts(b *testing.B, mode string) { + const perBank = 10000 + for _, workload := range []string{"compute_budget", "transfer"} { + b.Run(workload, func(b *testing.B) { + wires := make([][]byte, perBank) + for i := range wires { + if workload == "transfer" { + wires[i] = txfixture.MustSignedTransferWire(uint64(i)) + continue + } + tx, err := solana.NewTransaction([]solana.Instruction{ + computebudget.NewSetComputeUnitLimitInstruction(uint32(1000 + i)).Build(), + }, txfixture.TestBlockhash(), solana.TransactionPayer(txfixture.PayerPubkey())) + if err != nil { + b.Fatal(err) + } + key := txfixture.PayerPrivateKey() + if _, err = tx.Sign(func(solana.PublicKey) *solana.PrivateKey { return &key }); err != nil { + b.Fatal(err) + } + wires[i], err = tx.MarshalBinary() + if err != nil { + b.Fatal(err) + } + } + decoded := make([]*solana.Transaction, perBank) + for i, wire := range wires { + var err error + decoded[i], err = solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + } + var prepared []*replay.PreparedTransaction + var env *TestEnv + defer func() { + if env != nil { + env.Close() + } + }() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%perBank == 0 { + b.StopTimer() + if env != nil { + env.Close() + } + env = NewTestEnv(TestEnvConfig{}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.Features.EnableFeature(features.RaiseBlockLimitsTo100m, 0) + if err := env.SlotCtx.Accounts.SetAccount(&addresses.ComputeBudgetProgramAddr, &accounts.Account{ + Key: addresses.ComputeBudgetProgramAddr, Lamports: 1, + Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: math.MaxUint64, + }); err != nil { + b.Fatal(err) + } + + if mode == "prepared" { + // Bind the immutable snapshot after all fixture feature setup is complete. + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + if prepared == nil { + prepared = make([]*replay.PreparedTransaction, perBank) + for j, tx := range decoded { + prepared[j] = env.Bank.preparer.Prepare(tx) + if prepared[j] == nil { + b.Fatal("preparation failed") + } + } + } + } + b.StartTimer() + } + var result ForgeResult + var reason costmodel.ExceedReason + switch mode { + case "prepared": + result, reason = env.Bank.ForgePreparedTransaction(decoded[i%perBank], len(wires[i%perBank]), prepared[i%perBank]) + case "decoded": + result, reason = env.Bank.ForgeTransaction(decoded[i%perBank], len(wires[i%perBank])) + default: + result, reason = env.Bank.Forge(wires[i%perBank]) + } + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v / %v", i, result, reason) + } + } + }) + } +} diff --git a/pkg/blockprod/prepared_transaction_test.go b/pkg/blockprod/prepared_transaction_test.go new file mode 100644 index 000000000..9a6153d64 --- /dev/null +++ b/pkg/blockprod/prepared_transaction_test.go @@ -0,0 +1,142 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + computebudget "github.com/gagliardetto/solana-go/programs/compute-budget" + "github.com/gagliardetto/solana-go/programs/system" + "github.com/stretchr/testify/require" +) + +func TestPreparedBankPreservesOutcomes(t *testing.T) { + for _, kind := range []string{"success", "instruction_failure", "fees_only", "payer_changed", "expired", "duplicate", "foreign_message"} { + t.Run(kind, func(t *testing.T) { + reference := NewTestEnv(TestEnvConfig{}) + defer reference.Close() + candidate := NewTestEnv(TestEnvConfig{}) + defer candidate.Close() + tx := mustSignedTransfer(t, 7) + if kind == "instruction_failure" { + tx.Message.Instructions[0].Data[0] = 0xff + } + if kind == "fees_only" { + tx = mustSignBankTestTransaction(t, + computebudget.NewSetLoadedAccountsDataSizeLimitInstruction(1).Build(), + system.NewTransferInstruction(1, txfixture.PayerPubkey(), txfixture.DestPubkey()).Build()) + } + if kind == "expired" { + tx.Message.RecentBlockhash = solana.Hash{99} + } + prepared := replay.NewTransactionPreparer(candidate.SlotCtx.Features.Clone()).Prepare(tx) + require.NotNil(t, prepared) + estimate, err := costmodel.EstimateTransactionCost(tx, candidate.SlotCtx.Features) + require.NoError(t, err) + require.Equal(t, estimate, prepared.Cost()) + if kind == "payer_changed" { + // Preparation succeeded while the payer was funded. Admission must + // still reject after another transaction spends that balance. + setPayerLamports(t, reference, 1) + setPayerLamports(t, candidate, 1) + } + if kind == "foreign_message" { + tx = mustSignedTransfer(t, 19) + } + wire, err := tx.MarshalBinary() + require.NoError(t, err) + for i := 0; i < 2; i++ { + a, ar := reference.Bank.ForgeTransaction(tx, len(wire)) + b, br := candidate.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, a, b) + require.Equal(t, ar, br) + require.Equal(t, reference.Bank.CostTracker().BlockCost(), candidate.Bank.CostTracker().BlockCost()) + require.Equal(t, reference.Bank.TxFeeAccumulator(), candidate.Bank.TxFeeAccumulator()) + for _, key := range []solana.PublicKey{txfixture.PayerPubkey(), txfixture.DestPubkey()} { + ra, err := reference.SlotCtx.GetAccount(key) + require.NoError(t, err) + ca, err := candidate.SlotCtx.GetAccount(key) + require.NoError(t, err) + require.Equal(t, ra, ca) + } + if kind != "duplicate" { + break + } + } + after, err := tx.MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, after) + }) + } +} + +func TestPreparedBankRejectsStaleFeatures(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + future := env.SlotCtx.Features.Clone() + future.EnableFeature(features.EnableTxV1, 0) + tx := mustSignedTransfer(t, 1) + _, err := tx.Message.SetVersion(solana.MessageVersionV1) + require.NoError(t, err) + prepared := replay.NewTransactionPreparer(future).Prepare(tx) + require.NotNil(t, prepared) + require.False(t, env.Bank.preparer.Matches(prepared, tx, env.SlotCtx.Features)) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + result, _ := env.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, ForgeDroppedExecution, result) + require.Empty(t, env.Bank.ForgedTransactions()) + require.Zero(t, env.Bank.TxFeeAccumulator().TotalFees) +} + +func TestPreparedLeaderStillRejectsFeePayerNoOp(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + // Establish the complete feature snapshot before constructing the bank. + env.SlotCtx.Features.EnableFeature(features.RelaxFeePayerConstraint, 0) + bank := NewWorkingBank(BankConfig{SlotCtx: env.SlotCtx, Slot: env.SlotCtx.Slot, + TransactionStatuses: replay.NewTransactionStatusCache().View()}) + tx := mustSignedTransfer(t, 1) + prepared := bank.preparer.Prepare(tx) + require.NotNil(t, prepared) + setPayerLamports(t, env, 1) + preview := bank.preparer.LoadAndExecute(replay.LoadAndExecuteTransactionInput{ + SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true}, prepared) + require.True(t, preview.ProcessedAsNoOp) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + result, _ := bank.ForgePreparedTransaction(tx, len(wire), prepared) + require.Equal(t, ForgeDroppedExecution, result) + require.Empty(t, bank.ForgedTransactions()) + require.Zero(t, bank.TxFeeAccumulator().TotalFees) +} + +func TestPreparedExecutionRechecksStateWithoutMutatingPreparation(t *testing.T) { + env := NewTestEnv(TestEnvConfig{}) + defer env.Close() + tx := mustSignedTransfer(t, 1) + p := replay.NewTransactionPreparer(env.SlotCtx.Features) + prepared := p.Prepare(tx) + require.NotNil(t, prepared) + input := replay.LoadAndExecuteTransactionInput{SlotCtx: env.SlotCtx, Transaction: tx, LeanResult: true} + for _, balance := range []uint64{10_000_000, 1, 20_000_000} { + setPayerLamports(t, env, balance) + got := p.LoadAndExecute(input, prepared) + want := replay.LoadAndExecuteTransaction(input) + require.Equal(t, want.ProcessingResult, got.ProcessingResult) + require.Equal(t, want.FeeInfo, got.FeeInfo) + require.Equal(t, want.LoadedAccountsDataSize, got.LoadedAccountsDataSize) + if want.ExecCtx != nil { + require.Equal(t, want.ExecCtx.TransactionContext.Accounts.Accounts, got.ExecCtx.TransactionContext.Accounts.Accounts) + require.Equal(t, want.ExecCtx.ComputeMeter.Used(), got.ExecCtx.ComputeMeter.Used()) + } + } + // Rent-boundary checks still use the bank's current payer state. + rent := sealevel.NewDefaultRentSysvar() + setPayerLamports(t, env, rent.MinimumBalance(0)+4999) + require.Error(t, p.PayerCanFund(env.SlotCtx, tx, prepared)) +} diff --git a/pkg/blockprod/producer_block_bench_test.go b/pkg/blockprod/producer_block_bench_test.go new file mode 100644 index 000000000..c93c6849d --- /dev/null +++ b/pkg/blockprod/producer_block_bench_test.go @@ -0,0 +1,66 @@ +package blockprod + +import ( + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "testing" +) + +func BenchmarkProducerBlock50k(b *testing.B) { + const transactions = 50000 + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + txns[i] = *tx + } + leader := txfixture.PayerPrivateKey() + var lastBatches, lastPackets int + b.ReportAllocs() + b.ResetTimer() + for round := 0; round < b.N; round++ { + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + sink := &benchmarkPacketBroadcaster{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 1, Broadcaster: sink, + }) + if err := session.BroadcastHeader(solana.Hash{}); err != nil { + b.Fatal(err) + } + batches := 0 + for i := 0; i < transactions; i++ { + idx := i % len(txns) + entries, _, flushed := builder.Append(txns[idx], len(wires[idx])) + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + batches++ + } + } + if entries, _ := builder.Flush(); len(entries) > 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + batches++ + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + b.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + b.Fatal(err) + } + _ = session.BlockID(99, solana.Hash{}) + lastBatches, lastPackets = batches, sink.packets + } + b.StopTimer() + b.ReportMetric(transactions, "transactions/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(lastBatches), "entry-batches/op") + b.ReportMetric(float64(lastPackets), "packets/op") +} diff --git a/pkg/blockprod/producer_branch_bench_test.go b/pkg/blockprod/producer_branch_bench_test.go new file mode 100644 index 000000000..93b9d92e6 --- /dev/null +++ b/pkg/blockprod/producer_branch_bench_test.go @@ -0,0 +1,253 @@ +package blockprod + +import ( + "bytes" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + txwire "github.com/Overclock-Validator/mithril/pkg/tpu/wire" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" +) + +// These complete producer CPU workloads compare branch heads using identical +// pre-signed, parsed fixtures and a packet-counting sink. No execution, admission, +// worker queue, routing, UDP, or receiver work is timed. +const branchComparisonOneFECBytes = 32 * 963 +const branchComparisonSlotEntryCap = 20*1024*1024 - 48 + +func branchMaxSizeTransferPool(tb testing.TB) ([][]byte, []solana.Transaction) { + tb.Helper() + payer, destination := txfixture.PayerPubkey(), txfixture.DestPubkey() + privateKey := txfixture.PayerPrivateKey() + wires := make([][]byte, 512) + txns := make([]solana.Transaction, len(wires)) + for i := range wires { + memo := bytes.Repeat([]byte("m"), 981) + copy(memo, fmt.Sprintf("mithril-max-wire-%04d:", i)) + tx, err := solana.NewTransaction([]solana.Instruction{ + system.NewTransferInstruction(uint64(i+1), payer, destination).Build(), + solana.NewInstruction(solana.MemoProgramID, nil, memo), + }, txfixture.TestBlockhash(), solana.TransactionPayer(payer)) + if err != nil { + tb.Fatal(err) + } + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key == payer { + return &privateKey + } + return nil + }) + if err != nil { + tb.Fatal(err) + } + raw, err := tx.MarshalBinary() + if err != nil { + tb.Fatal(err) + } + if len(raw) != costmodel.PacketDataSize { + tb.Fatalf("wire size %d, expected %d", len(raw), costmodel.PacketDataSize) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + decoded, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := decoded.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := decoded.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("canonical transaction round trip changed bytes") + } + wires[i], txns[i] = raw, *decoded + } + return wires, txns +} + +func branchComparisonFixtures(tb testing.TB, maximum bool) ([][]byte, []solana.Transaction) { + tb.Helper() + if maximum { + return branchMaxSizeTransferPool(tb) + } + wires := txfixture.PrecomputeTransferPool(512) + txns := make([]solana.Transaction, len(wires)) + for i, raw := range wires { + if len(raw) != 215 { + tb.Fatalf("transfer size %d, want 215", len(raw)) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + tx, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := tx.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := tx.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("transfer round trip changed bytes") + } + txns[i] = *tx + } + return wires, txns +} + +type branchComparisonSink struct{ packets int } + +func (s *branchComparisonSink) Broadcast(packets [][]byte) error { + s.packets += len(packets) + return nil +} + +type branchComparisonStats struct { + batches, packets, transactions, maxDataShreds, maxEntryBytes int + blockID solana.Hash +} + +func branchComparisonRun(tb testing.TB, wires [][]byte, txns []solana.Transaction, slots []int, limits costmodel.Limits) branchComparisonStats { + tb.Helper() + var stats branchComparisonStats + var parentID, parentRoot solana.Hash + leader := txfixture.PayerPrivateKey() + for number, count := range slots { + slot := uint64(100 + number) + builder := NewEntryBuilder(limits, solana.Hash{}) + sink := &branchComparisonSink{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: slot, ParentSlot: slot - 1, Version: 1, Broadcaster: sink, + ParentBlockID: parentID, ParentChainedMerkleRoot: parentRoot, + }) + if err := session.BroadcastHeader(parentID); err != nil { + tb.Fatal(err) + } + entryBytes := 0 + for i := 0; i < count; i++ { + index := stats.transactions % len(txns) + entries, size, flushed := builder.Append(txns[index], len(wires[index])) + stats.transactions++ + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + tb.Fatal(err) + } + stats.batches++ + entryBytes += size + } + } + if entries, size := builder.Flush(); len(entries) != 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + tb.Fatal(err) + } + stats.batches++ + entryBytes += size + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + tb.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + tb.Fatal(err) + } + parentID = session.BlockID(slot-1, parentID) + parentRoot = session.ChainedMerkleRoot() + dataShreds := sink.packets / 2 + if dataShreds > costmodel.DefaultMaxDataShredsPerSlot { + tb.Fatal("slot exceeds data-shred budget") + } + if entryBytes > branchComparisonSlotEntryCap { + tb.Fatal("slot exceeds entry-byte budget") + } + if dataShreds > stats.maxDataShreds { + stats.maxDataShreds = dataShreds + } + if entryBytes > stats.maxEntryBytes { + stats.maxEntryBytes = entryBytes + } + stats.packets += sink.packets + } + stats.blockID = parentID + return stats +} + +func branchComparisonCheck(tb testing.TB, got branchComparisonStats, wireSize int, slots []int, batchLimit uint64) { + tb.Helper() + perBatch := (int(batchLimit) - 56) / wireSize + var batches, packets, transactions int + for _, count := range slots { + full, tail := count/perBatch, count%perBatch + batches += full + fecs := full * ((56 + perBatch*wireSize + branchComparisonOneFECBytes - 1) / branchComparisonOneFECBytes) + if tail != 0 { + batches++ + fecs += (56 + tail*wireSize + branchComparisonOneFECBytes - 1) / branchComparisonOneFECBytes + } + packets += (fecs + 3) * 64 // Header, footer, and signed ending tick each use one FEC set. + transactions += count + } + if got.batches != batches || got.packets != packets || got.transactions != transactions { + tb.Fatalf("workload counts %+v, want batches=%d packets=%d transactions=%d", got, batches, packets, transactions) + } + if got.blockID == (solana.Hash{}) { + tb.Fatal("empty block commitment") + } +} + +func TestProducerBranchComparisonFixtures(t *testing.T) { + for _, maximum := range []bool{false, true} { + wires, txns := branchComparisonFixtures(t, maximum) + limits := costmodel.DefaultLimits() + slots := []int{101, 100} + got := branchComparisonRun(t, wires, txns, slots, limits) + branchComparisonCheck(t, got, len(wires[0]), slots, limits.MaxBatchBytes) + t.Logf("wire=%d batch-target=%d batches=%d packets=%d", len(wires[0]), limits.MaxBatchBytes, got.batches, got.packets) + } +} + +func BenchmarkProducerBranchHeads(b *testing.B) { + for _, workload := range []struct { + name string + maximum bool + slots []int + }{ + {name: "small-215B", slots: []int{50000}}, + {name: "maximum-1232B", maximum: true, slots: []int{16667, 16667, 16666}}, + } { + wires, txns := branchComparisonFixtures(b, workload.maximum) + for _, target := range []struct { + name string + bytes uint64 + }{ + {name: "default"}, + {name: "one-fec", bytes: branchComparisonOneFECBytes}, + } { + b.Run(workload.name+"/"+target.name, func(b *testing.B) { + limits := costmodel.DefaultLimits() + if target.bytes != 0 { + limits.MaxBatchBytes = target.bytes + } + var last branchComparisonStats + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + last = branchComparisonRun(b, wires, txns, workload.slots, limits) + } + b.StopTimer() + branchComparisonCheck(b, last, len(wires[0]), workload.slots, limits.MaxBatchBytes) + b.ReportMetric(float64(last.transactions), "transactions/op") + b.ReportMetric(float64(len(workload.slots)), "slots/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(limits.MaxBatchBytes), "batch-target-B") + b.ReportMetric(float64(last.batches), "entry-batches/op") + b.ReportMetric(float64(last.packets), "packets/op") + b.ReportMetric(float64(last.maxDataShreds), "max-data-shreds/slot") + b.ReportMetric(float64(last.maxEntryBytes), "max-entry-B/slot") + }) + } + } +} diff --git a/pkg/blockprod/producer_maxsize_bench_test.go b/pkg/blockprod/producer_maxsize_bench_test.go new file mode 100644 index 000000000..e843baf81 --- /dev/null +++ b/pkg/blockprod/producer_maxsize_bench_test.go @@ -0,0 +1,153 @@ +package blockprod + +import ( + "bytes" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + txwire "github.com/Overclock-Validator/mithril/pkg/tpu/wire" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" +) + +// Generate exactly 1,232-byte, single-signature legacy transactions. Padding is +// a real UTF-8 memo instruction, rather than trailing bytes after a transaction. +func maxSizeTransferPool(tb testing.TB) ([][]byte, []solana.Transaction) { + tb.Helper() + payer, destination := txfixture.PayerPubkey(), txfixture.DestPubkey() + privateKey := txfixture.PayerPrivateKey() + wires := make([][]byte, 512) + txns := make([]solana.Transaction, len(wires)) + for i := range wires { + memo := bytes.Repeat([]byte("m"), 981) + copy(memo, fmt.Sprintf("mithril-max-wire-%04d:", i)) + tx, err := solana.NewTransaction([]solana.Instruction{ + system.NewTransferInstruction(uint64(i+1), payer, destination).Build(), + solana.NewInstruction(solana.MemoProgramID, nil, memo), + }, txfixture.TestBlockhash(), solana.TransactionPayer(payer)) + if err != nil { + tb.Fatal(err) + } + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key == payer { + return &privateKey + } + return nil + }) + if err != nil { + tb.Fatal(err) + } + raw, err := tx.MarshalBinary() + if err != nil { + tb.Fatal(err) + } + if len(raw) != costmodel.PacketDataSize { + tb.Fatalf("wire size %d, expected %d", len(raw), costmodel.PacketDataSize) + } + if _, err := txwire.Sanitize(raw); err != nil { + tb.Fatal(err) + } + decoded, err := solana.TransactionFromBytes(raw) + if err != nil { + tb.Fatal(err) + } + if err := decoded.VerifySignatures(); err != nil { + tb.Fatal(err) + } + encoded, err := decoded.MarshalBinary() + if err != nil || !bytes.Equal(encoded, raw) { + tb.Fatal("canonical transaction round trip changed bytes") + } + wires[i], txns[i] = raw, *decoded + } + return wires, txns +} + +func TestMaxSizeProducerFixture(t *testing.T) { + wires, txns := maxSizeTransferPool(t) + t.Logf("verified %d signed transactions at %d bytes each", len(wires), len(wires[0])) + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + for i := 0; i < 50; i++ { + entries, batchBytes, flushed := builder.Append(txns[i], len(wires[i])) + if i < 49 && flushed { + t.Fatalf("premature flush at %d", i) + } + if i == 49 { + if !flushed || len(entries) != 1 || len(entries[0].Txns) != 49 || batchBytes != 60424 { + t.Fatal("unexpected full-batch size") + } + } + } +} + +// 50k maximum-size transactions require two slots under the current 32,768 +// data-shred cap. Each iteration completes two slots of 25k transactions. +func BenchmarkProducer50kMaxSize(b *testing.B) { + const transactions = 50000 + const transactionsPerSlot = 25000 + wires, txns := maxSizeTransferPool(b) + leader := txfixture.PayerPrivateKey() + var lastBatches, lastPackets, lastMaxSlotDataShreds int + b.ReportAllocs() + b.ResetTimer() + for round := 0; round < b.N; round++ { + var parentID, parentRoot solana.Hash + totalBatches, totalPackets, maxSlotDataShreds := 0, 0, 0 + for block := 0; block < transactions/transactionsPerSlot; block++ { + slot := uint64(100 + block) + builder := NewEntryBuilder(costmodel.DefaultLimits(), solana.Hash{}) + sink := &benchmarkPacketBroadcaster{} + session := turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: leader, Slot: slot, ParentSlot: slot - 1, Version: 1, Broadcaster: sink, + ParentBlockID: parentID, ParentChainedMerkleRoot: parentRoot, + }) + if err := session.BroadcastHeader(parentID); err != nil { + b.Fatal(err) + } + for i := 0; i < transactionsPerSlot; i++ { + idx := (block*transactionsPerSlot + i) % len(txns) + entries, _, flushed := builder.Append(txns[idx], len(wires[idx])) + if flushed { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + totalBatches++ + } + } + if entries, _ := builder.Flush(); len(entries) > 0 { + if err := session.BroadcastEntryBatch(entries); err != nil { + b.Fatal(err) + } + totalBatches++ + } + if err := session.BroadcastFooter(solana.Hash{1}, 0, nil, nil); err != nil { + b.Fatal(err) + } + if err := session.BroadcastEndingTickLast(builder.CurrentEntryHash()); err != nil { + b.Fatal(err) + } + parentID = session.BlockID(slot-1, parentID) + parentRoot = session.ChainedMerkleRoot() + // The generator emits equal numbers of data and coding shreds. + dataShreds := sink.packets / 2 + if dataShreds > costmodel.DefaultMaxDataShredsPerSlot { + b.Fatal("slot exceeds data-shred limit") + } + if dataShreds > maxSlotDataShreds { + maxSlotDataShreds = dataShreds + } + totalPackets += sink.packets + } + lastBatches, lastPackets, lastMaxSlotDataShreds = totalBatches, totalPackets, maxSlotDataShreds + } + b.StopTimer() + b.ReportMetric(transactions, "transactions/op") + b.ReportMetric(transactions/transactionsPerSlot, "slots/op") + b.ReportMetric(float64(len(wires[0])), "wire-B/tx") + b.ReportMetric(float64(lastBatches), "entry-batches/op") + b.ReportMetric(float64(lastPackets), "packets/op") + b.ReportMetric(float64(lastMaxSlotDataShreds), "max-data-shreds/slot") +} diff --git a/pkg/blockprod/readonly_block_bench_test.go b/pkg/blockprod/readonly_block_bench_test.go new file mode 100644 index 000000000..c605243f5 --- /dev/null +++ b/pkg/blockprod/readonly_block_bench_test.go @@ -0,0 +1,189 @@ +package blockprod + +import ( + "crypto/ed25519" + "crypto/sha256" + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +const readonlyBlockAccepted = 48622 // 50M budget, including the next tx's upfront loaded-data reservation. +const readonlyActualCost = 1028 + +type readonlyBlockParent struct{ mem accounts.MemAccounts } + +func (p readonlyBlockParent) GetAccount(_ uint64, key solana.PublicKey) (*accounts.Account, error) { + raw := [32]byte(key) + return p.mem.GetAccount(&raw) +} + +type readonlyBlockFixture struct { + wires [][]byte + txs []*solana.Transaction + payers []solana.PublicKey + parent readonlyBlockParent +} + +func makeReadonlyBlockFixture(tb testing.TB, count int) readonlyBlockFixture { + tb.Helper() + f := readonlyBlockFixture{parent: readonlyBlockParent{accounts.NewMemAccounts()}} + keys := make([]ed25519.PrivateKey, 8) + for i := range keys { + seed := sha256.Sum256([]byte{byte(i), 73}) + keys[i] = ed25519.NewKeyFromSeed(seed[:]) + f.payers = append(f.payers, solana.PublicKeyFromBytes(keys[i].Public().(ed25519.PublicKey))) + } + pool := make([]solana.PublicKey, txfixture.ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + require.NoError(tb, f.parent.mem.SetAccountWithoutLock(pool[i], &accounts.Account{ + Key: pool[i], Lamports: 1_000_000, Data: make([]byte, 9), Owner: solana.PublicKey{11}, RentEpoch: math.MaxUint64, + })) + } + for i := 0; i < count; i++ { + wire, err := txfixture.ReadonlyPairWire(keys[i%8], txfixture.TestBlockhash(), pool, i/8) + require.NoError(tb, err) + tx, err := solana.TransactionFromBytes(wire) + require.NoError(tb, err) + f.wires = append(f.wires, wire) + f.txs = append(f.txs, tx) + } + return f +} + +func (f readonlyBlockFixture) bank(tb testing.TB, sink BatchSink) *TestEnv { + tb.Helper() + // Explicit active 200ms testnet budgets allow identical baseline/candidate + // benchmark fixtures; LimitsForSlot has separate epoch-transition tests. + limits := costmodel.DefaultLimits() + limits.BlockCost, limits.WritableAccountCost = 50_000_000, 20_000_000 + limits.AllocatedDataSizeDelta, limits.MaxEntryBytes = 50_000_000, 10*1024*1024-48 + env := NewTestEnv(TestEnvConfig{Limits: limits, Sink: sink}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.UnrootedRead = f.parent + for _, payer := range f.payers { + require.NoError(tb, env.SlotCtx.Accounts.SetAccountWithoutLock(payer, &accounts.Account{ + Key: payer, Lamports: 10_000_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + })) + } + return env +} + +// This is a capacity/correctness test, not a 200ms deadline assertion. It creates +// a full synthetic block, validates fee/cost accounting, and round-trips every +// emitted entry through the production shred generator and decoder. No network. +func TestReadonlyPairBlockCapacityAndShredRoundTrip(t *testing.T) { + f := makeReadonlyBlockFixture(t, readonlyBlockAccepted+1) + sink := &captureSink{} + env := f.bank(t, sink) + defer env.Close() + for i, tx := range f.txs { + result, reason := env.Bank.ForgeTransaction(tx, len(f.wires[i])) + if i < readonlyBlockAccepted { + require.Equal(t, ForgeAccepted, result, "transaction %d", i) + } else { + require.Equal(t, ForgeDroppedCost, result) + require.Equal(t, costmodel.ExceedBlockCost, reason) + } + } + env.Bank.Freeze() + require.Equal(t, uint64(readonlyBlockAccepted*readonlyActualCost), env.Bank.CostTracker().BlockCost()) + require.Equal(t, uint64(readonlyBlockAccepted), env.Bank.NumSignatures()) + require.Equal(t, uint64(readonlyBlockAccepted*5000), env.Bank.TxFeeAccumulator().TotalFees) + require.Len(t, env.Bank.ForgedTransactions(), readonlyBlockAccepted) + require.LessOrEqual(t, uint64(env.Bank.EntryBytes()), env.Bank.CostTracker().Limits().MaxEntryBytes) + for i, payer := range f.payers { + included := readonlyBlockAccepted / 8 + if i < readonlyBlockAccepted%8 { + included++ + } + acct, err := env.SlotCtx.GetAccount(payer) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000-included*5000), acct.Lamports) + } + gen := turbine.ShredGenerator{Slot: 42, ParentSlot: 41, Version: 7} + var root solana.Hash + var nextData, nextCode uint32 + count, totalBytes := 0, 0 + previous := solana.Hash{0xab} + for i, entries := range sink.batches { + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + raw, err := turbine.MarshalBlockComponent(component) + require.NoError(t, err) + require.Equal(t, len(raw), sink.bytes[i]) + totalBytes += len(raw) + packets, chained, d, c, err := gen.MakeShredsFromData(txfixture.PayerPrivateKey(), raw, false, root, nextData, nextCode) + require.NoError(t, err) + root, nextData, nextCode = chained, d, c + var shreds []*turbine.Shred + for _, packet := range packets { + sh, err := turbine.ParseShred(packet) + require.NoError(t, err) + if sh.Type == turbine.ShredTypeData { + shreds = append(shreds, sh) + } + } + decoded, err := turbine.DecodeEntriesFromDataShreds(shreds) + require.NoError(t, err) + require.Equal(t, entries, decoded) + for _, entry := range decoded { + require.Equal(t, turbine.NextAlpenglowEntryHash(previous, entry.NumHashes, entry.Txns), entry.Hash) + previous = entry.Hash + for _, tx := range entry.Txns { + require.Equal(t, f.txs[count].Signatures, tx.Signatures) + count++ + } + } + } + require.Equal(t, readonlyBlockAccepted, count) + require.Equal(t, env.Bank.EntryBytes(), totalBytes) + require.Equal(t, env.Bank.EntryHash(), previous) + require.Less(t, nextData, uint32(16384)) // Leaves room for header/footer/ending tick. +} + +// One serial caller; signing and fixture/bank setup are excluded. Measures +// admission, execution, account publication, entry building and final flush. +// It excludes signature verification, actual AccountsDB, network and consensus. +func BenchmarkReadonlyPairFullBlock(b *testing.B) { + f := makeReadonlyBlockFixture(b, readonlyBlockAccepted) + for _, mode := range []string{"wire", "decoded"} { + b.Run(mode, func(b *testing.B) { + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + env := f.bank(b, nil) + b.StartTimer() + for j, wire := range f.wires { + var result ForgeResult + if mode == "wire" { + result, _ = env.Bank.Forge(wire) + } else { + result, _ = env.Bank.ForgeTransaction(f.txs[j], len(wire)) + } + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v", j, result) + } + } + env.Bank.Freeze() + b.StopTimer() + if env.Bank.CostTracker().BlockCost() != readonlyBlockAccepted*readonlyActualCost { + b.Fatal("unexpected block cost") + } + env.Close() + b.StartTimer() + } + b.ReportMetric(float64(readonlyBlockAccepted), "tx/block") + }) + } +} diff --git a/pkg/blockprod/readonly_block_prepared_bench_test.go b/pkg/blockprod/readonly_block_prepared_bench_test.go new file mode 100644 index 000000000..68957fcef --- /dev/null +++ b/pkg/blockprod/readonly_block_prepared_bench_test.go @@ -0,0 +1,45 @@ +package blockprod + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/replay" +) + +// Models a queue already statically prepared before leadership. Preparation is +// shifted out of this timed bank phase, not eliminated from validator CPU work. +func BenchmarkReadonlyPairPreparedFullBlock(b *testing.B) { + f := makeReadonlyBlockFixture(b, readonlyBlockAccepted) + setup := f.bank(b, nil) + preparer := replay.NewTransactionPreparer(setup.SlotCtx.Features) + prepared := make([]*replay.PreparedTransaction, len(f.txs)) + for i, tx := range f.txs { + prepared[i] = preparer.Prepare(tx) + if prepared[i] == nil { + b.Fatal("preparation failed") + } + } + setup.Close() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + env := f.bank(b, nil) + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + b.StartTimer() + for j, tx := range f.txs { + result, _ := env.Bank.ForgePreparedTransaction(tx, len(f.wires[j]), prepared[j]) + if result != ForgeAccepted { + b.Fatalf("transaction %d: %v", j, result) + } + } + env.Bank.Freeze() + b.StopTimer() + if env.Bank.CostTracker().BlockCost() != readonlyBlockAccepted*readonlyActualCost { + b.Fatal("unexpected block cost") + } + env.Close() + b.StartTimer() + } + b.ReportMetric(float64(readonlyBlockAccepted), "tx/block") +} diff --git a/pkg/blockprod/readonly_load_bench_test.go b/pkg/blockprod/readonly_load_bench_test.go new file mode 100644 index 000000000..f5d24798d --- /dev/null +++ b/pkg/blockprod/readonly_load_bench_test.go @@ -0,0 +1,95 @@ +package blockprod + +import ( + "crypto/ed25519" + "math" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" +) + +type readonlyBenchParent struct{ mem accounts.MemAccounts } + +func (p readonlyBenchParent) GetAccount(_ uint64, key solana.PublicKey) (*accounts.Account, error) { + raw := [32]byte(key) + return p.mem.GetAccount(&raw) +} + +// Only bank admission/execution/entry building are timed. The immutable parent +// is in memory; this excludes actual AccountsDB, network and signing costs. +func BenchmarkReadonlyPairBank(b *testing.B) { + const perBank = 10000 + parent := readonlyBenchParent{accounts.NewMemAccounts()} + pool := make([]solana.PublicKey, 128) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + parent.mem.SetAccountWithoutLock(pool[i], &accounts.Account{Key: pool[i], Lamports: 1_000_000, Data: make([]byte, 9), Owner: solana.PublicKey{11}, RentEpoch: math.MaxUint64}) + } + txs := make([]*solana.Transaction, perBank) + key := txfixture.PayerPrivateKey() + payer := key.PublicKey() + hash := txfixture.TestBlockhash() + for i := range txs { + n := (i * 7919) % (128 * 127) + a, c := n/127, n%127 + if c >= a { + c++ + } + msg := []byte{1, 0, 2, 3} + msg = append(msg, payer[:]...) + msg = append(msg, pool[a][:]...) + msg = append(msg, pool[c][:]...) + msg = append(msg, hash[:]...) + msg = append(msg, 0) + wire := []byte{1} + wire = append(wire, ed25519.Sign(ed25519.PrivateKey(key), msg)...) + wire = append(wire, msg...) + var err error + txs[i], err = solana.TransactionFromBytes(wire) + if err != nil { + b.Fatal(err) + } + } + var env *TestEnv + var prepared []*replay.PreparedTransaction + defer func() { + if env != nil { + env.Close() + } + }() + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%perBank == 0 { + b.StopTimer() + if env != nil { + env.Close() + } + env = NewTestEnv(TestEnvConfig{}) + env.SlotCtx.Features.EnableFeature(features.RemoveAccountsDeltaHash, 0) + env.SlotCtx.UnrootedRead = parent + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + if prepared == nil { + prepared = make([]*replay.PreparedTransaction, perBank) + for j, t := range txs { + prepared[j] = env.Bank.preparer.Prepare(t) + if prepared[j] == nil { + b.Fatal("preparation failed") + } + } + } + b.StartTimer() + } + outcome, reason := env.Bank.ForgePreparedTransaction(txs[i%perBank], 198, prepared[i%perBank]) + if outcome != ForgeAccepted { + b.Fatalf("%v / %v", outcome, reason) + } + if i%perBank == perBank-1 && env.Bank.CostTracker().BlockCost() != perBank*1028 { + b.Fatalf("unexpected cost %d", env.Bank.CostTracker().BlockCost()) + } + } +} diff --git a/pkg/blockprod/scheduler/buffer.go b/pkg/blockprod/scheduler/buffer.go index 307dc0b28..38bb57a95 100644 --- a/pkg/blockprod/scheduler/buffer.go +++ b/pkg/blockprod/scheduler/buffer.go @@ -4,15 +4,17 @@ import ( "container/heap" "sync" + "github.com/Overclock-Validator/mithril/pkg/replay" "github.com/gagliardetto/solana-go" ) -// MaxBufferedTxns is the hard cap on cross-slot buffered transactions. +// MaxBufferedTxns is the default cap on cross-slot buffered transactions. const MaxBufferedTxns = 2 * 65536 // entry is one buffered, scored transaction. type entry struct { - tx *solana.Transaction + tx *solana.Transaction + prepared *replay.PreparedTransaction // wire is an owned copy of the packet bytes. Parsed tx fields may alias it // (solana-go decoder slices), so it must outlive any use of tx. wire []byte @@ -25,37 +27,34 @@ type entry struct { // (e.g. cost limit). The entry is retained for cross-slot retry. skipGen uint64 - alive bool - maxIdx int - minIdx int + alive bool + // Indexes belong to Buffer.mu. Every buffered entry appears exactly once + // in each heap; -1 denotes absence while an entry is owned by the consumer. + maxIndex, minIndex int } type maxHeap []*entry func (h maxHeap) Len() int { return len(h) } func (h maxHeap) Less(i, j int) bool { - if h[i].reward != h[j].reward { - return h[i].reward > h[j].reward - } - return h[i].seq < h[j].seq // older first on ties + return higherPriority(h[i], h[j]) } func (h maxHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] - h[i].maxIdx = i - h[j].maxIdx = j + h[i].maxIndex, h[j].maxIndex = i, j } func (h *maxHeap) Push(x any) { e := x.(*entry) - e.maxIdx = len(*h) + e.maxIndex = len(*h) *h = append(*h, e) } func (h *maxHeap) Pop() any { old := *h n := len(old) e := old[n-1] + e.maxIndex = -1 old[n-1] = nil *h = old[:n-1] - e.maxIdx = -1 return e } @@ -71,21 +70,20 @@ func (h minHeap) Less(i, j int) bool { } func (h minHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] - h[i].minIdx = i - h[j].minIdx = j + h[i].minIndex, h[j].minIndex = i, j } func (h *minHeap) Push(x any) { e := x.(*entry) - e.minIdx = len(*h) + e.minIndex = len(*h) *h = append(*h, e) } func (h *minHeap) Pop() any { old := *h n := len(old) e := old[n-1] + e.minIndex = -1 old[n-1] = nil *h = old[:n-1] - e.minIdx = -1 return e } @@ -173,21 +171,19 @@ func (b *Buffer) Cleanup(drop func(*entry) bool) int { b.mu.Lock() defer b.mu.Unlock() - var doomed []*entry + dropped := 0 for _, e := range b.byHash { - if e.alive && drop(e) { - doomed = append(doomed, e) + if drop(e) { + b.killLocked(e) + dropped++ } } - for _, e := range doomed { - b.killLocked(e) - } - b.drainDeadLocked() - return len(doomed) + return dropped } func (b *Buffer) pushAliveLocked(e *entry) { e.alive = true + e.maxIndex, e.minIndex = -1, -1 b.byHash[e.messageHash] = e heap.Push(&b.max, e) heap.Push(&b.min, e) @@ -198,56 +194,79 @@ func (b *Buffer) killLocked(e *entry) { if e == nil || !e.alive { return } + if e.maxIndex >= 0 { + heap.Remove(&b.max, e.maxIndex) + } + if e.minIndex >= 0 { + heap.Remove(&b.min, e.minIndex) + } e.alive = false delete(b.byHash, e.messageHash) b.alive-- } func (b *Buffer) popMaxAliveLocked() *entry { - for b.max.Len() > 0 { - e := heap.Pop(&b.max).(*entry) - if !e.alive { - continue - } + if b.max.Len() > 0 { + e := b.max.popEntry() b.killLocked(e) return e } return nil } -func (b *Buffer) peekMinAliveLocked() *entry { - for b.min.Len() > 0 { - if b.min[0].alive { - return b.min[0] +// popEntry moves the winning child into the hole at each level. This avoids +// interface dispatch and swapping two entries at every level of a large queue. +// Ordering is identical to maxHeap.Less, including FIFO for equal rewards. +func (h *maxHeap) popEntry() *entry { + nodes := *h + root := nodes[0] + root.maxIndex = -1 + last := nodes[len(nodes)-1] + nodes[len(nodes)-1] = nil + nodes = nodes[:len(nodes)-1] + if len(nodes) > 0 { + i := 0 + for { + child := 2*i + 1 + if child >= len(nodes) { + break + } + if child+1 < len(nodes) && higherPriority(nodes[child+1], nodes[child]) { + child++ + } + if !higherPriority(nodes[child], last) { + break + } + nodes[i] = nodes[child] + nodes[i].maxIndex = i + i = child } - heap.Pop(&b.min) + nodes[i] = last + last.maxIndex = i + } + *h = nodes + return root +} + +func higherPriority(a, b *entry) bool { + if a.reward != b.reward { + return a.reward > b.reward + } + return a.seq < b.seq +} + +func (b *Buffer) peekMinAliveLocked() *entry { + if b.min.Len() > 0 { + return b.min[0] } return nil } func (b *Buffer) popMinAliveLocked() *entry { - for b.min.Len() > 0 { + if b.min.Len() > 0 { e := heap.Pop(&b.min).(*entry) - if !e.alive { - continue - } b.killLocked(e) return e } return nil } - -func (b *Buffer) drainDeadLocked() { - for b.max.Len() > 0 { - if b.max[0].alive { - break - } - heap.Pop(&b.max) - } - for b.min.Len() > 0 { - if b.min[0].alive { - break - } - heap.Pop(&b.min) - } -} diff --git a/pkg/blockprod/scheduler/buffer_bench_test.go b/pkg/blockprod/scheduler/buffer_bench_test.go new file mode 100644 index 000000000..ccdaa3af0 --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_bench_test.go @@ -0,0 +1,45 @@ +package scheduler + +import ( + "encoding/binary" + "testing" + + "github.com/gagliardetto/solana-go" +) + +// BenchmarkBufferDrain measures selection from a large, already-filled TPU +// queue. Transaction decoding and queue filling are outside the timed region. +func BenchmarkBufferDrain(b *testing.B) { + const count = 120000 + for _, mixed := range []bool{false, true} { + name := "equal_rewards" + if mixed { + name = "mixed_rewards" + } + b.Run(name, func(b *testing.B) { + var buffer *Buffer + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if i%count == 0 { + b.StopTimer() + buffer = NewBuffer(count) + for j := 0; j < count; j++ { + e := &entry{tx: &solana.Transaction{}, seq: uint64(j), reward: 2500} + binary.LittleEndian.PutUint64(e.messageHash[:], uint64(j)) + if mixed { + e.reward = uint64(j*7919) % 1000 + } + if result, _ := buffer.Insert(e); result != InsertAccepted { + b.Fatal("queue fill failed") + } + } + b.StartTimer() + } + if buffer.PopMax() == nil { + b.Fatal("queue drained prematurely") + } + } + }) + } +} diff --git a/pkg/blockprod/scheduler/buffer_order_test.go b/pkg/blockprod/scheduler/buffer_order_test.go new file mode 100644 index 000000000..11a091bfb --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_order_test.go @@ -0,0 +1,106 @@ +package scheduler + +import ( + "container/heap" + "encoding/binary" + "math/rand" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestMaxHeapRemovalMatchesSortedOrder(t *testing.T) { + rng := rand.New(rand.NewSource(417)) + for _, count := range []int{1, 2, 3, 4, 7, 8, 9, 127, 128, 129, 4095, 4096, 4097} { + var h maxHeap + want := make([]*entry, count) + for i := range want { + want[i] = &entry{seq: uint64(i), reward: uint64(rng.Intn(17))} + heap.Push(&h, want[i]) + } + sort.Slice(want, func(i, j int) bool { + if want[i].reward == want[j].reward { + return want[i].seq < want[j].seq + } + return want[i].reward > want[j].reward + }) + backing := h[:cap(h)] + for i, expected := range want { + require.Same(t, expected, h.popEntry(), "count=%d pop=%d", count, i) + require.Nil(t, backing[len(h)], "removed pointer retained") + } + } +} + +// Compare interleaved insertion, eviction, cleanup and removal with a small +// unsorted reference model. Reusing hashes after removal exercises replacement +// entries as well as counterpart removal from both indexed heaps. +func TestBufferMixedOperationsMatchReference(t *testing.T) { + const capacity = 64 + rng := rand.New(rand.NewSource(418)) + b := NewBuffer(capacity) + model := make(map[[32]byte]*entry) + best := func(high bool) *entry { + var found *entry + for _, e := range model { + if found == nil || (high && (e.reward > found.reward || e.reward == found.reward && e.seq < found.seq)) || + (!high && (e.reward < found.reward || e.reward == found.reward && e.seq > found.seq)) { + found = e + } + } + return found + } + for step := 0; step < 10000; step++ { + switch action := rng.Intn(10); { + case action < 7: + e := &entry{tx: &solana.Transaction{}, seq: uint64(step), reward: uint64(rng.Intn(16))} + binary.LittleEndian.PutUint64(e.messageHash[:], uint64(rng.Intn(256))) + wantResult := InsertAccepted + var wantEvicted *entry + if _, duplicate := model[e.messageHash]; duplicate { + wantResult = InsertDuplicate + } else if len(model) == capacity { + lowest := best(false) + if e.reward <= lowest.reward { + wantResult = InsertRejectedCapacity + } else { + wantEvicted = lowest + delete(model, lowest.messageHash) + } + } + if wantResult == InsertAccepted { + model[e.messageHash] = e + } + got, evicted := b.Insert(e) + require.Equal(t, wantResult, got, "step=%d", step) + require.True(t, wantEvicted == evicted, "eviction differs at step=%d", step) + case action < 9: + want := best(true) + got := b.PopMax() + require.True(t, want == got, "selection differs at step=%d", step) + if want != nil { + delete(model, want.messageHash) + } + default: + mod := uint64(rng.Intn(11)) + want := 0 + for hash, e := range model { + if e.seq%11 == mod { + delete(model, hash) + want++ + } + } + require.Equal(t, want, b.Cleanup(func(e *entry) bool { return e.seq%11 == mod })) + } + require.Equal(t, len(model), b.Len(), "step=%d", step) + assertBufferIndexes(t, b) + } + for len(model) > 0 { + want := best(true) + require.Same(t, want, b.PopMax()) + delete(model, want.messageHash) + } + require.Nil(t, b.PopMax()) +} diff --git a/pkg/blockprod/scheduler/buffer_retention_test.go b/pkg/blockprod/scheduler/buffer_retention_test.go new file mode 100644 index 000000000..6b24aac8a --- /dev/null +++ b/pkg/blockprod/scheduler/buffer_retention_test.go @@ -0,0 +1,158 @@ +package scheduler + +import ( + "encoding/binary" + "sync" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func retainedTestEntry(id, reward uint64) *entry { + e := &entry{tx: &solana.Transaction{}, wire: []byte{1, 2, 3}, seq: id, reward: reward} + binary.LittleEndian.PutUint64(e.messageHash[:], id) + return e +} + +// Check both membership and backing-array references: shrinking a slice alone +// must not leave transaction payloads reachable outside its visible length. +func assertBufferIndexes(t *testing.T, b *Buffer) { + t.Helper() + b.mu.Lock() + defer b.mu.Unlock() + require.Equal(t, b.alive, len(b.byHash)) + require.Equal(t, b.alive, len(b.max)) + require.Equal(t, b.alive, len(b.min)) + require.LessOrEqual(t, b.alive, b.capacity) + for i, e := range b.max { + require.True(t, e.alive) + require.Equal(t, i, e.maxIndex) + require.Same(t, e, b.byHash[e.messageHash]) + require.Same(t, e, b.min[e.minIndex]) + if i > 0 { + require.False(t, b.max.Less(i, (i-1)/2)) + } + } + for i, e := range b.min { + require.Equal(t, i, e.minIndex) + require.Same(t, e, b.max[e.maxIndex]) + if i > 0 { + require.False(t, b.min.Less(i, (i-1)/2)) + } + } + for _, backing := range [][]*entry{b.max[:cap(b.max)], b.min[:cap(b.min)]} { + for _, e := range backing[b.alive:] { + require.Nil(t, e) + } + } +} + +func TestBufferConsumedEntriesReleaseBothHeapReferences(t *testing.T) { + b := NewBuffer(256) + pinned := retainedTestEntry(0, 1) + b.Insert(pinned) + for i := uint64(1); i <= 100000; i++ { + e := retainedTestEntry(i, 2) + result, _ := b.Insert(e) + require.Equal(t, InsertAccepted, result) + require.Same(t, e, b.PopMax()) + // The consumer still owns usable payloads after index removal. + require.NotNil(t, e.tx) + require.Equal(t, []byte{1, 2, 3}, e.wire) + if i%1000 == 0 { + assertBufferIndexes(t, b) + } + } + b.Cleanup(func(*entry) bool { return false }) + assertBufferIndexes(t, b) + t.Logf("capacity=%d active=%d max_refs=%d min_refs=%d", b.capacity, b.Len(), len(b.max), len(b.min)) + require.Same(t, pinned, b.PopMax()) + assertBufferIndexes(t, b) +} + +func TestBufferEvictionAndCleanupReleaseBothHeapReferences(t *testing.T) { + b := NewBuffer(2) + pinned := retainedTestEntry(0, 1000000) + b.Insert(pinned) + previous := retainedTestEntry(1, 1) + b.Insert(previous) + for i := uint64(2); i < 10000; i++ { + next := retainedTestEntry(i, i) + result, evicted := b.Insert(next) + require.Equal(t, InsertAccepted, result) + require.Same(t, previous, evicted) + require.False(t, evicted.alive) + require.Equal(t, -1, evicted.maxIndex) + require.Equal(t, -1, evicted.minIndex) + previous = next + if i%100 == 0 { + assertBufferIndexes(t, b) + } + } + require.Equal(t, 1, b.Cleanup(func(e *entry) bool { return e != pinned })) + assertBufferIndexes(t, b) + require.Same(t, pinned, b.PopMax()) + assertBufferIndexes(t, b) +} + +func TestBufferRepeatedRebufferPreservesNewHigherPriorityArrivals(t *testing.T) { + s := New(nil) + s.bankGen = 1 + skipped := retainedTestEntry(1, 10) + skipped.skipGen = s.bankGen + s.buffer.Insert(skipped) + for i := uint64(2); i < 10002; i++ { + low := retainedTestEntry(i, 1) + s.buffer.Insert(low) + picked, retry := s.popSchedulable(s.bankGen) + require.Same(t, low, picked) + require.Equal(t, []*entry{skipped}, retry) + for _, e := range retry { + s.rebuffer(e) + } + if i%100 == 0 { + assertBufferIndexes(t, s.buffer) + } + } + // Preserve the existing scan/retry policy when a higher-fee packet arrives. + high := retainedTestEntry(20000, 20) + s.buffer.Insert(high) + picked, retry := s.popSchedulable(s.bankGen) + require.Same(t, high, picked) + require.Empty(t, retry) + // A later bank may retry the previously skipped transaction. + s.bankGen++ + picked, retry = s.popSchedulable(s.bankGen) + require.Same(t, skipped, picked) + require.Empty(t, retry) + assertBufferIndexes(t, s.buffer) +} + +func TestBufferConcurrentInsertRemovalAndCleanup(t *testing.T) { + b := NewBuffer(64) + var workers sync.WaitGroup + for worker := uint64(0); worker < 4; worker++ { + workers.Go(func() { + for i := uint64(0); i < 2000; i++ { + id := worker*2000 + i + b.Insert(retainedTestEntry(id, id%17)) + } + }) + } + workers.Go(func() { + for i := 0; i < 8000; i++ { + b.PopMax() + } + }) + workers.Go(func() { + for i := 0; i < 100; i++ { + b.Cleanup(func(e *entry) bool { return e.seq%3 == 0 }) + } + }) + workers.Wait() + assertBufferIndexes(t, b) + for b.PopMax() != nil { + } + assertBufferIndexes(t, b) +} diff --git a/pkg/blockprod/scheduler/config_test.go b/pkg/blockprod/scheduler/config_test.go new file mode 100644 index 000000000..9f43ed511 --- /dev/null +++ b/pkg/blockprod/scheduler/config_test.go @@ -0,0 +1,28 @@ +package scheduler + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/stretchr/testify/require" +) + +func TestConfiguredQueueCapacityAndPreparation(t *testing.T) { + defaults := NewWithConfig(nil, Config{}) + require.Equal(t, MaxBufferedTxns, defaults.buffer.Capacity()) + feats := features.NewFeaturesDefault() + custom := NewWithConfig(nil, Config{MaxBufferedTransactions: 2, FeatureSource: func() *features.Features { return feats }}) + require.Equal(t, 2, custom.buffer.Capacity()) + require.NotNil(t, custom.preparer.Load()) + for i := byte(1); i <= 2; i++ { + result, _ := custom.buffer.Insert(testEntry(10, uint64(i), i)) + require.Equal(t, InsertAccepted, result) + } + result, evicted := custom.buffer.Insert(testEntry(9, 3, 3)) + require.Equal(t, InsertRejectedCapacity, result) + require.Nil(t, evicted) + result, evicted = custom.buffer.Insert(testEntry(11, 4, 4)) + require.Equal(t, InsertAccepted, result) + require.Equal(t, uint64(2), evicted.seq) + require.Equal(t, 2, custom.Buffered()) +} diff --git a/pkg/blockprod/scheduler/scheduler.go b/pkg/blockprod/scheduler/scheduler.go index de6196f30..88e6435bf 100644 --- a/pkg/blockprod/scheduler/scheduler.go +++ b/pkg/blockprod/scheduler/scheduler.go @@ -9,6 +9,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/blockprod" "github.com/Overclock-Validator/mithril/pkg/costmodel" "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/replay" "github.com/Overclock-Validator/mithril/pkg/tpu/packet" "github.com/gagliardetto/solana-go" ) @@ -43,9 +44,12 @@ type Stats struct { // Scheduler buffers verified TPU transactions in a reward-ordered heap and // drains into the active WorkingBank when one is published. type Scheduler struct { - banks BankSource - feats *features.Features - buffer *Buffer + banks BankSource + feats *features.Features + buffer *Buffer + preparer atomic.Pointer[replay.TransactionPreparer] + featureSource func() *features.Features + lastPreparationRefresh time.Time seq atomic.Uint64 wake chan struct{} @@ -80,6 +84,37 @@ func New(banks BankSource) *Scheduler { } } +// NewWithFeatureSource prepares queued messages against a replay-tip snapshot. +// The active bank independently verifies feature compatibility before reuse. +func NewWithFeatureSource(banks BankSource, source func() *features.Features) *Scheduler { + return NewWithConfig(banks, Config{FeatureSource: source}) +} + +// Config controls the bounded TPU queue and static preparation source. +type Config struct { + // MaxBufferedTransactions defaults to MaxBufferedTxns when zero. + MaxBufferedTransactions int + FeatureSource func() *features.Features +} + +func NewWithConfig(banks BankSource, cfg Config) *Scheduler { + s := New(banks) + if cfg.MaxBufferedTransactions > 0 { + s.buffer = NewBuffer(cfg.MaxBufferedTransactions) + } + s.featureSource = cfg.FeatureSource + s.refreshPreparation() + return s +} + +func (s *Scheduler) refreshPreparation() { + if s.featureSource == nil || time.Since(s.lastPreparationRefresh) < 100*time.Millisecond { + return + } + s.preparer.Store(replay.NewTransactionPreparer(s.featureSource())) + s.lastPreparationRefresh = time.Now() +} + // Start launches the bank-gated drain loop. func (s *Scheduler) Start(ctx context.Context) { s.startOnce.Do(func() { @@ -138,6 +173,7 @@ func (s *Scheduler) Receive(pkt packet.Packet) { e := &entry{ tx: tx, + prepared: s.preparer.Load().Prepare(tx), wire: wire, wireSize: len(wire), messageHash: messageHash, @@ -219,6 +255,7 @@ func (s *Scheduler) drainLoop(ctx context.Context) { if ctx.Err() != nil { return } + s.refreshPreparation() bank := s.banks.WorkingBank() s.noteBank(bank) if bank == nil { @@ -258,15 +295,10 @@ func (s *Scheduler) drainLoop(ctx context.Context) { continue } - // Prefer the owned wire so forge reparses from stable bytes even if the - // retained tx view was somehow mutated after buffering. - var result blockprod.ForgeResult - var reason costmodel.ExceedReason - if len(e.wire) > 0 { - result, reason = bank.Forge(e.wire) - } else { - result, reason = bank.ForgeTransaction(e.tx, e.wireSize) - } + // Receive owns the wire and its decoded transaction for the entire queue + // lifetime. Execution modifies transaction-local account clones, not + // this immutable message, so reuse the decoded transaction across banks. + result, reason := bank.ForgePreparedTransaction(e.tx, e.wireSize, e.prepared) switch result { case blockprod.ForgeDroppedNoLeader: s.rebuffer(e) diff --git a/pkg/blockprod/scheduler/scheduler_test.go b/pkg/blockprod/scheduler/scheduler_test.go index e4b74e092..f83711312 100644 --- a/pkg/blockprod/scheduler/scheduler_test.go +++ b/pkg/blockprod/scheduler/scheduler_test.go @@ -7,6 +7,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/blockprod" "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/fees" "github.com/Overclock-Validator/mithril/pkg/tpu/packet" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" @@ -244,7 +245,9 @@ func TestMaxBufferedTxnsConstant(t *testing.T) { } func TestSchedulerCopiesPooledPacketBytes(t *testing.T) { - sched := New(blockprod.NewController()) + env := blockprod.NewTestEnv(blockprod.TestEnvConfig{}) + defer env.Close() + sched := NewWithFeatureSource(blockprod.NewController(), func() *features.Features { return env.SlotCtx.Features.Clone() }) pool := packet.NewPool(1) buf, idx, ok := pool.Acquire() require.True(t, ok) @@ -269,11 +272,26 @@ func TestSchedulerCopiesPooledPacketBytes(t *testing.T) { e := sched.buffer.PopMax() require.NotNil(t, e) require.Equal(t, wire, e.wire) + require.NotNil(t, e.prepared) // Re-parse and verify the buffered transaction still has a valid signature. tx, err := solana.TransactionFromBytes(e.wire) require.NoError(t, err) require.Equal(t, e.tx.Signatures[0], tx.Signatures[0]) + + // Exercise the actual drain path using the retained decoded transaction, + // after its original pooled packet has already been overwritten. + res, _ := sched.buffer.Insert(e) + require.Equal(t, InsertAccepted, res) + sched.banks = env.Controller + sched.Start(context.Background()) + defer sched.Stop() + require.Eventually(t, func() bool { return sched.Stats().Accepted == 1 }, time.Second, time.Millisecond) + forged := env.Bank.ForgedTransactions() + require.Len(t, forged, 1) + forgedWire, err := forged[0].MarshalBinary() + require.NoError(t, err) + require.Equal(t, wire, forgedWire) } func TestClassifyBufferedExpired(t *testing.T) { diff --git a/pkg/config/config.go b/pkg/config/config.go index 1e7416c90..501dfa158 100644 --- a/pkg/config/config.go +++ b/pkg/config/config.go @@ -25,6 +25,7 @@ func ApplyDefaults(v *viper.Viper) { v.SetDefault("validator.tpu_quic_bind_addr", "") v.SetDefault("validator.advertised_ip", "") v.SetDefault("validator.tpu_sigverify_workers", 0) + v.SetDefault("validator.wait_to_vote_slot", uint64(0)) } // LedgerConfig holds ledger-related configuration (matches Firedancer [ledger] section) @@ -235,6 +236,9 @@ type ValidatorConfig struct { TPUQUICBindAddr string `toml:"tpu_quic_bind_addr" mapstructure:"tpu_quic_bind_addr"` AdvertisedIP string `toml:"advertised_ip" mapstructure:"advertised_ip"` TPUSigverifyWorkers int `toml:"tpu_sigverify_workers" mapstructure:"tpu_sigverify_workers"` + WaitToVoteSlot uint64 `toml:"wait_to_vote_slot" mapstructure:"wait_to_vote_slot"` // Minimum slot for new votes; automatic startup cutoff still applies + BlockCompletionReserveMs int `toml:"block_completion_reserve_ms" mapstructure:"block_completion_reserve_ms"` + TPUMaxBufferedTransactions int `toml:"tpu_max_buffered_transactions" mapstructure:"tpu_max_buffered_transactions"` } // Config holds all configuration options for Mithril (Firedancer-style hierarchy) diff --git a/pkg/consensus/engine.go b/pkg/consensus/engine.go index f57cb2ab5..5cf2cb203 100644 --- a/pkg/consensus/engine.go +++ b/pkg/consensus/engine.go @@ -389,7 +389,7 @@ func (e *AlpenglowObserverEngine) EnableVoting(cfg VotingConfig) error { // events, matching Agave's initial_parent_ready selection. if slot, parent, ok := voter.history.HighestParentReadyMatching(func(parent alpenglow.BlockID) bool { return !e.ensureChain().IsObjectivelyInvalidBlock(parent) - }); ok && slot > root.Slot { + }); ok && slot > root.Slot && (voter.reservation == nil || voter.reservation.recoverThrough == 0) { if !e.ensurePool().RestoreParentReady(slot, parent) { mlog.Log.FileOnlyf("ALPENGLOW voting: ignored persisted ParentReady slot=%d parent=%s because newer root/live tracker state is authoritative", slot, parent) } @@ -975,11 +975,18 @@ func (e *AlpenglowObserverEngine) injectLocalVote(message alpenglow.VoteMessage, } } -// alpenglowVoteActionFloor is the highest slot on which this validator must -// not initiate a new vote. The pool root is its strict admission boundary; -// direct finality is included because the pool deliberately retains a short -// reward-accounting tail behind finality where network votes remain useful. +// alpenglowVoteActionFloor is the retained pool's strict admission boundary. +// Network finality is not a voting root: a replayed block may still contribute +// a notarization to a later fast certificate and its slot+8 reward certificate. +// The voter also checks its own persisted history root before signing. func (e *AlpenglowObserverEngine) alpenglowVoteActionFloor() uint64 { + return e.ensurePool().Snapshot().RootSlot +} + +// alpenglowVerifiedFinalityFloor releases crash-recovery reservations. Keep this +// independent of live vote admission: a retained reward window must not weaken +// the requirement to pass every slot that may have been signed before a crash. +func (e *AlpenglowObserverEngine) alpenglowVerifiedFinalityFloor() uint64 { floor := e.ensurePool().Snapshot().RootSlot if finalized := e.ensureChain().Snapshot().LatestDirectFinalizedBlock.Slot; finalized > floor { floor = finalized @@ -1542,6 +1549,25 @@ func (e *AlpenglowObserverEngine) PruneAlpenglowBefore(slot uint64) { if slot == 0 { return } + // Replay enqueues its completed-block event before publishing a durable + // promotion. Retire the pool, execution proof and history on that same + // ordered voter stream, so a fast checkpoint cannot overtake the vote. + e.voterMu.RLock() + voter := e.voter + e.voterMu.RUnlock() + if voter != nil { + if err := voter.enqueue(voterEvent{kind: voterEventDurableRoot, slot: slot}); err != nil { + e.latchSafetyError(err) + } + return + } + e.applyAlpenglowDurableRoot(slot) +} + +// applyAlpenglowDurableRoot is called by the voter after earlier replay events, +// or synchronously by an observer without a voting loop. Startup root restore +// remains a separate, immediate barrier in SetAlpenglowRoot. +func (e *AlpenglowObserverEngine) applyAlpenglowDurableRoot(slot uint64) alpenglow.BlockID { e.poolOutputMu.Lock() defer e.poolOutputMu.Unlock() @@ -1557,13 +1583,11 @@ func (e *AlpenglowObserverEngine) PruneAlpenglowBefore(slot uint64) { if e.certPool != nil { e.certPool.ObserveFloor(slot) } - if err := e.enqueueVoter(voterEvent{kind: voterEventRoot, root: root}); err != nil { - e.latchSafetyError(err) - } // Replay calls this only after the fold through slot is durably committed. // Keep the transport peer window tied to that local root, not to speculative // certificate finality or the pool's reward-retention floor. e.advanceVotorPeerRoot(slot) + return root } func (e *AlpenglowObserverEngine) pruneInvalidBlockIDsBefore(slot uint64) { diff --git a/pkg/consensus/vote_history_writer.go b/pkg/consensus/vote_history_writer.go new file mode 100644 index 000000000..9d175d6d4 --- /dev/null +++ b/pkg/consensus/vote_history_writer.go @@ -0,0 +1,125 @@ +package consensus + +import ( + "errors" + "fmt" + "sync" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" +) + +// One serial writer, one in-flight snapshot and at most one newer pending +// snapshot. A newer complete history supersedes an unwritten snapshot; every +// retained, unrooted voting decision is still present in that newer history. +// The independently durable reservation, not this queue, authorizes signing. +// In-flight/pending snapshots may be lost on process death; even a completed +// unsynced replacement may be lost on host/power failure. Neither submitted nor +// written is a durable vote acknowledgement. Recovery must use the startup +// reservation unless the separate clean-history seal validates. +type voteHistoryWriter struct { + mu sync.Mutex + pending *alpenglow.VoteHistorySnapshot + closing bool + failure error + submitted uint64 + written uint64 + coalesced uint64 + wake chan struct{} + done chan struct{} + persist func(*alpenglow.VoteHistorySnapshot) error + onError func(error) +} + +func newVoteHistoryWriter(persist func(*alpenglow.VoteHistorySnapshot) error, onError func(error)) *voteHistoryWriter { + w := &voteHistoryWriter{wake: make(chan struct{}, 1), done: make(chan struct{}), persist: persist, onError: onError} + go w.run() + return w +} + +// submit does no I/O and never waits for the writer. The mutex only protects +// pointer/counter changes; neither persistence nor error callbacks hold it. +// A nil return means queued only. It must never replace the reservation check. +func (w *voteHistoryWriter) submit(snapshot *alpenglow.VoteHistorySnapshot) error { + if snapshot == nil { + return errors.New("nil vote-history snapshot") + } + w.mu.Lock() + if w.failure != nil { + err := w.failure + w.mu.Unlock() + return err + } + if w.closing { + w.mu.Unlock() + return errors.New("vote-history writer is closed") + } + if w.pending != nil { + w.coalesced++ + } + w.pending = snapshot + w.submitted++ + w.mu.Unlock() + w.notify() + return nil +} + +func (w *voteHistoryWriter) notify() { + select { + case w.wake <- struct{}{}: + default: + } +} + +func (w *voteHistoryWriter) run() { + defer close(w.done) + for range w.wake { + for { + w.mu.Lock() + snapshot := w.pending + w.pending = nil + closing := w.closing + w.mu.Unlock() + if snapshot == nil { + if closing { + return + } + break + } + if err := w.persist(snapshot); err != nil { + err = fmt.Errorf("background vote-history write: %w", err) + w.mu.Lock() + w.failure = err + w.pending = nil + w.closing = true + w.mu.Unlock() + if w.onError != nil { + w.onError(err) + } + return + } + w.mu.Lock() + w.written++ + w.mu.Unlock() + } + } +} + +// close rejects new submissions and drains every retained snapshot. The voter +// must join this worker before writing and syncing its final clean history, +// otherwise an older in-flight rename could overwrite the sealed history. +func (w *voteHistoryWriter) close() error { + w.mu.Lock() + w.closing = true + w.mu.Unlock() + w.notify() + <-w.done + w.mu.Lock() + defer w.mu.Unlock() + return w.failure +} + +func (w *voteHistoryWriter) counters() (submitted, written, coalesced uint64) { + w.mu.Lock() + defer w.mu.Unlock() + return w.submitted, w.written, w.coalesced +} diff --git a/pkg/consensus/vote_history_writer_test.go b/pkg/consensus/vote_history_writer_test.go new file mode 100644 index 000000000..74c7bbfcd --- /dev/null +++ b/pkg/consensus/vote_history_writer_test.go @@ -0,0 +1,201 @@ +package consensus + +import ( + "errors" + "os" + "os/exec" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestAsyncHistoryBlockedWriteDoesNotDelayVotesAndCleanCloseDrains(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + entered, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + unblock := func() { once.Do(func() { close(release) }) } + t.Cleanup(unblock) + first := true // Owned only by the serial writer. + v.historyWriter = newVoteHistoryWriter(func(s *alpenglow.VoteHistorySnapshot) error { + if first { + first = false + close(entered) + <-release + } + return alpenglow.SaveReservedVoteHistorySnapshot(cfg.HistoryDir, s) + }, nil) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + <-entered + castDone := make(chan error, 1) + go func() { + for _, slot := range []uint64{45, 46} { + ok, err := v.cast(alpenglow.NewSkipVote(slot), false) + if err != nil || !ok { + castDone <- errors.New("vote failed while history writer was blocked") + return + } + } + castDone <- nil + }() + select { + case err := <-castDone: + require.NoError(t, err) + case <-time.After(time.Second): + t.Fatal("disk writer blocked voting") + } + submitted, written, coalesced := v.historyWriter.counters() + require.Equal(t, uint64(3), submitted) + require.Zero(t, written) + require.Equal(t, uint64(1), coalesced) + onDisk, err := alpenglow.LoadVoteHistory(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.False(t, onDisk.HasSkipped(44), "I/O should still be blocked") + // Even when disk history lags, complete in-memory decisions forbid conflict. + voted, err = v.cast(alpenglow.NewNotarizationVote(44, solana.Hash{7}), false) + require.NoError(t, err) + require.False(t, voted) + closed := make(chan error, 1) + go func() { closed <- v.close() }() + require.Eventually(t, func() bool { + v.historyWriter.mu.Lock() + defer v.historyWriter.mu.Unlock() + return v.historyWriter.closing + }, time.Second, time.Millisecond) + select { + case err := <-closed: + t.Fatalf("close returned before draining its writer: %v", err) + default: + } + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) + unblock() + require.NoError(t, <-closed) + onDisk, err = alpenglow.LoadVoteHistory(cfg.HistoryDir, v.node) + require.NoError(t, err) + for _, slot := range []uint64{44, 45, 46} { + require.True(t, onDisk.HasSkipped(slot)) + } + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + digest, err := alpenglow.VoteHistoryDigest(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Equal(t, digest, r.CleanHistoryDigest) + _, written, _ = v.historyWriter.counters() + require.Equal(t, uint64(2), written, "old in-flight snapshot must finish before newest complete snapshot") +} + +func TestAsyncHistoryFailureIsStickyAndReportedWithoutMoreSubmissions(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + h := alpenglow.NewVoteHistory(voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount).Validators[0].NodePubkey, 39) + h.ReservationRequired = true + snapshot, err := alpenglow.PrepareReservedVoteHistory(h, cfg.Identity) + require.NoError(t, err) + diskErr := errors.New("injected disk failure") + reported := make(chan error, 1) + w := newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { return diskErr }, func(err error) { reported <- err }) + require.NoError(t, w.submit(snapshot)) + select { + case err := <-reported: + require.ErrorIs(t, err, diskErr) + case <-time.After(time.Second): + t.Fatal("background failure was not reported") + } + require.ErrorIs(t, w.submit(snapshot), diskErr) + require.ErrorIs(t, w.close(), diskErr) + require.ErrorIs(t, w.close(), diskErr) +} + +func TestAsyncHistoryFailureStopsVoterAndPreventsCleanMarker(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { _ = e.Close() }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{39}} + e.SetAlpenglowRoot(root) + v, err := newAlpenglowVoterUnstarted(e, cfg, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + t.Cleanup(func() { _ = v.close() }) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + diskErr := errors.New("injected background disk failure") + v.historyWriter = newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { return diskErr }, v.failHistoryWrite) + require.NoError(t, v.history.AddVote(alpenglow.NewSkipVote(44))) + require.NoError(t, v.saveHistory()) + select { + case <-v.done: + case <-time.After(time.Second): + t.Fatal("disk failure did not stop the voter") + } + require.ErrorIs(t, e.safetyError(), diskErr) + _, _, err = v.sign(alpenglow.NewSkipVote(45), false) + require.ErrorIs(t, err, diskErr) + require.Error(t, v.enqueue(voterEvent{kind: voterEventBlockTimeout, slot: 45})) + require.ErrorIs(t, v.close(), diskErr) + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) +} + +// Kill a subprocess while its first history write is blocked and a newer +// snapshot is pending. Successful earlier reservation syncs survive this +// process crash; this is deliberately not a host power-loss test. +func TestAsyncHistoryProcessCrashLosesPendingSnapshots(t *testing.T) { + const childEnv = "MITHRIL_ASYNC_HISTORY_TEST_DIR" + if dir := os.Getenv(childEnv); dir != "" { + cfg := reservedTestConfig(dir) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + require.NoError(t, v.historyWriter.close()) + entered := make(chan struct{}) + v.historyWriter = newVoteHistoryWriter(func(*alpenglow.VoteHistorySnapshot) error { + close(entered) + select {} + }, nil) + for _, slot := range []uint64{44, 45} { + voted, err := v.cast(alpenglow.NewSkipVote(slot), false) + require.NoError(t, err) + require.True(t, voted) + if slot == 44 { + <-entered + } + } + require.NoError(t, os.WriteFile(filepath.Join(dir, "ready"), []byte("ready"), 0600)) + select {} + } + dir := t.TempDir() + cmd := exec.Command(os.Args[0], "-test.run=^TestAsyncHistoryProcessCrashLosesPendingSnapshots$", "-test.count=1") + cmd.Env = append(os.Environ(), childEnv+"="+dir) + require.NoError(t, cmd.Start()) + t.Cleanup(func() { _ = cmd.Process.Kill() }) + require.Eventually(t, func() bool { _, err := os.Stat(filepath.Join(dir, "ready")); return err == nil }, 10*time.Second, time.Millisecond) + require.NoError(t, cmd.Process.Kill()) + require.Error(t, cmd.Wait()) + cfg := reservedTestConfig(dir) + cfg.InitializeVoteReservation = false + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.False(t, v.history.HasSkipped(44)) + require.False(t, v.history.HasSkipped(45)) + h := v.reservation.recoverThrough + require.GreaterOrEqual(t, h, uint64(45)) + _, _, err = v.sign(alpenglow.NewSkipVote(44), false) + require.ErrorIs(t, err, errVoterNotReady) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), false) + require.ErrorIs(t, err, errVoterNotReady, "verified finality must reach the lost history's bound") +} diff --git a/pkg/consensus/vote_reservation.go b/pkg/consensus/vote_reservation.go new file mode 100644 index 000000000..564cfc29d --- /dev/null +++ b/pkg/consensus/vote_reservation.go @@ -0,0 +1,296 @@ +package consensus + +import ( + "bytes" + "crypto/ed25519" + "errors" + "fmt" + "math" + "os" + "sync" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/gagliardetto/solana-go" +) + +const signingReserveSlots = uint64(32) +const signingRenewRemaining = uint64(16) + +// signingReservation bounds what a crash may erase from detailed vote history. +// Intended guarantee: losing recent history must not authorize conflicting +// voting/leader actions after restart. It does NOT guarantee that every vote +// survives on disk, immediate restart voting, or recovery from safety-file rollback. +// +// On an enrolled restart, H is the startup reservation. Without a clean-history +// seal, all vote types (including restored votes) are forbidden at slots <= H; +// slots > H also wait until verified finality/checkpoint state reaches H. +// Leaders always obey that startup barrier, even after a clean vote-history seal. +// The current acknowledged Through separately caps every new signing permission. +// Renewal can raise Through, but never moves this run's fixed recovery barrier. +// +// The worker owns record after startup. Only a successful file+directory sync +// publishes Through to signers; a request, queued write or uncertain sync cannot. +// This assumes storage honors sync and a single fenced identity owner preserves +// the current reservation independently of AccountsDB. See docs/reserved-vote-history.md. +type signingReservation struct { + uncertain bool // Worker only, read after halt. Failed sync may have reached storage. + record alpenglow.VoteReservation + through atomic.Uint64 + desired atomic.Uint64 + stopped atomic.Bool + recoverThrough uint64 // Startup H, or zero after first enrollment / a validated clean-history seal. + leaderThrough uint64 // Exact leader production history is not saved: always skip the old range. + wake chan struct{} + changed chan struct{} + stop chan struct{} + done chan struct{} + stopOnce sync.Once + persist func(alpenglow.VoteReservation) error +} + +func openSigningReservation(cfg VotingConfig, node solana.PublicKey, shredVersion uint16, history *alpenglow.VoteHistory) (*signingReservation, error) { + if cfg.Genesis == (solana.Hash{}) { + return nil, errors.New("reserved voting requires the bound genesis hash") + } + expected := alpenglow.VoteReservation{Version: 1, Node: node, VoteAccount: cfg.VoteAccount, AuthorizedVoter: solana.PublicKey(cfg.AuthorizedVoter.Public().(ed25519.PublicKey)), Genesis: cfg.Genesis, ShredVersion: shredVersion} + record, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, node) + initializing := errors.Is(err, os.ErrNotExist) + if initializing { + if !cfg.InitializeVoteReservation || history.ReservationRequired { + return nil, errors.New("missing vote reservation; explicit first enrollment with complete synchronous history is required") + } + record = expected + record.Generation = 1 + record.Through = history.Root + for slot := range history.VotesCast { + record.Through = max(record.Through, slot) + } + // The baseline is made durable before the first reservation is created. + if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { + return nil, err + } + } else if err != nil { + return nil, fmt.Errorf("refuse unsafe reservation reset: %w", err) + } + if record.Node != expected.Node || record.VoteAccount != expected.VoteAccount || record.AuthorizedVoter != expected.AuthorizedVoter || record.Genesis != expected.Genesis || record.ShredVersion != expected.ShredVersion { + return nil, errors.New("vote reservation cluster or signing identity mismatch; explicit domain migration is required") + } + if record.Through == math.MaxUint64 || record.Generation == math.MaxUint64 { + return nil, errors.New("vote reservation exhausted") + } + for slot := range history.VotesCast { + if slot > record.Through { + return nil, fmt.Errorf("history slot %d exceeds durable reservation %d", slot, record.Through) + } + } + r := &signingReservation{record: record, recoverThrough: record.Through, leaderThrough: record.Through, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.persist = func(next alpenglow.VoteReservation) error { + return alpenglow.SaveVoteReservation(cfg.HistoryDir, next, cfg.Identity) + } + if initializing { + r.recoverThrough = 0 + } else if len(record.CleanHistoryDigest) != 0 { + digest, err := alpenglow.VoteHistoryDigest(cfg.HistoryDir, node) + if err != nil { + return nil, err + } + if bytes.Equal(digest, record.CleanHistoryDigest) { + r.recoverThrough = 0 + } + } + // A matching digest proves exact history only for the sealed session. Consume + // that exception with a durably acknowledged dirty successor before allowing + // new vote/leader signing or new detailed-history decisions. A crash after + // this write must use H, + // even if the detailed history file still looks valid or matches the old seal. + r.record.CleanHistoryDigest = nil + r.record.Generation++ + if err := r.persist(r.record); err != nil { + return nil, fmt.Errorf("consume vote reservation session: %w", err) + } + history.ReservationRequired = true + // Version 2 is deliberately rejected by older binaries that do not enforce H. + if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { + return nil, err + } + r.through.Store(r.record.Through) + mlog.Log.Infof("ALPENGLOW signing reservation: through=%d recovery_through=%d leader_recovery_through=%d", r.record.Through, r.recoverThrough, r.leaderThrough) + go r.run() + return r, nil +} + +// allow is nonblocking and protects vote signing/restoration and leader slots. +// For a nonzero startup barrier H, require slot > H AND finalized >= H. Merely +// observing a new block, waiting elapsed time, or replaying past H is not proof +// that prior decisions can be forgotten. Finality comes from verified consensus/ +// checkpoint state, never an RPC tip; --wait-to-vote-slot cannot override it. +// Passing this gate is necessary, not sufficient: normal protocol checks apply. +func (r *signingReservation) allow(slot, finalized uint64, leader bool) bool { + if r.stopped.Load() { + return false + } + floor := r.recoverThrough + if leader { + floor = r.leaderThrough + } + if floor != 0 && (slot <= floor || finalized < floor) { + return false + } + r.request(slot) + return slot <= r.through.Load() +} + +func (r *signingReservation) request(slot uint64) { + for { + old := r.desired.Load() + if slot <= old || r.desired.CompareAndSwap(old, slot) { + break + } + } + through := r.through.Load() + if slot > through || through-slot <= signingRenewRemaining { + select { + case r.wake <- struct{}{}: + default: + } + } +} + +func (r *signingReservation) run() { + defer close(r.done) + retry := time.NewTicker(250 * time.Millisecond) + defer retry.Stop() + var warned bool + for { + select { + case <-r.stop: + return + case <-r.wake: + case <-retry.C: + } + if r.stopped.Load() { + return + } + slot := r.desired.Load() + if slot == 0 || (slot <= r.record.Through && r.record.Through-slot > signingRenewRemaining) { + continue + } + if slot > math.MaxUint64-signingReserveSlots || r.record.Generation == math.MaxUint64 { + continue + } // Never wrap or grant permission. + next := r.record + next.Through = max(next.Through, slot+signingReserveSlots) + next.Generation++ + next.CleanHistoryDigest = nil + if err := r.persist(next); err != nil { + r.uncertain = true + if !warned { + mlog.Log.Errorf("ALPENGLOW signing reservation renewal failed; permission remains through %d: %v", r.through.Load(), err) + warned = true + } + continue // Retry the same or a greater bound; an uncertain sync never grants permission. + } + warned = false + r.uncertain = false + r.record = next + r.through.Store(next.Through) + select { + case r.changed <- struct{}{}: + default: + } + } +} + +func (r *signingReservation) halt() { + r.stopOnce.Do(func() { r.stopped.Store(true); close(r.stop) }) + <-r.done +} + +// seal may be called only after the voter loop and leader producer have stopped, +// and after the ordered history writer has been drained/joined. The caller must +// also establish verified finality >= recoverThrough and no latched safety fault. +// Sync exact history first, then sync its digest in the reservation. A normal +// process exit or successful unsynced rename alone is not a clean seal. On an +// error the caller must not assume cleanliness; restart validates whichever +// durable record survived. The seal never relaxes the next run's leader barrier. +func (r *signingReservation) seal(dir string, history *alpenglow.VoteHistory, identity ed25519.PrivateKey) error { + r.halt() + if r.uncertain { + return errors.New("uncertain reservation write; retaining unclean recovery") + } + if err := alpenglow.SaveVoteHistory(dir, history, identity); err != nil { + return err + } + digest, err := alpenglow.VoteHistoryDigest(dir, history.NodePubkey) + if err != nil { + return err + } + if r.record.Generation == math.MaxUint64 { + return errors.New("vote reservation generation exhausted") + } + next := r.record + next.Generation++ + next.CleanHistoryDigest = digest + return r.persist(next) +} + +// Retain only events blocked on renewal, not historical catch-up traffic. +// Replay the original event after acknowledgement so normal finality, parent, +// execution and invalidation checks still decide whether to vote. +func (v *alpenglowVoter) retainReservationEvent(event voterEvent) { + r := v.reservation + if r == nil { + return + } + var slot uint64 + switch event.kind { + case voterEventBlock: + slot = event.block.Block.Slot + case voterEventBlockTimeout, voterEventCrashedLeaderTimeout: + slot = event.slot | (alpenglow.LeaderWindowSlots - 1) + case voterEventConsensus: + switch event.consensus.Kind { + case alpenglow.ConsensusEventBlockNotarized, alpenglow.ConsensusEventParentReady, alpenglow.ConsensusEventSafeToNotar, alpenglow.ConsensusEventSafeToSkip: + slot = event.consensus.Slot + if event.consensus.Kind == alpenglow.ConsensusEventSafeToNotar || event.consensus.Kind == alpenglow.ConsensusEventSafeToSkip { + slot |= alpenglow.LeaderWindowSlots - 1 + } + default: + return + } + default: + return + } + if slot <= r.through.Load() || slot <= v.admissionFloor() || slot < v.waitToVoteSlot || v.engine.alpenglowVerifiedFinalityFloor() < r.recoverThrough || slot <= r.recoverThrough { + return + } + if !v.votingStarted && v.readyToVote != nil && !v.readyToVote(slot) { + return + } + r.request(slot) + if len(v.reservationEvents) < votorEventQueueSize { + v.reservationEvents = append(v.reservationEvents, event) + } +} + +// AlpenglowCanSignLeaderSlot protects every produced slot, including the +// trailing slots of a leader window. A clean vote-history marker is not a +// complete leader-block history, so leaders always skip the old reservation. +func (e *AlpenglowObserverEngine) AlpenglowCanSignLeaderSlot(slot uint64) bool { + if e.safetyError() != nil { + return false + } + e.voterMu.RLock() + defer e.voterMu.RUnlock() + v := e.voter + if v == nil { + return false + } + if v.reservation == nil { + return true + } + return v.reservation.allow(slot, e.alpenglowVerifiedFinalityFloor(), true) +} diff --git a/pkg/consensus/vote_reservation_test.go b/pkg/consensus/vote_reservation_test.go new file mode 100644 index 000000000..f4e5c418d --- /dev/null +++ b/pkg/consensus/vote_reservation_test.go @@ -0,0 +1,330 @@ +package consensus + +import ( + "crypto/ed25519" + "errors" + "math" + "os" + "sync/atomic" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func reservedTestConfig(dir string) VotingConfig { + return VotingConfig{Identity: voterTestKey(11), AuthorizedVoter: voterTestKey(12), VoteAccount: solana.PublicKey(voterTestKey(13).Public().(ed25519.PublicKey)), HistoryDir: dir, Genesis: solana.Hash{1}, ReservedHistory: true, InitializeVoteReservation: true, WaitToVoteSlot: 40, ReadyToVote: func(uint64) bool { return true }, EpochForSlot: func(uint64) uint64 { return 7 }, Peers: func([]alpenglow.ValidatorStake) []alpenglow.VotorPeer { return nil }} +} + +func openReservedTestVoter(t *testing.T, cfg VotingConfig, root uint64) (*alpenglowVoter, error) { + t.Helper() + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, e.Close()) }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + block := alpenglow.BlockID{Slot: root, Hash: solana.Hash{byte(root)}} + e.SetAlpenglowRoot(block) + v, err := newAlpenglowVoterUnstarted(e, cfg, block, []alpenglow.ValidatorSet{set}) + if err == nil { + t.Cleanup(func() { require.NoError(t, v.close()) }) + } + return v, err +} + +func reserveThrough(t *testing.T, r *signingReservation, slot uint64) { + t.Helper() + r.request(slot) + require.Eventually(t, func() bool { return r.through.Load() >= slot }, time.Second, time.Millisecond) +} + +// Simulate loss of this process without executing the clean shutdown protocol. +func crashReservedTestVoter(t *testing.T, v *alpenglowVoter) { + t.Helper() + v.shutdownOnce.Do(func() { + v.closeOnce.Do(func() { close(v.done) }) + v.wg.Wait() + v.reservation.halt() + if v.historyWriter != nil { + require.NoError(t, v.historyWriter.close()) + } + require.NoError(t, v.broadcaster.Close()) + require.NoError(t, v.historyLock.Close()) + }) +} + +func TestReservedVotingLostHistorySuffixAndRepeatedCrash(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 60) + baseline, err := os.ReadFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node)) + require.NoError(t, err) + voted, err := v.cast(alpenglow.NewNotarizationVote(60, solana.Hash{1}), false) + require.NoError(t, err) + require.True(t, voted) + oldH := v.reservation.through.Load() + crashReservedTestVoter(t, v) + // Reproduce a host crash retaining a valid older version of detailed history. + require.NoError(t, os.WriteFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node), baseline, 0600)) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, oldH, resumed.reservation.recoverThrough) + for _, vote := range []alpenglow.Vote{alpenglow.NewNotarizationVote(60, solana.Hash{2}), alpenglow.NewSkipVote(60), alpenglow.NewFinalizationVote(60), alpenglow.NewNotarizationFallbackVote(60, solana.Hash{2}), alpenglow.NewSkipFallbackVote(60), alpenglow.NewSkipVote(oldH + 1)} { + // Check at the signing boundary, including the restoration bypass. + for _, normal := range []bool{false, true} { + _, _, err := resumed.sign(vote, normal) + require.ErrorIs(t, err, errVoterNotReady) + } + } + require.Equal(t, oldH, resumed.reservation.through.Load(), "recovery must not keep moving its target") + crashReservedTestVoter(t, resumed) + resumed, err = openReservedTestVoter(t, cfg, oldH) + require.NoError(t, err) + require.Equal(t, oldH, resumed.reservation.recoverThrough) + reserveThrough(t, resumed.reservation, oldH+1) + voted, err = resumed.cast(alpenglow.NewSkipVote(oldH+1), false) + require.NoError(t, err) + require.True(t, voted) + voted, err = resumed.cast(alpenglow.NewSkipVote(oldH), false) + require.NoError(t, err) + require.False(t, voted) +} + +func TestReservedVotingCleanMarkerConsumedBeforeSigning(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + require.NoError(t, v.close()) + r, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.NotEmpty(t, r.CleanHistoryDigest) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Zero(t, resumed.reservation.recoverThrough) + require.True(t, resumed.history.HasSkipped(44)) + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) + voted, err = resumed.cast(alpenglow.NewSkipVote(45), false) + require.NoError(t, err) + require.True(t, voted) + require.False(t, resumed.reservation.allow(45, 39, true), "clean vote history does not authorize repeating leader blocks") + crashReservedTestVoter(t, resumed) + again, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, r.Through, again.reservation.recoverThrough) + // A shutdown before recovering the uncertain range must not mark it clean. + require.NoError(t, again.close()) + r, err = alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Empty(t, r.CleanHistoryDigest) +} + +func TestReservedVotingRejectsMissingCorruptOrWrongDomain(t *testing.T) { + for _, which := range []string{"missing_history", "missing_bound", "corrupt_bound", "genesis", "authorized", "vote_account", "synchronous"} { + t.Run(which, func(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + switch which { + case "missing_history": + require.NoError(t, os.Remove(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node))) + case "missing_bound": + require.NoError(t, os.Remove(alpenglow.VoteReservationFilename(cfg.HistoryDir, v.node))) + cfg.InitializeVoteReservation = true + case "corrupt_bound": + require.NoError(t, os.WriteFile(alpenglow.VoteReservationFilename(cfg.HistoryDir, v.node), []byte("{"), 0600)) + case "genesis": + cfg.Genesis = solana.Hash{2} + case "authorized": + cfg.AuthorizedVoter = voterTestKey(19) + case "vote_account": + cfg.VoteAccount = solana.PublicKey{9} + case "synchronous": + cfg.ReservedHistory = false + } + _, err = openReservedTestVoter(t, cfg, 39) + require.Error(t, err) + }) + } +} + +func TestReservedVotingRequiresEnrollmentAndExclusiveOwner(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + cfg.InitializeVoteReservation = false + _, err := openReservedTestVoter(t, cfg, 39) + require.Error(t, err) + cfg.InitializeVoteReservation = true + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + _, err = openReservedTestVoter(t, cfg, 39) + require.ErrorContains(t, err, "already owned") + require.NoError(t, v.close()) +} + +func TestSigningReservationUnacknowledgedSyncCannotAuthorize(t *testing.T) { + entered, release := make(chan struct{}), make(chan struct{}) + r := &signingReservation{record: alpenglow.VoteReservation{Through: 64, Generation: 1}, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.through.Store(64) + var calls atomic.Uint64 + r.persist = func(next alpenglow.VoteReservation) error { + calls.Add(1) + close(entered) + <-release + return nil + } + go r.run() + require.True(t, r.allow(64, 64, false)) + <-entered + require.Equal(t, uint64(64), r.through.Load()) + require.False(t, r.allow(65, 64, false)) + close(release) + require.Eventually(t, func() bool { return r.through.Load() > 64 }, time.Second, time.Millisecond) + require.True(t, r.allow(65, 64, false)) + r.halt() + require.Equal(t, uint64(1), calls.Load()) +} + +func TestSigningReservationUncertainWriteSurvivesRestart(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + // Stop the original worker, then exercise a fresh worker against the real record. + v.reservation.halt() + record, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + r := &signingReservation{record: record, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{})} + r.through.Store(record.Through) + wrote := make(chan struct{}, 1) + r.persist = func(next alpenglow.VoteReservation) error { + err := alpenglow.SaveVoteReservation(cfg.HistoryDir, next, cfg.Identity) + select { + case wrote <- struct{}{}: + default: + } + if err != nil { + return err + } + return errors.New("injected lost sync acknowledgement") + } + v.reservation = r + go r.run() + require.False(t, r.allow(60, 39, false)) + <-wrote + r.halt() + require.Equal(t, record.Through, r.through.Load()) + durable, err := alpenglow.LoadVoteReservation(cfg.HistoryDir, v.node) + require.NoError(t, err) + require.Greater(t, durable.Through, record.Through) + require.ErrorContains(t, r.seal(cfg.HistoryDir, v.history, cfg.Identity), "uncertain") + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.Equal(t, durable.Through, resumed.reservation.recoverThrough) +} + +func TestSigningReservationNeverWraps(t *testing.T) { + r := &signingReservation{record: alpenglow.VoteReservation{Through: 64, Generation: 1}, wake: make(chan struct{}, 1), changed: make(chan struct{}, 1), stop: make(chan struct{}), done: make(chan struct{}), persist: func(alpenglow.VoteReservation) error { t.Error("overflow attempted persistence"); return nil }} + r.through.Store(64) + go r.run() + require.False(t, r.allow(math.MaxUint64, 64, false)) + r.halt() + require.Equal(t, uint64(64), r.through.Load()) +} + +func TestReservationRetryRechecksFinality(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + event := voterEvent{kind: voterEventBlockTimeout, slot: 44} + require.NoError(t, v.handle(event)) + require.NotEmpty(t, v.reservationEvents) + reserveThrough(t, v.reservation, 44) + v.engine.SetAlpenglowRoot(alpenglow.BlockID{Slot: 47, Hash: solana.Hash{47}}) + pending := v.reservationEvents + v.reservationEvents = nil + for _, e := range pending { + require.NoError(t, v.handle(e)) + } + require.False(t, v.history.HasSkipped(44), "finalized work must not be signed after a delayed ack") +} + +func TestReservedVotingEverySignatureTypeAtBound(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + h := v.reservation.through.Load() + // Freeze acknowledgement while leaving the signing guard active. + v.reservation.halt() + v.reservation.stopped.Store(false) + defer v.reservation.stopped.Store(true) + for _, slot := range []uint64{h, h + 1} { + votes := []alpenglow.Vote{alpenglow.NewNotarizationVote(slot, solana.Hash{2}), alpenglow.NewSkipVote(slot), alpenglow.NewFinalizationVote(slot), alpenglow.NewNotarizationFallbackVote(slot, solana.Hash{2}), alpenglow.NewSkipFallbackVote(slot)} + for _, vote := range votes { + for _, normal := range []bool{false, true} { + _, _, err := v.sign(vote, normal) + if slot == h { + require.NoError(t, err) + } else { + require.ErrorIs(t, err, errVoterNotReady) + } + } + } + } +} + +func TestReservedCleanDigestMismatchUsesCrashRecovery(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + baseline, err := os.ReadFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node)) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + voted, err := v.cast(alpenglow.NewSkipVote(44), false) + require.NoError(t, err) + require.True(t, voted) + require.NoError(t, v.close()) + require.NoError(t, os.WriteFile(alpenglow.VoteHistoryFilename(cfg.HistoryDir, v.node), baseline, 0600)) + resumed, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + require.NotZero(t, resumed.reservation.recoverThrough) +} + +func TestReservationLoopRetriesAfterAcknowledgement(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + v.start() + require.NoError(t, v.enqueue(voterEvent{kind: voterEventBlockTimeout, slot: 44})) + // Read via stats, not mutable voter history, while its loop is running. + require.Eventually(t, func() bool { v.landingMu.RLock(); defer v.landingMu.RUnlock(); return v.stats.VotesCastThisRun == 4 }, time.Second, time.Millisecond) + require.NoError(t, v.close()) + for slot := uint64(44); slot <= 47; slot++ { + require.True(t, v.history.HasSkipped(slot)) + } +} + +func TestReservationRetainsWindowCrossingBound(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 44) + h := v.reservation.through.Load() // 76: window ends at 79. + v.retainReservationEvent(voterEvent{kind: voterEventBlockTimeout, slot: h}) + require.Len(t, v.reservationEvents, 1, "trailing skip slots require retry even if the first slot fits") +} diff --git a/pkg/consensus/voter.go b/pkg/consensus/voter.go index b77a7689b..4a9cbbc15 100644 --- a/pkg/consensus/voter.go +++ b/pkg/consensus/voter.go @@ -5,6 +5,7 @@ import ( "errors" "fmt" "math" + "os" "sort" "sync" "time" @@ -38,43 +39,57 @@ type VotingPeerSource func(validators []alpenglow.ValidatorStake) []alpenglow.Vo // account address; AuthorizedVoter is the Ed25519 signer from which its BLS key // was registered. type VotingConfig struct { - Identity ed25519.PrivateKey - AuthorizedVoter ed25519.PrivateKey - VoteAccount solana.PublicKey - HistoryDir string - EpochForSlot func(slot uint64) uint64 - Peers VotingPeerSource - SlotDuration time.Duration - WaitToVoteSlot uint64 - ReadyToVote func(slot uint64) bool + Identity ed25519.PrivateKey + AuthorizedVoter ed25519.PrivateKey + VoteAccount solana.PublicKey + HistoryDir string + ReservedHistory bool + InitializeVoteReservation bool + Genesis solana.Hash + EpochForSlot func(slot uint64) uint64 + Peers VotingPeerSource + SlotDuration time.Duration + WaitToVoteSlot uint64 // Inclusive minimum for new votes; authenticated history restoration is separate + ReadyToVote func(slot uint64) bool } // VotingStats exposes positive network evidence separately from local casting. // NetworkLandedVotes counts unique persisted votes whose rank appeared in the // exact BLS-verified certificate proof received over Votor QUIC. type VotingStats struct { - Enabled bool `json:"enabled"` - VotesCastThisRun uint64 `json:"votes_cast_this_run"` - NetworkLandedVotes uint64 `json:"network_landed_votes"` - LastNetworkLandedSlot uint64 `json:"last_network_landed_slot,omitempty"` - LastNetworkLandedVoteType alpenglow.VoteType `json:"last_network_landed_vote_type,omitempty"` - LastNetworkCertificateType alpenglow.CertificateType `json:"last_network_certificate_type,omitempty"` - LastNetworkLandedAt time.Time `json:"last_network_landed_at,omitempty"` - BroadcastMessagesQueued uint64 `json:"broadcast_messages_queued"` - BroadcastMessagesDropped uint64 `json:"broadcast_messages_dropped"` - BroadcastPeerSends uint64 `json:"broadcast_peer_sends"` - BroadcastPeerSendsSkipped uint64 `json:"broadcast_peer_sends_skipped"` - BroadcastPeerSendErrors uint64 `json:"broadcast_peer_send_errors"` - BroadcastDesiredPeers int `json:"broadcast_desired_peers"` - BroadcastActiveConnections int `json:"broadcast_active_connections"` - BroadcastPendingConnections int `json:"broadcast_pending_connections"` - BroadcastConnectionAttempts uint64 `json:"broadcast_connection_attempts"` - BroadcastConnectionErrors uint64 `json:"broadcast_connection_errors"` - BroadcastConnectionJobsDropped uint64 `json:"broadcast_connection_jobs_dropped"` - BroadcastLastPeerSendError string `json:"broadcast_last_peer_send_error,omitempty"` - BroadcastLastPeerSendErrorAt time.Time `json:"broadcast_last_peer_send_error_at,omitempty"` - BroadcastLastConnectionError string `json:"broadcast_last_connection_error,omitempty"` - BroadcastLastConnectionErrorAt time.Time `json:"broadcast_last_connection_error_at,omitempty"` + HistorySnapshotsSubmitted uint64 `json:"history_snapshots_submitted,omitempty"` + HistorySnapshotsWritten uint64 `json:"history_snapshots_written,omitempty"` + HistorySnapshotsCoalesced uint64 `json:"history_snapshots_coalesced,omitempty"` + ReservedHistory bool `json:"reserved_history"` + SigningReservedThrough uint64 `json:"signing_reserved_through,omitempty"` + RecoveryThrough uint64 `json:"recovery_through,omitempty"` + Enabled bool `json:"enabled"` + VotesCastThisRun uint64 `json:"votes_cast_this_run"` + NetworkLandedVotes uint64 `json:"network_landed_votes"` + LastNetworkLandedSlot uint64 `json:"last_network_landed_slot,omitempty"` + LastNetworkLandedVoteType alpenglow.VoteType `json:"last_network_landed_vote_type,omitempty"` + LastNetworkCertificateType alpenglow.CertificateType `json:"last_network_certificate_type,omitempty"` + LastNetworkLandedAt time.Time `json:"last_network_landed_at,omitempty"` + BroadcastMessagesQueued uint64 `json:"broadcast_messages_queued"` + BroadcastMessagesDropped uint64 `json:"broadcast_messages_dropped"` + BroadcastPeerSends uint64 `json:"broadcast_peer_sends"` + BroadcastPeerSendsSkipped uint64 `json:"broadcast_peer_sends_skipped"` + BroadcastPeerSendErrors uint64 `json:"broadcast_peer_send_errors"` + BroadcastPeerQueueDrops uint64 `json:"broadcast_peer_queue_drops"` + BroadcastPeerQueueDiscarded uint64 `json:"broadcast_peer_queue_discarded"` + BroadcastPeerSendTimeouts uint64 `json:"broadcast_peer_send_timeouts"` + BroadcastPeerQueueMaxDelay time.Duration `json:"broadcast_peer_queue_max_delay_ns"` + BroadcastPeerQueues []alpenglow.VotorPeerQueueStats `json:"broadcast_peer_queues,omitempty"` + BroadcastDesiredPeers int `json:"broadcast_desired_peers"` + BroadcastActiveConnections int `json:"broadcast_active_connections"` + BroadcastPendingConnections int `json:"broadcast_pending_connections"` + BroadcastConnectionAttempts uint64 `json:"broadcast_connection_attempts"` + BroadcastConnectionErrors uint64 `json:"broadcast_connection_errors"` + BroadcastConnectionJobsDropped uint64 `json:"broadcast_connection_jobs_dropped"` + BroadcastLastPeerSendError string `json:"broadcast_last_peer_send_error,omitempty"` + BroadcastLastPeerSendErrorAt time.Time `json:"broadcast_last_peer_send_error_at,omitempty"` + BroadcastLastConnectionError string `json:"broadcast_last_connection_error,omitempty"` + BroadcastLastConnectionErrorAt time.Time `json:"broadcast_last_connection_error_at,omitempty"` } type voterEventKind uint8 @@ -88,6 +103,7 @@ const ( voterEventValidatorSet voterEventRoot voterEventNetworkCertificate + voterEventDurableRoot ) type voterEvent struct { @@ -109,43 +125,49 @@ type pendingVotorBlock struct { // history decisions run on loop; validator-set snapshots are protected only so // the outbound peer callback can read them from broadcast workers. type alpenglowVoter struct { - engine *AlpenglowObserverEngine - identity ed25519.PrivateKey - node solana.PublicKey - voteAccount solana.PublicKey - signer *alpenglow.BLSSigner - historyDir string - history *alpenglow.VoteHistory - epochForSlot func(uint64) uint64 - peerSource VotingPeerSource - slotDuration time.Duration - waitToVoteSlot uint64 - readyToVote func(slot uint64) bool - broadcaster *alpenglow.VotorBroadcaster - events chan voterEvent - done chan struct{} - startOnce sync.Once - closeOnce sync.Once - wg sync.WaitGroup - setsMu sync.RWMutex - sets map[uint64]alpenglow.ValidatorSet - restored map[alpenglow.VoteMessageKey]bool - pending map[uint64][]pendingVotorBlock - receivedShred map[uint64]bool - timeoutsSet map[uint64]bool - executedBlocks map[alpenglow.BlockID]bool - highestFinal uint64 - lastFinalizedAt time.Time - votingStarted bool - latestLiveSlot uint64 - standstillSlot *uint64 - refreshQueue []alpenglow.Message - refreshCursor int - lastWarn map[uint64]time.Time - landingMu sync.RWMutex - landed map[alpenglow.VoteMessageKey]struct{} - stats VotingStats - lastStatsLog time.Time + engine *AlpenglowObserverEngine + identity ed25519.PrivateKey + node solana.PublicKey + voteAccount solana.PublicKey + signer *alpenglow.BLSSigner + historyDir string + historyLock *os.File + reservation *signingReservation + historyWriter *voteHistoryWriter + reservationEvents []voterEvent + shutdownOnce sync.Once + shutdownErr error + history *alpenglow.VoteHistory + epochForSlot func(uint64) uint64 + peerSource VotingPeerSource + slotDuration time.Duration + waitToVoteSlot uint64 + readyToVote func(slot uint64) bool + broadcaster *alpenglow.VotorBroadcaster + events chan voterEvent + done chan struct{} + startOnce sync.Once + closeOnce sync.Once + wg sync.WaitGroup + setsMu sync.RWMutex + sets map[uint64]alpenglow.ValidatorSet + restored map[alpenglow.VoteMessageKey]bool + pending map[uint64][]pendingVotorBlock + receivedShred map[uint64]bool + timeoutsSet map[uint64]bool + executedBlocks map[alpenglow.BlockID]bool + highestFinal uint64 + lastFinalizedAt time.Time + votingStarted bool + latestLiveSlot uint64 + standstillSlot *uint64 + refreshQueue []alpenglow.Message + refreshCursor int + lastWarn map[uint64]time.Time + landingMu sync.RWMutex + landed map[alpenglow.VoteMessageKey]struct{} + stats VotingStats + lastStatsLog time.Time // beforeVoteGuard is a deterministic test seam for invalidation races. It // is nil in production. beforeVoteGuard func(alpenglow.BlockID) @@ -163,6 +185,9 @@ func newAlpenglowVoterUnstarted(engine *AlpenglowObserverEngine, cfg VotingConfi } func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfig, root alpenglow.BlockID, start bool, sets []alpenglow.ValidatorSet) (*alpenglowVoter, error) { + if cfg.InitializeVoteReservation && !cfg.ReservedHistory { + return nil, errors.New("initialize-vote-reservation requires reserved-vote-history") + } if engine == nil { return nil, fmt.Errorf("enable Alpenglow voting: nil consensus engine") } @@ -192,17 +217,52 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi if node != engineNode { return nil, fmt.Errorf("enable Alpenglow voting: identity %s does not match consensus transport identity %s", node, engineNode) } + historyLock, err := alpenglow.LockVoteHistory(cfg.HistoryDir, node) + if err != nil { + return nil, err + } + keepLock := false + defer func() { + if !keepLock { + historyLock.Close() + } + }() + if !cfg.ReservedHistory { + if _, err := os.Stat(alpenglow.VoteReservationFilename(cfg.HistoryDir, node)); !errors.Is(err, os.ErrNotExist) { + return nil, fmt.Errorf("existing or unreadable vote reservation requires reserved history mode") + } + } history, err := alpenglow.LoadVoteHistory(cfg.HistoryDir, node) if err != nil { if !errors.Is(err, alpenglow.ErrVoteHistoryNotFound) { return nil, fmt.Errorf("enable Alpenglow voting: refuse unsafe vote-history reset: %w", err) } + if cfg.ReservedHistory { + if _, err := os.Stat(alpenglow.VoteReservationFilename(cfg.HistoryDir, node)); !errors.Is(err, os.ErrNotExist) || !cfg.InitializeVoteReservation { + return nil, fmt.Errorf("missing history for reserved voter; refusing automatic reset") + } + } history = alpenglow.NewVoteHistory(node, root.Slot) if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { return nil, fmt.Errorf("initialize Alpenglow vote history: %w", err) } mlog.Log.FileOnlyf("ALPENGLOW voting: created new vote history for %s at root %d; do not reuse this vote account on another validator", node, root.Slot) } + if history.ReservationRequired && !cfg.ReservedHistory { + return nil, errors.New("reserved vote history cannot be opened in synchronous mode") + } + var reservation *signingReservation + if cfg.ReservedHistory { + reservation, err = openSigningReservation(cfg, node, engine.shredVersion, history) + if err != nil { + return nil, err + } + defer func() { + if !keepLock { + reservation.halt() + } + }() + } if history.Root < root.Slot { history.SetRoot(root.Slot) if err := alpenglow.SaveVoteHistory(cfg.HistoryDir, history, cfg.Identity); err != nil { @@ -220,6 +280,8 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi voteAccount: cfg.VoteAccount, signer: signer, historyDir: cfg.HistoryDir, + historyLock: historyLock, + reservation: reservation, history: history, epochForSlot: cfg.EpochForSlot, peerSource: cfg.Peers, @@ -261,6 +323,12 @@ func newAlpenglowVoterWithStart(engine *AlpenglowObserverEngine, cfg VotingConfi return nil, err } v.broadcaster = broadcaster + if reservation != nil { + v.historyWriter = newVoteHistoryWriter(func(snapshot *alpenglow.VoteHistorySnapshot) error { + return alpenglow.SaveReservedVoteHistorySnapshot(v.historyDir, snapshot) + }, v.failHistoryWrite) + } + keepLock = true if start { v.start() } @@ -300,12 +368,25 @@ func (v *alpenglowVoter) loop() { v.closeOnce.Do(func() { close(v.done) }) v.wg.Done() }() + var reservationChanged <-chan struct{} + if v.reservation != nil { + reservationChanged = v.reservation.changed + } ticker := time.NewTicker(time.Second) defer ticker.Stop() for { select { case <-v.done: return + case <-reservationChanged: + pending := v.reservationEvents + v.reservationEvents = nil + for _, event := range pending { + if err := v.handle(event); err != nil { + v.engine.latchSafetyError(fmt.Errorf("reservation retry: %w", err)) + return + } + } case event := <-v.events: if err := v.handle(event); err != nil { v.engine.latchSafetyError(fmt.Errorf("voting engine: %w", err)) @@ -324,6 +405,11 @@ func (v *alpenglowVoter) loop() { } func (v *alpenglowVoter) handle(event voterEvent) error { + v.retainReservationEvent(event) + return v.handleEvent(event) +} + +func (v *alpenglowVoter) handleEvent(event voterEvent) error { floor := v.admissionFloor() if event.kind == voterEventBlock && v.engine.ensureChain().IsObjectivelyInvalidBlock(event.block.Block) { return nil @@ -363,6 +449,9 @@ func (v *alpenglowVoter) handle(event voterEvent) error { return nil } return v.saveHistory() + case voterEventDurableRoot: + root := v.engine.applyAlpenglowDurableRoot(event.slot) + return v.handleEvent(voterEvent{kind: voterEventRoot, root: root}) case voterEventNetworkCertificate: v.recordNetworkCertificate(event.certificate) return nil @@ -477,9 +566,9 @@ func (v *alpenglowVoter) handleConsensus(event alpenglow.ConsensusEvent) error { func (v *alpenglowVoter) admissionFloor() uint64 { floor := v.history.Root - if v.highestFinal > floor { - floor = v.highestFinal - } + // highestFinal tracks network progress and standstill, not retirement of + // our own decisions. Keep ParentReady, pending replay and exact vote history + // available until the retained pool or an ordered durable root retires them. if engineFloor := v.engine.alpenglowVoteActionFloor(); engineFloor > floor { floor = engineFloor } @@ -661,7 +750,7 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded if !restoring && v.beforeVoteGuard != nil { v.beforeVoteGuard(guardedBlock) } - // Hold through signing, durable history, pool admission, and broadcast. + // Hold through signing, history recording, pool admission, and broadcast. // Objective invalidation takes the write side before changing the chain, // so a new or restored vote is wholly before it or sees the tombstone. v.engine.invalidActionMu.RLock() @@ -682,18 +771,22 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded return false, nil } if !restoring { - // Finality can advance while the BLS signature is computed. Avoid a - // durable stale record when that race is already visible here; if it - // advances later, atomic admission below classifies it benignly. + // Retention/root pruning can advance while the BLS signature is computed. + // Avoid an expired record if that race is already visible here; atomic + // admission below classifies a later pruning race benignly. if vote.Slot <= v.admissionFloor() { return false, nil } if err := v.history.AddVote(vote); err != nil { return false, fmt.Errorf("record %s vote at slot %d: %w", vote.Type, vote.Slot, err) } - // Pool admission may synchronously assemble and publish a certificate. - // Persist the anti-equivocation record first so no externally visible - // proof can survive a crash without its signed local history. + // Pool admission can publish a certificate. In reserved mode the durable + // upper bound covers loss of this unsynchronized history replacement; + // synchronous mode still persists the exact history before admission. + // sign computed BLS bytes in RAM, but nothing may expose them before + // this boundary succeeds. In reserved mode saveHistory only queues the + // snapshot: restart safety comes from the durable reservation checked + // before sign, not from assuming this snapshot reached durable storage. if err := v.saveHistory(); err != nil { return false, err } @@ -727,7 +820,16 @@ func (v *alpenglowVoter) castTarget(vote alpenglow.Vote, restoring bool, guarded return true, nil } +// sign checks reservation recovery even when restoration bypasses the live +// joining gate. Re-signing a saved vote is still signing; the presence of an +// older valid history file cannot prove that its lost suffix was conflict-free. func (v *alpenglowVoter) sign(vote alpenglow.Vote, respectVotingGate bool) (alpenglow.VoteMessage, alpenglow.VoteVerifyResult, error) { + if err := v.engine.safetyError(); err != nil { + return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, err + } + if v.reservation != nil && !v.reservation.allow(vote.Slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { + return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, fmt.Errorf("%w: waiting for verified recovery or durable signing reservation", errVoterNotReady) + } if respectVotingGate { if err := v.votingGateError(vote.Slot); err != nil { return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, err @@ -768,7 +870,7 @@ func (v *alpenglowVoter) votingGateError(slot uint64) error { return fmt.Errorf("%w: slot %d is at or below consensus action floor %d", errVoterNotReady, slot, floor) } if slot < v.waitToVoteSlot { - return fmt.Errorf("%w: waiting for startup watermark slot %d", errVoterNotReady, v.waitToVoteSlot) + return fmt.Errorf("%w: waiting for voting cutoff slot %d", errVoterNotReady, v.waitToVoteSlot) } // ReadyToVote is a startup join guard, not a perpetual clock check. Once an // accepted live block or vote joins Votor, verified ParentReady and timeout @@ -777,6 +879,9 @@ func (v *alpenglowVoter) votingGateError(slot uint64) error { if !v.votingStarted && v.readyToVote != nil && !v.readyToVote(slot) { return fmt.Errorf("%w: slot is still behind the startup live voting window", errVoterNotReady) } + if v.reservation != nil && !v.reservation.allow(slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { + return fmt.Errorf("%w: waiting for verified recovery or durable signing reservation", errVoterNotReady) + } return nil } @@ -837,6 +942,9 @@ func (v *alpenglowVoter) votorTransportValidators() []alpenglow.ValidatorStake { } func (v *alpenglowVoter) restoreVotesForEpoch(epoch uint64) error { + if v.reservation != nil && v.reservation.recoverThrough != 0 { + return nil + } floor := v.admissionFloor() for _, vote := range v.history.VotesAfter(v.history.Root - minU64(v.history.Root, 1)) { // Keep every signed history entry for anti-equivocation, but do not @@ -884,6 +992,9 @@ func (v *alpenglowVoter) restoreVotesForEpoch(epoch uint64) error { // persist-before-admission crash window without trusting process-local invalid // block state across a restart. func (v *alpenglowVoter) restoreVotesForBlock(block alpenglow.BlockID) (bool, error) { + if v.reservation != nil && v.reservation.recoverThrough != 0 { + return false, nil + } if block.Slot <= v.admissionFloor() { return false, nil } @@ -1086,13 +1197,29 @@ func (v *alpenglowVoter) isIdentityStaked(slot uint64) bool { return false } +// saveHistory is a publication boundary with different persistence semantics: +// synchronous mode acknowledges durable exact history; reserved mode acknowledges +// an immutable queued snapshot only. The latter relies on the signing reservation. func (v *alpenglowVoter) saveHistory() error { + if v.historyWriter != nil { + snapshot, err := alpenglow.PrepareReservedVoteHistory(v.history, v.identity) + if err != nil { + return err + } + return v.historyWriter.submit(snapshot) + } if err := alpenglow.SaveVoteHistory(v.historyDir, v.history, v.identity); err != nil { return fmt.Errorf("persist vote history before consensus publication: %w", err) } return nil } +func (v *alpenglowVoter) failHistoryWrite(err error) { + v.engine.latchSafetyError(err) + mlog.Log.Errorf("ALPENGLOW VOTING SAFETY: %v", err) + v.closeOnce.Do(func() { close(v.done) }) +} + func (v *alpenglowVoter) recordNetworkCertificate(cert alpenglow.Certificate) { set, ok := v.validatorSet(cert.Slot) if !ok { @@ -1201,6 +1328,14 @@ func (v *alpenglowVoter) snapshot() VotingStats { stats := v.stats v.landingMu.RUnlock() stats.Enabled = true + if v.historyWriter != nil { + stats.HistorySnapshotsSubmitted, stats.HistorySnapshotsWritten, stats.HistorySnapshotsCoalesced = v.historyWriter.counters() + } + if v.reservation != nil { + stats.ReservedHistory = true + stats.SigningReservedThrough = v.reservation.through.Load() + stats.RecoveryThrough = v.reservation.recoverThrough + } if v.broadcaster != nil { broadcast := v.broadcaster.Stats() stats.BroadcastMessagesQueued = broadcast.MessagesQueued @@ -1208,6 +1343,11 @@ func (v *alpenglowVoter) snapshot() VotingStats { stats.BroadcastPeerSends = broadcast.PeerSends stats.BroadcastPeerSendsSkipped = broadcast.PeerSendsSkipped stats.BroadcastPeerSendErrors = broadcast.PeerSendErrors + stats.BroadcastPeerQueueDrops = broadcast.PeerQueueDrops + stats.BroadcastPeerQueueDiscarded = broadcast.PeerQueueDiscarded + stats.BroadcastPeerSendTimeouts = broadcast.PeerSendTimeouts + stats.BroadcastPeerQueueMaxDelay = broadcast.PeerQueueMaxDelay + stats.BroadcastPeerQueues = broadcast.PeerQueues stats.BroadcastDesiredPeers = broadcast.DesiredPeers stats.BroadcastActiveConnections = broadcast.Connections stats.BroadcastPendingConnections = broadcast.PendingConnections @@ -1228,7 +1368,7 @@ func (v *alpenglowVoter) maybeLogStats() { } v.lastStatsLog = time.Now() stats := v.snapshot() - mlog.Log.FileOnlyf("alpenglow voting stats: votes_cast_this_run=%d network_landed=%d last_landed_slot=%d broadcast_queued=%d broadcast_dropped=%d peer_sends=%d peer_sends_skipped=%d peer_send_errors=%d desired_peers=%d active_connections=%d pending_connections=%d connection_attempts=%d connection_errors=%d connection_jobs_dropped=%d", + mlog.Log.FileOnlyf("alpenglow voting stats: votes_cast_this_run=%d network_landed=%d last_landed_slot=%d broadcast_queued=%d broadcast_dropped=%d peer_sends=%d peer_sends_skipped=%d peer_send_errors=%d peer_queue_drops=%d peer_queue_discarded=%d peer_send_timeouts=%d peer_queue_max_delay=%s desired_peers=%d active_connections=%d pending_connections=%d connection_attempts=%d connection_errors=%d connection_jobs_dropped=%d reserved_history=%t signing_through=%d recovery_through=%d history_submitted=%d history_written=%d history_coalesced=%d", stats.VotesCastThisRun, stats.NetworkLandedVotes, stats.LastNetworkLandedSlot, @@ -1237,12 +1377,18 @@ func (v *alpenglowVoter) maybeLogStats() { stats.BroadcastPeerSends, stats.BroadcastPeerSendsSkipped, stats.BroadcastPeerSendErrors, + stats.BroadcastPeerQueueDrops, + stats.BroadcastPeerQueueDiscarded, + stats.BroadcastPeerSendTimeouts, + stats.BroadcastPeerQueueMaxDelay, stats.BroadcastDesiredPeers, stats.BroadcastActiveConnections, stats.BroadcastPendingConnections, stats.BroadcastConnectionAttempts, stats.BroadcastConnectionErrors, stats.BroadcastConnectionJobsDropped, + stats.ReservedHistory, stats.SigningReservedThrough, stats.RecoveryThrough, + stats.HistorySnapshotsSubmitted, stats.HistorySnapshotsWritten, stats.HistorySnapshotsCoalesced, ) } @@ -1274,9 +1420,28 @@ func (v *alpenglowVoter) close() error { if v == nil { return nil } - v.closeOnce.Do(func() { close(v.done) }) - v.wg.Wait() - return v.broadcaster.Close() + v.shutdownOnce.Do(func() { + v.closeOnce.Do(func() { close(v.done) }) + v.wg.Wait() + if v.reservation != nil { + v.reservation.halt() + if v.historyWriter != nil { + v.shutdownErr = v.historyWriter.close() + } + // Exiting normally is insufficient: an unresolved recovery barrier, + // writer failure or safety fault must leave the session unsealed. + floor := v.engine.alpenglowVerifiedFinalityFloor() + if v.shutdownErr == nil && v.engine.safetyError() == nil && floor >= v.reservation.recoverThrough { + v.history.SetRoot(floor) + v.shutdownErr = v.reservation.seal(v.historyDir, v.history, v.identity) + if v.shutdownErr == nil { + mlog.Log.Infof("ALPENGLOW signing reservation: clean history sealed at root=%d through=%d", v.history.Root, v.reservation.through.Load()) + } + } + } + v.shutdownErr = errors.Join(v.shutdownErr, v.broadcaster.Close(), v.historyLock.Close()) + }) + return v.shutdownErr } func pendingContains(blocks []pendingVotorBlock, candidate pendingVotorBlock) bool { diff --git a/pkg/consensus/voter_finality_ordering_test.go b/pkg/consensus/voter_finality_ordering_test.go new file mode 100644 index 000000000..a53abb640 --- /dev/null +++ b/pkg/consensus/voter_finality_ordering_test.go @@ -0,0 +1,192 @@ +package consensus + +import ( + "context" + "crypto/ed25519" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Keep the actor unstarted so tests can place finality, replay and durable +// promotion in a deterministic order, using the real engine event queue. +func newOrderingTestVoter(t *testing.T, reserved bool) (*AlpenglowObserverEngine, *alpenglowVoter) { + t.Helper() + cfg := reservedTestConfig(t.TempDir()) + cfg.ReservedHistory = reserved + cfg.InitializeVoteReservation = reserved + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{39}} + e, err := NewEngine(Config{AlpenglowIdentity: cfg.Identity, AlpenglowShredVersion: 0x1234}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, e.Close()) }) + set := voterTestValidatorSet(t, cfg.Identity, cfg.AuthorizedVoter, cfg.VoteAccount) + e.SetAlpenglowEpochLookup(cfg.EpochForSlot) + require.NoError(t, e.SetAlpenglowValidatorSet(set)) + e.SetAlpenglowRoot(root) + v, err := newAlpenglowVoterUnstarted(e, cfg, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + e.voter = v + if reserved { + reserveThrough(t, v.reservation, 44) + } + // Equivalent to startup's trusted-root ParentReady seed. + require.True(t, v.history.AddParentReady(40, root)) + return e, v +} + +func drainOrderingEvents(t *testing.T, v *alpenglowVoter) { + t.Helper() + for i := 0; i < 1000; i++ { + select { + case event := <-v.events: + require.NoError(t, v.handle(event)) + default: + return + } + } + t.Fatal("voter event queue did not drain") +} + +func observeOrderingBlock(t *testing.T, e *AlpenglowObserverEngine, slot uint64) alpenglow.BlockID { + t.Helper() + id := alpenglow.BlockID{Slot: slot, Hash: solana.Hash{byte(slot)}} + require.NoError(t, e.ObserveBlock(context.Background(), BlockObservation{Source: "ordering-test", Block: &block.Block{ + Slot: slot, ParentSlot: slot - 1, + AlpenglowBlockID: [32]byte(id.Hash), HasAlpenglowBlockID: true, + AlpenglowParentBlockID: [32]byte{byte(slot - 1)}, HasAlpenglowParentBlockID: true, + }})) + return id +} + +func finalizeOrderingBlock(t *testing.T, e *AlpenglowObserverEngine, v *alpenglowVoter, id alpenglow.BlockID) { + t.Helper() + // The two peers supply the 60% slow-finality quorum without our vote; + // adding our 30% notarization later can produce a fast certificate. + for _, vote := range []alpenglow.Vote{alpenglow.NewNotarizationVote(id.Slot, id.Hash), alpenglow.NewFinalizationVote(id.Slot)} { + for rank, key := range []ed25519.PrivateKey{voterTestKey(21), voterTestKey(22)} { + peer := signedVerifiedVoterPeerVote(t, e, v.sets[7], key, uint16(rank+1), vote) + _, err := e.acceptVerifiedVoteResult(peer) + require.NoError(t, err) + } + } + require.Equal(t, id.Slot, e.ensureChain().Snapshot().LatestDirectFinalizedBlock.Slot) +} + +func TestAlpenglowVoterReplaysFourBlocksAfterNetworkFinality(t *testing.T) { + for _, reserved := range []bool{false, true} { + name := "synchronous" + if reserved { + name = "reserved" + } + t.Run(name, func(t *testing.T) { + e, v := newOrderingTestVoter(t, reserved) + for slot := uint64(40); slot <= 43; slot++ { + id := observeOrderingBlock(t, e, slot) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.Equal(t, slot, v.highestFinal) + require.Less(t, v.admissionFloor(), slot) + require.False(t, v.history.VotedAt(slot), "network finality does not prove local execution") + before := v.snapshot() + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: slot, Source: "ordering-test"})) + drainOrderingEvents(t, v) + hash, ok := v.history.NotarizedVote(slot) + require.True(t, ok, "replay must still notarize after slow finality") + require.Equal(t, id.Hash, hash) + require.Greater(t, v.snapshot().BroadcastMessagesQueued, before.BroadcastMessagesQueued) + require.NoError(t, e.AlpenglowSafetyError()) + } + }) + } +} + +func TestAlpenglowVoterNetworkFinalityDuringLocalAdmission(t *testing.T) { + e, v := newOrderingTestVoter(t, false) + id := observeOrderingBlock(t, e, 40) + called := false + v.beforeLocalVoteInject = func(vote alpenglow.Vote) { + if called { + return + } + called = true + require.Equal(t, alpenglow.NewNotarizationVote(id.Slot, id.Hash), vote) + finalizeOrderingBlock(t, e, v, id) + } + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: id.Slot})) + drainOrderingEvents(t, v) + require.True(t, called) + require.Positive(t, v.snapshot().VotesCastThisRun) + message, _, err := v.sign(alpenglow.NewNotarizationVote(id.Slot, id.Hash), false) + require.NoError(t, err) + require.True(t, e.ensurePool().HasVerifiedVote(message)) + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestAlpenglowVoterDurableRootCannotOvertakeQueuedReplay(t *testing.T) { + e, v := newOrderingTestVoter(t, true) + id := observeOrderingBlock(t, e, 40) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: id.Slot})) + e.PruneAlpenglowBefore(id.Slot) + require.Equal(t, uint64(39), e.ensurePool().Snapshot().RootSlot) + require.Contains(t, e.executedReplayBlocks, id, "queued replay must retain its execution proof") + queuedBefore := v.snapshot().BroadcastMessagesQueued + drainOrderingEvents(t, v) + require.Greater(t, v.snapshot().BroadcastMessagesQueued, queuedBefore) + require.Equal(t, id.Slot, v.history.Root) + require.Equal(t, id.Slot, e.ensurePool().Snapshot().RootSlot) + require.NotContains(t, e.executedReplayBlocks, id) + _, ok := v.history.NotarizedVote(id.Slot) + require.True(t, ok, "root vote must remain available to its intra-window child") + child := observeOrderingBlock(t, e, 41) + finalizeOrderingBlock(t, e, v, child) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: child.Slot})) + drainOrderingEvents(t, v) + hash, ok := v.history.NotarizedVote(child.Slot) + require.True(t, ok) + require.Equal(t, child.Hash, hash) + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestAlpenglowVoterFinalityPreservesEarlierSkipDecision(t *testing.T) { + e, v := newOrderingTestVoter(t, true) + require.NoError(t, v.history.AddVote(alpenglow.NewSkipVote(40))) + id := observeOrderingBlock(t, e, 40) + finalizeOrderingBlock(t, e, v, id) + drainOrderingEvents(t, v) + require.NoError(t, e.OnReplayResult(context.Background(), SlotReplayResult{Slot: 40})) + drainOrderingEvents(t, v) + require.True(t, v.history.HasSkipped(40)) + _, ok := v.history.NotarizedVote(40) + require.False(t, ok, "late replay must never replace an earlier round-one decision") + require.NoError(t, e.AlpenglowSafetyError()) +} + +func TestReservedRecoveryUsesVerifiedFinalityNotLiveAdmissionFloor(t *testing.T) { + cfg := reservedTestConfig(t.TempDir()) + v, err := openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + reserveThrough(t, v.reservation, 40) + h := v.reservation.through.Load() + crashReservedTestVoter(t, v) + cfg.InitializeVoteReservation = false + v, err = openReservedTestVoter(t, cfg, 39) + require.NoError(t, err) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), false) + require.ErrorIs(t, err, errVoterNotReady) + id := observeOrderingBlock(t, v.engine, h) + finalizeOrderingBlock(t, v.engine, v, id) + require.Less(t, v.admissionFloor(), h) + require.Equal(t, h, v.engine.alpenglowVerifiedFinalityFloor()) + reserveThrough(t, v.reservation, h+1) + for _, normal := range []bool{false, true} { + _, _, err = v.sign(alpenglow.NewSkipVote(h), normal) + require.ErrorIs(t, err, errVoterNotReady) + _, _, err = v.sign(alpenglow.NewSkipVote(h+1), normal) + require.NoError(t, err) + } +} diff --git a/pkg/consensus/voter_wait_slot_test.go b/pkg/consensus/voter_wait_slot_test.go new file mode 100644 index 000000000..b06a35819 --- /dev/null +++ b/pkg/consensus/voter_wait_slot_test.go @@ -0,0 +1,91 @@ +package consensus + +import ( + "crypto/ed25519" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/alpenglow" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func waitSlotTestVoter(t *testing.T, cutoff uint64) *alpenglowVoter { + t.Helper() + identity, authorized := voterTestKey(11), voterTestKey(12) + voteAccount := solana.PublicKey(voterTestKey(13).Public().(ed25519.PublicKey)) + set := voterTestValidatorSet(t, identity, authorized, voteAccount) + root := alpenglow.BlockID{Slot: 39, Hash: solana.Hash{0x39}} + engine, err := NewEngine(Config{AlpenglowShredVersion: 0x1234, AlpenglowIdentity: identity}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, engine.Close()) }) + engine.SetAlpenglowEpochLookup(func(uint64) uint64 { return set.Epoch }) + require.NoError(t, engine.SetAlpenglowValidatorSet(set)) + engine.SetAlpenglowRoot(root) + voter, err := newAlpenglowVoterUnstarted(engine, VotingConfig{ + Identity: identity, AuthorizedVoter: authorized, VoteAccount: voteAccount, + HistoryDir: t.TempDir(), EpochForSlot: func(uint64) uint64 { return set.Epoch }, + Peers: func([]alpenglow.ValidatorStake) []alpenglow.VotorPeer { return nil }, + WaitToVoteSlot: cutoff, ReadyToVote: func(uint64) bool { return true }, + }, root, []alpenglow.ValidatorSet{set}) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, voter.close()) }) + return voter +} + +func TestWaitToVoteSlotGatesEveryNewVoteType(t *testing.T) { + voter := waitSlotTestVoter(t, 44) + constructors := []func(uint64) alpenglow.Vote{ + func(slot uint64) alpenglow.Vote { return alpenglow.NewNotarizationVote(slot, solana.Hash{1}) }, + alpenglow.NewFinalizationVote, + alpenglow.NewSkipVote, + func(slot uint64) alpenglow.Vote { return alpenglow.NewNotarizationFallbackVote(slot, solana.Hash{1}) }, + alpenglow.NewSkipFallbackVote, + } + for _, started := range []bool{false, true} { + voter.votingStarted = started + for _, voteAt := range constructors { + vote := voteAt(43) + _, _, err := voter.sign(vote, true) + require.ErrorIs(t, err, errVoterNotReady, "%s started=%t", vote.Type, started) + for _, slot := range []uint64{44, 45} { + message, _, err := voter.sign(voteAt(slot), true) + require.NoError(t, err, "%s slot=%d started=%t", vote.Type, slot, started) + require.Equal(t, voteAt(slot), message.Vote) + } + } + } +} + +func TestWaitToVoteSlotSplitsSkipWindowAndPersistsOnlyAllowedVotes(t *testing.T) { + voter := waitSlotTestVoter(t, 42) + require.NoError(t, voter.trySkipWindow(40)) + for _, slot := range []uint64{40, 41} { + require.False(t, voter.history.VotedAt(slot)) + } + for _, slot := range []uint64{42, 43} { + require.True(t, voter.history.HasSkipped(slot)) + } + restored, err := alpenglow.LoadVoteHistory(voter.historyDir, voter.node) + require.NoError(t, err) + require.Equal(t, voter.history.VotesCast, restored.VotesCast) + require.EqualValues(t, 2, voter.engine.ensurePool().Snapshot().VerifiedVotes) + // Joining live voting must not make older slots eligible afterward. + require.True(t, voter.votingStarted) + voted, err := voter.cast(alpenglow.NewSkipVote(41), false) + require.NoError(t, err) + require.False(t, voted) + require.False(t, voter.history.VotedAt(41)) +} + +func TestWaitToVoteSlotPreservesAuthenticatedHistoryRestoration(t *testing.T) { + voter := waitSlotTestVoter(t, 44) + require.NoError(t, voter.history.AddVote(alpenglow.NewSkipVote(40))) + require.NoError(t, voter.saveHistory()) + restored, err := alpenglow.LoadVoteHistory(voter.historyDir, voter.node) + require.NoError(t, err) + voter.history = restored + require.NoError(t, voter.restoreVotesForEpoch(voter.epochForSlot(40))) + require.EqualValues(t, 1, voter.engine.ensurePool().Snapshot().VerifiedVotes) + require.False(t, voter.votingStarted, "restoring a recorded vote must not bypass startup readiness") + require.ErrorIs(t, voter.votingGateError(41), errVoterNotReady) +} diff --git a/pkg/costmodel/entry_bytes.go b/pkg/costmodel/entry_bytes.go index ab91b8d5a..7cfec6481 100644 --- a/pkg/costmodel/entry_bytes.go +++ b/pkg/costmodel/entry_bytes.go @@ -28,8 +28,11 @@ func PackEntryBytesMax(slotMaxDataShreds, maxMicroblock uint64) uint64 { // DefaultPackEntryBytes is min(shred-safe, SIMD-0525) minus one ending tick. func DefaultPackEntryBytes() uint64 { - shredSafe := PackEntryBytesMax(DefaultMaxDataShredsPerSlot, MaxMicroblockBytes) - cap := uint64(DefaultMaxEntryBytesPerSlot) + return packEntryBytes(DefaultMaxDataShredsPerSlot, DefaultMaxEntryBytesPerSlot) +} + +func packEntryBytes(maxDataShreds, cap uint64) uint64 { + shredSafe := PackEntryBytesMax(maxDataShreds, MaxMicroblockBytes) if shredSafe > 0 && shredSafe < cap { cap = shredSafe } diff --git a/pkg/costmodel/limits.go b/pkg/costmodel/limits.go index 2e4766569..4e3f726ff 100644 --- a/pkg/costmodel/limits.go +++ b/pkg/costmodel/limits.go @@ -1,7 +1,11 @@ package costmodel import ( + "fmt" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/safemath" + "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/gagliardetto/solana-go" ) @@ -47,11 +51,11 @@ const ( TypicalDataShredPayloadBytes = 963 // DataShredsPerFECSet matches turbine's 32:32 erasure batch. DataShredsPerFECSet = 32 - // FECSetsPerBatch is the close watermark: hold until one FEC set is full. - FECSetsPerBatch = 1 + // FECSetsPerBatch is the close watermark: hold until two FEC sets are full. + FECSetsPerBatch = 2 // TypicalFECSetPayloadBytes is one full unsigned FEC set. TypicalFECSetPayloadBytes = DataShredsPerFECSet * TypicalDataShredPayloadBytes - // DefaultTargetBatchBytes is one FEC set. A short leftover is only + // DefaultTargetBatchBytes is two FEC sets. A short leftover is only // emitted at slot end (Freeze / ending tick). DefaultTargetBatchBytes = FECSetsPerBatch * TypicalFECSetPayloadBytes ) @@ -75,11 +79,53 @@ func DefaultLimits() Limits { } } -// LimitsForFeatures returns the cost limits selected by the bank's feature set. +// LimitsForFeatures returns the legacy 400ms budgets. Live banks must use +// LimitsForSlot to apply slot-time reductions at the correct epoch boundary. func LimitsForFeatures(feats *features.Features) Limits { limits := DefaultLimits() if feats != nil && feats.IsActive(features.RaiseBlockLimitsTo100m) { limits.BlockCost = MaxBlockUnitsSIMD0286 + limits.WritableAccountCost = 40_000_000 } return limits } + +// LimitsForSlot mirrors Agave v4.3.0-rc.1 runtime/slot_params.rs. A slot-time +// gate takes effect in the epoch after activation; among effective gates the +// shortest duration wins, even if longer-duration gates activate later. +func LimitsForSlot(feats *features.Features, schedule *sealevel.SysvarEpochSchedule, slot uint64) (Limits, error) { + limits := DefaultLimits() + for _, transition := range []struct { + gate features.FeatureGate + account, block, data, shreds, entries uint64 + }{ + {features.ReduceSlotTimeTo350ms, 21_000_000, 52_500_000, 87_500_000, 28_672, 18_350_080}, + {features.ReduceSlotTimeTo300ms, 18_000_000, 45_000_000, 75_000_000, 24_576, 15_728_640}, + {features.ReduceSlotTimeTo250ms, 15_000_000, 37_500_000, 62_500_000, 20_480, 13_107_200}, + {features.ReduceSlotTimeTo200ms, 12_000_000, 30_000_000, 50_000_000, 16_384, 10_485_760}, + } { + if feats == nil { + break + } + activation, active := feats.ActivationSlot(transition.gate) + if !active { + continue + } + if schedule == nil || schedule.SlotsPerEpoch == 0 { + return Limits{}, fmt.Errorf("epoch schedule required for slot-time cost limits") + } + effective := schedule.FirstSlotInEpoch(safemath.SaturatingAddU64(schedule.GetEpoch(activation), 1)) + if effective > slot { + continue + } + limits.WritableAccountCost = transition.account + limits.BlockCost = transition.block + limits.AllocatedDataSizeDelta = transition.data + limits.MaxEntryBytes = packEntryBytes(transition.shreds, transition.entries) + } + if feats != nil && feats.IsActive(features.RaiseBlockLimitsTo100m) { + limits.BlockCost = limits.BlockCost * 100 / 60 + limits.WritableAccountCost = limits.WritableAccountCost * 100 / 60 + } + return limits, nil +} diff --git a/pkg/costmodel/limits_test.go b/pkg/costmodel/limits_test.go new file mode 100644 index 000000000..60de4062a --- /dev/null +++ b/pkg/costmodel/limits_test.go @@ -0,0 +1,13 @@ +package costmodel + +import "testing" + +func TestDefaultTargetBatchBytesMatchesTwoTypicalFECSets(t *testing.T) { + const want = 61_632 + if DefaultTargetBatchBytes != want { + t.Fatalf("DefaultTargetBatchBytes = %d, want %d", DefaultTargetBatchBytes, want) + } + if DefaultTargetBatchBytes != 2*DataShredsPerFECSet*TypicalDataShredPayloadBytes { + t.Fatal("default batch target must remain two complete typical FEC payloads") + } +} diff --git a/pkg/costmodel/slot_limits_test.go b/pkg/costmodel/slot_limits_test.go new file mode 100644 index 000000000..0d521e744 --- /dev/null +++ b/pkg/costmodel/slot_limits_test.go @@ -0,0 +1,88 @@ +package costmodel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/stretchr/testify/require" +) + +func TestSlotLimitsMatchAgaveTable(t *testing.T) { + schedule := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100} + for _, tc := range []struct { + name string + gate features.FeatureGate + account, block, data, shreds, entries uint64 + }{ + {"400ms", features.FeatureGate{}, 24_000_000, 60_000_000, 100_000_000, 32768, 20 * 1024 * 1024}, + {"350ms", features.ReduceSlotTimeTo350ms, 21_000_000, 52_500_000, 87_500_000, 28672, 18_350_080}, + {"300ms", features.ReduceSlotTimeTo300ms, 18_000_000, 45_000_000, 75_000_000, 24576, 15_728_640}, + {"250ms", features.ReduceSlotTimeTo250ms, 15_000_000, 37_500_000, 62_500_000, 20480, 13_107_200}, + {"200ms", features.ReduceSlotTimeTo200ms, 12_000_000, 30_000_000, 50_000_000, 16384, 10_485_760}, + } { + t.Run(tc.name, func(t *testing.T) { + f := features.NewFeaturesDefault() + if tc.name != "400ms" { + f.EnableFeature(tc.gate, 50) + } + before, err := LimitsForSlot(f, schedule, 99) + require.NoError(t, err) + require.Equal(t, DefaultLimits(), before) + for _, raise := range []bool{false, true} { + if raise { + f.EnableFeature(features.RaiseBlockLimitsTo100m, 0) + } + got, err := LimitsForSlot(f, schedule, 100) + require.NoError(t, err) + account, block := tc.account, tc.block + if raise { + account = account * 100 / 60 + block = block * 100 / 60 + } + require.Equal(t, account, got.WritableAccountCost) + require.Equal(t, block, got.BlockCost) + require.Equal(t, tc.data, got.AllocatedDataSizeDelta) + require.Equal(t, tc.entries-EntryHeaderBytes, got.MaxEntryBytes) + require.LessOrEqual(t, got.MaxEntryBytes, PackEntryBytesMax(tc.shreds, MaxMicroblockBytes)) + require.Equal(t, uint64(DefaultTargetBatchBytes), got.MaxBatchBytes) + } + }) + } +} + +func TestSlotLimitsDoNotLengthenSlotsForLaterGates(t *testing.T) { + f := features.NewFeaturesDefault() + f.EnableFeature(features.ReduceSlotTimeTo200ms, 50) + f.EnableFeature(features.ReduceSlotTimeTo350ms, 150) + schedule := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100} + for _, slot := range []uint64{100, 199, 200, 400} { + limits, err := LimitsForSlot(f, schedule, slot) + require.NoError(t, err) + require.Equal(t, uint64(30_000_000), limits.BlockCost) + } +} + +func TestSlotLimitsActivationDuringWarmup(t *testing.T) { + f := features.NewFeaturesDefault() + f.EnableFeature(features.ReduceSlotTimeTo200ms, 40) + // Epoch 0: [0,32), epoch 1: [32,96), normal epoch 2: [96,224). + schedule := &sealevel.SysvarEpochSchedule{Warmup: true, SlotsPerEpoch: 128, FirstNormalEpoch: 2, FirstNormalSlot: 96} + before, err := LimitsForSlot(f, schedule, 95) + require.NoError(t, err) + require.Equal(t, uint64(60_000_000), before.BlockCost) + after, err := LimitsForSlot(f, schedule, 96) + require.NoError(t, err) + require.Equal(t, uint64(30_000_000), after.BlockCost) +} + +func TestSlotLimitsRequireScheduleForActiveReductions(t *testing.T) { + f := features.NewFeaturesDefault() + _, err := LimitsForSlot(f, nil, 10) + require.NoError(t, err) + f.EnableFeature(features.ReduceSlotTimeTo200ms, 0) + _, err = LimitsForSlot(f, nil, 10) + require.Error(t, err) + _, err = LimitsForSlot(f, &sealevel.SysvarEpochSchedule{}, 10) + require.Error(t, err) +} diff --git a/pkg/costmodel/transaction_cost.go b/pkg/costmodel/transaction_cost.go index 1dd4f0c71..3a804b037 100644 --- a/pkg/costmodel/transaction_cost.go +++ b/pkg/costmodel/transaction_cost.go @@ -55,6 +55,13 @@ func EstimateTransactionCost(tx *solana.Transaction, feats *features.Features) ( }, nil } + return EstimatePreparedTransactionCost(tx, instrs, limits, feats), nil +} + +// EstimatePreparedTransactionCost reuses successfully parsed instructions and +// compute limits from the same immutable transaction and feature snapshot. +func EstimatePreparedTransactionCost(tx *solana.Transaction, instrs []sealevel.Instruction, limits *sealevel.ComputeBudgetLimits, feats *features.Features) TransactionCost { + writable := writableAccounts(tx) loadedDataCost := loadedAccountsDataSizeCost(limits.LoadedAccountBytes) // Banking-stage admission must reserve at least one page for the fee // payer, including V1 transactions whose inline loaded-data limit is zero. @@ -62,13 +69,13 @@ func EstimateTransactionCost(tx *solana.Transaction, feats *features.Features) ( loadedDataCost = max(loadedDataCost, uint64(HeapCost)) return TransactionCost{ SignatureCost: signatureCost(tx, instrs, feats), - WriteLockCost: writeLockCost(countWriteLocks(tx)), + WriteLockCost: writeLockCost(uint64(len(writable))), DataBytesCost: instructionDataCost(tx), ProgramsExecutionCost: uint64(limits.ComputeUnitLimit), LoadedAccountsDataSizeCost: loadedDataCost, AllocatedAccountsDataSize: estimateAllocDelta(instrs, feats), - WritableAccounts: writableAccounts(tx), - }, nil + WritableAccounts: writable, + } } func signatureCost(tx *solana.Transaction, instrs []sealevel.Instruction, feats *features.Features) uint64 { @@ -177,7 +184,7 @@ func replayInstrsAndAcctMetas(tx *solana.Transaction, feats *features.Features) } upgradeableLoaderPresent := false for _, key := range tx.Message.AccountKeys { - if key.String() == "BPFLoaderUpgradeab1e11111111111111111111111" { + if key == addresses.BpfLoaderUpgradeableAddr { upgradeableLoaderPresent = true break } diff --git a/pkg/merkletree/merkletree.go b/pkg/merkletree/merkletree.go index 988f2633c..0de1d98f3 100644 --- a/pkg/merkletree/merkletree.go +++ b/pkg/merkletree/merkletree.go @@ -46,8 +46,29 @@ func (n *Nodes) GetRoot() (out *[32]byte) { return &n.Nodes[len(n.Nodes)-1] } -// TODO provide a method for memory-efficient Merkle construction when only the root is requested. -// Can be implemented using recursion root level downwards +// HashRoot computes the same root as HashNodes without retaining proof nodes. +// Empty input returns zero. Each level overwrites the preceding level, so only +// one hash per leaf is allocated. Leaves are never modified. +func HashRoot(leaves [][]byte) (root [32]byte) { + if len(leaves) == 0 { + return root + } + if len(leaves) == 1 { + return HashLeaf(leaves[0]) + } + nodes := make([][32]byte, len(leaves)) + for i, leaf := range leaves { + nodes[i] = HashLeaf(leaf) + } + for len(nodes) > 1 { + for i := 0; i < len(nodes); i += 2 { + right := min(i+1, len(nodes)-1) + nodes[i/2] = HashIntermediate(&nodes[i], &nodes[right]) + } + nodes = nodes[:(len(nodes)+1)/2] + } + return nodes[0] +} // HashNodes constructs proof data from a set of leaves. // @@ -97,6 +118,14 @@ func HashNodes(leaves [][]byte) (out Nodes) { // HashLeaf returns the hash of a leaf node. func HashLeaf(data []byte) (out [32]byte) { + if len(data) == 64 { + // Transaction signatures are fixed-width leaves. A single buffer avoids + // incremental hash writes while retaining the leaf domain separator. + var input [65]byte + input[0] = TypeLeaf + copy(input[1:], data) + return sha256.Sum256(input[:]) + } h := sha256.New() h.Write([]byte{TypeLeaf}) h.Write(data) @@ -106,12 +135,11 @@ func HashLeaf(data []byte) (out [32]byte) { // HashIntermediate returns the hash of an intermediate node. func HashIntermediate(left *[32]byte, right *[32]byte) (out [32]byte) { - h := sha256.New() - h.Write([]byte{TypeIntermediate}) - h.Write(left[:]) - h.Write(right[:]) - h.Sum(out[:0]) - return + var input [65]byte + input[0] = TypeIntermediate + copy(input[1:33], left[:]) + copy(input[33:], right[:]) + return sha256.Sum256(input[:]) } // nextLevelLen returns the amount of nodes in the layer above the current one, diff --git a/pkg/merkletree/root_test.go b/pkg/merkletree/root_test.go new file mode 100644 index 000000000..1437c3833 --- /dev/null +++ b/pkg/merkletree/root_test.go @@ -0,0 +1,60 @@ +package merkletree + +import ( + "bytes" + "crypto/sha256" + "fmt" + "testing" +) + +// Independent construction: retain separate levels and use one-shot SHA256 +// over explicit domain-prefixed bytes, without the production hash helpers. +func referenceRoot(leaves [][]byte) [32]byte { + if len(leaves) == 0 { + return [32]byte{} + } + level := make([][32]byte, len(leaves)) + for i, leaf := range leaves { + level[i] = sha256.Sum256(append([]byte{0}, leaf...)) + } + for len(level) > 1 { + next := make([][32]byte, (len(level)+1)/2) + for i := range next { + left, right := level[2*i], level[min(2*i+1, len(level)-1)] + data := append([]byte{1}, left[:]...) + data = append(data, right[:]...) + next[i] = sha256.Sum256(data) + } + level = next + } + return level[0] +} + +func TestHashRootMatchesCanonicalTree(t *testing.T) { + counts := []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 63, 64, 65, 127, 128, 129, 255, 256, 257, 311, 511, 512, 513, 1023, 1024, 1025} + for _, size := range []int{0, 31, 32, 55, 56, 63, 64, 65, 1232} { + for _, count := range counts { + t.Run(fmt.Sprintf("%dx%d", count, size), func(t *testing.T) { + leaves := make([][]byte, count) + for i := range leaves { + leaves[i] = make([]byte, size) + for j := range leaves[i] { + leaves[i][j] = byte(i*71 + i/256 + j*17) + } + } + before := bytes.Join(leaves, nil) + want := referenceRoot(leaves) + if got := HashRoot(leaves); got != want { + t.Fatalf("root %x, want %x", got, want) + } + nodes := HashNodes(leaves) + if got := nodes.GetRoot(); got != nil && *got != want { + t.Fatalf("proof root %x, want %x", *got, want) + } + if !bytes.Equal(before, bytes.Join(leaves, nil)) { + t.Fatal("hashing modified input leaves") + } + }) + } + } +} diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 10548fdd4..3af188d16 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -15,8 +15,18 @@ func (t *Timing) AddTiming(d time.Duration) { atomic.AddUint64(&t.SumNanoseconds, uint64(d.Nanoseconds())) } +// StartTiming avoids reading the clock when a caller does not record timings. +func StartTiming(enabled bool) time.Time { + if enabled { + return time.Now() + } + return time.Time{} +} + func (t *Timing) AddTimingSince(start time.Time) { - t.AddTiming(time.Since(start)) + if !start.IsZero() { + t.AddTiming(time.Since(start)) + } } // AccountLoader is the per-slot decomposition of LoadBlockAccounts. Counters @@ -94,15 +104,26 @@ type AccountLoader struct { SysvarCachePublicationEpochRejects uint64 } -// TurbineIngress is the exact per-slot pre-replay pipeline decomposition. +// TurbineIngress records per-slot pre-replay pipeline observations. // It is written to replay_timings.jsonl without high-cardinality metric labels. type TurbineIngress struct { ShredCollection Timing CompletionQueueDelay Timing BlockDecode Timing + // Completion-only parse and outstanding-signature join/verification time. TransactionParse Timing TransactionSigverify Timing ReplayAdmission Timing + // Summed completed prefetched component durations, including any discarded + // optimistic prefix. Overlap reception; not CPU time or additive wall stages. + // Early sigverify includes queueing. + EarlyTransactionParse Timing + EarlyTransactionSigverify Timing + // Completion wait for claimed background parsing/submission, outside BlockDecode. + EarlyPreparationWait Timing + EarlyVerifiedTransactions uint64 + // FullToReady contains the completion stages above, excluding admission. + FullToReady Timing } // VoteRewardDetails decomposes RewardCertificatePreflight and @@ -171,16 +192,20 @@ type BlockReplay struct { // BlockUpdateAccounts is synchronous critical-path work: rooted-tail // buffering (including its callback) or legacy store enqueue. It excludes // legacy asynchronous disk completion. - BlockUpdateAccounts Timing - TransactionStatusCommit Timing - SignatureVerificationJoin Timing - AccountsDeltaHash Timing - LtHashDedupe Timing - LtHashWorkerCompute Timing - LtHashPartialReduce Timing - BankHashFinalize Timing - BankHash Timing - AlpenglowFooterVerification Timing + BlockUpdateAccounts Timing + TransactionStatusCommit Timing + // Preparation overlaps execution and is not additive with replay wall time. + // PreparationWait is the residual join nested within TransactionStatusCommit. + TransactionStatusPreparation Timing + TransactionStatusPreparationWait Timing + SignatureVerificationJoin Timing + AccountsDeltaHash Timing + LtHashDedupe Timing + LtHashWorkerCompute Timing + LtHashPartialReduce Timing + BankHashFinalize Timing + BankHash Timing + AlpenglowFooterVerification Timing // PostProcessBlock is caller-side state publication and replay // bookkeeping after ProcessBlock returns. TransactionStatusView, // ChainTipUpdate, and ResumeContext are nested sub-phases; logging, summary diff --git a/pkg/replay/async_checkpoint_capture_test.go b/pkg/replay/async_checkpoint_capture_test.go new file mode 100644 index 000000000..a556c5a37 --- /dev/null +++ b/pkg/replay/async_checkpoint_capture_test.go @@ -0,0 +1,172 @@ +package replay + +import ( + "encoding/json" + "errors" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/state" + "github.com/stretchr/testify/require" +) + +type testCheckpointEncoder func() ([]byte, error) + +func (f testCheckpointEncoder) MarshalBinary() ([]byte, error) { return f() } + +func testCheckpointBytes(payload []byte) TransactionStatusSnapshot { + owned := append([]byte(nil), payload...) + return testCheckpointEncoder(func() ([]byte, error) { return append([]byte(nil), owned...), nil }) +} + +func TestAsyncCheckpointEncodingDoesNotRunDuringJobBuild(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + started, release := make(chan struct{}), make(chan struct{}) + blockEncoding := false + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: func(through uint64) (TransactionStatusSnapshot, error) { + require.Equal(t, uint64(6), through) + return testCheckpointEncoder(func() ([]byte, error) { + if !blockEncoding { + return nil, errors.New("encoder ran during job construction") + } + close(started) + <-release + return []byte("encoded-on-worker"), nil + }), nil + }, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(6, false) + require.NoError(t, err) + select { + case <-started: + t.Fatal("job construction ran the encoder") + default: + } + promoter := newAsyncPromoter(fc) + // Always unblock the worker before draining it, including a failed assertion. + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + defer promoter.stop() + defer unblock() + blockEncoding = true + promoter.enqueue(job) + select { + case <-started: + case <-time.After(2 * time.Second): + t.Fatal("checkpoint worker did not start encoding") + } + // Replay can publish a later bank while the worker's encoder is blocked. + tail.Add(7, []*accounts.Account{testAccount(3, 7)}, testHashBytes(7)) + tail.SetContext(7, &state.ResumeContext{Slot: 7}) + require.Equal(t, 3, tail.overlay.HeldSlots()) + require.Nil(t, promoter.poll()) + + unblock() + result := promoter.drain() + require.NotNil(t, result) + require.NoError(t, result.err) + require.Nil(t, result.job.transactionStatusSnapshot) + tail.applyFoldJob(result.job) + require.Equal(t, 1, tail.overlay.HeldSlots()) +} + +func TestCheckpointCaptureFailureOrdering(t *testing.T) { + cases := []struct { + name string + capture func(uint64) (TransactionStatusSnapshot, error) + want string + buildFails bool + }{ + {"nil capture", func(uint64) (TransactionStatusSnapshot, error) { return nil, nil }, "capture is nil", true}, + {"capture error", func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("capture failed") }, "capture failed", true}, + {"encode error", func(uint64) (TransactionStatusSnapshot, error) { + return testCheckpointEncoder(func() ([]byte, error) { return nil, errors.New("encode failed") }), nil + }, "encode failed", false}, + {"empty encoding", func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes(nil), nil }, "snapshot is empty", false}, + } + for _, tc := range cases { + for _, forced := range []bool{false, true} { + name := tc.name + "/async" + if forced { + name = tc.name + "/forced" + } + t.Run(name, func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + installed := false + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: tc.capture, + Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { + installed = true + return nil, errors.New("unexpected install") + }, + })) + if forced { + through, _, err := tail.flush(6) + require.ErrorContains(t, err, tc.want) + require.Zero(t, through) + } else { + job, err := tail.buildFoldJob(6, false) + if tc.buildFails { + require.ErrorContains(t, err, tc.want) + require.Nil(t, job) + } else { + require.NoError(t, err) + require.ErrorContains(t, runFoldJob(fc, job), tc.want) + require.Nil(t, job.transactionStatusSnapshot, "failed result retained its captured deltas") + } + } + require.False(t, installed) + require.Empty(t, fc.throughs) + require.Equal(t, 2, tail.overlay.HeldSlots()) + }) + } + } +} + +func TestFoldCheckpointKeepsCapturedRootAfterLiveCacheAdvances(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 319, 320, 321) + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: c.CaptureSnapshotThrough, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(321, false) + require.NoError(t, err) + for slot := uint64(351); slot <= 660; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + c.Root(slot - 20) + } + require.NoError(t, runFoldJob(fc, job)) + require.Nil(t, job.transactionStatusSnapshot, "completed result retained captured deltas") + require.Positive(t, job.checkpointCaptureTime) + require.Positive(t, job.checkpointEncodeTime) + require.Equal(t, len(want), job.checkpointBytes) + var manifest state.ResumeContext + require.NoError(t, json.Unmarshal(fc.ctxs[320], &manifest)) + require.Equal(t, uint64(320), manifest.TransactionStatusCheckpoint.Root) + got, err := ReadTransactionStatusCheckpoint(rootDir, manifest.TransactionStatusCheckpoint) + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) +} diff --git a/pkg/replay/async_promotion_test.go b/pkg/replay/async_promotion_test.go index 49d47a49e..2780e425d 100644 --- a/pkg/replay/async_promotion_test.go +++ b/pkg/replay/async_promotion_test.go @@ -22,7 +22,7 @@ type slowCommitter struct { delay time.Duration } -func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { +func TestFoldJobCapturesStatusOnLoopAndReferenceRidesManifest(t *testing.T) { rootDir := t.TempDir() fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6, 7) @@ -32,10 +32,10 @@ func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { installCalled := false afterCommitCalled := false hooks := TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(6), through) snapshotCalled = true - return scratch, nil + return testCheckpointBytes(scratch), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { require.True(t, snapshotCalled, "worker install ran before loop snapshot") @@ -83,7 +83,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return nil, errors.New("snapshot boom") }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("snapshot boom") }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { t.Fatal("install must not run") return nil, nil @@ -101,7 +101,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { return nil, errors.New("fsync boom") }, @@ -120,7 +120,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { tail := asyncTestTail(fc, 5, 6) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) }, @@ -144,9 +144,9 @@ func TestForcedFoldCarriesStatusCheckpointReference(t *testing.T) { tail := asyncTestTail(fc, 5) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(5), through) - return []byte("forced-partial-status"), nil + return testCheckpointBytes([]byte("forced-partial-status")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) @@ -175,7 +175,7 @@ func TestCheckpointAfterCommitRequiresDurabilityHooks(t *testing.T) { err := tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ AfterCommit: func(*state.TransactionStatusCheckpointRef) error { return nil }, }) - require.ErrorContains(t, err, "requires Snapshot and Install") + require.ErrorContains(t, err, "requires Capture and Install") } func (c *slowCommitter) CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) { @@ -444,3 +444,26 @@ func TestShutdownFlushCannotFoldPastGateTarget(t *testing.T) { target = safePromoteTarget(9, true, 7, 6) assert.Equal(t, uint64(5), target, "persisted-divergence floor holds promotion below the disputed slot") } + +// Model repeated replay/skip iterations while a nearly full checkpoint batch +// waits for one more held bank. Account writes must not be copied on this path. +func BenchmarkBuildFoldJobWaitingForBatch(b *testing.B) { + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 128, "") + writes := make([]*accounts.Account, 512) + for i := range writes { + var key [32]byte + key[0], key[1] = byte(i), byte(i>>8) + writes[i] = &accounts.Account{Key: key, Lamports: 1} + } + for slot := uint64(1); slot <= 127; slot++ { + tail.Add(slot, writes, nil) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + job, err := tail.buildFoldJob(127, false) + if err != nil || job != nil { + b.Fatalf("unexpected fold admission: job=%v err=%v", job, err) + } + } +} diff --git a/pkg/replay/block.go b/pkg/replay/block.go index bb465bc70..646554962 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1747,6 +1747,7 @@ func ReplayBlocks( var unwoundParentBankSysvars *sealevel.BankSysvars var partitionedEpochRewardsEnabled bool var partitionedRewardsInfo *rewards.PartitionedRewardDistributionInfo + var rewardsCompletion partitionedRewardsCompletion var featuresActivatedInFirstSlot []*accounts.Account var parentFeaturesActivatedInFirstSlot []*accounts.Account @@ -1987,9 +1988,9 @@ func ReplayBlocks( checkpointAfterCommit = consensusOpts.TransactionStatusCheckpointAfterCommit } if hookErr := unrootedTailState.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - // Snapshot runs here on the replay loop during fold-job construction; - // only its immutable bytes cross to the async worker. - Snapshot: transactionStatuses.SnapshotThrough, + // Pin the exact immutable view on replay. Sorting and encoding run + // on the existing fold worker, after releasing the live cache lock. + Capture: transactionStatuses.CaptureSnapshotThrough, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(acctsDbPath, through, payload) }, @@ -2061,6 +2062,10 @@ func ReplayBlocks( mithrilState.LastRootedSlot = promotedThrough mithrilState.LastRootedBankhash = rootedCtx.Bankhash mithrilState.LastRootedContext = rootedCtx + if rewardsCompletion.retire(&partitionedRewardsInfo, promotedThrough) { + rewardsHoldBelowSlot = 0 + mlog.Log.Infof("epoch rewards bookkeeping retired through durable slot %d; later fork switches may unwind in memory", promotedThrough) + } if transactionStatuses.Root(promotedThrough) { mlog.Log.Infof("transaction status cache reconstructed complete %d-root coverage through durable slot %d", maxTransactionStatusRoots, promotedThrough) @@ -2831,6 +2836,11 @@ func ReplayBlocks( record.TransactionParse.AddTiming(ingressTimings.TransactionParse) record.TransactionSigverify.AddTiming(ingressTimings.TransactionSigverify) record.ReplayAdmission.AddTiming(ingressTimings.ReplayAdmission) + record.EarlyTransactionParse.AddTiming(ingressTimings.EarlyTransactionParse) + record.EarlyTransactionSigverify.AddTiming(ingressTimings.EarlyTransactionSigverify) + record.EarlyPreparationWait.AddTiming(ingressTimings.EarlyPreparationWait) + record.EarlyVerifiedTransactions = ingressTimings.EarlyVerifiedTransactions + record.FullToReady.AddTiming(ingressTimings.FullToReady) } start := time.Now() @@ -2909,6 +2919,7 @@ func ReplayBlocks( boundaryParentCtx = epochBoundaryParentCtx(acctsDb, block, currentEpoch, replayCtx.CurrentFeatures) } partitionedRewardsInfo = handleEpochTransition(acctsDb, partitionedEpochRewardsEnabled, boundaryParentCtx, replayCtx, epochSchedule, replayCtx.CurrentFeatures, block, currentEpoch, rpcc, dbgOpts) + rewardsCompletion = partitionedRewardsCompletion{} currentEpoch = block.Epoch justCrossedEpochBoundary = true // While partitioned rewards are distributing, hold durable promotion @@ -3021,6 +3032,7 @@ func ReplayBlocks( } // The successful child now owns its derived snapshot. Any later bank uses // lastSlotCtx; the one-shot retained unwind bridge is no longer needed. + rewardsCompletion.observeBank(partitionedRewardsInfo, lastSlotCtx.BankSysvars()) unwoundParentBankSysvars = nil postProcessBlockStart := processBlockEnd statusViewStart := time.Now() @@ -4147,11 +4159,20 @@ func ProcessBlock( return nil, fmt.Errorf("validate transaction messages for slot %d: %w", block.Slot, err) } statusValidationStart := time.Now() - statusValidationErr := transactionStatuses.validateBlockWithPlan(block, executionPlan) + statusValidation, statusValidationErr := transactionStatuses.validateBlockForPublication(block, executionPlan) metrics.GlobalBlockReplay.TransactionStatusValidation.AddTimingSince(statusValidationStart) if statusValidationErr != nil { return nil, fmt.Errorf("validate transaction statuses for slot %d: %w", block.Slot, statusValidationErr) } + statusPreparation := transactionStatuses.startStatusPreparation(executionPlan) + defer func() { + // Join before returning so a rejected bank cannot leave work behind or + // charge its preparation time to the next block's metrics record. + statusPreparation.wait() + if statusPreparation != nil { + metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) + } + }() ctx, task := trace.NewTask(context.Background(), "ProcessBlock") defer task.End() trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) @@ -4411,7 +4432,10 @@ func ProcessBlock( return slotCtx, err } statusCommitStart := time.Now() - statusErr := transactionStatuses.commitBlockWithPlan(block, executionPlan) + statusWaitStart := time.Now() + preparedStatuses := statusPreparation.wait() + metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) + statusErr := transactionStatuses.commitBlockWithValidation(block, executionPlan, preparedStatuses, statusValidation) metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) if statusErr != nil { return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) diff --git a/pkg/replay/chain_state.go b/pkg/replay/chain_state.go index 649146c93..70c404b22 100644 --- a/pkg/replay/chain_state.go +++ b/pkg/replay/chain_state.go @@ -266,3 +266,14 @@ func ChainTipFeatureActive(gate features.FeatureGate) bool { defer chainTipMu.RUnlock() return chainTipFeatures != nil && chainTipFeatures.IsActive(gate) } + +// ChainTipFeatures returns an independent feature snapshot for queued transaction +// preparation. Bank admission checks compatibility again before reuse. +func ChainTipFeatures() *features.Features { + chainTipMu.RLock() + defer chainTipMu.RUnlock() + if chainTipFeatures == nil { + return nil + } + return chainTipFeatures.Clone() +} diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index c750009ea..5472ec4ed 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -27,14 +27,13 @@ type batchCommitter interface { CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) } -// TransactionStatusCheckpointHooks deliberately split status-cache capture -// from sidecar I/O. Snapshot runs on the replay loop while its mutable cache is -// coherent; Install runs on the fold worker using only those immutable bytes. -// This makes it impossible for the async worker to traverse concurrently -// changing replay lineage. The later AccountsDB manifest remains the selector. +// TransactionStatusCheckpointHooks split immutable status capture from encoding +// and sidecar I/O. Capture runs on replay; the fold worker serializes the captured +// view and then calls Install. Neither worker operation revisits live lineage. +// The later AccountsDB manifest remains the durable checkpoint selector. type TransactionStatusCheckpointHooks struct { - Snapshot func(through uint64) ([]byte, error) - Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) + Capture func(through uint64) (TransactionStatusSnapshot, error) + Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) // AfterCommit is an advisory retention hook. It runs only after CommitBatch // has durably selected the manifest carrying selected. Its error is logged // and ignored: once CommitBatch succeeds, the fold must remain successful. @@ -378,7 +377,10 @@ type foldJob struct { ctx *state.ResumeContext ctxJSON []byte stakeIdxDir string - transactionStatusCheckpointPayload []byte + transactionStatusSnapshot TransactionStatusSnapshot + checkpointCaptureTime time.Duration + checkpointEncodeTime time.Duration + checkpointBytes int installTransactionStatusCheckpoint func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) afterTransactionStatusCheckpointCommit func(selected *state.TransactionStatusCheckpointRef) error } @@ -398,16 +400,13 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. if err != nil { return nil, err } - prefix := t.overlay.PromotionPrefix(through) - if len(prefix) == 0 { + // Check chunk eligibility before materializing account-write lists. Replay + // calls this on every iteration, including skipped slots; a partial batch + // remains in RAM without rescanning all of its accounts each time. + chunk := t.overlay.PromotionChunk(through, t.batchSlots, force) + if len(chunk) == 0 { return nil, nil } - chunk := prefix - if len(chunk) > t.batchSlots { - chunk = chunk[:t.batchSlots] - } else if len(chunk) < t.batchSlots && !force { - return nil, nil // trailing partial chunk stays in RAM - } through = chunk[len(chunk)-1].Slot ctx := t.contexts[through] @@ -415,18 +414,18 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. return nil, fmt.Errorf("fold chunk through slot %d: no resume context recorded for chunk-top slot", through) } ctx = cloneResumeContextForFold(ctx) - var checkpointPayload []byte - if hooks.Snapshot != nil { - checkpointPayload, err = hooks.Snapshot(through) + var snapshot TransactionStatusSnapshot + var captureTime time.Duration + if hooks.Capture != nil { + start := time.Now() + snapshot, err = hooks.Capture(through) + captureTime = time.Since(start) if err != nil { - return nil, fmt.Errorf("fold chunk through slot %d: snapshot transaction status checkpoint: %w", through, err) + return nil, fmt.Errorf("fold chunk through slot %d: capture transaction status checkpoint: %w", through, err) } - if len(checkpointPayload) == 0 { - return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint snapshot is empty", through) + if snapshot == nil { + return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint capture is nil", through) } - // The worker owns this immutable copy. Even a future Snapshot - // implementation that reuses a scratch buffer cannot race it. - checkpointPayload = append([]byte(nil), checkpointPayload...) } bankhashes := make(map[uint64][32]byte, len(chunk)) for _, sd := range chunk { @@ -440,7 +439,8 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. bankhashes: bankhashes, ctx: ctx, stakeIdxDir: t.stakeIdxDir, - transactionStatusCheckpointPayload: checkpointPayload, + transactionStatusSnapshot: snapshot, + checkpointCaptureTime: captureTime, installTransactionStatusCheckpoint: hooks.Install, afterTransactionStatusCheckpointCommit: hooks.AfterCommit, }, nil @@ -450,12 +450,26 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. // state). Stake-index entries flush (fsync'd) BEFORE the batch commit — see // promoteRootedBatched for why that order is a correctness requirement. func runFoldJob(committer batchCommitter, job *foldJob) error { - if job == nil || job.ctx == nil { + if job == nil { + return errors.New("fold job has no resume context") + } + // Failed folds are rebuilt from the retained tail. Neither a failed result + // nor a completed-but-unapplied job should keep checkpoint deltas alive. + defer func() { job.transactionStatusSnapshot = nil }() + if job.ctx == nil { return errors.New("fold job has no resume context") } var selectedCheckpoint *state.TransactionStatusCheckpointRef if job.installTransactionStatusCheckpoint != nil { - ref, err := job.installTransactionStatusCheckpoint(job.through, job.transactionStatusCheckpointPayload) + start := time.Now() + payload, err := encodeTransactionStatusCheckpoint(job.transactionStatusSnapshot) + job.checkpointEncodeTime = time.Since(start) + job.transactionStatusSnapshot = nil + if err != nil { + return fmt.Errorf("fold chunk through slot %d: encode transaction status checkpoint: %w", job.through, err) + } + job.checkpointBytes = len(payload) + ref, err := job.installTransactionStatusCheckpoint(job.through, payload) if err != nil { return fmt.Errorf("fold chunk through slot %d: prepare transaction status checkpoint: %w", job.through, err) } @@ -539,7 +553,9 @@ func (p *asyncPromoter) run() { start := time.Now() err := runFoldJob(p.committer, job) if err == nil { - mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s", len(job.chunk), job.through, time.Since(start).Round(time.Millisecond)) + mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s checkpoint_capture=%s checkpoint_encode=%s checkpoint_bytes=%d", + len(job.chunk), job.through, time.Since(start).Round(time.Millisecond), + job.checkpointCaptureTime, job.checkpointEncodeTime, job.checkpointBytes) } p.results <- foldResult{job: job, err: err} } @@ -713,14 +729,15 @@ func promoteRootedBatched( } ctx = cloneResumeContextForFold(ctx) var selectedCheckpoint *state.TransactionStatusCheckpointRef - if hooks.Snapshot != nil { - payload, serr := hooks.Snapshot(chunkThrough) + if hooks.Capture != nil { + snapshot, serr := hooks.Capture(chunkThrough) if serr != nil { - err = fmt.Errorf("promote chunk through slot %d: snapshot transaction status checkpoint: %w", chunkThrough, serr) + err = fmt.Errorf("promote chunk through slot %d: capture transaction status checkpoint: %w", chunkThrough, serr) break } - if len(payload) == 0 { - err = fmt.Errorf("promote chunk through slot %d: transaction status checkpoint snapshot is empty", chunkThrough) + payload, serr := encodeTransactionStatusCheckpoint(snapshot) + if serr != nil { + err = fmt.Errorf("promote chunk through slot %d: encode transaction status checkpoint: %w", chunkThrough, serr) break } ref, perr := hooks.Install(chunkThrough, payload) @@ -792,15 +809,29 @@ func resolveTransactionStatusCheckpointHooks(configured TransactionStatusCheckpo } func validateTransactionStatusCheckpointHooks(hooks TransactionStatusCheckpointHooks) error { - if (hooks.Snapshot == nil) != (hooks.Install == nil) { - return errors.New("transaction status checkpoint Snapshot and Install hooks must either both be set or both be nil") + if (hooks.Capture == nil) != (hooks.Install == nil) { + return errors.New("transaction status checkpoint Capture and Install hooks must either both be set or both be nil") } if hooks.AfterCommit != nil && hooks.Install == nil { - return errors.New("transaction status checkpoint AfterCommit hook requires Snapshot and Install hooks") + return errors.New("transaction status checkpoint AfterCommit hook requires Capture and Install hooks") } return nil } +func encodeTransactionStatusCheckpoint(snapshot TransactionStatusSnapshot) ([]byte, error) { + if snapshot == nil { + return nil, errors.New("transaction status checkpoint capture is nil") + } + payload, err := snapshot.MarshalBinary() + if err != nil { + return nil, err + } + if len(payload) == 0 { + return nil, errors.New("transaction status checkpoint snapshot is empty") + } + return payload, nil +} + func cloneResumeContextForFold(ctx *state.ResumeContext) *state.ResumeContext { if ctx == nil { return nil diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go new file mode 100644 index 000000000..a03a00cbf --- /dev/null +++ b/pkg/replay/rewards_retirement.go @@ -0,0 +1,49 @@ +package replay + +import ( + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" +) + +// partitionedRewardsCompletion is replay-thread-owned, process-local evidence +// that a successfully executed bank contains all effects of this distribution. +// It is not a checkpoint or signing authority. Until that bank is durable, +// tryInLoopUnwind must still reject even a zero-remaining distribution: its +// spool has been consumed and cannot be rolled back with the account overlay. +type partitionedRewardsCompletion struct { + info *rewards.PartitionedRewardDistributionInfo + slot uint64 +} + +// observeBank must run only after successful block execution/publication, using +// that bank's immutable sysvars (never the speculative global sysvar cache). +// If the first completed bank lacks evidence, recording a later descendant is +// conservative: retirement then waits for that later bank to become durable. +func (c *partitionedRewardsCompletion) observeBank(info *rewards.PartitionedRewardDistributionInfo, bank *sealevel.BankSysvars) { + if c.info != info { + *c = partitionedRewardsCompletion{info: info} + } + if info == nil || c.slot != 0 || info.NumRewardPartitionsRemaining != 0 || bank == nil || bank.Slot() == 0 { + return + } + epochRewards, ok := bank.EpochRewards() + if ok && !epochRewards.Active { + c.slot = bank.Slot() + } +} + +// retire is called only when replay applies a successfully committed fold and +// advances LastRootedSlot. Finality, an enqueued/in-flight fold, and a failed +// commit do not acknowledge durability. At this boundary every rewards effect +// is in AccountsDB; in-memory switches above it cannot undo distribution. +// Switches at/below it still take durable recovery, whose persisted +// EpochRewards validation remains unchanged. Restart loses this optional +// evidence and reconstructs state through the existing recovery path. +func (c *partitionedRewardsCompletion) retire(info **rewards.PartitionedRewardDistributionInfo, durableSlot uint64) bool { + if *info == nil || *info != c.info || c.slot == 0 || durableSlot < c.slot || (*info).NumRewardPartitionsRemaining != 0 { + return false + } + *info = nil + *c = partitionedRewardsCompletion{} + return true +} diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go new file mode 100644 index 000000000..460584eee --- /dev/null +++ b/pkg/replay/rewards_retirement_test.go @@ -0,0 +1,119 @@ +package replay + +import ( + "bytes" + "encoding/base64" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/state" + bin "github.com/gagliardetto/binary" + "github.com/mr-tron/base58" + "github.com/stretchr/testify/require" +) + +func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { + active := &sealevel.SysvarEpochRewards{Active: true} + var raw bytes.Buffer + require.NoError(t, active.MarshalWithEncoder(bin.NewBinEncoder(&raw))) + activeBank, err := sealevel.NewBankSysvars(5, &accounts.Account{Key: sealevel.SysvarEpochRewardsAddr, Data: raw.Bytes()}) + require.NoError(t, err) + missingBank, err := sealevel.NewBankSysvars(5) + require.NoError(t, err) + for _, tc := range []struct { + name string + remaining uint64 + bank *sealevel.BankSysvars + }{ + {"active distribution", 1, testUnwindBankSysvars(t, 5, 50)}, + {"active bank", 0, activeBank}, + {"missing bank", 0, nil}, + {"missing rewards", 0, missingBank}, + {"unknown slot", 0, testUnwindBankSysvars(t, 0, 50)}, + } { + t.Run(tc.name, func(t *testing.T) { + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: tc.remaining} + var completed partitionedRewardsCompletion + completed.observeBank(info, tc.bank) + require.False(t, completed.retire(&info, 100)) + require.NotNil(t, info) + }) + } + + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + require.False(t, completed.retire(&info, 100), "zero remaining without observed completion is insufficient") + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + require.False(t, completed.retire(&info, 4), "uncommitted completion must retain the guard") + require.True(t, completed.retire(&info, 5), "later observations must not postpone recorded completion") + require.Nil(t, info) + require.False(t, completed.retire(&info, 100), "retirement is one-shot") +} + +func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { + old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} + next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} + var completed partitionedRewardsCompletion + completed.observeBank(old, testUnwindBankSysvars(t, 5, 50)) + require.False(t, completed.retire(&next, 100), "old completion cannot retire new bookkeeping") + completed.observeBank(next, testUnwindBankSysvars(t, 11, 60)) + require.False(t, completed.retire(&next, 10)) + require.True(t, completed.retire(&next, 11)) +} + +func TestRewardsRetirementWaitsForSuccessfulFold(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 5} + tail := asyncTestTail(fc, 5, 6) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + job, err := tail.buildFoldJob(6, true) + require.NoError(t, err) + require.NotNil(t, job) + root := uint64(4) + require.False(t, completed.retire(&info, root), "capturing a job does not make its bank durable") + require.Error(t, runFoldJob(fc, job)) + require.False(t, completed.retire(&info, root), "a failed fold leaves the old durable root") + fc.failOn = 0 + require.NoError(t, runFoldJob(fc, job)) + ctx := tail.applyFoldJob(job) + require.NotNil(t, ctx) + root = job.through + require.True(t, completed.retire(&info, root)) +} + +func TestRewardsRetirementAllowsExactParentUnwind(t *testing.T) { + resetVoteStakeDirty() + t.Cleanup(resetVoteStakeDirty) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + parent := &state.ResumeContext{Slot: 7, Bankhash: base58.Encode(make([]byte, 32)), AcctsLtHash: base64.StdEncoding.EncodeToString(make([]byte, 2048)), Capitalization: 700} + bank := testUnwindBankSysvars(t, 7, 50) + tail.Add(7, []*accounts.Account{testAccount(1, 71)}, testHashBytes(7)) + tail.SetContext(7, parent, bank) + tail.Add(8, []*accounts.Account{testAccount(1, 81)}, testHashBytes(8)) + tail.SetContext(8, &state.ResumeContext{Slot: 8}, testUnwindBankSysvars(t, 8, 999)) + sw := &CertifiedSwitch{Slot: 8} + ms := &state.MithrilState{LastRootedSlot: 4} + sched := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000} + rs, _, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Nil(t, rs) + require.Equal(t, unwindFallbackRewardsWindow, reason) + ms.LastRootedSlot = 5 + markVoteStakeDirty(5) // completed reward writes are also below the durable root + require.True(t, completed.retire(&info, ms.LastRootedSlot)) + rs, restored, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Empty(t, reason) + require.Same(t, bank, restored, "use the surviving bank, never abandoned reward sysvars") + want, err := ResumeStateFromRootedContext(parent, nil) + require.NoError(t, err) + require.Equal(t, want, rs, "resume state must match rebuilding the exact retained parent") + acct, err := tail.GetAccount(8, testAccount(1, 0).Key) + require.NoError(t, err) + require.Equal(t, uint64(71), acct.Lamports, "abandoned account writes must be removed") +} diff --git a/pkg/replay/transaction_preparation.go b/pkg/replay/transaction_preparation.go new file mode 100644 index 000000000..303dcc3e3 --- /dev/null +++ b/pkg/replay/transaction_preparation.go @@ -0,0 +1,133 @@ +package replay + +import ( + "crypto/sha256" + "encoding/binary" + "sort" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/fees" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// TransactionPreparer binds static transaction preparation to an immutable bank +// feature snapshot. Its source must remain immutable for the bank's lifetime. +// Prepared messages must also remain immutable, including their backing bytes. +// Account contents, transaction age and duplicate status are never cached here. +type TransactionPreparer struct { + source *features.Features + feats *features.Features + key [32]byte +} + +// PreparedTransaction contains only message/feature-derived data. Private fields +// prevent callers from substituting an unchecked hash, cost or instruction list. +type PreparedTransaction struct { + tx *solana.Transaction + key [32]byte + hash [32]byte + cost costmodel.TransactionCost + instrs []sealevel.Instruction + instructionAccts [][]sealevel.InstructionAccount + accountMetas []*solana.AccountMeta + limits *sealevel.ComputeBudgetLimits +} + +func NewTransactionPreparer(f *features.Features) *TransactionPreparer { + if f == nil { + return nil + } + clone := f.Clone() + gates := make([]features.FeatureGate, 0, len(*clone)) + for gate := range *clone { + gates = append(gates, gate) + } + sort.Slice(gates, func(i, j int) bool { + if gates[i].Address != gates[j].Address { + return string(gates[i].Address[:]) < string(gates[j].Address[:]) + } + return gates[i].Name < gates[j].Name + }) + h := sha256.New() + var value [8]byte + for _, gate := range gates { + info := (*clone)[gate] + h.Write(gate.Address[:]) + binary.LittleEndian.PutUint64(value[:], uint64(len(gate.Name))) + h.Write(value[:]) + h.Write([]byte(gate.Name)) + if info.Enabled { + h.Write([]byte{1}) + } else { + h.Write([]byte{0}) + } + binary.LittleEndian.PutUint64(value[:], info.ActivationSlot) + h.Write(value[:]) + } + p := &TransactionPreparer{source: f, feats: clone} + copy(p.key[:], h.Sum(nil)) + return p +} + +// Prepare returns nil on a static validation failure. Callers retain their normal +// processing path in that case, preserving its error classification and ordering. +func (p *TransactionPreparer) Prepare(tx *solana.Transaction) *PreparedTransaction { + if p == nil || tx == nil { + return nil + } + if tx.Message.GetVersion() == solana.MessageVersionV1 && !p.feats.IsActive(features.EnableTxV1) { + return nil + } + if txverify.SanitizeTransaction(tx) != nil { + return nil + } + if p.feats.IsActive(features.StaticInstructionLimit) && len(tx.Message.Instructions) > maxInstrTraceCapacity { + return nil + } + instrs, instructionAccts, metas, err := instrsAndAcctMetasFromTx(tx, p.feats) + if err != nil { + return nil + } + limits, err := sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, p.feats) + if err != nil { + return nil + } + hash, err := TransactionMessageHash(tx) + if err != nil { + return nil + } + return &PreparedTransaction{tx: tx, key: p.key, hash: hash, + cost: costmodel.EstimatePreparedTransactionCost(tx, instrs, limits, p.feats), + instrs: instrs, instructionAccts: instructionAccts, accountMetas: metas, limits: limits} +} + +func (p *TransactionPreparer) Matches(prepared *PreparedTransaction, tx *solana.Transaction, f *features.Features) bool { + return p != nil && p.source == f && prepared != nil && prepared.tx == tx && p.key == prepared.key +} + +func (p *PreparedTransaction) MessageHash() [32]byte { return p.hash } + +// Cost returns the estimate; its writable-account slice is read-only. +func (p *PreparedTransaction) Cost() costmodel.TransactionCost { return p.cost } + +// LoadAndExecute reuses preparation only for the same immutable message and a +// matching feature snapshot. All bank-dependent checks still run on every call. +func (p *TransactionPreparer) LoadAndExecute(input LoadAndExecuteTransactionInput, prepared *PreparedTransaction) LoadAndExecuteTransactionOutput { + if input.SlotCtx == nil || p == nil || p.source != input.SlotCtx.Features || !p.Matches(prepared, input.Transaction, input.SlotCtx.Features) { + return LoadAndExecuteTransaction(input) + } + return loadAndExecuteTransaction(input, prepared) +} + +// PayerCanFund keeps strict leader admission, using current payer state while +// sharing the already-validated instructions and compute limits. +func (p *TransactionPreparer) PayerCanFund(slotCtx *sealevel.SlotCtx, tx *solana.Transaction, prepared *PreparedTransaction) error { + if slotCtx == nil || !p.Matches(prepared, tx, slotCtx.Features) { + return fees.PayerCanFund(slotCtx, tx) + } + _, err := fees.ValidateTransactionFeePayer(slotCtx, tx, prepared.instrs, prepared.limits) + return err +} diff --git a/pkg/replay/transaction_processing_pure.go b/pkg/replay/transaction_processing_pure.go index 061703db1..434346460 100644 --- a/pkg/replay/transaction_processing_pure.go +++ b/pkg/replay/transaction_processing_pure.go @@ -3,7 +3,6 @@ package replay import ( "errors" "math" - "time" "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/arena" @@ -39,6 +38,10 @@ type LoadAndExecuteTransactionInput struct { // banks also skip writable-account result materialization; RPC simulation // leaves this false to retain the rich result. LeanResult bool + // SkipTimingMetrics omits the detailed transaction and instruction-dispatch + // timings for leader execution. Replay and simulation retain their default + // instrumentation; program-specific instrumentation is independent. + SkipTimingMetrics bool // CapturePreBalances retains pre-fee balances in lean mode. Rich mode // always captures them for RPC compatibility. CapturePreBalances bool @@ -124,6 +127,10 @@ func feeOnlyRollbackAccountsDataSize(slotCtx *sealevel.SlotCtx, tx *solana.Trans } func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExecuteTransactionOutput { + return loadAndExecuteTransaction(input, nil) +} + +func loadAndExecuteTransaction(input LoadAndExecuteTransactionInput, prepared *PreparedTransaction) LoadAndExecuteTransactionOutput { tx := input.Transaction slotCtx := input.SlotCtx @@ -131,64 +138,76 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec input.Arena.Reset() } - if tx == nil || slotCtx == nil || slotCtx.Features == nil { - return sanitizeFailureOutput() - } + var instrs []sealevel.Instruction + var instructionAcctsPerInstr [][]sealevel.InstructionAccount + var txAcctMetas []*solana.AccountMeta + var computeBudgetLimits *sealevel.ComputeBudgetLimits + var err error + start := metrics.StartTiming(false) + if prepared != nil { + instrs, instructionAcctsPerInstr, txAcctMetas = prepared.instrs, prepared.instructionAccts, prepared.accountMetas + computeBudgetLimits = prepared.limits + } else { + if tx == nil || slotCtx == nil || slotCtx.Features == nil { + return sanitizeFailureOutput() + } - // Match Agave's Bank verification order: once a transaction has decoded as - // v1, the feature gate is checked before structural sanitization. - if tx.Message.GetVersion() == solana.MessageVersionV1 && !slotCtx.Features.IsActive(features.EnableTxV1) { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorUnsupportedVersion, - InstructionError: TxErrUnsupportedVersion, + // Match Agave's Bank verification order: once a transaction has decoded as + // v1, the feature gate is checked before structural sanitization. + if tx.Message.GetVersion() == solana.MessageVersionV1 && !slotCtx.Features.IsActive(features.EnableTxV1) { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorUnsupportedVersion, + InstructionError: TxErrUnsupportedVersion, + }, }, - }, + } + } + // Reject malformed transactions before account-indexed code can observe + // them. The helper also handles Mithril's already-resolved v0 messages. + if err := txverify.SanitizeTransaction(tx); err != nil { + return sanitizeFailureOutput() + } + // Mirror block-replay's StaticInstructionLimit cap so pre-activation + // clusters fail mid-execution like Agave instead of SanitizeFailure. + if slotCtx.Features.IsActive(features.StaticInstructionLimit) && + len(tx.Message.Instructions) > maxInstrTraceCapacity { + return sanitizeFailureOutput() } - } - // Reject malformed transactions before account-indexed code can observe - // them. The helper also handles Mithril's already-resolved v0 messages. - if err := txverify.SanitizeTransaction(tx); err != nil { - return sanitizeFailureOutput() - } - // Mirror block-replay's StaticInstructionLimit cap so pre-activation - // clusters fail mid-execution like Agave instead of SanitizeFailure. - if slotCtx.Features.IsActive(features.StaticInstructionLimit) && - len(tx.Message.Instructions) > maxInstrTraceCapacity { - return sanitizeFailureOutput() - } - // Parse instructions and account metas - start := time.Now() - instrs, instructionAcctsPerInstr, txAcctMetas, err := instrsAndAcctMetasFromTx(tx, slotCtx.Features) - if err != nil { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorSanitizeFailure, - InstructionError: err, + // Parse instructions and account metas + start = metrics.StartTiming(!input.SkipTimingMetrics) + instrs, instructionAcctsPerInstr, txAcctMetas, err = instrsAndAcctMetasFromTx(tx, slotCtx.Features) + if err != nil { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorSanitizeFailure, + InstructionError: err, + }, }, - }, + } } - } - metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.AddTimingSince(start) - - // Compute budget limits - start = time.Now() - computeBudgetLimits, err := sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, slotCtx.Features) - if err != nil { - return LoadAndExecuteTransactionOutput{ - ProcessingResult: TransactionProcessingResult{ - TransactionError: &TransactionError{ - ErrorType: TransactionErrorInstructionError, - InstructionError: err, + metrics.GlobalBlockReplay.InstructionsAndAccountMetasFromTx.AddTimingSince(start) + + // Compute budget limits + start = metrics.StartTiming(!input.SkipTimingMetrics) + computeBudgetLimits, err = sealevel.ComputeBudgetLimitsForTransaction(tx, instrs, slotCtx.Features) + if err != nil { + return LoadAndExecuteTransactionOutput{ + ProcessingResult: TransactionProcessingResult{ + TransactionError: &TransactionError{ + ErrorType: TransactionErrorInstructionError, + InstructionError: err, + }, }, - }, - Instrs: instrs, + Instrs: instrs, + } } + metrics.GlobalBlockReplay.ComputeBudgetExecutionInstructions.AddTimingSince(start) + } - metrics.GlobalBlockReplay.ComputeBudgetExecutionInstructions.AddTimingSince(start) // Validate transaction age if !sealevel.IsTransactionAgeValid(tx, instrs, slotCtx) { @@ -236,7 +255,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec } // Load and validate accounts - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) instructionsSysvarIdx := instructionsSysvarAccountIndex(tx) var instrsAcct *accounts.Account if instructionsSysvarIdx >= 0 { @@ -303,6 +322,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec execCtx.TransactionContext.Signature = tx.Signatures[0] execCtx.TransactionContext.BorrowedAccountArena = input.Arena execCtx.IsSimulation = input.IsSimulation + execCtx.SkipTimingMetrics = input.SkipTimingMetrics execCtx.RecordInnerInstructions = input.RecordInnerInstructions // Capture pre-balance lamports (before fee deduction) @@ -332,7 +352,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec // Calculate and deduct fees. RentForSlot supplies the exemption // minimum so a rent-exempt payer is rejected before instructions run. - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) txFeeInfo, _, err := fees.CalculateAndDeductTxFees(tx, input.TxMeta, instrs, &execCtx.TransactionContext.Accounts, computeBudgetLimits, slotCtx.Features, fees.RentForSlot(slotCtx), input.IsSimulation) if err != nil { errType, accountIndex := feePayerTransactionError(err) @@ -355,7 +375,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.CalcAndDeductFees.AddTimingSince(start) // Read rent sysvar - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) rentSysvar, err := sealevel.ReadRentSysvar(execCtx) if err != nil { // Rent sysvar unreadable; return cleanly so the RPC worker @@ -374,18 +394,18 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.ReadRentSysvar.AddTimingSince(start) // Set rent-exempt rent epoch max and compute pre-tx rent states - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) rent.MaybeSetRentExemptRentEpochMax(slotCtx, &rentSysvar, &execCtx.Features, &execCtx.TransactionContext.Accounts) preTxRentStates := rent.NewRentStateInfo(&rentSysvar, execCtx.TransactionContext, &execCtx.Features) metrics.GlobalBlockReplay.PreTxRentStates.AddTimingSince(start) // Execute all instructions var instrErr error - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) for instrIdx, instr := range tx.Message.Instructions { execCtx.SetCurrentTopLevelInstr(uint8(instrIdx)) if instructionsSysvarIdx >= 0 { - ixStart := time.Now() + ixStart := metrics.StartTiming(!input.SkipTimingMetrics) err = fixupInstructionsSysvarAcct(execCtx, instructionsSysvarIdx, uint16(instrIdx)) if err != nil { instrErr = err @@ -421,7 +441,7 @@ func LoadAndExecuteTransaction(input LoadAndExecuteTransactionInput) LoadAndExec metrics.GlobalBlockReplay.IxLoop.AddTimingSince(start) // Check rent state transitions - start = time.Now() + start = metrics.StartTiming(!input.SkipTimingMetrics) postTxRentStates := rent.NewRentStateInfo(&rentSysvar, execCtx.TransactionContext, &execCtx.Features) rentStateErr := rent.VerifyRentStateChanges(preTxRentStates, postTxRentStates, execCtx.TransactionContext) metrics.GlobalBlockReplay.PostTxRentStates.AddTimingSince(start) diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 8320df056..bb4a157fb 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -10,6 +10,7 @@ import ( "path/filepath" "sort" "sync" + "sync/atomic" b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/state" @@ -48,6 +49,38 @@ type transactionStatusNode struct { hasBlockID bool parent *transactionStatusNode delta transactionStatusDelta + + // Shared by parentless checkpoint copies and relinked retained nodes. + // Only the memoized encoding changes after publication; lineage and delta + // remain immutable. Never copy the atomic field after its first use. + encoding atomic.Pointer[transactionStatusNodeEncoding] +} + +type transactionStatusNodeEncoding struct { + once sync.Once + data []byte +} + +func (n *transactionStatusNode) encodingCache() *transactionStatusNodeEncoding { + if cache := n.encoding.Load(); cache != nil { + return cache + } + cache := new(transactionStatusNodeEncoding) + if n.encoding.CompareAndSwap(nil, cache) { + return cache + } + return n.encoding.Load() +} + +// copyInto initializes a fresh node, sharing its encoding without retaining +// excluded ancestry or copying a used synchronization primitive. The encoding excludes +// parent links and depends only on the immutable slot, block ID and delta. +func (n *transactionStatusNode) copyInto(copy *transactionStatusNode, parent *transactionStatusNode) { + *copy = transactionStatusNode{ + slot: n.slot, blockID: n.blockID, hasBlockID: n.hasBlockID, + parent: parent, delta: n.delta, + } + copy.encoding.Store(n.encodingCache()) } type visibleTransactionStatusGroup struct { @@ -72,6 +105,9 @@ type TransactionStatusCache struct { // from a known-empty genesis cache. Without this bit, completeness requires // the full 300 retained roots; a serialized boolean alone is not evidence. coverageFromGenesis bool + + // Protected by mu; see transactionStatusValidation. Never serialized. + validationVersion uint64 } // TransactionStatusView is an immutable view of one bank lineage. It lazily @@ -412,6 +448,7 @@ func (c *TransactionStatusCache) BindTipBlockID(slot uint64, blockID solana.Hash if c.tip.hasBlockID && c.tip.blockID != blockID { return fmt.Errorf("transaction status tip at slot %d has block id %s, cannot bind %s", slot, c.tip.blockID, blockID) } + c.invalidateValidationLocked() c.tip = &transactionStatusNode{ slot: slot, blockID: blockID, hasBlockID: true, parent: c.tip.parent, delta: c.tip.delta, @@ -515,25 +552,33 @@ func (c *TransactionStatusCache) ValidateBlock(block *b.Block) error { // validateBlockWithPlan preserves the status-cache checks while letting // replay reuse the exact immutable identities used for execution planning. func (c *TransactionStatusCache) validateBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + _, err := c.validateBlockForPublication(block, plan) + return err +} + +func (c *TransactionStatusCache) validateBlockForPublication(block *b.Block, plan blockTransactionExecutionPlan) (transactionStatusValidation, error) { if block == nil { - return errors.New("nil block") + return transactionStatusValidation{}, errors.New("nil block") } if plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { - return errors.New("prepared transaction message identities do not match block") + return transactionStatusValidation{}, errors.New("prepared transaction message identities do not match block") } if c == nil { - return &IncompleteTransactionStatusCoverageError{} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{} } c.mu.RLock() defer c.mu.RUnlock() if !c.coverageComplete { - return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } if err := c.validateParentLocked(block); err != nil { - return err + return transactionStatusValidation{}, err + } + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return transactionStatusValidation{}, err } - return c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities) + return transactionStatusValidation{cache: c, identities: plan.messageIdentities, version: c.validationVersion}, nil } func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { @@ -586,6 +631,14 @@ func (c *TransactionStatusCache) CommitBlock(block *b.Block) error { // commitBlockWithPlan atomically rechecks the mutable lineage/status state and // publishes the already-prepared immutable transaction identities. func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + return c.commitBlockWithPreparedDelta(block, plan, nil) +} + +func (c *TransactionStatusCache) commitBlockWithPreparedDelta(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta) error { + return c.commitBlockWithValidation(block, plan, prepared, transactionStatusValidation{}) +} + +func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta, validation transactionStatusValidation) error { if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { return errors.New("prepared transaction message identities do not match block") } @@ -594,33 +647,40 @@ func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockT if !c.coverageComplete { return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } - // Parent lineage and ancestor status are mutable, so both remain under the - // publication lock even when hashing and same-bank deduplication happened - // earlier. This keeps commit safe across a concurrent branch transition. + // Always check coverage, block binding and parent lineage. Reuse the earlier + // ancestor scan only under this lock and only for the same unchanged cache + // and immutable identities. A branch transition (including away and back) + // or root/prune invalidates it, requiring a fresh scan before publication. if err := c.validateParentLocked(block); err != nil { return err } - if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { - return err + if !validation.reusableForLocked(c, plan.messageIdentities) { + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } } - delta := make(transactionStatusDelta) - for index := 0; index < plan.messageIdentities.Len(); index++ { - identity := plan.messageIdentities.Identity(index) - blockhash := identity.RecentBlockhash - group := delta[blockhash] - if group == nil { - keyIndex := uint8(0) - if visible := c.visible[blockhash]; visible != nil { - keyIndex = visible.keyIndex + delta := transactionStatusDelta(nil) + if prepared != nil && prepared.identities == plan.messageIdentities { + delta = prepared.delta + // A restore or branch transition can change a blockhash's slice offset. + // Rebuild from full identities if any current group uses another offset. + for blockhash, group := range delta { + if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { + delta = nil + break } - group = &transactionStatusGroup{ - keyIndex: keyIndex, - keys: make(map[transactionStatusKey]struct{}), + } + } + if delta == nil { + counts := countTransactionStatusGroups(plan.messageIdentities) + indexes := make(map[solana.Hash]uint8, len(counts)) + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex } - delta[blockhash] = group } - group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + delta = buildTransactionStatusDelta(plan.messageIdentities, counts, indexes) } if err := c.addDeltaVisibleLocked(delta); err != nil { @@ -663,6 +723,7 @@ func (c *TransactionStatusCache) Root(through uint64) bool { } c.mu.Lock() defer c.mu.Unlock() + c.invalidateValidationLocked() wasComplete := c.coverageComplete newlyRooted := c.countNodesBetweenLocked(c.rootedThrough, through) if through > c.rootedThrough { @@ -681,10 +742,32 @@ func (c *TransactionStatusCache) Root(through uint64) bool { return !wasComplete && c.coverageComplete } -// SnapshotThrough serializes only the rooted lineage needed at through. It is -// called while constructing a fold job, so the blob rides in that exact durable -// manifest without being copied into every speculative ResumeContext. -func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { +// TransactionStatusSnapshot pins an immutable checkpoint view. MarshalBinary +// must use only captured data, without locking or revisiting the live cache, +// and return an owned payload. It can run on the checkpoint worker while replay +// commits, roots, or unwinds its current lineage. +type TransactionStatusSnapshot interface { + MarshalBinary() ([]byte, error) +} + +type transactionStatusSnapshot struct { + nodes []*transactionStatusNode + rootedSinceSeed uint16 + complete bool + coverageFromGenesis bool +} + +func (s *transactionStatusSnapshot) MarshalBinary() ([]byte, error) { + if s == nil { + return nil, nil + } + return marshalTransactionStatusNodes(s.nodes, s.rootedSinceSeed, s.complete, s.coverageFromGenesis) +} + +// CaptureSnapshotThrough selects the exact checkpoint lineage and coverage on +// replay, but leaves transaction-key sorting and serialization to the worker. +// Published deltas are immutable; only small node headers are copied here. +func (c *TransactionStatusCache) CaptureSnapshotThrough(through uint64) (TransactionStatusSnapshot, error) { if c == nil { return nil, nil } @@ -700,7 +783,27 @@ func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) if rootedSinceSeed > maxTransactionStatusRoots { rootedSinceSeed = maxTransactionStatusRoots } - return marshalTransactionStatusNodes(nodes, uint16(rootedSinceSeed), complete, c.coverageFromGenesis) + owned := make([]transactionStatusNode, len(nodes)) + pinned := make([]*transactionStatusNode, len(nodes)) + for i, node := range nodes { + // Do not keep the old parent chain or copy its atomic field. + node.copyInto(&owned[i], nil) + pinned[i] = &owned[i] + } + return &transactionStatusSnapshot{ + nodes: pinned, rootedSinceSeed: uint16(rootedSinceSeed), complete: complete, + coverageFromGenesis: c.coverageFromGenesis, + }, nil +} + +// SnapshotThrough is the synchronous convenience API. Serialization still +// happens after releasing the cache lock; normal folds use CaptureSnapshotThrough. +func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { + snapshot, err := c.CaptureSnapshotThrough(through) + if err != nil || snapshot == nil { + return nil, err + } + return snapshot.MarshalBinary() } func (c *TransactionStatusCache) processedSlotLocked(blockhash solana.Hash, key transactionStatusKey) uint64 { @@ -749,6 +852,7 @@ func (c *TransactionStatusCache) validateParentLocked(block *b.Block) error { } func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDelta) error { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", @@ -760,7 +864,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, - keys: make(map[transactionStatusKey]uint16), + keys: make(map[transactionStatusKey]uint16, len(deltaGroup.keys)), } c.visible[blockhash] = group } @@ -772,6 +876,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe } func (c *TransactionStatusCache) removeDeltaVisibleLocked(delta transactionStatusDelta) { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { group := c.visible[blockhash] if group == nil { @@ -839,20 +944,76 @@ func (c *TransactionStatusCache) pruneLocked(through uint64) { if drop <= 0 { return } - for _, node := range nodes[:drop] { - c.removeDeltaVisibleLocked(node.delta) - } retained := nodes[drop:] + c.expireVisibleLocked(nodes[:drop], retained) var parent *transactionStatusNode for _, old := range retained { - parent = &transactionStatusNode{ - slot: old.slot, blockID: old.blockID, hasBlockID: old.hasBlockID, - parent: parent, delta: old.delta, - } + next := new(transactionStatusNode) + old.copyInto(next, parent) + parent = next } c.tip = parent } +// expireVisibleLocked expires a whole rooted batch. Most old blockhash groups +// have no surviving bank and can be removed without visiting their transaction +// keys. For a group crossing the boundary, update whichever side is smaller. +// Immutable node deltas (including those pinned by producer views/checkpoints) +// are never mutated. Unrooted retained banks count as survivors too. +func (c *TransactionStatusCache) expireVisibleLocked(expired, retained []*transactionStatusNode) { + type groupExpiry struct { + expiredKeys int + retainedKeys int + survivors []*transactionStatusGroup + } + groups := make(map[solana.Hash]*groupExpiry) + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if g == nil { + g = &groupExpiry{} + groups[hash] = g + } + g.expiredKeys += len(delta.keys) + } + } + for _, node := range retained { + for hash, delta := range node.delta { + if g := groups[hash]; g != nil { + g.retainedKeys += len(delta.keys) + g.survivors = append(g.survivors, delta) + } + } + } + for hash, g := range groups { + if len(g.survivors) == 0 { + delete(c.visible, hash) + } else if g.retainedKeys < g.expiredKeys { + rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex, keys: make(map[transactionStatusKey]uint16)} + for _, delta := range g.survivors { + for key := range delta.keys { + rebuilt.keys[key]++ + } + } + c.visible[hash] = rebuilt + if len(rebuilt.keys) == 0 { + delete(c.visible, hash) + } + } + } + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if len(g.survivors) == 0 || g.retainedKeys < g.expiredKeys { + continue + } + // The existing removal path preserves reference counts for keys + // occurring in more than one retained/expired bank. + c.removeDeltaVisibleLocked(transactionStatusDelta{hash: delta}) + } + } +} + func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transactionStatusKey { // Match Agave's saturating_sub(CACHED_KEY_SIZE + 1), including its // deliberate exclusion of the final possible starting offset. @@ -867,7 +1028,16 @@ func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transaction } func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { - var buf bytes.Buffer + encoded := make([][]byte, len(nodes)) + size := 9 // magic, flags, rooted count and node count + for i, node := range nodes { + cache := node.encodingCache() + cache.once.Do(func() { cache.data = marshalTransactionStatusNode(node) }) + encoded[i] = cache.data + size += len(cache.data) + } + // Every caller owns its result. Never return or append into a cached slice. + buf := bytes.NewBuffer(make([]byte, 0, size)) buf.Write(transactionStatusSnapshotMagic[:]) flags := byte(0) if complete { @@ -877,47 +1047,61 @@ func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSe flags |= 2 } buf.WriteByte(flags) - _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) - _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) - for _, node := range nodes { - _ = binary.Write(&buf, binary.LittleEndian, node.slot) - nodeFlags := byte(0) - if node.hasBlockID { - nodeFlags = 1 - } - buf.WriteByte(nodeFlags) - if node.hasBlockID { - buf.Write(node.blockID[:]) - } - blockhashes := make([]solana.Hash, 0, len(node.delta)) - for blockhash := range node.delta { - blockhashes = append(blockhashes, blockhash) + _ = binary.Write(buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(buf, binary.LittleEndian, uint16(len(nodes))) + for _, data := range encoded { + buf.Write(data) + } + return buf.Bytes(), nil +} + +func marshalTransactionStatusNode(node *transactionStatusNode) []byte { + size := 8 + 1 + 4 // slot, flags and group count + if node.hasBlockID { + size += len(node.blockID) + } + for _, group := range node.delta { + size += 32 + 1 + 4 + transactionStatusKeySize*len(group.keys) + } + buf := bytes.NewBuffer(make([]byte, 0, size)) + _ = binary.Write(buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) } - sort.Slice(blockhashes, func(i, j int) bool { - return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) - for _, blockhash := range blockhashes { - group := node.delta[blockhash] - buf.Write(blockhash[:]) - buf.WriteByte(group.keyIndex) - keys := make([]transactionStatusKey, 0, len(group.keys)) - for key := range group.keys { - keys = append(keys, key) - } - sort.Slice(keys, func(i, j int) bool { - return bytes.Compare(keys[i][:], keys[j][:]) < 0 - }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) - for _, key := range keys { - buf.Write(key[:]) - } + _ = binary.Write(buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) } } - return buf.Bytes(), nil + return buf.Bytes() } func (c *TransactionStatusCache) restore(data []byte) error { + c.invalidateValidationLocked() reader := bytes.NewReader(data) var magic [4]byte if _, err := io.ReadFull(reader, magic[:]); err != nil { diff --git a/pkg/replay/transaction_status_capture_bench_test.go b/pkg/replay/transaction_status_capture_bench_test.go new file mode 100644 index 000000000..234743222 --- /dev/null +++ b/pkg/replay/transaction_status_capture_bench_test.go @@ -0,0 +1,113 @@ +package replay + +import ( + "crypto/sha256" + "encoding/binary" + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" +) + +var checkpointBenchmarkPayload []byte +var checkpointBenchmarkCapture TransactionStatusSnapshot + +func checkpointEncodingFixture() *TransactionStatusCache { + // A private, not-yet-published fixture with the same complete 300-root + // metadata as an imported cache. 1.5 million keys encode to roughly 30 MB. + c := newTransactionStatusCache(true) + c.coverageFromGenesis = false + c.rootedSinceSeed = maxTransactionStatusRoots + c.rootedThrough = maxTransactionStatusRoots + for slot := uint64(1); slot <= maxTransactionStatusRoots; slot++ { + keys := make(map[transactionStatusKey]struct{}, 5000) + var seed [16]byte + binary.LittleEndian.PutUint64(seed[:8], slot) + for i := uint64(0); i < 5000; i++ { + binary.LittleEndian.PutUint64(seed[8:], i) + hash := sha256.Sum256(seed[:]) + var key transactionStatusKey + copy(key[:], hash[:]) + keys[key] = struct{}{} + } + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: keys}}} + } + return c +} + +func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { + c := checkpointEncodingFixture() + view, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + b.Run("SynchronousBaseline", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = legacyStatusSnapshotForTest(c, maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("CaptureOnReplay", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkCapture, err = c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("EncodeOnWorker", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = view.MarshalBinary() + if err != nil { + b.Fatal(err) + } + } + }) +} + +// Moving 300-root windows at several checkpoint cadences. Fixtures and initial +// cache warming are excluded; allocating new node headers and encoding/output +// allocation are included. This measures encoding only, not fsync or account +// checkpoint work. Cold encodes represent first startup/all-new windows. +func BenchmarkTransactionStatusCheckpointEncoding(b *testing.B) { + c := checkpointEncodingFixture() + view, err := c.CaptureSnapshotThrough(300) + if err != nil { + b.Fatal(err) + } + seed := view.(*transactionStatusSnapshot).nodes + for _, advance := range []int{0, 1, 8, 32, defaultFoldBatchSlots, 300} { + for _, cached := range []bool{false, true} { + b.Run(fmt.Sprintf("new=%d/cached=%t", advance, cached), func(b *testing.B) { + nodes := append([]*transactionStatusNode(nil), seed...) + if cached { + _, _ = marshalTransactionStatusNodes(nodes, 300, true, false) + } + b.ReportAllocs() + b.ResetTimer() + for n := 0; n < b.N; n++ { + copy(nodes, nodes[advance:]) + for i := 300 - advance; i < 300; i++ { + nodes[i] = &transactionStatusNode{slot: uint64(301 + n*advance + i), delta: seed[i].delta} + } + var err error + if cached { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodes(nodes, 300, true, false) + } else { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodesUncached(nodes, 300, true, false) + } + if err != nil { + b.Fatal(err) + } + } + b.SetBytes(int64(len(checkpointBenchmarkPayload))) + }) + } + } +} diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go new file mode 100644 index 000000000..8e1344c6f --- /dev/null +++ b/pkg/replay/transaction_status_capture_test.go @@ -0,0 +1,299 @@ +package replay + +import ( + "bytes" + "encoding/binary" + "fmt" + "sort" + "sync" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Keep the pre-split selection/metadata calculation as a differential oracle. +// Use the original uncached wire encoder to check byte-for-byte compatibility. +func legacyStatusSnapshotForTest(c *TransactionStatusCache, through uint64) ([]byte, error) { + c.mu.RLock() + defer c.mu.RUnlock() + nodes := c.nodesThroughLocked(through) + if len(nodes) > maxTransactionStatusRoots { + nodes = nodes[len(nodes)-maxTransactionStatusRoots:] + } + rooted := uint32(c.rootedSinceSeed) + uint32(c.countNodesBetweenLocked(c.rootedThrough, through)) + complete := c.coverageComplete || rooted >= maxTransactionStatusRoots + if rooted > maxTransactionStatusRoots { + rooted = maxTransactionStatusRoots + } + return marshalTransactionStatusNodesUncached(nodes, uint16(rooted), complete, c.coverageFromGenesis) +} + +func importedStatusCacheForTest(t *testing.T) *TransactionStatusCache { + t.Helper() + roots := make([]txstatus.SnapshotSlotDelta, maxTransactionStatusRoots) + for i := range roots { + roots[i] = txstatus.SnapshotSlotDelta{Slot: uint64(i + 1), IsRoot: true} + } + c, err := NewTransactionStatusCacheFromAgaveSnapshot(roots, maxTransactionStatusRoots) + require.NoError(t, err) + return c +} + +func captureTestBlock(slot uint64, branch byte) *b.Block { + tx := statusCacheTestTransaction(1, 2, branch) + data := make([]byte, 9) + binary.LittleEndian.PutUint64(data, slot) + data[8] = branch + tx.Message.Instructions[0].Data = data + return statusCacheTestBlock(slot, tx) +} + +func TestTransactionStatusCaptureSurvivesConcurrentPruneAndUnwind(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + captured, err := c.CaptureSnapshotThrough(320) + require.NoError(t, err) + view := captured.(*transactionStatusSnapshot) + require.Len(t, view.nodes, maxTransactionStatusRoots) + require.Equal(t, uint64(21), view.nodes[0].slot) + require.Equal(t, uint64(320), view.nodes[len(view.nodes)-1].slot) + for _, node := range view.nodes { + require.Nil(t, node.parent, "capture retained excluded ancestry") + } + + var wg sync.WaitGroup + wg.Add(1) + errs := make(chan error, 1) + go func() { + defer wg.Done() + for slot := uint64(351); slot <= 750; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 1)); err != nil { + errs <- err + return + } + c.Root(slot - 20) + } + if err := c.Unwind(741); err != nil { + errs <- err + return + } + for slot := uint64(741); slot <= 755; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 2)); err != nil { + errs <- err + return + } + } + }() + for i := 0; i < 50; i++ { + got, err := captured.MarshalBinary() + if err != nil || string(want) != string(got) { + t.Errorf("captured bytes changed during replay: %v", err) + break + } + } + wg.Wait() + close(errs) + for err := range errs { + require.NoError(t, err) + } + got, err := captured.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) + retry := statusCacheTestBlock(321, captureTestBlock(301, 1).Transactions[0]) + require.Error(t, restored.ValidateBlock(retry), "captured ancestor was forgotten") + // A bank after the capture's through-slot must not leak into recovery. + future := statusCacheTestBlock(321, captureTestBlock(350, 1).Transactions[0]) + require.NoError(t, restored.ValidateBlock(future)) +} + +func TestTransactionStatusCapturePreservesCoverageAndOwnedBytes(t *testing.T) { + for _, complete := range []bool{false, true} { + c := newTransactionStatusCache(complete) + // Exercise metadata selection without changing its pre-existing rules. + for slot := uint64(1); slot <= 310; slot++ { + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: map[transactionStatusKey]struct{}{{byte(slot), byte(slot >> 8)}: {}}}}} + } + for _, through := range []uint64{0, 1, 299, 300, 310, 400} { + want, err := legacyStatusSnapshotForTest(c, through) + require.NoError(t, err) + view, err := c.CaptureSnapshotThrough(through) + require.NoError(t, err) + got, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + got[0] ^= 0xff + again, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, again, "caller mutated the captured data through encoded bytes") + } + } + var absent *TransactionStatusCache + view, err := absent.CaptureSnapshotThrough(1) + require.NoError(t, err) + require.Nil(t, view) +} + +func TestTransactionStatusCaptureEncodingDoesNotLockLiveCache(t *testing.T) { + c := importedStatusCacheForTest(t) + require.NoError(t, c.CommitBlock(captureTestBlock(301, 1))) + view, err := c.CaptureSnapshotThrough(301) + require.NoError(t, err) + c.mu.Lock() + done := make(chan error, 1) + go func() { _, err := view.MarshalBinary(); done <- err }() + select { + case err := <-done: + c.mu.Unlock() + require.NoError(t, err) + case <-time.After(2 * time.Second): + c.mu.Unlock() + t.Fatal("checkpoint encoding waited for the live cache lock") + } +} + +func marshalTransactionStatusNodesUncached(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { + var buf bytes.Buffer + buf.Write(transactionStatusSnapshotMagic[:]) + flags := byte(0) + if complete { + flags = 1 + } + if coverageFromGenesis { + flags |= 2 + } + buf.WriteByte(flags) + _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) + for _, node := range nodes { + _ = binary.Write(&buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) + } + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) + } + } + } + return buf.Bytes(), nil +} + +func TestTransactionStatusEncodingSharedAcrossCaptureAndPrune(t *testing.T) { + for _, warmBeforePrune := range []bool{false, true} { + t.Run(fmt.Sprintf("warm=%t", warmBeforePrune), func(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 305; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + first, err := c.CaptureSnapshotThrough(304) + require.NoError(t, err) + pinned := first.(*transactionStatusSnapshot) + want, err := legacyStatusSnapshotForTest(c, 304) + require.NoError(t, err) + if warmBeforePrune { + got, err := first.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + } + // Force relinking of retained nodes after the snapshot has copied + // their headers, including the still-unencoded case. + c.Root(305) + second, err := c.CaptureSnapshotThrough(305) + require.NoError(t, err) + current := second.(*transactionStatusSnapshot) + caches := make(map[uint64]*transactionStatusNodeEncoding) + for _, node := range pinned.nodes { + caches[node.slot] = node.encodingCache() + } + for _, node := range current.nodes { + if prior := caches[node.slot]; prior != nil { + require.Same(t, prior, node.encodingCache()) + } + } + var wg sync.WaitGroup + for i := 0; i < 8; i++ { + wg.Add(1) + go func() { + defer wg.Done() + got, err := first.MarshalBinary() + assert.NoError(t, err) + assert.Equal(t, want, got) + // Mutate the node body as well as the header; neither may + // alias the memoized node data or another caller's result. + clear(got) + }() + } + wg.Wait() + currentWant, err := legacyStatusSnapshotForTest(c, 305) + require.NoError(t, err) + got, err := second.MarshalBinary() + require.NoError(t, err) + require.Equal(t, currentWant, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + roundTrip, err := restored.SnapshotThrough(305) + require.NoError(t, err) + require.Equal(t, got, roundTrip) + }) + } +} + +func TestTransactionStatusEncodingMatchesOriginalWireFormat(t *testing.T) { + // Deliberately unsorted groups/keys, nonzero offsets, empty deltas and + // mixed block-ID presence exercise every independently cached field. + nodes := []*transactionStatusNode{ + {slot: 3, hasBlockID: true, blockID: solana.Hash{9}, delta: transactionStatusDelta{ + solana.Hash{7}: {keyIndex: 11, keys: map[transactionStatusKey]struct{}{{8}: {}, {1}: {}, {4}: {}}}, + solana.Hash{1}: {keyIndex: 2, keys: map[transactionStatusKey]struct{}{{9}: {}, {2}: {}}}, + }}, + {slot: 5}, + {slot: 8, delta: transactionStatusDelta{solana.Hash{3}: {keyIndex: 0, keys: map[transactionStatusKey]struct{}{}}}}, + } + for _, complete := range []bool{false, true} { + for _, genesis := range []bool{false, true} { + want, err := marshalTransactionStatusNodesUncached(nodes, 3, complete, genesis) + require.NoError(t, err) + got, err := marshalTransactionStatusNodes(nodes, 3, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got) + } + } +} diff --git a/pkg/replay/transaction_status_expiry_test.go b/pkg/replay/transaction_status_expiry_test.go new file mode 100644 index 000000000..84a11cd28 --- /dev/null +++ b/pkg/replay/transaction_status_expiry_test.go @@ -0,0 +1,129 @@ +package replay + +import ( + "encoding/binary" + "fmt" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +func TestTransactionStatusBatchExpiryMatchesPerKeyRemoval(t *testing.T) { + for seed := int64(0); seed < 100; seed++ { + rng := rand.New(rand.NewSource(seed)) + fast, ref := NewTransactionStatusCache(), NewTransactionStatusCache() + var nodes []*transactionStatusNode + for slot := 0; slot < 40; slot++ { + d := make(transactionStatusDelta) + for j := 0; j < 6; j++ { + h := solana.Hash{byte(rng.Intn(12))} + g := &transactionStatusGroup{keyIndex: h[0], keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < rng.Intn(30); k++ { + g.keys[transactionStatusKey{byte(rng.Intn(40))}] = struct{}{} + } + d[h] = g + } + nodes = append(nodes, &transactionStatusNode{slot: uint64(slot), delta: d}) + require.NoError(t, fast.addDeltaVisibleLocked(d)) + require.NoError(t, ref.addDeltaVisibleLocked(d)) + } + cut := 1 + rng.Intn(len(nodes)-1) + fast.expireVisibleLocked(nodes[:cut], nodes[cut:]) + for _, n := range nodes[:cut] { + ref.removeDeltaVisibleLocked(n.delta) + } + require.Equal(t, ref.visible, fast.visible, "seed %d", seed) + for i := len(nodes) - 1; i >= cut; i-- { + fast.removeDeltaVisibleLocked(nodes[i].delta) + ref.removeDeltaVisibleLocked(nodes[i].delta) + } + require.Equal(t, ref.visible, fast.visible, "unwind seed %d", seed) + } +} + +func TestTransactionStatusBatchExpiryPinnedViewsAndSnapshot(t *testing.T) { + c := NewTransactionStatusCache() + old := statusCacheTestTransaction(1, 1, 1) + keep := statusCacheTestTransaction(2, 2, 2) + require.NoError(t, c.CommitBlock(statusCacheTestBlock(1, old))) + for slot := uint64(2); slot <= maxTransactionStatusRoots+1; slot++ { + blk := statusCacheTestBlock(slot) + if slot == maxTransactionStatusRoots+1 { + blk.Transactions = append(blk.Transactions, keep) + } + require.NoError(t, c.CommitBlock(blk)) + } + pinned := c.View() + snapshot, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + before, err := snapshot.MarshalBinary() + require.NoError(t, err) + c.coverageComplete = false // Exercise completion once 300 banks become rooted. + c.Root(maxTransactionStatusRoots + 1) + after, err := snapshot.MarshalBinary() + require.NoError(t, err) + require.Equal(t, before, after) + found, err := pinned.ContainsTransaction(old) + require.NoError(t, err) + require.True(t, found) + found, err = c.View().ContainsTransaction(old) + require.NoError(t, err) + require.False(t, found) + require.NoError(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + blob, err := c.SnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + require.NoError(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + require.Error(t, c.Unwind(maxTransactionStatusRoots+1)) +} + +func BenchmarkTransactionStatusBatchExpiry(b *testing.B) { + for _, shape := range []string{"four-bank-groups", "one-expired-group", "crossing-group"} { + for _, legacy := range []bool{true, false} { + b.Run(fmt.Sprintf("%s/legacy=%t", shape, legacy), func(b *testing.B) { + for i := 0; i < b.N; i++ { + b.StopTimer() + c := NewTransactionStatusCache() + var expired, retained []*transactionStatusNode + for slot := 0; slot < 129; slot++ { + var h solana.Hash + if shape == "four-bank-groups" || (shape == "one-expired-group" && slot == 128) { + binary.LittleEndian.PutUint64(h[:], uint64(slot/4+1)) + } + g := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < 33760; k++ { + var key transactionStatusKey + binary.LittleEndian.PutUint64(key[:], uint64(slot*33760+k)) + g.keys[key] = struct{}{} + } + n := &transactionStatusNode{delta: transactionStatusDelta{h: g}} + if err := c.addDeltaVisibleLocked(n.delta); err != nil { + b.Fatal(err) + } + if slot < 128 { + expired = append(expired, n) + } else { + retained = append(retained, n) + } + } + b.StartTimer() + if legacy { + for _, n := range expired { + c.removeDeltaVisibleLocked(n.delta) + } + } else { + c.expireVisibleLocked(expired, retained) + } + b.StopTimer() + if len(c.visible) != 1 { + b.Fatal("retained group missing") + } + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_overlap_benchmark_test.go b/pkg/replay/transaction_status_overlap_benchmark_test.go new file mode 100644 index 000000000..0da5f739f --- /dev/null +++ b/pkg/replay/transaction_status_overlap_benchmark_test.go @@ -0,0 +1,110 @@ +package replay + +import ( + "fmt" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" +) + +// This controlled workload runs 4,096 executions of the transfer fixture while +// preparing 33,760 independent status keys. It measures scheduling/GC contention, +// not full replay: no accounts are committed, and the status fixture differs +// from the repeated transfer fixture. Run with -cpu=1,2 to compare contention +// without and with a spare execution thread. Check live replay separately. +func BenchmarkTransactionStatusExecutionOverlap(tb *testing.B) { + for _, mode := range []string{"legacy", "sized", "overlap"} { + tb.Run(mode, func(tb *testing.B) { + slotCtx, cleanup := newCommitTestSlotCtx() + defer cleanup() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(0)) + if err != nil { + tb.Fatal(err) + } + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + var execution, commit time.Duration + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var p *transactionStatusPreparation + if mode == "overlap" { + p = cache.startStatusPreparation(plan) + } + start := time.Now() + for range 4096 { + output := LoadAndExecuteTransaction(LoadAndExecuteTransactionInput{SlotCtx: slotCtx, Transaction: tx, LeanResult: true}) + if output.ProcessingResult.TransactionError != nil { + tb.Fatal(output.ProcessingResult.TransactionError) + } + } + execution += time.Since(start) + start = time.Now() + switch mode { + case "legacy": + err = cache.legacyCommitStatusForBenchmark(blk, plan) + case "sized": + err = cache.commitBlockWithPlan(blk, plan) + case "overlap": + err = cache.commitBlockWithPreparedDelta(blk, plan, p.wait()) + } + commit += time.Since(start) + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + tb.StopTimer() + tb.ReportMetric(float64(execution.Nanoseconds())/float64(tb.N), "execution-ns/op") + tb.ReportMetric(float64(commit.Nanoseconds())/float64(tb.N), "commit-with-wait-ns/op") + }) + } +} + +func BenchmarkTransactionStatusSmallPublication(tb *testing.B) { + for _, count := range []int{0, 1, 32} { + for _, mode := range []string{"legacy", "prepared_total"} { + tb.Run(fmt.Sprintf("txs_%d/%s", count, mode), func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + if mode == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPreparedDelta(blk, plan, cache.startStatusPreparation(plan).wait()) + } + if err != nil { + tb.Fatal(err) + } + // Include unwind equally in this small-work benchmark, avoiding + // timer start/stop overhead around microsecond operations. + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_plan_binding_test.go b/pkg/replay/transaction_status_plan_binding_test.go index 30b456d3a..faf0b2540 100644 --- a/pkg/replay/transaction_status_plan_binding_test.go +++ b/pkg/replay/transaction_status_plan_binding_test.go @@ -15,9 +15,10 @@ func TestPreparedCommitRejectsTransactionReplacement(t *testing.T) { if err != nil { t.Fatal(err) } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) candidate.Transactions[0] = statusCacheTestTransaction(4, 5, 6) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithPreparedDelta(candidate, plan, prepared) if err == nil || err.Error() != "prepared transaction message identities do not match block" { t.Fatalf("commit error = %v, want prepared-plan binding failure", err) } diff --git a/pkg/replay/transaction_status_prepared_test.go b/pkg/replay/transaction_status_prepared_test.go index 44a42ca83..0172fffa6 100644 --- a/pkg/replay/transaction_status_prepared_test.go +++ b/pkg/replay/transaction_status_prepared_test.go @@ -25,7 +25,9 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { candidate := statusCacheTestBlock(12, retried, unique) plan, err := planBlockTransactionExecution(candidate) requireNoError(err) - requireNoError(cache.validateBlockWithPlan(candidate, plan)) + validation, err := cache.validateBlockForPublication(candidate, plan) + requireNoError(err) + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) requireNoError(cache.Unwind(11)) replacement := statusCacheTestBlock( @@ -34,7 +36,7 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { ) requireNoError(cache.CommitBlock(replacement)) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithValidation(candidate, plan, prepared, validation) var ancestorErr *AncestorAlreadyProcessedTransactionMessagesError if !errors.As(err, &ancestorErr) { t.Fatalf("prepared commit error = %v, want ancestor AlreadyProcessed", err) @@ -76,22 +78,26 @@ func TestConcurrentPreparedSiblingCommitsPublishExactlyOne(t *testing.T) { if err != nil { t.Fatal(err) } - if err := cache.validateBlockWithPlan(left, leftPlan); err != nil { + leftValidation, err := cache.validateBlockForPublication(left, leftPlan) + if err != nil { t.Fatalf("prevalidate left sibling: %v", err) } - if err := cache.validateBlockWithPlan(right, rightPlan); err != nil { + rightValidation, err := cache.validateBlockForPublication(right, rightPlan) + if err != nil { t.Fatalf("prevalidate right sibling: %v", err) } + leftPrepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + rightPrepared := cache.prepareTransactionStatusDelta(rightPlan.messageIdentities) start := make(chan struct{}) results := make(chan error, 2) go func() { <-start - results <- cache.commitBlockWithPlan(left, leftPlan) + results <- cache.commitBlockWithValidation(left, leftPlan, leftPrepared, leftValidation) }() go func() { <-start - results <- cache.commitBlockWithPlan(right, rightPlan) + results <- cache.commitBlockWithValidation(right, rightPlan, rightPrepared, rightValidation) }() close(start) diff --git a/pkg/replay/transaction_status_publication.go b/pkg/replay/transaction_status_publication.go new file mode 100644 index 000000000..e821071ad --- /dev/null +++ b/pkg/replay/transaction_status_publication.go @@ -0,0 +1,83 @@ +package replay + +import ( + "runtime" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +// Preparation owns private, immutable maps. It never publishes a status or +// authorizes a bank: commit still checks coverage, lineage, and duplicates. +type preparedTransactionStatusDelta struct { + identities *b.PreparedTransactionMessageIdentities + delta transactionStatusDelta +} + +type transactionStatusPreparation struct { + done chan struct{} + prepared *preparedTransactionStatusDelta + duration time.Duration +} + +// Replay joins this task on every exit, including rejected banks. It only reads +// the immutable identities, so account loading and ALT resolution can proceed. +func (c *TransactionStatusCache) startStatusPreparation(plan blockTransactionExecutionPlan) *transactionStatusPreparation { + // Small-block measurements show dispatch/join costs as much as the work. + // With one Go execution thread preparation cannot overlap execution at all. + if plan.messageIdentities.Len() <= 32 || runtime.GOMAXPROCS(0) == 1 { + return nil + } + p := &transactionStatusPreparation{done: make(chan struct{})} + go func() { + defer close(p.done) + start := time.Now() + p.prepared = c.prepareTransactionStatusDelta(plan.messageIdentities) + p.duration = time.Since(start) + }() + return p +} + +func (p *transactionStatusPreparation) wait() *preparedTransactionStatusDelta { + if p == nil { + return nil + } + <-p.done + return p.prepared +} + +func countTransactionStatusGroups(identities *b.PreparedTransactionMessageIdentities) map[solana.Hash]int { + counts := make(map[solana.Hash]int) + for i := 0; i < identities.Len(); i++ { + counts[identities.Identity(i).RecentBlockhash]++ + } + return counts +} + +func buildTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8) transactionStatusDelta { + delta := make(transactionStatusDelta, len(counts)) + for blockhash, count := range counts { + delta[blockhash] = &transactionStatusGroup{keyIndex: indexes[blockhash], keys: make(map[transactionStatusKey]struct{}, count)} + } + for i := 0; i < identities.Len(); i++ { + identity := identities.Identity(i) + group := delta[identity.RecentBlockhash] + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + return delta +} + +func (c *TransactionStatusCache) prepareTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities) *preparedTransactionStatusDelta { + counts := countTransactionStatusGroups(identities) + indexes := make(map[solana.Hash]uint8, len(counts)) + // Copy offsets only, never share mutable visible maps with the worker. + c.mu.RLock() + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex + } + } + c.mu.RUnlock() + return &preparedTransactionStatusDelta{identities: identities, delta: buildTransactionStatusDelta(identities, counts, indexes)} +} diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go new file mode 100644 index 000000000..2fc7d0f45 --- /dev/null +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -0,0 +1,166 @@ +package replay + +import ( + "encoding/binary" + "errors" + "fmt" + "testing" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, plan blockTransactionExecutionPlan) error { + if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { + return errors.New("prepared transaction message identities do not match block") + } + c.mu.Lock() + defer c.mu.Unlock() + if !c.coverageComplete { + return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + } + // Parent lineage and ancestor status are mutable, so both remain under the + // publication lock even when hashing and same-bank deduplication happened + // earlier. This keeps commit safe across a concurrent branch transition. + if err := c.validateParentLocked(block); err != nil { + return err + } + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } + + delta := make(transactionStatusDelta) + for index := 0; index < plan.messageIdentities.Len(); index++ { + identity := plan.messageIdentities.Identity(index) + blockhash := identity.RecentBlockhash + group := delta[blockhash] + if group == nil { + keyIndex := uint8(0) + if visible := c.visible[blockhash]; visible != nil { + keyIndex = visible.keyIndex + } + group = &transactionStatusGroup{ + keyIndex: keyIndex, + keys: make(map[transactionStatusKey]struct{}), + } + delta[blockhash] = group + } + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + + if err := c.legacyAddStatusForBenchmark(delta); err != nil { + return err + } + c.tip = &transactionStatusNode{ + slot: block.Slot, + blockID: solana.Hash(block.AlpenglowBlockID), + hasBlockID: block.HasAlpenglowBlockID, + parent: c.tip, + delta: delta, + } + return nil +} + +// Frozen production commit algorithm before publication optimization. This is +// an independent baseline, including its original visible-index allocation. +func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { + for blockhash, deltaGroup := range delta { + if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { + return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", + blockhash, group.keyIndex, deltaGroup.keyIndex) + } + } + for blockhash, deltaGroup := range delta { + group := c.visible[blockhash] + if group == nil { + group = &visibleTransactionStatusGroup{ + keyIndex: deltaGroup.keyIndex, + keys: make(map[transactionStatusKey]uint16), + } + c.visible[blockhash] = group + } + for key := range deltaGroup.keys { + group.keys[key]++ + } + } + return nil +} + +// BenchmarkTransactionStatusPublication times only status publication, with +// prepared message identities. No execution, disk I/O, signing or networking. +// prepared_commit excludes delta preparation; prepared_total includes it and +// goroutine dispatch/join, with no execution overlap. Neither measures replay. +// validated_commit also excludes the successful pre-execution ancestor scan. +// invalidated_commit roots between validation and commit, forcing a full recheck. +// Each iteration restores the same ancestor contents; existing maps retain +// steady-state capacity. Fixture creation, seeding and unwind are not timed. +func BenchmarkTransactionStatusPublication(tb *testing.B) { + const count = 33760 + for _, groups := range []int{1, 4} { + for _, existing := range []bool{false, true} { + tb.Run(fmt.Sprintf("groups_%d/existing_%t", groups, existing), func(tb *testing.B) { + txs := benchmarkUniqueTransactions(count * 2) + for i, tx := range txs { + binary.LittleEndian.PutUint32(tx.Message.RecentBlockhash[:], uint32(i%groups+1)) + } + parent := statusCacheTestBlock(10, txs[:count]...) + if !existing { + parent.Transactions = nil + } + blk := statusCacheTestBlock(11, txs[count:]...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + for _, name := range []string{"legacy", "sized", "prepared_total", "prepared_commit", "validated_commit", "invalidated_commit"} { + tb.Run(name, func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(parent); err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var err error + if name == "prepared_commit" || name == "validated_commit" || name == "invalidated_commit" { + tb.StopTimer() + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + validation, validationErr := cache.validateBlockForPublication(blk, plan) + if validationErr != nil { + tb.Fatal(validationErr) + } + if name == "invalidated_commit" { + cache.Root(10) + } + tb.StartTimer() + if name == "prepared_commit" { + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else { + err = cache.commitBlockWithValidation(blk, plan, prepared, validation) + } + } else if name == "prepared_total" { + prepared := cache.startStatusPreparation(plan).wait() + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else if name == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPlan(blk, plan) + } + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if got := cache.tip.slot; got != 11 { + tb.Fatalf("tip=%d", got) + } + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + }) + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_publication_test.go b/pkg/replay/transaction_status_publication_test.go new file mode 100644 index 000000000..58a24b06f --- /dev/null +++ b/pkg/replay/transaction_status_publication_test.go @@ -0,0 +1,133 @@ +package replay + +import ( + "fmt" + "runtime" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/stretchr/testify/require" +) + +func TestPreparedStatusDeltaRebindsSnapshotOffsets(t *testing.T) { + for _, from := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + for _, to := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + t.Run(fmt.Sprintf("%d_to_%d", from, to), func(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed := func(offset uint64) *TransactionStatusCache { + cache, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, offset)}}, + }, 0) + require.NoError(t, err) + return cache + } + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed(from).prepareTransactionStatusDelta(plan.messageIdentities) + cache := seed(to) + pinned := cache.View() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + require.Equal(t, uint8(to), cache.tip.delta[candidate.Transactions[0].Message.RecentBlockhash].keyIndex) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + found, err = pinned.ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + blob, err := cache.SnapshotThrough(1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + found, err = restored.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + require.NoError(t, cache.Unwind(1)) + found, err = cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(ancestor) + require.NoError(t, err) + require.True(t, found) + }) + } + } +} + +func TestPreparedStatusDeltaDoesNotPublishUntilCommit(t *testing.T) { + prior := runtime.GOMAXPROCS(2) + defer runtime.GOMAXPROCS(prior) + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + candidate := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + // A rejected bank joins and discards the prepared maps. Waiting is also + // idempotent for the normal commit followed by ProcessBlock's deferred join. + require.Same(t, p.wait(), p.wait()) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + require.Equal(t, uint64(10), cache.tip.slot) + cache.mu.Lock() + cache.coverageComplete = false + cache.mu.Unlock() + var incomplete *IncompleteTransactionStatusCoverageError + require.ErrorAs(t, cache.commitBlockWithPreparedDelta(candidate, plan, p.wait()), &incomplete) + require.Equal(t, uint64(10), cache.tip.slot) +} + +func TestPreparedStatusDeltaRejectsWrongPlanWithoutPublishingIt(t *testing.T) { + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + left := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + right := statusCacheTestBlock(11, statusCacheTestTransaction(1, 4, 5)) + leftPlan, err := planBlockTransactionExecution(left) + require.NoError(t, err) + rightPlan, err := planBlockTransactionExecution(right) + require.NoError(t, err) + prepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + // A mismatched prepared delta falls back to the actual block's identities. + require.NoError(t, cache.commitBlockWithPreparedDelta(right, rightPlan, prepared)) + found, err := cache.View().ContainsTransaction(left.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(right.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} + +func TestPreparedStatusDeltaEmptyBlock(t *testing.T) { + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + require.Nil(t, p, "empty bank must not queue background work") + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) +} + +func TestPreparedStatusDeltaScheduling(t *testing.T) { + for _, threads := range []int{1, 2} { + for _, count := range []int{1, 32, 33} { + t.Run(fmt.Sprintf("threads_%d/txs_%d", threads, count), func(t *testing.T) { + previous := runtime.GOMAXPROCS(threads) + defer runtime.GOMAXPROCS(previous) + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + defer p.wait() + require.Equal(t, threads > 1 && count > 32, p != nil) + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) + for _, tx := range block.Transactions { + found, err := cache.View().ContainsTransaction(tx) + require.NoError(t, err) + require.True(t, found) + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_validation.go b/pkg/replay/transaction_status_validation.go new file mode 100644 index 000000000..4413b8b8f --- /dev/null +++ b/pkg/replay/transaction_status_validation.go @@ -0,0 +1,37 @@ +package replay + +import ( + "math" + + b "github.com/Overclock-Validator/mithril/pkg/block" +) + +// transactionStatusValidation records a successful ancestor scan under cache.mu. +// It authorizes skipping only that scan, never the coverage, parent or exact +// block-identity checks. The receipt is private, bound to one cache instance and +// one immutable identity set, and checked under the publication lock. +// +// Mutating the visible index, binding the tip, rooting/pruning or restoring +// invalidates earlier receipts. In particular, committing and then unwinding to +// the same tip cannot resurrect one. Snapshot/Agave constructors create a new +// cache instance; this receipt is neither persisted nor usable after recovery. +// This optimization changes no crash-recovery or durable-checkpoint guarantee. +type transactionStatusValidation struct { + cache *TransactionStatusCache + identities *b.PreparedTransactionMessageIdentities + version uint64 +} + +func (v transactionStatusValidation) reusableForLocked(c *TransactionStatusCache, identities *b.PreparedTransactionMessageIdentities) bool { + return v.cache == c && v.identities == identities && + v.version == c.validationVersion && c.validationVersion != math.MaxUint64 +} + +// invalidateValidationLocked requires exclusive access (mu, or an unpublished +// constructor). Saturation permanently disables reuse instead of wrapping into +// an old generation. Even empty commits invalidate, because they change lineage. +func (c *TransactionStatusCache) invalidateValidationLocked() { + if c.validationVersion != math.MaxUint64 { + c.validationVersion++ + } +} diff --git a/pkg/replay/transaction_status_validation_test.go b/pkg/replay/transaction_status_validation_test.go new file mode 100644 index 000000000..9a1346125 --- /dev/null +++ b/pkg/replay/transaction_status_validation_test.go @@ -0,0 +1,151 @@ +package replay + +import ( + "errors" + "math" + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestStatusValidationInvalidation(t *testing.T) { + for _, action := range []string{"commit", "unwind", "round_trip", "root", "bind", "restore"} { + t.Run(action, func(t *testing.T) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if !receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("unchanged receipt not reusable") + } + switch action { + case "commit", "round_trip": + err = cache.CommitBlock(statusCacheTestBlock(11)) + if err == nil && action == "round_trip" { + err = cache.Unwind(11) + } + case "unwind": + err = cache.Unwind(10) + case "root": + cache.Root(10) + case "bind": + err = cache.BindTipBlockID(10, solana.Hash{1}) + case "restore": + var data []byte + data, err = cache.SnapshotThrough(10) + if err == nil { + cache, err = NewTransactionStatusCacheFromSnapshot(data) + } + } + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("receipt survived " + action) + } + }) + } +} + +func TestStatusValidationCannotCrossCacheOrIdentity(t *testing.T) { + good := NewTransactionStatusCache() + bad := NewTransactionStatusCache() + tx := statusCacheTestTransaction(1, 2, 3) + if err := good.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + if err := bad.CommitBlock(statusCacheTestBlock(10, tx)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, tx) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := good.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + // Both caches have the same version and parent slot, but different contents. + if good.validationVersion != bad.validationVersion { + t.Fatal("fixture must have equal versions") + } + prepared := bad.prepareTransactionStatusDelta(plan.messageIdentities) + var already *AncestorAlreadyProcessedTransactionMessagesError + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign cache: %v", err) + } + + unique := statusCacheTestBlock(11, statusCacheTestTransaction(4, 5, 6)) + uniquePlan, err := planBlockTransactionExecution(unique) + if err != nil { + t.Fatal(err) + } + receipt, err = bad.validateBlockForPublication(unique, uniquePlan) + if err != nil { + t.Fatal(err) + } + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign identities: %v", err) + } + failed, err := bad.validateBlockForPublication(blk, plan) + if err == nil || failed.cache != nil { + t.Fatalf("failed validation returned a receipt: %+v, %v", failed, err) + } +} + +func TestStatusValidationSaturation(t *testing.T) { + cache := NewTransactionStatusCache() + cache.validationVersion = math.MaxUint64 - 1 + blk := statusCacheTestBlock(1) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + cache.Root(0) + cache.Root(0) + if cache.validationVersion != math.MaxUint64 || receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("generation wrapped or old receipt reusable") + } + receipt, err = cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("saturated cache allowed reuse") + } + if err := cache.commitBlockWithValidation(blk, plan, nil, receipt); err != nil { + t.Fatal(err) + } +} + +func TestStatusValidationStillChecksBlockBinding(t *testing.T) { + cache := NewTransactionStatusCache() + blk := statusCacheTestBlock(1, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + blk.Transactions[0] = statusCacheTestTransaction(4, 5, 6) + if err := cache.commitBlockWithValidation(blk, plan, prepared, receipt); err == nil { + t.Fatal("replaced transaction accepted") + } +} diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go new file mode 100644 index 000000000..38d1028f8 --- /dev/null +++ b/pkg/sbpf/pooling_test.go @@ -0,0 +1,75 @@ +package sbpf + +import ( + "bytes" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/stretchr/testify/require" +) + +func poolingInterpreter(heap int) *Interpreter { + meter := cu.NewComputeMeter(100) + return NewInterpreter(testV3Program([]Slot{testSlot(OpExit, 0, 0, 0, 0)}, nil), + &VMOpts{HeapMax: heap, ComputeMeter: &meter}) +} + +func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { + old := UsePool + UsePool = true + t.Cleanup(func() { UsePool = old }) + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for i := 0; i < 30; i++ { + parent := poolingInterpreter(32 * 1024) + if !bytes.Equal(parent.heap, make([]byte, len(parent.heap))) || + !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { + t.Error("pooled VM exposed data from an earlier execution") + } + parent.heap[0] = byte(worker + 1) + parent.stack.mem[0] = byte(worker + 1) + child := poolingInterpreter(256 * 1024) + for j := range child.heap { + child.heap[j] = 0xab + } + for j := range child.stack.mem { + child.stack.mem[j] = 0xcd + } + child.Finish() + if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { + t.Error("nested VM storage aliased its active parent") + } + parent.Finish() + } + }(worker) + } + wg.Wait() + ip := poolingInterpreter(32 * 1024) + defer ip.Finish() + require.Len(t, ip.heap, 32*1024) + _, err := ip.Translate(VaddrHeap+32*1024, 1, false) + require.Error(t, err, "pool capacity must not widen the requested heap mapping") +} + +func BenchmarkVMCreateAndFinish(b *testing.B) { + old := UsePool + b.Cleanup(func() { UsePool = old }) + for _, pooled := range []bool{false, true} { + name := "fresh" + if pooled { + name = "pooled" + } + b.Run(name, func(b *testing.B) { + UsePool = pooled + b.ReportAllocs() + for i := 0; i < b.N; i++ { + ip := poolingInterpreter(32 * 1024) + ip.Finish() + } + }) + } +} diff --git a/pkg/sealevel/execution_ctx.go b/pkg/sealevel/execution_ctx.go index 348b4647c..ba54cc267 100644 --- a/pkg/sealevel/execution_ctx.go +++ b/pkg/sealevel/execution_ctx.go @@ -5,7 +5,6 @@ import ( "fmt" "sync" "sync/atomic" - "time" "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/accountsdb" @@ -19,6 +18,9 @@ import ( ) type ExecutionCtx struct { + // SkipTimingMetrics disables instruction-dispatch timing collection for + // leader execution; it never changes instruction validation or CU charging. + SkipTimingMetrics bool Log Logger Accounts accounts.Accounts TransactionContext *TransactionCtx @@ -237,14 +239,14 @@ func (execCtx *ExecutionCtx) PrepareInstruction(ix Instruction, signers []solana } func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAccts []InstructionAccount, programIndices []uint64) error { - start := time.Now() + start := metrics.StartTiming(!execCtx.SkipTimingMetrics) nextInstrCtx, err := execCtx.TransactionContext.NextInstructionCtx() if err != nil { return err } metrics.GlobalBlockReplay.GetNextIxCtx.AddTimingSince(start) - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) nextInstrCtx.Configure(programIndices, instructionAccts, instrData) metrics.GlobalBlockReplay.NextIxCtxConfigure.AddTimingSince(start) @@ -267,7 +269,7 @@ func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAcc }) } - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err = execCtx.Push() if err != nil { return err @@ -283,7 +285,7 @@ func (execCtx *ExecutionCtx) ProcessInstruction(instrData []byte, instructionAcc err1 := execCtx.ExecuteInstruction() - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err2 := execCtx.Pop() metrics.GlobalBlockReplay.IxPop.AddTimingSince(start) @@ -304,7 +306,7 @@ func (execCtx *ExecutionCtx) AddModifiedVoteState(pubkey solana.PublicKey, state } func (execCtx *ExecutionCtx) ExecuteInstruction() error { - start := time.Now() + start := metrics.StartTiming(!execCtx.SkipTimingMetrics) txCtx := execCtx.TransactionContext instrCtx, err := txCtx.CurrentInstructionCtx() @@ -334,7 +336,7 @@ func (execCtx *ExecutionCtx) ExecuteInstruction() error { } metrics.GlobalBlockReplay.ExecIxResolveNativeProgram.AddTimingSince(start) - start = time.Now() + start = metrics.StartTiming(!execCtx.SkipTimingMetrics) err = nativeProgramFn(execCtx) switch nativeProgramStr { case a.SystemProgramAddrStr: diff --git a/pkg/sealevel/vote_deque_ownership_test.go b/pkg/sealevel/vote_deque_ownership_test.go new file mode 100644 index 000000000..b7d50f43b --- /dev/null +++ b/pkg/sealevel/vote_deque_ownership_test.go @@ -0,0 +1,29 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" + "github.com/gammazero/deque" + "github.com/stretchr/testify/require" +) + +func TestProcessNewVoteStateOwnsRetainedDeque(t *testing.T) { + // Model the TowerSync scratch deque being returned to its pool and reused. + scratch := new(deque.Deque[LandedVote]) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 100, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 101, ConfirmationCount: 1}}) + state := new(VoteState) + require.NoError(t, processNewVoteState(state, scratch, nil, nil, 0, 101, features.Features{})) + cached := newVoteState4FromCurrent(state, solana.PublicKey{}) + want := []LandedVote{state.Votes.At(0), state.Votes.At(1)} + + scratch.Clear() + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 200, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 201, ConfirmationCount: 1}}) + for i, vote := range want { + require.Equal(t, vote, state.Votes.At(i)) + require.Equal(t, vote, cached.Votes.At(i)) + } +} diff --git a/pkg/sealevel/vote_program.go b/pkg/sealevel/vote_program.go index 43a8500c2..679d1787d 100644 --- a/pkg/sealevel/vote_program.go +++ b/pkg/sealevel/vote_program.go @@ -1951,7 +1951,14 @@ func processNewVoteState(voteState *VoteState, newState *deque.Deque[LandedVote] } voteState.RootSlot = newRoot - voteState.Votes = *newState + // newState may be a pooled deque. Own the backing storage before its + // caller returns it to the pool: the resulting state can escape into the + // shared vote cache after this instruction completes. + var owned deque.Deque[LandedVote] + for i := 0; i < newState.Len(); i++ { + owned.PushBack(newState.At(i)) + } + voteState.Votes = owned return nil } diff --git a/pkg/sigverify/config_policy_test.go b/pkg/sigverify/config_policy_test.go new file mode 100644 index 000000000..52e8948ce --- /dev/null +++ b/pkg/sigverify/config_policy_test.go @@ -0,0 +1,63 @@ +package sigverify + +import ( + "runtime" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestResolveConfigPolicyDefaultsAndOverrides(t *testing.T) { + zero, err := ResolveConfig(Config{}) + require.NoError(t, err) + require.Equal(t, BackendAuto, zero.Backend) + require.Equal(t, min(2, runtime.GOMAXPROCS(0)), zero.Workers) + require.Equal(t, 8, zero.BatchTarget) + require.False(t, zero.DisableShredOverlap) + defaults, err := ResolveConfig(Defaults()) + require.NoError(t, err) + require.Equal(t, zero, defaults) + + override := Config{Backend: BackendGeneric, Workers: 4, BatchTarget: 4, DisableShredOverlap: true} + resolved, err := ResolveConfig(override) + require.NoError(t, err) + require.Equal(t, override, resolved) +} + +func TestInvalidPolicyDoesNotLatchBackend(t *testing.T) { + if !inChild() { + out, err := runConfigureChild(t, t.Name()) + require.NoError(t, err, "child output:\n%s", out) + require.Contains(t, out, "PASS") + return + } + before := Cfg + for _, cfg := range []Config{ + {Backend: BackendGeneric, Workers: -1}, + {Backend: BackendGeneric, BatchTarget: -1}, + {Backend: BackendGeneric, BatchTarget: 3}, + {Backend: BackendGeneric, BatchTarget: 16}, + } { + _, err := Configure(cfg) + require.Error(t, err) + require.Equal(t, before, Cfg, "invalid policy must not become live") + require.Empty(t, configuredBackend, "invalid policy must not latch the backend") + } + _, err := Configure(Config{Backend: BackendGeneric, Workers: 2, BatchTarget: 4}) + require.NoError(t, err, "valid configuration must remain possible after rejection") + require.Equal(t, 2, TransactionWorkers()) + require.Equal(t, 4, TransactionBatchTarget()) +} + +func TestTransactionPolicyWithoutConfigure(t *testing.T) { + if !inChild() { + out, err := runConfigureChild(t, t.Name()) + require.NoError(t, err, "child output:\n%s", out) + require.Contains(t, out, "PASS") + return + } + Cfg = Config{} + require.Equal(t, min(2, runtime.GOMAXPROCS(0)), TransactionWorkers()) + require.Equal(t, 8, TransactionBatchTarget()) + require.False(t, Cfg.DisableShredOverlap) +} diff --git a/pkg/sigverify/sigverify.go b/pkg/sigverify/sigverify.go index fae00447d..a0502712c 100644 --- a/pkg/sigverify/sigverify.go +++ b/pkg/sigverify/sigverify.go @@ -20,6 +20,7 @@ package sigverify import ( "fmt" + "runtime" "sync" narya "github.com/Overclock-Validator/narya-ed25519/ed25519" @@ -40,15 +41,60 @@ const ( BackendStdlib = "stdlib" ) -// Config selects the verification backend. It is deliberately tiny: the -// library's own defaults are good, and every knob here is a consensus-visible -// or performance-visible choice that an operator should have to state. +// Config selects the backend and the Turbine transaction-verification policy. +// Worker and batching settings do not change the TPU or fallback replay pools. type Config struct { - Backend string + Backend string + Workers int + BatchTarget int + DisableShredOverlap bool } // Defaults returns the configuration used when the operator sets nothing. -func Defaults() Config { return Config{Backend: BackendAuto} } +func Defaults() Config { return Config{Backend: BackendAuto, BatchTarget: BatchTarget} } + +// ResolveConfig validates every setting before the one-shot backend selection. +// Zero workers selects at most two transaction-verification workers; zero batch +// target uses eight signatures. Available short batches are never held to fill. +func ResolveConfig(cfg Config) (Config, error) { + if cfg.Backend == "" { + cfg.Backend = BackendAuto + } + switch cfg.Backend { + case BackendAuto, BackendR51, BackendGeneric, BackendStdlib: + default: + return Config{}, fmt.Errorf("sigverify.backend must be one of %q, %q, %q, %q; got %q", BackendAuto, BackendR51, BackendGeneric, BackendStdlib, cfg.Backend) + } + if cfg.Workers < 0 { + return Config{}, fmt.Errorf("sigverify.workers must be >= 0; got %d", cfg.Workers) + } + if cfg.Workers == 0 { + cfg.Workers = min(2, max(1, runtime.GOMAXPROCS(0))) + } + if cfg.BatchTarget == 0 { + cfg.BatchTarget = BatchTarget + } + if cfg.BatchTarget != 4 && cfg.BatchTarget != 8 { + return Config{}, fmt.Errorf("sigverify.batch_target must be 4 or 8 (0 uses 8); got %d", cfg.BatchTarget) + } + return cfg, nil +} + +// TransactionWorkers also supports callers that do not run node Configure. +func TransactionWorkers() int { + if Cfg.Workers > 0 { + return Cfg.Workers + } + return min(2, max(1, runtime.GOMAXPROCS(0))) +} + +// TransactionBatchTarget also supports the zero configuration outside startup. +func TransactionBatchTarget() int { + if Cfg.BatchTarget == 4 { + return 4 + } + return BatchTarget +} // Cfg is the live configuration, set once by Configure during startup and // read-only afterwards. It follows the same shape as replay.TrailingVerifierCfg. @@ -64,10 +110,6 @@ var Cfg = Defaults() // underlying library pins its backend on first use and a late switch would // leave the process in a state neither caller asked for. func Configure(cfg Config) (string, error) { - if cfg.Backend == "" { - cfg.Backend = Defaults().Backend - } - configureMu.Lock() defer configureMu.Unlock() @@ -79,12 +121,10 @@ func Configure(cfg Config) (string, error) { // Validate before publishing anything. Assigning Cfg first would leave a // rejected backend name visible to Backend() and to the startup log. - switch cfg.Backend { - case BackendAuto, BackendR51, BackendGeneric, BackendStdlib: - default: - return "", fmt.Errorf( - "sigverify.backend must be one of %q, %q, %q, %q; got %q", - BackendAuto, BackendR51, BackendGeneric, BackendStdlib, cfg.Backend) + var err error + cfg, err = ResolveConfig(cfg) + if err != nil { + return "", err } resolved, err := installBackend(cfg.Backend) diff --git a/pkg/statsd/statsd.go b/pkg/statsd/statsd.go index bbf3eea9c..2418cc025 100644 --- a/pkg/statsd/statsd.go +++ b/pkg/statsd/statsd.go @@ -111,6 +111,7 @@ var ( TxsPerBlock = Metric{"txs_per_block"} SnapshotTarBytesRead = Metric{"snapshot_tar_bytes_read"} SlotReplays = Metric{"slot_replays"} + BlockProductionEntrySerializationErrors = Metric{"block_production_entry_serialization_errors_total"} BlockProductionLeaderSlots = Metric{"block_production_leader_slots_total"} BlockProductionLeaderSlotTerminals = Metric{"block_production_leader_slot_terminals_total"} BlockProductionParentReady = Metric{"block_production_parent_ready_activations_total"} @@ -124,6 +125,13 @@ var ( TurbineBlockDecode = Metric{"turbine_block_decode_duration_seconds"} TurbineTransactionParse = Metric{"turbine_transaction_parse_duration_seconds"} TurbineTransactionSigverify = Metric{"turbine_transaction_sigverify_duration_seconds"} + // Early durations sum elapsed component work, including verifier queueing; + // they overlap shred collection and are neither CPU nor pipeline wall time. + TurbineEarlyTransactionParse = Metric{"turbine_early_transaction_parse_duration_seconds"} + TurbineEarlyTransactionSigverify = Metric{"turbine_early_transaction_sigverify_elapsed_seconds"} + TurbineEarlyPreparationWait = Metric{"turbine_early_preparation_wait_seconds"} + TurbineEarlyVerifiedTransactions = Metric{"turbine_early_verified_transactions_total"} + TurbineFullToReady = Metric{"turbine_full_to_ready_duration_seconds"} // ReplaySigverifyGroup times one drained group of transaction signatures // and ReplaySigverifyGroupSignatures counts how many signatures were in it. // The pair is what tells an operator whether batching is actually happening: @@ -229,11 +237,12 @@ var MetricToType = map[Metric]metricType{ SlotReplayDurationMs: TimingT, TxsPerBlock: TimingT, - SnapshotTarBytesRead: CountT, - SlotReplays: CountT, - BlockProductionLeaderSlots: CountT, - BlockProductionLeaderSlotTerminals: CountT, - BlockProductionParentReady: CountT, + SnapshotTarBytesRead: CountT, + SlotReplays: CountT, + BlockProductionEntrySerializationErrors: CountT, + BlockProductionLeaderSlots: CountT, + BlockProductionLeaderSlotTerminals: CountT, + BlockProductionParentReady: CountT, BlockProductionParentReadyAge: TimingT, BlockProductionStartCutoffLate: TimingT, @@ -245,6 +254,11 @@ var MetricToType = map[Metric]metricType{ TurbineBlockDecode: TimingT, TurbineTransactionParse: TimingT, TurbineTransactionSigverify: TimingT, + TurbineEarlyTransactionParse: TimingT, + TurbineEarlyTransactionSigverify: TimingT, + TurbineEarlyPreparationWait: TimingT, + TurbineEarlyVerifiedTransactions: CountT, + TurbineFullToReady: TimingT, ReplaySigverifyGroup: TimingT, ReplaySigverifyGroupSignatures: CountT, TurbineReplayAdmission: TimingT, @@ -335,13 +349,14 @@ var MetricToLabels = map[Metric][]string{ TasksIndexEntryBuilderLatency: {}, TasksAppendVecCopyingLatency: {}, - SlotReplayDurationMs: {}, - TxsPerBlock: {}, - SnapshotTarBytesRead: {}, - SlotReplays: {}, - BlockProductionLeaderSlots: {"outcome", "reason"}, - BlockProductionLeaderSlotTerminals: {"outcome", "terminal", "cause"}, - BlockProductionParentReady: {"activation", "status"}, + SlotReplayDurationMs: {}, + TxsPerBlock: {}, + SnapshotTarBytesRead: {}, + SlotReplays: {}, + BlockProductionEntrySerializationErrors: {}, + BlockProductionLeaderSlots: {"outcome", "reason"}, + BlockProductionLeaderSlotTerminals: {"outcome", "terminal", "cause"}, + BlockProductionParentReady: {"activation", "status"}, BlockProductionParentReadyAge: {"activation"}, BlockProductionStartCutoffLate: {"phase"}, @@ -353,6 +368,11 @@ var MetricToLabels = map[Metric][]string{ TurbineBlockDecode: {}, TurbineTransactionParse: {}, TurbineTransactionSigverify: {}, + TurbineEarlyTransactionParse: {}, + TurbineEarlyTransactionSigverify: {}, + TurbineEarlyPreparationWait: {}, + TurbineEarlyVerifiedTransactions: {}, + TurbineFullToReady: {}, ReplaySigverifyGroup: {}, ReplaySigverifyGroupSignatures: {}, TurbineReplayAdmission: {}, @@ -390,6 +410,10 @@ var MetricToBuckets = map[Metric][]float64{ TurbineBlockDecode: turbinePipelineDurationBuckets, TurbineTransactionParse: turbinePipelineDurationBuckets, TurbineTransactionSigverify: turbinePipelineDurationBuckets, + TurbineEarlyTransactionParse: turbinePipelineDurationBuckets, + TurbineEarlyTransactionSigverify: turbinePipelineDurationBuckets, + TurbineEarlyPreparationWait: turbinePipelineDurationBuckets, + TurbineFullToReady: turbinePipelineDurationBuckets, ReplaySigverifyGroup: turbinePipelineDurationBuckets, TurbineReplayAdmission: turbinePipelineDurationBuckets, AlpenglowVoteRewards: turbinePipelineDurationBuckets, diff --git a/pkg/statsd/statsd_test.go b/pkg/statsd/statsd_test.go index 830ea99eb..6ec3cd0c5 100644 --- a/pkg/statsd/statsd_test.go +++ b/pkg/statsd/statsd_test.go @@ -203,6 +203,10 @@ func TestTurbinePipelineDurationMetricsUseSecondsAndBoundedSchema(t *testing.T) TurbineBlockDecode, TurbineTransactionParse, TurbineTransactionSigverify, + TurbineEarlyTransactionParse, + TurbineEarlyTransactionSigverify, + TurbineEarlyPreparationWait, + TurbineFullToReady, TurbineReplayAdmission, } duration := 25 * time.Millisecond @@ -228,6 +232,11 @@ func TestTurbinePipelineDurationMetricsUseSecondsAndBoundedSchema(t *testing.T) } } +func TestTurbineEarlyVerifiedTransactionsHasBoundedCountSchema(t *testing.T) { + assert.Equal(t, CountT, MetricToType[TurbineEarlyVerifiedTransactions]) + assert.Equal(t, []string{}, MetricToLabels[TurbineEarlyVerifiedTransactions]) +} + func TestBlockProductionMetricLabelsStayBounded(t *testing.T) { assert.Equal(t, []string{"outcome", "reason"}, MetricToLabels[BlockProductionLeaderSlots]) assert.Equal(t, []string{"outcome", "terminal", "cause"}, MetricToLabels[BlockProductionLeaderSlotTerminals]) diff --git a/pkg/tpu/txfixture/readonly_pair.go b/pkg/tpu/txfixture/readonly_pair.go new file mode 100644 index 000000000..c1de0a074 --- /dev/null +++ b/pkg/tpu/txfixture/readonly_pair.go @@ -0,0 +1,42 @@ +package txfixture + +import ( + "crypto/ed25519" + "fmt" + + "github.com/gagliardetto/solana-go" +) + +const ReadonlyPairPoolSize = 128 +const ReadonlyPairCapacity = ReadonlyPairPoolSize * (ReadonlyPairPoolSize - 1) + +// ReadonlyPairWire builds the 198-byte, single-signature, zero-instruction +// workload used for leader packing tests. Varying ordered pairs of existing +// readonly accounts gives distinct messages without adding instructions or +// forcing a lookup of a new nonexistent account for every transaction. +// Reusing an ordinal requires a different payer or recent blockhash. +func ReadonlyPairWire(key ed25519.PrivateKey, hash solana.Hash, pool []solana.PublicKey, ordinal int) ([]byte, error) { + if len(key) != ed25519.PrivateKeySize || len(pool) != ReadonlyPairPoolSize || ordinal < 0 || ordinal >= ReadonlyPairCapacity { + return nil, fmt.Errorf("invalid readonly-pair fixture key, pool or ordinal") + } + n := ordinal * 7919 % ReadonlyPairCapacity + a, b := n/(ReadonlyPairPoolSize-1), n%(ReadonlyPairPoolSize-1) + if b >= a { + b++ + } + payer := solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + if payer == pool[a] || payer == pool[b] || pool[a] == pool[b] { + return nil, fmt.Errorf("readonly-pair accounts must be distinct") + } + message := make([]byte, 0, 133) + message = append(message, 1, 0, 2, 3) + message = append(message, payer[:]...) + message = append(message, pool[a][:]...) + message = append(message, pool[b][:]...) + message = append(message, hash[:]...) + message = append(message, 0) + wire := make([]byte, 0, 198) + wire = append(wire, 1) + wire = append(wire, ed25519.Sign(key, message)...) + return append(wire, message...), nil +} diff --git a/pkg/tpu/txfixture/readonly_pair_test.go b/pkg/tpu/txfixture/readonly_pair_test.go new file mode 100644 index 000000000..70efa16dd --- /dev/null +++ b/pkg/tpu/txfixture/readonly_pair_test.go @@ -0,0 +1,68 @@ +package txfixture + +import ( + "crypto/ed25519" + "crypto/sha256" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Reproduce the full preloaded four-slot experiment without RPC, funds or sends. +func TestReadonlyPair200KDistinctMessages(t *testing.T) { + keys := make([]ed25519.PrivateKey, 8) + for i := range keys { + seed := sha256.Sum256([]byte{byte(i), 73}) + keys[i] = ed25519.NewKeyFromSeed(seed[:]) + } + pool := make([]solana.PublicKey, ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + } + seen := make(map[[32]byte]struct{}, 200000) + for i := 0; i < 200000; i++ { + hash, index := solana.Hash{1}, i + if i >= 120000 { + hash, index = solana.Hash{2}, i-120000 + } + wire, err := ReadonlyPairWire(keys[index%8], hash, pool, index/8) + require.NoError(t, err) + require.Len(t, wire, 198) + h := sha256.Sum256(wire[65:]) + if _, ok := seen[h]; ok { + t.Fatalf("duplicate message %d", i) + } + seen[h] = struct{}{} + if i%ReadonlyPairCapacity == 0 || i == 119999 || i == 120000 || i == 199999 { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + require.Len(t, tx.Signatures, 1) + require.Empty(t, tx.Message.Instructions) + require.Equal(t, hash, tx.Message.RecentBlockhash) + require.True(t, ed25519.Verify(ed25519.PublicKey(tx.Message.AccountKeys[0][:]), wire[65:], wire[1:65])) + } + } +} + +func TestReadonlyPairRejectsInvalidInputs(t *testing.T) { + key := ed25519.PrivateKey(PayerPrivateKey()) + pool := make([]solana.PublicKey, ReadonlyPairPoolSize) + for i := range pool { + pool[i] = solana.PublicKey{byte(i + 1), 77} + } + for _, ordinal := range []int{-1, ReadonlyPairCapacity} { + _, err := ReadonlyPairWire(key, TestBlockhash(), pool, ordinal) + require.Error(t, err) + } + _, err := ReadonlyPairWire(nil, TestBlockhash(), pool, 0) + require.Error(t, err) + _, err = ReadonlyPairWire(key, TestBlockhash(), nil, 0) + require.Error(t, err) + pool[0] = solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)) + _, err = ReadonlyPairWire(key, TestBlockhash(), pool, 0) + require.Error(t, err) + pool[0], pool[1] = solana.PublicKey{9}, solana.PublicKey{9} + _, err = ReadonlyPairWire(key, TestBlockhash(), pool, 0) + require.Error(t, err) +} diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 9b69f4370..d2bb4acee 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -10,6 +10,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/statsd" + "github.com/Overclock-Validator/mithril/pkg/turbine/internal/rsrecover" "github.com/gagliardetto/solana-go" "github.com/klauspost/reedsolomon" ) @@ -43,20 +44,26 @@ const ( ) type SlotAssembler struct { - mu sync.Mutex - slots map[uint64]*slotState - completedSlots map[uint64]struct{} - knownBlockIDs map[uint64]solana.Hash - rejectedBlockIDs map[uint64]map[solana.Hash]struct{} - protectedKnownIDs map[uint64]struct{} - protectedBlockIDs map[uint64]struct{} - priorityRepairSlots map[uint64]struct{} - priorityRepairOrder []uint64 - encoders map[fecLayout]reedsolomon.Encoder - partialShredObs map[uint64]PartialShredObservation // shreds seen for slots that never became full (retained for skip observability) - retentionFloor uint64 // when non-zero, slots >= floor are never "too old" (repair catchup holds a window far behind the live edge) - edgeScanLag uint64 // how far behind the shred edge the freshness-repair scan reaches (0 = repairScanSlotWindow) - maxObservedSlot uint64 + mu sync.Mutex + slots map[uint64]*slotState + completedSlots map[uint64]struct{} + knownBlockIDs map[uint64]solana.Hash + rejectedBlockIDs map[uint64]map[solana.Hash]struct{} + protectedKnownIDs map[uint64]struct{} + protectedBlockIDs map[uint64]struct{} + priorityRepairSlots map[uint64]struct{} + priorityRepairOrder []uint64 + encoders map[fecLayout]reedsolomon.Encoder + partialShredObs map[uint64]PartialShredObservation // shreds seen for slots that never became full (retained for skip observability) + retentionFloor uint64 // when non-zero, slots >= floor are never "too old" (repair catchup holds a window far behind the live edge) + edgeScanLag uint64 // how far behind the shred edge the freshness-repair scan reaches (0 = repairScanSlotWindow) + maxObservedSlot uint64 + // Age sweeps depend on the edge, repair floor, and mutations that can add + // old metadata or release a completing generation's protected identities. + retentionSwept bool + retentionDirty bool + retentionSweepEdge uint64 + retentionSweepFloor uint64 highestFullSlot uint64 // monotonic: highest slot reconstructed from shreds ("full", Agave SlotMeta/is_full sense) recoveredDataShreds uint64 usefulRepairShreds uint64 // distinct data shreds delivered BY repair (the throughput signal) @@ -72,6 +79,7 @@ type SlotAssembler struct { // Configured before ingestion; tests may replace it with a blocking probe. // Production uses the process-wide bounded transaction verifier. verifyTransactions func(context.Context, *block.Block) error + entryPrefetch *entryPrefetchPool } type SlotRepairRequest struct { @@ -91,14 +99,15 @@ type PartialShredObservation struct { } type slotState struct { - slot uint64 - parentSlot uint64 - shreds map[uint32]*Shred - fecSets map[uint32]*fecState - lastIndex uint32 - haveLast bool - shredVer uint16 - firstParent bool + pipelineTrace *entryPipelineTrace + slot uint64 + parentSlot uint64 + shreds map[uint32]*Shred + fecSets map[uint32]*fecState + lastIndex uint32 + haveLast bool + shredVer uint16 + firstParent bool // Observability: when the slot's first shred was accepted, and how many of // its shreds arrived via repair rather than turbine. @@ -106,7 +115,10 @@ type slotState struct { fullAt time.Time repairedShreds int // completing makes the immutable full state a single-owner generation token. - completing bool + completing bool + batchIndex *entryBatchIndex + completeBatches []shredBatchRange + prefetch *slotEntryPrefetch // Assembly failures for this slot (mixed variants/signatures, FEC layout // conflicts, ...). A slot frozen below completion while repair responses // flow is usually poisoned state — the latest error names the poison. @@ -161,6 +173,7 @@ type fecState struct { haveSig bool dataVariant byte codeVariant byte + rootCache *authenticatedFECRoot } func NewSlotAssembler() *SlotAssembler { @@ -174,7 +187,6 @@ func NewSlotAssembler() *SlotAssembler { priorityRepairSlots: make(map[uint64]struct{}), partialShredObs: make(map[uint64]PartialShredObservation), encoders: make(map[fecLayout]reedsolomon.Encoder), - verifyTransactions: validateBlockTransactionsContext, } } @@ -185,6 +197,7 @@ func (a *SlotAssembler) recordPartialObsLocked(state *slotState) { if state == nil || len(state.shreds) == 0 { return } + a.retentionDirty = true a.partialShredObs[state.slot] = PartialShredObservation{ DataShreds: len(state.shreds), RepairedShreds: state.repairedShreds, @@ -239,6 +252,13 @@ func (a *SlotAssembler) AddShredFrom(shred *Shred, fromRepair bool) (*block.Bloc // reconstructable it returns a single immutable completion token; decoding, // parsing, and signature verification must happen after this method unlocks. func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotCompletionWork, error) { + return a.addShredFromWithRoot(shred, fromRepair, nil) +} + +// addShredFromWithRoot accepts an optional result from successful authentication +// of this immutable shred. Unauthenticated callers and spool hydration use nil +// and retain the normal root-computation fallback. +func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root *solana.Hash) (*slotCompletionWork, error) { if shred == nil { return nil, nil } @@ -287,9 +307,16 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple state.noteError(err) return nil, err } + state.traceAcceptedShred(shred) + a.notePrefetchShredLocked(state, shred) if state.firstShredAt.IsZero() { state.firstShredAt = time.Now() } + if root != nil { + if fec := state.fecSets[shred.FECSetIndex]; fec != nil { + fec.rememberAuthenticatedRoot(shred, *root) + } + } recovered, err := a.recoverFEC(state, shred.FECSetIndex) if err != nil { @@ -303,10 +330,13 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple return nil, err } if err == nil { + state.traceAcceptedShred(recoveredShred) + a.notePrefetchShredLocked(state, recoveredShred) a.recoveredDataShreds++ } } + a.prefetchEntriesLocked(state) if !state.complete() { return nil, nil } @@ -318,6 +348,9 @@ func (a *SlotAssembler) claimCompletionLocked(state *slotState, reportNonCanonic return nil } now := time.Now() + if state.pipelineTrace != nil { + state.pipelineTrace.sealed = true + } observeCollection := state.fullAt.IsZero() if observeCollection { state.fullAt = now @@ -348,6 +381,7 @@ func (a *SlotAssembler) abortCompletion(work *slotCompletionWork) { } a.mu.Lock() if a.slots[work.state.slot] == work.state && work.state.completing { + a.retentionDirty = true work.state.completing = false } a.mu.Unlock() @@ -363,6 +397,7 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet if ctx.Err() != nil { return processedSlotCompletion{canceled: true} } + ctx = withEntryPipelineTrace(ctx, work.state.pipelineTrace) startedAt := time.Now() timings := block.TurbineIngressTimings{CompletionQueueDelay: startedAt.Sub(work.queuedAt)} _ = statsd.Duration(statsd.TurbineBlockCompletionQueueDelay, timings.CompletionQueueDelay, nil) @@ -374,19 +409,27 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet } decodeStartedAt := time.Now() - var decodeTimings entryDecodeTimings + decodeTimings := entryDecodeTimings{ctx: ctx} + if work.state.prefetch != nil { + decodeTimings.prefetched = work.state.prefetch.batches + } blk, parentInfo, roots, err := work.state.decodeBlock(&decodeTimings) decodeTotal := time.Since(decodeStartedAt) - decodeOnly := decodeTotal - decodeTimings.transactionParse + decodeOnly := decodeTotal - decodeTimings.transactionParse - decodeTimings.prefetchWait if decodeOnly < 0 { decodeOnly = 0 } timings.BlockDecode = decodeOnly timings.TransactionParse = decodeTimings.transactionParse + timings.EarlyPreparationWait = decodeTimings.prefetchWait _ = statsd.Duration(statsd.TurbineBlockDecode, timings.BlockDecode, nil) _ = statsd.Duration(statsd.TurbineTransactionParse, timings.TransactionParse, nil) processed := processedSlotCompletion{block: blk, parentInfo: parentInfo, roots: roots, err: err, timings: timings} if err != nil { + if ctx.Err() != nil { + processed.canceled = true + processed.err = nil + } return processed } if ctx.Err() != nil { @@ -395,7 +438,12 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet } sigverifyStartedAt := time.Now() - processed.err = work.verifyTransactions(ctx, blk) + if work.state.prefetch != nil && len(decodeTimings.retained) > 0 { + processed.err = verifyDecodedEntryBatchesWithTimings(ctx, blk, decodeTimings.retained, work.state.prefetch.pool.verifier, &decodeTimings) + } else { + processed.err = work.verifyTransactions(ctx, blk) + } + earlyEntryTimings(&decodeTimings, work.state.fullAt, &processed.timings) processed.timings.TransactionSigverify = time.Since(sigverifyStartedAt) _ = statsd.Duration(statsd.TurbineTransactionSigverify, processed.timings.TransactionSigverify, nil) if ctx.Err() != nil { @@ -406,6 +454,13 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet if processed.err == nil { blk.MarkTransactionSignaturesVerified() processed.completionReadyAt = time.Now() + processed.timings.FullToReady = processed.completionReadyAt.Sub(work.state.fullAt) + _ = statsd.Duration(statsd.TurbineFullToReady, processed.timings.FullToReady, nil) + _ = statsd.Duration(statsd.TurbineEarlyPreparationWait, processed.timings.EarlyPreparationWait, nil) + _ = statsd.Duration(statsd.TurbineEarlyTransactionParse, processed.timings.EarlyTransactionParse, nil) + _ = statsd.Duration(statsd.TurbineEarlyTransactionSigverify, processed.timings.EarlyTransactionSigverify, nil) + _ = statsd.Count(statsd.TurbineEarlyVerifiedTransactions, int64(processed.timings.EarlyVerifiedTransactions), nil) + queueEntryPipelineReport(work.state, blk, &decodeTimings, startedAt, processed.completionReadyAt) } return processed } @@ -420,6 +475,13 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p a.mu.Unlock() return nil, nil } + // Every terminal outcome releases this generation's retention protection. + a.retentionDirty = true + if processed.canceled { + state.completing = false + a.mu.Unlock() + return nil, nil + } if processed.err != nil { // Retain a deterministic decode/verification failure on the live full // state so catchup diagnostics report poison instead of a missing slot. @@ -434,6 +496,7 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p if !a.acceptAlpenglowBlockIDLocked(blk) { a.trackNonCanonicalBlockIDLocked(blk) a.recordPartialObsLocked(state) + a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.mu.Unlock() if work.reportNonCanonical { @@ -442,6 +505,7 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p return nil, nil } + a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.completedSlots[state.slot] = struct{}{} a.trackBlockIDLocked(blk) @@ -510,6 +574,9 @@ func (a *SlotAssembler) SetKnownAlpenglowBlockID(slot uint64, blockID solana.Has if _, rejected := a.rejectedBlockIDs[slot][blockID]; rejected { return } + if _, exists := a.knownBlockIDs[slot]; !exists { + a.retentionDirty = true + } a.knownBlockIDs[slot] = blockID } @@ -530,6 +597,7 @@ func (a *SlotAssembler) RejectAlpenglowBlockID(slot uint64, blockID solana.Hash) if ids == nil { ids = make(map[solana.Hash]struct{}) a.rejectedBlockIDs[slot] = ids + a.retentionDirty = true } ids[blockID] = struct{}{} if a.knownBlockIDs[slot] == blockID { @@ -541,7 +609,9 @@ func (a *SlotAssembler) ResetSlot(slot uint64) { a.mu.Lock() defer a.mu.Unlock() + a.retentionDirty = true a.recordPartialObsLocked(a.slots[slot]) + a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) delete(a.completedSlots, slot) } @@ -635,6 +705,32 @@ func (a *SlotAssembler) slotTooOldLocked(slot uint64) bool { } func (a *SlotAssembler) pruneOldSlotsLocked() { + if !a.retentionSwept || a.retentionDirty || a.retentionSweepEdge != a.maxObservedSlot || a.retentionSweepFloor != a.retentionFloor { + a.sweepRetentionMapsLocked() + a.retentionSwept = true + a.retentionDirty = false + a.retentionSweepEdge = a.maxObservedSlot + a.retentionSweepFloor = a.retentionFloor + } + // New incomplete generations can exceed the cap without advancing the + // edge (especially during catch-up). Never cache the capacity check. + + if len(a.slots) == 0 { + return + } + for len(a.slots) > maxRetainedIncompleteSlotCap { + victim, ok := a.capEvictionCandidateLocked() + if !ok { + return + } + a.recordPartialObsLocked(a.slots[victim]) + a.releasePrefetchLocked(a.slots[victim]) + delete(a.slots, victim) + a.evictedSlots++ + } +} + +func (a *SlotAssembler) sweepRetentionMapsLocked() { if len(a.slots) > 0 && a.maxObservedSlot > maxRetainedIncompleteSlotLag { minSlot := a.maxObservedSlot - maxRetainedIncompleteSlotLag if a.retentionFloor > 0 && a.retentionFloor < minSlot { @@ -643,6 +739,7 @@ func (a *SlotAssembler) pruneOldSlotsLocked() { for slot, state := range a.slots { if slot < minSlot && !state.completing { a.recordPartialObsLocked(state) + a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) a.evictedSlots++ } @@ -686,18 +783,6 @@ func (a *SlotAssembler) pruneOldSlotsLocked() { } a.prunePriorityRepairSlotsLocked() - if len(a.slots) == 0 { - return - } - for len(a.slots) > maxRetainedIncompleteSlotCap { - victim, ok := a.capEvictionCandidateLocked() - if !ok { - return - } - a.recordPartialObsLocked(a.slots[victim]) - delete(a.slots, victim) - a.evictedSlots++ - } } // capEvictionCandidateLocked chooses state furthest ahead of replay, rather @@ -1018,6 +1103,9 @@ func (a *SlotAssembler) trackBlockIDLocked(blk *block.Block) { if known, ok := a.knownBlockIDs[blk.Slot]; ok && known != (solana.Hash{}) && known != blockID { return } + if _, exists := a.knownBlockIDs[blk.Slot]; !exists { + a.retentionDirty = true + } a.knownBlockIDs[blk.Slot] = blockID } @@ -1300,21 +1388,43 @@ func (a *SlotAssembler) recoverFEC(state *slotState, fecSetIndex uint32) ([]*Shr } shards[int(layout.dataShreds)+int(pos)] = shard } - encoder, err := a.fecEncoder(layout) - if err != nil { - return nil, err - } - required := make([]bool, int(layout.dataShreds)+int(layout.codingShreds)) var missingData int + missingDataIndex := -1 for idx := 0; idx < int(layout.dataShreds); idx++ { if fec.data[uint32(idx)] == nil { - required[idx] = true missingData++ + missingDataIndex = idx } } if missingData == 0 { return nil, nil } + if missingData == 1 && + layout.dataShreds == rsrecover.DataShards && + layout.codingShreds == rsrecover.CodingShards { + presence, err := rsrecover.Presence(shards) + if err != nil { + return nil, err + } + dst := make([]byte, layout.shardSize) + if err := rsrecover.RecoverOneData(presence, missingDataIndex, shards, dst); err != nil { + return nil, fmt.Errorf("recover one FEC data shred slot %d fec_set=%d: %w", state.slot, fecSetIndex, err) + } + shred, err := fec.recoveredDataShred(uint32(missingDataIndex), dst) + if err != nil { + return nil, err + } + return []*Shred{shred}, nil + } + + required := make([]bool, int(layout.dataShreds)+int(layout.codingShreds)) + for idx := 0; idx < int(layout.dataShreds); idx++ { + required[idx] = fec.data[uint32(idx)] == nil + } + encoder, err := a.fecEncoder(layout) + if err != nil { + return nil, err + } if err := encoder.ReconstructSome(shards, required); err != nil { if errors.Is(err, reedsolomon.ErrTooFewShards) { return nil, nil @@ -1498,6 +1608,24 @@ func (s *slotState) complete() bool { } func (s *slotState) orderedShreds() []*Shred { + // Normal completed slots contain exactly the contiguous range 0..lastIndex. + // Keep the sparse fallback: malformed tails and focused partial-state callers + // must not silently lose shreds beyond the last-in-slot marker. + if s.haveLast && uint64(len(s.shreds)) == uint64(s.lastIndex)+1 { + out := make([]*Shred, len(s.shreds)) + for idx := range out { + shred := s.shreds[uint32(idx)] + if shred == nil || shred.Index != uint32(idx) { + return s.sortedShreds() + } + out[idx] = shred + } + return out + } + return s.sortedShreds() +} + +func (s *slotState) sortedShreds() []*Shred { indexes := make([]int, 0, len(s.shreds)) for idx := range s.shreds { indexes = append(indexes, int(idx)) @@ -1507,6 +1635,7 @@ func (s *slotState) orderedShreds() []*Shred { for _, idx := range indexes { out = append(out, s.shreds[uint32(idx)]) } + sort.Slice(out, func(i, j int) bool { return out[i].Index < out[j].Index }) return out } @@ -1514,7 +1643,7 @@ func (s *slotState) orderedShreds() []*Shred { // transaction signature verification. Parent/child identity hints are applied // later under the assembler lock so hints learned while this runs still win. func (s *slotState) decodeBlock(timings *entryDecodeTimings) (*block.Block, *AlpenglowParentInfo, []solana.Hash, error) { - entries, parentInfo, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(s.orderedShreds(), timings) + entries, parentInfo, footer, err := decodeEntriesFromOrderedDataShreds(s.orderedShreds(), timings) if err != nil { return nil, nil, nil, err } @@ -1660,35 +1789,36 @@ func (s *slotState) fecSetMerkleRoots() ([]solana.Hash, error) { } func (f *fecState) merkleRoot() (solana.Hash, bool, error) { - for _, idx := range sortedUint32Keys(f.data) { - shred := f.data[idx] - if shred == nil || shred.Recovered { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - if errors.Is(err, ErrUnsupportedShred) { - continue + // Preserve the old lowest-data-index, then lowest-coding-position choice, + // including its first error. Unsupported variants and recovered data have + // no usable proof. Selecting the minimum needs neither sorting nor scratch. + selected := f.data[0] + var first uint32 + // Relative index zero is already the minimum. Repair or legacy/malformed + // inputs may lack that proof and still take the general selection path. + if !hasMerkleRootProof(selected) || selected.Recovered { + selected = nil + for idx, shred := range f.data { + if hasMerkleRootProof(shred) && !shred.Recovered && (selected == nil || idx < first) { + selected, first = shred, idx } - return solana.Hash{}, false, err } - return root, true, nil } - for _, pos := range sortedUint16Keys(f.coding) { - shred := f.coding[pos] - if shred == nil { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - if errors.Is(err, ErrUnsupportedShred) { - continue + if selected == nil { + for pos, shred := range f.coding { + if hasMerkleRootProof(shred) && (selected == nil || uint32(pos) < first) { + selected, first = shred, uint32(pos) } - return solana.Hash{}, false, err } - return root, true, nil } - return solana.Hash{}, false, nil + if selected == nil { + return solana.Hash{}, false, nil + } + if cached := f.rootCache; cached != nil && cached.matches(selected) { + return cached.root, true, nil + } + root, err := selected.MerkleRoot() + return root, err == nil, err } func merkleTreeRoot(leaves []solana.Hash) solana.Hash { diff --git a/pkg/turbine/broadcast.go b/pkg/turbine/broadcast.go index fee229158..814951132 100644 --- a/pkg/turbine/broadcast.go +++ b/pkg/turbine/broadcast.go @@ -3,7 +3,6 @@ package turbine import ( "fmt" "net" - "sort" "sync" "github.com/gagliardetto/solana-go" @@ -116,8 +115,8 @@ type BroadcastSessionConfig struct { // It seeds the chained merkle root embedded in this slot's first FEC batch. ParentChainedMerkleRoot solana.Hash Broadcaster PacketBroadcaster - UserAgent []byte - Version uint16 + UserAgent []byte + Version uint16 } func NewBroadcastSession(cfg BroadcastSessionConfig) *BroadcastSession { @@ -192,7 +191,7 @@ func (s *BroadcastSession) broadcastComponent(component BlockComponent, isLastIn if s.broadcaster == nil { return nil } - batch, nextData, nextCode, err := s.shredder.MakeMerkleShredsFromComponent( + batch, nextData, nextCode, err := s.shredder.makeMerklePacketsFromComponent( s.leader, component, isLastInSlot, @@ -203,44 +202,9 @@ func (s *BroadcastSession) broadcastComponent(component BlockComponent, isLastIn if err != nil { return err } - s.chainedMerkleRoot = batch.ChainedMerkleRoot + s.chainedMerkleRoot = batch.chainedMerkleRoot s.nextDataIndex = nextData s.nextCodeIndex = nextCode - if len(batch.DataShreds) > 0 { - s.fecSetRoots = appendFECSetMerkleRoots(s.fecSetRoots, batch.DataShreds) - } - return s.broadcaster.Broadcast(batch.Packets) -} - -func appendFECSetMerkleRoots(roots []solana.Hash, dataShreds []*Shred) []solana.Hash { - if len(dataShreds) == 0 { - return roots - } - indices := make([]uint32, 0) - seen := make(map[uint32]struct{}) - for _, shred := range dataShreds { - if shred == nil { - continue - } - if _, ok := seen[shred.FECSetIndex]; ok { - continue - } - seen[shred.FECSetIndex] = struct{}{} - indices = append(indices, shred.FECSetIndex) - } - sort.Slice(indices, func(i, j int) bool { return indices[i] < indices[j] }) - for _, fecSetIndex := range indices { - for _, shred := range dataShreds { - if shred == nil || shred.FECSetIndex != fecSetIndex { - continue - } - root, err := shred.MerkleRoot() - if err != nil { - continue - } - roots = append(roots, root) - break - } - } - return roots + s.fecSetRoots = append(s.fecSetRoots, batch.fecSetRoots...) + return s.broadcaster.Broadcast(batch.packets) } diff --git a/pkg/turbine/broadcast_test.go b/pkg/turbine/broadcast_test.go index 09e82a397..11c211c76 100644 --- a/pkg/turbine/broadcast_test.go +++ b/pkg/turbine/broadcast_test.go @@ -52,6 +52,70 @@ func TestBroadcastSessionHeaderAndFooter(t *testing.T) { require.NotEqual(t, parentBlockID, chainedRoot) } +func TestBroadcastSessionMatchesParsedShreds(t *testing.T) { + leader := testBroadcastLeader(t) + parentID, parentRoot := solana.Hash{0xaa}, solana.Hash{0xbb} + capture := &packetCapture{} + session := NewBroadcastSession(BroadcastSessionConfig{ + Leader: leader, Slot: 100, ParentSlot: 99, Version: 7, + ParentChainedMerkleRoot: parentRoot, Broadcaster: capture, + }) + shredder := Shredder{Slot: 100, ParentSlot: 99, Version: 7} + txns := make([]solana.Transaction, 400) + for i := range txns { + txns[i] = mustParseTransferTx(t, uint64(i)) + } + entries, err := NewEntryBatch([]Entry{{NumHashes: 1, Hash: solana.Hash{1}, Txns: txns}}) + require.NoError(t, err) + tick, err := NewEntryBatch([]Entry{{NumHashes: 1, Hash: solana.Hash{2}}}) + require.NoError(t, err) + components := []BlockComponent{ + NewBlockHeader(99, parentID), + entries, // More than two FEC sets: every root must enter the block ID. + NewUpdateParent(98, solana.Hash{3}), + NewBlockFooter(BlockFooter{BankHash: solana.Hash{4}}), + tick, + } + var nextData, nextCode uint32 + root := parentRoot + var roots []solana.Hash + for i, component := range components { + last := i == len(components)-1 + batch, data, code, err := shredder.MakeMerkleShredsFromComponent( + leader, component, last, root, nextData, nextCode, + ) + require.NoError(t, err) + if i == 1 { + require.Greater(t, len(batch.DataShreds), 2*dataShredsPerFECBlock) + } + for j, shred := range batch.DataShreds { + if j == 0 || shred.FECSetIndex != batch.DataShreds[j-1].FECSetIndex { + fecRoot, err := shred.MerkleRoot() + require.NoError(t, err) + roots = append(roots, fecRoot) + } + } + capture.packets = nil + require.NoError(t, session.BroadcastComponent(component, last)) + require.Equal(t, batch.Packets, capture.packets) + require.Equal(t, batch.ChainedMerkleRoot, session.ChainedMerkleRoot()) + require.Equal(t, data, session.nextDataIndex) + require.Equal(t, code, session.nextCodeIndex) + require.Equal(t, roots, session.fecSetRoots) + require.Equal(t, DoubleMerkleBlockID(99, parentID, roots), session.BlockID(99, parentID)) + root, nextData, nextCode = batch.ChainedMerkleRoot, data, code + } + + // Invalid components must not publish packets or advance the commitment. + capture.packets = nil + require.Error(t, session.BroadcastComponent(BlockComponent{Marker: &BlockMarker{Kind: 255}}, false)) + require.Empty(t, capture.packets) + require.Equal(t, root, session.ChainedMerkleRoot()) + require.Equal(t, nextData, session.nextDataIndex) + require.Equal(t, nextCode, session.nextCodeIndex) + require.Equal(t, roots, session.fecSetRoots) +} + func TestUDPBroadcasterLoopback(t *testing.T) { recvAddr, err := net.ResolveUDPAddr("udp", "127.0.0.1:0") require.NoError(t, err) diff --git a/pkg/turbine/cancellation_regression_test.go b/pkg/turbine/cancellation_regression_test.go index 1f85e4df9..719931216 100644 --- a/pkg/turbine/cancellation_regression_test.go +++ b/pkg/turbine/cancellation_regression_test.go @@ -17,86 +17,52 @@ func TestTransactionVerifierCancellationDuringBlockedAdmissionJoinsAdmittedJobs( const workers = 2 blocker := verifierTestBlock(workers) target := verifierTestBlock(2 * workers) - filler := &solana.Transaction{} - release := make(chan struct{}) var releaseOnce sync.Once started := make(chan struct{}, workers) var targetFirstCalls atomic.Int32 var targetLaterCalls atomic.Int32 - - verifier := newTransactionVerifier(workers, workers, func(tx *solana.Transaction) error { + // Single-transaction groups make the blocked-admission boundary exact. + verifier := newTransactionVerifierWithBatchTarget(workers, 1, 1, func(tx *solana.Transaction) error { switch tx { case blocker.Transactions[0], blocker.Transactions[1]: started <- struct{}{} <-release case target.Transactions[0]: targetFirstCalls.Add(1) - case target.Transactions[1], target.Transactions[2], target.Transactions[3]: + default: targetLaterCalls.Add(1) } return nil }) - + defer verifier.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) ctx, cancel := context.WithCancel(context.Background()) + defer cancel() blockerDone := make(chan error, 1) targetDone := make(chan error, 1) - var calls sync.WaitGroup - calls.Add(1) - go func() { - defer calls.Done() - blockerDone <- verifier.verifyBlock(blocker) - }() + go func() { blockerDone <- verifier.verifyBlock(blocker) }() for range workers { - select { - case <-started: - case <-time.After(3 * time.Second): - cancel() - releaseOnce.Do(func() { close(release) }) - calls.Wait() - verifier.closeAndWait() - t.Fatal("timed out occupying transaction verifier workers") - } + waitSignal(t, started, "occupied verifier worker") } - // Leave one queued job ahead of the target. With both workers occupied and - // a two-entry queue, the target admits transaction 0 and then blocks trying - // to admit transaction 1. Cancellation must wait for transaction 0 to drain, - // while transactions 1 and all later chunks must never reach a worker. - var fillerErr error - var fillerDone sync.WaitGroup - fillerDone.Add(1) - verifier.jobs <- transactionVerifyJob{tx: filler, err: &fillerErr, done: &fillerDone} - - calls.Add(1) - go func() { - defer calls.Done() - targetDone <- verifier.verifyBlockContext(ctx, target) - }() + // Both workers are occupied. The target's first group fills the one-entry + // queue, and its second blocks on admission. Cancellation must still join + // the first group while preventing any later transactions from running. + go func() { targetDone <- verifier.verifyBlockContext(ctx, target) }() deadline := time.Now().Add(3 * time.Second) for len(verifier.jobs) != cap(verifier.jobs) { if time.Now().After(deadline) { - cancel() - releaseOnce.Do(func() { close(release) }) - calls.Wait() - fillerDone.Wait() - verifier.closeAndWait() - t.Fatalf("transaction queue did not fill: len=%d cap=%d", len(verifier.jobs), cap(verifier.jobs)) + t.Fatal("transaction queue did not fill") } time.Sleep(time.Millisecond) } - cancel() select { case err := <-targetDone: - releaseOnce.Do(func() { close(release) }) - calls.Wait() - fillerDone.Wait() - verifier.closeAndWait() t.Fatalf("canceled verifier returned before its admitted job joined: %v", err) case <-time.After(50 * time.Millisecond): } - releaseOnce.Do(func() { close(release) }) select { case err := <-targetDone: @@ -114,13 +80,6 @@ func TestTransactionVerifierCancellationDuringBlockedAdmissionJoinsAdmittedJobs( case <-time.After(3 * time.Second): t.Fatal("blocker verification did not drain") } - fillerDone.Wait() - calls.Wait() - verifier.closeAndWait() - - if fillerErr != nil { - t.Fatalf("filler verification: %v", fillerErr) - } if got := targetFirstCalls.Load(); got != 1 { t.Fatalf("admitted target transaction calls = %d, want 1", got) } diff --git a/pkg/turbine/cluster_nodes.go b/pkg/turbine/cluster_nodes.go index 8446a3966..ff6e0bdc0 100644 --- a/pkg/turbine/cluster_nodes.go +++ b/pkg/turbine/cluster_nodes.go @@ -89,6 +89,13 @@ func newClusterNodes(cfg ClusterNodesConfig, broadcast bool) *ClusterNodes { // shuffle because it already owns the shred. Tree placement and fanout match // Agave ClusterNodes::get_retransmit_addrs. func (c *ClusterNodes) RetransmitPeers(leader solana.PublicKey, shred ShredID, fanout int) (uint8, []*net.UDPAddr, error) { + return c.retransmitPeersInto(leader, shred, fanout, nil) +} + +// retransmitPeersInto uses caller-owned result storage. The caller must finish +// using the returned slice before reusing dst; addresses still belong to this +// immutable cluster snapshot. Public callers retain the allocating API above. +func (c *ClusterNodes) retransmitPeersInto(leader solana.PublicKey, shred ShredID, fanout int, dst []*net.UDPAddr) (uint8, []*net.UDPAddr, error) { if c == nil || fanout <= 0 { return maxTurbineHops - 1, nil, nil } @@ -123,7 +130,10 @@ func (c *ClusterNodes) RetransmitPeers(leader solana.PublicKey, shred ShredID, f step = 1 } position := anchor*fanout + offset + 1 - peers := make([]*net.UDPAddr, 0, fanout) + peers := dst[:0] + if dst == nil { + peers = make([]*net.UDPAddr, 0, fanout) + } shufflePosition := selfPos for range fanout { var index int diff --git a/pkg/turbine/completion_order_root_test.go b/pkg/turbine/completion_order_root_test.go new file mode 100644 index 000000000..5fd8fd7f7 --- /dev/null +++ b/pkg/turbine/completion_order_root_test.go @@ -0,0 +1,228 @@ +package turbine + +import ( + "bytes" + "context" + "errors" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestCompletedShredOrderPreservesSparseAndExtraTails(t *testing.T) { + for _, indexes := range [][]uint32{{0, 1, 2}, {0, 2, 3}, {0, 1, 2, 9}, {1, 3, 7}} { + s := &slotState{haveLast: true, lastIndex: 2, shreds: make(map[uint32]*Shred)} + for _, idx := range indexes { + s.shreds[idx] = &Shred{Index: idx, Type: ShredTypeData} + } + got := s.orderedShreds() + require.Len(t, got, len(indexes)) + for i, idx := range indexes { + require.Same(t, s.shreds[idx], got[i]) + } + } +} + +func TestOrderedEntryDecodeMatchesPublicUnorderedDecode(t *testing.T) { + packets := agavePaddedSlot1752420Packets(t) + s := &slotState{shreds: make(map[uint32]*Shred)} + var shuffled []*Shred + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + s.shreds[shred.Index] = shred + shuffled = append(shuffled, shred) + } + for i, j := 0, len(shuffled)-1; i < j; i, j = i+1, j-1 { + shuffled[i], shuffled[j] = shuffled[j], shuffled[i] + } + want, parent, footer, err := DecodeEntriesAndAlpenglowMarkersFromDataShreds(shuffled) + require.NoError(t, err) + got, gotParent, gotFooter, err := decodeEntriesFromOrderedDataShreds(s.orderedShreds(), nil) + require.NoError(t, err) + require.Equal(t, want, got) + require.Equal(t, parent, gotParent) + require.Equal(t, footer, gotFooter) +} + +func TestAuthenticatedFECRootRejectsChangedInputAndReplacement(t *testing.T) { + shred, leader := buildSignedTestShred(t, 100, 42) + var verifier shredSigCache + root, err := verifier.verifyShredRoot(shred, leader) + require.NoError(t, err) + f := &fecState{data: map[uint32]*Shred{0: shred}} + f.rememberAuthenticatedRoot(shred, root) + cached := f.rootCache + require.True(t, cached.matches(shred)) + for i := range shred.Payload { + shred.Payload[i] ^= 1 + require.False(t, cached.matches(shred), "payload byte %d", i) + shred.Payload[i] ^= 1 + } + for _, mutate := range []func(*Shred){ + func(s *Shred) { s.Variant ^= 1 }, func(s *Shred) { s.Type = ShredTypeCode }, + func(s *Shred) { s.Index++ }, func(s *Shred) { s.FECSetIndex++ }, + func(s *Shred) { s.NumDataShreds++ }, func(s *Shred) { s.Position++ }, + } { + original := *shred + mutate(shred) + require.False(t, cached.matches(shred)) + *shred = original + } + shred.Payload[dataHeaderSize] ^= 1 + want, err := shred.MerkleRoot() + require.NoError(t, err) + got, ok, err := f.merkleRoot() + require.NoError(t, err) + require.True(t, ok) + require.Equal(t, want, got) + require.NotEqual(t, root, got) + _, err = verifier.verifyShredRoot(shred, leader) + require.ErrorIs(t, err, ErrInvalidSignature) + shred.Payload[dataHeaderSize] ^= 1 + replacement := *shred + require.False(t, cached.matches(&replacement)) + f.data[0] = &replacement + got, ok, err = f.merkleRoot() + require.NoError(t, err) + require.True(t, ok) + require.Equal(t, root, got) +} + +func TestAuthenticatedFECRootAdmissionAndReset(t *testing.T) { + shred, leader := buildSignedTestShred(t, 100, 42) + var verifier shredSigCache + root, err := verifier.verifyShredRoot(shred, leader) + require.NoError(t, err) + a := NewSlotAssembler() + work, err := a.addShredFromWithRoot(shred, false, &root) + require.NoError(t, err) + require.NotNil(t, work) + f := work.state.fecSets[0] + require.True(t, f.rootCache.matches(shred)) + // A duplicate cannot overwrite the admitted root, even while completion is + // retried after cancellation. Reset starts a separate FEC generation. + a.abortCompletion(work) + wrong := solana.Hash{99} + _, err = a.addShredFromWithRoot(shred, true, &wrong) + require.NoError(t, err) + require.Equal(t, root, f.rootCache.root) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + require.True(t, a.processCompletion(ctx, work).canceled) + a.ResetSlot(100) + work, err = a.addShredFrom(shred, false) + require.NoError(t, err) + require.NotNil(t, work) + require.Nil(t, work.state.fecSets[0].rootCache) +} + +func referenceFECRoot(f *fecState) (solana.Hash, bool, error) { + for _, idx := range sortedUint32Keys(f.data) { + s := f.data[idx] + if s == nil || s.Recovered { + continue + } + root, err := s.MerkleRoot() + if errors.Is(err, ErrUnsupportedShred) { + continue + } + return root, err == nil, err + } + for _, idx := range sortedUint16Keys(f.coding) { + s := f.coding[idx] + if s == nil { + continue + } + root, err := s.MerkleRoot() + if errors.Is(err, ErrUnsupportedShred) { + continue + } + return root, err == nil, err + } + return solana.Hash{}, false, nil +} + +func FuzzFECRootSelectionMatchesSortedReference(f *testing.F) { + f.Add([]byte{0, 1, 2, 3, 4, 5, 6, 7}) + f.Add([]byte{9, 0, 0, 0, 2, 1, 3, 7}) + f.Fuzz(func(t *testing.T, choices []byte) { + if len(choices) > 512 { + t.Skip() + } + state := &fecState{data: make(map[uint32]*Shred), coding: make(map[uint16]*Shred)} + for i, choice := range choices { + s := &Shred{Variant: merkleDataVariant, Type: ShredType(choice % 3), Payload: bytes.Repeat([]byte{choice}, dataPayloadSize)} + switch choice % 5 { + case 0: + s = nil + case 1: + s.Recovered = true + case 2: + s.Variant = legacyDataVariant + case 3: + s.Payload = s.Payload[:20] + } + if i%2 == 0 { + state.data[uint32(choice)] = s + } else { + state.coding[uint16(choice)] = s + } + } + want, wantOK, wantErr := referenceFECRoot(state) + got, gotOK, gotErr := state.merkleRoot() + require.Equal(t, want, got) + require.Equal(t, wantOK, gotOK) + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + }) +} + +func TestFECRootCacheKeepsDeterministicDataPrecedence(t *testing.T) { + packets := append(localnetMerkleShreds(t, "d"), localnetMerkleShreds(t, "c")...) + f := &fecState{data: make(map[uint32]*Shred), coding: make(map[uint16]*Shred)} + var shreds []*Shred + for _, packet := range packets { + s, err := ParseShred(packet) + require.NoError(t, err) + if s.FECSetIndex != 0 { + continue + } + shreds = append(shreds, s) + } + require.NotEmpty(t, shreds) + sort.Slice(shreds, func(i, j int) bool { + if shreds[i].Type != shreds[j].Type { + return shreds[i].Type == ShredTypeCode + } + return shreds[i].Index > shreds[j].Index + }) + for _, s := range shreds { + root, err := s.MerkleRoot() + require.NoError(t, err) + if s.Type == ShredTypeData { + f.data[s.Index-s.FECSetIndex] = s + } else { + f.coding[s.Position] = s + } + f.rememberAuthenticatedRoot(s, root) + want, wantOK, wantErr := referenceFECRoot(f) + got, gotOK, gotErr := f.merkleRoot() + require.Equal(t, wantErr, gotErr) + require.Equal(t, wantOK, gotOK) + require.Equal(t, want, got) + } + for _, s := range f.data { + s.Recovered = true + } + want, wantOK, wantErr := referenceFECRoot(f) + got, gotOK, gotErr := f.merkleRoot() + require.Equal(t, wantErr, gotErr) + require.Equal(t, wantOK, gotOK) + require.Equal(t, want, got) +} diff --git a/pkg/turbine/component_shredder.go b/pkg/turbine/component_shredder.go index ad4c9229b..109408a4c 100644 --- a/pkg/turbine/component_shredder.go +++ b/pkg/turbine/component_shredder.go @@ -14,7 +14,7 @@ type Shredder struct { ReferenceTick uint8 } -// ShredBatch is one FEC batch emitted for a single block component. +// ShredBatch contains the FEC batches emitted for a single block component. type ShredBatch struct { Slot uint64 Component BlockComponent @@ -34,23 +34,8 @@ func (s *Shredder) MakeMerkleShredsFromComponent( nextShredIndex uint32, nextCodeIndex uint32, ) (ShredBatch, uint32, uint32, error) { - bytes, err := MarshalBlockComponent(component) - if err != nil { - return ShredBatch{}, nextShredIndex, nextCodeIndex, err - } - gen := ShredGenerator{ - Slot: s.Slot, - ParentSlot: s.ParentSlot, - Version: s.Version, - ReferenceTick: s.ReferenceTick, - } - packets, root, nextData, nextCode, err := gen.MakeShredsFromData( - leader, - bytes, - isLastInSlot, - chainedMerkleRoot, - nextShredIndex, - nextCodeIndex, + generated, nextData, nextCode, err := s.makeMerklePacketsFromComponent( + leader, component, isLastInSlot, chainedMerkleRoot, nextShredIndex, nextCodeIndex, ) if err != nil { return ShredBatch{}, nextShredIndex, nextCodeIndex, err @@ -58,11 +43,11 @@ func (s *Shredder) MakeMerkleShredsFromComponent( batch := ShredBatch{ Slot: s.Slot, Component: component, - Packets: packets, - ChainedMerkleRoot: root, + Packets: generated.packets, + ChainedMerkleRoot: generated.chainedMerkleRoot, IsLastInSlot: isLastInSlot, } - for _, packet := range packets { + for _, packet := range batch.Packets { shred, err := ParseShred(packet) if err != nil { return ShredBatch{}, nextShredIndex, nextCodeIndex, fmt.Errorf("parse generated shred: %w", err) @@ -75,3 +60,37 @@ func (s *Shredder) MakeMerkleShredsFromComponent( } return batch, nextData, nextCode, nil } + +// makeMerklePacketsFromComponent serves the producer, which needs the wire +// packets and FEC roots but not the owning Shred objects exposed by the public API. +func (s *Shredder) makeMerklePacketsFromComponent( + leader solana.PrivateKey, + component BlockComponent, + isLastInSlot bool, + chainedMerkleRoot solana.Hash, + nextShredIndex uint32, + nextCodeIndex uint32, +) (shredPackets, uint32, uint32, error) { + bytes, err := MarshalBlockComponent(component) + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err + } + gen := ShredGenerator{ + Slot: s.Slot, + ParentSlot: s.ParentSlot, + Version: s.Version, + ReferenceTick: s.ReferenceTick, + } + batch, nextData, nextCode, err := gen.makeShredsFromData( + leader, + bytes, + isLastInSlot, + chainedMerkleRoot, + nextShredIndex, + nextCodeIndex, + ) + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err + } + return batch, nextData, nextCode, nil +} diff --git a/pkg/turbine/component_test.go b/pkg/turbine/component_test.go index 64b05760c..2617715be 100644 --- a/pkg/turbine/component_test.go +++ b/pkg/turbine/component_test.go @@ -131,6 +131,36 @@ func TestShredEntryBatchRoundTrip(t *testing.T) { require.Equal(t, entry.NumHashes, components[0].EntryBatch[0].NumHashes) } +func TestShredMultiFECEntryBatchRoundTrip(t *testing.T) { + leader := testLeader(t) + entries := make([]turbine.Entry, 1300) + for i := range entries { + entries[i] = turbine.Entry{NumHashes: 1, Hash: solana.Hash{byte(i), byte(i >> 8)}} + } + component, err := turbine.NewEntryBatch(entries) + require.NoError(t, err) + + shredder := turbine.Shredder{Slot: 100, ParentSlot: 99, Version: 42, ReferenceTick: 63} + batch, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, true, solana.Hash{}, 0, 0, + ) + require.NoError(t, err) + require.Greater(t, len(batch.DataShreds), 32) + for i, shred := range batch.DataShreds[:len(batch.DataShreds)-1] { + require.False(t, shred.DataComplete(), "intermediate data shred %d ended the component", i) + } + require.True(t, batch.DataShreds[len(batch.DataShreds)-1].DataComplete()) + require.True(t, batch.DataShreds[len(batch.DataShreds)-1].LastInSlot()) + + components, err := turbine.DecodeComponentsFromDataShreds(batch.DataShreds) + require.NoError(t, err) + require.Len(t, components, 1) + require.Len(t, components[0].EntryBatch, len(entries)) + for i := range entries { + require.Equal(t, entries[i].Hash, components[0].EntryBatch[i].Hash) + } +} + func TestShredBlockHeaderMarkerRoundTrip(t *testing.T) { leader := testLeader(t) parentID := solana.Hash{8} diff --git a/pkg/turbine/entries.go b/pkg/turbine/entries.go index 9f6251318..e9372204d 100644 --- a/pkg/turbine/entries.go +++ b/pkg/turbine/entries.go @@ -1,6 +1,8 @@ package turbine import ( + "bytes" + "context" "encoding/binary" "fmt" "sort" @@ -40,6 +42,12 @@ type AlpenglowParentInfo struct { type entryDecodeTimings struct { transactionParse time.Duration + ctx context.Context + prefetched map[uint32]*prefetchedShredBatch + retained []*prefetchedShredBatch + all []*prefetchedShredBatch + prefetchWait time.Duration + traceFallback *transactionVerification } func (e *Entry) UnmarshalWithDecoder(decoder *bin.Decoder) error { @@ -124,69 +132,109 @@ func decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds []*Shred, timings *en sort.Slice(shreds, func(i, j int) bool { return shreds[i].Index < shreds[j].Index }) + return decodeEntriesFromOrderedDataShreds(shreds, timings) +} - type decodedEntryBatch struct { - start uint32 - entries []Entry - } - var entryBatches []decodedEntryBatch +// decodeEntriesFromOrderedDataShreds requires increasing shred indexes. The +// assembler supplies that order directly; public decoding still sorts input. +func decodeEntriesFromOrderedDataShreds(shreds []*Shred, timings *entryDecodeTimings) ([]Entry, *AlpenglowParentInfo, *BlockFooter, error) { + var entryBatches []*prefetchedShredBatch var parentInfo *AlpenglowParentInfo var blockFooter *BlockFooter - var batchBytes []byte var batchStart uint32 + var batchStartPos, batchSize int var haveBatch bool - for _, shred := range shreds { + for shredPos, shred := range shreds { if shred == nil || shred.Type != ShredTypeData { continue } if !haveBatch { batchStart = shred.Index + batchStartPos = shredPos haveBatch = true } - batchBytes = append(batchBytes, shred.Data...) + batchSize += len(shred.Data) if !shred.DataComplete() { continue } - if parent, footer, ok, err := decodeAlpenglowMarkerFromShredBatch(batchBytes, batchStart); err != nil { - return nil, nil, nil, fmt.Errorf("decode alpenglow block marker ending at shred %d: %w", shred.Index, err) - } else if ok { - if parent != nil { - parentInfo, err = mergeAlpenglowParentInfo(parentInfo, parent) - if err != nil { - return nil, nil, nil, fmt.Errorf("merge alpenglow parent marker ending at shred %d: %w", shred.Index, err) + batchShreds := shreds[batchStartPos : shredPos+1] + var batch *prefetchedShredBatch + if timings != nil { + if cached := timings.prefetched[batchStart]; cached != nil && cached.start == batchStart && cached.end == shred.Index { + ctx := timings.ctx + if ctx == nil { + ctx = context.Background() + } + if cached.ready != nil { + waitStarted := time.Now() + select { + case <-cached.ready: + case <-ctx.Done(): + timings.prefetchWait += time.Since(waitStarted) + return nil, nil, nil, ctx.Err() + } + timings.prefetchWait += time.Since(waitStarted) + } + if err := ctx.Err(); err != nil { + return nil, nil, nil, err + } + // Bounds alone cannot prove identity after repair or replacement. + // Read decoded fields only after the preparation channel closes. + // Compare the original slices directly: a cache hit needs no + // second component buffer or copies of already decoded bytes. + if len(cached.raw) == batchSize && dataShredBatchMatches(batchShreds, cached.raw) { + batch = cached } } - if footer != nil { - blockFooter = footer + } + if batch == nil { + // A miss owns a fresh, exactly sized backing array. Transactions + // retain instruction-data slices into it after this call returns. + var traceStart int64 + if timings != nil && entryTraceContext(timings.ctx) { + traceStart = entryTraceNow() + } + batchBytes := make([]byte, 0, batchSize) + for _, part := range batchShreds { + if part != nil && part.Type == ShredTypeData { + batchBytes = append(batchBytes, part.Data...) + } + } + batch = decodeClosedShredBatch(batchBytes, batchStart, shred.Index) + if traceStart != 0 { + batch.traceDecodeStart, batch.traceDecodeEnd = traceStart, entryTraceNow() + } + if timings != nil { + timings.transactionParse += batch.parseDuration } - batchBytes = nil - haveBatch = false - continue } - parseStart := time.Now() - batchEntries, consumed, err := decodeEntryBatchPrefix(batchBytes) if timings != nil { - timings.transactionParse += time.Since(parseStart) + timings.all = append(timings.all, batch) } - // A zero entry count with more bytes denotes a marker. Preserve the - // rejection of unrecognized or misplaced markers in this fallback path. - if err == nil && len(batchEntries) == 0 && consumed != len(batchBytes) { - err = fmt.Errorf("entry batch has %d trailing bytes", len(batchBytes)-consumed) + if batch.err != nil { + return nil, nil, nil, batch.err } - if err != nil { - return nil, nil, nil, fmt.Errorf("decode entry batch ending at shred %d: %w", shred.Index, err) + if batch.marker { + if batch.parent != nil { + var err error + parentInfo, err = mergeAlpenglowParentInfo(parentInfo, batch.parent) + if err != nil { + return nil, nil, nil, fmt.Errorf("merge alpenglow parent marker ending at shred %d: %w", shred.Index, err) + } + } + if batch.footer != nil { + blockFooter = batch.footer + } + batchSize = 0 + haveBatch = false + continue } - entryBatches = append(entryBatches, decodedEntryBatch{ - start: batchStart, - entries: batchEntries, - }) - // Decoded transactions retain slices into the batch buffer for instruction data. - // Keep the backing array alive instead of reusing and overwriting it. - batchBytes = nil + entryBatches = append(entryBatches, batch) + batchSize = 0 haveBatch = false } - if len(batchBytes) != 0 { - return nil, nil, nil, fmt.Errorf("slot ended with %d undecoded entry bytes", len(batchBytes)) + if batchSize != 0 { + return nil, nil, nil, fmt.Errorf("slot ended with %d undecoded entry bytes", batchSize) } var entries []Entry for _, batch := range entryBatches { @@ -199,10 +247,60 @@ func decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds []*Shred, timings *en continue } entries = append(entries, batch.entries...) + if timings != nil { + timings.retained = append(timings.retained, batch) + } } return entries, parentInfo, blockFooter, nil } +// dataShredBatchMatches is equivalent to comparing raw with the concatenated +// data bytes, including all padding. Neither equal prefixes nor changed lengths +// can reuse a cached signature verdict. It does not retain or allocate buffers. +func dataShredBatchMatches(shreds []*Shred, raw []byte) bool { + offset := 0 + for _, shred := range shreds { + if shred == nil || shred.Type != ShredTypeData { + continue + } + if len(shred.Data) > len(raw)-offset || !bytes.Equal(shred.Data, raw[offset:offset+len(shred.Data)]) { + return false + } + offset += len(shred.Data) + } + return offset == len(raw) +} + +// decodeClosedShredBatch owns raw through the returned decoded transactions. +// It decodes the same padded component envelope for early and full assembly; +// marker merging and UpdateParent selection still require the full slot. +func decodeClosedShredBatch(raw []byte, start, end uint32) *prefetchedShredBatch { + batch := &prefetchedShredBatch{start: start, end: end, raw: raw} + parent, footer, marker, err := decodeAlpenglowMarkerFromShredBatch(raw, start) + if err != nil { + batch.err = fmt.Errorf("decode alpenglow block marker ending at shred %d: %w", end, err) + return batch + } + if marker { + batch.parent, batch.footer, batch.marker = parent, footer, true + return batch + } + parseStarted := time.Now() + entries, consumed, err := decodeEntryBatchPrefix(raw) + batch.parseDuration = time.Since(parseStarted) + // A zero entry count with more bytes denotes a marker. Preserve rejection + // of unknown or misplaced markers instead of treating them as an empty batch. + if err == nil && len(entries) == 0 && consumed != len(raw) { + err = fmt.Errorf("entry batch has %d trailing bytes", len(raw)-consumed) + } + if err != nil { + batch.err = fmt.Errorf("decode entry batch ending at shred %d: %w", end, err) + return batch + } + batch.entries = entries + return batch +} + func decodeEntryBatch(data []byte) ([]Entry, error) { entries, consumed, err := decodeEntryBatchPrefix(data) if err != nil { diff --git a/pkg/turbine/entries_direct_compare_test.go b/pkg/turbine/entries_direct_compare_test.go new file mode 100644 index 000000000..4fb55637e --- /dev/null +++ b/pkg/turbine/entries_direct_compare_test.go @@ -0,0 +1,100 @@ +package turbine + +import ( + "bytes" + "context" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestDataShredBatchMatchesChecksLengthsAndEverySlice(t *testing.T) { + raw := []byte("0123456789abcdefghijklmnopqrstuvwxyz") + shreds := []*Shred{ + {Type: ShredTypeData, Data: raw[:10]}, + nil, + {Type: ShredTypeCode, Data: []byte("coding bytes are not component data")}, + {Type: ShredTypeData}, + {Type: ShredTypeData, Data: raw[10:20]}, + {Type: ShredTypeData, Data: raw[20:]}, + } + require.True(t, dataShredBatchMatches(shreds, bytes.Clone(raw))) + for i := range raw { + changed := bytes.Clone(raw) + changed[i] ^= 1 + require.False(t, dataShredBatchMatches(shreds, changed), "changed byte %d", i) + } + for i := 0; i < len(raw); i++ { + require.False(t, dataShredBatchMatches(shreds, raw[:i]), "truncated at %d", i) + } + require.False(t, dataShredBatchMatches(shreds, append(bytes.Clone(raw), 0))) + require.True(t, dataShredBatchMatches(nil, nil)) + require.False(t, dataShredBatchMatches(nil, raw)) +} + +func TestEntryDecodeDirectComparisonCannotReuseChangedSignatureVerdict(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 1) + raw := prefetchTestPayload(t, txs) + cache := decodedPrefetchForTest(paddedComponentShreds(t, raw, 0, 0xa5)) + cached := cache[0] + var err error + cached.verification, err = v.submitTransactions(context.Background(), entryBatchTransactions(cached.entries)) + require.NoError(t, err) + _, err = cached.verification.wait() + require.NoError(t, err) + + bad := *txs[0] + bad.Signatures = append([]solana.Signature(nil), bad.Signatures...) + bad.Signatures[0][0] ^= 1 + changed := prefetchTestPayload(t, []*solana.Transaction{&bad}) + require.Len(t, changed, len(raw)) + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(paddedComponentShreds(t, changed, 0, 0xa5), &timings) + require.NoError(t, err) + require.Len(t, timings.retained, 1) + require.NotSame(t, cached, timings.retained[0]) + require.Nil(t, timings.retained[0].verification) + blk := BlockFromEntries(100, 99, entries) + require.Equal(t, bad.Signatures[0], blk.Transactions[0].Signatures[0]) + require.ErrorContains(t, verifyDecodedEntryBatches(context.Background(), blk, timings.retained, v), "failed signature verification") + require.False(t, blk.TransactionSignaturesVerified()) +} + +func TestEntryDecodeDirectComparisonRejectsChangedCachedLength(t *testing.T) { + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for _, delta := range []int{-1, 1} { + shreds := paddedComponentShreds(t, raw, 0, 0xa5) + cache := decodedPrefetchForTest(shreds) + cached := cache[0] + if delta < 0 { + cached.raw = cached.raw[:len(cached.raw)-1] + } else { + cached.raw = append(cached.raw, 0) + } + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.NotSame(t, cached, timings.retained[0]) + } +} + +func FuzzDataShredBatchMatchesConcatenatedBytes(f *testing.F) { + f.Add([]byte("a component spanning several shreds"), []byte("a component spanning several shreds"), uint8(3)) + f.Add([]byte("truncated"), []byte("truncate"), uint8(1)) + f.Add([]byte{}, []byte{}, uint8(0)) + f.Fuzz(func(t *testing.T, data, candidate []byte, split uint8) { + if len(data) > 64*1024 || len(candidate) > 64*1024 { + t.Skip() + } + shreds := []*Shred{nil, {Type: ShredTypeCode, Data: []byte{1, 2, 3}}} + width := int(split) + 1 + for start := 0; start < len(data); start += width { + shreds = append(shreds, &Shred{Type: ShredTypeData, Data: data[start:min(start+width, len(data))]}) + } + require.Equal(t, bytes.Equal(data, candidate), dataShredBatchMatches(shreds, candidate)) + }) +} diff --git a/pkg/turbine/entries_prefetch_test.go b/pkg/turbine/entries_prefetch_test.go new file mode 100644 index 000000000..9ecdbf599 --- /dev/null +++ b/pkg/turbine/entries_prefetch_test.go @@ -0,0 +1,146 @@ +package turbine + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func decodedPrefetchForTest(shreds []*Shred) map[uint32]*prefetchedShredBatch { + cache := make(map[uint32]*prefetchedShredBatch) + var raw []byte + var start uint32 + for _, shred := range shreds { + if raw == nil { + start = shred.Index + } + raw = append(raw, shred.Data...) + if shred.DataComplete() { + batch := decodeClosedShredBatch(raw, start, shred.Index) + batch.ready = make(chan struct{}) + close(batch.ready) + cache[start] = batch + raw = nil + } + } + return cache +} + +func TestEntryDecodeReusesOnlyExactPrefetchedBytesAndBounds(t *testing.T) { + component, err := NewEntryBatch([]Entry{{Hash: solana.Hash{9}, Txns: []solana.Transaction{mustParseTransferTx(t, 21)}}}) + require.NoError(t, err) + raw, err := MarshalBlockComponent(component) + require.NoError(t, err) + for _, mode := range []string{"match", "different_bytes", "different_end", "different_start"} { + t.Run(mode, func(t *testing.T) { + shreds := paddedComponentShreds(t, raw, 0, 0xa5) + cache := decodedPrefetchForTest(shreds) + cached := cache[0] + require.NoError(t, cached.err) + cached.parseDuration = time.Hour // must not enter final parse accounting + if mode != "match" { + cached.err = errors.New("stale cached failure must be ignored") + } + switch mode { + case "different_bytes": + cached.raw[len(cached.raw)-1] ^= 1 // even differing padding invalidates reuse + case "different_end": + cached.end++ + cached.ready = make(chan struct{}) // mismatched bounds must never wait + case "different_start": + cached.start++ + cached.ready = make(chan struct{}) + } + timings := entryDecodeTimings{prefetched: cache} + entries, _, _, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.Len(t, timings.all, 1) + require.Len(t, timings.retained, 1) + if mode == "match" { + require.Same(t, cached, timings.retained[0]) + require.Same(t, &cached.entries[0].Txns[0], &entries[0].Txns[0]) + require.Zero(t, timings.transactionParse) + } else { + require.NotSame(t, cached, timings.retained[0]) + require.Less(t, timings.transactionParse, time.Hour) + } + }) + } +} + +func TestEntryDecodePrefetchPreparationWaitHonorsCancellation(t *testing.T) { + raw, err := marshalEntryBatch([]Entry{{Hash: solana.Hash{3}}}) + require.NoError(t, err) + shreds := paddedComponentShreds(t, raw, 0, 0) + cache := decodedPrefetchForTest(shreds) + cache[0].ready = make(chan struct{}) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + timings := entryDecodeTimings{ctx: ctx, prefetched: cache} + _, _, _, err = decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.ErrorIs(t, err, context.Canceled) + require.Empty(t, timings.retained) + require.Zero(t, timings.transactionParse) +} + +func TestEntryDecodePrefetchPreservesUpdateParentAndIgnoresVerificationErrors(t *testing.T) { + prefix, err := NewEntryBatch([]Entry{{Hash: solana.Hash{1}, Txns: []solana.Transaction{mustParseTransferTx(t, 20)}}}) + require.NoError(t, err) + suffix, err := NewEntryBatch([]Entry{{Hash: solana.Hash{2}, Txns: []solana.Transaction{mustParseTransferTx(t, 21)}}}) + require.NoError(t, err) + components := []BlockComponent{NewBlockHeader(99, solana.Hash{9}), prefix, NewUpdateParent(98, solana.Hash{8}), suffix, NewBlockFooter(BlockFooter{BankHash: solana.Hash{7}})} + var shreds []*Shred + for _, component := range components { + raw, err := MarshalBlockComponent(component) + require.NoError(t, err) + shreds = append(shreds, paddedComponentShreds(t, raw, uint32(len(shreds)), 0xa5)...) + } + cache := decodedPrefetchForTest(shreds) + for _, start := range []uint32{32, 96} { + // Decoding must not make a signature decision, even for retained entries. + cache[start].verification = &transactionVerification{err: errors.New("signature result belongs to completion")} + cache[start].submitErr = errors.New("submission result belongs to completion") + } + timings := entryDecodeTimings{prefetched: cache} + entries, parent, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Len(t, entries, 1) + require.Equal(t, solana.Hash{2}, entries[0].Hash) + require.Equal(t, uint32(64), parent.ReplayFECSetIndex) + require.Equal(t, solana.Hash{7}, footer.BankHash) + require.Len(t, timings.all, 5) + require.Len(t, timings.retained, 1) + require.Same(t, cache[96], timings.retained[0]) + + // Parsing the discarded prefix remains mandatory, unlike its signatures. + cache[32].err = errors.New("malformed optimistic prefix") + timings = entryDecodeTimings{prefetched: cache} + _, _, _, err = decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.ErrorContains(t, err, "malformed optimistic prefix") +} + +func TestEntryDecodePrefetchedAgavePaddedCaptureMatchesFullDecode(t *testing.T) { + packets := agavePaddedSlot1752420Packets(t) + shreds := make([]*Shred, len(packets)) + for i, packet := range packets { + var err error + shreds[i], err = ParseShred(packet) + require.NoError(t, err) + } + wantEntries, wantParent, wantFooter, err := DecodeEntriesAndAlpenglowMarkersFromDataShreds(shreds) + require.NoError(t, err) + timings := entryDecodeTimings{prefetched: decodedPrefetchForTest(shreds)} + entries, parent, footer, err := decodeEntriesAndAlpenglowMarkersFromDataShreds(shreds, &timings) + require.NoError(t, err) + require.Equal(t, wantEntries, entries) + require.Equal(t, wantParent, parent) + require.Equal(t, wantFooter, footer) + require.Len(t, timings.all, 4) + require.Len(t, timings.retained, 2) + require.Zero(t, timings.transactionParse) +} diff --git a/pkg/turbine/entry_batch_index.go b/pkg/turbine/entry_batch_index.go new file mode 100644 index 000000000..3bf3f2640 --- /dev/null +++ b/pkg/turbine/entry_batch_index.go @@ -0,0 +1,162 @@ +package turbine + +import "math/bits" + +// Three levels cover all 65,536 permitted data-shred indexes. Successor and +// predecessor queries touch a bounded number of words, even in adversarial order. +// The index is bounded (~24 KiB per retained slot) and allocated only when +// streaming preparation is enabled. It never rescans a slot's retained map. +const batchIndexWords = maxDataShredsPerSlot / 64 + +type shredIndexBits struct { + words [batchIndexWords]uint64 + groups [batchIndexWords / 64]uint64 + top uint64 +} + +func (b *shredIndexBits) set(i uint32) { + w, g := i/64, i/4096 + b.words[w] |= uint64(1) << (i % 64) + b.groups[g] |= uint64(1) << (w % 64) + b.top |= uint64(1) << g +} +func (b *shredIndexBits) clear(i uint32) { + w, g := i/64, i/4096 + b.words[w] &^= uint64(1) << (i % 64) + if b.words[w] == 0 { + b.groups[g] &^= uint64(1) << (w % 64) + if b.groups[g] == 0 { + b.top &^= uint64(1) << g + } + } +} +func (b *shredIndexBits) next(i uint32) (uint32, bool) { + if i >= maxDataShredsPerSlot { + return 0, false + } + w, g := i/64, i/4096 + if x := b.words[w] & (^uint64(0) << (i % 64)); x != 0 { + return w*64 + uint32(bits.TrailingZeros64(x)), true + } + x := b.groups[g] & (^uint64(0) << (w%64 + 1)) + if x == 0 { + top := b.top & (^uint64(0) << (g + 1)) + if top == 0 { + return 0, false + } + g = uint32(bits.TrailingZeros64(top)) + x = b.groups[g] + } + w = g*64 + uint32(bits.TrailingZeros64(x)) + return w*64 + uint32(bits.TrailingZeros64(b.words[w])), true +} +func (b *shredIndexBits) previous(i uint32) (uint32, bool) { + if i >= maxDataShredsPerSlot { + i = maxDataShredsPerSlot - 1 + } + w, g := i/64, i/4096 + if x := b.words[w] & (^uint64(0) >> (63 - i%64)); x != 0 { + return w*64 + uint32(63-bits.LeadingZeros64(x)), true + } + x := b.groups[g] & ((uint64(1) << (w % 64)) - 1) + if x == 0 { + top := b.top & ((uint64(1) << g) - 1) + if top == 0 { + return 0, false + } + g = uint32(63 - bits.LeadingZeros64(top)) + x = b.groups[g] + } + w = g*64 + uint32(63-bits.LeadingZeros64(x)) + return w*64 + uint32(63-bits.LeadingZeros64(b.words[w])), true +} + +type entryBatchIndex struct { + missing shredIndexBits + ends shredIndexBits + emitted [batchIndexWords]uint64 +} + +func newEntryBatchIndex() *entryBatchIndex { + b := new(entryBatchIndex) + for i := range b.missing.words { + b.missing.words[i] = ^uint64(0) + } + for i := range b.missing.groups { + b.missing.groups[i] = ^uint64(0) + } + b.missing.top = (uint64(1) << len(b.missing.groups)) - 1 + return b +} + +// One insertion can complete its containing batch, and, if it provides a new +// DATA_COMPLETE boundary, the immediately following batch. All other batches +// are unchanged. A range is emitted once, only when every data index is present. +// The preceding boundary is mandatory unless the range starts at index zero. +func (b *entryBatchIndex) add(i uint32, dataComplete bool) (ready [2]shredBatchRange, n int) { + if i >= maxDataShredsPerSlot { + return ready, 0 + } + b.missing.clear(i) + if dataComplete { + b.ends.set(i) + } + if end, ok := b.ends.next(i); ok { + if r, ok := b.complete(end); ok { + ready[n] = r + n++ + } + } + if dataComplete { + if end, ok := b.ends.next(i + 1); ok { + if r, ok := b.complete(end); ok { + ready[n] = r + n++ + } + } + } + return +} +func (b *entryBatchIndex) complete(end uint32) (shredBatchRange, bool) { + if b.emitted[end/64]&(uint64(1)<<(end%64)) != 0 { + return shredBatchRange{}, false + } + start := uint32(0) + if end > 0 { + if prev, ok := b.ends.previous(end - 1); ok { + start = prev + 1 + } + } + if missing, ok := b.missing.next(start); ok && missing <= end { + return shredBatchRange{}, false + } + b.emitted[end/64] |= uint64(1) << (end % 64) + return shredBatchRange{start, end}, true +} + +func (s *slotState) discoverEntryBatch(sh *Shred) { + ready, n := s.batchIndex.add(sh.Index, sh.DataComplete()) + for _, r := range ready[:n] { + s.completeBatches = append(s.completeBatches, r) + if s.pipelineTrace != nil && !s.pipelineTrace.sealed { + s.pipelineTrace.discovered[r.start] = entryTraceNow() + } + } +} + +// Seed once if preparation is installed on an assembler with existing data. +// Normal ingress creates the index with the first accepted data shred. +func (a *SlotAssembler) notePrefetchShredLocked(s *slotState, sh *Shred) { + p := a.entryPrefetch + if p == nil || p.closed || p.ctx.Err() != nil || sh.Type != ShredTypeData { + return + } + if s.batchIndex == nil { + s.batchIndex = newEntryBatchIndex() + for _, existing := range s.shreds { + s.discoverEntryBatch(existing) + } + } else { + s.discoverEntryBatch(sh) + } +} diff --git a/pkg/turbine/entry_batch_index_test.go b/pkg/turbine/entry_batch_index_test.go new file mode 100644 index 000000000..5f812be8a --- /dev/null +++ b/pkg/turbine/entry_batch_index_test.go @@ -0,0 +1,106 @@ +package turbine + +import ( + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +func TestShredIndexBitsBoundaries(t *testing.T) { + var b shredIndexBits + indexes := []uint32{0, 1, 63, 64, 65, 4095, 4096, 4097, 65534, 65535} + for _, i := range indexes { + b.set(i) + } + for i := uint32(0); i <= 65536; i++ { + var wantNext, wantPrev uint32 + var hasNext, hasPrev bool + for _, j := range indexes { + if j >= i && !hasNext { + wantNext, hasNext = j, true + } + if j <= i { + wantPrev, hasPrev = j, true + } + } + got, ok := b.next(i) + require.Equal(t, hasNext, ok) + if ok { + require.Equal(t, wantNext, got) + } + got, ok = b.previous(i) + require.Equal(t, hasPrev, ok) + if ok { + require.Equal(t, wantPrev, got) + } + } + for _, i := range indexes { + b.clear(i) + } + _, ok := b.next(0) + require.False(t, ok) + _, ok = b.previous(65535) + require.False(t, ok) + require.Zero(t, b.top) +} + +func TestEntryBatchIndexRandomArrivalMatchesOracle(t *testing.T) { + rng := rand.New(rand.NewSource(42)) + for trial := 0; trial < 100; trial++ { + const size = 257 + var ends, present [size]bool + for i := range ends { + ends[i] = rng.Intn(8) == 0 + } + ends[size-1] = true + index := newEntryBatchIndex() + emitted := map[shredBatchRange]bool{} + for _, i := range rng.Perm(size) { + present[i] = true + got, n := index.add(uint32(i), ends[i]) + want := map[shredBatchRange]bool{} + start, complete := 0, true + for j := 0; j < size; j++ { + complete = complete && present[j] + if ends[j] && present[j] { + r := shredBatchRange{uint32(start), uint32(j)} + if complete && !emitted[r] { + want[r] = true + } + start, complete = j+1, true + } + } + require.Len(t, want, n, "trial %d index %d", trial, i) + for _, r := range got[:n] { + require.True(t, want[r]) + require.False(t, emitted[r]) + emitted[r] = true + } + _, n = index.add(uint32(i), ends[i]) + require.Zero(t, n, "duplicate emitted") + } + } +} + +func BenchmarkEntryBatchIndex(b *testing.B) { + for _, reverse := range []bool{false, true} { + name := "ordered" + if reverse { + name = "reverse" + } + b.Run(name, func(b *testing.B) { + b.ReportAllocs() + for n := 0; n < b.N; n++ { + idx := newEntryBatchIndex() + for k := uint32(0); k < 65536; k++ { + i := k + if reverse { + i = 65535 - k + } + idx.add(i, i%64 == 63) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/65536, "ns/shred") + }) + } +} diff --git a/pkg/turbine/entry_batch_transactions_test.go b/pkg/turbine/entry_batch_transactions_test.go new file mode 100644 index 000000000..28329ab03 --- /dev/null +++ b/pkg/turbine/entry_batch_transactions_test.go @@ -0,0 +1,37 @@ +package turbine + +import ( + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryBatchTransactionsPreservesEntryOwnershipAndOrder(t *testing.T) { + entries := []Entry{{}, {Txns: make([]solana.Transaction, 3)}, {}, {Txns: make([]solana.Transaction, 2)}} + txs := entryBatchTransactions(entries) + require.Len(t, txs, 5) + for i := 0; i < 3; i++ { + require.Same(t, &entries[1].Txns[i], txs[i]) + } + for i := 0; i < 2; i++ { + require.Same(t, &entries[3].Txns[i], txs[3+i]) + } + require.Empty(t, entryBatchTransactions(nil)) + require.Empty(t, entryBatchTransactions([]Entry{{}, {}})) +} + +var entryBatchBenchmarkSink []*solana.Transaction + +func BenchmarkEntryBatchTransactions(b *testing.B) { + for _, count := range []int{4, 49, 269, 33760} { + b.Run(fmt.Sprintf("tx_%d", count), func(b *testing.B) { + entries := []Entry{{Txns: make([]solana.Transaction, count/2)}, {}, {Txns: make([]solana.Transaction, count-count/2)}} + b.ReportAllocs() + for b.Loop() { + entryBatchBenchmarkSink = entryBatchTransactions(entries) + } + }) + } +} diff --git a/pkg/turbine/entry_hash.go b/pkg/turbine/entry_hash.go index 72af6bdb8..35ee7561c 100644 --- a/pkg/turbine/entry_hash.go +++ b/pkg/turbine/entry_hash.go @@ -90,14 +90,7 @@ func hashTransactions(txns []solana.Transaction) solana.Hash { } func hashSignatures(signatures [][]byte) solana.Hash { - if len(signatures) == 0 { - return solana.Hash{} - } - nodes := merkletree.HashNodes(signatures) - if root := nodes.GetRoot(); root != nil { - return solana.Hash(*root) - } - return solana.Hash{} + return solana.Hash(merkletree.HashRoot(signatures)) } func sha256Hash(data []byte) solana.Hash { diff --git a/pkg/turbine/entry_hash_bench_test.go b/pkg/turbine/entry_hash_bench_test.go new file mode 100644 index 000000000..e8d7d1038 --- /dev/null +++ b/pkg/turbine/entry_hash_bench_test.go @@ -0,0 +1,24 @@ +package turbine + +import ( + "encoding/binary" + "fmt" + "testing" +) + +func BenchmarkEntrySignatureRoot(b *testing.B) { + for _, count := range []int{1, 64, 311, 512, 1024} { + b.Run(fmt.Sprint(count), func(b *testing.B) { + sigs := make([][]byte, count) + for i := range sigs { + sigs[i] = make([]byte, 64) + binary.LittleEndian.PutUint64(sigs[i], uint64(i)) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + _ = hashSignatures(sigs) + } + }) + } +} diff --git a/pkg/turbine/entry_identity_recovery_test.go b/pkg/turbine/entry_identity_recovery_test.go new file mode 100644 index 000000000..aab96f794 --- /dev/null +++ b/pkg/turbine/entry_identity_recovery_test.go @@ -0,0 +1,131 @@ +package turbine + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryIdentityRecoveryVerifiesFinalTransactions(t *testing.T) { + for _, fault := range []string{"partial_identities", "wrong_pointer", "missing_range", "oversized_range", "nil_batch"} { + for _, invalidFinal := range []bool{false, true} { + name := fault + "/valid" + if invalidFinal { + name = fault + "/invalid" + } + t.Run(name, func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 3) + entries := []Entry{{Txns: []solana.Transaction{*txs[0], *txs[1], *txs[2]}}} + blk := &block.Block{Slot: 812, Transactions: entryBatchTransactions(entries)} + request, err := v.submitTransactions(context.Background(), blk.Transactions) + require.NoError(t, err) + _, err = request.wait() + require.NoError(t, err) + batches := []*prefetchedShredBatch{{entries: entries, verification: request}} + switch fault { + case "partial_identities": + request.identities = request.identities[:2] + case "wrong_pointer": + copyTx := *blk.Transactions[0] + blk.Transactions[0] = ©Tx + case "missing_range": + batches = nil + case "oversized_range": + batches[0].entries = append(batches[0].entries, Entry{Txns: []solana.Transaction{*txs[0]}}) + case "nil_batch": + batches = []*prefetchedShredBatch{nil} + } + if invalidFinal { + // The old request verified a different, valid transaction. + // A successful old verdict must not bless these final bytes. + copyTx := *blk.Transactions[0] + copyTx.Signatures = append([]solana.Signature(nil), copyTx.Signatures...) + copyTx.Signatures[0][0] ^= 1 + blk.Transactions[0] = ©Tx + } + err = verifyDecodedEntryBatches(context.Background(), blk, batches, v) + if invalidFinal { + require.ErrorContains(t, err, "failed signature verification") + return + } + require.NoError(t, err) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } + }) + } + } +} + +func TestEntryIdentityRecoveryPreservesCancellationAndVerifierFailure(t *testing.T) { + blk := &block.Block{Slot: 813, Transactions: verifierSignedTransactions(t, 2)} + v := newTransactionVerifier(2, 16, nil) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + require.ErrorIs(t, verifyDecodedEntryBatches(ctx, blk, nil, v), context.Canceled) + v.closeAndWait() + require.ErrorIs(t, verifyDecodedEntryBatches(context.Background(), blk, nil, v), errTransactionVerifierClosed) +} + +func TestEntryIdentityRecoveryJoinsCanceledReaders(t *testing.T) { + started := make(chan struct{}, 1) + release := make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + select { + case started <- struct{}{}: + default: + } + <-release + return nil + }) + defer v.closeAndWait() + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + defer unblock() + txs := verifierSignedTransactions(t, 2) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + request, err := v.submitTransactions(ctx, txs) + require.NoError(t, err) + select { + case <-started: + case <-time.After(time.Second): + t.Fatal("reader never started") + } + result := make(chan error, 1) + // This deliberately inconsistent range sends us through recovery while + // the old verifier still owns the block's transaction buffers. + go func() { + result <- verifyDecodedEntryBatches(ctx, &block.Block{Transactions: txs}, []*prefetchedShredBatch{{verification: request}}, v) + }() + cancel() + select { + case err := <-result: + t.Fatalf("returned before old reader released its buffers: %v", err) + case <-time.After(20 * time.Millisecond): + } + unblock() + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(time.Second): + t.Fatal("recovery did not join canceled readers") + } + select { + case <-request.done: + default: + t.Fatal("request ownership was not released") + } +} diff --git a/pkg/turbine/entry_pipeline_trace.go b/pkg/turbine/entry_pipeline_trace.go new file mode 100644 index 000000000..c2c83b454 --- /dev/null +++ b/pkg/turbine/entry_pipeline_trace.go @@ -0,0 +1,212 @@ +package turbine + +// Temporary, opt-in pipeline diagnostics. No transaction bytes or keys are logged. +// Timestamps are monotonic nanoseconds relative to origin_unix_ns. Worker elapsed +// time includes descheduling; summed job durations are NOT wall-clock critical paths. +import ( + "context" + "encoding/json" + "fmt" + "os" + "strconv" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" +) + +var entryTraceOrigin = time.Now() +var entryTraceDropped atomic.Uint64 +var entryTraceConfig = configureEntryTrace() + +type entryTraceSettings struct { + modulo uint64 + until time.Time + reports chan entryPipelineReport +} + +func configureEntryTrace() entryTraceSettings { + n, err := strconv.ParseUint(os.Getenv("MITHRIL_ENTRY_TRACE_MOD"), 10, 32) + if err != nil || n == 0 { + return entryTraceSettings{} + } + seconds, err := strconv.Atoi(os.Getenv("MITHRIL_ENTRY_TRACE_SECONDS")) + if err != nil || seconds < 1 || seconds > 1800 { + seconds = 600 + } + path := os.Getenv("MITHRIL_ENTRY_TRACE_FILE") + if path == "" { + return entryTraceSettings{} + } + f, err := os.OpenFile(path, os.O_CREATE|os.O_APPEND|os.O_WRONLY, 0600) + if err != nil { + fmt.Fprintf(os.Stderr, "entry pipeline trace disabled: %v\n", err) + return entryTraceSettings{} + } + ch := make(chan entryPipelineReport, 8) + go func() { + defer f.Close() + enc := json.NewEncoder(f) + for r := range ch { + r.Dropped = entryTraceDropped.Load() + r.finish() + if err := enc.Encode(r); err != nil { + entryTraceDropped.Add(1) + } + } + }() + return entryTraceSettings{n, time.Now().Add(time.Duration(seconds) * time.Second), ch} +} + +func entryTraceNow() int64 { return time.Since(entryTraceOrigin).Nanoseconds() } +func entryTraceTime(t time.Time) int64 { return t.Sub(entryTraceOrigin).Nanoseconds() } + +type entryTraceContextKey struct{} + +func withEntryPipelineTrace(ctx context.Context, t *entryPipelineTrace) context.Context { + if t == nil { + return ctx + } + return context.WithValue(ctx, entryTraceContextKey{}, true) +} +func entryTraceContext(ctx context.Context) bool { + return ctx != nil && ctx.Value(entryTraceContextKey{}) == true +} + +type entryPipelineTrace struct { + arrivals map[uint32]int64 + discovered map[uint32]int64 + sealed bool // frozen at first completion claim, including retry/error paths +} + +// Called only after successful admission, under the assembler lock. Includes +// FEC-reconstructed data. This is local availability, not a NIC timestamp. +func (s *slotState) traceAcceptedShred(sh *Shred) { + if sh.Type != ShredTypeData { + return + } + if s.pipelineTrace == nil { + c := entryTraceConfig + if c.modulo == 0 || s.slot%c.modulo != 0 || time.Now().After(c.until) { + return + } + s.pipelineTrace = &entryPipelineTrace{arrivals: make(map[uint32]int64), discovered: make(map[uint32]int64)} + } + if !s.pipelineTrace.sealed { + s.pipelineTrace.arrivals[sh.Index] = entryTraceNow() + } +} + +type entryVerificationTrace struct { + Transactions int `json:"transactions"` + Submit int64 `json:"submit_ns"` + Admitted int64 `json:"admitted_ns"` + FirstWorker int64 `json:"first_worker_ns"` + LastWorker int64 `json:"last_worker_ns"` + Finished int64 `json:"finished_ns"` + Jobs int `json:"jobs"` + JobWaitSum int64 `json:"job_offer_to_start_sum_ns"` + JobWaitMax int64 `json:"job_offer_to_start_max_ns"` + WorkerSum int64 `json:"worker_elapsed_sum_ns"` + WorkerMax int64 `json:"worker_elapsed_max_ns"` +} + +func (t *entryVerificationTrace) observe(j *transactionVerifyJob) { + if t.FirstWorker == 0 || j.workerStart < t.FirstWorker { + t.FirstWorker = j.workerStart + } + t.LastWorker = max(t.LastWorker, j.workerEnd) + t.Jobs++ + wait, work := j.workerStart-j.offeredAt, j.workerEnd-j.workerStart + t.JobWaitSum += wait + t.JobWaitMax = max(t.JobWaitMax, wait) + t.WorkerSum += work + t.WorkerMax = max(t.WorkerMax, work) +} + +type entryBatchTraceReport struct { + Start uint32 `json:"start"` + End uint32 `json:"end"` + Transactions int `json:"transactions"` + Retained bool `json:"retained"` + Prefetched bool `json:"prefetched"` + AvailabilityKnown bool `json:"availability_known"` + Available int64 `json:"available_ns"` + Discovered int64 `json:"discovered_ns"` + DecodeStart int64 `json:"decode_start_ns"` + DecodeEnd int64 `json:"decode_end_ns"` + Verification *entryVerificationTrace `json:"verification,omitempty"` +} + +type entryPipelineReport struct { + Origin int64 `json:"origin_unix_ns"` + Slot uint64 `json:"slot"` + Transactions int `json:"transactions"` + Full int64 `json:"full_ns"` + CompletionStart int64 `json:"completion_start_ns"` + Ready int64 `json:"ready_ns"` + Dropped uint64 `json:"dropped_reports"` + Batches []entryBatchTraceReport `json:"batches"` + Fallback *entryVerificationTrace `json:"fallback,omitempty"` + source *entryPipelineTrace + all, retained []*prefetchedShredBatch + fallback *transactionVerification +} + +func completedEntryVerification(r *transactionVerification) *entryVerificationTrace { + if r == nil { + return nil + } + select { + case <-r.done: + return r.trace + default: + return nil + } +} + +// All source maps are sealed, and decode has joined all preparation readers. +// Reading request metrics additionally requires the verification done barrier. +func (r *entryPipelineReport) finish() { + retained := make(map[*prefetchedShredBatch]bool, len(r.retained)) + for _, b := range r.retained { + retained[b] = true + } + for _, b := range r.all { + row := entryBatchTraceReport{Start: b.start, End: b.end, Retained: retained[b], Prefetched: b.ready != nil, + DecodeStart: b.traceDecodeStart, DecodeEnd: b.traceDecodeEnd, Discovered: r.source.discovered[b.start], + AvailabilityKnown: true, Verification: completedEntryVerification(b.verification)} + for _, e := range b.entries { + row.Transactions += len(e.Txns) + } + // Require the preceding DATA_COMPLETE boundary as well as every shred in + // this batch. An end marker alone cannot establish an independent start. + start := b.start + if start > 0 { + start-- + } + for i := start; i <= b.end; i++ { + at, ok := r.source.arrivals[i] + if !ok { + row.AvailabilityKnown = false + } + row.Available = max(row.Available, at) + } + r.Batches = append(r.Batches, row) + } + r.Fallback = completedEntryVerification(r.fallback) +} + +func queueEntryPipelineReport(s *slotState, b *block.Block, d *entryDecodeTimings, start, ready time.Time) { + if s.pipelineTrace == nil || len(b.Transactions) < 10000 || entryTraceConfig.reports == nil { + return + } + r := entryPipelineReport{Origin: entryTraceOrigin.UnixNano(), Slot: s.slot, Transactions: len(b.Transactions), + Full: entryTraceTime(s.fullAt), CompletionStart: entryTraceTime(start), Ready: entryTraceTime(ready), + source: s.pipelineTrace, all: d.all, retained: d.retained, fallback: d.traceFallback} + select { + case entryTraceConfig.reports <- r: + default: + entryTraceDropped.Add(1) + } +} diff --git a/pkg/turbine/entry_pipeline_trace_test.go b/pkg/turbine/entry_pipeline_trace_test.go new file mode 100644 index 000000000..6aa493601 --- /dev/null +++ b/pkg/turbine/entry_pipeline_trace_test.go @@ -0,0 +1,68 @@ +package turbine + +import ( + "context" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestEntryPipelineTraceRequiresCompleteBatchAndBoundary(t *testing.T) { + source := &entryPipelineTrace{arrivals: map[uint32]int64{0: 10, 1: 100, 2: 20, 3: 30, 4: 40}, discovered: map[uint32]int64{0: 110, 3: 120}, sealed: true} + first := &prefetchedShredBatch{start: 0, end: 2} + later := &prefetchedShredBatch{start: 3, end: 4} + r := entryPipelineReport{source: source, all: []*prefetchedShredBatch{first, later}, retained: []*prefetchedShredBatch{later}} + r.finish() + require.Equal(t, int64(100), r.Batches[0].Available) + require.Equal(t, int64(40), r.Batches[1].Available) + require.Equal(t, int64(80), r.Batches[1].Discovered-r.Batches[1].Available, "a gap in the earlier batch delays discovery, not availability of the later one") + require.False(t, r.Batches[0].Retained) + require.True(t, r.Batches[1].Retained) + delete(source.arrivals, 2) + r.Batches = nil + r.finish() + require.False(t, r.Batches[1].AvailabilityKnown, "the preceding boundary must also have been observed") +} + +func TestEntryVerificationTraceJoinsWorkerTimings(t *testing.T) { + entered, release := make(chan struct{}), make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + select { + case <-entered: + default: + close(entered) + } + <-release + return nil + }) + defer v.closeAndWait() + ctx := withEntryPipelineTrace(context.Background(), &entryPipelineTrace{}) + r, err := v.submitTransactions(ctx, []*solana.Transaction{{}}) + require.NoError(t, err) + <-entered + require.Nil(t, completedEntryVerification(r), "unfinished mutable metrics must not be read") + close(release) + _, err = r.wait() + require.NoError(t, err) + m := completedEntryVerification(r) + require.NotNil(t, m) + require.Equal(t, 1, m.Jobs) + require.LessOrEqual(t, m.Submit, m.Admitted) + require.LessOrEqual(t, m.Admitted, m.FirstWorker) + require.Less(t, m.FirstWorker, m.LastWorker) + require.LessOrEqual(t, m.LastWorker, m.Finished) + require.Positive(t, m.WorkerSum) + require.GreaterOrEqual(t, m.JobWaitSum, int64(0)) + plain, err := v.submitTransactions(context.Background(), []*solana.Transaction{{}}) + require.NoError(t, err) + _, err = plain.wait() + require.NoError(t, err) + require.Nil(t, plain.trace, "ordinary verification does not collect job timestamps") +} + +func TestEntryPipelineTraceSealedGeneration(t *testing.T) { + s := &slotState{pipelineTrace: &entryPipelineTrace{arrivals: map[uint32]int64{1: 42}, discovered: map[uint32]int64{}, sealed: true}} + s.traceAcceptedShred(&Shred{Type: ShredTypeData, Index: 2}) + require.Len(t, s.pipelineTrace.arrivals, 1, "completion/retry cannot mutate a report's frozen arrival map") +} diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go new file mode 100644 index 000000000..2b082e52b --- /dev/null +++ b/pkg/turbine/entry_prefetch.go @@ -0,0 +1,428 @@ +package turbine + +import ( + "context" + "errors" + "sync" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +const ( + entryPrefetchSlots = 8 + // Covers a large block of maximum-size transactions while bounding the + // extra retained encoded bytes across all in-flight generations. + entryPrefetchBytes = 64 << 20 + entryPrefetchBatchBytes = 1 << 20 +) + +type shredBatchRange struct{ start, end uint32 } + +// A result belongs to one exact DATA_COMPLETE range in one slot generation. +// Fields are immutable after ready closes; signature readers own its decoded +// transactions until verification.done closes. +type prefetchedShredBatch struct { + start, end uint32 + raw []byte + entries []Entry + parent *AlpenglowParentInfo + footer *BlockFooter + marker bool + traceDecodeStart int64 + traceDecodeEnd int64 + parseDuration time.Duration + err error + ready chan struct{} + verification *transactionVerification + submittedAt time.Time + submitErr error +} + +type slotEntryPrefetch struct { + pool *entryPrefetchPool + ctx context.Context + cancel context.CancelFunc + batches map[uint32]*prefetchedShredBatch + next int + queued, released bool + queueDone chan struct{} // closed after the queued/running token retires + bytes int +} + +// All scheduling and accounting use assembler.mu. The packet reader only +// indexes complete data ranges and attempts a nonblocking, coalesced enqueue. +// Decoding and bounded verifier admission run on separate background workers. +type entryPrefetchPool struct { + a *SlotAssembler + ctx context.Context + cancel context.CancelFunc + verifier *transactionVerifier + jobs chan *slotState + workers, cleanup sync.WaitGroup + slots, bytes int + closed bool + close sync.Once +} + +func newEntryPrefetchPool(ctx context.Context, a *SlotAssembler, verifier *transactionVerifier) *entryPrefetchPool { + ctx, cancel := context.WithCancel(ctx) + p := &entryPrefetchPool{a: a, ctx: ctx, cancel: cancel, verifier: verifier, jobs: make(chan *slotState, entryPrefetchSlots)} + a.mu.Lock() + a.entryPrefetch = p + a.mu.Unlock() + p.workers.Add(2) + for i := 0; i < 2; i++ { + go p.run() + } + return p +} + +func (a *SlotAssembler) prefetchEntriesLocked(s *slotState) { + p := a.entryPrefetch + if p == nil || p.closed || p.ctx.Err() != nil { + return + } + if s.batchIndex == nil && len(s.shreds) != 0 { + s.batchIndex = newEntryBatchIndex() + for _, sh := range s.shreds { + s.discoverEntryBatch(sh) + } + } + if s.prefetch == nil && len(s.completeBatches) > 0 && p.slots < entryPrefetchSlots { + ctx, cancel := context.WithCancel(withEntryPipelineTrace(p.ctx, s.pipelineTrace)) + s.prefetch = &slotEntryPrefetch{pool: p, ctx: ctx, cancel: cancel, batches: make(map[uint32]*prefetchedShredBatch)} + p.slots++ + } + p.enqueueLocked(s) +} + +func (p *entryPrefetchPool) enqueueLocked(s *slotState) { + f := s.prefetch + if f == nil || f.pool != p || f.queued || f.released || s.completing || p.closed || f.next >= len(s.completeBatches) { + return + } + select { + case p.jobs <- s: + f.queued = true + f.queueDone = make(chan struct{}) + default: + } +} + +func (p *entryPrefetchPool) run() { + defer p.workers.Done() + for s := range p.jobs { + p.a.mu.Lock() + f := s.prefetch + if p.closed || f.released || f.ctx.Err() != nil || p.a.slots[s.slot] != s || s.completing { + f.queued = false + close(f.queueDone) + p.a.mu.Unlock() + continue + } + var batch *prefetchedShredBatch + var shreds []*Shred + var rawSize int + for f.next < len(s.completeBatches) { + r := s.completeBatches[f.next] + size := 0 + for i := r.start; i <= r.end; i++ { + size += len(s.shreds[i].Data) + if size > entryPrefetchBatchBytes { + break + } + } + if size > entryPrefetchBatchBytes { + f.next++ + continue + } + if p.bytes+size > entryPrefetchBytes { + break + } + f.next++ + p.bytes += size + f.bytes += size + rawSize = size + batch = &prefetchedShredBatch{start: r.start, end: r.end, ready: make(chan struct{})} + f.batches[r.start] = batch + shreds = make([]*Shred, 0, r.end-r.start+1) + for i := r.start; i <= r.end; i++ { + shreds = append(shreds, s.shreds[i]) + } + break + } + if batch == nil { + f.queued = false + close(f.queueDone) + p.a.mu.Unlock() + continue + } + p.a.mu.Unlock() + + if s.pipelineTrace != nil { + batch.traceDecodeStart = entryTraceNow() + } + raw := make([]byte, 0, rawSize) + for _, sh := range shreds { + raw = append(raw, sh.Data...) + } + decoded := decodeClosedShredBatch(raw, batch.start, batch.end) + ready := batch.ready + batch.raw, batch.entries = decoded.raw, decoded.entries + batch.parent, batch.footer, batch.marker = decoded.parent, decoded.footer, decoded.marker + batch.parseDuration, batch.err = decoded.parseDuration, decoded.err + if s.pipelineTrace != nil { + batch.traceDecodeEnd = entryTraceNow() + } + if batch.err == nil && !batch.marker && f.ctx.Err() == nil { + txs := entryBatchTransactions(batch.entries) + if len(txs) > 0 { + batch.submittedAt = time.Now() + batch.verification, batch.submitErr = p.verifier.submitPrefetchTransactions(f.ctx, txs) + } + } + close(ready) + p.a.mu.Lock() + f.queued = false + close(f.queueDone) + p.enqueueLocked(s) + p.a.mu.Unlock() + } +} + +func entryBatchTransactions(entries []Entry) []*solana.Transaction { + count := 0 + for i := range entries { + count += len(entries[i].Txns) + } + // Entries are already decoded: size this pointer view once instead of + // repeatedly reallocating and copying it while preparing each component. + txs := make([]*solana.Transaction, 0, count) + for i := range entries { + for j := range entries[i].Txns { + txs = append(txs, &entries[i].Txns[j]) + } + } + return txs +} + +// Keep reservations until canceled readers have actually relinquished their +// buffers. Repeated resets cannot evade the memory or slot bounds. +func (a *SlotAssembler) releasePrefetchLocked(s *slotState) { + if s == nil || s.prefetch == nil || s.prefetch.released { + return + } + f := s.prefetch + f.released = true + f.cancel() + p := f.pool + queueDone := f.queueDone + p.cleanup.Add(1) + go func() { + defer p.cleanup.Done() + for _, b := range f.batches { + <-b.ready + if b.verification != nil { + b.verification.wait() + } + } + if queueDone != nil { + <-queueDone + } + p.a.mu.Lock() + p.slots-- + p.bytes -= f.bytes + p.a.mu.Unlock() + }() +} + +func (p *entryPrefetchPool) closeAndWait() { + p.close.Do(func() { + p.cancel() + p.a.mu.Lock() + p.closed = true + if p.a.entryPrefetch == p { + p.a.entryPrefetch = nil + } + for _, s := range p.a.slots { + if s.prefetch != nil && s.prefetch.pool == p { + p.a.releasePrefetchLocked(s) + } + } + close(p.jobs) + p.a.mu.Unlock() + p.workers.Wait() + p.cleanup.Wait() + }) +} + +// Reuse only retained entry results. UpdateParent may intentionally discard an +// invalid optimistic prefix. Unprefetched transactions form one immediately +// available request, overlapping any early requests still running. +func verifyDecodedEntryBatches(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier) error { + return verifyDecodedEntryBatchesWithTimings(ctx, blk, batches, verifier, nil) +} + +func verifyDecodedEntryBatchesWithTimings(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier, timings *entryDecodeTimings) error { + if ctx == nil { + ctx = context.Background() + } + if blk == nil { + return errors.New("verify decoded entries: nil block") + } + type pending struct { + future *transactionVerification + offset int + count int + } + var early []pending + var missing []*solana.Transaction + var indices []int + offset := 0 + for _, b := range batches { + if b == nil { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("nil retained entry batch")) + } + count := 0 + for _, e := range b.entries { + count += len(e.Txns) + } + if count > len(blk.Transactions)-offset { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry transaction range exceeds final block")) + } + reusable := b.verification != nil + if reusable { + select { + case <-b.verification.done: + // Cancellation is not a signature verdict. A completion canceled + // after preparation may be retried on this same slot generation. + reusable = !errors.Is(b.verification.err, context.Canceled) && !errors.Is(b.verification.err, context.DeadlineExceeded) + default: + } + } + if reusable { + early = append(early, pending{b.verification, offset, count}) + } else { + missing = append(missing, blk.Transactions[offset:offset+count]...) + for i := 0; i < count; i++ { + indices = append(indices, offset+i) + } + } + offset += count + } + if offset != len(blk.Transactions) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry identity coverage mismatch")) + } + var fallback *transactionVerification + var err error + if len(missing) > 0 { + fallback, err = verifier.submitTransactions(ctx, missing) + if timings != nil { + timings.traceFallback = fallback + } + } + firstIndex := len(blk.Transactions) + firstErr := err + for _, p := range early { + i, e := p.future.waitContext(ctx) + if e != nil && (firstErr == nil || (i >= 0 && p.offset+i < firstIndex)) { + firstErr = e + if i >= 0 { + firstIndex = p.offset + i + } + } + } + if fallback != nil { + i, e := fallback.waitContext(ctx) + if e != nil && (firstErr == nil || (i >= 0 && indices[i] < firstIndex)) { + firstErr = e + if i >= 0 { + firstIndex = indices[i] + } + } + } + if ctx.Err() != nil { + return ctx.Err() + } + if firstErr != nil && firstIndex < len(blk.Transactions) { + return formatTransactionVerificationError(blk, firstIndex, firstErr) + } + if firstErr != nil { + return firstErr + } + // Custom verification hooks do not produce trusted message identities. + // Preserve their existing lazy preparation path (primarily test fixtures). + if verifier.verify != nil { + return nil + } + identities := make([]txverify.VerifiedMessageIdentity, len(blk.Transactions)) + for _, p := range early { + if len(p.future.identities) != p.count || p.offset+len(p.future.identities) > len(identities) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("entry identity range mismatch")) + } + copy(identities[p.offset:], p.future.identities) + } + if fallback != nil { + if len(fallback.identities) != len(indices) { + return recoverEntryVerification(ctx, blk, batches, verifier, errors.New("fallback identity coverage mismatch")) + } + for i, index := range indices { + identities[index] = fallback.identities[i] + } + } + if err := blk.CacheVerifiedTransactionMessageIdentities(identities); err != nil { + return recoverEntryVerification(ctx, blk, batches, verifier, err) + } + return nil +} + +// Prefetch metadata is an optimization, never a substitute for verifying the +// final block. Join old readers, then verify every final transaction afresh. +// This also repairs pointer/coverage mismatches without treating a successful +// verdict for another transaction as proof for this one. Normal signature +// failures above are still rejected directly. Failed recovery stays an error. +func recoverEntryVerification(ctx context.Context, blk *block.Block, batches []*prefetchedShredBatch, verifier *transactionVerifier, reason error) error { + for _, batch := range batches { + if batch != nil && batch.verification != nil { + _, _ = batch.verification.waitContext(ctx) + } + } + if err := ctx.Err(); err != nil { + return err + } + mlog.Log.Warnf("slot %d: discarded inconsistent entry verification metadata; re-verifying final transactions: %v", blk.Slot, reason) + return verifier.verifyBlockContext(ctx, blk) +} + +func earlyEntryTimings(t *entryDecodeTimings, fullAt time.Time, timings *block.TurbineIngressTimings) { + for _, b := range t.all { + if b.ready == nil { + continue + } + timings.EarlyTransactionParse += b.parseDuration + if b.verification != nil { + select { + case <-b.verification.done: + timings.EarlyTransactionSigverify += b.verification.finishedAt.Sub(b.submittedAt) + default: + } + } + } + for _, b := range t.retained { + if b.verification != nil { + select { + case <-b.verification.done: + if b.verification.err == nil && !b.verification.finishedAt.After(fullAt) { + for _, e := range b.entries { + timings.EarlyVerifiedTransactions += uint64(len(e.Txns)) + } + } + default: + } + } + } +} diff --git a/pkg/turbine/entry_prefetch_benchmark_test.go b/pkg/turbine/entry_prefetch_benchmark_test.go new file mode 100644 index 000000000..be1ef6dcd --- /dev/null +++ b/pkg/turbine/entry_prefetch_benchmark_test.go @@ -0,0 +1,288 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/binary" + "fmt" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/gagliardetto/solana-go" +) + +// BenchmarkEntryPrefetchAssembly runs actual assembler ingestion, component +// decoding, final Merkle identity checks, transaction verification, and final +// publication. Source transactions use the same generated 228/1232-byte or +// captured fixtures as BenchmarkTransactionVerificationFlow. Generation, +// packet parsing and shred-signature authentication happen outside the timer. +// There is no network I/O, loss/recovery, replay, PoH or footer-bankhash execution. +// +// Complete component bursts are scheduled across 200 ms for tip scenarios; +// catchup offers all shreds immediately. This is a workload model, not a +// measured cluster arrival distribution. A 60 KiB signed-transaction budget +// defines components; actual entry overhead and FEC padding are generated. +// Pools persist across iterations; the identical fixture slot is reset between +// samples outside the timer. Use fixed counts (e.g. -benchtime=3x). +func BenchmarkEntryPrefetchAssembly(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, workers := range []int{2, 4} { + b.Run(fmt.Sprintf("workers_%d", workers), func(b *testing.B) { + for _, target := range []int{4, 8} { + b.Run(fmt.Sprintf("target_%d", target), func(b *testing.B) { + for _, arrival := range []struct { + name string + span time.Duration + }{{"catchup", 0}, {"tip_200ms", 200 * time.Millisecond}} { + b.Run(arrival.name, func(b *testing.B) { + for _, overlap := range []bool{false, true} { + name := "overlap_off" + if overlap { + name = "overlap_on" + } + b.Run(name, func(b *testing.B) { + runAssemblyFlowBenchmark(b, source.blk, fixture, workers, target, arrival.span, overlap, false) + }) + } + }) + } + }) + } + }) + } + }) + } +} + +type assemblyFlowFixture struct { + components [][]*Shred + authenticatedRoots [][]solana.Hash + blockID solana.Hash + parentID solana.Hash + bankhash solana.Hash + shreds int + holdGap bool +} + +func makeAssemblyFlowFixture(tb testing.TB, source *block.Block) assemblyFlowFixture { + tb.Helper() + fixture := assemblyFlowFixture{parentID: solana.Hash{31}, bankhash: solana.Hash{47}} + var seed [ed25519.SeedSize]byte + seed[0] = 193 // Public deterministic benchmark leader, not validator material. + leader := solana.PrivateKey(ed25519.NewKeyFromSeed(seed[:])) + public := solana.PublicKeyFromBytes(ed25519.PrivateKey(leader).Public().(ed25519.PublicKey)) + generator := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7, ReferenceTick: 63} + var nextData, nextCode uint32 + var chained solana.Hash + var roots []solana.Hash + appendComponent := func(component BlockComponent, last bool) { + raw, err := MarshalBlockComponent(component) + if err != nil { + tb.Fatal(err) + } + packets, finalRoot, dataEnd, codeEnd, err := generator.MakeShredsFromData(leader, raw, last, chained, nextData, nextCode) + if err != nil { + tb.Fatal(err) + } + chained, nextData, nextCode = finalRoot, dataEnd, codeEnd + var data []*Shred + var authenticatedRoots []solana.Hash + for _, packet := range packets { + shred, err := ParseShred(packet) + if err != nil { + tb.Fatal(err) + } + if shred.Type != ShredTypeData { + continue + } + if err := shred.VerifySignature(public); err != nil { + tb.Fatal(err) + } + root, err := shred.MerkleRoot() + if err != nil { + tb.Fatal(err) + } + if shred.Index == shred.FECSetIndex { + roots = append(roots, root) + } + authenticatedRoots = append(authenticatedRoots, root) + data = append(data, shred) + } + fixture.components = append(fixture.components, data) + fixture.authenticatedRoots = append(fixture.authenticatedRoots, authenticatedRoots) + fixture.shreds += len(data) + } + appendComponent(NewBlockHeader(99, fixture.parentID), false) + for i, txs := range flowComponents(source.Transactions, 60*1024) { + entry := Entry{NumHashes: 1, Txns: make([]solana.Transaction, len(txs))} + binary.LittleEndian.PutUint64(entry.Hash[:], uint64(i+1)) + for j, tx := range txs { + entry.Txns[j] = *tx + } + appendComponent(BlockComponent{EntryBatch: []Entry{entry}}, false) + } + appendComponent(NewBlockFooter(BlockFooter{BankHash: fixture.bankhash}), true) + if nextData > maxDataShredsPerSlot { + tb.Fatalf("fixture requires %d data shreds; assembler limit is %d", nextData, maxDataShredsPerSlot) + } + fixture.blockID = DoubleMerkleBlockID(99, fixture.parentID, roots) + return fixture +} + +func runAssemblyFlowBenchmark(b *testing.B, source *block.Block, fixture assemblyFlowFixture, workers, target int, span time.Duration, overlap, prepareIdentities bool) { + v := newTransactionVerifierWithBatchTarget(workers, 2*workers*target, target, nil) + defer v.closeAndWait() + if err := v.verifyBlock(source); err != nil { + b.Fatal(err) + } + a := NewSlotAssembler() + a.verifyTransactions = v.verifyBlockContext + a.SetKnownAlpenglowBlockID(99, fixture.parentID) + a.SetKnownAlpenglowBlockID(100, fixture.blockID) + if overlap { + prefetch := newEntryPrefetchPool(context.Background(), a, v) + defer prefetch.closeAndWait() + } + var ready, parse, preparation, joins, arrivals, identityPreparation, fullToIdentities []time.Duration + var early uint64 + var cpu float64 + before := sigverify.Stats() + b.ReportAllocs() + b.ResetTimer() + for range b.N { + b.StopTimer() + a.ResetSlot(100) + cpuStarted := flowCPUSeconds(b) + b.StartTimer() + started := time.Now() + var work *slotCompletionWork + for componentIndex, component := range fixture.components { + if span > 0 { + time.Sleep(time.Until(started.Add(flowArrivalOffset(componentIndex, len(fixture.components), span)))) + } + for shredIndex, shred := range component { + if fixture.holdGap && componentIndex == len(fixture.components)*3/4 && shredIndex == 1 { + continue + } + candidate, err := a.addShredFromWithRoot(shred, false, &fixture.authenticatedRoots[componentIndex][shredIndex]) + if err != nil { + b.Fatal(err) + } + if candidate != nil { + if work != nil { + b.Fatal("slot claimed completion more than once") + } + work = candidate + } + } + } + if fixture.holdGap { + ci := len(fixture.components) * 3 / 4 + candidate, err := a.addShredFromWithRoot(fixture.components[ci][1], false, &fixture.authenticatedRoots[ci][1]) + if err != nil || candidate == nil || work != nil { + b.Fatalf("held gap completion failed: %v", err) + } + work = candidate + } + if work == nil { + b.Fatal("all generated data shreds did not complete the slot") + } + processed := a.processCompletion(context.Background(), work) + completed, err := a.finalizeCompletion(work, processed) + if prepareIdentities && err == nil && completed != nil { + start := time.Now() + _, err = completed.PrepareTransactionMessageIdentities() + identityPreparation = append(identityPreparation, time.Since(start)) + fullToIdentities = append(fullToIdentities, time.Since(work.state.fullAt)) + } + b.StopTimer() + cpu += flowCPUSeconds(b) - cpuStarted + if err != nil || completed == nil { + b.Fatalf("completion failed: block=%v error=%v", completed != nil, err) + } + if !completed.TransactionSignaturesVerified() || len(completed.Transactions) != len(source.Transactions) || + !completed.HasAlpenglowBlockID || solana.Hash(completed.AlpenglowBlockID) != fixture.blockID || + !completed.HasExpectedBankhash || completed.ExpectedBankhash != fixture.bankhash { + b.Fatal("completion changed transaction coverage or authenticated block metadata") + } + ready = append(ready, processed.timings.FullToReady) + parse = append(parse, processed.timings.TransactionParse) + preparation = append(preparation, processed.timings.EarlyPreparationWait) + joins = append(joins, processed.timings.TransactionSigverify) + arrivals = append(arrivals, processed.timings.ShredCollection) + early += processed.timings.EarlyVerifiedTransactions + } + after := sigverify.Stats() + b.ReportMetric(cpu*1000/float64(b.N), "cpu-ms/block") + b.ReportMetric(cpu/b.Elapsed().Seconds(), "avg_cpu_cores") + b.ReportMetric(float64(early)/float64(b.N), "early_verified_tx/block") + b.ReportMetric(float64(len(source.Transactions)), "tx/block") + b.ReportMetric(float64(fixture.shreds), "data_shreds/block") + b.ReportMetric(float64(len(fixture.components)), "components/block") + if batches := after.Batches - before.Batches; batches > 0 { + b.ReportMetric(float64(after.Signatures-before.Signatures)/float64(batches), "mean_width") + } + flowReportPercentiles(b, ready, "full_to_ready") + flowReportPercentiles(b, parse, "completion_parse") + flowReportPercentiles(b, preparation, "preparation_wait") + flowReportPercentiles(b, joins, "completion_sigverify") + flowReportPercentiles(b, arrivals, "collection") + if prepareIdentities { + flowReportPercentiles(b, identityPreparation, "identity_admission") + flowReportPercentiles(b, fullToIdentities, "full_to_identities") + } + if after.InternalFaultFallbacks != before.InternalFaultFallbacks { + b.Fatal("signature verifier used an internal fault fallback") + } + var signatures uint64 + for _, tx := range source.Transactions { + signatures += uint64(len(tx.Signatures)) + } + if got, want := after.Signatures-before.Signatures, signatures*uint64(b.N); got != want { + b.Fatalf("verified %d signatures; want %d, exactly once per retained transaction", got, want) + } +} + +// BenchmarkEntryMessageIdentityArrival includes the first admission-time message +// identity lookup after assembly. Compare unchanged baseline and candidate with +// identical fixtures; full_to_identities includes any moved completion work. +// This does not include replay's whole-block duplicate map or transaction loop. +func BenchmarkEntryMessageIdentityArrival(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, arrival := range []struct { + name string + span time.Duration + }{{"catchup", 0}, {"tip_200ms", 200 * time.Millisecond}} { + b.Run(arrival.name, func(b *testing.B) { + runAssemblyFlowBenchmark(b, source.blk, fixture, 2, 8, arrival.span, true, true) + }) + } + }) + } +} + +// Holds one data shred in a batch three quarters through the block until the +// footer has arrived. Other complete batches continue arriving over 200 ms. +// This isolates discovery behind a gap; it is not a measured network replay. +func BenchmarkEntryPrefetchGapArrival(b *testing.B) { + flowConfigureBackend(b) + for _, source := range flowBenchmarkFixtures(b) { + b.Run(source.name, func(b *testing.B) { + fixture := makeAssemblyFlowFixture(b, source.blk) + for _, gap := range []bool{false, true} { + b.Run(fmt.Sprintf("gap_%t", gap), func(b *testing.B) { + fixture.holdGap = gap + runAssemblyFlowBenchmark(b, source.blk, fixture, 2, 8, 200*time.Millisecond, true, true) + }) + } + }) + } +} diff --git a/pkg/turbine/entry_prefetch_bounds_test.go b/pkg/turbine/entry_prefetch_bounds_test.go new file mode 100644 index 000000000..189ff25c3 --- /dev/null +++ b/pkg/turbine/entry_prefetch_bounds_test.go @@ -0,0 +1,50 @@ +package turbine + +import ( + "context" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +func TestEntryPrefetchByteBoundsFallBackToCompleteVerification(t *testing.T) { + for _, mode := range []string{"budget_full", "oversized_component"} { + t.Run(mode, func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + if mode == "budget_full" { + // Model other generations owning the entire encoded-byte budget. + a.mu.Lock() + p.bytes = entryPrefetchBytes + a.mu.Unlock() + defer func() { a.mu.Lock(); p.bytes -= entryPrefetchBytes; a.mu.Unlock() }() + } + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + if mode == "oversized_component" { + // Ordinary entry components permit trailing FEC padding. The + // complete decoder must accept this even though prefetch skips it. + raw = append(raw, make([]byte, entryPrefetchBatchBytes+1-len(raw))...) + } + const slot = 400 + batches := prefetchTestShreds(t, slot, raw, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + f := a.slots[slot].prefetch + return f != nil && !f.queued && len(f.batches) == 0 + }, 3*time.Second, time.Millisecond) + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 3) + require.True(t, blk.TransactionSignaturesVerified()) + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Zero(t, timings.EarlyVerifiedTransactions) + }) + } +} diff --git a/pkg/turbine/entry_prefetch_test.go b/pkg/turbine/entry_prefetch_test.go new file mode 100644 index 000000000..679ee18b5 --- /dev/null +++ b/pkg/turbine/entry_prefetch_test.go @@ -0,0 +1,528 @@ +package turbine + +import ( + "context" + "errors" + "fmt" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func prefetchTestPayload(t *testing.T, txs []*solana.Transaction) []byte { + t.Helper() + entry := Entry{NumHashes: 1, Hash: solana.Hash{7}, Txns: make([]solana.Transaction, len(txs))} + for i, tx := range txs { + entry.Txns[i] = *tx + } + raw, err := marshalEntryBatch([]Entry{entry}) + require.NoError(t, err) + return raw +} + +func prefetchTestShreds(t *testing.T, slot uint64, payloads ...[]byte) [][]*Shred { + t.Helper() + gen := ShredGenerator{Slot: slot, ParentSlot: slot - 1, Version: 1} + var nextData, nextCode uint32 + var root solana.Hash + batches := make([][]*Shred, len(payloads)) + for i, raw := range payloads { + packets, nextRoot, data, code, err := gen.MakeShredsFromData(testShredLeader(t), raw, i == len(payloads)-1, root, nextData, nextCode) + require.NoError(t, err) + root, nextData, nextCode = nextRoot, data, code + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + if shred.Type == ShredTypeData { + batches[i] = append(batches[i], shred) + } + } + } + return batches +} + +func feedPrefetchShreds(t *testing.T, a *SlotAssembler, shreds []*Shred) *block.Block { + t.Helper() + var result *block.Block + for _, shred := range shreds { + blk, err := a.AddShred(shred) + require.NoError(t, err) + if blk != nil { + require.Nil(t, result) + result = blk + } + } + return result +} + +func waitPrefetchedBatch(t *testing.T, a *SlotAssembler, slot uint64, start uint32) *prefetchedShredBatch { + t.Helper() + var batch *prefetchedShredBatch + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + if s := a.slots[slot]; s != nil && s.prefetch != nil { + batch = s.prefetch.batches[start] + } + return batch != nil + }, 3*time.Second, time.Millisecond) + waitSignal(t, batch.ready, "prefetched component preparation") + require.NoError(t, batch.err) + require.NoError(t, batch.submitErr) + return batch +} + +func TestEntryPrefetchVerifiesBeforeLastShredAndReusesResults(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 16, func(tx *solana.Transaction) error { + calls.Add(1) + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 100 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + require.NotNil(t, cached.verification) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(3), calls.Load()) + for range 5 { + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + } + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 7) + require.True(t, blk.TransactionSignaturesVerified()) + require.Equal(t, int32(7), calls.Load(), "cached transactions must not be verified twice") + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Equal(t, uint64(3), timings.EarlyVerifiedTransactions) + require.LessOrEqual(t, cached.verification.finishedAt.UnixNano(), blk.ShredFullNanos) +} + +func TestEntryPrefetchWaitsForGapAcrossMultipleFECSets(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { calls.Add(1); return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 104 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 300)), buildAlpenglowEndingTick(t)) + require.Greater(t, len(batches[0]), dataShredsPerFECBlock) + // Arrival of DATA_COMPLETE and later FEC sets cannot bypass the hole. + for i := len(batches[0]) - 1; i >= 0; i-- { + if i != 1 { + require.Nil(t, feedPrefetchShreds(t, a, batches[0][i:i+1])) + } + } + a.mu.Lock() + require.Empty(t, a.slots[slot].completeBatches) + require.Nil(t, a.slots[slot].prefetch) + a.mu.Unlock() + require.Zero(t, calls.Load()) + require.Nil(t, feedPrefetchShreds(t, a, batches[0][1:2])) + cached := waitPrefetchedBatch(t, a, slot, 0) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(300), calls.Load()) + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + require.Equal(t, int32(300), calls.Load()) +} + +func TestEntryPrefetchResetKeepsOldReservationsUntilReadersJoin(t *testing.T) { + txs := verifierSignedTransactions(t, 2) + txs[0].Signatures[0][9] ^= 0x40 + oldSignature := txs[0].Signatures[0] + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(tx *solana.Transaction) error { + if tx.Signatures[0] == oldSignature { + close(started) + <-release + } + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + const slot = 108 + old := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[:1]), buildAlpenglowEndingTick(t)) + fresh := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[1:]), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, old[0])) + waitSignal(t, started, "old generation verifier") + oldBatch := waitPrefetchedBatch(t, a, slot, 0) + a.ResetSlot(slot) + a.mu.Lock() + require.Equal(t, 1, p.slots) + require.Positive(t, p.bytes) + a.mu.Unlock() + require.Nil(t, feedPrefetchShreds(t, a, fresh[0])) + a.mu.Lock() + require.Equal(t, 2, p.slots) + a.mu.Unlock() + // With one verifier worker only one request may prefetch. The fresh + // generation keeps its pool reservation while admission waits for the old + // reader to join; it must not release or reuse the old generation's bytes. + releaseOnce.Do(func() { close(release) }) + _, err := oldBatch.verification.wait() + require.ErrorIs(t, err, context.Canceled) + newBatch := waitPrefetchedBatch(t, a, slot, 0) + require.NotSame(t, oldBatch, newBatch) + _, err = newBatch.verification.wait() + require.NoError(t, err) + blk := feedPrefetchShreds(t, a, fresh[1]) + require.NotNil(t, blk) + require.Equal(t, txs[1].Signatures[0], blk.Transactions[0].Signatures[0]) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots == 0 && p.bytes == 0 + }, 3*time.Second, time.Millisecond) +} + +func TestEntryPrefetchSaturationDoesNotBlockShredAdmissionAndShutdownJoins(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first, releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + first.Do(func() { close(started) }) + <-release + return nil + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + payload := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for i := 0; i < entryPrefetchSlots+1; i++ { + batches := prefetchTestShreds(t, uint64(200+i), payload, buildAlpenglowEndingTick(t)) + admitted := make(chan struct{}) + go func() { + defer close(admitted) + for _, sh := range batches[0] { + _, err := a.addShredFrom(sh, false) + if err != nil { + t.Errorf("admit shred: %v", err) + } + } + }() + waitSignal(t, admitted, "nonblocking ingress while prefetch saturated") + } + waitSignal(t, started, "blocked verifier") + a.mu.Lock() + require.Equal(t, entryPrefetchSlots, p.slots) + require.LessOrEqual(t, p.bytes, entryPrefetchBytes) + require.Nil(t, a.slots[200+entryPrefetchSlots].prefetch) + a.mu.Unlock() + done := make(chan struct{}) + go func() { p.closeAndWait(); close(done) }() + select { + case <-done: + t.Fatal("early pool released buffers before signature worker joined") + case <-time.After(20 * time.Millisecond): + } + releaseOnce.Do(func() { close(release) }) + waitSignal(t, done, "saturated pool shutdown") + a.mu.Lock() + require.Zero(t, p.slots) + require.Zero(t, p.bytes) + require.Nil(t, a.entryPrefetch) + a.mu.Unlock() +} + +func TestEntryPrefetchInvalidRetainedTransactionFailsClosed(t *testing.T) { + txs := verifierSignedTransactions(t, 3) + txs[1].Signatures[0][3] ^= 0x80 + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 300, prefetchTestPayload(t, txs), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, 300, 0) + index, err := cached.verification.wait() + require.Equal(t, 1, index) + require.ErrorContains(t, err, "invalid signature") + var finalErr error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + finalErr = err + } + } + require.ErrorContains(t, finalErr, "transaction 1") + require.False(t, a.SlotCompleted(300)) +} + +func TestEntryPrefetchUpdateParentDiscardsInvalidOptimisticPrefix(t *testing.T) { + txs := verifierSignedTransactions(t, 2) + txs[0].Signatures[0][3] ^= 0x80 + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 304 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, txs[:1]), + testAlpenglowParentMarkerBytes(blockMarkerVariantUpdateParent, slot-2, solana.Hash{12}), + prefetchTestPayload(t, txs[1:])) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + _, err := cached.verification.wait() + require.ErrorContains(t, err, "invalid signature") + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 1) + require.Equal(t, txs[1].Signatures[0], blk.Transactions[0].Signatures[0]) + require.Equal(t, uint64(slot-2), blk.SourceParentSlot) + require.True(t, blk.TransactionSignaturesVerified()) +} + +func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first, releaseOnce sync.Once + v := newTransactionVerifier(1, 8, func(tx *solana.Transaction) error { + first.Do(func() { close(started); <-release }) + return txverify.VerifyTransaction(tx) + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + const slot = 308 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitSignal(t, started, "early verification") + cached := waitPrefetchedBatch(t, a, slot, 0) + var work *slotCompletionWork + for _, sh := range batches[1] { + var err error + work, err = a.addShredFrom(sh, false) + require.NoError(t, err) + } + require.NotNil(t, work) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + canceled := make(chan struct{}) + originalCancel := cached.verification.cancel + cached.verification.cancel = func() { close(canceled); originalCancel() } + done := make(chan processedSlotCompletion, 1) + go func() { done <- a.processCompletion(ctx, work) }() + // The completion has no expensive decode left and blocks joining this + // one already-prepared future; cancel while it owns those transactions. + time.Sleep(20 * time.Millisecond) + cancel() + waitSignal(t, canceled, "completion canceled its signature request") + releaseOnce.Do(func() { close(release) }) + var processed processedSlotCompletion + select { + case processed = <-done: + case <-time.After(3 * time.Second): + t.Fatal("canceled completion failed to join") + } + require.True(t, processed.canceled) + _, err := a.finalizeCompletion(work, processed) + require.NoError(t, err) + _, err = cached.verification.wait() + require.True(t, errors.Is(err, context.Canceled)) + a.mu.Lock() + retry := a.claimCompletionLocked(a.slots[slot], false) + a.mu.Unlock() + require.NotNil(t, retry) + processed = a.processCompletion(context.Background(), retry) + require.NoError(t, processed.err) + blk, err := a.finalizeCompletion(retry, processed) + require.NoError(t, err) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) +} + +// A complete later batch must verify while an earlier batch still has a gap. +// Its preceding DATA_COMPLETE shred remains necessary to establish its start. +func TestEntryPrefetchBypassesEarlierGap(t *testing.T) { + for _, lateBoundary := range []bool{false, true} { + t.Run(fmt.Sprint("lateBoundary=", lateBoundary), func(t *testing.T) { + var calls atomic.Int32 + v := newTransactionVerifier(2, 8, func(tx *solana.Transaction) error { calls.Add(1); return txverify.VerifyTransaction(tx) }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 909 + txs := verifierSignedTransactions(t, 60) + batches := prefetchTestShreds(t, slot, prefetchTestPayload(t, txs[:30]), prefetchTestPayload(t, txs[30:]), buildAlpenglowEndingTick(t)) + require.Greater(t, len(batches[0]), 2) + end := len(batches[0]) - 1 + for i, sh := range batches[0] { + if i == 1 || (lateBoundary && i == end) { + continue + } + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + if lateBoundary { + a.mu.Lock() + empty := len(a.slots[slot].completeBatches) == 0 + a.mu.Unlock() + require.True(t, empty, "unknown preceding boundary must prevent speculation") + require.Nil(t, feedPrefetchShreds(t, a, batches[0][end:])) + } + later := waitPrefetchedBatch(t, a, slot, batches[1][0].Index) + _, err := later.verification.wait() + require.NoError(t, err) + require.Equal(t, int32(30), calls.Load()) + require.Nil(t, feedPrefetchShreds(t, a, batches[0][1:2])) + earlier := waitPrefetchedBatch(t, a, slot, 0) + _, err = earlier.verification.wait() + require.NoError(t, err) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) + require.Len(t, blk.Transactions, 60) + require.Equal(t, int32(60), calls.Load(), "every signature verified exactly once") + for i, tx := range txs { + require.Equal(t, tx.Signatures[0], blk.Transactions[i].Signatures[0]) + } + }) + } +} + +func TestEntryPrefetchDiscoversRecoveredBoundary(t *testing.T) { + v := newTransactionVerifier(2, 8, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + const slot = 910 + txs := verifierSignedTransactions(t, 60) + gen := ShredGenerator{Slot: slot, ParentSlot: slot - 1, Version: 1} + raw := prefetchTestPayload(t, txs[:30]) + packets, root, nextData, nextCode, err := gen.MakeShredsFromData(testShredLeader(t), raw, false, solana.Hash{}, 0, 0) + require.NoError(t, err) + var code []*Shred + for _, packet := range packets { + sh, err := ParseShred(packet) + require.NoError(t, err) + if sh.Type == ShredTypeCode { + code = append(code, sh) + continue + } + if !sh.DataComplete() { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + } + packets, _, _, _, err = gen.MakeShredsFromData(testShredLeader(t), prefetchTestPayload(t, txs[30:]), false, root, nextData, nextCode) + require.NoError(t, err) + for _, packet := range packets { + sh, err := ParseShred(packet) + require.NoError(t, err) + if sh.Type == ShredTypeData { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + } + a.mu.Lock() + empty := len(a.slots[slot].completeBatches) == 0 + a.mu.Unlock() + require.True(t, empty) + require.NotEmpty(t, code) + for _, sh := range code { + require.Nil(t, feedPrefetchShreds(t, a, []*Shred{sh})) + } + later := waitPrefetchedBatch(t, a, slot, nextData) + _, err = later.verification.wait() + require.NoError(t, err) + first := waitPrefetchedBatch(t, a, slot, 0) + _, err = first.verification.wait() + require.NoError(t, err) +} + +func TestEntryPrefetchIndexDisabledAndLateInstall(t *testing.T) { + a := NewSlotAssembler() + batches := prefetchTestShreds(t, 911, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + a.mu.Lock() + absent := a.slots[911].batchIndex == nil + a.mu.Unlock() + require.True(t, absent) + v := newTransactionVerifier(2, 8, nil) + defer v.closeAndWait() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + // Seeding also works on a coding-only admission with no new data recovery. + a.mu.Lock() + a.prefetchEntriesLocked(a.slots[911]) + a.mu.Unlock() + cached := waitPrefetchedBatch(t, a, 911, 0) + _, err := cached.verification.wait() + require.NoError(t, err) +} + +func TestEntryPrefetchResetRetainsQueuedReservations(t *testing.T) { + a := NewSlotAssembler() + ctx, cancel := context.WithCancel(context.Background()) + // Hold workers until after reset so all tokens remain in the channel. + p := &entryPrefetchPool{a: a, ctx: ctx, cancel: cancel, jobs: make(chan *slotState, entryPrefetchSlots)} + a.entryPrefetch = p + startWorker := sync.OnceFunc(func() { + p.workers.Add(1) + go p.run() + }) + defer func() { + startWorker() + p.closeAndWait() + }() + for i := 0; i < entryPrefetchSlots; i++ { + s := &slotState{slot: uint64(i), completeBatches: []shredBatchRange{{0, 0}}} + a.mu.Lock() + a.slots[s.slot] = s + a.prefetchEntriesLocked(s) + a.releasePrefetchLocked(s) + a.mu.Unlock() + } + require.Equal(t, entryPrefetchSlots, len(p.jobs)) + // Cleanup must not admit another generation while stale queue tokens live. + require.Never(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots != entryPrefetchSlots + }, 50*time.Millisecond, time.Millisecond) + fresh := &slotState{slot: 100, completeBatches: []shredBatchRange{{0, 0}}} + a.mu.Lock() + a.prefetchEntriesLocked(fresh) + reserved := fresh.prefetch != nil + a.mu.Unlock() + // Start the worker before assertions so test failures cannot strand cleanup. + startWorker() + require.False(t, reserved) + p.cleanup.Wait() + a.mu.Lock() + slots := p.slots + a.mu.Unlock() + require.Zero(t, slots) +} diff --git a/pkg/turbine/fec_root_cache.go b/pkg/turbine/fec_root_cache.go new file mode 100644 index 000000000..cb4395b01 --- /dev/null +++ b/pkg/turbine/fec_root_cache.go @@ -0,0 +1,56 @@ +package turbine + +import ( + "bytes" + + "github.com/gagliardetto/solana-go" +) + +// One snapshot per FEC generation binds an authenticated root to its source. +// It is written only during assembly, before the state is frozen. Completion +// never trusts pointer identity alone or changes the deterministic root choice. +type authenticatedFECRoot struct { + source *Shred + input shredRootInput + payload []byte + root solana.Hash +} + +// These are every parsed field read by MerkleRoot. The exact payload comparison +// covers headers, data, proof, and trailers, even for noncanonical test input. +type shredRootInput struct { + variant byte + kind ShredType + index, fecSetIndex uint32 + dataCount, position uint16 +} + +func rootInput(s *Shred) shredRootInput { + return shredRootInput{s.Variant, s.Type, s.Index, s.FECSetIndex, s.NumDataShreds, s.Position} +} + +func hasMerkleRootProof(s *Shred) bool { + return s != nil && isMerkleVariant(s.Variant) && (s.Type == ShredTypeData || s.Type == ShredTypeCode) +} + +func (c *authenticatedFECRoot) matches(s *Shred) bool { + return s == c.source && rootInput(s) == c.input && bytes.Equal(s.Payload, c.payload) +} + +func (f *fecState) rememberAuthenticatedRoot(s *Shred, root solana.Hash) { + if s.Recovered || !isMerkleVariant(s.Variant) { + return + } + if cached := f.rootCache; cached != nil { + // Data proofs precede coding proofs, and the lowest index wins. An + // unauthenticated earlier arrival can still force fallback at completion. + old := cached.input + if old.kind == ShredTypeData && (s.Type != ShredTypeData || s.Index >= old.index) { + return + } + if old.kind == ShredTypeCode && s.Type == ShredTypeCode && s.Position >= old.position { + return + } + } + f.rootCache = &authenticatedFECRoot{source: s, input: rootInput(s), payload: bytes.Clone(s.Payload), root: root} +} diff --git a/pkg/turbine/generate.go b/pkg/turbine/generate.go index 14626e214..f8a4e6343 100644 --- a/pkg/turbine/generate.go +++ b/pkg/turbine/generate.go @@ -4,6 +4,7 @@ import ( "crypto/ed25519" "encoding/binary" "fmt" + "sync" "github.com/gagliardetto/solana-go" "github.com/klauspost/reedsolomon" @@ -14,6 +15,21 @@ const ( proofEntriesFor32x32 = 6 ) +var erasureEncoderPool sync.Pool + +func acquireErasureEncoder() (reedsolomon.Encoder, error) { + if encoder := erasureEncoderPool.Get(); encoder != nil { + return encoder.(reedsolomon.Encoder), nil + } + return reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) +} + +func releaseErasureEncoder(encoder reedsolomon.Encoder) { + if encoder != nil { + erasureEncoderPool.Put(encoder) + } +} + // ShredGenerator builds merkle FEC shreds from a serialized byte buffer. type ShredGenerator struct { Slot uint64 @@ -22,6 +38,14 @@ type ShredGenerator struct { ReferenceTick uint8 } +// shredPackets retains the roots already computed during generation, in FEC +// order, so broadcast can commit to them without parsing the packets again. +type shredPackets struct { + packets [][]byte + fecSetRoots []solana.Hash + chainedMerkleRoot solana.Hash +} + func dataCapacity(proofSize uint8, resigned bool) int { capacity := dataPayloadSize - dataHeaderSize - merkleRootSize - int(proofSize)*merkleProofEntrySize if resigned { @@ -66,9 +90,31 @@ func (g *ShredGenerator) MakeShredsFromData( nextShredIndex uint32, nextCodeIndex uint32, ) ([][]byte, solana.Hash, uint32, uint32, error) { + batch, nextData, nextCode, err := g.makeShredsFromData( + leader, data, isLastInSlot, chainedMerkleRoot, nextShredIndex, nextCodeIndex, + ) + return batch.packets, batch.chainedMerkleRoot, nextData, nextCode, err +} + +func (g *ShredGenerator) makeShredsFromData( + leader solana.PrivateKey, + data []byte, + isLastInSlot bool, + chainedMerkleRoot solana.Hash, + nextShredIndex uint32, + nextCodeIndex uint32, +) (shredPackets, uint32, uint32, error) { if g.Slot < g.ParentSlot || g.Slot-g.ParentSlot > uint64(^uint16(0)) { - return nil, solana.Hash{}, nextShredIndex, nextCodeIndex, fmt.Errorf("invalid parent slot %d for slot %d", g.ParentSlot, g.Slot) + return shredPackets{}, nextShredIndex, nextCodeIndex, fmt.Errorf("invalid parent slot %d for slot %d", g.ParentSlot, g.Slot) + } + // The 32+32 coding matrix is invariant across every FEC set in this + // operation. Building it requires a Vandermonde inversion, so retain the + // encoder for the whole payload rather than reconstructing it per set. + encoder, err := acquireErasureEncoder() + if err != nil { + return shredPackets{}, nextShredIndex, nextCodeIndex, err } + defer releaseErasureEncoder(encoder) proofSize := uint8(proofEntriesFor32x32) unsignedCap := dataCapacity(proofSize, false) signedCap := dataCapacity(proofSize, true) @@ -92,49 +138,62 @@ func (g *ShredGenerator) MakeShredsFromData( } var packets [][]byte + var fecSetRoots []solana.Hash dataIndex := nextShredIndex codeIndex := nextCodeIndex chainedRoot := chainedMerkleRoot + // DATA_COMPLETE ends the serialized component, which may span FEC sets. for len(unsignedData) >= unsignedBatch { batch := unsignedData[:unsignedBatch] unsignedData = unsignedData[unsignedBatch:] - batchPackets, root, err := g.makeFECBatch(leader, batch, unsignedCap, proofSize, false, parentOffset, flags, false, chainedRoot, dataIndex, codeIndex) + // DATA_COMPLETE marks the end of the serialized component, not the end + // of every FEC set. A full unsigned batch is complete only when no + // unsigned remainder or signed-last batch follows it. + dataComplete := len(unsignedData) == 0 && len(signedData) == 0 + batchPackets, root, err := g.makeFECBatch(encoder, leader, batch, unsignedCap, proofSize, false, parentOffset, flags, dataComplete, false, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } if len(unsignedData) > 0 || (len(packets) == 0 && !isLastInSlot) { - batchPackets, root, err := g.makeFECBatch(leader, unsignedData, unsignedCap, proofSize, false, parentOffset, flags, false, chainedRoot, dataIndex, codeIndex) + dataComplete := len(signedData) == 0 + batchPackets, root, err := g.makeFECBatch(encoder, leader, unsignedData, unsignedCap, proofSize, false, parentOffset, flags, dataComplete, false, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } if len(signedData) > 0 || (len(packets) == 0 && isLastInSlot) { - batchPackets, root, err := g.makeFECBatch(leader, signedData, signedCap, proofSize, true, parentOffset, flags, isLastInSlot, chainedRoot, dataIndex, codeIndex) + batchPackets, root, err := g.makeFECBatch(encoder, leader, signedData, signedCap, proofSize, true, parentOffset, flags, true, isLastInSlot, chainedRoot, dataIndex, codeIndex) if err != nil { - return nil, solana.Hash{}, dataIndex, codeIndex, err + return shredPackets{}, dataIndex, codeIndex, err } packets = append(packets, batchPackets...) + fecSetRoots = append(fecSetRoots, root) chainedRoot = root dataIndex += dataShredsPerFECBlock codeIndex += codingShredsPerFECBlock } - return packets, chainedRoot, dataIndex, codeIndex, nil + return shredPackets{ + packets: packets, fecSetRoots: fecSetRoots, chainedMerkleRoot: chainedRoot, + }, dataIndex, codeIndex, nil } func (g *ShredGenerator) makeFECBatch( + encoder reedsolomon.Encoder, leader solana.PrivateKey, data []byte, dataCap int, @@ -142,6 +201,7 @@ func (g *ShredGenerator) makeFECBatch( resigned bool, parentOffset uint16, flags byte, + dataComplete bool, isLastInSlot bool, chainedMerkleRoot solana.Hash, dataIndex uint32, @@ -196,11 +256,11 @@ func (g *ShredGenerator) makeFECBatch( dataPackets[i][dataFlagsOffset] |= shredFlagLastShredInSlot break } - } else if len(dataPackets) > 0 { + } else if dataComplete && len(dataPackets) > 0 { dataPackets[len(dataPackets)-1][dataFlagsOffset] |= shredFlagDataComplete } - root, err := finishErasureBatch(leader, allPackets, chainedMerkleRoot, proofSize, resigned) + root, err := finishErasureBatch(encoder, leader, allPackets, chainedMerkleRoot, proofSize, resigned) if err != nil { return nil, solana.Hash{}, err } @@ -208,101 +268,70 @@ func (g *ShredGenerator) makeFECBatch( } func finishErasureBatch( + encoder reedsolomon.Encoder, leader solana.PrivateKey, packets [][]byte, chainedMerkleRoot solana.Hash, proofSize uint8, resigned bool, ) (solana.Hash, error) { - encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if len(packets) != dataShredsPerFECBlock+codingShredsPerFECBlock { + return solana.Hash{}, fmt.Errorf("invalid FEC packet count %d", len(packets)) + } + dataCap, err := merkleCapacity(dataPayloadSize, dataHeaderSize, proofSize, true, resigned) if err != nil { return solana.Hash{}, err } + codeCap, err := merkleCapacity(codingPayloadSize, codingHeaderSize, proofSize, true, resigned) + if err != nil { + return solana.Hash{}, err + } + dataVariant := chainedDataVariant(proofSize, resigned) + codeVariant := chainedCodeVariant(proofSize, resigned) + // These packets were constructed immediately above, so retain direct views + // of their erasure regions. ParseShred is intentionally a defensive, + // owning parser for untrusted network packets; using it here would allocate + // and copy every packet several times only to copy the same bytes back. shards := make([][]byte, len(packets)) for i, packet := range packets { - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, fmt.Errorf("parse batch shred %d: %w", i, err) + if i < dataShredsPerFECBlock { + if len(packet) < dataPayloadSize || packet[shredVariantOffset] != dataVariant { + return solana.Hash{}, fmt.Errorf("invalid generated data shred %d", i) + } + shards[i] = packet[shredSignatureSize : dataHeaderSize+dataCap] + continue } - shard, err := shred.erasureShard() - if err != nil { - return solana.Hash{}, fmt.Errorf("erasure shard %d: %w", i, err) + if len(packet) < codingPayloadSize || packet[shredVariantOffset] != codeVariant { + return solana.Hash{}, fmt.Errorf("invalid generated coding shred %d", i-dataShredsPerFECBlock) } - shards[i] = shard + shards[i] = packet[codingHeaderSize : codingHeaderSize+codeCap] } if err := encoder.Encode(shards); err != nil { return solana.Hash{}, fmt.Errorf("reed-solomon encode: %w", err) } for i, packet := range packets { - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, err - } - proofSizeInfo, chained, resignedFlag, ok := merkleVariantInfo(shred.Variant) - if !ok { - return solana.Hash{}, ErrUnsupportedShred - } - _ = proofSizeInfo - _ = chained - _ = resignedFlag - - capacity, err := merkleCapacity(len(packet), dataHeaderSize, proofSize, true, resigned) - if shred.Type == ShredTypeCode { - capacity, err = merkleCapacity(len(packet), codingHeaderSize, proofSize, true, resigned) - } - if err != nil { - return solana.Hash{}, err - } - rootOffset := dataHeaderSize + capacity - if shred.Type == ShredTypeCode { - rootOffset = codingHeaderSize + capacity + rootOffset := dataHeaderSize + dataCap + if i >= dataShredsPerFECBlock { + rootOffset = codingHeaderSize + codeCap } copy(packet[rootOffset:rootOffset+merkleRootSize], chainedMerkleRoot[:]) - - if shred.Type == ShredTypeCode { - start := codingHeaderSize - end := start + capacity - copy(packet[start:end], shards[i]) - } else { - start := shredSignatureSize - end := dataHeaderSize + capacity - copy(packet[start:end], shards[i]) - } } - nodes, err := buildMerkleTree(packets) - if err != nil { - return solana.Hash{}, err - } + nodes := buildGeneratedMerkleTree(packets, dataCap, codeCap) root := nodes[len(nodes)-1] sig := ed25519.Sign(ed25519.PrivateKey(leader), root[:]) - for _, packet := range packets { + for i, packet := range packets { copy(packet[shredSignatureOffset:shredSignatureSize], sig) - shred, err := ParseShred(packet) - if err != nil { - return solana.Hash{}, err - } - leafIndex, err := shred.merkleLeafIndex() - if err != nil { - return solana.Hash{}, err - } - proof := makeMerkleProof(nodes, leafIndex, len(packets)) - capacity, err := merkleCapacity(len(packet), dataHeaderSize, proofSize, true, resigned) - if shred.Type == ShredTypeCode { - capacity, err = merkleCapacity(len(packet), codingHeaderSize, proofSize, true, resigned) - } - if err != nil { - return solana.Hash{}, err - } - proofOffset := dataHeaderSize + capacity + merkleRootSize - if shred.Type == ShredTypeCode { - proofOffset = codingHeaderSize + capacity + merkleRootSize + proofOffset := dataHeaderSize + dataCap + merkleRootSize + if i >= dataShredsPerFECBlock { + proofOffset = codingHeaderSize + codeCap + merkleRootSize } - for j, entry := range proof { - copy(packet[proofOffset+j*merkleProofEntrySize:], entry[:]) + proofEntries := writeMerkleProof(packet[proofOffset:], nodes, i, len(packets)) + if proofEntries != int(proofSize) { + return solana.Hash{}, fmt.Errorf("generated merkle proof has %d entries, want %d", proofEntries, proofSize) } if resigned { retransmitOffset := proofOffset + int(proofSize)*merkleProofEntrySize @@ -312,6 +341,54 @@ func finishErasureBatch( return root, nil } +// buildGeneratedMerkleTree hashes the fixed packet order emitted by +// makeFECBatch: 32 data shreds followed by 32 coding shreds. Callers must have +// already validated the packet sizes and variants in finishErasureBatch. +func buildGeneratedMerkleTree(packets [][]byte, dataCap, codeCap int) []solana.Hash { + leaves := make([]solana.Hash, len(packets)) + for i, packet := range packets { + end := dataHeaderSize + dataCap + merkleRootSize + if i >= dataShredsPerFECBlock { + end = codingHeaderSize + codeCap + merkleRootSize + } + leaves[i] = merkleHashLeaf(packet[shredSignatureSize:end]) + } + + nodes := make([]solana.Hash, 0, merkleTreeSize(len(leaves))) + nodes = append(nodes, leaves...) + for size := len(leaves); size > 1; size = (size + 1) >> 1 { + offset := len(nodes) - size + for index := offset; index < offset+size; index += 2 { + other := index + 1 + if other >= offset+size { + other = offset + size - 1 + } + nodes = append(nodes, merkleHashNode(nodes[index][:merkleProofEntrySize], nodes[other][:merkleProofEntrySize])) + } + } + return nodes +} + +// writeMerkleProof writes the truncated sibling hashes directly into a packet. +// The generated FEC tree has fixed depth, so materializing a temporary proof +// slice for every one of its 64 packets only adds allocator and copy traffic. +func writeMerkleProof(dst []byte, nodes []solana.Hash, index, size int) int { + entries := 0 + offset := 0 + for size > 1 { + sibling := index ^ 1 + if sibling >= size { + sibling = size - 1 + } + copy(dst[entries*merkleProofEntrySize:], nodes[offset+sibling][:merkleProofEntrySize]) + entries++ + offset += size + size = (size + 1) >> 1 + index >>= 1 + } + return entries +} + func buildMerkleTree(packets [][]byte) ([]solana.Hash, error) { leaves := make([]solana.Hash, len(packets)) for i, packet := range packets { diff --git a/pkg/turbine/generate_bench_test.go b/pkg/turbine/generate_bench_test.go new file mode 100644 index 000000000..2301eefaa --- /dev/null +++ b/pkg/turbine/generate_bench_test.go @@ -0,0 +1,242 @@ +package turbine + +import ( + "crypto/ed25519" + "crypto/sha256" + "encoding/hex" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/costmodel" + "github.com/gagliardetto/solana-go" + "github.com/klauspost/reedsolomon" +) + +const ( + benchmarkTransactionCount = 50_000 + benchmarkTransactionBytes = 1_232 +) + +var ( + benchmarkEncoderSink reedsolomon.Encoder + benchmarkPacketsSink [][]byte + benchmarkRootSink solana.Hash + benchmarkByteSink byte +) + +func benchmarkLeaderKey() solana.PrivateKey { + var seed [ed25519.SeedSize]byte + for i := range seed { + seed[i] = byte(i + 1) + } + return solana.PrivateKey(ed25519.NewKeyFromSeed(seed[:])) +} + +func benchmarkPayload(size int) []byte { + payload := make([]byte, size) + var state uint64 = 0x9e3779b97f4a7c15 + for i := range payload { + // A deterministic, non-zero corpus avoids accidentally benchmarking a + // special all-zero input while keeping fixture construction out of the + // timed region. + state ^= state << 7 + state ^= state >> 9 + state ^= state << 8 + payload[i] = byte(state) + } + return payload +} + +// BenchmarkReedSolomonEncode32x32 isolates the arithmetic kernel used by one +// unsigned 32+32 chained FEC set. Encoder construction, shred parsing, Merkle +// hashing, signing, and packet copies are intentionally outside this result. +func BenchmarkReedSolomonEncode32x32(b *testing.B) { + const shardBytes = 987 // unsigned chained 32+32 shreds with proof size 6 + encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if err != nil { + b.Fatal(err) + } + shards := make([][]byte, dataShredsPerFECBlock+codingShredsPerFECBlock) + for i := range shards { + shards[i] = make([]byte, shardBytes) + if i < dataShredsPerFECBlock { + copy(shards[i], benchmarkPayload(shardBytes)) + } + } + + b.SetBytes(dataShredsPerFECBlock * shardBytes) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := encoder.Encode(shards); err != nil { + b.Fatal(err) + } + } + benchmarkByteSink = shards[len(shards)-1][shardBytes-1] +} + +// BenchmarkReedSolomonNew32x32 measures work that finishErasureBatch currently +// repeats for every FEC set even though the 32+32 shape never changes. +func BenchmarkReedSolomonNew32x32(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + encoder, err := reedsolomon.New(dataShredsPerFECBlock, codingShredsPerFECBlock) + if err != nil { + b.Fatal(err) + } + benchmarkEncoderSink = encoder + } +} + +// BenchmarkMakeShredsFromData reports the complete current generator cost, +// including packet construction, Reed-Solomon coding, chained Merkle trees, +// one Ed25519 signature per FEC set, and proof materialization. +func BenchmarkMakeShredsFromData(b *testing.B) { + const blockBytes = benchmarkTransactionCount * benchmarkTransactionBytes + cases := []struct { + name string + size int + isLastInSlot bool + }{ + {name: "one-unsigned-fec", size: dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false)}, + {name: "block-50000x1232", size: blockBytes, isLastInSlot: true}, + } + + for _, tc := range cases { + b.Run(tc.name, func(b *testing.B) { + leader := benchmarkLeaderKey() + payload := benchmarkPayload(tc.size) + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + + b.SetBytes(int64(tc.size)) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + packets, root, _, _, err := gen.MakeShredsFromData(leader, payload, tc.isLastInSlot, solana.Hash{}, 0, 0) + if err != nil { + b.Fatal(err) + } + benchmarkPacketsSink = packets + benchmarkRootSink = root + } + b.StopTimer() + if len(benchmarkPacketsSink) == 0 || len(benchmarkPacketsSink)%64 != 0 { + b.Fatalf("unexpected packet count %d", len(benchmarkPacketsSink)) + } + b.ReportMetric(float64(len(benchmarkPacketsSink)), "packets/op") + b.ReportMetric(float64(len(benchmarkPacketsSink)/64), "FEC-sets/op") + if tc.size == blockBytes { + b.ReportMetric(benchmarkTransactionCount, "transactions/op") + } + }) + } +} + +// BenchmarkMakeShreds50000TargetBatches1232 models the producer's target-sized +// component stream without retaining a multi-gigabyte output. It measures only +// the 61.6 MB transaction payload; entry framing is deliberately outside this +// erasure-coding benchmark. +func BenchmarkMakeShreds50000TargetBatches1232(b *testing.B) { + const inputBytes = benchmarkTransactionCount * benchmarkTransactionBytes + leader := benchmarkLeaderKey() + payload := benchmarkPayload(inputBytes) + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + + b.SetBytes(inputBytes) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + var ( + root solana.Hash + dataIndex uint32 + codeIndex uint32 + ) + var ( + offset int + components int + packetCount int + fecSetCount int + ) + for offset < len(payload) { + end := min(offset+costmodel.DefaultTargetBatchBytes, len(payload)) + packets, nextRoot, nextData, nextCode, err := gen.MakeShredsFromData( + leader, + payload[offset:end], + end == len(payload), + root, + dataIndex, + codeIndex, + ) + if err != nil { + b.Fatal(err) + } + benchmarkPacketsSink = packets + components++ + packetCount += len(packets) + fecSetCount += len(packets) / (dataShredsPerFECBlock + codingShredsPerFECBlock) + root, dataIndex, codeIndex = nextRoot, nextData, nextCode + offset = end + } + b.ReportMetric(float64(components), "components/op") + b.ReportMetric(float64(fecSetCount), "FEC-sets/op") + b.ReportMetric(float64(packetCount), "packets/op") + benchmarkRootSink = root + } + b.StopTimer() + b.ReportMetric(benchmarkTransactionCount, "transactions/op") +} + +func TestBenchmarkBlockPayloadAccounting(t *testing.T) { + const blockBytes = benchmarkTransactionCount * benchmarkTransactionBytes + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + unsignedBytes := blockBytes - signedBatch + unsignedFECs := (unsignedBytes + unsignedBatch - 1) / unsignedBatch + totalFECs := unsignedFECs + 1 + if totalFECs != 2000 { + t.Fatalf("50k x 1232 payload maps to %d FEC sets, want 2000 (%s)", totalFECs, fmt.Sprintf("%d bytes", blockBytes)) + } +} + +func TestProducerTargetMatchesTwoTypicalFECPayloads(t *testing.T) { + want := 2 * dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + if costmodel.DefaultTargetBatchBytes != want { + t.Fatalf("producer target = %d, want two typical FEC payloads = %d", costmodel.DefaultTargetBatchBytes, want) + } +} + +func TestMakeShredsFromDataStableBytes(t *testing.T) { + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + tests := []struct { + name string + size int + isLastInSlot bool + want string + }{ + {name: "unsigned-one-fec", size: unsignedBatch, want: "f9334f1835240df21d4b48a09f35b3ff90578122d30d0527608c10d38d0911f7"}, + {name: "signed-one-fec", size: signedBatch, isLastInSlot: true, want: "bfa398c445509c5e1345553bbe84fea04e86001caf441008b4014d07c6e36ccd"}, + {name: "two-unsigned-one-signed", size: 2*unsignedBatch + signedBatch, isLastInSlot: true, want: "dffae840e4c247680b5e1667747a63138872a0080c51a6fcf4cc5002eb7778ac"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1, ReferenceTick: 17} + packets, _, _, _, err := gen.MakeShredsFromData( + benchmarkLeaderKey(), benchmarkPayload(tt.size), tt.isLastInSlot, + solana.Hash{3}, 7, 11, + ) + if err != nil { + t.Fatal(err) + } + h := sha256.New() + for _, packet := range packets { + _, _ = h.Write(packet) + } + got := hex.EncodeToString(h.Sum(nil)) + if got != tt.want { + t.Fatalf("packet digest %s, want %s", got, tt.want) + } + }) + } +} diff --git a/pkg/turbine/generate_test.go b/pkg/turbine/generate_test.go index 083133954..64c5d1091 100644 --- a/pkg/turbine/generate_test.go +++ b/pkg/turbine/generate_test.go @@ -4,6 +4,7 @@ import ( "bytes" "crypto/ed25519" "encoding/binary" + "fmt" "testing" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" @@ -174,6 +175,46 @@ func TestMakeShredsFromDataRoundTrip(t *testing.T) { } } +func TestGeneratedFECRootsMatchPacketProofs(t *testing.T) { + unsignedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signedBatch := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + for _, last := range []bool{false, true} { + for _, size := range []int{0, 1, signedBatch, signedBatch + 1, unsignedBatch, unsignedBatch + 1, 2 * unsignedBatch, 2*unsignedBatch + signedBatch} { + t.Run(fmt.Sprintf("last=%t/bytes=%d", last, size), func(t *testing.T) { + gen := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7, ReferenceTick: 17} + parentRoot := solana.Hash{5} + leader := testShredLeader(t) + batch, nextData, nextCode, err := gen.makeShredsFromData( + leader, benchmarkPayload(size), last, parentRoot, 7, 11, + ) + require.NoError(t, err) + require.NotEmpty(t, batch.fecSetRoots) + require.Len(t, batch.packets, len(batch.fecSetRoots)*(dataShredsPerFECBlock+codingShredsPerFECBlock)) + require.Equal(t, uint32(7+len(batch.fecSetRoots)*dataShredsPerFECBlock), nextData) + require.Equal(t, uint32(11+len(batch.fecSetRoots)*codingShredsPerFECBlock), nextCode) + require.Equal(t, batch.fecSetRoots[len(batch.fecSetRoots)-1], batch.chainedMerkleRoot) + for i, packet := range batch.packets { + fec := i / (dataShredsPerFECBlock + codingShredsPerFECBlock) + shred, err := ParseShred(packet) + require.NoError(t, err) + root, err := shred.MerkleRoot() + require.NoError(t, err) + require.Equal(t, batch.fecSetRoots[fec], root) + require.Equal(t, uint32(7+fec*dataShredsPerFECBlock), shred.FECSetIndex) + previousRoot := parentRoot + if fec > 0 { + previousRoot = batch.fecSetRoots[fec-1] + } + embeddedRoot, err := shred.EmbeddedChainedMerkleRoot() + require.NoError(t, err) + require.Equal(t, previousRoot, embeddedRoot) + require.NoError(t, shred.VerifySignature(leader.PublicKey())) + } + }) + } + } +} + func TestMakeShredsFromAlpenglowBlock(t *testing.T) { leader := testShredLeader(t) gen := ShredGenerator{ @@ -184,10 +225,10 @@ func TestMakeShredsFromAlpenglowBlock(t *testing.T) { } var ( - chainedRoot = solana.Hash{5} - nextData uint32 = 0 - nextCode uint32 = 0 - allDataShreds []*Shred + chainedRoot = solana.Hash{5} + nextData uint32 = 0 + nextCode uint32 = 0 + allDataShreds []*Shred ) for _, component := range buildAlpenglowSlot(t) { packets, root, newData, newCode, err := gen.MakeShredsFromData( diff --git a/pkg/turbine/generated_component_boundary_test.go b/pkg/turbine/generated_component_boundary_test.go new file mode 100644 index 000000000..871843089 --- /dev/null +++ b/pkg/turbine/generated_component_boundary_test.go @@ -0,0 +1,47 @@ +package turbine + +import ( + "bytes" + "fmt" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Exercise exact FEC boundaries and the smaller resigned final-set capacity. +// The receiver must see one DATA_COMPLETE per serialized component, regardless +// of how many FEC sets carry it, with proofs authenticating the final flags. +func TestGeneratedComponentHasOneCompletionBoundary(t *testing.T) { + unsigned := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, false) + signed := dataShredsPerFECBlock * dataCapacity(proofEntriesFor32x32, true) + for _, last := range []bool{false, true} { + for _, size := range []int{0, 1, signed, signed + 1, unsigned, unsigned + 1, 2 * unsigned, 2*unsigned + signed} { + t.Run(fmt.Sprintf("last=%t/bytes=%d", last, size), func(t *testing.T) { + leader := testShredLeader(t) + gen := ShredGenerator{Slot: 100, ParentSlot: 99, Version: 7} + payload := bytes.Repeat([]byte{0x5a}, size) + packets, _, nextData, _, err := gen.MakeShredsFromData(leader, payload, last, solana.Hash{}, 0, 0) + require.NoError(t, err) + var decoded []byte + var completed int + for _, packet := range packets { + shred, err := ParseShred(packet) + require.NoError(t, err) + require.NoError(t, shred.VerifySignature(leader.PublicKey())) + if shred.Type != ShredTypeData { + continue + } + decoded = append(decoded, shred.Data...) + if shred.DataComplete() { + completed++ + require.Equal(t, nextData-1, shred.Index) + } + require.Equal(t, last && shred.Index == nextData-1, shred.LastInSlot()) + } + require.Equal(t, 1, completed) + require.True(t, bytes.Equal(payload, decoded)) + }) + } + } +} diff --git a/pkg/turbine/internal/rsrecover/doc.go b/pkg/turbine/internal/rsrecover/doc.go new file mode 100644 index 000000000..38bc6e7f5 --- /dev/null +++ b/pkg/turbine/internal/rsrecover/doc.go @@ -0,0 +1,5 @@ +// Package rsrecover contains fixed-shape Reed-Solomon recovery plans for +// Solana's 32 data + 32 coding shred FEC sets. Production dispatch uses only +// exactly-one-missing-data recovery. Reduced-subset and all-coding plans are +// reference/benchmark alternatives and are not selected by SlotAssembler. +package rsrecover diff --git a/pkg/turbine/internal/rsrecover/recover.go b/pkg/turbine/internal/rsrecover/recover.go new file mode 100644 index 000000000..6a16e91e9 --- /dev/null +++ b/pkg/turbine/internal/rsrecover/recover.go @@ -0,0 +1,599 @@ +package rsrecover + +import ( + "errors" + "fmt" + "math/bits" + "sync" + + "github.com/klauspost/reedsolomon" +) + +const ( + DataShards = 32 + CodingShards = 32 + TotalShards = DataShards + CodingShards + + cauchyGamma = byte(0xa5) +) + +var ( + ErrInvalidPattern = errors.New("invalid erasure pattern") + ErrPatternChanged = errors.New("availability differs from prepared plan") + ErrInvalidBuffers = errors.New("invalid recovery buffers") + + gfLog [256]byte + gfExp [512]byte + + allCodingEncoderOnce sync.Once + allCodingEncoder reedsolomon.Encoder + allCodingEncoderErr error + + // oneDataCoefficientRows[missing][coding] contains one coefficient for + // each data input followed by the selected coding input. Every one of the + // fixed 32x32 rows is exhaustively differential-tested against the general + // Reed-Solomon decoder. + oneDataCoefficientRows [DataShards][CodingShards][DataShards + 1]byte +) + +func init() { + value := uint16(1) + for exponent := 0; exponent < 255; exponent++ { + gfExp[exponent] = byte(value) + gfLog[byte(value)] = byte(exponent) + value <<= 1 + if value&0x100 != 0 { + value ^= 0x11d + } + } + for exponent := 255; exponent < len(gfExp); exponent++ { + gfExp[exponent] = gfExp[exponent-255] + } + for missing := 0; missing < DataShards; missing++ { + for coding := 0; coding < CodingShards; coding++ { + // If c = sum(a_i*d_i), then the missing d_m is + // inv(a_m) * (c + sum(i != m, a_i*d_i)) over GF(2^8). + coefficientInv := reedsolomon.Inv(codingCoefficient(coding, missing)) + for data := 0; data < DataShards; data++ { + if data != missing { + oneDataCoefficientRows[missing][coding][data] = gfMul( + coefficientInv, + codingCoefficient(coding, data), + ) + } + } + oneDataCoefficientRows[missing][coding][DataShards] = coefficientInv + } + } +} + +// OneDataPlan recovers exactly one absent data shard from the other 31 data +// shards and one available coding shard. It avoids a general 32x32 decode +// matrix inversion. The plan is immutable and safe for concurrent execution +// when callers provide independent destinations. +type OneDataPlan struct { + presence uint64 + missing uint8 + sources [DataShards]uint8 + coefficients [DataShards]byte +} + +// DataSubsetPlan recovers every absent data shard using the present data and +// the lowest-indexed coding rows required to reach the 32-shard threshold. +// Setup solves only the m x m system induced by the missing data columns. +type DataSubsetPlan struct { + presence uint64 + missing []uint8 + sources [DataShards]uint8 + weights [][]byte +} + +// AllCodingPlan recovers all 32 data rows from all 32 coding rows. For the +// fixed Solana matrix C*C=I, so the package's optimized encoder can apply C a +// second time instead of constructing a decode matrix. +type AllCodingPlan struct { + presence uint64 + encoder reedsolomon.Encoder +} + +// Presence reports which of the 64 input shards are non-empty. A zero-length +// shard is absent, matching reedsolomon.ReconstructSome. +func Presence(shards [][]byte) (uint64, error) { + if len(shards) != TotalShards { + return 0, fmt.Errorf("%w: got %d shards, want %d", ErrInvalidBuffers, len(shards), TotalShards) + } + var mask uint64 + for index, shard := range shards { + if len(shard) != 0 { + mask |= uint64(1) << index + } + } + return mask, nil +} + +// PrepareRecoverOneData constructs the direct coefficient row for a pattern +// with exactly one missing data shard. Additional coding shards may be present; +// the lowest-indexed one is selected deterministically. +func PrepareRecoverOneData(presence uint64, missingDataIndex int) (OneDataPlan, error) { + if missingDataIndex < 0 || missingDataIndex >= DataShards { + return OneDataPlan{}, fmt.Errorf("%w: missing data index %d", ErrInvalidPattern, missingDataIndex) + } + for index := 0; index < DataShards; index++ { + present := presence&(uint64(1)<= DataShards { + return fmt.Errorf("%w: missing data index %d", ErrInvalidPattern, missingDataIndex) + } + const dataMask = uint64(1)<> DataShards) + if codingMask == 0 { + return fmt.Errorf("%w: no coding shard is available", ErrInvalidPattern) + } + shardSize, err := validateExecution(presence, shards, [][]byte{dst}) + if err != nil { + return err + } + if len(dst) != shardSize { + return fmt.Errorf("%w: destination has %d bytes, want %d", ErrInvalidBuffers, len(dst), shardSize) + } + + codingPosition := bits.TrailingZeros32(codingMask) + row := &oneDataCoefficientRows[missingDataIndex][codingPosition] + var lowLevel reedsolomon.LowLevel + first := true + for dataIndex := 0; dataIndex < DataShards; dataIndex++ { + coefficient := row[dataIndex] + if coefficient == 0 { + continue + } + if first { + lowLevel.GalMulSlice(coefficient, shards[dataIndex], dst) + first = false + } else { + lowLevel.GalMulSliceXor(coefficient, shards[dataIndex], dst) + } + } + lowLevel.GalMulSliceXor(row[DataShards], shards[DataShards+codingPosition], dst) + return nil +} + +// PrepareRecoverDataSubset constructs direct output rows for all absent data +// shards. It first inverts only the reduced m x m coding/data matrix, then +// expands those rows over exactly 32 selected input shards so byte execution +// can be compared fairly with a general decoder. +func PrepareRecoverDataSubset(presence uint64) (DataSubsetPlan, error) { + plan := DataSubsetPlan{presence: presence} + knownData := make([]uint8, 0, DataShards) + for index := 0; index < DataShards; index++ { + if presence&(uint64(1)<>uint((i%8)*8)) ^ byte(i*29+7) + } + digest := sha256.Sum256(append([]byte("mithril-repair-sim-leader-v1"), input[:]...)) + return solana.PrivateKey(ed25519.NewKeyFromSeed(digest[:])) +} + +func deterministicEntries(seed int64, slot uint64, count int) []turbine.Entry { + entries := make([]turbine.Entry, count) + for i := range entries { + material := fmt.Sprintf("mithril-repair-sim-entry-v1:%d:%d:%d", seed, slot, i) + h := sha256.Sum256([]byte(material)) + entries[i] = turbine.Entry{NumHashes: 1, Hash: solana.Hash(h)} + } + return entries +} + +// findEntryCount uses the generator itself as the capacity oracle. This avoids +// duplicating signed-last-FEC payload constants in the harness. +func findEntryCount(cfg LedgerConfig, leader solana.PrivateKey) (int, error) { + lo, hi := 1, cfg.FECsPerSlot*900 + for lo < hi { + mid := lo + (hi-lo)/2 + entries := deterministicEntries(cfg.Seed, cfg.StartSlot, mid) + slot, err := generateSlot(cfg, leader, cfg.StartSlot, cfg.StartSlot-1, entries) + if err != nil { + return 0, err + } + if len(slot.FECs) < cfg.FECsPerSlot { + lo = mid + 1 + } else { + hi = mid + } + } + entries := deterministicEntries(cfg.Seed, cfg.StartSlot, lo) + slot, err := generateSlot(cfg, leader, cfg.StartSlot, cfg.StartSlot-1, entries) + if err != nil { + return 0, err + } + if len(slot.FECs) != cfg.FECsPerSlot { + return 0, fmt.Errorf("cannot derive %d FEC sets within %d entries (got %d)", cfg.FECsPerSlot, hi, len(slot.FECs)) + } + return lo, nil +} + +func generateSlot(cfg LedgerConfig, leader solana.PrivateKey, number, parent uint64, entries []turbine.Entry) (Slot, error) { + component, err := turbine.NewEntryBatch(entries) + if err != nil { + return Slot{}, err + } + shredder := turbine.Shredder{ + Slot: number, + ParentSlot: parent, + Version: cfg.ShredVersion, + ReferenceTick: cfg.ReferenceTick, + } + batch, _, _, err := shredder.MakeMerkleShredsFromComponent( + leader, component, true, solana.Hash{}, 0, 0, + ) + if err != nil { + return Slot{}, err + } + + byFEC := make(map[uint32]*FECSet) + packetByKey := make(map[packetKey][]byte, len(batch.Packets)) + for _, raw := range batch.Packets { + shred, err := turbine.ParseShred(raw) + if err != nil { + return Slot{}, err + } + key := keyForShred(shred) + packetByKey[key] = append([]byte(nil), raw...) + } + data := make(map[uint32]Packet, len(batch.DataShreds)) + var highest uint32 + for _, shred := range append(append([]*turbine.Shred(nil), batch.DataShreds...), batch.CodeShreds...) { + fec := byFEC[shred.FECSetIndex] + if fec == nil { + fec = &FECSet{Index: shred.FECSetIndex} + byFEC[shred.FECSetIndex] = fec + } + packet := Packet{ + Bytes: packetByKey[keyForShred(shred)], + Slot: shred.Slot, + Type: shred.Type, + Index: shred.Index, + FECSetIndex: shred.FECSetIndex, + Position: shred.Position, + } + if shred.Type == turbine.ShredTypeData { + fec.Data = append(fec.Data, packet) + data[shred.Index] = packet + if shred.Index > highest { + highest = shred.Index + } + } else { + fec.Coding = append(fec.Coding, packet) + } + } + fecs := make([]FECSet, 0, len(byFEC)) + for index := uint32(0); len(fecs) < len(byFEC); index += dataShredsPerFEC { + fec := byFEC[index] + if fec == nil { + return Slot{}, fmt.Errorf("non-contiguous FEC sets: missing %d", index) + } + if len(fec.Data) != dataShredsPerFEC || len(fec.Coding) != codeShredsPerFEC { + return Slot{}, fmt.Errorf("FEC %d has %d+%d shreds", index, len(fec.Data), len(fec.Coding)) + } + fecs = append(fecs, *fec) + } + return Slot{Number: number, ParentSlot: parent, Entries: entries, FECs: fecs, Data: data, Highest: highest}, nil +} + +type packetKey struct { + type_ turbine.ShredType + index uint32 + fec uint32 + position uint16 +} + +func keyForShred(shred *turbine.Shred) packetKey { + return packetKey{type_: shred.Type, index: shred.Index, fec: shred.FECSetIndex, position: shred.Position} +} diff --git a/pkg/turbine/repairsim/sim.go b/pkg/turbine/repairsim/sim.go new file mode 100644 index 000000000..c89e867e1 --- /dev/null +++ b/pkg/turbine/repairsim/sim.go @@ -0,0 +1,713 @@ +package repairsim + +import ( + "container/heap" + "errors" + "fmt" + "math/rand" + "os" + "runtime" + "sort" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +type Scenario string + +const ( + ScenarioNearTip Scenario = "near-tip" + ScenarioDeepCatchup Scenario = "deep-catchup" +) + +type Availability string + +const ( + AvailabilityComplete Availability = "complete" + AvailabilityNearLoss Availability = "near-loss" + AvailabilitySparse Availability = "sparse" + AvailabilityMixed Availability = "mixed" +) + +// Config controls the deterministic network and local starting state. +type Config struct { + Scenario Scenario `json:"scenario"` + Availability Availability `json:"availability"` + RepairEnabled bool `json:"repair_enabled"` + RepairLatency time.Duration `json:"repair_latency_ns"` + RepairJitter time.Duration `json:"repair_jitter_ns"` + PacketLoss float64 `json:"packet_loss"` + DuplicateProbability float64 `json:"duplicate_probability"` + BandwidthBytesPerSec int64 `json:"bandwidth_bytes_per_sec"` + MaxConcurrent int `json:"max_concurrent_requests"` + MaxRequestSlots int `json:"max_request_slots"` + MaxMissingPerSlot int `json:"max_missing_per_slot"` + CorruptResponses int `json:"corrupt_responses"` + NaturalLateShreds bool `json:"natural_late_shreds"` + CollectTrace bool `json:"collect_trace"` + Seed int64 `json:"seed"` + SpoolDir string `json:"spool_dir,omitempty"` + SpoolMaxBytes int64 `json:"spool_max_bytes"` +} + +// DefaultConfig returns a deterministic starting point for a scenario. +func DefaultConfig(scenario Scenario) Config { + cfg := Config{ + Scenario: scenario, + RepairEnabled: true, + RepairLatency: 20 * time.Millisecond, + RepairJitter: 2 * time.Millisecond, + DuplicateProbability: 0.02, + BandwidthBytesPerSec: 100 * 1024 * 1024, + MaxConcurrent: 256, + MaxRequestSlots: 64, + MaxMissingPerSlot: 256, + NaturalLateShreds: scenario == ScenarioNearTip, + CollectTrace: true, + Seed: 1, + SpoolMaxBytes: 1 << 30, + } + if scenario == ScenarioDeepCatchup { + cfg.Availability = AvailabilityMixed + } else { + cfg.Availability = AvailabilityNearLoss + } + return cfg +} + +// TraceEvent is a deterministic logical-time event. CPU durations are kept in +// Result.StageCPU so trace equality does not depend on scheduler noise. +type TraceEvent struct { + Sequence int `json:"sequence"` + AtNanos int64 `json:"at_ns"` + Stage string `json:"stage"` + Slot uint64 `json:"slot,omitempty"` + FECSetIndex uint32 `json:"fec_set_index,omitempty"` + ShredIndex uint32 `json:"shred_index,omitempty"` + ShredType turbine.ShredType `json:"shred_type,omitempty"` + Bytes int `json:"bytes,omitempty"` + Detail string `json:"detail,omitempty"` +} + +type LatencySummary struct { + P50 time.Duration `json:"p50_ns"` + P95 time.Duration `json:"p95_ns"` + P99 time.Duration `json:"p99_ns"` +} + +// Result separates simulated-network time from actual local execution time. +type Result struct { + Scenario Scenario `json:"scenario"` + Availability Availability `json:"availability"` + Slots int `json:"slots"` + CompletedSlots int `json:"completed_slots"` + LogicalElapsed time.Duration `json:"logical_elapsed_ns"` + WallElapsed time.Duration `json:"wall_elapsed_ns"` + TimeToFirstReplayable time.Duration `json:"time_to_first_replayable_ns"` + TimeToFirstRecoveredData time.Duration `json:"time_to_first_recovered_data_ns"` + RepairEligibleToRecovery time.Duration `json:"repair_eligible_to_first_recovery_ns"` + CompletionLatency LatencySummary `json:"completion_latency"` + SlotsPerLogicalSecond float64 `json:"slots_per_logical_second"` + SlotsPerCPUSecond float64 `json:"slots_per_cpu_second"` + RepairRequests uint64 `json:"repair_requests"` + RepairResponses uint64 `json:"repair_responses"` + RepairBytesRequested uint64 `json:"repair_bytes_requested"` + RepairBytesReceived uint64 `json:"repair_bytes_received"` + UsefulNetworkDataShreds uint64 `json:"useful_network_data_shreds"` + LocallyRecoveredDataShreds uint64 `json:"locally_recovered_data_shreds"` + LocallyRecoveredDataBytes uint64 `json:"locally_recovered_data_bytes"` + InitialMissingDataShreds uint64 `json:"initial_missing_data_shreds"` + FractionRecoveredLocally float64 `json:"fraction_missing_recovered_locally"` + FECDecodes uint64 `json:"fec_decodes"` + DuplicateResponses uint64 `json:"duplicate_responses"` + CanceledOrLateResponses uint64 `json:"canceled_or_late_responses"` + LostResponses uint64 `json:"lost_responses"` + RejectedCorruptResponses uint64 `json:"rejected_corrupt_responses"` + ShredSignatureCacheHits uint64 `json:"shred_signature_cache_hits"` + ShredEd25519Verifications uint64 `json:"shred_ed25519_verifications"` + QueueHighWater int `json:"queue_high_water"` + SpoolBytes int64 `json:"spool_bytes"` + SpoolCompleteSlots int `json:"spool_complete_slots"` + DataShredBytesReplayable uint64 `json:"data_shred_bytes_replayable"` + Allocations uint64 `json:"allocations"` + StageCPU map[string]time.Duration `json:"stage_cpu_ns"` + Trace []TraceEvent `json:"trace"` + Limitations []string `json:"limitations"` +} + +// Run executes the virtual network around production parsing, validation, +// repair selection, reconstruction, spool insertion, and block completion. +func Run(ledger *Ledger, cfg Config) (Result, error) { + if ledger == nil || len(ledger.Slots) == 0 { + return Result{}, errors.New("empty ledger") + } + if cfg.Scenario != ScenarioNearTip && cfg.Scenario != ScenarioDeepCatchup { + return Result{}, fmt.Errorf("unsupported scenario %q", cfg.Scenario) + } + if cfg.Availability == "" { + cfg.Availability = DefaultConfig(cfg.Scenario).Availability + } + if cfg.MaxConcurrent <= 0 { + cfg.MaxConcurrent = 1 + } + if cfg.MaxRequestSlots <= 0 { + cfg.MaxRequestSlots = 64 + } + if cfg.MaxMissingPerSlot <= 0 { + cfg.MaxMissingPerSlot = 256 + } + if cfg.SpoolMaxBytes <= 0 { + cfg.SpoolMaxBytes = 1 << 30 + } + + spoolDir := cfg.SpoolDir + if spoolDir == "" { + var err error + spoolDir, err = os.MkdirTemp("", "mithril-repair-sim-") + if err != nil { + return Result{}, err + } + defer os.RemoveAll(spoolDir) + } + spool, err := turbine.OpenShredSpool(spoolDir, cfg.SpoolMaxBytes) + if err != nil { + return Result{}, err + } + defer spool.Close() + + var memBefore runtime.MemStats + runtime.ReadMemStats(&memBefore) + wallStarted := time.Now() + s := &simulation{ + ledger: ledger, + cfg: cfg, + assembler: turbine.NewSlotAssembler(), + spool: spool, + rng: rand.New(rand.NewSource(cfg.Seed)), + firstShred: make(map[uint64]time.Duration), + completed: make(map[uint64]*block.Block), + replayableAt: make(map[uint64]time.Duration), + pending: make(map[repairKey]struct{}), + stageCPU: make(map[string]time.Duration), + } + s.assembler.SetRetentionFloor(ledger.Slots[0].Number) + s.assembler.SetOnComplete(spool.MarkComplete) + s.nextReplaySlot = ledger.Slots[0].Number + + if err := s.seedLocalState(); err != nil { + return Result{}, err + } + if err := s.runDeliveries(); err != nil { + return Result{}, err + } + + var memAfter runtime.MemStats + runtime.ReadMemStats(&memAfter) + _, spoolBytes := spool.Stats() + result := s.result + result.Scenario = cfg.Scenario + result.Availability = cfg.Availability + result.Slots = len(ledger.Slots) + result.CompletedSlots = len(s.completed) + result.LogicalElapsed = s.now + result.WallElapsed = time.Since(wallStarted) + result.LocallyRecoveredDataShreds = s.assembler.RecoveredDataShreds() + result.InitialMissingDataShreds = initialMissingDataShreds(ledger, cfg.Availability) + if result.InitialMissingDataShreds > 0 { + result.FractionRecoveredLocally = float64(result.LocallyRecoveredDataShreds) / float64(result.InitialMissingDataShreds) + } + if len(ledger.Slots[0].FECs) > 0 && len(ledger.Slots[0].FECs[0].Data) > 0 { + result.LocallyRecoveredDataBytes = result.LocallyRecoveredDataShreds * uint64(len(ledger.Slots[0].FECs[0].Data[0].Bytes)) + } + if s.haveRecoverAt { + result.TimeToFirstRecoveredData = s.firstRecoverAt + if s.haveRepairAt && s.firstRecoverAt >= s.firstRepairAt { + result.RepairEligibleToRecovery = s.firstRecoverAt - s.firstRepairAt + } + } + result.SpoolBytes = spoolBytes + result.SpoolCompleteSlots = spool.CompleteSlots() + result.Allocations = memAfter.Mallocs - memBefore.Mallocs + result.StageCPU = s.stageCPU + result.Trace = s.trace + result.ShredSignatureCacheHits, result.ShredEd25519Verifications = s.shredVerifier.Stats() + result.Limitations = []string{ + "remote peers and latency are simulated in process; no UDP/IP stack is measured", + "synthetic entries contain no transactions, so transaction execution is not measured", + "slot offered to replay means SlotAssembler emitted a verified block; replay execution is not run", + "FEC decode start is observed at the AddShredFrom call boundary, not inside the Reed-Solomon library", + } + latencies := make([]time.Duration, 0, len(s.replayableAt)) + for slot, completedAt := range s.replayableAt { + if first, ok := s.firstShred[slot]; ok { + latencies = append(latencies, completedAt-first) + } + } + result.CompletionLatency = summarizeLatencies(latencies) + if len(s.replayableAt) > 0 { + first := ledger.Slots[0].Number + result.TimeToFirstReplayable = s.replayableAt[first] + } + if result.LogicalElapsed > 0 { + result.SlotsPerLogicalSecond = float64(result.CompletedSlots) / result.LogicalElapsed.Seconds() + } + if result.WallElapsed > 0 { + result.SlotsPerCPUSecond = float64(result.CompletedSlots) / result.WallElapsed.Seconds() + } + return result, nil +} + +type simulation struct { + ledger *Ledger + cfg Config + assembler *turbine.SlotAssembler + shredVerifier turbine.ShredSignatureVerifier + spool *turbine.ShredSpool + rng *rand.Rand + now time.Duration + nextWireAt time.Duration + sequence int + trace []TraceEvent + queue deliveryHeap + pending map[repairKey]struct{} + firstShred map[uint64]time.Duration + completed map[uint64]*block.Block + replayableAt map[uint64]time.Duration + nextReplaySlot uint64 + stageCPU map[string]time.Duration + result Result + corruptLeft int + firstRepairAt time.Duration + haveRepairAt bool + firstRecoverAt time.Duration + haveRecoverAt bool +} + +func (s *simulation) seedLocalState() error { + s.corruptLeft = s.cfg.CorruptResponses + packetsBySlot := make([][]Packet, len(s.ledger.Slots)) + for slotIdx := range s.ledger.Slots { + packetsBySlot[slotIdx] = initialPackets(&s.ledger.Slots[slotIdx], s.cfg.Availability) + } + for ordinal := 0; ; ordinal++ { + added := false + for slotIdx := range s.ledger.Slots { + packets := packetsBySlot[slotIdx] + if ordinal >= len(packets) { + continue + } + added = true + if err := s.ingest(packets[ordinal], false, false); err != nil { + return fmt.Errorf("seed slot %d: %w", s.ledger.Slots[slotIdx].Number, err) + } + s.now++ + } + if !added { + break + } + } + if s.cfg.NaturalLateShreds && s.cfg.Availability == AvailabilityNearLoss { + for i := range s.ledger.Slots { + slot := &s.ledger.Slots[i] + if len(slot.FECs) == 0 || len(slot.FECs[0].Data) < 31 { + continue + } + at := s.now + s.cfg.RepairLatency/2 + time.Duration(i)*time.Microsecond + heap.Push(&s.queue, delivery{at: at, sequence: s.sequence, packet: slot.FECs[0].Data[30]}) + s.sequence++ + } + } + return nil +} + +func initialPackets(slot *Slot, availability Availability) []Packet { + var out []Packet + for i := range slot.FECs { + fec := &slot.FECs[i] + switch availability { + case AvailabilityComplete: + out = append(out, fec.Data...) + case AvailabilityNearLoss: + if i%2 == 0 { + out = append(out, fec.Data[:30]...) + out = append(out, fec.Coding[0]) + } else { + out = append(out, fec.Data[:31]...) + } + case AvailabilitySparse: + out = append(out, fec.Data[:2]...) + case AvailabilityMixed: + out = append(out, fec.Data[:16]...) + out = append(out, fec.Coding[:15]...) + } + } + return out +} + +func (s *simulation) runDeliveries() error { + const maxIterations = 10_000_000 + for iterations := 0; len(s.completed) < len(s.ledger.Slots); iterations++ { + if iterations >= maxIterations { + return errors.New("repair simulation exceeded iteration limit") + } + if s.cfg.RepairEnabled { + s.prioritizeHeadWindow() + s.scheduleRequests() + } + if len(s.queue) == 0 { + // Without repair, exhaust the live arrivals and report any remaining + // holes. No more traffic is expected to complete those slots. + if !s.cfg.RepairEnabled { + return nil + } + return fmt.Errorf("repair stalled with %d/%d completed", len(s.completed), len(s.ledger.Slots)) + } + event := heap.Pop(&s.queue).(delivery) + if event.at > s.now { + s.now = event.at + } + if event.primary { + delete(s.pending, event.key) + } + if event.drop { + s.result.LostResponses++ + s.record("repair_response_lost", event.packet, "") + continue + } + if event.duplicate { + s.result.DuplicateResponses++ + } + if event.fromRepair { + s.result.RepairResponses++ + s.result.RepairBytesReceived += uint64(len(event.packet.Bytes)) + } + beforeUseful := s.assembler.UsefulRepairShreds() + if err := s.ingest(event.packet, event.fromRepair, event.corrupt); err != nil { + if event.corrupt && errors.Is(err, turbine.ErrInvalidSignature) { + s.result.RejectedCorruptResponses++ + s.record("repair_response_rejected", event.packet, "invalid signature or Merkle proof") + continue + } + return err + } + if event.fromRepair { + afterUseful := s.assembler.UsefulRepairShreds() + if afterUseful == beforeUseful { + s.result.CanceledOrLateResponses++ + } else { + s.result.UsefulNetworkDataShreds += afterUseful - beforeUseful + } + } + } + return nil +} + +func (s *simulation) prioritizeHeadWindow() { + var head uint64 + found := false + for i := range s.ledger.Slots { + slot := s.ledger.Slots[i].Number + if _, complete := s.completed[slot]; !complete { + head, found = slot, true + break + } + } + if !found { + return + } + end := head + 63 + last := s.ledger.Slots[len(s.ledger.Slots)-1].Number + if end > last { + end = last + } + s.assembler.PrioritizeRepairRange(head, end) +} + +func (s *simulation) scheduleRequests() { + capacity := s.cfg.MaxConcurrent - len(s.pending) + if capacity <= 0 { + return + } + requests := s.assembler.RepairRequests(s.cfg.MaxRequestSlots, s.cfg.MaxMissingPerSlot) + for _, req := range requests { + if !s.haveRepairAt { + s.firstRepairAt = s.now + s.haveRepairAt = true + } + s.recordAt("repair_needed_decision", req.Slot, 0, 0, 0, 0, + fmt.Sprintf("missing=%d need_highest=%t", len(req.MissingDataShreds), req.NeedHighestDataShred)) + slot, ok := s.ledger.Slot(req.Slot) + if !ok { + continue + } + indexes := append([]uint32(nil), req.MissingDataShreds...) + if req.NeedHighestDataShred { + indexes = append(indexes, slot.Highest) + } + seen := make(map[uint32]struct{}, len(indexes)) + for _, index := range indexes { + if capacity == 0 { + return + } + if _, duplicate := seen[index]; duplicate { + continue + } + seen[index] = struct{}{} + packet, ok := slot.Data[index] + if !ok { + continue + } + key := repairKey{slot: req.Slot, index: index} + if _, outstanding := s.pending[key]; outstanding { + continue + } + s.pending[key] = struct{}{} + capacity-- + s.result.RepairRequests++ + s.result.RepairBytesRequested += uint64(len(packet.Bytes)) + s.record("repair_request_enqueue", packet, "data shred") + s.record("repair_request_send", packet, "data shred") + s.scheduleResponse(key, packet) + } + } +} + +func (s *simulation) scheduleResponse(key repairKey, packet Packet) { + jitter := time.Duration(0) + if s.cfg.RepairJitter > 0 { + span := int64(s.cfg.RepairJitter)*2 + 1 + jitter = time.Duration(s.rng.Int63n(span)) - s.cfg.RepairJitter + } + at := s.now + s.cfg.RepairLatency + jitter + if at < s.now { + at = s.now + } + if at < s.nextWireAt { + at = s.nextWireAt + } + if s.cfg.BandwidthBytesPerSec > 0 { + wire := time.Duration(float64(len(packet.Bytes)) / float64(s.cfg.BandwidthBytesPerSec) * float64(time.Second)) + if wire < time.Nanosecond { + wire = time.Nanosecond + } + at += wire + s.nextWireAt = at + } + d := delivery{at: at, sequence: s.sequence, packet: packet, key: key, primary: true, fromRepair: true} + s.sequence++ + if s.cfg.PacketLoss > 0 && s.rng.Float64() < s.cfg.PacketLoss { + d.drop = true + } + if s.corruptLeft > 0 { + d.corrupt = true + s.corruptLeft-- + } + heap.Push(&s.queue, d) + if !d.drop && s.cfg.DuplicateProbability > 0 && s.rng.Float64() < s.cfg.DuplicateProbability { + dup := d + dup.at++ + dup.sequence = s.sequence + dup.primary = false + dup.duplicate = true + dup.corrupt = false + s.sequence++ + heap.Push(&s.queue, dup) + } + if len(s.queue) > s.result.QueueHighWater { + s.result.QueueHighWater = len(s.queue) + } +} + +func (s *simulation) ingest(packet Packet, fromRepair, corrupt bool) error { + raw := packet.Bytes + if corrupt { + raw = append([]byte(nil), raw...) + if len(raw) > 200 { + raw[200] ^= 0x80 + } else if len(raw) > 0 { + raw[len(raw)-1] ^= 0x80 + } + } + started := time.Now() + shred, err := turbine.ParseShred(raw) + s.stageCPU["shred_parse"] += time.Since(started) + if err != nil { + return err + } + started = time.Now() + err = s.shredVerifier.Verify(shred, s.ledger.LeaderPub) + s.stageCPU["shred_validation"] += time.Since(started) + if err != nil { + return err + } + s.record("shred_validation", packet, "Merkle proof and leader signature valid") + if _, ok := s.firstShred[shred.Slot]; !ok { + s.firstShred[shred.Slot] = s.now + s.record("first_shred", packet, "") + } + started = time.Now() + spooled := s.spool.AppendShred(shred, raw) + s.stageCPU["blockstore_insert"] += time.Since(started) + if spooled { + s.record("blockstore_insert", packet, "verified shred spool") + } + beforeRecovered := s.assembler.RecoveredDataShreds() + started = time.Now() + blk, err := s.assembler.AddShredFrom(shred, fromRepair) + s.stageCPU["assembler_ingest_and_recovery"] += time.Since(started) + if err != nil { + return err + } + afterRecovered := s.assembler.RecoveredDataShreds() + if afterRecovered > beforeRecovered { + s.result.FECDecodes++ + if !s.haveRecoverAt { + s.firstRecoverAt = s.now + s.haveRecoverAt = true + } + s.record("fec_threshold_reached", packet, "observed at AddShredFrom call boundary") + s.record("fec_decode_complete", packet, fmt.Sprintf("recovered_data=%d", afterRecovered-beforeRecovered)) + } + if fromRepair { + s.record("repair_response_receive", packet, "") + } else { + s.record("live_shred_receive", packet, "") + } + if blk != nil { + canonical, ok := s.ledger.Slot(blk.Slot) + if !ok { + return fmt.Errorf("completed unknown slot %d", blk.Slot) + } + if err := compareBlock(canonical, blk); err != nil { + return err + } + if _, ok := s.spool.IsComplete(blk.Slot); !ok { + return fmt.Errorf("slot %d completed without spool completion record", blk.Slot) + } + s.completed[blk.Slot] = blk + for _, data := range canonical.Data { + s.result.DataShredBytesReplayable += uint64(len(data.Bytes)) + } + s.record("slot_offered_to_replay", packet, "verified block emitted") + s.advanceReplayable() + } + return nil +} + +func (s *simulation) advanceReplayable() { + for { + if _, ok := s.completed[s.nextReplaySlot]; !ok { + return + } + s.replayableAt[s.nextReplaySlot] = s.now + s.recordAt("slot_replayable", s.nextReplaySlot, 0, 0, 0, 0, "contiguous parent chain available") + s.nextReplaySlot++ + } +} + +func compareBlock(canonical *Slot, got *block.Block) error { + if got.Slot != canonical.Number || got.SourceParentSlot != canonical.ParentSlot { + return fmt.Errorf("block identity got slot=%d parent=%d, want slot=%d parent=%d", got.Slot, got.SourceParentSlot, canonical.Number, canonical.ParentSlot) + } + if len(got.Entries) != len(canonical.Entries) { + return fmt.Errorf("slot %d entries=%d, want %d", got.Slot, len(got.Entries), len(canonical.Entries)) + } + for i := range canonical.Entries { + want := canonical.Entries[i] + entry := got.Entries[i] + if entry.NumHashes != want.NumHashes || string(entry.Hash) != string(want.Hash[:]) || len(entry.Indices) != len(want.Txns) { + return fmt.Errorf("slot %d entry %d differs from canonical ledger", got.Slot, i) + } + } + if !got.TransactionSignaturesVerified() { + return fmt.Errorf("slot %d block was not signature-verified", got.Slot) + } + return nil +} + +func (s *simulation) record(stage string, packet Packet, detail string) { + s.recordAt(stage, packet.Slot, packet.FECSetIndex, packet.Index, packet.Type, len(packet.Bytes), detail) +} + +func (s *simulation) recordAt(stage string, slot uint64, fec, index uint32, typ turbine.ShredType, bytes int, detail string) { + if s.cfg.CollectTrace { + s.trace = append(s.trace, TraceEvent{ + Sequence: s.sequence, AtNanos: int64(s.now), Stage: stage, Slot: slot, + FECSetIndex: fec, ShredIndex: index, ShredType: typ, Bytes: bytes, Detail: detail, + }) + } + s.sequence++ +} + +func summarizeLatencies(values []time.Duration) LatencySummary { + if len(values) == 0 { + return LatencySummary{} + } + sort.Slice(values, func(i, j int) bool { return values[i] < values[j] }) + percentile := func(p float64) time.Duration { + index := int(float64(len(values)-1)*p + 0.5) + return values[index] + } + return LatencySummary{P50: percentile(.50), P95: percentile(.95), P99: percentile(.99)} +} + +func initialMissingDataShreds(ledger *Ledger, availability Availability) uint64 { + var heldPerFEC int + switch availability { + case AvailabilityComplete: + heldPerFEC = 32 + case AvailabilityNearLoss: + var missing uint64 + for i := range ledger.Slots { + for fec := range ledger.Slots[i].FECs { + if fec%2 == 0 { + missing += 2 + } else { + missing++ + } + } + } + return missing + case AvailabilitySparse: + heldPerFEC = 2 + case AvailabilityMixed: + heldPerFEC = 16 + } + return uint64(len(ledger.Slots) * ledger.Config.FECsPerSlot * (dataShredsPerFEC - heldPerFEC)) +} + +type repairKey struct { + slot uint64 + index uint32 +} + +type delivery struct { + at time.Duration + sequence int + packet Packet + key repairKey + primary bool + fromRepair bool + drop bool + duplicate bool + corrupt bool +} + +type deliveryHeap []delivery + +func (h deliveryHeap) Len() int { return len(h) } +func (h deliveryHeap) Less(i, j int) bool { + if h[i].at != h[j].at { + return h[i].at < h[j].at + } + return h[i].sequence < h[j].sequence +} +func (h deliveryHeap) Swap(i, j int) { h[i], h[j] = h[j], h[i] } +func (h *deliveryHeap) Push(x any) { *h = append(*h, x.(delivery)) } +func (h *deliveryHeap) Pop() any { + old := *h + last := old[len(old)-1] + *h = old[:len(old)-1] + return last +} diff --git a/pkg/turbine/repairsim/sim_test.go b/pkg/turbine/repairsim/sim_test.go new file mode 100644 index 000000000..fc161768f --- /dev/null +++ b/pkg/turbine/repairsim/sim_test.go @@ -0,0 +1,254 @@ +package repairsim + +import ( + "reflect" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +func testLedger(t *testing.T, slots, fecs int) *Ledger { + t.Helper() + ledger, err := GenerateLedger(LedgerConfig{ + StartSlot: 20_000, + Slots: slots, + FECsPerSlot: fecs, + Seed: 7, + ShredVersion: 11, + ReferenceTick: 63, + }) + if err != nil { + t.Fatal(err) + } + return ledger +} + +func deterministicConfig(scenario Scenario) Config { + cfg := DefaultConfig(scenario) + cfg.RepairLatency = 10 * time.Millisecond + cfg.RepairJitter = time.Millisecond + cfg.DuplicateProbability = 0 + cfg.BandwidthBytesPerSec = 0 + cfg.MaxConcurrent = 32 + cfg.Seed = 19 + return cfg +} + +func TestGenerateLedgerHasExactFECCountAndValidPackets(t *testing.T) { + ledger := testLedger(t, 2, 3) + if ledger.Config.EntriesPerSlot == 0 { + t.Fatal("entry count was not resolved") + } + for _, slot := range ledger.Slots { + if len(slot.FECs) != 3 { + t.Fatalf("slot %d FECs=%d, want 3", slot.Number, len(slot.FECs)) + } + for _, fec := range slot.FECs { + if len(fec.Data) != 32 || len(fec.Coding) != 32 { + t.Fatalf("slot %d FEC %d shape=%d+%d", slot.Number, fec.Index, len(fec.Data), len(fec.Coding)) + } + for _, packet := range append(append([]Packet(nil), fec.Data...), fec.Coding...) { + shred, err := parseAndVerify(packet, ledger) + if err != nil { + t.Fatalf("slot %d FEC %d: %v", slot.Number, fec.Index, err) + } + if shred.Slot != slot.Number || shred.FECSetIndex != fec.Index { + t.Fatalf("packet routing got slot=%d FEC=%d", shred.Slot, shred.FECSetIndex) + } + } + } + } +} + +func TestNearTipRepairCompletesAndTraceIsDeterministic(t *testing.T) { + ledger := testLedger(t, 3, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.NaturalLateShreds = true + + first, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + second, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if first.CompletedSlots != 3 || first.SpoolCompleteSlots != 3 { + t.Fatalf("completed=%d spool=%d, want 3", first.CompletedSlots, first.SpoolCompleteSlots) + } + if first.LocallyRecoveredDataShreds == 0 || first.FECDecodes == 0 { + t.Fatalf("recovered=%d decodes=%d, want both nonzero", first.LocallyRecoveredDataShreds, first.FECDecodes) + } + if first.CanceledOrLateResponses == 0 { + t.Fatal("natural late-shred scenario did not produce canceled/late repair work") + } + if !reflect.DeepEqual(first.Trace, second.Trace) { + t.Fatal("same seed/config produced different logical traces") + } +} + +func TestNearTipWithoutRepairRemainsIncomplete(t *testing.T) { + ledger := testLedger(t, 2, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.RepairEnabled = false + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 0 || result.SpoolCompleteSlots != 0 { + t.Fatalf("completed=%d spool=%d without repair", result.CompletedSlots, result.SpoolCompleteSlots) + } +} + +func TestNaturalLateShredsWithoutRepair(t *testing.T) { + for _, tc := range []struct { + name string + fecs int + wantCompleted int + }{ + {name: "live-arrivals-complete-slots", fecs: 1, wantCompleted: 2}, + {name: "live-arrivals-leave-other-holes", fecs: 2, wantCompleted: 0}, + } { + t.Run(tc.name, func(t *testing.T) { + ledger := testLedger(t, 2, tc.fecs) + cfg := deterministicConfig(ScenarioNearTip) + cfg.RepairEnabled = false + cfg.NaturalLateShreds = true + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != tc.wantCompleted || result.SpoolCompleteSlots != tc.wantCompleted { + t.Fatalf("completed=%d spool=%d, want %d", result.CompletedSlots, result.SpoolCompleteSlots, tc.wantCompleted) + } + // Each live arrival provides the 32nd shard in its slot's first FEC, + // recovering one missing data shred without any repair response. + if result.LocallyRecoveredDataShreds != 2 { + t.Fatalf("recovered=%d, want 2 after both live arrivals", result.LocallyRecoveredDataShreds) + } + if result.RepairRequests != 0 || result.RepairResponses != 0 || result.RepairBytesRequested != 0 { + t.Fatalf("repair disabled: requests=%d responses=%d bytes requested=%d", result.RepairRequests, result.RepairResponses, result.RepairBytesRequested) + } + if result.LogicalElapsed < cfg.RepairLatency/2+time.Microsecond { + t.Fatalf("elapsed=%v, stopped before the last live arrival", result.LogicalElapsed) + } + }) + } +} + +func TestCompleteDeliveryEstablishesZeroRepairBaseline(t *testing.T) { + ledger := testLedger(t, 2, 2) + cfg := deterministicConfig(ScenarioNearTip) + cfg.Availability = AvailabilityComplete + cfg.RepairEnabled = false + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 2 { + t.Fatalf("completed=%d, want 2", result.CompletedSlots) + } + if result.RepairRequests != 0 || result.LocallyRecoveredDataShreds != 0 { + t.Fatalf("baseline requests=%d recovered=%d, want zero", result.RepairRequests, result.LocallyRecoveredDataShreds) + } + if result.ShredEd25519Verifications != 4 || result.ShredSignatureCacheHits != 124 { + t.Fatalf("signature cache verifies=%d hits=%d, want 4/124 for four FEC roots", + result.ShredEd25519Verifications, result.ShredSignatureCacheHits) + } +} + +func TestDeepCatchupMixedUsesThresholdRecovery(t *testing.T) { + ledger := testLedger(t, 8, 2) + cfg := deterministicConfig(ScenarioDeepCatchup) + cfg.Availability = AvailabilityMixed + cfg.NaturalLateShreds = false + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != len(ledger.Slots) { + t.Fatalf("completed=%d, want %d", result.CompletedSlots, len(ledger.Slots)) + } + if result.LocallyRecoveredDataShreds <= result.UsefulNetworkDataShreds { + t.Fatalf("local recovery=%d, network data=%d; mixed threshold scenario should recover most losses locally", result.LocallyRecoveredDataShreds, result.UsefulNetworkDataShreds) + } + if result.RepairRequests == 0 || result.RepairBytesReceived == 0 { + t.Fatal("deep catch-up completed without exercising repair") + } +} + +func TestCorruptRepairResponseRejectedThenRetried(t *testing.T) { + ledger := testLedger(t, 1, 1) + cfg := deterministicConfig(ScenarioDeepCatchup) + cfg.Availability = AvailabilityMixed + cfg.CorruptResponses = 1 + result, err := Run(ledger, cfg) + if err != nil { + t.Fatal(err) + } + if result.CompletedSlots != 1 { + t.Fatalf("completed=%d, want 1", result.CompletedSlots) + } + if result.RejectedCorruptResponses != 1 { + t.Fatalf("rejected corrupt=%d, want 1", result.RejectedCorruptResponses) + } + if result.RepairRequests < 2 { + t.Fatalf("requests=%d, want retry after corruption", result.RepairRequests) + } +} + +func parseAndVerify(packet Packet, ledger *Ledger) (*turbine.Shred, error) { + shred, err := turbine.ParseShred(packet.Bytes) + if err != nil { + return nil, err + } + if err := shred.VerifySignature(ledger.LeaderPub); err != nil { + return nil, err + } + return shred, nil +} + +func BenchmarkScenarios(b *testing.B) { + ledger, err := GenerateLedger(LedgerConfig{ + StartSlot: 30_000, Slots: 8, FECsPerSlot: 2, Seed: 23, ShredVersion: 1, ReferenceTick: 63, + }) + if err != nil { + b.Fatal(err) + } + tests := []struct { + name string + scenario Scenario + availability Availability + }{ + {name: "near-tip", scenario: ScenarioNearTip, availability: AvailabilityNearLoss}, + {name: "deep-mixed", scenario: ScenarioDeepCatchup, availability: AvailabilityMixed}, + {name: "deep-sparse", scenario: ScenarioDeepCatchup, availability: AvailabilitySparse}, + } + for _, tt := range tests { + b.Run(tt.name, func(b *testing.B) { + cfg := DefaultConfig(tt.scenario) + cfg.Availability = tt.availability + cfg.RepairLatency = 0 + cfg.RepairJitter = 0 + cfg.DuplicateProbability = 0 + cfg.BandwidthBytesPerSec = 0 + cfg.NaturalLateShreds = false + cfg.CollectTrace = false + b.ReportAllocs() + for i := 0; i < b.N; i++ { + result, err := Run(ledger, cfg) + if err != nil { + b.Fatal(err) + } + if result.CompletedSlots != len(ledger.Slots) { + b.Fatalf("completed=%d", result.CompletedSlots) + } + b.ReportMetric(float64(result.RepairRequests), "repair-requests/op") + b.ReportMetric(float64(result.LocallyRecoveredDataShreds), "recovered-shreds/op") + } + }) + } +} diff --git a/pkg/turbine/retention_sweep_test.go b/pkg/turbine/retention_sweep_test.go new file mode 100644 index 000000000..840640461 --- /dev/null +++ b/pkg/turbine/retention_sweep_test.go @@ -0,0 +1,137 @@ +package turbine + +import ( + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func sweepForTest(a *SlotAssembler) { + a.mu.Lock() + a.pruneOldSlotsLocked() + a.mu.Unlock() +} + +func TestRetentionSweepFloorMovesWithoutNewShreds(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + a.SetRetentionFloor(1000) + a.slots[1000] = &slotState{slot: 1000} + a.slots[1001] = &slotState{slot: 1001} + a.completedSlots[1001] = struct{}{} + a.SetKnownAlpenglowBlockID(1001, solana.Hash{1}) + a.PrioritizeRepairSlot(1000) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) + a.SetRetentionFloor(1001) + sweepForTest(a) + require.NotContains(t, a.slots, uint64(1000)) + require.NotContains(t, a.priorityRepairSlots, uint64(1000)) + require.Contains(t, a.slots, uint64(1001)) + a.SetRetentionFloor(0) + sweepForTest(a) + require.Empty(t, a.slots) + require.Empty(t, a.completedSlots) + require.Empty(t, a.knownBlockIDs) + + // Lowering the floor must permit new old repair state again. + a.SetRetentionFloor(1000) + a.slotState(1000, 1) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) +} + +func TestRetentionSweepReleasesCompletingProtectionAtFixedEdge(t *testing.T) { + for _, outcome := range []string{"abort", "cancel", "error", "complete", "reset"} { + t.Run(outcome, func(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + s := &slotState{slot: 1000, parentSlot: 999, completing: true, shreds: map[uint32]*Shred{0: {}}} + a.slots[s.slot] = s + a.SetKnownAlpenglowBlockID(999, solana.Hash{1}) + a.SetKnownAlpenglowBlockID(1000, solana.Hash{2}) + a.RejectAlpenglowBlockID(1000, solana.Hash{3}) + sweepForTest(a) + require.Contains(t, a.slots, uint64(1000)) + require.Contains(t, a.knownBlockIDs, uint64(999)) + require.Contains(t, a.rejectedBlockIDs, uint64(1000)) + work := &slotCompletionWork{state: s} + switch outcome { + case "abort": + a.abortCompletion(work) + case "cancel": + _, err := a.finalizeCompletion(work, processedSlotCompletion{canceled: true}) + require.NoError(t, err) + case "error": + _, err := a.finalizeCompletion(work, processedSlotCompletion{err: errors.New("decode failure")}) + require.Error(t, err) + case "complete": + _, err := a.finalizeCompletion(work, processedSlotCompletion{block: &block.Block{Slot: 1000}}) + require.NoError(t, err) + case "reset": + a.ResetSlot(1000) + } + sweepForTest(a) + require.Empty(t, a.slots) + require.Empty(t, a.completedSlots) + require.Empty(t, a.knownBlockIDs) + require.Empty(t, a.rejectedBlockIDs) + require.Empty(t, a.partialShredObs) + }) + } +} + +func TestRetentionSweepOldHintsAddedAtFixedEdge(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + sweepForTest(a) + a.SetKnownAlpenglowBlockID(1000, solana.Hash{1}) + a.RejectAlpenglowBlockID(1001, solana.Hash{2}) + sweepForTest(a) + require.Empty(t, a.knownBlockIDs) + require.Empty(t, a.rejectedBlockIDs) + a.mu.Lock() + a.trackBlockIDLocked(&block.Block{Slot: 1002, HasAlpenglowBlockID: true, AlpenglowBlockID: solana.Hash{3}}) + a.mu.Unlock() + sweepForTest(a) + require.Empty(t, a.knownBlockIDs) +} + +func TestRetentionSweepCapacityAtFixedEdge(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + a.SetRetentionFloor(1000) + a.PrioritizeRepairSlot(1001) + sweepForTest(a) + for i := 0; i < maxRetainedIncompleteSlotCap+2; i++ { + a.slotState(1000+uint64(i), 1) + } + sweepForTest(a) + require.Len(t, a.slots, maxRetainedIncompleteSlotCap) + require.Contains(t, a.slots, uint64(1000)) + require.Contains(t, a.slots, uint64(1001)) + require.NotContains(t, a.slots, uint64(1000+maxRetainedIncompleteSlotCap+1)) +} + +func BenchmarkRetentionRepeatedCompletedShred(b *testing.B) { + a := NewSlotAssembler() + a.maxObservedSlot = 10000 + for slot := uint64(9488); slot <= 10000; slot++ { + a.completedSlots[slot] = struct{}{} + a.knownBlockIDs[slot] = solana.Hash{1} + a.rejectedBlockIDs[slot] = map[solana.Hash]struct{}{{2}: {}} + a.partialShredObs[slot] = PartialShredObservation{DataShreds: 1} + } + sh := &Shred{Slot: 10000, Type: ShredTypeData} + // The public ingestion path still acquires the lock and performs its + // ordinary completed-slot rejection on every packet. + _, _ = a.AddShred(sh) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + _, _ = a.AddShred(sh) + } +} diff --git a/pkg/turbine/retransmit.go b/pkg/turbine/retransmit.go index d8584e6cf..dfb0a04a2 100644 --- a/pkg/turbine/retransmit.go +++ b/pkg/turbine/retransmit.go @@ -60,10 +60,17 @@ type RetransmitConfig struct { type retransmitWork struct { packet []byte - shred ShredID - leader solana.PublicKey + // storage is exclusively owned by this work until send (including retries) + // completes. Nil denotes the unpooled oversized compatibility path. + storage *retransmitPacket + shred ShredID + leader solana.PublicKey } +// Canonical shreds fit in one Solana packet. Keep the pool fixed-size rather +// than retaining arbitrary caller-provided capacities. +type retransmitPacket [packetDataSize]byte + type cachedRetransmitNodes struct { asof time.Time nodes *ClusterNodes @@ -91,6 +98,8 @@ type retransmitParentSigCache struct { } type packetBatchSender interface { + // Send borrows packet and peers only until it returns, including on errors + // or partial sends. Implementations must copy anything they retain. Send(packet []byte, peers []*net.UDPAddr) (int, error) Close() error } @@ -120,8 +129,13 @@ type Retransmitter struct { // immediately visible alongside short-send counters. sendBufferBytes int - queue chan retransmitWork - senders []packetBatchSender + queue chan retransmitWork + senders []packetBatchSender + packetPool sync.Pool + // Only guards queue admission versus shutdown; never held during crypto, + // peer selection or socket writes. A stopped queue cannot retain late work. + submitMu sync.RWMutex + stopped bool cacheMu sync.Mutex cache map[uint64]cachedRetransmitNodes @@ -283,20 +297,53 @@ func (r *Retransmitter) Run(ctx context.Context) { sender := sender go func() { defer workers.Done() + var peers [dataPlaneFanout]*net.UDPAddr for { select { case <-ctx.Done(): return case work := <-r.queue: - r.send(work, sender) + r.send(work, sender, peers[:0]) } } }() } <-ctx.Done() + r.submitMu.Lock() + r.stopped = true + r.submitMu.Unlock() workers.Wait() - for _, sender := range r.senders { - _ = sender.Close() + // Admission is closed and senders have returned their leases. Drain the + // remaining queue without closing a channel still visible to submitters. + for { + select { + case work := <-r.queue: + r.releasePacket(work.storage) + default: + for _, sender := range r.senders { + _ = sender.Close() + } + return + } + } +} + +func (r *Retransmitter) copyPacket(packet []byte) ([]byte, *retransmitPacket) { + if len(packet) > len(retransmitPacket{}) { + return append([]byte(nil), packet...), nil + } + storage, _ := r.packetPool.Get().(*retransmitPacket) + if storage == nil { + storage = new(retransmitPacket) + } + out := storage[:len(packet)] + copy(out, packet) + return out, storage +} + +func (r *Retransmitter) releasePacket(storage *retransmitPacket) { + if storage != nil { + r.packetPool.Put(storage) } } @@ -340,25 +387,37 @@ func (r *Retransmitter) SubmitFrom(packet []byte, shred *Shred, leader solana.Pu if packetSize == 0 || packetSize > len(packet) { return fmt.Errorf("turbine retransmit: invalid canonical packet size %d/%d", packetSize, len(packet)) } - out := append([]byte(nil), packet[:packetSize]...) + // Validate the signing range before borrowing storage, so all error exits + // either precede ownership or release it through send/the queue-drop path. + offset := 0 if root != nil { - offset, err := shred.retransmitterSignatureOffset() + offset, err = shred.retransmitterSignatureOffset() if err != nil { return fmt.Errorf("turbine retransmit: locate retransmitter signature: %w", err) } - if offset+ed25519.SignatureSize > len(out) { - return fmt.Errorf("turbine retransmit: retransmitter signature slice %d:%d exceeds packet size %d", offset, offset+ed25519.SignatureSize, len(out)) + if offset+ed25519.SignatureSize > packetSize { + return fmt.Errorf("turbine retransmit: retransmitter signature slice %d:%d exceeds packet size %d", offset, offset+ed25519.SignatureSize, packetSize) } + } + out, storage := r.copyPacket(packet[:packetSize]) + if root != nil { signature := ed25519.Sign(r.cfg.Identity, root[:]) copy(out[offset:offset+ed25519.SignatureSize], signature) r.resignedShreds.Add(1) } r.submitted.Add(1) + r.submitMu.RLock() + defer r.submitMu.RUnlock() + if r.stopped { + r.releasePacket(storage) + return nil + } select { - case r.queue <- retransmitWork{packet: out, shred: id, leader: leader}: + case r.queue <- retransmitWork{packet: out, storage: storage, shred: id, leader: leader}: default: r.queueDrops.Add(1) + r.releasePacket(storage) } return nil } @@ -417,9 +476,13 @@ func (r *Retransmitter) verifyParentSignature(shred *Shred, leader solana.Public return &root, nil } -func (r *Retransmitter) send(work retransmitWork, sender packetBatchSender) { +func (r *Retransmitter) send(work retransmitWork, sender packetBatchSender, scratch []*net.UDPAddr) { + defer r.releasePacket(work.storage) + // Clear even the unused tail after selection: worker scratch must not pin + // old snapshot addresses after a topology refresh or a no-peer/error path. + defer clear(scratch[:cap(scratch)]) nodes := r.clusterNodesForSlot(work.shred.Slot) - distance, peers, err := nodes.RetransmitPeers(work.leader, work.shred, dataPlaneFanout) + distance, peers, err := nodes.retransmitPeersInto(work.leader, work.shred, dataPlaneFanout, scratch) if errors.Is(err, ErrRetransmitLoopback) { r.loopbacks.Add(1) return diff --git a/pkg/turbine/retransmit_allocation_bench_test.go b/pkg/turbine/retransmit_allocation_bench_test.go new file mode 100644 index 000000000..3f376652a --- /dev/null +++ b/pkg/turbine/retransmit_allocation_bench_test.go @@ -0,0 +1,89 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/binary" + "fmt" + "net" + "runtime" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/gagliardetto/solana-go" +) + +type discardRelaySender struct{} + +func (*discardRelaySender) Send(_ []byte, peers []*net.UDPAddr) (int, error) { return len(peers), nil } +func (*discardRelaySender) Close() error { return nil } + +// Includes Submit's dedupe/copy, channel handoff, routing and sender dispatch. +// The sender performs no syscalls; this isolates relay CPU/allocations rather +// than claiming a network-throughput improvement. Authentication before Submit +// and resigned-shred crypto are covered by functional tests, not this fixture. +func BenchmarkRetransmitPipeline(b *testing.B) { + for _, count := range []int{90, 512} { + b.Run(fmt.Sprint(count), func(b *testing.B) { + nodes, leader, key := relayAllocationNodes(count, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 256}, []packetBatchSender{&discardRelaySender{}}) + if err != nil { + b.Fatal(err) + } + r.cache[10] = cachedRetransmitNodes{asof: time.Now().Add(time.Hour), nodes: nodes} + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { r.Run(ctx); close(done) }() + packet := make([]byte, dataPayloadSize) + shred := &Shred{Slot: 10, Type: ShredTypeData, Payload: packet} + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + // Exactly one producer; leave capacity before submitting so every + // iteration measures actual forwarding rather than dropped work. + for len(r.queue) == cap(r.queue) { + runtime.Gosched() + } + binary.LittleEndian.PutUint64(packet[:8], uint64(i)) + shred.Index = uint32(i) + if err := r.Submit(packet, shred, leader, false); err != nil { + b.Fatal(err) + } + } + for { + var n uint64 + for i := range r.rootDistance { + n += r.rootDistance[i].Load() + } + if n >= uint64(b.N) { + break + } + runtime.Gosched() + } + cancel() + <-done + b.StopTimer() + if r.queueDrops.Load() != 0 { + b.Fatal("unexpected queue drop") + } + }) + } +} + +func relayAllocationNodes(count int, chacha8 bool) (*ClusterNodes, solana.PublicKey, ed25519.PrivateKey) { + key := ed25519.NewKeyFromSeed(make([]byte, ed25519.SeedSize)) + var self solana.PublicKey + copy(self[:], key.Public().(ed25519.PublicKey)) + leader := solana.PublicKey{255} + peers := make([]gossip.TVUPeer, 0, count) + stakes := map[solana.PublicKey]uint64{self: 100, leader: 50} + for i := 0; i < count; i++ { + var pub solana.PublicKey + binary.LittleEndian.PutUint64(pub[:], uint64(i+1)) + addr := &net.UDPAddr{IP: net.IPv4(127, 1, byte(i/250), byte(i%250+1)), Port: 8001} + peers = append(peers, gossip.TVUPeer{Pubkey: gossip.Pubkey(pub), TVUAddr: addr}) + stakes[pub] = uint64(i % 101) + } + return NewRetransmitClusterNodes(ClusterNodesConfig{Self: self, TVUPeers: peers, Stakes: stakes, UseChaCha8: chacha8}), leader, key +} diff --git a/pkg/turbine/retransmit_allocation_test.go b/pkg/turbine/retransmit_allocation_test.go new file mode 100644 index 000000000..3cced3129 --- /dev/null +++ b/pkg/turbine/retransmit_allocation_test.go @@ -0,0 +1,267 @@ +package turbine + +import ( + "bytes" + "context" + "net" + "sync" + "syscall" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Derive expected children from the full weighted permutation, independently +// of the streaming child-selection loop and its caller-owned output storage. +func expectedRelayPeers(nodes *ClusterNodes, leader solana.PublicKey, id ShredID, fanout int) (uint8, []*net.UDPAddr) { + order := nodes.retransmitShuffle(leader, id) + self := -1 + for i, index := range order { + if nodes.nodes[index].pubkey == nodes.selfPubkey { + self = i + break + } + } + if self < 0 { + return maxTurbineHops - 1, nil + } + offset := 0 + if self > 0 { + offset = (self - 1) % fanout + } + step := fanout + if self == 0 { + step = 1 + } + var peers []*net.UDPAddr + for position, n := (self-offset)*fanout+offset+1, 0; position < len(order) && n < fanout; position, n = position+step, n+1 { + node := nodes.nodes[order[position]] + if node.hasContact { + if addr, ok := broadcastTVUUDP(node.tvuAddr); ok { + peers = append(peers, addr) + } + } + } + return turbineRootDistance(self, fanout), peers +} + +func TestRetransmitScratchMatchesPermutation(t *testing.T) { + for _, chacha8 := range []bool{false, true} { + for _, count := range []int{0, 31, 90, 512} { + nodes, leader, _ := relayAllocationNodes(count, chacha8) + // Exercise absent contacts and unroutable peers without changing stake order. + for i := range nodes.nodes { + if i%13 == 0 { + nodes.nodes[i].hasContact = false + } + if i%17 == 0 { + nodes.nodes[i].tvuAddr = &net.UDPAddr{IP: net.ParseIP("::1"), Port: 8001} + } + } + for _, fanout := range []int{1, 3, 200} { + var scratch [dataPlaneFanout]*net.UDPAddr + for index := uint32(0); index < 40; index++ { + id := ShredID{Slot: uint64(10 + index/4), Index: index, Type: ShredType(index % 2)} + wantDistance, want := expectedRelayPeers(nodes, leader, id, fanout) + distance, got, err := nodes.retransmitPeersInto(leader, id, fanout, scratch[:0]) + require.NoError(t, err) + require.Equal(t, wantDistance, distance) + require.Equal(t, len(want), len(got)) + for i := range want { + require.Same(t, want[i], got[i]) + } + _, owned, err := nodes.RetransmitPeers(leader, id, fanout) + require.NoError(t, err) + copyOfOwned := append([]*net.UDPAddr(nil), owned...) + clear(scratch[:]) + require.Equal(t, copyOfOwned, append([]*net.UDPAddr(nil), owned...)) + } + } + } + } +} + +func TestRetransmitScratchConcurrent(t *testing.T) { + nodes, leader, _ := relayAllocationNodes(300, true) + var wg sync.WaitGroup + for range 8 { + wg.Go(func() { + var scratch [dataPlaneFanout]*net.UDPAddr + for index := uint32(0); index < 50; index++ { + id := ShredID{Slot: 10, Index: index, Type: ShredTypeData} + wd, wp := expectedRelayPeers(nodes, leader, id, 3) + d, p, err := nodes.retransmitPeersInto(leader, id, 3, scratch[:0]) + if err != nil || d != wd || len(p) != len(wp) { + t.Error("concurrent routing mismatch") + return + } + for i := range p { + if p[i] != wp[i] { + t.Error("concurrent peer mismatch") + return + } + } + } + }) + } + wg.Wait() +} + +func TestRetransmitPacketPoolOwnership(t *testing.T) { + r := &Retransmitter{} + input := bytes.Repeat([]byte{7}, packetDataSize) + a, ownerA := r.copyPacket(input) + b, ownerB := r.copyPacket(input) + require.NotSame(t, ownerA, ownerB) + clear(input) + require.Equal(t, byte(7), a[0]) + require.Equal(t, byte(7), b[0]) + r.releasePacket(ownerA) + for range 50 { + p, owner := r.copyPacket(bytes.Repeat([]byte{9}, 1203)) + clear(p) + r.releasePacket(owner) + } + require.Equal(t, bytes.Repeat([]byte{7}, packetDataSize), b) + r.releasePacket(ownerB) + oversized := bytes.Repeat([]byte{4}, packetDataSize+1) + p, owner := r.copyPacket(oversized) + require.Nil(t, owner) + clear(oversized) + require.Equal(t, byte(4), p[0]) +} + +func TestRetransmitQueuedPacketOwnsInput(t *testing.T) { + nodes, leader, key := relayAllocationNodes(90, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 1}, []packetBatchSender{newCaptureBatchSender()}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + packet := bytes.Repeat([]byte{3}, dataPayloadSize) + packet[0] = 1 + shred := &Shred{Slot: 10, Index: 1, Type: ShredTypeData, Payload: packet} + require.NoError(t, r.Submit(packet, shred, leader, false)) + want := append([]byte(nil), packet...) + for i := 2; i < 20; i++ { + packet[0] = byte(i) + shred.Index = uint32(i) + require.NoError(t, r.Submit(packet, shred, leader, false)) + } + require.Equal(t, uint64(18), r.queueDrops.Load()) + clear(packet) + work := <-r.queue + require.Equal(t, want, work.packet) + // Cancellation leaves no owned copies queued after workers finish. + r.queue <- work + ctx, cancel := context.WithCancel(context.Background()) + cancel() + r.Run(ctx) + require.Empty(t, r.queue) + packet[0] = 99 + shred.Index = 99 + require.NoError(t, r.Submit(packet, shred, leader, false)) + require.Empty(t, r.queue, "late submit retained storage after workers stopped") +} + +type borrowedRelaySender struct { + t *testing.T + want []byte + entered chan struct{} + resume chan struct{} + calls int +} + +func (s *borrowedRelaySender) Send(packet []byte, peers []*net.UDPAddr) (int, error) { + s.calls++ + if s.calls == 1 { + close(s.entered) + <-s.resume + } + require.Equal(s.t, s.want, packet) + if s.calls == 1 { + return 0, syscall.EAGAIN + } + return len(peers), nil +} +func (*borrowedRelaySender) Close() error { return nil } + +func TestRetransmitPoolLeaseSurvivesSendRetries(t *testing.T) { + nodes, leader, key := relayAllocationNodes(31, true) + id := ShredID{Slot: 10, Type: ShredTypeData} + // Select a shred for which this validator is the root and has children. + for ; id.Index < 10000; id.Index++ { + d, p, err := nodes.RetransmitPeers(leader, id, 200) + require.NoError(t, err) + if d == 0 && len(p) > 1 { + break + } + } + require.Less(t, id.Index, uint32(10000)) + want := bytes.Repeat([]byte{7}, dataPayloadSize) + sender := &borrowedRelaySender{t: t, want: want, entered: make(chan struct{}), resume: make(chan struct{})} + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }}, []packetBatchSender{sender}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + packet, owner := r.copyPacket(want) + var scratch [dataPlaneFanout]*net.UDPAddr + done := make(chan struct{}) + go func() { + defer close(done) + r.send(retransmitWork{packet: packet, storage: owner, shred: id, leader: leader}, sender, scratch[:0]) + }() + select { + case <-sender.entered: + case <-time.After(5 * time.Second): + t.Fatal("send did not start") + } + for range 100 { + p, owned := r.copyPacket(want) + clear(p) + r.releasePacket(owned) + } + close(sender.resume) + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("send did not finish") + } + require.Equal(t, 2, sender.calls) + for _, addr := range scratch { + require.Nil(t, addr, "scratch pinned prior snapshot") + } +} + +func TestRetransmitConcurrentSubmitAndStop(t *testing.T) { + nodes, leader, key := relayAllocationNodes(90, true) + r, err := newRetransmitterWithSenders(RetransmitConfig{Identity: key, Peers: &mutableTVUPeers{}, Stakes: func(uint64) map[solana.PublicKey]uint64 { return nil }, QueueDepth: 8}, []packetBatchSender{&discardRelaySender{}, &discardRelaySender{}}) + require.NoError(t, err) + r.cache[10] = cachedRetransmitNodes{asof: time.Now(), nodes: nodes} + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { r.Run(ctx); close(done) }() + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Go(func() { + packet := make([]byte, dataPayloadSize) + for i := 0; i < 100; i++ { + packet[0], packet[1] = byte(worker), byte(i) + shred := &Shred{Slot: 10, Index: uint32(worker*100 + i), Type: ShredTypeData, Payload: packet} + if err := r.Submit(packet, shred, leader, false); err != nil { + t.Error(err) + } + if i == 50 { + cancel() + } + } + }) + } + wg.Wait() + cancel() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("shutdown blocked") + } + require.Empty(t, r.queue) +} diff --git a/pkg/turbine/shred.go b/pkg/turbine/shred.go index 4361f7cbd..23a4d002a 100644 --- a/pkg/turbine/shred.go +++ b/pkg/turbine/shred.go @@ -457,13 +457,12 @@ func merkleHashNode(left []byte, right []byte) solana.Hash { return hashv([][]byte{[]byte(merkleHashPrefixNode), left, right}) } -func hashv(parts [][]byte) solana.Hash { +func hashv(parts [][]byte) (out solana.Hash) { h := sha256.New() for _, part := range parts { _, _ = h.Write(part) } - var out solana.Hash - copy(out[:], h.Sum(nil)) + _ = h.Sum(out[:0]) return out } diff --git a/pkg/turbine/shredspool.go b/pkg/turbine/shredspool.go index 8f8003eb1..964a98ce3 100644 --- a/pkg/turbine/shredspool.go +++ b/pkg/turbine/shredspool.go @@ -32,25 +32,28 @@ import ( // re-fetch near the tip, the low end borders replay and is what repair would // otherwise pay for dearly. type ShredSpool struct { - mu sync.Mutex - dir string - open map[uint64]*spoolFile - sizes map[uint64]int64 // per-slot bytes on disk (open writers included) - seen map[uint64]map[spoolShredKey]struct{} // distinct shreds appended this run - validated map[uint64]bool // adopted files whose record tail was checked this run - complete map[uint64]SpoolSlotMeta - journal *os.File // append-only completeness journal (complete.idx) - bytes int64 - maxBytes int64 - highestSlot uint64 - haveHighest bool - floor uint64 - closed bool + mu sync.Mutex + dir string + open map[uint64]*spoolFile + sizes map[uint64]int64 // per-slot bytes on disk (open writers included) + seen map[uint64]map[spoolShredKey]struct{} // distinct shreds appended this run + validated map[uint64]bool // adopted files whose record tail was checked this run + complete map[uint64]SpoolSlotMeta + journal *spoolCompletionJournal // ordered completeness hints (complete.idx) + journalOverflow bool // Close must retry the current hints after queue overflow + bytes int64 + maxBytes int64 + highestSlot uint64 + haveHighest bool + floor uint64 + closed bool } // SpoolSlotMeta records a slot proven FULLY assembled: every data shred // 0..LastIndex was held when the assembler completed it. The completeness -// index is what turns the spool from a byte cache into the seed of a +// index is a repair hint, not proof that all buffered packets survived a crash. +// The assembler still validates coverage when hydrating a slot. This turns +// the spool from a byte cache into the seed of a // repair-serving shredstore: complete slots need zero network on restart, // answer HighestWindowIndex honestly, and define the serving/retention set. type SpoolSlotMeta struct { @@ -170,10 +173,10 @@ func (s *ShredSpool) loadJournal() { if err != nil { return // journal unavailable: completeness degrades to per-run only } + s.journal = newSpoolCompletionJournal(f) for slot, meta := range s.complete { - f.Write(spoolJournalRecord(slot, meta)) + s.journal.complete(slot, meta) } - s.journal = f } func spoolJournalRecord(slot uint64, meta SpoolSlotMeta) []byte { @@ -186,11 +189,13 @@ func spoolJournalRecord(slot uint64, meta SpoolSlotMeta) []byte { // MarkComplete records that the slot fully assembled (data shreds // 0..lastIndex all held). Called by the assembler's completion hook, so -// hydrating an adopted file re-marks it for free. Idempotent. +// hydrating an adopted file re-marks it for free. Idempotent. Journal submission +// never waits for storage or queue space; a crash may lose this repair hint, +// causing reassembly/repair, but cannot authorize a vote or advance a checkpoint. func (s *ShredSpool) MarkComplete(slot uint64, lastIndex uint32, shreds uint32) { s.mu.Lock() defer s.mu.Unlock() - if slot < s.floor || shreds == 0 { + if s.closed || slot < s.floor || shreds == 0 { return } if _, done := s.complete[slot]; done { @@ -198,8 +203,8 @@ func (s *ShredSpool) MarkComplete(slot uint64, lastIndex uint32, shreds uint32) } meta := SpoolSlotMeta{LastIndex: lastIndex, Shreds: shreds} s.complete[slot] = meta - if s.journal != nil { - s.journal.Write(spoolJournalRecord(slot, meta)) + if s.journal != nil && !s.journal.tryComplete(slot, meta) { + s.journalOverflow = true } } @@ -371,12 +376,17 @@ func (s *ShredSpool) ensureRoomLocked(slot uint64, additional int64) bool { if slot >= s.highestSlot { return false } - s.dropSlotLocked(s.highestSlot) + if !s.dropSlotLocked(s.highestSlot) { + return false + } } return true } -func (s *ShredSpool) dropSlotLocked(slot uint64) { +func (s *ShredSpool) dropSlotLocked(slot uint64) bool { + if err := s.invalidateCompleteLocked(slot); err != nil { + return false + } s.closeSlotLocked(slot) s.bytes -= s.sizes[slot] delete(s.sizes, slot) @@ -384,14 +394,20 @@ func (s *ShredSpool) dropSlotLocked(slot uint64) { delete(s.validated, slot) delete(s.complete, slot) _ = os.Remove(s.pathFor(slot)) - if s.journal != nil { - // Supersede any older completion record if this slot number is later - // re-created before the journal is compacted on restart. - s.journal.Write(spoolJournalRecord(slot, SpoolSlotMeta{})) - } if s.haveHighest && slot == s.highestSlot { s.recomputeHighestLocked() } + return true +} + +// Remove the live hint immediately, but do not mutate its file until older +// journal hints have been superseded. A failed fence leaves the file intact. +func (s *ShredSpool) invalidateCompleteLocked(slot uint64) error { + delete(s.complete, slot) + if s.journal != nil { + return s.journal.invalidate(slot) + } + return nil } func (s *ShredSpool) recomputeHighestLocked() { @@ -483,16 +499,15 @@ func (s *ShredSpool) readSlotLocked(slot uint64) ([][]byte, error) { validEnd = packetEnd } if validEnd != len(data) { + if err := s.invalidateCompleteLocked(slot); err != nil { + return nil, err + } if err := os.Truncate(path, int64(validEnd)); err != nil { return nil, fmt.Errorf("truncate corrupt shred spool tail for slot %d: %w", slot, err) } oldSize := s.sizes[slot] s.sizes[slot] = int64(validEnd) s.bytes += int64(validEnd) - oldSize - delete(s.complete, slot) - if s.journal != nil { - s.journal.Write(spoolJournalRecord(slot, SpoolSlotMeta{})) - } } s.validated[slot] = true return packets, nil @@ -533,12 +548,20 @@ func (s *ShredSpool) Stats() (slots int, bytes int64) { func (s *ShredSpool) Close() { s.mu.Lock() defer s.mu.Unlock() + if s.closed { + return + } s.closed = true for slot := range s.open { s.closeSlotLocked(slot) } if s.journal != nil { - _ = s.journal.Close() + if s.journalOverflow { + for slot, meta := range s.complete { + s.journal.complete(slot, meta) + } + } + s.journal.close() s.journal = nil } } diff --git a/pkg/turbine/shredspool_benchmark_test.go b/pkg/turbine/shredspool_benchmark_test.go new file mode 100644 index 000000000..d3e30cf6f --- /dev/null +++ b/pkg/turbine/shredspool_benchmark_test.go @@ -0,0 +1,37 @@ +package turbine + +import ( + "path/filepath" + "sort" + "strconv" + "testing" + "time" +) + +// Run this identical public-API benchmark on both revisions. Each sample owns +// a fresh spool, so no completed-slot dedupe or full-queue dropping is timed. +// Close is untimed but drains the writer before the next sample. These ordinary +// filesystem measurements do not simulate rare storage stalls or whole replay. +func BenchmarkShredSpoolMarkComplete(b *testing.B) { + root := b.TempDir() + elapsed := make([]int64, 0, b.N) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + b.StopTimer() + s, err := OpenShredSpool(filepath.Join(root, strconv.Itoa(i)), 0) + if err != nil { + b.Fatal(err) + } + s.Append(100, []byte("packet")) + b.StartTimer() + start := time.Now() + s.MarkComplete(100, 0, 1) + duration := time.Since(start).Nanoseconds() + b.StopTimer() + elapsed = append(elapsed, duration) + s.Close() + } + sort.Slice(elapsed, func(i, j int) bool { return elapsed[i] < elapsed[j] }) + b.ReportMetric(float64(elapsed[(len(elapsed)-1)/2]), "p50-ns") + b.ReportMetric(float64(elapsed[(99*len(elapsed)+99)/100-1]), "p99-ns") +} diff --git a/pkg/turbine/shredspool_journal.go b/pkg/turbine/shredspool_journal.go new file mode 100644 index 000000000..a537e202f --- /dev/null +++ b/pkg/turbine/shredspool_journal.go @@ -0,0 +1,102 @@ +package turbine + +import ( + "fmt" + "io" +) + +// Completion records are repair-cache hints, not voting or checkpoint state. +// A bounded writer removes their disk I/O from block delivery. Only completion +// hints may be dropped when the queue is full; live completeness stays in memory +// and Close retries the current hints before the next opener takes ownership. +const spoolJournalQueueSize = 256 + +type spoolJournalFile interface { + io.Writer + Truncate(int64) error + Close() error +} + +type spoolJournalRequest struct { + record [spoolJournalRecordSize]byte + done chan error // non-nil for an invalidation that must precede file mutation +} + +type spoolCompletionJournal struct { + requests chan spoolJournalRequest + done chan struct{} +} + +func newSpoolCompletionJournal(file spoolJournalFile) *spoolCompletionJournal { + j := &spoolCompletionJournal{requests: make(chan spoolJournalRequest, spoolJournalQueueSize), done: make(chan struct{})} + go j.run(file) + return j +} + +func journalRequest(slot uint64, meta SpoolSlotMeta) spoolJournalRequest { + var req spoolJournalRequest + copy(req.record[:], spoolJournalRecord(slot, meta)) + return req +} + +// The spool mutex serializes submissions and Close, but the worker never takes +// that mutex. A stalled write therefore cannot directly stall MarkComplete. +func (j *spoolCompletionJournal) tryComplete(slot uint64, meta SpoolSlotMeta) bool { + select { + case j.requests <- journalRequest(slot, meta): + return true + default: + return false + } +} + +func (j *spoolCompletionJournal) complete(slot uint64, meta SpoolSlotMeta) { + j.requests <- journalRequest(slot, meta) +} + +// Never drop or reorder invalidations. Wait for earlier completions and this +// tombstone before deleting/replacing/truncating a slot file. These rare paths +// may still wait for storage while holding the spool mutex. Queueing tombstones +// without this fence could resurrect an old completion after a crash. +func (j *spoolCompletionJournal) invalidate(slot uint64) error { + req := journalRequest(slot, SpoolSlotMeta{}) + req.done = make(chan error, 1) + j.requests <- req + return <-req.done +} + +func (j *spoolCompletionJournal) close() { + close(j.requests) + <-j.done +} + +func (j *spoolCompletionJournal) run(file spoolJournalFile) { + defer close(j.done) + defer file.Close() + failed, invalidated := false, false + for req := range j.requests { + var err error + if !failed { + var n int + n, err = file.Write(req.record[:]) + if err == nil && n != len(req.record) { + err = io.ErrShortWrite + } + failed = err != nil + } + if failed && !invalidated { + // Never append behind a short record, or acknowledge an invalidation + // while old completion hints remain. Empty the disposable journal and + // disable further hint writes for this opener. If even truncation fails, + // the caller must leave the slot file unchanged and retry later. + err = file.Truncate(0) + invalidated = err == nil + if err != nil { + err = fmt.Errorf("invalidate failed shred completeness journal: %w", err) + } + } + if req.done != nil { + req.done <- err + } + } +} diff --git a/pkg/turbine/shredspool_journal_test.go b/pkg/turbine/shredspool_journal_test.go new file mode 100644 index 000000000..1ba9169ce --- /dev/null +++ b/pkg/turbine/shredspool_journal_test.go @@ -0,0 +1,205 @@ +package turbine + +import ( + "bytes" + "errors" + "os" + "path/filepath" + "sync" + "testing" + "time" +) + +type gatedSpoolJournal struct { + *os.File + started chan struct{} + release chan struct{} + once sync.Once +} + +func (f *gatedSpoolJournal) Write(p []byte) (int, error) { + f.once.Do(func() { close(f.started); <-f.release }) + return f.File.Write(p) +} +func gateSpoolJournal(t *testing.T, s *ShredSpool) *gatedSpoolJournal { + t.Helper() + s.journal.close() + file, err := os.OpenFile(filepath.Join(s.dir, spoolJournalName), os.O_WRONLY|os.O_APPEND, 0) + if err != nil { + t.Fatal(err) + } + gate := &gatedSpoolJournal{File: file, started: make(chan struct{}), release: make(chan struct{})} + s.journal = newSpoolCompletionJournal(gate) + return gate +} +func waitSpoolTest(t *testing.T, done <-chan struct{}) { + t.Helper() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("spool operation did not complete") + } +} + +func TestShredSpoolCompletionDoesNotWaitForJournalAndCloseDrainsOverflow(t *testing.T) { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + // Seed real slot files before blocking only the completeness writer. + for slot := uint64(1); slot <= spoolJournalQueueSize+8; slot++ { + s.Append(slot, []byte("packet")) + } + gate := gateSpoolJournal(t, s) + var release sync.Once + t.Cleanup(func() { release.Do(func() { close(gate.release) }); s.Close() }) + s.MarkComplete(1, 0, 1) + waitSpoolTest(t, gate.started) + done := make(chan struct{}) + go func() { + for slot := uint64(2); slot <= spoolJournalQueueSize+8; slot++ { + s.MarkComplete(slot, 0, 1) + } + close(done) + }() + waitSpoolTest(t, done) + if !s.journalOverflow { + t.Fatal("test did not overflow the bounded queue") + } + if s.CompleteSlots() != spoolJournalQueueSize+8 { + t.Fatal("overflow lost live completeness") + } + closed := make(chan struct{}) + go func() { s.Close(); close(closed) }() + select { + case <-closed: + t.Fatal("Close returned before the blocked writer drained") + default: + } + release.Do(func() { close(gate.release) }) + waitSpoolTest(t, closed) + s.Close() // idempotent; no closed-channel send + s.MarkComplete(9999, 0, 1) + if _, ok := s.IsComplete(9999); ok { + t.Fatal("post-Close completion was accepted") + } + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + defer reopened.Close() + if reopened.CompleteSlots() != spoolJournalQueueSize+8 { + t.Fatal("clean handoff lost overflowed completion hints") + } +} + +func TestShredSpoolInvalidationWaitsBeforeReplacement(t *testing.T) { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.Append(800, []byte("old-file")) + if _, err := s.ReadSlot(800); err != nil { + t.Fatal(err) + } + gate := gateSpoolJournal(t, s) + var release sync.Once + t.Cleanup(func() { release.Do(func() { close(gate.release) }); s.Close() }) + s.MarkComplete(800, 0, 1) + waitSpoolTest(t, gate.started) + done := make(chan struct{}) + go func() { s.DiscardSlot(800); s.Append(800, []byte("replacement-partial")); close(done) }() + // While the earlier completion write is stalled, replacement must wait. + select { + case <-done: + t.Fatal("replacement passed an undrained invalidation") + case <-time.After(20 * time.Millisecond): + } + data, err := os.ReadFile(s.pathFor(800)) + if err != nil || !bytes.Contains(data, []byte("old-file")) { + t.Fatalf("old file mutated before invalidation: %v", err) + } + release.Do(func() { close(gate.release) }) + waitSpoolTest(t, done) + s.Close() + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + defer reopened.Close() + if _, ok := reopened.IsComplete(800); ok { + t.Fatal("older completion resurrected for replacement") + } + packets, err := reopened.ReadSlot(800) + if err != nil || len(packets) != 1 || string(packets[0]) != "replacement-partial" { + t.Fatalf("replacement: %q %v", packets, err) + } +} + +type faultySpoolJournal struct { + *os.File + failTruncate bool // only changed while worker is fenced by invalidate's reply +} + +func (f *faultySpoolJournal) Write(p []byte) (int, error) { return f.File.Write(p[:len(p)/2]) } +func (f *faultySpoolJournal) Truncate(n int64) error { + if f.failTruncate { + return errors.New("injected truncate failure") + } + return f.File.Truncate(n) +} + +func TestShredSpoolJournalShortWriteDisablesHintsAndFencesMutations(t *testing.T) { + for _, failTruncate := range []bool{false, true} { + dir := t.TempDir() + s, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.Append(100, []byte("old-file")) + s.MarkComplete(100, 0, 1) + s.Close() + s, err = OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + s.journal.close() + file, err := os.OpenFile(filepath.Join(dir, spoolJournalName), os.O_WRONLY|os.O_APPEND, 0) + if err != nil { + t.Fatal(err) + } + faulty := &faultySpoolJournal{File: file, failTruncate: failTruncate} + s.journal = newSpoolCompletionJournal(faulty) + s.mu.Lock() + dropped := s.dropSlotLocked(100) + s.mu.Unlock() + if dropped == failTruncate { + t.Fatalf("drop=%v with truncate failure=%v", dropped, failTruncate) + } + if failTruncate { + data, err := os.ReadFile(s.pathFor(100)) + if err != nil || !bytes.Contains(data, []byte("old-file")) { + t.Fatal("failed journal fence mutated slot file") + } + faulty.failTruncate = false + s.DiscardSlot(100) // retry can now invalidate all old hints + } + s.Append(100, []byte("partial")) + s.MarkComplete(101, 0, 1) + s.Close() + data, err := os.ReadFile(filepath.Join(dir, spoolJournalName)) + if err != nil || len(data) != 0 { + t.Fatalf("failed journal must stay empty, got %d bytes: %v", len(data), err) + } + reopened, err := OpenShredSpool(dir, 0) + if err != nil { + t.Fatal(err) + } + if _, ok := reopened.IsComplete(100); ok { + t.Fatal("failed journal resurrected completion") + } + reopened.Close() + } +} diff --git a/pkg/turbine/sigcache.go b/pkg/turbine/sigcache.go index 7f3b13139..990023d9b 100644 --- a/pkg/turbine/sigcache.go +++ b/pkg/turbine/sigcache.go @@ -20,7 +20,13 @@ import ( // hit reproduces exactly the result of re-running it on the same inputs. // Tampered content can never hit — different bytes yield a different root, // hence a different key. Failures are never cached. -type shredSigCache struct { +// ShredSignatureVerifier authenticates Merkle shreds with the same bounded, +// per-root result cache used by UDPReceiver. The cache never stores failures; +// each packet's Merkle proof is still evaluated before a cache lookup. +// +// It is exported so deterministic and loopback ingress harnesses can exercise +// production validation without constructing a UDPReceiver. +type ShredSignatureVerifier struct { mu sync.Mutex cur map[shredSigCacheKey]struct{} prev map[shredSigCacheKey]struct{} @@ -29,6 +35,10 @@ type shredSigCache struct { verifies atomic.Uint64 } +// Keep the internal receiver/test name as an alias; there is one +// implementation and one cache contract. +type shredSigCache = ShredSignatureVerifier + type shredSigCacheKey struct { leader solana.PublicKey root solana.Hash @@ -42,10 +52,17 @@ const shredSigCacheGenCap = 4096 // verifyShred authenticates a shred exactly like Shred.VerifySignature, with // the per-root ed25519 result cached. -func (c *shredSigCache) verifyShred(s *Shred, leader solana.PublicKey) error { +func (c *ShredSignatureVerifier) verifyShred(s *Shred, leader solana.PublicKey) error { + _, err := c.verifyShredRoot(s, leader) + return err +} + +// verifyShredRoot also returns the root authenticated for these exact bytes. +// Callers must keep the shred immutable through assembler admission. +func (c *ShredSignatureVerifier) verifyShredRoot(s *Shred, leader solana.PublicKey) (solana.Hash, error) { root, err := s.MerkleRoot() if err != nil { - return err + return solana.Hash{}, err } key := shredSigCacheKey{leader: leader, root: root, sig: s.Signature} @@ -53,28 +70,34 @@ func (c *shredSigCache) verifyShred(s *Shred, leader solana.PublicKey) error { if _, ok := c.cur[key]; ok { c.mu.Unlock() c.hits.Add(1) - return nil + return root, nil } if _, ok := c.prev[key]; ok { // Promote: a set straddling a rotation keeps its entry hot. c.addLocked(key) c.mu.Unlock() c.hits.Add(1) - return nil + return root, nil } c.mu.Unlock() c.verifies.Add(1) if !narya.VerifyStrict(leader[:], root[:], s.Signature[:]) { - return fmt.Errorf("%w: slot %d shred %d", ErrInvalidSignature, s.Slot, s.Index) + return solana.Hash{}, fmt.Errorf("%w: slot %d shred %d", ErrInvalidSignature, s.Slot, s.Index) } c.mu.Lock() c.addLocked(key) c.mu.Unlock() - return nil + return root, nil } -func (c *shredSigCache) addLocked(key shredSigCacheKey) { +// Verify authenticates one shred and retains successful root/signature tuples +// for sibling shreds in the same FEC set. +func (c *ShredSignatureVerifier) Verify(s *Shred, leader solana.PublicKey) error { + return c.verifyShred(s, leader) +} + +func (c *ShredSignatureVerifier) addLocked(key shredSigCacheKey) { if c.cur == nil { c.cur = make(map[shredSigCacheKey]struct{}, shredSigCacheGenCap) } @@ -85,6 +108,11 @@ func (c *shredSigCache) addLocked(key shredSigCacheKey) { c.cur[key] = struct{}{} } -func (c *shredSigCache) stats() (hits, verifies uint64) { +func (c *ShredSignatureVerifier) stats() (hits, verifies uint64) { return c.hits.Load(), c.verifies.Load() } + +// Stats reports cache hits and actual Ed25519 verifications. +func (c *ShredSignatureVerifier) Stats() (hits, verifies uint64) { + return c.stats() +} diff --git a/pkg/turbine/streaming_message_identity_test.go b/pkg/turbine/streaming_message_identity_test.go new file mode 100644 index 000000000..4fce6debd --- /dev/null +++ b/pkg/turbine/streaming_message_identity_test.go @@ -0,0 +1,104 @@ +package turbine + +import ( + "context" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestMessageIdentitiesPreparedBeforeFinalShred(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 100, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, 100, 0) + _, err := cached.verification.wait() + require.NoError(t, err) + require.Len(t, cached.verification.identities, 3) + for i := range cached.entries[0].Txns { + tx := &cached.entries[0].Txns[i] + got, ok := cached.verification.identities[i].ForTransaction(tx) + require.True(t, ok) + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, got) + } + blk := feedPrefetchShreds(t, a, batches[1]) + require.NotNil(t, blk) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } +} + +func TestVerifiedIdentityCacheRejectsMismatchedAndPartialCoverage(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + txs := verifierSignedTransactions(t, 3) + request, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + _, err = request.wait() + require.NoError(t, err) + blk := &block.Block{Transactions: txs} + require.NoError(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities)) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Error(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities[:2])) + copyTx := *txs[0] + for _, changed := range [][]*solana.Transaction{{txs[1], txs[0], txs[2]}, {©Tx, txs[1], txs[2]}} { + other := &block.Block{Transactions: changed} + require.Error(t, other.CacheVerifiedTransactionMessageIdentities(request.identities)) + } + again, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + require.Same(t, prepared, again, "failed cache adoption must not replace a valid cache") + txs[0].Message.RecentBlockhash[0] ^= 1 + require.Error(t, blk.CacheVerifiedTransactionMessageIdentities(request.identities)) +} + +func TestMessageIdentityFallbackPreservesRetainedTransactionOrder(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 100, + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4)), + prefetchTestPayload(t, verifierSignedTransactions(t, 5))) + for i := 0; i < 2; i++ { + require.Nil(t, feedPrefetchShreds(t, a, batches[i])) + cached := waitPrefetchedBatch(t, a, 100, batches[i][0].Index) + _, err := cached.verification.wait() + require.NoError(t, err) + if i == 1 { + // Force a canceled, joined result in the middle of the retained + // sequence. Completion must reverify and scatter its identities. + done := make(chan struct{}) + close(done) + cached.verification = &transactionVerification{done: done, err: context.Canceled, index: -1, cancel: func() {}} + } + } + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 12) + prepared, err := blk.PrepareTransactionMessageIdentities() + require.NoError(t, err) + for i, tx := range blk.Transactions { + want, err := txstatus.IdentityForTransaction(tx) + require.NoError(t, err) + require.Equal(t, want, prepared.Identity(i)) + } +} diff --git a/pkg/turbine/transaction_job_groups_test.go b/pkg/turbine/transaction_job_groups_test.go new file mode 100644 index 000000000..ab2edb554 --- /dev/null +++ b/pkg/turbine/transaction_job_groups_test.go @@ -0,0 +1,118 @@ +package turbine + +import ( + "context" + "fmt" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestWideVerificationJobsPreserveSignatureFailureIndex(t *testing.T) { + txs := verifierSignedTransactions(t, 320) + txs[33].Signatures[0][0] ^= 1 + txs[98].Signatures[0][0] ^= 1 + for _, groups := range []int{1, 4, 8} { + v := newTransactionVerifierWithJobGroups(2, 32, 8, groups, nil) + r, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + index, err := r.wait() + require.Error(t, err) + require.Equal(t, 33, index) + v.closeAndWait() + } +} + +func TestWideVerificationJobsYieldToReadySmallRequest(t *testing.T) { + for _, groups := range []int{4, 8} { + t.Run(fmt.Sprint(groups), func(t *testing.T) { + large, small := verifierTestBlock(800), verifierTestBlock(4) + started, release := make(chan struct{}), make(chan struct{}) + var releaseOnce sync.Once + var calls, beforeSmall atomic.Int32 + v := newTransactionVerifierWithJobGroups(1, 16, 8, groups, func(tx *solana.Transaction) error { + for _, s := range small.Transactions { + if tx == s { + beforeSmall.Store(calls.Load()) + return nil + } + } + calls.Add(1) + if tx == large.Transactions[0] { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + big, err := v.submitTransactions(context.Background(), large.Transactions) + require.NoError(t, err) + waitSignal(t, started, "large job") + little, err := v.submitTransactions(context.Background(), small.Transactions) + require.NoError(t, err) + require.Eventually(t, func() bool { return len(v.jobs) == 1 }, 3*time.Second, time.Millisecond) + releaseOnce.Do(func() { close(release) }) + _, err = little.wait() + require.NoError(t, err) + require.Equal(t, int32(groups*8), beforeSmall.Load(), "one large job, not an entire catch-up request, precedes small work") + _, err = big.wait() + require.NoError(t, err) + require.Equal(t, int32(800), calls.Load()) + }) + } +} + +func TestWideVerificationJobsCancelBetweenVectorsAndJoin(t *testing.T) { + for _, groups := range []int{4, 8} { + t.Run(fmt.Sprint(groups), func(t *testing.T) { + started, release := make(chan struct{}), make(chan struct{}) + var releaseOnce sync.Once + var calls atomic.Int32 + v := newTransactionVerifierWithJobGroups(1, 16, 8, groups, func(*solana.Transaction) error { + if calls.Add(1) == 1 { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + r, err := v.submitTransactions(ctx, verifierTestBlock(800).Transactions) + require.NoError(t, err) + waitSignal(t, started, "first vector") + cancel() + select { + case <-r.done: + t.Fatal("released transactions still read by a worker") + default: + } + releaseOnce.Do(func() { close(release) }) + _, err = r.wait() + require.ErrorIs(t, err, context.Canceled) + require.Equal(t, int32(8), calls.Load(), "canceled large job finishes its admitted vector only") + }) + } +} + +func TestWideVerificationJobsDoNotWaitForFourTransactionBatch(t *testing.T) { + for _, groups := range []int{4, 8} { + v := newTransactionVerifierWithJobGroups(2, 32, 8, groups, func(*solana.Transaction) error { return nil }) + r, err := v.submitTransactions(context.Background(), verifierTestBlock(4).Transactions) + require.NoError(t, err) + select { + case <-r.done: + case <-time.After(3 * time.Second): + t.Fatal("partial work waited for another submission") + } + _, err = r.wait() + require.NoError(t, err) + v.closeAndWait() + } +} diff --git a/pkg/turbine/transaction_verifier.go b/pkg/turbine/transaction_verifier.go index e64e36696..050e126b3 100644 --- a/pkg/turbine/transaction_verifier.go +++ b/pkg/turbine/transaction_verifier.go @@ -3,8 +3,8 @@ package turbine import ( "context" "fmt" - "runtime" "sync" + "time" "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/sigverify" @@ -12,105 +12,340 @@ import ( "github.com/gagliardetto/solana-go" ) -var errNilTransaction = fmt.Errorf("nil transaction") +var ( + errNilTransaction = fmt.Errorf("nil transaction") + errTransactionVerifierClosed = fmt.Errorf("transaction verifier closed") +) + +// Four vector groups amortize dispatch for large ready requests while bounding +// the work that can precede another component on a worker. +const defaultTransactionJobGroups = 4 +// A job is formed before admission, from transactions which are already +// available. Workers never wait for more transactions to fill a vector group. type transactionVerifyJob struct { - tx *solana.Transaction - err *error - done *sync.WaitGroup + trace bool + offeredAt, workerStart, workerEnd int64 + ctx context.Context + txs []*solana.Transaction + identities []txverify.VerifiedMessageIdentity + errs []error + start int + done chan<- *transactionVerifyJob } -type transactionVerifier struct { - jobs chan transactionVerifyJob - verify func(*solana.Transaction) error - workers int - // wave is how many transactions verifyBlockContext admits at once. It is - // workers * sigverify.BatchTarget so each worker can actually accumulate a - // full vector group rather than being handed one transaction at a time. - wave int - close sync.Once +// transactionVerification owns an asynchronous request until done closes. +// Transactions submitted to it must remain immutable until wait returns. +type transactionVerification struct { + trace *entryVerificationTrace + done chan struct{} + cancel context.CancelFunc + index int + err error + finishedAt time.Time + identities []txverify.VerifiedMessageIdentity +} - worker sync.WaitGroup +func (r *transactionVerification) wait() (int, error) { + return r.waitContext(context.Background()) } -func newTransactionVerifier(workers, queueDepth int, verify func(*solana.Transaction) error) *transactionVerifier { - if workers < 1 { - workers = 1 +// waitContext cancels further admission when ctx is canceled, but joins every +// admitted job before returning. The caller may then safely release or mutate +// the transaction objects, including their backing message byte slices. +func (r *transactionVerification) waitContext(ctx context.Context) (int, error) { + if ctx == nil { + ctx = context.Background() + } + select { + case <-r.done: + case <-ctx.Done(): + r.cancel() + <-r.done + return -1, ctx.Err() } - wave := workers * sigverify.BatchTarget - if queueDepth < 1 { - queueDepth = 1 + if err := ctx.Err(); err != nil { + return -1, err } + return r.index, r.err +} + +type transactionVerifier struct { + jobs chan *transactionVerifyJob + verify func(*solana.Transaction) error + workers int + batchTarget int + jobGroups int + // Each accepted request owns at most workers outstanding jobs. The + // admission semaphore also bounds asynchronous request goroutines; callers + // apply backpressure before handing off another decoded component. + requests chan struct{} + // Protected by mu. Reserve one request permit for completion/recovery. + prefetchRequests int + completionWaiters int + admissionChanged chan struct{} + request sync.WaitGroup + mu sync.Mutex + closed bool + stopped chan struct{} + close sync.Once + worker sync.WaitGroup +} + +func newTransactionVerifier(workers, queueDepth int, verify func(*solana.Transaction) error) *transactionVerifier { + return newTransactionVerifierWithBatchTarget(workers, queueDepth, sigverify.BatchTarget, verify) +} + +// queueDepth is a transaction budget, rounded up to whole jobs. batchTarget +// counts signature lanes: multi-signature transactions stay indivisible and +// may exceed the target. Four/eight targets can be compared without changing +// the admission or cancellation policy. +func newTransactionVerifierWithBatchTarget(workers, queueDepth, batchTarget int, verify func(*solana.Transaction) error) *transactionVerifier { + return newTransactionVerifierWithJobGroups(workers, queueDepth, batchTarget, defaultTransactionJobGroups, verify) +} + +// Job groups amortize dispatch over already available vector groups. They do +// not change vector width or wait for future transactions to arrive. +func newTransactionVerifierWithJobGroups(workers, queueDepth, batchTarget, jobGroups int, verify func(*solana.Transaction) error) *transactionVerifier { + workers = max(1, workers) + batchTarget = max(1, min(batchTarget, sigverify.BatchTarget)) + jobGroups = max(1, min(jobGroups, 8)) + jobCapacity := batchTarget * jobGroups + queueGroups := max(1, (queueDepth+jobCapacity-1)/jobCapacity) v := &transactionVerifier{ - jobs: make(chan transactionVerifyJob, queueDepth), - verify: verify, - workers: workers, - wave: wave, + jobs: make(chan *transactionVerifyJob, queueGroups), + verify: verify, + workers: workers, + batchTarget: batchTarget, + jobGroups: jobGroups, + requests: make(chan struct{}, 2*workers), + stopped: make(chan struct{}), } v.worker.Add(workers) for i := 0; i < workers; i++ { go func() { defer v.worker.Done() - // Worker-local scratch reused across groups. - var ( - group []transactionVerifyJob - scr verifyScratch - ) + var batch txverify.BatchVerifier for job := range v.jobs { - group = sigverify.Drain(group, job, v.jobs, - sigverify.FairShare(len(v.jobs), v.workers, sigverify.BatchTarget)) - v.verifyGroup(group, &scr) - // Do not keep finished jobs reachable through the scratch. - clear(group) + v.verifyGroup(job, &batch) } }() } return v } -// verifyScratch is one worker's reusable buffers. -type verifyScratch struct { - txs []*solana.Transaction - errs []error - batch txverify.BatchVerifier -} - -// verifyGroup verifies a drained group and releases every job in it. -// -// Releasing happens in a defer covering the whole group, so no caller can be -// left waiting on a job that was drained into a batch which then failed — -// a stranded job would hang verifyBlockContext's done.Wait() forever. -func (v *transactionVerifier) verifyGroup(group []transactionVerifyJob, scr *verifyScratch) { +// verifyGroup releases its job even if signature verification panics. The +// request's bounded completion channel always has room for every pending job. +func (v *transactionVerifier) verifyGroup(job *transactionVerifyJob, batch *txverify.BatchVerifier) { + if job.trace { + job.workerStart = entryTraceNow() + } defer func() { - for _, job := range group { - job.done.Done() + if job.trace { + job.workerEnd = entryTraceNow() } + job.done <- job }() - - // An injected verifier is a per-transaction function and stays that way; - // only the default path can batch. This seam is used by tests. - if v.verify != nil { - for _, job := range group { - *job.err = verifyTransactionSafely(v.verify, job.tx) + for start := 0; start < len(job.txs); { + // An admitted job always finishes its first vector group, preserving + // ownership/join semantics. Cancellation can skip additional groups. + if err := job.ctx.Err(); start > 0 && err != nil { + for i := start; i < len(job.errs); i++ { + job.errs[i] = err + } + return } - return + end := transactionVerifyGroupEnd(job.txs, start, v.batchTarget) + if v.verify != nil { + for i := start; i < end; i++ { + tx := job.txs[i] + if tx == nil { + job.errs[i] = errNilTransaction + } else { + job.errs[i] = verifyTransactionSafely(v.verify, tx) + } + } + } else { + if job.identities != nil { + verifyBatchWithIdentitiesSafely(batch, job.txs[start:end], job.errs[start:end], job.identities[start:end]) + } else { + verifyBatchSafely(batch, job.txs[start:end], job.errs[start:end]) + } + } + start = end } +} + +// submitTransactions admits one immutable decoded component or complete block. +// Admission is bounded and may block: call it from a decode/completion worker, +// never the UDP reader or while holding the assembler mutex. Cancellation of +// ctx stops further groups but still joins every admitted group. +func (v *transactionVerifier) submitTransactions(ctx context.Context, txs []*solana.Transaction) (*transactionVerification, error) { + return v.submitRequest(ctx, txs, false) +} + +// submitPrefetchTransactions applies backpressure before allocating a request: +// prefetch may use at most 2*workers-1 of the existing 2*workers permits. +func (v *transactionVerifier) submitPrefetchTransactions(ctx context.Context, txs []*solana.Transaction) (*transactionVerification, error) { + return v.submitRequest(ctx, txs, true) +} + +func (v *transactionVerifier) submitRequest(ctx context.Context, txs []*solana.Transaction, prefetch bool) (*transactionVerification, error) { + if ctx == nil { + ctx = context.Background() + } + if err := ctx.Err(); err != nil { + return nil, err + } + var trace *entryVerificationTrace + if entryTraceContext(ctx) { + trace = &entryVerificationTrace{Submit: entryTraceNow(), Transactions: len(txs)} + } + if err := v.acquireRequest(ctx, prefetch); err != nil { + return nil, err + } + ctx, cancel := context.WithCancel(ctx) + if trace != nil { + trace.Admitted = entryTraceNow() + } + r := &transactionVerification{done: make(chan struct{}), cancel: cancel, index: -1, trace: trace} + if v.verify == nil { + r.identities = make([]txverify.VerifiedMessageIdentity, len(txs)) + } + go func() { + defer v.request.Done() + defer v.releaseRequest(prefetch) + defer cancel() + r.index, r.err = v.verifyTransactionsWithTiming(ctx, txs, r.identities, trace) + if trace != nil { + trace.Finished = entryTraceNow() + } + r.finishedAt = time.Now() + close(r.done) + }() + return r, nil +} + +// verifyTransactions keeps a rolling window instead of waiting for an entire +// worker wave. A slow job cannot idle workers whose earlier jobs finished. +// One caller can queue at most workers jobs, so a large catch-up block cannot +// put all its transactions ahead of a newly available component. +func (v *transactionVerifier) verifyTransactions(ctx context.Context, txs []*solana.Transaction) (int, error) { + return v.verifyTransactionsWithIdentities(ctx, txs, nil) +} - scr.txs = scr.txs[:0] - for _, job := range group { - scr.txs = append(scr.txs, job.tx) +func (v *transactionVerifier) verifyTransactionsWithIdentities(ctx context.Context, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) (int, error) { + return v.verifyTransactionsWithTiming(ctx, txs, identities, nil) +} + +func (v *transactionVerifier) verifyTransactionsWithTiming(ctx context.Context, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity, trace *entryVerificationTrace) (int, error) { + if len(txs) == 0 { + return -1, ctx.Err() + } + window := min(v.workers, len(txs)) + jobGroups := v.jobGroups + // Keep short components responsive and enough independent jobs to supply + // every worker. This is a ready-work threshold, never a batching timer. + if len(txs) < 2*v.workers*v.batchTarget*jobGroups { + jobGroups = 1 } - if cap(scr.errs) < len(scr.txs) { - scr.errs = make([]error, len(scr.txs)) + jobCapacity := v.batchTarget * jobGroups + completed := make(chan *transactionVerifyJob, window) + groups := make([]transactionVerifyJob, window) + errs := make([]error, window*jobCapacity) + free := make([]*transactionVerifyJob, window) + for i := range groups { + groups[i].trace = trace != nil + groups[i].ctx = ctx + groups[i].errs = errs[i*jobCapacity : (i+1)*jobCapacity] + groups[i].done = completed + free[i] = &groups[i] } - scr.errs = scr.errs[:len(scr.txs)] - verifyBatchSafely(&scr.batch, scr.txs, scr.errs) + nextIndex, active := 0, 0 + failureIndex := -1 + var failure error + var pending *transactionVerifyJob + ctxDone := ctx.Done() + stopped := false + for active > 0 || (!stopped && nextIndex < len(txs)) { + if !stopped && ctx.Err() != nil { + stopped = true + ctxDone = nil + } + if stopped && active == 0 { + break + } + if !stopped && pending == nil && nextIndex < len(txs) && len(free) > 0 { + pending = free[len(free)-1] + free = free[:len(free)-1] + end := nextIndex + for group := 0; group < jobGroups && end < len(txs); group++ { + end = transactionVerifyGroupEnd(txs, end, v.batchTarget) + } + if trace != nil { + pending.offeredAt = entryTraceNow() + } + pending.start = nextIndex + pending.txs = txs[nextIndex:end] + if identities != nil { + pending.identities = identities[nextIndex:end] + } + pending.errs = pending.errs[:end-nextIndex] + clear(pending.errs) + } + var admission chan *transactionVerifyJob + if !stopped && pending != nil { + admission = v.jobs + } + select { + case admission <- pending: + nextIndex += len(pending.txs) + active++ + pending = nil + case job := <-completed: + if trace != nil { + trace.observe(job) + } + active-- + for i, err := range job.errs { + if err != nil && (failureIndex < 0 || job.start+i < failureIndex) { + failureIndex, failure = job.start+i, err + stopped = true + } + } + job.txs = nil + job.identities = nil + clear(job.errs) + free = append(free, job) + case <-ctxDone: + stopped = true + ctxDone = nil + } + } + if err := ctx.Err(); err != nil { + return -1, err + } + return failureIndex, failure +} - for i, job := range group { - *job.err = scr.errs[i] +func transactionVerifyGroupEnd(txs []*solana.Transaction, start, target int) int { + end, signatures := start, 0 + for end < len(txs) && end-start < target { + count := 1 + if txs[end] != nil { + count = max(1, len(txs[end].Signatures)) + } + if end > start && signatures+count > target { + break + } + signatures += count + end++ + if signatures >= target { + break + } } - clear(scr.txs) + return end } // verifyBatchSafely mirrors verifyTransactionSafely: a panic in the verifier @@ -129,11 +364,28 @@ func verifyBatchSafely(batch *txverify.BatchVerifier, txs []*solana.Transaction, batch.Verify(txs, errs) } +func verifyBatchWithIdentitiesSafely(batch *txverify.BatchVerifier, txs []*solana.Transaction, errs []error, identities []txverify.VerifiedMessageIdentity) { + defer func() { + if recovered := recover(); recovered != nil { + clear(identities) + for i := range errs { + errs[i] = fmt.Errorf("signature verifier panic: %v", recovered) + } + } + }() + batch.VerifyWithMessageIdentities(txs, errs, identities) +} + func (v *transactionVerifier) closeAndWait() { if v == nil { return } v.close.Do(func() { + v.mu.Lock() + v.closed = true + close(v.stopped) + v.mu.Unlock() + v.request.Wait() close(v.jobs) v.worker.Wait() }) @@ -162,47 +414,18 @@ func (v *transactionVerifier) verifyBlockContext(ctx context.Context, blk *block if blk == nil || len(blk.Transactions) == 0 { return nil } - // Admit one worker-wave at a time. A monster block still occupies every - // verifier lane, but cannot park tens of thousands of jobs ahead of a newly - // completed small block in the shared bounded queue. - for chunkStart := 0; chunkStart < len(blk.Transactions); chunkStart += v.wave { - if err := ctx.Err(); err != nil { - return err - } - chunkEnd := min(chunkStart+v.wave, len(blk.Transactions)) - errs := make([]error, chunkEnd-chunkStart) - var done sync.WaitGroup - for txIdx := chunkStart; txIdx < chunkEnd; txIdx++ { - if err := ctx.Err(); err != nil { - done.Wait() - return err - } - tx := blk.Transactions[txIdx] - errIdx := txIdx - chunkStart - if tx == nil { - errs[errIdx] = errNilTransaction - continue - } - done.Add(1) - select { - case v.jobs <- transactionVerifyJob{tx: tx, err: &errs[errIdx], done: &done}: - case <-ctx.Done(): - done.Done() - done.Wait() - return ctx.Err() - } - } - done.Wait() - if err := ctx.Err(); err != nil { - return err - } - for errIdx, err := range errs { - if err != nil { - return formatTransactionVerificationError(blk, chunkStart+errIdx, err) - } - } + request, err := v.submitTransactions(ctx, blk.Transactions) + if err != nil { + return err + } + index, err := request.wait() + if err != nil && index >= 0 { + return formatTransactionVerificationError(blk, index, err) } - return nil + if err == nil && request.identities != nil { + return blk.CacheVerifiedTransactionMessageIdentities(request.identities) + } + return err } func formatTransactionVerificationError(blk *block.Block, txIdx int, err error) error { @@ -226,12 +449,17 @@ var ( defaultTransactionVerifier *transactionVerifier ) -func validateBlockTransactionsContext(ctx context.Context, blk *block.Block) error { +func getDefaultTransactionVerifier() *transactionVerifier { defaultTransactionVerifierOnce.Do(func() { - workers := max(1, (runtime.GOMAXPROCS(0)+1)/2) - defaultTransactionVerifier = newTransactionVerifier(workers, 2*workers*sigverify.BatchTarget, nil) + workers := sigverify.TransactionWorkers() + target := sigverify.TransactionBatchTarget() + defaultTransactionVerifier = newTransactionVerifierWithBatchTarget(workers, 2*workers*target, target, nil) }) - return defaultTransactionVerifier.verifyBlockContext(ctx, blk) + return defaultTransactionVerifier +} + +func validateBlockTransactionsContext(ctx context.Context, blk *block.Block) error { + return getDefaultTransactionVerifier().verifyBlockContext(ctx, blk) } func validateBlockTransactions(blk *block.Block) error { diff --git a/pkg/turbine/transaction_verifier_admission.go b/pkg/turbine/transaction_verifier_admission.go new file mode 100644 index 000000000..c0a20ed2c --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission.go @@ -0,0 +1,76 @@ +package turbine + +import "context" + +// acquireRequest keeps the total request/job bounds unchanged while reserving +// one permit for completion or full-block recovery. Waiting completions win the +// next available permit over prefetch; completions are otherwise equal priority. +// This is not replay-head scheduling: unfinished prefetch already admitted for +// the head keeps its existing rolling job window, and future-slot completions +// also use the reservation. No worker is reserved and no admitted job is evicted. +// +// Prefetch can wait while completions remain queued. It is speculative work and +// resumes when the completion backlog drains. Cancellation/close wake waiters +// without admitting a request; accepted requests retain the full join contract. +func (v *transactionVerifier) acquireRequest(ctx context.Context, prefetch bool) error { + v.mu.Lock() + waitingCompletion := false + defer func() { + if waitingCompletion { + v.completionWaiters-- + v.wakeAdmissionLocked() + } + v.mu.Unlock() + }() + for { + if v.closed { + return errTransactionVerifierClosed + } + if err := ctx.Err(); err != nil { + return err + } + if len(v.requests) < cap(v.requests) && + (!prefetch || (v.prefetchRequests < cap(v.requests)-1 && v.completionWaiters == 0)) { + v.requests <- struct{}{} + if prefetch { + v.prefetchRequests++ + } + // Add under the same lock as close, so closeAndWait cannot finish + // while an accepted request has yet to start its goroutine. + v.request.Add(1) + return nil + } + if !prefetch && !waitingCompletion { + v.completionWaiters++ + waitingCompletion = true + } + if v.admissionChanged == nil { + v.admissionChanged = make(chan struct{}) + } + changed := v.admissionChanged + v.mu.Unlock() + select { + case <-changed: + case <-ctx.Done(): + case <-v.stopped: + } + v.mu.Lock() + } +} + +func (v *transactionVerifier) releaseRequest(prefetch bool) { + v.mu.Lock() + <-v.requests + if prefetch { + v.prefetchRequests-- + } + v.wakeAdmissionLocked() + v.mu.Unlock() +} + +func (v *transactionVerifier) wakeAdmissionLocked() { + if v.admissionChanged != nil { + close(v.admissionChanged) + v.admissionChanged = nil + } +} diff --git a/pkg/turbine/transaction_verifier_admission_benchmark_test.go b/pkg/turbine/transaction_verifier_admission_benchmark_test.go new file mode 100644 index 000000000..9971224f9 --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission_benchmark_test.go @@ -0,0 +1,118 @@ +package turbine + +import ( + "context" + "fmt" + "testing" + "time" + + "github.com/gagliardetto/solana-go" +) + +// Compare the request-class policy with the same workers, rolling job window, +// Narya backend and total work. Shared submits prefetch as ordinary completion +// requests to reproduce the old four-permit occupancy; reserved labels it as +// prefetch. This is a saturation microbenchmark, not observed live p99 or true +// replay-head scheduling. All signatures are verified and every request joined. +func BenchmarkVerifierCompletionReservation(b *testing.B) { + flowConfigureBackend(b) + txs := make([]*solana.Transaction, 4096) + for i := range txs { + txs[i] = flowGeneratedTransaction(b, 228, uint64(i)) + } + for _, size := range []int{256, 4096} { + for _, reserved := range []bool{false, true} { + b.Run(fmt.Sprintf("prefetch_%d/reserved_%t", size, reserved), func(b *testing.B) { + v := newTransactionVerifierWithBatchTarget(2, 32, 8, nil) + defer v.closeAndWait() + submit := v.submitTransactions + if reserved { + submit = v.submitPrefetchTransactions + } + var admission, finish, total []time.Duration + occupied := 0 + ctx := context.WithValue(context.Background(), entryTraceContextKey{}, true) + warm, err := v.submitTransactions(context.Background(), txs[:256]) + if err != nil { + b.Fatal(err) + } + if _, err = warm.wait(); err != nil { + b.Fatal(err) + } + b.ResetTimer() + for range b.N { + start := time.Now() + prior := make([]*transactionVerification, 0, 4) + for range 3 { + r, err := submit(context.Background(), txs[:size]) + if err != nil { + b.Fatal(err) + } + prior = append(prior, r) + } + type result struct { + r *transactionVerification + err error + } + fourth := make(chan result, 1) + attempting := make(chan struct{}) + go func() { + close(attempting) + r, err := submit(context.Background(), txs[:size]) + fourth <- result{r, err} + }() + <-attempting + if !reserved { + last := <-fourth + if last.err != nil { + b.Fatal(last.err) + } + prior = append(prior, last.r) + } + expectedActive := cap(v.requests) + if reserved { + expectedActive-- + } + if len(v.requests) == expectedActive { + occupied++ + } + r, err := v.submitTransactions(ctx, txs[:32]) + if err != nil { + b.Fatal(err) + } + if _, err = r.wait(); err != nil { + b.Fatal(err) + } + admission = append(admission, time.Duration(r.trace.Admitted-r.trace.Submit)) + finish = append(finish, time.Duration(r.trace.Finished-r.trace.Submit)) + if reserved { + last := <-fourth + if last.err != nil { + b.Fatal(last.err) + } + prior = append(prior, last.r) + } + for _, r := range prior { + if _, err = r.wait(); err != nil { + b.Fatal(err) + } + } + total = append(total, time.Since(start)) + } + b.StopTimer() + b.ReportMetric(float64(occupied)/float64(b.N), "occupied_fraction") + for _, metric := range []struct { + name string + values []time.Duration + }{ + {"completion_admit", admission}, {"completion_done", finish}, {"all_work", total}, + } { + flowReportPercentiles(b, metric.values, metric.name) + index := max(0, (len(metric.values)*99+99)/100-1) + b.ReportMetric(float64(metric.values[index])/float64(time.Millisecond), metric.name+"_p99-ms") + b.ReportMetric(float64(metric.values[len(metric.values)-1])/float64(time.Millisecond), metric.name+"_max-ms") + } + }) + } + } +} diff --git a/pkg/turbine/transaction_verifier_admission_test.go b/pkg/turbine/transaction_verifier_admission_test.go new file mode 100644 index 000000000..eef2c9604 --- /dev/null +++ b/pkg/turbine/transaction_verifier_admission_test.go @@ -0,0 +1,172 @@ +package turbine + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func TestPrefetchLeavesCompletionPermitAndCancellationJoins(t *testing.T) { + release := make(chan struct{}) + v := newTransactionVerifier(2, 32, func(*solana.Transaction) error { <-release; return nil }) + defer v.closeAndWait() + defer close(release) + for range 3 { + _, err := v.submitPrefetchTransactions(context.Background(), verifierTestBlock(64).Transactions) + require.NoError(t, err) + } + require.Equal(t, 3, len(v.requests)) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + blocked := make(chan error, 1) + go func() { _, err := v.submitPrefetchTransactions(ctx, verifierTestBlock(1).Transactions); blocked <- err }() + accepted := make(chan *transactionVerification, 1) + go func() { + r, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + if err != nil { + t.Error(err) + } + accepted <- r + }() + select { + case r := <-accepted: + require.NotNil(t, r) + case <-time.After(3 * time.Second): + t.Fatal("prefetch occupied the reserved completion permit") + } + require.Equal(t, cap(v.requests), len(v.requests), "total bound must not increase") + cancel() + select { + case err := <-blocked: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("canceled prefetch admission did not return") + } + // The admitted jobs are still reading transactions until release closes. + require.Equal(t, 4, len(v.requests)) +} + +// Exercise permit arbitration without depending on cryptographic job duration. +// Holding permits models admitted requests; each release also joins its request. +func TestCompletionWinsAdmissionAndPrefetchResumes(t *testing.T) { + v := newTransactionVerifier(1, 8, nil) + ctx, cancel := context.WithCancel(context.Background()) + var cleanup sync.WaitGroup + defer v.closeAndWait() + defer cleanup.Wait() + defer cancel() + release := func(prefetch bool) { v.releaseRequest(prefetch); v.request.Done() } + for range 2 { + require.NoError(t, v.acquireRequest(ctx, false)) + } + firstReleased := false + defer func() { + if !firstReleased { + release(false) + } + release(false) + }() + completionAdmitted := make(chan struct{}) + completionRelease := make(chan struct{}) + var once sync.Once + defer once.Do(func() { close(completionRelease) }) + cleanup.Go(func() { + if err := v.acquireRequest(ctx, false); err != nil { + return + } + close(completionAdmitted) + select { + case <-completionRelease: + case <-ctx.Done(): + } + release(false) + }) + require.Eventually(t, func() bool { v.mu.Lock(); defer v.mu.Unlock(); return v.completionWaiters == 1 }, 3*time.Second, time.Millisecond) + prefetchAdmitted := make(chan struct{}) + cleanup.Go(func() { + if err := v.acquireRequest(ctx, true); err != nil { + return + } + close(prefetchAdmitted) + release(true) + }) + release(false) + firstReleased = true + waitSignal(t, completionAdmitted, "priority completion admission") + select { + case <-prefetchAdmitted: + t.Fatal("prefetch bypassed waiting completion") + default: + } + once.Do(func() { close(completionRelease) }) + waitSignal(t, prefetchAdmitted, "prefetch resumed after completion") +} + +func TestCanceledCompletionDoesNotBlockPrefetch(t *testing.T) { + v := newTransactionVerifier(1, 8, nil) + defer v.closeAndWait() + for range 2 { + require.NoError(t, v.acquireRequest(context.Background(), false)) + } + defer func() { + for range 2 { + v.releaseRequest(false) + v.request.Done() + } + }() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + done := make(chan error, 1) + go func() { done <- v.acquireRequest(ctx, false) }() + require.Eventually(t, func() bool { v.mu.Lock(); defer v.mu.Unlock(); return v.completionWaiters == 1 }, 3*time.Second, time.Millisecond) + cancel() + select { + case err := <-done: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("completion cancellation stranded admission") + } + v.mu.Lock() + require.Zero(t, v.completionWaiters) + v.mu.Unlock() +} + +func TestCloseWakesBothAdmissionClassesAndJoinsAcceptedWork(t *testing.T) { + release := make(chan struct{}) + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { <-release; return nil }) + var once sync.Once + defer v.closeAndWait() + defer once.Do(func() { close(release) }) + _, err := v.submitPrefetchTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + _, err = v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + results := make(chan error, 2) + for _, prefetch := range []bool{true, false} { + go func() { + _, err := v.submitRequest(context.Background(), verifierTestBlock(1).Transactions, prefetch) + results <- err + }() + } + closed := make(chan struct{}) + go func() { v.closeAndWait(); close(closed) }() + for range 2 { + select { + case err := <-results: + require.ErrorIs(t, err, errTransactionVerifierClosed) + case <-time.After(3 * time.Second): + t.Fatal("close stranded admission") + } + } + select { + case <-closed: + t.Fatal("close returned while jobs still own transactions") + default: + } + once.Do(func() { close(release) }) + waitSignal(t, closed, "close joined accepted work") +} diff --git a/pkg/turbine/transaction_verifier_flow_benchmark_test.go b/pkg/turbine/transaction_verifier_flow_benchmark_test.go new file mode 100644 index 000000000..0c8ce131b --- /dev/null +++ b/pkg/turbine/transaction_verifier_flow_benchmark_test.go @@ -0,0 +1,479 @@ +package turbine + +import ( + "context" + "crypto/ed25519" + "encoding/base64" + "encoding/binary" + "encoding/json" + "fmt" + "os" + "path/filepath" + "sort" + "strconv" + "sync" + "syscall" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// BenchmarkTransactionVerificationFlow measures the real verifier pool under +// simulated transaction availability, not actual network reception or replay. +// Decode and fixture generation are outside the timer. "tip" spreads complete +// components over 200 ms; this is an explicit workload model, not a claim about +// the cluster's observed component sizes or arrival distribution. No component +// waits for another component to fill a verification batch. +// +// Optional environment: +// +// MITHRIL_SIGVERIFY_FLOW_BACKEND=r51 (use -run '^$' in a fresh test process) +// MITHRIL_SIGVERIFY_FLOW_FIXTURES=/path/to/fixtures (block-*.json, base64 txs) +// MITHRIL_SIGVERIFY_FLOW_COUNT=33760 (generated count or captured prefix limit) +// +// Without captured fixtures, use reproducible signed 228-byte and 1232-byte +// legacy memo transactions. They are signature workloads, not replay fixtures. +// Use -benchtime=3x or another fixed count when comparing configurations so the +// deliberate arrival waits do not change the number of observations. +func BenchmarkTransactionVerificationFlow(b *testing.B) { + flowConfigureBackend(b) + for _, fixture := range flowBenchmarkFixtures(b) { + b.Run(fixture.name, func(b *testing.B) { + for _, workers := range []int{2, 4} { + b.Run(fmt.Sprintf("workers_%d", workers), func(b *testing.B) { + for _, target := range []int{4, 8} { + b.Run(fmt.Sprintf("target_%d", target), func(b *testing.B) { + for _, scenario := range flowScenarios(fixture.blk) { + b.Run(scenario.name, func(b *testing.B) { + flowRunBenchmark(b, workers, target, fixture.blk, scenario) + }) + } + }) + } + }) + } + }) + } +} + +type flowFixture struct { + name string + blk *block.Block +} + +type flowScenario struct { + name string + components [][]*solana.Transaction + arrivalSpan time.Duration + overlap bool +} + +func flowScenarios(blk *block.Block) []flowScenario { + // 60 KiB is a benchmark parameter only. Include signed transaction bytes; + // actual serialized entry/component overhead and shred recovery are omitted. + components := flowComponents(blk.Transactions, 60*1024) + scenarios := []flowScenario{ + {name: "catchup", components: [][]*solana.Transaction{blk.Transactions}}, + {name: "tip_200ms_after_complete", components: components, arrivalSpan: 200 * time.Millisecond}, + {name: "tip_200ms_overlap", components: components, arrivalSpan: 200 * time.Millisecond, overlap: true}, + } + // Sparse components expose tail behavior without requiring microsecond + // timer precision to deliver tens of thousands of tiny arrival events. + for _, width := range []int{4, 7, 8} { + txs := blk.Transactions[:min(256, len(blk.Transactions))] + var small [][]*solana.Transaction + for start := 0; start < len(txs); start += width { + small = append(small, txs[start:min(start+width, len(txs))]) + } + scenarios = append(scenarios, flowScenario{ + name: fmt.Sprintf("sparse_%dtx_200ms_overlap", width), components: small, + arrivalSpan: 200 * time.Millisecond, overlap: true, + }) + } + return scenarios +} + +func flowComponents(txs []*solana.Transaction, bytesPerComponent int) [][]*solana.Transaction { + var components [][]*solana.Transaction + start, size := 0, 0 + for i, tx := range txs { + wireSize, err := txverify.TransactionWireSize(tx) + if err != nil { + panic(err) // fixtures are validated before entering this helper + } + if i > start && size+wireSize > bytesPerComponent { + components = append(components, txs[start:i]) + start, size = i, 0 + } + size += wireSize + } + if start < len(txs) { + components = append(components, txs[start:]) + } + return components +} + +type flowObservation struct { + latencies []time.Duration + submits []time.Duration + feedLags []time.Duration + residual time.Duration + err error +} + +func flowArrivalOffset(index, count int, span time.Duration) time.Duration { + if count <= 1 { + return span + } + return time.Duration(int64(span) * int64(index) / int64(count-1)) +} + +func flowObserve(v *transactionVerifier, blk *block.Block, scenario flowScenario) flowObservation { + observation := flowObservation{ + latencies: make([]time.Duration, len(scenario.components)), + submits: make([]time.Duration, len(scenario.components)), + feedLags: make([]time.Duration, len(scenario.components)), + } + started := time.Now() + finalArrival := started.Add(scenario.arrivalSpan) + if !scenario.overlap { + time.Sleep(time.Until(finalArrival)) + submitStarted := time.Now() + future, err := v.submitTransactions(context.Background(), blk.Transactions) + submitDuration := time.Since(submitStarted) + if err != nil { + observation.err = err + return observation + } + _, observation.err = future.wait() + finished := future.finishedAt + for i := range observation.latencies { + available := started.Add(flowArrivalOffset(i, len(scenario.components), scenario.arrivalSpan)) + observation.latencies[i] = finished.Sub(available) + observation.submits[i] = submitDuration + observation.feedLags[i] = submitStarted.Sub(available) + } + observation.residual = max(0, finished.Sub(finalArrival)) + return observation + } + + var waiters sync.WaitGroup + errs := make([]error, len(scenario.components)) + finished := make([]time.Time, len(scenario.components)) + for i, txs := range scenario.components { + available := started.Add(flowArrivalOffset(i, len(scenario.components), scenario.arrivalSpan)) + time.Sleep(time.Until(available)) + submitStarted := time.Now() + observation.feedLags[i] = submitStarted.Sub(available) + future, err := v.submitTransactions(context.Background(), txs) + observation.submits[i] = time.Since(submitStarted) + if err != nil { + errs[i] = err + break + } + waiters.Add(1) + go func() { + defer waiters.Done() + _, errs[i] = future.wait() + finished[i] = future.finishedAt + observation.latencies[i] = finished[i].Sub(available) + }() + } + waiters.Wait() + for _, completed := range finished { + observation.residual = max(observation.residual, completed.Sub(finalArrival)) + } + for _, err := range errs { + if err != nil { + observation.err = err + break + } + } + return observation +} + +func flowRunBenchmark(b *testing.B, workers, target int, blk *block.Block, scenario flowScenario) { + v := newTransactionVerifierWithBatchTarget(workers, 2*workers*8, target, nil) + defer v.closeAndWait() + if err := v.verifyBlock(blk); err != nil { + b.Fatal(err) + } + flowBenchmarkGate(b) + var signatureCount int + for _, component := range scenario.components { + for _, tx := range component { + signatureCount += len(tx.Signatures) + } + } + var latencies, submits, feedLags, residuals []time.Duration + before := sigverify.Stats() + cpuBefore := flowCPUSeconds(b) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + observation := flowObserve(v, blk, scenario) + if observation.err != nil { + b.Fatal(observation.err) + } + latencies = append(latencies, observation.latencies...) + submits = append(submits, observation.submits...) + feedLags = append(feedLags, observation.feedLags...) + residuals = append(residuals, observation.residual) + } + b.StopTimer() + cpuSeconds := flowCPUSeconds(b) - cpuBefore + after := sigverify.Stats() + b.ReportMetric(1000*cpuSeconds/float64(b.N), "cpu-ms/block") + b.ReportMetric(cpuSeconds/b.Elapsed().Seconds(), "avg_cpu_cores") + b.ReportMetric(float64(b.N*signatureCount)/b.Elapsed().Seconds(), "signatures/s") + b.ReportMetric(float64(signatureCount), "signatures/block") + b.ReportMetric(float64(len(scenario.components)), "components/block") + b.ReportMetric(float64(after.Signatures-before.Signatures)/float64(after.Batches-before.Batches), "mean_width") + flowReportPercentiles(b, latencies, "ready") + flowReportPercentiles(b, submits, "submit") + flowReportPercentiles(b, feedLags, "feed_lag") + flowReportPercentiles(b, residuals, "residual") + if after.InternalFaultFallbacks != before.InternalFaultFallbacks { + b.Fatal("signature verifier used an internal fault fallback") + } + if want := uint64(b.N * signatureCount); after.Signatures-before.Signatures != want { + b.Fatalf("verified signature count = %d, want %d", after.Signatures-before.Signatures, want) + } +} + +// The optional rendezvous is outside the timer. An external contention runner +// starts the real execution probe after seeing READY, then creates START. Both +// paths are explicit files in that runner's output directory. This is only for +// coordination; ordinary benchmark invocations do no filesystem polling. +func flowBenchmarkGate(b *testing.B) { + b.Helper() + ready, start := os.Getenv("MITHRIL_SIGVERIFY_FLOW_READY"), os.Getenv("MITHRIL_SIGVERIFY_FLOW_START") + if ready == "" && start == "" { + return + } + if ready == "" || start == "" { + b.Fatal("set both MITHRIL_SIGVERIFY_FLOW_READY and MITHRIL_SIGVERIFY_FLOW_START") + } + // Go's first N=1 calibration and warmup must finish before the external + // execution probe starts. The runner selects a fixed N greater than one. + if b.N == 1 { + return + } + if err := os.WriteFile(ready, []byte(b.Name()+"\n"), 0600); err != nil { + b.Fatal(err) + } + deadline := time.Now().Add(30 * time.Second) + for { + if _, err := os.Stat(start); err == nil { + return + } else if !os.IsNotExist(err) { + b.Fatal(err) + } + if time.Now().After(deadline) { + b.Fatal("contention runner did not release benchmark within 30 seconds") + } + time.Sleep(time.Millisecond) + } +} + +func flowReportPercentiles(b *testing.B, values []time.Duration, prefix string) { + sort.Slice(values, func(i, j int) bool { return values[i] < values[j] }) + for _, percentile := range []int{50, 95} { + index := max(0, (len(values)*percentile+99)/100-1) + b.ReportMetric(float64(values[index])/float64(time.Millisecond), fmt.Sprintf("%s_p%d-ms", prefix, percentile)) + } +} + +func flowCPUSeconds(tb testing.TB) float64 { + tb.Helper() + var usage syscall.Rusage + if err := syscall.Getrusage(syscall.RUSAGE_SELF, &usage); err != nil { + tb.Fatal(err) + } + return float64(usage.Utime.Sec+usage.Stime.Sec) + float64(usage.Utime.Usec+usage.Stime.Usec)/1e6 +} + +var flowBackendOnce sync.Once +var flowBackendError error + +func flowConfigureBackend(tb testing.TB) { + tb.Helper() + flowBackendOnce.Do(func() { + if backend := os.Getenv("MITHRIL_SIGVERIFY_FLOW_BACKEND"); backend != "" { + _, flowBackendError = sigverify.Configure(sigverify.Config{Backend: backend}) + } + }) + if flowBackendError != nil { + tb.Fatal(flowBackendError) + } +} + +func flowBenchmarkFixtures(tb testing.TB) []flowFixture { + tb.Helper() + count := 33760 + limit := false + if value := os.Getenv("MITHRIL_SIGVERIFY_FLOW_COUNT"); value != "" { + var err error + count, err = strconv.Atoi(value) + if err != nil || count < 1 { + tb.Fatal("MITHRIL_SIGVERIFY_FLOW_COUNT must be a positive integer") + } + limit = true + } + if dir := os.Getenv("MITHRIL_SIGVERIFY_FLOW_FIXTURES"); dir != "" { + paths, err := filepath.Glob(filepath.Join(dir, "block-*.json")) + if err != nil || len(paths) == 0 { + tb.Fatalf("captured fixtures: %v, files=%d", err, len(paths)) + } + var fixtures []flowFixture + for _, path := range paths { + data, err := os.ReadFile(path) + if err != nil { + tb.Fatal(err) + } + var captured struct { + Slot uint64 + Transactions []string + } + if err := json.Unmarshal(data, &captured); err != nil { + tb.Fatal(err) + } + blk := &block.Block{Slot: captured.Slot} + for i, encoded := range captured.Transactions { + if limit && i == count { + break + } + wire, err := base64.StdEncoding.DecodeString(encoded) + if err != nil { + tb.Fatal(err) + } + tx, err := solana.TransactionFromBytes(wire) + if err != nil { + tb.Fatal(err) + } + if err := txverify.SanitizeTransaction(tx); err != nil { + tb.Fatal(err) + } + blk.Transactions = append(blk.Transactions, tx) + } + if len(blk.Transactions) == 0 { + tb.Fatalf("fixture %s contains no transactions", path) + } + fixtures = append(fixtures, flowFixture{fmt.Sprintf("captured_%d", blk.Slot), blk}) + } + return fixtures + } + var fixtures []flowFixture + for _, wireSize := range []int{228, txverify.MaxLegacyTransactionSize} { + blk := &block.Block{Slot: 1, Transactions: make([]*solana.Transaction, count)} + for i := range blk.Transactions { + blk.Transactions[i] = flowGeneratedTransaction(tb, wireSize, uint64(i)) + } + fixtures = append(fixtures, flowFixture{fmt.Sprintf("generated_%dB", wireSize), blk}) + } + return fixtures +} + +func flowGeneratedTransaction(tb testing.TB, wireSize int, index uint64) *solana.Transaction { + tb.Helper() + // Public, deterministic benchmark material, never a validator identity. + var seed [ed25519.SeedSize]byte + binary.LittleEndian.PutUint64(seed[:], index+1) + private := ed25519.NewKeyFromSeed(seed[:]) + public := solana.PublicKeyFromBytes(private.Public().(ed25519.PublicKey)) + tx := &solana.Transaction{ + Signatures: make([]solana.Signature, 1), + Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: []solana.PublicKey{public, solana.MemoProgramID}, + Instructions: []solana.CompiledInstruction{{ProgramIDIndex: 1, Accounts: []uint16{0}}}, + }, + } + binary.LittleEndian.PutUint64(tx.Message.RecentBlockhash[:], index+1) + baseSize, err := txverify.TransactionWireSize(tx) + if err != nil { + tb.Fatal(err) + } + padding := wireSize - baseSize + for attempts := 0; attempts < 3 && padding >= 0; attempts++ { + tx.Message.Instructions[0].Data = make([]byte, padding) + size, err := txverify.TransactionWireSize(tx) + if err != nil { + tb.Fatal(err) + } + if size != wireSize { + padding += wireSize - size + continue + } + for i := range tx.Message.Instructions[0].Data { + tx.Message.Instructions[0].Data[i] = 'a' + } + message, err := txverify.MessageBytes(tx) + if err != nil { + tb.Fatal(err) + } + copy(tx.Signatures[0][:], ed25519.Sign(private, message)) + if err := txverify.SanitizeTransaction(tx); err != nil { + tb.Fatal(err) + } + return tx + } + tb.Fatalf("cannot construct a %d-byte transaction", wireSize) + return nil +} + +func TestTransactionVerificationFlowFixtureShape(t *testing.T) { + for _, size := range []int{228, 1232} { + first := flowGeneratedTransaction(t, size, 0) + second := flowGeneratedTransaction(t, size, 1) + for _, tx := range []*solana.Transaction{first, second} { + wire, err := tx.MarshalBinary() + if err != nil { + t.Fatal(err) + } + if len(wire) != size || len(tx.Signatures) != 1 { + t.Fatalf("fixture wire=%d signatures=%d, want %d bytes and one signature", len(wire), len(tx.Signatures), size) + } + if err := txverify.VerifyTransaction(tx); err != nil { + t.Fatal(err) + } + } + if first.Signatures[0] == second.Signatures[0] || first.Message.AccountKeys[0] == second.Message.AccountKeys[0] { + t.Fatal("different fixture indices reused a message signature or signer") + } + } +} + +func TestTransactionVerificationFlowComponentBoundaries(t *testing.T) { + txs := make([]*solana.Transaction, 7) + for i := range txs { + txs[i] = flowGeneratedTransaction(t, 228, uint64(i)) + } + components := flowComponents(txs, 500) + if len(components) != 4 { + t.Fatalf("got %d components, want four", len(components)) + } + next := 0 + for i, component := range components { + want := 2 + if i == 3 { + want = 1 + } + if len(component) != want { + t.Fatalf("component %d contains %d transactions, want %d", i, len(component), want) + } + for _, tx := range component { + if tx != txs[next] { + t.Fatal("component split changed transaction order or identity") + } + next++ + } + } + if flowArrivalOffset(0, 4, 200*time.Millisecond) != 0 || flowArrivalOffset(3, 4, 200*time.Millisecond) != 200*time.Millisecond { + t.Fatal("availability schedule does not span the requested interval") + } +} diff --git a/pkg/turbine/transaction_verifier_test.go b/pkg/turbine/transaction_verifier_test.go index 8eb19c63d..dbb73e859 100644 --- a/pkg/turbine/transaction_verifier_test.go +++ b/pkg/turbine/transaction_verifier_test.go @@ -1,9 +1,12 @@ package turbine import ( + "context" + "crypto/ed25519" "errors" "fmt" "strings" + "sync" "sync/atomic" "testing" "time" @@ -43,12 +46,12 @@ func TestTransactionVerifierBoundsConcurrencyAndQueue(t *testing.T) { return nil }) defer verifier.closeAndWait() - if cap(verifier.jobs) != 2*workers { - t.Fatalf("queue capacity = %d, want %d", cap(verifier.jobs), 2*workers) + if got, want := cap(verifier.jobs), 1; got != want { + t.Fatalf("group queue capacity = %d, want %d", got, want) } done := make(chan error, 1) - go func() { done <- verifier.verifyBlock(verifierTestBlock(12)) }() + go func() { done <- verifier.verifyBlock(verifierTestBlock(24)) }() deadline := time.After(3 * time.Second) for active.Load() != workers { select { @@ -76,7 +79,7 @@ func TestTransactionVerifierBoundsConcurrencyAndQueue(t *testing.T) { } func TestTransactionVerifierReturnsLowestFailingIndex(t *testing.T) { - blk := verifierTestBlock(6) + blk := verifierTestBlock(24) lowErr := errors.New("low index failure") highErr := errors.New("high index failure") verifier := newTransactionVerifier(4, 8, func(tx *solana.Transaction) error { @@ -84,7 +87,7 @@ func TestTransactionVerifierReturnsLowestFailingIndex(t *testing.T) { case blk.Transactions[1]: time.Sleep(10 * time.Millisecond) return lowErr - case blk.Transactions[3]: + case blk.Transactions[9]: return highErr default: return nil @@ -132,8 +135,8 @@ func TestTransactionVerifierRejectsNilAtDeterministicIndex(t *testing.T) { // Every transaction in a block must be verified and joined, whatever the count. // Workers group transactions, so a count that divides badly into groups must -// not leave a remainder waiting for company: verifyBlockContext joins each -// wave with done.Wait(), and a stranded job would hang it forever. +// not leave a remainder waiting for company: every tail is dispatched as +// soon as it is available, without waiting for another request. // // A counting verifier is injected so the assertion is on what was actually // verified, not merely on returning without error. @@ -161,3 +164,290 @@ func TestTransactionVerifierVerifiesEveryTransactionForAwkwardCounts(t *testing. }) } } + +func TestTransactionVerifierRefillsWhileEarlierGroupIsBlocked(t *testing.T) { + blk := verifierTestBlock(24) + started := make(chan struct{}) + release := make(chan struct{}) + refilled := make(chan struct{}) + v := newTransactionVerifierWithBatchTarget(2, 16, 8, func(tx *solana.Transaction) error { + switch tx { + case blk.Transactions[0]: + close(started) + <-release + case blk.Transactions[16]: + close(refilled) + } + return nil + }) + defer v.closeAndWait() + defer close(release) + r, err := v.submitTransactions(context.Background(), blk.Transactions) + require.NoError(t, err) + waitSignal(t, started, "slow first group") + waitSignal(t, refilled, "rolling refill before first group finishes") + select { + case <-r.done: + t.Fatal("request finished without joining its blocked group") + default: + } +} + +func TestTransactionVerifierPartialTailStartsWithoutAnotherSubmission(t *testing.T) { + for _, target := range []int{4, 8} { + t.Run(fmt.Sprintf("target=%d", target), func(t *testing.T) { + seen := make(chan struct{}, 3) + v := newTransactionVerifierWithBatchTarget(2, 16, target, func(*solana.Transaction) error { + seen <- struct{}{} + return nil + }) + defer v.closeAndWait() + r, err := v.submitTransactions(context.Background(), verifierTestBlock(3).Transactions) + require.NoError(t, err) + for range 3 { + waitSignal(t, seen, "available partial batch transaction") + } + _, err = r.wait() + require.NoError(t, err) + require.False(t, r.finishedAt.IsZero()) + }) + } +} + +func TestTransactionVerifierLargeRequestDoesNotQueuePastSmallRequest(t *testing.T) { + large := verifierTestBlock(800) + small := verifierTestBlock(1) + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + var largeCalls atomic.Int32 + var callsBeforeSmall atomic.Int32 + v := newTransactionVerifierWithBatchTarget(1, 16, 8, func(tx *solana.Transaction) error { + if tx == small.Transactions[0] { + callsBeforeSmall.Store(largeCalls.Load()) + return nil + } + largeCalls.Add(1) + if tx == large.Transactions[0] { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + largeRequest, err := v.submitTransactions(context.Background(), large.Transactions) + require.NoError(t, err) + waitSignal(t, started, "large request first group") + smallRequest, err := v.submitTransactions(context.Background(), small.Transactions) + require.NoError(t, err) + require.Eventually(t, func() bool { return len(v.jobs) == 1 }, 3*time.Second, time.Millisecond) + releaseOnce.Do(func() { close(release) }) + _, err = smallRequest.wait() + require.NoError(t, err) + require.Equal(t, int32(v.batchTarget*v.jobGroups), callsBeforeSmall.Load(), "large request may only stay one job ahead") + _, err = largeRequest.wait() + require.NoError(t, err) + require.Equal(t, int32(800), largeCalls.Load()) +} + +func TestTransactionVerifierAsyncAdmissionAppliesCancelableBackpressure(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var first sync.Once + v := newTransactionVerifierWithBatchTarget(1, 8, 8, func(*solana.Transaction) error { + first.Do(func() { close(started) }) + <-release + return nil + }) + defer v.closeAndWait() + defer close(release) + for range 2 { + _, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.NoError(t, err) + } + waitSignal(t, started, "occupied request slots") + require.Equal(t, cap(v.requests), len(v.requests)) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + result := make(chan error, 1) + go func() { + _, err := v.submitTransactions(ctx, verifierTestBlock(1).Transactions) + result <- err + }() + select { + case err := <-result: + t.Fatalf("unbounded request admitted instead of waiting: %v", err) + case <-time.After(20 * time.Millisecond): + } + cancel() + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("request admission ignored cancellation") + } +} + +func TestTransactionVerificationWaitContextCancelsAndJoins(t *testing.T) { + started := make(chan struct{}) + release := make(chan struct{}) + var releaseOnce sync.Once + var seen atomic.Int32 + v := newTransactionVerifierWithBatchTarget(1, 8, 8, func(*solana.Transaction) error { + if seen.Add(1) == 1 { + close(started) + <-release + } + return nil + }) + defer v.closeAndWait() + defer releaseOnce.Do(func() { close(release) }) + r, err := v.submitTransactions(context.Background(), verifierTestBlock(800).Transactions) + require.NoError(t, err) + waitSignal(t, started, "first admitted group") + ctx, cancel := context.WithCancel(context.Background()) + cancel() + result := make(chan error, 1) + go func() { _, err := r.waitContext(ctx); result <- err }() + select { + case err := <-result: + t.Fatalf("wait returned while transactions still owned by worker: %v", err) + case <-time.After(20 * time.Millisecond): + } + releaseOnce.Do(func() { close(release) }) + select { + case err := <-result: + require.ErrorIs(t, err, context.Canceled) + case <-time.After(3 * time.Second): + t.Fatal("canceled request did not join") + } + require.Equal(t, int32(8), seen.Load(), "cancellation must stop later group admission") +} + +func TestTransactionVerifierCloseRacesAdmissionWithoutStrandingRequests(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + var callers sync.WaitGroup + for range 32 { + callers.Go(func() { + r, err := v.submitTransactions(context.Background(), verifierTestBlock(17).Transactions) + if err != nil { + if !errors.Is(err, errTransactionVerifierClosed) { + t.Errorf("submit error: %v", err) + } + return + } + _, err = r.wait() + if err != nil { + t.Errorf("admitted request error: %v", err) + } + }) + } + v.closeAndWait() + callers.Wait() + _, err := v.submitTransactions(context.Background(), verifierTestBlock(1).Transactions) + require.ErrorIs(t, err, errTransactionVerifierClosed) +} + +func verifierSignedTransactions(t *testing.T, count int) []*solana.Transaction { + t.Helper() + seed := make([]byte, ed25519.SeedSize) + seed[0] = 71 // Deterministic test-only key; never a validator identity. + key := ed25519.NewKeyFromSeed(seed) + var public solana.PublicKey + copy(public[:], key[32:]) + txs := make([]*solana.Transaction, count) + for i := range txs { + tx := &solana.Transaction{ + Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1}, + AccountKeys: []solana.PublicKey{public}, + RecentBlockhash: solana.Hash{byte(i), byte(i >> 8)}, + }, + Signatures: make([]solana.Signature, 1), + } + message, err := tx.Message.MarshalBinary() + require.NoError(t, err) + copy(tx.Signatures[0][:], ed25519.Sign(key, message)) + txs[i] = tx + } + return txs +} + +func TestTransactionVerifierRejectsEveryInvalidSignatureLane(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + for invalid := range 8 { + t.Run(fmt.Sprintf("lane=%d", invalid), func(t *testing.T) { + txs := verifierSignedTransactions(t, 8) + txs[invalid].Signatures[0][13] ^= 0x40 + r, err := v.submitTransactions(context.Background(), txs) + require.NoError(t, err) + index, err := r.wait() + require.ErrorContains(t, err, "invalid signature") + require.Equal(t, invalid, index) + }) + } + r, err := v.submitTransactions(context.Background(), verifierSignedTransactions(t, 17)) + require.NoError(t, err) + _, err = r.wait() + require.NoError(t, err, "valid transactions must still verify after invalid lanes") +} + +func TestTransactionVerifierKeepsMultisignatureTransactionsIntactAcrossTargets(t *testing.T) { + counts := []int{2, 2, 1, 4, 9, 3, 2, 1} + txs := make([]*solana.Transaction, len(counts)) + for i, count := range counts { + keys := make([]ed25519.PrivateKey, count) + public := make([]solana.PublicKey, count) + for signer := range keys { + seed := make([]byte, ed25519.SeedSize) + seed[0], seed[1] = 83, byte(signer) + keys[signer] = ed25519.NewKeyFromSeed(seed) + copy(public[signer][:], keys[signer][32:]) + } + tx := &solana.Transaction{ + Message: solana.Message{ + Header: solana.MessageHeader{ + NumRequiredSignatures: uint8(count), + NumReadonlySignedAccounts: uint8(count - 1), + }, + AccountKeys: public, + RecentBlockhash: solana.Hash{byte(i)}, + }, + Signatures: make([]solana.Signature, count), + } + message, err := tx.Message.MarshalBinary() + require.NoError(t, err) + for signer, key := range keys { + copy(tx.Signatures[signer][:], ed25519.Sign(key, message)) + } + txs[i] = tx + } + for _, target := range []int{4, 8} { + for _, groups := range []int{1, 4, 8} { + t.Run(fmt.Sprintf("target=%d/groups=%d", target, groups), func(t *testing.T) { + v := newTransactionVerifierWithJobGroups(2, 16, target, groups, nil) + defer v.closeAndWait() + var large []*solana.Transaction + for range 40 { + large = append(large, txs...) + } + r, err := v.submitTransactions(context.Background(), large) + require.NoError(t, err) + _, err = r.wait() + require.NoError(t, err) + + // Corrupt a non-first signer after an oversized (nine-signature) + // transaction. Results must still map to the original tx index. + txs[6].Signatures[1][11] ^= 0x20 + r, err = v.submitTransactions(context.Background(), large) + require.NoError(t, err) + index, err := r.wait() + require.ErrorContains(t, err, "invalid signature") + require.Equal(t, 6, index) + txs[6].Signatures[1][11] ^= 0x20 + }) + } + } +} diff --git a/pkg/txstatus/message_identity.go b/pkg/txstatus/message_identity.go index 4d41f5450..4541a2a12 100644 --- a/pkg/txstatus/message_identity.go +++ b/pkg/txstatus/message_identity.go @@ -27,11 +27,18 @@ func TransactionMessageHash(tx *solana.Transaction) ([32]byte, error) { return messageHash, fmt.Errorf("serialize transaction message: %w", err) } + return HashCanonicalMessage(message), nil +} + +// HashCanonicalMessage hashes the exact canonical bytes used for transaction +// signature verification, including any message-version prefix. +func HashCanonicalMessage(message []byte) [32]byte { + var messageHash [32]byte hasher := blake3.New() _, _ = hasher.Write([]byte(transactionMessageHashDomain)) _, _ = hasher.Write(message) hasher.Sum(messageHash[:0]) - return messageHash, nil + return messageHash } // IdentityForTransaction captures both components needed for a status-cache diff --git a/pkg/txverify/message_identity.go b/pkg/txverify/message_identity.go new file mode 100644 index 000000000..4ba83bfc2 --- /dev/null +++ b/pkg/txverify/message_identity.go @@ -0,0 +1,25 @@ +package txverify + +import ( + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) + +// VerifiedMessageIdentity is an immutable result of signature verification. +// Its zero value is unusable. Signed message contents must remain immutable; +// as with Block's existing cache, arbitrary in-place edits require invalidation. +type VerifiedMessageIdentity struct { + transaction *solana.Transaction + version solana.MessageVersion + identity txstatus.TransactionMessageIdentity + verified bool +} + +// ForTransaction checks the binding without reserializing. Address-table +// resolution is allowed because it does not change the canonical message. +func (v VerifiedMessageIdentity) ForTransaction(tx *solana.Transaction) (txstatus.TransactionMessageIdentity, bool) { + if !v.verified || tx == nil || tx != v.transaction || tx.Message.GetVersion() != v.version || tx.Message.RecentBlockhash != v.identity.RecentBlockhash { + return txstatus.TransactionMessageIdentity{}, false + } + return v.identity, true +} diff --git a/pkg/txverify/message_identity_test.go b/pkg/txverify/message_identity_test.go new file mode 100644 index 000000000..29d265497 --- /dev/null +++ b/pkg/txverify/message_identity_test.go @@ -0,0 +1,70 @@ +package txverify + +import ( + "crypto/ed25519" + "encoding/hex" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func identitySignedTransaction(t *testing.T, version solana.MessageVersion) *solana.Transaction { + t.Helper() + key := ed25519.NewKeyFromSeed(make([]byte, 32)) + tx := &solana.Transaction{Message: solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: []solana.PublicKey{solana.PublicKeyFromBytes(key.Public().(ed25519.PublicKey)), {2}}, + RecentBlockhash: solana.Hash{3}, + Instructions: []solana.CompiledInstruction{{ProgramIDIndex: 1, Accounts: []uint16{0}, Data: []byte{4}}}, + }} + _, err := tx.Message.SetVersion(version) + require.NoError(t, err) + msg, err := MessageBytes(tx) + require.NoError(t, err) + tx.Signatures = []solana.Signature{solana.SignatureFromBytes(ed25519.Sign(key, msg))} + return tx +} + +func TestVerifiedMessageIdentityCanonicalVersionsAndFailures(t *testing.T) { + wire, err := hex.DecodeString(rustV1FeeHeapTransaction) + require.NoError(t, err) + v1, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + txs := []*solana.Transaction{identitySignedTransaction(t, solana.MessageVersionLegacy), identitySignedTransaction(t, solana.MessageVersionV0), v1} + bad := identitySignedTransaction(t, solana.MessageVersionLegacy) + bad.Signatures[0][0] ^= 1 + txs = append(txs, bad, nil) + errs := make([]error, len(txs)) + ids := make([]VerifiedMessageIdentity, len(txs)) + var verifier BatchVerifier + verifier.VerifyWithMessageIdentities(txs, errs, ids) + for i := range txs { + id, ok := ids[i].ForTransaction(txs[i]) + if i >= 3 { + require.Error(t, errs[i]) + require.False(t, ok) + continue + } + require.NoError(t, errs[i]) + require.True(t, ok) + want, err := txstatus.IdentityForTransaction(txs[i]) + require.NoError(t, err) + require.Equal(t, want, id) + } + // Scratch reuse cannot invalidate a prior successful request, and reusing + // an output lane for failure must not leave a usable old identity behind. + saved := ids[0] + verifier.VerifyWithMessageIdentities([]*solana.Transaction{bad}, errs[:1], ids[:1]) + _, ok := saved.ForTransaction(txs[0]) + require.True(t, ok) + _, ok = ids[0].ForTransaction(txs[0]) + require.False(t, ok) + copyTx := *txs[0] + _, ok = saved.ForTransaction(©Tx) + require.False(t, ok) + txs[0].Message.RecentBlockhash[0] ^= 1 + _, ok = saved.ForTransaction(txs[0]) + require.False(t, ok) +} diff --git a/pkg/txverify/txverify.go b/pkg/txverify/txverify.go index 7221399c4..af58b9cb5 100644 --- a/pkg/txverify/txverify.go +++ b/pkg/txverify/txverify.go @@ -4,6 +4,7 @@ import ( "fmt" "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/txstatus" "github.com/gagliardetto/solana-go" ) @@ -222,6 +223,21 @@ type BatchVerifier struct { // Every transaction gets an independent verdict: one bad transaction does not // mask the others, so a caller can report precisely which one failed. func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { + v.verify(txs, errs, nil) +} + +// VerifyWithMessageIdentities additionally retains identities derived from the +// same canonical bytes used by signature verification. Only successful verdicts +// produce usable identities. The output is caller-owned, not verifier scratch. +func (v *BatchVerifier) VerifyWithMessageIdentities(txs []*solana.Transaction, errs []error, identities []VerifiedMessageIdentity) { + if len(identities) != len(txs) { + panic("txverify: identities and txs length mismatch") + } + clear(identities) + v.verify(txs, errs, identities) +} + +func (v *BatchVerifier) verify(txs []*solana.Transaction, errs []error, identities []VerifiedMessageIdentity) { if len(errs) != len(txs) { panic("txverify: errs and txs length mismatch") } @@ -239,6 +255,16 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { v.signers = append(v.signers, nil) continue } + if identities != nil { + identities[i] = VerifiedMessageIdentity{ + transaction: tx, + version: tx.Message.GetVersion(), + identity: txstatus.TransactionMessageIdentity{ + MessageHash: txstatus.HashCanonicalMessage(msg), + RecentBlockhash: tx.Message.RecentBlockhash, + }, + } + } for j := range tx.Signatures { v.batch.Add((*[32]byte)(&signers[j]), msg, tx.Signatures[j][:]) } @@ -247,6 +273,9 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { } if v.batch.Verify() { + for i := range identities { + identities[i].verified = errs[i] == nil + } return } @@ -257,6 +286,9 @@ func (v *BatchVerifier) Verify(txs []*solana.Transaction, errs []error) { errs[i] = fmt.Errorf("invalid signature by %s", v.signers[i][j]) } } + if identities != nil { + identities[i].verified = errs[i] == nil + } lane += count } } From 28e32380e6ed221beed256d69d840c591d5ccac2 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 13 Sep 2026 23:12:41 -0500 Subject: [PATCH 002/111] Move transaction-status checkpoint encoding off the replay loop --- pkg/replay/async_checkpoint_capture_test.go | 172 ++++++++++++++++++ pkg/replay/async_promotion_test.go | 18 +- pkg/replay/block.go | 6 +- pkg/replay/promotion.go | 92 +++++++--- pkg/replay/transaction_status_cache.go | 54 +++++- .../transaction_status_capture_bench_test.go | 66 +++++++ pkg/replay/transaction_status_capture_test.go | 162 +++++++++++++++++ 7 files changed, 524 insertions(+), 46 deletions(-) create mode 100644 pkg/replay/async_checkpoint_capture_test.go create mode 100644 pkg/replay/transaction_status_capture_bench_test.go create mode 100644 pkg/replay/transaction_status_capture_test.go diff --git a/pkg/replay/async_checkpoint_capture_test.go b/pkg/replay/async_checkpoint_capture_test.go new file mode 100644 index 000000000..a556c5a37 --- /dev/null +++ b/pkg/replay/async_checkpoint_capture_test.go @@ -0,0 +1,172 @@ +package replay + +import ( + "encoding/json" + "errors" + "sync" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/state" + "github.com/stretchr/testify/require" +) + +type testCheckpointEncoder func() ([]byte, error) + +func (f testCheckpointEncoder) MarshalBinary() ([]byte, error) { return f() } + +func testCheckpointBytes(payload []byte) TransactionStatusSnapshot { + owned := append([]byte(nil), payload...) + return testCheckpointEncoder(func() ([]byte, error) { return append([]byte(nil), owned...), nil }) +} + +func TestAsyncCheckpointEncodingDoesNotRunDuringJobBuild(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + started, release := make(chan struct{}), make(chan struct{}) + blockEncoding := false + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: func(through uint64) (TransactionStatusSnapshot, error) { + require.Equal(t, uint64(6), through) + return testCheckpointEncoder(func() ([]byte, error) { + if !blockEncoding { + return nil, errors.New("encoder ran during job construction") + } + close(started) + <-release + return []byte("encoded-on-worker"), nil + }), nil + }, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(6, false) + require.NoError(t, err) + select { + case <-started: + t.Fatal("job construction ran the encoder") + default: + } + promoter := newAsyncPromoter(fc) + // Always unblock the worker before draining it, including a failed assertion. + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + defer promoter.stop() + defer unblock() + blockEncoding = true + promoter.enqueue(job) + select { + case <-started: + case <-time.After(2 * time.Second): + t.Fatal("checkpoint worker did not start encoding") + } + // Replay can publish a later bank while the worker's encoder is blocked. + tail.Add(7, []*accounts.Account{testAccount(3, 7)}, testHashBytes(7)) + tail.SetContext(7, &state.ResumeContext{Slot: 7}) + require.Equal(t, 3, tail.overlay.HeldSlots()) + require.Nil(t, promoter.poll()) + + unblock() + result := promoter.drain() + require.NotNil(t, result) + require.NoError(t, result.err) + require.Nil(t, result.job.transactionStatusSnapshot) + tail.applyFoldJob(result.job) + require.Equal(t, 1, tail.overlay.HeldSlots()) +} + +func TestCheckpointCaptureFailureOrdering(t *testing.T) { + cases := []struct { + name string + capture func(uint64) (TransactionStatusSnapshot, error) + want string + buildFails bool + }{ + {"nil capture", func(uint64) (TransactionStatusSnapshot, error) { return nil, nil }, "capture is nil", true}, + {"capture error", func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("capture failed") }, "capture failed", true}, + {"encode error", func(uint64) (TransactionStatusSnapshot, error) { + return testCheckpointEncoder(func() ([]byte, error) { return nil, errors.New("encode failed") }), nil + }, "encode failed", false}, + {"empty encoding", func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes(nil), nil }, "snapshot is empty", false}, + } + for _, tc := range cases { + for _, forced := range []bool{false, true} { + name := tc.name + "/async" + if forced { + name = tc.name + "/forced" + } + t.Run(name, func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 5, 6) + installed := false + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: tc.capture, + Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { + installed = true + return nil, errors.New("unexpected install") + }, + })) + if forced { + through, _, err := tail.flush(6) + require.ErrorContains(t, err, tc.want) + require.Zero(t, through) + } else { + job, err := tail.buildFoldJob(6, false) + if tc.buildFails { + require.ErrorContains(t, err, tc.want) + require.Nil(t, job) + } else { + require.NoError(t, err) + require.ErrorContains(t, runFoldJob(fc, job), tc.want) + require.Nil(t, job.transactionStatusSnapshot, "failed result retained its captured deltas") + } + } + require.False(t, installed) + require.Empty(t, fc.throughs) + require.Equal(t, 2, tail.overlay.HeldSlots()) + }) + } + } +} + +func TestFoldCheckpointKeepsCapturedRootAfterLiveCacheAdvances(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + fc := &fakeCommitter{durable: accounts.NewMemAccounts()} + tail := asyncTestTail(fc, 319, 320, 321) + rootDir := t.TempDir() + require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ + Capture: c.CaptureSnapshotThrough, + Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { + return PrepareTransactionStatusCheckpoint(rootDir, through, payload) + }, + })) + job, err := tail.buildFoldJob(321, false) + require.NoError(t, err) + for slot := uint64(351); slot <= 660; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + c.Root(slot - 20) + } + require.NoError(t, runFoldJob(fc, job)) + require.Nil(t, job.transactionStatusSnapshot, "completed result retained captured deltas") + require.Positive(t, job.checkpointCaptureTime) + require.Positive(t, job.checkpointEncodeTime) + require.Equal(t, len(want), job.checkpointBytes) + var manifest state.ResumeContext + require.NoError(t, json.Unmarshal(fc.ctxs[320], &manifest)) + require.Equal(t, uint64(320), manifest.TransactionStatusCheckpoint.Root) + got, err := ReadTransactionStatusCheckpoint(rootDir, manifest.TransactionStatusCheckpoint) + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) +} diff --git a/pkg/replay/async_promotion_test.go b/pkg/replay/async_promotion_test.go index 49d47a49e..eff0de206 100644 --- a/pkg/replay/async_promotion_test.go +++ b/pkg/replay/async_promotion_test.go @@ -22,7 +22,7 @@ type slowCommitter struct { delay time.Duration } -func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { +func TestFoldJobCapturesStatusOnLoopAndReferenceRidesManifest(t *testing.T) { rootDir := t.TempDir() fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6, 7) @@ -32,10 +32,10 @@ func TestFoldJobSnapshotsStatusOnLoopAndReferenceRidesManifest(t *testing.T) { installCalled := false afterCommitCalled := false hooks := TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(6), through) snapshotCalled = true - return scratch, nil + return testCheckpointBytes(scratch), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { require.True(t, snapshotCalled, "worker install ran before loop snapshot") @@ -83,7 +83,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return nil, errors.New("snapshot boom") }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return nil, errors.New("snapshot boom") }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { t.Fatal("install must not run") return nil, nil @@ -101,7 +101,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { fc := &fakeCommitter{durable: accounts.NewMemAccounts()} tail := asyncTestTail(fc, 5, 6) require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(uint64, []byte) (*state.TransactionStatusCheckpointRef, error) { return nil, errors.New("fsync boom") }, @@ -120,7 +120,7 @@ func TestFoldJobCheckpointFailuresCannotReachCommitBatch(t *testing.T) { tail := asyncTestTail(fc, 5, 6) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(uint64) ([]byte, error) { return []byte("captured"), nil }, + Capture: func(uint64) (TransactionStatusSnapshot, error) { return testCheckpointBytes([]byte("captured")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) }, @@ -144,9 +144,9 @@ func TestForcedFoldCarriesStatusCheckpointReference(t *testing.T) { tail := asyncTestTail(fc, 5) afterCommitCalled := false require.NoError(t, tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - Snapshot: func(through uint64) ([]byte, error) { + Capture: func(through uint64) (TransactionStatusSnapshot, error) { require.Equal(t, uint64(5), through) - return []byte("forced-partial-status"), nil + return testCheckpointBytes([]byte("forced-partial-status")), nil }, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(rootDir, through, payload) @@ -175,7 +175,7 @@ func TestCheckpointAfterCommitRequiresDurabilityHooks(t *testing.T) { err := tail.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ AfterCommit: func(*state.TransactionStatusCheckpointRef) error { return nil }, }) - require.ErrorContains(t, err, "requires Snapshot and Install") + require.ErrorContains(t, err, "requires Capture and Install") } func (c *slowCommitter) CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) { diff --git a/pkg/replay/block.go b/pkg/replay/block.go index c45c6cb9a..d28f3b8ca 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1987,9 +1987,9 @@ func ReplayBlocks( checkpointAfterCommit = consensusOpts.TransactionStatusCheckpointAfterCommit } if hookErr := unrootedTailState.SetTransactionStatusCheckpointHooks(TransactionStatusCheckpointHooks{ - // Snapshot runs here on the replay loop during fold-job construction; - // only its immutable bytes cross to the async worker. - Snapshot: transactionStatuses.SnapshotThrough, + // Pin the exact immutable view on replay. Sorting and encoding run + // on the existing fold worker, after releasing the live cache lock. + Capture: transactionStatuses.CaptureSnapshotThrough, Install: func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) { return PrepareTransactionStatusCheckpoint(acctsDbPath, through, payload) }, diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index c750009ea..752e0cddc 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -27,14 +27,13 @@ type batchCommitter interface { CommitBatch(deltas []accounts.SlotDelta, throughSlot uint64, bankhashes map[uint64][32]byte, resumeCtx []byte) (accountsdb.BatchCommitResult, error) } -// TransactionStatusCheckpointHooks deliberately split status-cache capture -// from sidecar I/O. Snapshot runs on the replay loop while its mutable cache is -// coherent; Install runs on the fold worker using only those immutable bytes. -// This makes it impossible for the async worker to traverse concurrently -// changing replay lineage. The later AccountsDB manifest remains the selector. +// TransactionStatusCheckpointHooks split immutable status capture from encoding +// and sidecar I/O. Capture runs on replay; the fold worker serializes the captured +// view and then calls Install. Neither worker operation revisits live lineage. +// The later AccountsDB manifest remains the durable checkpoint selector. type TransactionStatusCheckpointHooks struct { - Snapshot func(through uint64) ([]byte, error) - Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) + Capture func(through uint64) (TransactionStatusSnapshot, error) + Install func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) // AfterCommit is an advisory retention hook. It runs only after CommitBatch // has durably selected the manifest carrying selected. Its error is logged // and ignored: once CommitBatch succeeds, the fold must remain successful. @@ -378,7 +377,10 @@ type foldJob struct { ctx *state.ResumeContext ctxJSON []byte stakeIdxDir string - transactionStatusCheckpointPayload []byte + transactionStatusSnapshot TransactionStatusSnapshot + checkpointCaptureTime time.Duration + checkpointEncodeTime time.Duration + checkpointBytes int installTransactionStatusCheckpoint func(through uint64, payload []byte) (*state.TransactionStatusCheckpointRef, error) afterTransactionStatusCheckpointCommit func(selected *state.TransactionStatusCheckpointRef) error } @@ -415,18 +417,18 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. return nil, fmt.Errorf("fold chunk through slot %d: no resume context recorded for chunk-top slot", through) } ctx = cloneResumeContextForFold(ctx) - var checkpointPayload []byte - if hooks.Snapshot != nil { - checkpointPayload, err = hooks.Snapshot(through) + var snapshot TransactionStatusSnapshot + var captureTime time.Duration + if hooks.Capture != nil { + start := time.Now() + snapshot, err = hooks.Capture(through) + captureTime = time.Since(start) if err != nil { - return nil, fmt.Errorf("fold chunk through slot %d: snapshot transaction status checkpoint: %w", through, err) + return nil, fmt.Errorf("fold chunk through slot %d: capture transaction status checkpoint: %w", through, err) } - if len(checkpointPayload) == 0 { - return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint snapshot is empty", through) + if snapshot == nil { + return nil, fmt.Errorf("fold chunk through slot %d: transaction status checkpoint capture is nil", through) } - // The worker owns this immutable copy. Even a future Snapshot - // implementation that reuses a scratch buffer cannot race it. - checkpointPayload = append([]byte(nil), checkpointPayload...) } bankhashes := make(map[uint64][32]byte, len(chunk)) for _, sd := range chunk { @@ -440,7 +442,8 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. bankhashes: bankhashes, ctx: ctx, stakeIdxDir: t.stakeIdxDir, - transactionStatusCheckpointPayload: checkpointPayload, + transactionStatusSnapshot: snapshot, + checkpointCaptureTime: captureTime, installTransactionStatusCheckpoint: hooks.Install, afterTransactionStatusCheckpointCommit: hooks.AfterCommit, }, nil @@ -450,12 +453,26 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. // state). Stake-index entries flush (fsync'd) BEFORE the batch commit — see // promoteRootedBatched for why that order is a correctness requirement. func runFoldJob(committer batchCommitter, job *foldJob) error { - if job == nil || job.ctx == nil { + if job == nil { + return errors.New("fold job has no resume context") + } + // Failed folds are rebuilt from the retained tail. Neither a failed result + // nor a completed-but-unapplied job should keep checkpoint deltas alive. + defer func() { job.transactionStatusSnapshot = nil }() + if job.ctx == nil { return errors.New("fold job has no resume context") } var selectedCheckpoint *state.TransactionStatusCheckpointRef if job.installTransactionStatusCheckpoint != nil { - ref, err := job.installTransactionStatusCheckpoint(job.through, job.transactionStatusCheckpointPayload) + start := time.Now() + payload, err := encodeTransactionStatusCheckpoint(job.transactionStatusSnapshot) + job.checkpointEncodeTime = time.Since(start) + job.transactionStatusSnapshot = nil + if err != nil { + return fmt.Errorf("fold chunk through slot %d: encode transaction status checkpoint: %w", job.through, err) + } + job.checkpointBytes = len(payload) + ref, err := job.installTransactionStatusCheckpoint(job.through, payload) if err != nil { return fmt.Errorf("fold chunk through slot %d: prepare transaction status checkpoint: %w", job.through, err) } @@ -539,7 +556,9 @@ func (p *asyncPromoter) run() { start := time.Now() err := runFoldJob(p.committer, job) if err == nil { - mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s", len(job.chunk), job.through, time.Since(start).Round(time.Millisecond)) + mlog.Log.FileOnlyf("async fold: committed %d slots through %d in %s checkpoint_capture=%s checkpoint_encode=%s checkpoint_bytes=%d", + len(job.chunk), job.through, time.Since(start).Round(time.Millisecond), + job.checkpointCaptureTime, job.checkpointEncodeTime, job.checkpointBytes) } p.results <- foldResult{job: job, err: err} } @@ -713,14 +732,15 @@ func promoteRootedBatched( } ctx = cloneResumeContextForFold(ctx) var selectedCheckpoint *state.TransactionStatusCheckpointRef - if hooks.Snapshot != nil { - payload, serr := hooks.Snapshot(chunkThrough) + if hooks.Capture != nil { + snapshot, serr := hooks.Capture(chunkThrough) if serr != nil { - err = fmt.Errorf("promote chunk through slot %d: snapshot transaction status checkpoint: %w", chunkThrough, serr) + err = fmt.Errorf("promote chunk through slot %d: capture transaction status checkpoint: %w", chunkThrough, serr) break } - if len(payload) == 0 { - err = fmt.Errorf("promote chunk through slot %d: transaction status checkpoint snapshot is empty", chunkThrough) + payload, serr := encodeTransactionStatusCheckpoint(snapshot) + if serr != nil { + err = fmt.Errorf("promote chunk through slot %d: encode transaction status checkpoint: %w", chunkThrough, serr) break } ref, perr := hooks.Install(chunkThrough, payload) @@ -792,15 +812,29 @@ func resolveTransactionStatusCheckpointHooks(configured TransactionStatusCheckpo } func validateTransactionStatusCheckpointHooks(hooks TransactionStatusCheckpointHooks) error { - if (hooks.Snapshot == nil) != (hooks.Install == nil) { - return errors.New("transaction status checkpoint Snapshot and Install hooks must either both be set or both be nil") + if (hooks.Capture == nil) != (hooks.Install == nil) { + return errors.New("transaction status checkpoint Capture and Install hooks must either both be set or both be nil") } if hooks.AfterCommit != nil && hooks.Install == nil { - return errors.New("transaction status checkpoint AfterCommit hook requires Snapshot and Install hooks") + return errors.New("transaction status checkpoint AfterCommit hook requires Capture and Install hooks") } return nil } +func encodeTransactionStatusCheckpoint(snapshot TransactionStatusSnapshot) ([]byte, error) { + if snapshot == nil { + return nil, errors.New("transaction status checkpoint capture is nil") + } + payload, err := snapshot.MarshalBinary() + if err != nil { + return nil, err + } + if len(payload) == 0 { + return nil, errors.New("transaction status checkpoint snapshot is empty") + } + return payload, nil +} + func cloneResumeContextForFold(ctx *state.ResumeContext) *state.ResumeContext { if ctx == nil { return nil diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 8320df056..03437ba02 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -681,10 +681,32 @@ func (c *TransactionStatusCache) Root(through uint64) bool { return !wasComplete && c.coverageComplete } -// SnapshotThrough serializes only the rooted lineage needed at through. It is -// called while constructing a fold job, so the blob rides in that exact durable -// manifest without being copied into every speculative ResumeContext. -func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { +// TransactionStatusSnapshot pins an immutable checkpoint view. MarshalBinary +// must use only captured data, without locking or revisiting the live cache, +// and return an owned payload. It can run on the checkpoint worker while replay +// commits, roots, or unwinds its current lineage. +type TransactionStatusSnapshot interface { + MarshalBinary() ([]byte, error) +} + +type transactionStatusSnapshot struct { + nodes []*transactionStatusNode + rootedSinceSeed uint16 + complete bool + coverageFromGenesis bool +} + +func (s *transactionStatusSnapshot) MarshalBinary() ([]byte, error) { + if s == nil { + return nil, nil + } + return marshalTransactionStatusNodes(s.nodes, s.rootedSinceSeed, s.complete, s.coverageFromGenesis) +} + +// CaptureSnapshotThrough selects the exact checkpoint lineage and coverage on +// replay, but leaves transaction-key sorting and serialization to the worker. +// Published deltas are immutable; only small node headers are copied here. +func (c *TransactionStatusCache) CaptureSnapshotThrough(through uint64) (TransactionStatusSnapshot, error) { if c == nil { return nil, nil } @@ -700,7 +722,29 @@ func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) if rootedSinceSeed > maxTransactionStatusRoots { rootedSinceSeed = maxTransactionStatusRoots } - return marshalTransactionStatusNodes(nodes, uint16(rootedSinceSeed), complete, c.coverageFromGenesis) + owned := make([]transactionStatusNode, len(nodes)) + pinned := make([]*transactionStatusNode, len(nodes)) + for i, node := range nodes { + owned[i] = *node + // The encoder consumes only these node deltas. Do not keep the old + // parent chain, which could retain roots excluded from this snapshot. + owned[i].parent = nil + pinned[i] = &owned[i] + } + return &transactionStatusSnapshot{ + nodes: pinned, rootedSinceSeed: uint16(rootedSinceSeed), complete: complete, + coverageFromGenesis: c.coverageFromGenesis, + }, nil +} + +// SnapshotThrough is the synchronous convenience API. Serialization still +// happens after releasing the cache lock; normal folds use CaptureSnapshotThrough. +func (c *TransactionStatusCache) SnapshotThrough(through uint64) ([]byte, error) { + snapshot, err := c.CaptureSnapshotThrough(through) + if err != nil || snapshot == nil { + return nil, err + } + return snapshot.MarshalBinary() } func (c *TransactionStatusCache) processedSlotLocked(blockhash solana.Hash, key transactionStatusKey) uint64 { diff --git a/pkg/replay/transaction_status_capture_bench_test.go b/pkg/replay/transaction_status_capture_bench_test.go new file mode 100644 index 000000000..64bf7a7eb --- /dev/null +++ b/pkg/replay/transaction_status_capture_bench_test.go @@ -0,0 +1,66 @@ +package replay + +import ( + "crypto/sha256" + "encoding/binary" + "testing" + + "github.com/gagliardetto/solana-go" +) + +var checkpointBenchmarkPayload []byte +var checkpointBenchmarkCapture TransactionStatusSnapshot + +func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { + // A private, not-yet-published fixture with the same complete 300-root + // metadata as an imported cache. 1.5 million keys encode to roughly 30 MB. + c := newTransactionStatusCache(true) + c.coverageFromGenesis = false + c.rootedSinceSeed = maxTransactionStatusRoots + c.rootedThrough = maxTransactionStatusRoots + for slot := uint64(1); slot <= maxTransactionStatusRoots; slot++ { + keys := make(map[transactionStatusKey]struct{}, 5000) + var seed [16]byte + binary.LittleEndian.PutUint64(seed[:8], slot) + for i := uint64(0); i < 5000; i++ { + binary.LittleEndian.PutUint64(seed[8:], i) + hash := sha256.Sum256(seed[:]) + var key transactionStatusKey + copy(key[:], hash[:]) + keys[key] = struct{}{} + } + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: keys}}} + } + view, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + b.Run("SynchronousBaseline", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = legacyStatusSnapshotForTest(c, maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("CaptureOnReplay", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkCapture, err = c.CaptureSnapshotThrough(maxTransactionStatusRoots) + if err != nil { + b.Fatal(err) + } + } + }) + b.Run("EncodeOnWorker", func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + checkpointBenchmarkPayload, err = view.MarshalBinary() + if err != nil { + b.Fatal(err) + } + } + }) +} diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go new file mode 100644 index 000000000..755fa9dfd --- /dev/null +++ b/pkg/replay/transaction_status_capture_test.go @@ -0,0 +1,162 @@ +package replay + +import ( + "encoding/binary" + "sync" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// Keep the pre-split selection/metadata calculation as a differential oracle. +// The wire encoder itself did not change. +func legacyStatusSnapshotForTest(c *TransactionStatusCache, through uint64) ([]byte, error) { + c.mu.RLock() + defer c.mu.RUnlock() + nodes := c.nodesThroughLocked(through) + if len(nodes) > maxTransactionStatusRoots { + nodes = nodes[len(nodes)-maxTransactionStatusRoots:] + } + rooted := uint32(c.rootedSinceSeed) + uint32(c.countNodesBetweenLocked(c.rootedThrough, through)) + complete := c.coverageComplete || rooted >= maxTransactionStatusRoots + if rooted > maxTransactionStatusRoots { + rooted = maxTransactionStatusRoots + } + return marshalTransactionStatusNodes(nodes, uint16(rooted), complete, c.coverageFromGenesis) +} + +func importedStatusCacheForTest(t *testing.T) *TransactionStatusCache { + t.Helper() + roots := make([]txstatus.SnapshotSlotDelta, maxTransactionStatusRoots) + for i := range roots { + roots[i] = txstatus.SnapshotSlotDelta{Slot: uint64(i + 1), IsRoot: true} + } + c, err := NewTransactionStatusCacheFromAgaveSnapshot(roots, maxTransactionStatusRoots) + require.NoError(t, err) + return c +} + +func captureTestBlock(slot uint64, branch byte) *b.Block { + tx := statusCacheTestTransaction(1, 2, branch) + data := make([]byte, 9) + binary.LittleEndian.PutUint64(data, slot) + data[8] = branch + tx.Message.Instructions[0].Data = data + return statusCacheTestBlock(slot, tx) +} + +func TestTransactionStatusCaptureSurvivesConcurrentPruneAndUnwind(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 350; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + want, err := legacyStatusSnapshotForTest(c, 320) + require.NoError(t, err) + captured, err := c.CaptureSnapshotThrough(320) + require.NoError(t, err) + view := captured.(*transactionStatusSnapshot) + require.Len(t, view.nodes, maxTransactionStatusRoots) + require.Equal(t, uint64(21), view.nodes[0].slot) + require.Equal(t, uint64(320), view.nodes[len(view.nodes)-1].slot) + for _, node := range view.nodes { + require.Nil(t, node.parent, "capture retained excluded ancestry") + } + + var wg sync.WaitGroup + wg.Add(1) + errs := make(chan error, 1) + go func() { + defer wg.Done() + for slot := uint64(351); slot <= 750; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 1)); err != nil { + errs <- err + return + } + c.Root(slot - 20) + } + if err := c.Unwind(741); err != nil { + errs <- err + return + } + for slot := uint64(741); slot <= 755; slot++ { + if err := c.CommitBlock(captureTestBlock(slot, 2)); err != nil { + errs <- err + return + } + } + }() + for i := 0; i < 50; i++ { + got, err := captured.MarshalBinary() + if err != nil || string(want) != string(got) { + t.Errorf("captured bytes changed during replay: %v", err) + break + } + } + wg.Wait() + close(errs) + for err := range errs { + require.NoError(t, err) + } + got, err := captured.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + require.Equal(t, uint64(320), restored.RootedThrough()) + require.True(t, restored.CoverageComplete()) + retry := statusCacheTestBlock(321, captureTestBlock(301, 1).Transactions[0]) + require.Error(t, restored.ValidateBlock(retry), "captured ancestor was forgotten") + // A bank after the capture's through-slot must not leak into recovery. + future := statusCacheTestBlock(321, captureTestBlock(350, 1).Transactions[0]) + require.NoError(t, restored.ValidateBlock(future)) +} + +func TestTransactionStatusCapturePreservesCoverageAndOwnedBytes(t *testing.T) { + for _, complete := range []bool{false, true} { + c := newTransactionStatusCache(complete) + // Exercise metadata selection without changing its pre-existing rules. + for slot := uint64(1); slot <= 310; slot++ { + c.tip = &transactionStatusNode{slot: slot, parent: c.tip, + delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: map[transactionStatusKey]struct{}{{byte(slot), byte(slot >> 8)}: {}}}}} + } + for _, through := range []uint64{0, 1, 299, 300, 310, 400} { + want, err := legacyStatusSnapshotForTest(c, through) + require.NoError(t, err) + view, err := c.CaptureSnapshotThrough(through) + require.NoError(t, err) + got, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + got[0] ^= 0xff + again, err := view.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, again, "caller mutated the captured data through encoded bytes") + } + } + var absent *TransactionStatusCache + view, err := absent.CaptureSnapshotThrough(1) + require.NoError(t, err) + require.Nil(t, view) +} + +func TestTransactionStatusCaptureEncodingDoesNotLockLiveCache(t *testing.T) { + c := importedStatusCacheForTest(t) + require.NoError(t, c.CommitBlock(captureTestBlock(301, 1))) + view, err := c.CaptureSnapshotThrough(301) + require.NoError(t, err) + c.mu.Lock() + done := make(chan error, 1) + go func() { _, err := view.MarshalBinary(); done <- err }() + select { + case err := <-done: + c.mu.Unlock() + require.NoError(t, err) + case <-time.After(2 * time.Second): + c.mu.Unlock() + t.Fatal("checkpoint encoding waited for the live cache lock") + } +} From f1a15c25631abaa7f7012b637ae781681ead524d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 14:51:49 -0500 Subject: [PATCH 003/111] Expire transaction-status groups in batches on durable promotion --- docs/transaction-status-expiry.md | 47 +++++++ pkg/replay/transaction_status_cache.go | 63 ++++++++- pkg/replay/transaction_status_expiry_test.go | 129 +++++++++++++++++++ 3 files changed, 236 insertions(+), 3 deletions(-) create mode 100644 docs/transaction-status-expiry.md create mode 100644 pkg/replay/transaction_status_expiry_test.go diff --git a/docs/transaction-status-expiry.md b/docs/transaction-status-expiry.md new file mode 100644 index 000000000..4d3f665a4 --- /dev/null +++ b/docs/transaction-status-expiry.md @@ -0,0 +1,47 @@ +# Batched transaction-status expiry + +Applying an asynchronous checkpoint still calls `TransactionStatusCache.Root` +on replay. Live Zen 5 instruction probes measured 43–101 ms inside that function. +The previous expiry path visited every key in every retired bank, even when an +entire recent-blockhash group could be discarded. + +Expiry now examines the expired and retained bank deltas by blockhash. It drops +fully expired groups directly. For a group spanning the cutoff, it either +subtracts the expired keys or rebuilds the visible reference counts from the +retained deltas, whichever requires fewer key visits. Retained unrooted banks +are included. Physical map reclamation is still Go GC work; this is not a claim +that memory reclamation costs disappear. + +The 300-root retention rule, immediate logical expiry, duplicate-key reference +counts, selected-parent validation, checkpoint format and immutable producer +views are unchanged. All index changes remain under the existing cache lock. +This does not move unsafe mutable state to another goroutine or delay expiry. +A long-lived blockhash with many transactions on both sides of the cutoff can +still require substantial per-key work. This patch reduces that work to the +smaller side; it does not give a constant-time worst-case bound. + +## Validation + +The replay race suite, replay vet and validator production build pass. New tests +compare exact visible indexes against the original per-key removal for 100 +random lineages with shared hashes, collisions and empty groups, then unwind +surviving banks. A Root integration test checks pinned producer views, +checkpoint bytes, restored duplicate detection and rooted-unwind rejection. + +M4 Pro, Go benchmark, single caller, two iterations per case. Each iteration +expires 128 banks of 33,760 unique keys (4,321,280 entries) and retains another +33,760 entries. Setup is outside the timer. The baseline invokes the original +per-key removal; the new path invokes batched expiry. These are **expiry-path** +measurements, not end-to-end Root/replay or a prediction of live FAST scores. + +| Recent-blockhash grouping | Old expiry | Batched expiry | +|---|---:|---:| +| Groups shared by four expired banks | 185–189 ms | 0.037–0.080 ms | +| One fully expired group | 604–614 ms | 0.025–0.026 ms | +| One group shared by expired and retained banks | 590 ms | 1.63–2.36 ms | + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusBatchExpiry$' -benchtime=1x -count=2`. + +Prepared on `7layer/status-expiry-performance` above the isolated Votor fix. +This source is not the exact live FEC-integrated source. No deployment or public +PR change is implied by these local results. diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 03437ba02..bf043fb17 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -883,10 +883,8 @@ func (c *TransactionStatusCache) pruneLocked(through uint64) { if drop <= 0 { return } - for _, node := range nodes[:drop] { - c.removeDeltaVisibleLocked(node.delta) - } retained := nodes[drop:] + c.expireVisibleLocked(nodes[:drop], retained) var parent *transactionStatusNode for _, old := range retained { parent = &transactionStatusNode{ @@ -897,6 +895,65 @@ func (c *TransactionStatusCache) pruneLocked(through uint64) { c.tip = parent } +// expireVisibleLocked expires a whole rooted batch. Most old blockhash groups +// have no surviving bank and can be removed without visiting their transaction +// keys. For a group crossing the boundary, update whichever side is smaller. +// Immutable node deltas (including those pinned by producer views/checkpoints) +// are never mutated. Unrooted retained banks count as survivors too. +func (c *TransactionStatusCache) expireVisibleLocked(expired, retained []*transactionStatusNode) { + type groupExpiry struct { + expiredKeys int + retainedKeys int + survivors []*transactionStatusGroup + } + groups := make(map[solana.Hash]*groupExpiry) + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if g == nil { + g = &groupExpiry{} + groups[hash] = g + } + g.expiredKeys += len(delta.keys) + } + } + for _, node := range retained { + for hash, delta := range node.delta { + if g := groups[hash]; g != nil { + g.retainedKeys += len(delta.keys) + g.survivors = append(g.survivors, delta) + } + } + } + for hash, g := range groups { + if len(g.survivors) == 0 { + delete(c.visible, hash) + } else if g.retainedKeys < g.expiredKeys { + rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex, keys: make(map[transactionStatusKey]uint16)} + for _, delta := range g.survivors { + for key := range delta.keys { + rebuilt.keys[key]++ + } + } + c.visible[hash] = rebuilt + if len(rebuilt.keys) == 0 { + delete(c.visible, hash) + } + } + } + for _, node := range expired { + for hash, delta := range node.delta { + g := groups[hash] + if len(g.survivors) == 0 || g.retainedKeys < g.expiredKeys { + continue + } + // The existing removal path preserves reference counts for keys + // occurring in more than one retained/expired bank. + c.removeDeltaVisibleLocked(transactionStatusDelta{hash: delta}) + } + } +} + func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transactionStatusKey { // Match Agave's saturating_sub(CACHED_KEY_SIZE + 1), including its // deliberate exclusion of the final possible starting offset. diff --git a/pkg/replay/transaction_status_expiry_test.go b/pkg/replay/transaction_status_expiry_test.go new file mode 100644 index 000000000..84a11cd28 --- /dev/null +++ b/pkg/replay/transaction_status_expiry_test.go @@ -0,0 +1,129 @@ +package replay + +import ( + "encoding/binary" + "fmt" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +func TestTransactionStatusBatchExpiryMatchesPerKeyRemoval(t *testing.T) { + for seed := int64(0); seed < 100; seed++ { + rng := rand.New(rand.NewSource(seed)) + fast, ref := NewTransactionStatusCache(), NewTransactionStatusCache() + var nodes []*transactionStatusNode + for slot := 0; slot < 40; slot++ { + d := make(transactionStatusDelta) + for j := 0; j < 6; j++ { + h := solana.Hash{byte(rng.Intn(12))} + g := &transactionStatusGroup{keyIndex: h[0], keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < rng.Intn(30); k++ { + g.keys[transactionStatusKey{byte(rng.Intn(40))}] = struct{}{} + } + d[h] = g + } + nodes = append(nodes, &transactionStatusNode{slot: uint64(slot), delta: d}) + require.NoError(t, fast.addDeltaVisibleLocked(d)) + require.NoError(t, ref.addDeltaVisibleLocked(d)) + } + cut := 1 + rng.Intn(len(nodes)-1) + fast.expireVisibleLocked(nodes[:cut], nodes[cut:]) + for _, n := range nodes[:cut] { + ref.removeDeltaVisibleLocked(n.delta) + } + require.Equal(t, ref.visible, fast.visible, "seed %d", seed) + for i := len(nodes) - 1; i >= cut; i-- { + fast.removeDeltaVisibleLocked(nodes[i].delta) + ref.removeDeltaVisibleLocked(nodes[i].delta) + } + require.Equal(t, ref.visible, fast.visible, "unwind seed %d", seed) + } +} + +func TestTransactionStatusBatchExpiryPinnedViewsAndSnapshot(t *testing.T) { + c := NewTransactionStatusCache() + old := statusCacheTestTransaction(1, 1, 1) + keep := statusCacheTestTransaction(2, 2, 2) + require.NoError(t, c.CommitBlock(statusCacheTestBlock(1, old))) + for slot := uint64(2); slot <= maxTransactionStatusRoots+1; slot++ { + blk := statusCacheTestBlock(slot) + if slot == maxTransactionStatusRoots+1 { + blk.Transactions = append(blk.Transactions, keep) + } + require.NoError(t, c.CommitBlock(blk)) + } + pinned := c.View() + snapshot, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + before, err := snapshot.MarshalBinary() + require.NoError(t, err) + c.coverageComplete = false // Exercise completion once 300 banks become rooted. + c.Root(maxTransactionStatusRoots + 1) + after, err := snapshot.MarshalBinary() + require.NoError(t, err) + require.Equal(t, before, after) + found, err := pinned.ContainsTransaction(old) + require.NoError(t, err) + require.True(t, found) + found, err = c.View().ContainsTransaction(old) + require.NoError(t, err) + require.False(t, found) + require.NoError(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, c.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + blob, err := c.SnapshotThrough(maxTransactionStatusRoots + 1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + require.NoError(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, old))) + require.Error(t, restored.ValidateBlock(statusCacheTestBlock(maxTransactionStatusRoots+2, keep))) + require.Error(t, c.Unwind(maxTransactionStatusRoots+1)) +} + +func BenchmarkTransactionStatusBatchExpiry(b *testing.B) { + for _, shape := range []string{"four-bank-groups", "one-expired-group", "crossing-group"} { + for _, legacy := range []bool{true, false} { + b.Run(fmt.Sprintf("%s/legacy=%t", shape, legacy), func(b *testing.B) { + for i := 0; i < b.N; i++ { + b.StopTimer() + c := NewTransactionStatusCache() + var expired, retained []*transactionStatusNode + for slot := 0; slot < 129; slot++ { + var h solana.Hash + if shape == "four-bank-groups" || (shape == "one-expired-group" && slot == 128) { + binary.LittleEndian.PutUint64(h[:], uint64(slot/4+1)) + } + g := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} + for k := 0; k < 33760; k++ { + var key transactionStatusKey + binary.LittleEndian.PutUint64(key[:], uint64(slot*33760+k)) + g.keys[key] = struct{}{} + } + n := &transactionStatusNode{delta: transactionStatusDelta{h: g}} + if err := c.addDeltaVisibleLocked(n.delta); err != nil { + b.Fatal(err) + } + if slot < 128 { + expired = append(expired, n) + } else { + retained = append(retained, n) + } + } + b.StartTimer() + if legacy { + for _, n := range expired { + c.removeDeltaVisibleLocked(n.delta) + } + } else { + c.expireVisibleLocked(expired, retained) + } + b.StopTimer() + if len(c.visible) != 1 { + b.Fatal("retained group missing") + } + } + }) + } + } +} From 852aae99c91197d2093ec1bc6778b1bd86ed5a09 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 15:05:52 -0500 Subject: [PATCH 004/111] Record native Zen 5 expiry benchmark results --- docs/transaction-status-expiry.md | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/docs/transaction-status-expiry.md b/docs/transaction-status-expiry.md index 4d3f665a4..179202360 100644 --- a/docs/transaction-status-expiry.md +++ b/docs/transaction-status-expiry.md @@ -45,3 +45,28 @@ Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusBatchExpi Prepared on `7layer/status-expiry-performance` above the isolated Votor fix. This source is not the exact live FEC-integrated source. No deployment or public PR change is implied by these local results. + +## Zen 5 validation — 20:05 UTC + +Native AMD Ryzen 7 9700X tests used an isolated copy of the preserved live +FEC-integrated source at `/srv/mithril-status-expiry-test-20260914/source`. +The original status-cache file was byte-identical to the change's parent. +Only the reviewed cache source and new tests were overlaid, with SHA256 checks. +The full replay race suite passed (2.189 s), and vet passed. No binary was deployed. + +Benchmarks ran with GOMAXPROCS=2, nice=15, one caller and three iterations per +case, while the validator and loader remained active. Setup and later GC are +excluded from the expiry timer. Each case expires 4,321,280 entries (128 banks +of 33,760) and retains 33,760 entries. These synthetic batches exceed the earlier +live stall samples and are not an end-to-end replay or FAST-score comparison. + +| Shape | Original expiry | New expiry | +|---|---:|---:| +| Four-bank blockhash groups | 306–311 ms | 0.049–0.057 ms | +| One fully expired blockhash group | 700–718 ms | 0.024–0.031 ms | +| Group crossing the retention boundary | 717–735 ms | 1.85–2.05 ms | + +At 20:05:13 UTC the enrolled validator PID 291548 was at RPC/local slot 3,708,093, +last vote 3,708,092. Loader unpaused; validator, loader, FAST and Titan services +all active. Source/implementation and deployment status remain unchanged. +See zen5-benchmark.log, zen5-race.log, zen5-vet.log and zen5-health.json. From 6e5278985d415063cdd902899333ab014e9e74e9 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 20:45:28 -0500 Subject: [PATCH 005/111] Document isolated PR validation and retain relevant benchmark evidence --- .gitattributes | 7 ++ .../pr-split-2026-09-15/status/README.md | 7 ++ .../status/status-tests.log | 1 + .../pr-split-2026-09-15/status/status-vet.log | 0 .../baseline-failure-comparison.json | 26 ++++ .../2026-09-14/checkpoint-live.json | 118 ++++++++++++++++++ .../checkpoint-native-benchmark.log | 15 +++ .../2026-09-14/zen5-benchmark.log | 24 ++++ .../status-cache/2026-09-14/zen5-race.log | 1 + .../status-cache/2026-09-14/zen5-vet.log | 0 docs/status-checkpoint-capture.md | 7 ++ 11 files changed, 206 insertions(+) create mode 100644 .gitattributes create mode 100644 docs/results/pr-split-2026-09-15/status/README.md create mode 100644 docs/results/pr-split-2026-09-15/status/status-tests.log create mode 100644 docs/results/pr-split-2026-09-15/status/status-vet.log create mode 100644 docs/results/status-cache/2026-09-14/baseline-failure-comparison.json create mode 100644 docs/results/status-cache/2026-09-14/checkpoint-live.json create mode 100644 docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log create mode 100644 docs/results/status-cache/2026-09-14/zen5-benchmark.log create mode 100644 docs/results/status-cache/2026-09-14/zen5-race.log create mode 100644 docs/results/status-cache/2026-09-14/zen5-vet.log create mode 100644 docs/status-checkpoint-capture.md diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 000000000..d64360d9a --- /dev/null +++ b/.gitattributes @@ -0,0 +1,7 @@ +# Keep raw benchmark evidence available without overwhelming the review. +# Retain raw Go CPU padding and test-framework space/tab indentation. +docs/results/**/*.json linguist-generated=true +docs/results/**/*.jsonl linguist-generated=true +docs/results/**/*.txt linguist-generated=true whitespace=-blank-at-eol +docs/results/**/*.log linguist-generated=true -whitespace +docs/results/**/*.tar.gz linguist-generated=true diff --git a/docs/results/pr-split-2026-09-15/status/README.md b/docs/results/pr-split-2026-09-15/status/README.md new file mode 100644 index 000000000..134adf562 --- /dev/null +++ b/docs/results/pr-split-2026-09-15/status/README.md @@ -0,0 +1,7 @@ +# Review branch validation, September 15 + +This PR is split from #279 plus the later working-tree improvements. The tested code commit before this documentation commit was `a111ba892fb88ea95768930f56dd2f9a5091e6ce`. Tests ran locally on Apple M4 Pro, Go1.26.4, GOMAXPROCS3 and package parallelism2. Logs beside this file are the fresh split-branch checks, not native measurements. Existing Zen5 benchmark documents retain their original baselines and scope. No validator restart or deployment occurred during this reorganization. + +Four independent branches start at current alpenglow-dev33dde405. Voting is based on the certificate-processing PR; leader packing is based on the streaming-preparation PR. Runtime changes and status-cache changes are independent. Shared CLI/configuration additions need an ordinary three-file merge reconciliation when combining leader packing and voting. A separate audit checkout reconciled these additions and matched the preserved full implementation exactly across Go sources, module files, TOML configuration and CI. + +The branch-specific race suites and vet passed. The combined audit has a separately documented pre-existing intermittent peer reconnect timeout; this is not reported as an entirely green combined race run. diff --git a/docs/results/pr-split-2026-09-15/status/status-tests.log b/docs/results/pr-split-2026-09-15/status/status-tests.log new file mode 100644 index 000000000..b6b40ef23 --- /dev/null +++ b/docs/results/pr-split-2026-09-15/status/status-tests.log @@ -0,0 +1 @@ +ok github.com/Overclock-Validator/mithril/pkg/replay 4.657s diff --git a/docs/results/pr-split-2026-09-15/status/status-vet.log b/docs/results/pr-split-2026-09-15/status/status-vet.log new file mode 100644 index 000000000..e69de29bb diff --git a/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json b/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json new file mode 100644 index 000000000..5353cc6eb --- /dev/null +++ b/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json @@ -0,0 +1,26 @@ +{ + "failed_tests": [ + "TestExecute_Tx_BpfLoader_Write_Success", + "TestExecute_Tx_BpfLoader_Write_Offset_Too_Large_Failure", + "TestExecute_Tx_BpfLoader_Write_Buffer_Authority_Didnt_Sign_Failure", + "TestExecute_Tx_BpfLoader_Write_Incorrect_Authority_Failure", + "TestExecute_Tx_BpfLoader_SetAuthority_Not_Enough_Instr_Accts_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Not_Enough_Instr_Accts_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Success", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Immutable_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Wrong_Upgrade_Authority_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Authority_Didnt_Sign_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_New_Authority_Didnt_Sign_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Uninitialized_Account_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Immutable_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Authority_Didnt_Sign_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_New_Authority_Didnt_Sign_Failure", + "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Wrong_Authority_Failure", + "TestExecute_Tx_BpfLoader_Close_Buffer_Not_Enough_Accounts", + "TestExecute_Tx_BpfLoader_Close_ProgramData_Success" + ], + "same_failed_test_names": true, + "same_panic_function": "UpgradeableLoaderClose", + "same_failed_test_names_on_native_base": true +} diff --git a/docs/results/status-cache/2026-09-14/checkpoint-live.json b/docs/results/status-cache/2026-09-14/checkpoint-live.json new file mode 100644 index 000000000..0072e9189 --- /dev/null +++ b/docs/results/status-cache/2026-09-14/checkpoint-live.json @@ -0,0 +1,118 @@ +{ + "utc": "2026-09-14T03:34:45Z", + "count": 10, + "log_bytes": 1607896, + "rows": [ + { + "slots": 128, + "through": 3427284, + "worker_ms": 331.0, + "capture_us": 34.174, + "encode_ms": 236.108854, + "bytes": 31488298, + "line": "(+ 28s) async fold: committed 128 slots through 3427284 in 331ms checkpoint_capture=34.174\u00b5s checkpoint_encode=236.108854ms checkpoint_bytes=31488298" + }, + { + "slots": 128, + "through": 3427428, + "worker_ms": 264.0, + "capture_us": 30.366, + "encode_ms": 182.148884, + "bytes": 25219543, + "line": "(+ 59s) async fold: committed 128 slots through 3427428 in 264ms checkpoint_capture=30.366\u00b5s checkpoint_encode=182.148884ms checkpoint_bytes=25219543" + }, + { + "slots": 128, + "through": 3427569, + "worker_ms": 318.0, + "capture_us": 30.417, + "encode_ms": 225.601155, + "bytes": 29886172, + "line": "(+ 1m29s) async fold: committed 128 slots through 3427569 in 318ms checkpoint_capture=30.417\u00b5s checkpoint_encode=225.601155ms checkpoint_bytes=29886172" + }, + { + "slots": 128, + "through": 3427701, + "worker_ms": 258.0, + "capture_us": 29.495, + "encode_ms": 179.07318000000004, + "bytes": 24530250, + "line": "(+ 1m58s) async fold: committed 128 slots through 3427701 in 258ms checkpoint_capture=29.495\u00b5s checkpoint_encode=179.07318ms checkpoint_bytes=24530250" + }, + { + "slots": 128, + "through": 3427841, + "worker_ms": 305.0, + "capture_us": 30.888, + "encode_ms": 216.143473, + "bytes": 29469092, + "line": "(+ 2m28s) async fold: committed 128 slots through 3427841 in 305ms checkpoint_capture=30.888\u00b5s checkpoint_encode=216.143473ms checkpoint_bytes=29469092" + }, + { + "slots": 128, + "through": 3427989, + "worker_ms": 286.0, + "capture_us": 28.694, + "encode_ms": 201.186937, + "bytes": 27782921, + "line": "(+ 2m59s) async fold: committed 128 slots through 3427989 in 286ms checkpoint_capture=28.694\u00b5s checkpoint_encode=201.186937ms checkpoint_bytes=27782921" + }, + { + "slots": 128, + "through": 3428133, + "worker_ms": 376.0, + "capture_us": 28.874, + "encode_ms": 278.613922, + "bytes": 35692638, + "line": "(+ 3m29s) async fold: committed 128 slots through 3428133 in 376ms checkpoint_capture=28.874\u00b5s checkpoint_encode=278.613922ms checkpoint_bytes=35692638" + }, + { + "slots": 128, + "through": 3428269, + "worker_ms": 431.0, + "capture_us": 32.671, + "encode_ms": 329.865906, + "bytes": 37615609, + "line": "(+ 3m58s) async fold: committed 128 slots through 3428269 in 431ms checkpoint_capture=32.671\u00b5s checkpoint_encode=329.865906ms checkpoint_bytes=37615609" + }, + { + "slots": 128, + "through": 3428409, + "worker_ms": 403.0, + "capture_us": 31.108, + "encode_ms": 301.569555, + "bytes": 38403988, + "line": "(+ 4m28s) async fold: committed 128 slots through 3428409 in 403ms checkpoint_capture=31.108\u00b5s checkpoint_encode=301.569555ms checkpoint_bytes=38403988" + }, + { + "slots": 128, + "through": 3428549, + "worker_ms": 477.0, + "capture_us": 34.464, + "encode_ms": 366.581817, + "bytes": 43443732, + "line": "(+ 4m58s) async fold: committed 128 slots through 3428549 in 477ms checkpoint_capture=34.464\u00b5s checkpoint_encode=366.581817ms checkpoint_bytes=43443732" + } + ], + "capture_us": { + "min": 28.694, + "median": 30.652500000000003, + "max": 34.464 + }, + "encode_ms": { + "min": 179.07318000000004, + "median": 230.8550045, + "max": 366.581817 + }, + "bytes": { + "min": 24530250, + "median": 30687235.0, + "max": 43443732 + }, + "worker_ms": { + "min": 258.0, + "median": 324.5, + "max": 477.0 + }, + "error_lines": [] +} diff --git a/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log b/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log new file mode 100644 index 000000000..33f4c2cd6 --- /dev/null +++ b/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log @@ -0,0 +1,15 @@ +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 206554597 ns/op 99120032 B/op 2731 allocs/op +BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 207352232 ns/op 99120072 B/op 2732 allocs/op +BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 202442401 ns/op 99120048 B/op 2731 allocs/op +BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 47412 5936 ns/op 35168 B/op 11 allocs/op +BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 39506 6057 ns/op 35168 B/op 11 allocs/op +BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 37850 5922 ns/op 35168 B/op 11 allocs/op +BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 2 202522908 ns/op 99108084 B/op 2723 allocs/op +BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 2 201841085 ns/op 99108076 B/op 2723 allocs/op +BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 1 200834149 ns/op 99108104 B/op 2724 allocs/op +PASS +ok github.com/Overclock-Validator/mithril/pkg/replay 3.133s diff --git a/docs/results/status-cache/2026-09-14/zen5-benchmark.log b/docs/results/status-cache/2026-09-14/zen5-benchmark.log new file mode 100644 index 000000000..0229c19af --- /dev/null +++ b/docs/results/status-cache/2026-09-14/zen5-benchmark.log @@ -0,0 +1,24 @@ +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 308302685 ns/op +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 311104834 ns/op +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 306059722 ns/op +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 56685 ns/op +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 49743 ns/op +BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 48581 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 700015562 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 718445604 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 716785553 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 30758 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 26119 ns/op +BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 24436 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 734956959 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 722327226 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 717188987 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 2045462 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 1845859 ns/op +BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 1983167 ns/op +PASS +ok github.com/Overclock-Validator/mithril/pkg/replay 22.725s diff --git a/docs/results/status-cache/2026-09-14/zen5-race.log b/docs/results/status-cache/2026-09-14/zen5-race.log new file mode 100644 index 000000000..100c86a11 --- /dev/null +++ b/docs/results/status-cache/2026-09-14/zen5-race.log @@ -0,0 +1 @@ +ok github.com/Overclock-Validator/mithril/pkg/replay 2.189s diff --git a/docs/results/status-cache/2026-09-14/zen5-vet.log b/docs/results/status-cache/2026-09-14/zen5-vet.log new file mode 100644 index 000000000..e69de29bb diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md new file mode 100644 index 000000000..e47e67e2b --- /dev/null +++ b/docs/status-checkpoint-capture.md @@ -0,0 +1,7 @@ +# Transaction-status checkpoint capture + +Replay used to encode and sort the full status checkpoint while preparing a promotion. Capture now pins immutable lineage and coverage metadata; the existing promotion worker encodes the same checkpoint later. Capture occurs before pruning and publication ordering stays unchanged. Pinned snapshots retain their nodes across pruning/unwind. + +The native Zen5 benchmark captured roughly30MB of status data: original202–207ms versus5.92–6.06microseconds on replay. Encoding moved to the worker and was not eliminated. Ten live captures took29–34microseconds; worker encoding179–367ms. These are historical stage measurements, not a promise of total validator speedup; see docs/results/status-cache/2026-09-14. + +This PR also includes batched expiry; see transaction-status-expiry.md. The proposed transaction-status publication optimization has not been implemented or included. Fresh standalone replay race and vet checks are under docs/results/pr-split-2026-09-15/status. From 6c31ff511693b542818669d9dcc1ae6d8cd79f0a Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:01:40 -0500 Subject: [PATCH 006/111] Prepare transaction-status deltas during block execution --- .../2026-09-15/native-benchmark-summary.json | 82 ++++++++++ .../native-overlap-final-summary.json | 104 ++++++++++++ .../2026-09-15/native-overlap-summary.json | 104 ++++++++++++ .../2026-09-15/native-window.json | 4 + .../2026-09-15/post-test-health.json | 1 + .../2026-09-15/tested-source.json | 16 ++ docs/transaction-status-publication.md | 52 ++++++ pkg/metrics/metrics.go | 24 +-- pkg/replay/block.go | 14 +- pkg/replay/transaction_status_cache.go | 38 +++-- ...ansaction_status_overlap_benchmark_test.go | 110 +++++++++++++ .../transaction_status_plan_binding_test.go | 3 +- .../transaction_status_prepared_test.go | 9 +- pkg/replay/transaction_status_publication.go | 83 ++++++++++ ...ction_status_publication_benchmark_test.go | 153 ++++++++++++++++++ .../transaction_status_publication_test.go | 133 +++++++++++++++ 16 files changed, 900 insertions(+), 30 deletions(-) create mode 100644 docs/results/status-publication/2026-09-15/native-benchmark-summary.json create mode 100644 docs/results/status-publication/2026-09-15/native-overlap-final-summary.json create mode 100644 docs/results/status-publication/2026-09-15/native-overlap-summary.json create mode 100644 docs/results/status-publication/2026-09-15/native-window.json create mode 100644 docs/results/status-publication/2026-09-15/post-test-health.json create mode 100644 docs/results/status-publication/2026-09-15/tested-source.json create mode 100644 docs/transaction-status-publication.md create mode 100644 pkg/replay/transaction_status_overlap_benchmark_test.go create mode 100644 pkg/replay/transaction_status_publication.go create mode 100644 pkg/replay/transaction_status_publication_benchmark_test.go create mode 100644 pkg/replay/transaction_status_publication_test.go diff --git a/docs/results/status-publication/2026-09-15/native-benchmark-summary.json b/docs/results/status-publication/2026-09-15/native-benchmark-summary.json new file mode 100644 index 000000000..3936921d1 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-benchmark-summary.json @@ -0,0 +1,82 @@ +{ + "BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2": { + "ns/op": 4990269.0, + "B/op": 6302003.0, + "allocs/op": 555.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2": { + "ns/op": 3331600.0, + "B/op": 3151475.0, + "allocs/op": 265.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2": { + "ns/op": 3393252.0, + "B/op": 3151718.0, + "allocs/op": 269.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2": { + "ns/op": 1587430.0, + "B/op": 1575587.0, + "allocs/op": 132.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2": { + "ns/op": 4991180.0, + "B/op": 3466193.0, + "allocs/op": 304.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2": { + "ns/op": 4657032.0, + "B/op": 1891049.0, + "allocs/op": 159.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2": { + "ns/op": 4434153.0, + "B/op": 1891281.0, + "allocs/op": 163.0 + }, + "BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2": { + "ns/op": 2756954.0, + "B/op": 315161.0, + "allocs/op": 26.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2": { + "ns/op": 4624988.0, + "B/op": 6301427.0, + "allocs/op": 659.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2": { + "ns/op": 3743889.0, + "B/op": 3151859.0, + "allocs/op": 283.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2": { + "ns/op": 5915974.0, + "B/op": 3152091.0, + "allocs/op": 287.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2": { + "ns/op": 2205044.0, + "B/op": 1575779.0, + "allocs/op": 141.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2": { + "ns/op": 5224367.0, + "B/op": 3465532.0, + "allocs/op": 357.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2": { + "ns/op": 4644107.0, + "B/op": 1891228.0, + "allocs/op": 169.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2": { + "ns/op": 4949162.0, + "B/op": 1891460.0, + "allocs/op": 173.0 + }, + "BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2": { + "ns/op": 2858236.0, + "B/op": 315148.0, + "allocs/op": 27.0 + } +} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json b/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json new file mode 100644 index 000000000..7f2b767e3 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json @@ -0,0 +1,104 @@ +{ + "BenchmarkTransactionStatusExecutionOverlap/legacy": { + "ns/op": 23370663.0, + "commit-with-wait-ns/op": 6062517.0, + "execution-ns/op": 17289566.0, + "B/op": 23276118.0, + "allocs/op": 242222.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/legacy-2": { + "ns/op": 20677847.0, + "commit-with-wait-ns/op": 5394296.0, + "execution-ns/op": 15069600.0, + "B/op": 23276627.0, + "allocs/op": 242225.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/sized": { + "ns/op": 21296975.0, + "commit-with-wait-ns/op": 4299800.0, + "execution-ns/op": 17436468.0, + "B/op": 20125587.0, + "allocs/op": 241932.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/sized-2": { + "ns/op": 18573776.0, + "commit-with-wait-ns/op": 4299496.0, + "execution-ns/op": 14369209.0, + "B/op": 20126012.0, + "allocs/op": 241934.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/overlap": { + "ns/op": 22188363.0, + "commit-with-wait-ns/op": 4616295.0, + "execution-ns/op": 17694970.0, + "B/op": 20125587.0, + "allocs/op": 241932.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/overlap-2": { + "ns/op": 17090947.0, + "commit-with-wait-ns/op": 2119131.0, + "execution-ns/op": 14966507.0, + "B/op": 20126212.0, + "allocs/op": 241938.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/legacy": { + "ns/op": 95.12, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2": { + "ns/op": 74.71, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total": { + "ns/op": 119.4, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2": { + "ns/op": 100.0, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/legacy": { + "ns/op": 593.0, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2": { + "ns/op": 466.8, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total": { + "ns/op": 708.3, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2": { + "ns/op": 609.1, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/legacy": { + "ns/op": 6085.0, + "B/op": 6320.0, + "allocs/op": 23.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2": { + "ns/op": 5145.0, + "B/op": 6320.0, + "allocs/op": 23.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total": { + "ns/op": 4456.0, + "B/op": 3616.0, + "allocs/op": 13.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2": { + "ns/op": 3770.0, + "B/op": 3616.0, + "allocs/op": 13.0 + } +} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-overlap-summary.json b/docs/results/status-publication/2026-09-15/native-overlap-summary.json new file mode 100644 index 000000000..efa0017aa --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-overlap-summary.json @@ -0,0 +1,104 @@ +{ + "BenchmarkTransactionStatusExecutionOverlap/legacy": { + "ns/op": 22833095.0, + "commit-with-wait-ns/op": 5848843.0, + "execution-ns/op": 16779392.0, + "B/op": 23276116.0, + "allocs/op": 242222.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/legacy-2": { + "ns/op": 19163664.0, + "commit-with-wait-ns/op": 5114189.0, + "execution-ns/op": 14034884.0, + "B/op": 23276650.0, + "allocs/op": 242225.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/sized": { + "ns/op": 21164557.0, + "commit-with-wait-ns/op": 4745710.0, + "execution-ns/op": 17057597.0, + "B/op": 20125556.0, + "allocs/op": 241932.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/sized-2": { + "ns/op": 18233386.0, + "commit-with-wait-ns/op": 4143998.0, + "execution-ns/op": 14088971.0, + "B/op": 20126009.0, + "allocs/op": 241934.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/overlap": { + "ns/op": 21351433.0, + "commit-with-wait-ns/op": 3481211.0, + "execution-ns/op": 18286875.0, + "B/op": 20125816.0, + "allocs/op": 241936.0 + }, + "BenchmarkTransactionStatusExecutionOverlap/overlap-2": { + "ns/op": 16814463.0, + "commit-with-wait-ns/op": 2297246.0, + "execution-ns/op": 14538161.0, + "B/op": 20126252.0, + "allocs/op": 241938.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/legacy": { + "ns/op": 94.81, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2": { + "ns/op": 76.19, + "B/op": 112.0, + "allocs/op": 2.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total": { + "ns/op": 181.7, + "B/op": 264.0, + "allocs/op": 5.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2": { + "ns/op": 142.9, + "B/op": 264.0, + "allocs/op": 5.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/legacy": { + "ns/op": 810.9, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2": { + "ns/op": 633.3, + "B/op": 960.0, + "allocs/op": 9.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total": { + "ns/op": 1645.0, + "B/op": 1192.0, + "allocs/op": 13.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2": { + "ns/op": 1511.0, + "B/op": 1192.0, + "allocs/op": 13.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/legacy": { + "ns/op": 6067.0, + "B/op": 6320.0, + "allocs/op": 23.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2": { + "ns/op": 5064.0, + "B/op": 6320.0, + "allocs/op": 23.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total": { + "ns/op": 5837.0, + "B/op": 3848.0, + "allocs/op": 17.0 + }, + "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2": { + "ns/op": 5961.0, + "B/op": 3848.0, + "allocs/op": 17.0 + } +} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-window.json b/docs/results/status-publication/2026-09-15/native-window.json new file mode 100644 index 000000000..f283937ee --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-window.json @@ -0,0 +1,4 @@ +{ + "start": "2026-09-15T02:57:29.581963+00:00", + "end": "2026-09-15T02:58:07.245339+00:00" +} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/post-test-health.json b/docs/results/status-publication/2026-09-15/post-test-health.json new file mode 100644 index 000000000..df730357e --- /dev/null +++ b/docs/results/status-publication/2026-09-15/post-test-health.json @@ -0,0 +1 @@ +{"utc": "2026-09-15T02:58:13.874813+00:00", "health": {"utc": "2026-09-15T02:58:13.949166+00:00", "pid": 568418, "rpc_slot": 3824460, "local_slot": 3824460, "last_vote": 3824459, "vote_lag": 1}, "pid": "MainPID=568418", "services": ["active", "active", "active", "active"]} diff --git a/docs/results/status-publication/2026-09-15/tested-source.json b/docs/results/status-publication/2026-09-15/tested-source.json new file mode 100644 index 000000000..c4889f6a2 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/tested-source.json @@ -0,0 +1,16 @@ +{ + "base": "33dde4050d9250557583395810799aaac2f54017", + "branch_before_change": "31e0d8c0", + "files": { + "pkg/replay/transaction_status_overlap_benchmark_test.go": "6310ce381e6e88151b14a7f0e5f56e7370a8e1d94af466911dd69707379d8cdb", + "pkg/replay/transaction_status_publication.go": "a94c1227ce0b0e329dfff390c2303bb36525493e0cde3fce7d3fe177eaa1681b", + "pkg/replay/transaction_status_publication_benchmark_test.go": "0ff3a1f0c2a8d0ceafb5d7ba300c6cb5358082fd019f75a4b6fe0c35fde1575e", + "pkg/replay/transaction_status_publication_test.go": "eae775d2d10e6ca5b13b1e6f201cff15d8d0174b40ff20a69aa4279004799ca1", + "pkg/metrics/metrics.go": "17005056d872f9b8acf75fee15dc2c175bc4addad1b2dafee61445124dfbe307", + "pkg/replay/block.go": "19e7e931ec5e8aaab2e910808cb2f6e19f5721ff2ad953541f42828a88076a0d", + "pkg/replay/transaction_status_cache.go": "731febda7fdfe09b84d4281c53c3d0f4381e8f7d67d9d47f084c8a10c1523935", + "pkg/replay/transaction_status_plan_binding_test.go": "6c1fe9c92af2011402025ca96601847a1313bf2ffb564082eb06aa89088e250b", + "pkg/replay/transaction_status_prepared_test.go": "1944b443624d1075f6ea3e89630cdca42a6245a95a3116bbd47bd670311ad876" + }, + "baseline_check": "Frozen commitBlockWithPlan and addDeltaVisibleLocked exactly match alpenglow-dev at the recorded base after renaming benchmark helper methods." +} diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md new file mode 100644 index 000000000..d182bc6fb --- /dev/null +++ b/docs/transaction-status-publication.md @@ -0,0 +1,52 @@ +# Preparing transaction-status publication during execution + +Replay previously built the immutable per-bank transaction-status delta and grew the visible duplicate index only after execution and bank-state publication. In a prior live sample of 25 large blocks, TransactionStatusCommit took 7.704 ms median and 10.206 ms maximum. Those live timings motivate this change; they are not the controlled benchmark baseline below. + +Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. + +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit still checks exact block/identity binding, complete coverage, parent lineage and all ancestor duplicates under the publication lock. A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. + +TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. + +## Native benchmark + +AMD Ryzen 7 9700X (Zen 5), Go 1.26.4, GOMAXPROCS=2. Tests ran in a separate process on the validator host with Nice=15 and a 200% CPU quota; the validator and loader continued running. This is a shared-host microbenchmark, with observable timing variation. Five samples per case, ten iterations per sample; values below are medians of sample means, not per-block percentiles. + +Each block has 33,760 unique prepared message identities spread across one or four recent blockhashes. Existing-group cases seed 33,760 different ancestor transactions. Fixture creation, hashing, seeding and unwind are untimed. Existing maps retain capacity after unwind: the first timed commit's growth is amortized across the ten iterations. This does not model an index growing indefinitely across live blocks. + +The frozen baseline functions exactly match alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions use the same prepared identities, parent/duplicate checks and fixtures. + +| Recent blockhash groups | Parent has keys in these groups | Baseline commit | Sized maps, inline | Preparation + commit, no overlap | Commit after preparation | +|---|---|---:|---:|---:|---:| +| 1 | No | 4.990 ms | 3.332 ms | 3.393 ms | 1.587 ms | +| 1 | Yes | 4.991 ms | 4.657 ms | 4.434 ms | 2.757 ms | +| 4 | No | 4.625 ms | 3.744 ms | 5.916 ms | 2.205 ms | +| 4 | Yes | 5.224 ms | 4.644 ms | 4.949 ms | 2.858 ms | + +The last column deliberately excludes delta preparation: it measures the work remaining if execution hides preparation completely. It is not total replay or CPU work. Total publication allocations with new groups fell from approximately 6.30 MB to 3.15 MB per block. Existing-group allocation figures include the amortized first growth described above. + +The four-new-group total-work sample was slower. Preserve that result rather than claiming improvement in every sample. A subsequent baseline/candidate/candidate/baseline comparison of that same case, with 50 iterations per sample, measured baseline **4.400 and 4.565 ms**, candidate **2.985 and 3.131 ms**. This supports a reduction in work but does not isolate the cause of the earlier timing variation. + +## Execution contention and small blocks + +A separate controlled benchmark performs 4,096 load-and-execute calls using the existing transfer fixture while preparing 33,760 independent status keys. It does not commit transfer accounts, and its status fixture differs from the repeated transfer fixture. It tests scheduling/allocation contention, not whole-block replay or a valid block workload. + +With two Go execution threads, the final implementation measured **20.678 ms baseline**, **18.574 ms with sizing alone**, and **17.091 ms with overlap**. Execution itself measured 15.070, 14.369 and 14.967 ms respectively. Thus preparation competed with execution relative to sizing alone, but the shorter final stage outweighed that cost in this controlled workload. These are separate medians and need not add exactly. + +The initial unrestricted version showed no additional total-time benefit from overlap with GOMAXPROCS=1. Tiny-block measurements also showed roughly a microsecond of avoidable scheduling overhead. The final implementation therefore does no background preparation with one Go execution thread or at most 32 transactions. Empty and one-transaction cases retain the baseline allocation counts. The 32-transaction case benefits from sizing without launching a worker. Threshold and single-thread behavior have regression coverage. + +## Validation and limits + +Full replay and block race suites passed on both Zen 5 and M4 Pro. Metrics has no tests. Native vet for replay/metrics and the validator build passed. Tests cover fork replacement introducing a duplicate after preparation, concurrent sibling publication, stale identity binding, changed snapshot slice offsets, rejected/incomplete banks, mismatched preparation, pinned views, snapshot restore, unwind, empty banks and scheduling boundaries. + +Raw logs, source hashes, summaries and the alternating recheck are in [results/status-publication/2026-09-15](results/status-publication/2026-09-15). The baseline comparison covers only status publication. No live replay or FAST improvement is claimed. The staging binary was not deployed; the existing validator remained active and voting throughout the tests. + +Reproduce from this branch: + +```sh +GOMAXPROCS=2 go test -race -p 2 ./pkg/replay ./pkg/block ./pkg/metrics -count=1 +GOMAXPROCS=2 go vet -p 2 ./pkg/replay ./pkg/metrics +GOMAXPROCS=2 go build -p 2 ./cmd/mithril +GOMAXPROCS=2 go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusPublication$' -benchtime=10x -count=5 +go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatus(ExecutionOverlap|SmallPublication)$' -benchtime=100ms -count=5 -cpu=1,2 +``` diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 10548fdd4..ae2b8a4ac 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -171,16 +171,20 @@ type BlockReplay struct { // BlockUpdateAccounts is synchronous critical-path work: rooted-tail // buffering (including its callback) or legacy store enqueue. It excludes // legacy asynchronous disk completion. - BlockUpdateAccounts Timing - TransactionStatusCommit Timing - SignatureVerificationJoin Timing - AccountsDeltaHash Timing - LtHashDedupe Timing - LtHashWorkerCompute Timing - LtHashPartialReduce Timing - BankHashFinalize Timing - BankHash Timing - AlpenglowFooterVerification Timing + BlockUpdateAccounts Timing + TransactionStatusCommit Timing + // Preparation overlaps execution and is not additive with replay wall time. + // PreparationWait is the residual join nested within TransactionStatusCommit. + TransactionStatusPreparation Timing + TransactionStatusPreparationWait Timing + SignatureVerificationJoin Timing + AccountsDeltaHash Timing + LtHashDedupe Timing + LtHashWorkerCompute Timing + LtHashPartialReduce Timing + BankHashFinalize Timing + BankHash Timing + AlpenglowFooterVerification Timing // PostProcessBlock is caller-side state publication and replay // bookkeeping after ProcessBlock returns. TransactionStatusView, // ChainTipUpdate, and ResumeContext are nested sub-phases; logging, summary diff --git a/pkg/replay/block.go b/pkg/replay/block.go index d28f3b8ca..22cecfc18 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -4115,6 +4115,15 @@ func ProcessBlock( if statusValidationErr != nil { return nil, fmt.Errorf("validate transaction statuses for slot %d: %w", block.Slot, statusValidationErr) } + statusPreparation := transactionStatuses.startStatusPreparation(executionPlan) + defer func() { + // Join before returning so a rejected bank cannot leave work behind or + // charge its preparation time to the next block's metrics record. + statusPreparation.wait() + if statusPreparation != nil { + metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) + } + }() ctx, task := trace.NewTask(context.Background(), "ProcessBlock") defer task.End() trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) @@ -4374,7 +4383,10 @@ func ProcessBlock( return slotCtx, err } statusCommitStart := time.Now() - statusErr := transactionStatuses.commitBlockWithPlan(block, executionPlan) + statusWaitStart := time.Now() + preparedStatuses := statusPreparation.wait() + metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) + statusErr := transactionStatuses.commitBlockWithPreparedDelta(block, executionPlan, preparedStatuses) metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) if statusErr != nil { return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index bf043fb17..2703c4199 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -586,6 +586,10 @@ func (c *TransactionStatusCache) CommitBlock(block *b.Block) error { // commitBlockWithPlan atomically rechecks the mutable lineage/status state and // publishes the already-prepared immutable transaction identities. func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + return c.commitBlockWithPreparedDelta(block, plan, nil) +} + +func (c *TransactionStatusCache) commitBlockWithPreparedDelta(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta) error { if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { return errors.New("prepared transaction message identities do not match block") } @@ -604,23 +608,27 @@ func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockT return err } - delta := make(transactionStatusDelta) - for index := 0; index < plan.messageIdentities.Len(); index++ { - identity := plan.messageIdentities.Identity(index) - blockhash := identity.RecentBlockhash - group := delta[blockhash] - if group == nil { - keyIndex := uint8(0) - if visible := c.visible[blockhash]; visible != nil { - keyIndex = visible.keyIndex + delta := transactionStatusDelta(nil) + if prepared != nil && prepared.identities == plan.messageIdentities { + delta = prepared.delta + // A restore or branch transition can change a blockhash's slice offset. + // Rebuild from full identities if any current group uses another offset. + for blockhash, group := range delta { + if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { + delta = nil + break } - group = &transactionStatusGroup{ - keyIndex: keyIndex, - keys: make(map[transactionStatusKey]struct{}), + } + } + if delta == nil { + counts := countTransactionStatusGroups(plan.messageIdentities) + indexes := make(map[solana.Hash]uint8, len(counts)) + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex } - delta[blockhash] = group } - group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + delta = buildTransactionStatusDelta(plan.messageIdentities, counts, indexes) } if err := c.addDeltaVisibleLocked(delta); err != nil { @@ -804,7 +812,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, - keys: make(map[transactionStatusKey]uint16), + keys: make(map[transactionStatusKey]uint16, len(deltaGroup.keys)), } c.visible[blockhash] = group } diff --git a/pkg/replay/transaction_status_overlap_benchmark_test.go b/pkg/replay/transaction_status_overlap_benchmark_test.go new file mode 100644 index 000000000..0da5f739f --- /dev/null +++ b/pkg/replay/transaction_status_overlap_benchmark_test.go @@ -0,0 +1,110 @@ +package replay + +import ( + "fmt" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" +) + +// This controlled workload runs 4,096 executions of the transfer fixture while +// preparing 33,760 independent status keys. It measures scheduling/GC contention, +// not full replay: no accounts are committed, and the status fixture differs +// from the repeated transfer fixture. Run with -cpu=1,2 to compare contention +// without and with a spare execution thread. Check live replay separately. +func BenchmarkTransactionStatusExecutionOverlap(tb *testing.B) { + for _, mode := range []string{"legacy", "sized", "overlap"} { + tb.Run(mode, func(tb *testing.B) { + slotCtx, cleanup := newCommitTestSlotCtx() + defer cleanup() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(0)) + if err != nil { + tb.Fatal(err) + } + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + var execution, commit time.Duration + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var p *transactionStatusPreparation + if mode == "overlap" { + p = cache.startStatusPreparation(plan) + } + start := time.Now() + for range 4096 { + output := LoadAndExecuteTransaction(LoadAndExecuteTransactionInput{SlotCtx: slotCtx, Transaction: tx, LeanResult: true}) + if output.ProcessingResult.TransactionError != nil { + tb.Fatal(output.ProcessingResult.TransactionError) + } + } + execution += time.Since(start) + start = time.Now() + switch mode { + case "legacy": + err = cache.legacyCommitStatusForBenchmark(blk, plan) + case "sized": + err = cache.commitBlockWithPlan(blk, plan) + case "overlap": + err = cache.commitBlockWithPreparedDelta(blk, plan, p.wait()) + } + commit += time.Since(start) + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + tb.StopTimer() + tb.ReportMetric(float64(execution.Nanoseconds())/float64(tb.N), "execution-ns/op") + tb.ReportMetric(float64(commit.Nanoseconds())/float64(tb.N), "commit-with-wait-ns/op") + }) + } +} + +func BenchmarkTransactionStatusSmallPublication(tb *testing.B) { + for _, count := range []int{0, 1, 32} { + for _, mode := range []string{"legacy", "prepared_total"} { + tb.Run(fmt.Sprintf("txs_%d/%s", count, mode), func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + tb.Fatal(err) + } + blk := statusCacheTestBlock(11, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + if mode == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPreparedDelta(blk, plan, cache.startStatusPreparation(plan).wait()) + } + if err != nil { + tb.Fatal(err) + } + // Include unwind equally in this small-work benchmark, avoiding + // timer start/stop overhead around microsecond operations. + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_plan_binding_test.go b/pkg/replay/transaction_status_plan_binding_test.go index 30b456d3a..faf0b2540 100644 --- a/pkg/replay/transaction_status_plan_binding_test.go +++ b/pkg/replay/transaction_status_plan_binding_test.go @@ -15,9 +15,10 @@ func TestPreparedCommitRejectsTransactionReplacement(t *testing.T) { if err != nil { t.Fatal(err) } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) candidate.Transactions[0] = statusCacheTestTransaction(4, 5, 6) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithPreparedDelta(candidate, plan, prepared) if err == nil || err.Error() != "prepared transaction message identities do not match block" { t.Fatalf("commit error = %v, want prepared-plan binding failure", err) } diff --git a/pkg/replay/transaction_status_prepared_test.go b/pkg/replay/transaction_status_prepared_test.go index 44a42ca83..fd71b6857 100644 --- a/pkg/replay/transaction_status_prepared_test.go +++ b/pkg/replay/transaction_status_prepared_test.go @@ -26,6 +26,7 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { plan, err := planBlockTransactionExecution(candidate) requireNoError(err) requireNoError(cache.validateBlockWithPlan(candidate, plan)) + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) requireNoError(cache.Unwind(11)) replacement := statusCacheTestBlock( @@ -34,7 +35,7 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { ) requireNoError(cache.CommitBlock(replacement)) - err = cache.commitBlockWithPlan(candidate, plan) + err = cache.commitBlockWithPreparedDelta(candidate, plan, prepared) var ancestorErr *AncestorAlreadyProcessedTransactionMessagesError if !errors.As(err, &ancestorErr) { t.Fatalf("prepared commit error = %v, want ancestor AlreadyProcessed", err) @@ -83,15 +84,17 @@ func TestConcurrentPreparedSiblingCommitsPublishExactlyOne(t *testing.T) { t.Fatalf("prevalidate right sibling: %v", err) } + leftPrepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + rightPrepared := cache.prepareTransactionStatusDelta(rightPlan.messageIdentities) start := make(chan struct{}) results := make(chan error, 2) go func() { <-start - results <- cache.commitBlockWithPlan(left, leftPlan) + results <- cache.commitBlockWithPreparedDelta(left, leftPlan, leftPrepared) }() go func() { <-start - results <- cache.commitBlockWithPlan(right, rightPlan) + results <- cache.commitBlockWithPreparedDelta(right, rightPlan, rightPrepared) }() close(start) diff --git a/pkg/replay/transaction_status_publication.go b/pkg/replay/transaction_status_publication.go new file mode 100644 index 000000000..e821071ad --- /dev/null +++ b/pkg/replay/transaction_status_publication.go @@ -0,0 +1,83 @@ +package replay + +import ( + "runtime" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +// Preparation owns private, immutable maps. It never publishes a status or +// authorizes a bank: commit still checks coverage, lineage, and duplicates. +type preparedTransactionStatusDelta struct { + identities *b.PreparedTransactionMessageIdentities + delta transactionStatusDelta +} + +type transactionStatusPreparation struct { + done chan struct{} + prepared *preparedTransactionStatusDelta + duration time.Duration +} + +// Replay joins this task on every exit, including rejected banks. It only reads +// the immutable identities, so account loading and ALT resolution can proceed. +func (c *TransactionStatusCache) startStatusPreparation(plan blockTransactionExecutionPlan) *transactionStatusPreparation { + // Small-block measurements show dispatch/join costs as much as the work. + // With one Go execution thread preparation cannot overlap execution at all. + if plan.messageIdentities.Len() <= 32 || runtime.GOMAXPROCS(0) == 1 { + return nil + } + p := &transactionStatusPreparation{done: make(chan struct{})} + go func() { + defer close(p.done) + start := time.Now() + p.prepared = c.prepareTransactionStatusDelta(plan.messageIdentities) + p.duration = time.Since(start) + }() + return p +} + +func (p *transactionStatusPreparation) wait() *preparedTransactionStatusDelta { + if p == nil { + return nil + } + <-p.done + return p.prepared +} + +func countTransactionStatusGroups(identities *b.PreparedTransactionMessageIdentities) map[solana.Hash]int { + counts := make(map[solana.Hash]int) + for i := 0; i < identities.Len(); i++ { + counts[identities.Identity(i).RecentBlockhash]++ + } + return counts +} + +func buildTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8) transactionStatusDelta { + delta := make(transactionStatusDelta, len(counts)) + for blockhash, count := range counts { + delta[blockhash] = &transactionStatusGroup{keyIndex: indexes[blockhash], keys: make(map[transactionStatusKey]struct{}, count)} + } + for i := 0; i < identities.Len(); i++ { + identity := identities.Identity(i) + group := delta[identity.RecentBlockhash] + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + return delta +} + +func (c *TransactionStatusCache) prepareTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities) *preparedTransactionStatusDelta { + counts := countTransactionStatusGroups(identities) + indexes := make(map[solana.Hash]uint8, len(counts)) + // Copy offsets only, never share mutable visible maps with the worker. + c.mu.RLock() + for blockhash := range counts { + if visible := c.visible[blockhash]; visible != nil { + indexes[blockhash] = visible.keyIndex + } + } + c.mu.RUnlock() + return &preparedTransactionStatusDelta{identities: identities, delta: buildTransactionStatusDelta(identities, counts, indexes)} +} diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go new file mode 100644 index 000000000..1bb19d66c --- /dev/null +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -0,0 +1,153 @@ +package replay + +import ( + "encoding/binary" + "errors" + "fmt" + "testing" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/gagliardetto/solana-go" +) + +func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, plan blockTransactionExecutionPlan) error { + if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { + return errors.New("prepared transaction message identities do not match block") + } + c.mu.Lock() + defer c.mu.Unlock() + if !c.coverageComplete { + return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + } + // Parent lineage and ancestor status are mutable, so both remain under the + // publication lock even when hashing and same-bank deduplication happened + // earlier. This keeps commit safe across a concurrent branch transition. + if err := c.validateParentLocked(block); err != nil { + return err + } + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } + + delta := make(transactionStatusDelta) + for index := 0; index < plan.messageIdentities.Len(); index++ { + identity := plan.messageIdentities.Identity(index) + blockhash := identity.RecentBlockhash + group := delta[blockhash] + if group == nil { + keyIndex := uint8(0) + if visible := c.visible[blockhash]; visible != nil { + keyIndex = visible.keyIndex + } + group = &transactionStatusGroup{ + keyIndex: keyIndex, + keys: make(map[transactionStatusKey]struct{}), + } + delta[blockhash] = group + } + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + } + + if err := c.legacyAddStatusForBenchmark(delta); err != nil { + return err + } + c.tip = &transactionStatusNode{ + slot: block.Slot, + blockID: solana.Hash(block.AlpenglowBlockID), + hasBlockID: block.HasAlpenglowBlockID, + parent: c.tip, + delta: delta, + } + return nil +} + +// Frozen production commit algorithm before publication optimization. This is +// an independent baseline, including its original visible-index allocation. +func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { + for blockhash, deltaGroup := range delta { + if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { + return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", + blockhash, group.keyIndex, deltaGroup.keyIndex) + } + } + for blockhash, deltaGroup := range delta { + group := c.visible[blockhash] + if group == nil { + group = &visibleTransactionStatusGroup{ + keyIndex: deltaGroup.keyIndex, + keys: make(map[transactionStatusKey]uint16), + } + c.visible[blockhash] = group + } + for key := range deltaGroup.keys { + group.keys[key]++ + } + } + return nil +} + +// BenchmarkTransactionStatusPublication times only status publication, with +// prepared message identities. No execution, disk I/O, signing or networking. +// prepared_commit excludes delta preparation; prepared_total includes it and +// goroutine dispatch/join, with no execution overlap. Neither measures replay. +// Each iteration restores the same ancestor contents; existing maps retain +// steady-state capacity. Fixture creation, seeding and unwind are not timed. +func BenchmarkTransactionStatusPublication(tb *testing.B) { + const count = 33760 + for _, groups := range []int{1, 4} { + for _, existing := range []bool{false, true} { + tb.Run(fmt.Sprintf("groups_%d/existing_%t", groups, existing), func(tb *testing.B) { + txs := benchmarkUniqueTransactions(count * 2) + for i, tx := range txs { + binary.LittleEndian.PutUint32(tx.Message.RecentBlockhash[:], uint32(i%groups+1)) + } + parent := statusCacheTestBlock(10, txs[:count]...) + if !existing { + parent.Transactions = nil + } + blk := statusCacheTestBlock(11, txs[count:]...) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + tb.Fatal(err) + } + for _, name := range []string{"legacy", "sized", "prepared_total", "prepared_commit"} { + tb.Run(name, func(tb *testing.B) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(parent); err != nil { + tb.Fatal(err) + } + tb.ReportAllocs() + tb.ResetTimer() + for range tb.N { + var err error + if name == "prepared_commit" { + tb.StopTimer() + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + tb.StartTimer() + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else if name == "prepared_total" { + prepared := cache.startStatusPreparation(plan).wait() + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else if name == "legacy" { + err = cache.legacyCommitStatusForBenchmark(blk, plan) + } else { + err = cache.commitBlockWithPlan(blk, plan) + } + if err != nil { + tb.Fatal(err) + } + tb.StopTimer() + if got := cache.tip.slot; got != 11 { + tb.Fatalf("tip=%d", got) + } + if err := cache.Unwind(11); err != nil { + tb.Fatal(err) + } + tb.StartTimer() + } + }) + } + }) + } + } +} diff --git a/pkg/replay/transaction_status_publication_test.go b/pkg/replay/transaction_status_publication_test.go new file mode 100644 index 000000000..58a24b06f --- /dev/null +++ b/pkg/replay/transaction_status_publication_test.go @@ -0,0 +1,133 @@ +package replay + +import ( + "fmt" + "runtime" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/stretchr/testify/require" +) + +func TestPreparedStatusDeltaRebindsSnapshotOffsets(t *testing.T) { + for _, from := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + for _, to := range []uint64{0, 7, txstatus.MaxCachedKeyIndex} { + t.Run(fmt.Sprintf("%d_to_%d", from, to), func(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed := func(offset uint64) *TransactionStatusCache { + cache, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, offset)}}, + }, 0) + require.NoError(t, err) + return cache + } + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed(from).prepareTransactionStatusDelta(plan.messageIdentities) + cache := seed(to) + pinned := cache.View() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + require.Equal(t, uint8(to), cache.tip.delta[candidate.Transactions[0].Message.RecentBlockhash].keyIndex) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + found, err = pinned.ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + blob, err := cache.SnapshotThrough(1) + require.NoError(t, err) + restored, err := NewTransactionStatusCacheFromSnapshot(blob) + require.NoError(t, err) + found, err = restored.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) + require.NoError(t, cache.Unwind(1)) + found, err = cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(ancestor) + require.NoError(t, err) + require.True(t, found) + }) + } + } +} + +func TestPreparedStatusDeltaDoesNotPublishUntilCommit(t *testing.T) { + prior := runtime.GOMAXPROCS(2) + defer runtime.GOMAXPROCS(prior) + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + candidate := statusCacheTestBlock(11, benchmarkUniqueTransactions(33760)...) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + // A rejected bank joins and discards the prepared maps. Waiting is also + // idempotent for the normal commit followed by ProcessBlock's deferred join. + require.Same(t, p.wait(), p.wait()) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + require.Equal(t, uint64(10), cache.tip.slot) + cache.mu.Lock() + cache.coverageComplete = false + cache.mu.Unlock() + var incomplete *IncompleteTransactionStatusCoverageError + require.ErrorAs(t, cache.commitBlockWithPreparedDelta(candidate, plan, p.wait()), &incomplete) + require.Equal(t, uint64(10), cache.tip.slot) +} + +func TestPreparedStatusDeltaRejectsWrongPlanWithoutPublishingIt(t *testing.T) { + cache := NewTransactionStatusCache() + require.NoError(t, cache.CommitBlock(statusCacheTestBlock(10))) + left := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + right := statusCacheTestBlock(11, statusCacheTestTransaction(1, 4, 5)) + leftPlan, err := planBlockTransactionExecution(left) + require.NoError(t, err) + rightPlan, err := planBlockTransactionExecution(right) + require.NoError(t, err) + prepared := cache.prepareTransactionStatusDelta(leftPlan.messageIdentities) + // A mismatched prepared delta falls back to the actual block's identities. + require.NoError(t, cache.commitBlockWithPreparedDelta(right, rightPlan, prepared)) + found, err := cache.View().ContainsTransaction(left.Transactions[0]) + require.NoError(t, err) + require.False(t, found) + found, err = cache.View().ContainsTransaction(right.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} + +func TestPreparedStatusDeltaEmptyBlock(t *testing.T) { + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + require.Nil(t, p, "empty bank must not queue background work") + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) +} + +func TestPreparedStatusDeltaScheduling(t *testing.T) { + for _, threads := range []int{1, 2} { + for _, count := range []int{1, 32, 33} { + t.Run(fmt.Sprintf("threads_%d/txs_%d", threads, count), func(t *testing.T) { + previous := runtime.GOMAXPROCS(threads) + defer runtime.GOMAXPROCS(previous) + cache := NewTransactionStatusCache() + block := statusCacheTestBlock(10, benchmarkUniqueTransactions(count)...) + plan, err := planBlockTransactionExecution(block) + require.NoError(t, err) + p := cache.startStatusPreparation(plan) + defer p.wait() + require.Equal(t, threads > 1 && count > 32, p != nil) + require.NoError(t, cache.commitBlockWithPreparedDelta(block, plan, p.wait())) + for _, tx := range block.Transactions { + found, err := cache.View().ContainsTransaction(tx) + require.NoError(t, err) + require.True(t, found) + } + }) + } + } +} From c241d6b512fd99da44fbf6ec2192eb4f86570b2d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:02:01 -0500 Subject: [PATCH 007/111] Include raw status-publication benchmark and validation logs --- .../2026-09-15/local-race-final.log | 3 + .../2026-09-15/native-benchmark.log | 85 +++++++++++++++ .../2026-09-15/native-build.log | 0 .../2026-09-15/native-final-run.log | 12 +++ .../2026-09-15/native-overlap-final.log | 95 ++++++++++++++++ .../2026-09-15/native-overlap.log | 102 ++++++++++++++++++ .../2026-09-15/native-race.log | 3 + .../2026-09-15/native-vet.log | 0 .../2026-09-15/recheck-abba.log | 34 ++++++ 9 files changed, 334 insertions(+) create mode 100644 docs/results/status-publication/2026-09-15/local-race-final.log create mode 100644 docs/results/status-publication/2026-09-15/native-benchmark.log create mode 100644 docs/results/status-publication/2026-09-15/native-build.log create mode 100644 docs/results/status-publication/2026-09-15/native-final-run.log create mode 100644 docs/results/status-publication/2026-09-15/native-overlap-final.log create mode 100644 docs/results/status-publication/2026-09-15/native-overlap.log create mode 100644 docs/results/status-publication/2026-09-15/native-race.log create mode 100644 docs/results/status-publication/2026-09-15/native-vet.log create mode 100644 docs/results/status-publication/2026-09-15/recheck-abba.log diff --git a/docs/results/status-publication/2026-09-15/local-race-final.log b/docs/results/status-publication/2026-09-15/local-race-final.log new file mode 100644 index 000000000..17b7d4d2a --- /dev/null +++ b/docs/results/status-publication/2026-09-15/local-race-final.log @@ -0,0 +1,3 @@ +ok github.com/Overclock-Validator/mithril/pkg/replay 4.618s +ok github.com/Overclock-Validator/mithril/pkg/block 1.774s +? github.com/Overclock-Validator/mithril/pkg/metrics [no test files] diff --git a/docs/results/status-publication/2026-09-15/native-benchmark.log b/docs/results/status-publication/2026-09-15/native-benchmark.log new file mode 100644 index 000000000..ea4b94603 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-benchmark.log @@ -0,0 +1,85 @@ +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4971792 ns/op 6302003 B/op 555 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 5003752 ns/op 6302003 B/op 555 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4990269 ns/op 6302003 B/op 555 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 5067117 ns/op 6302003 B/op 555 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4855132 ns/op 6302003 B/op 555 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3869972 ns/op 3151475 B/op 265 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3593875 ns/op 3151477 B/op 265 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3116164 ns/op 3151475 B/op 265 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3292134 ns/op 3151475 B/op 265 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3331600 ns/op 3151475 B/op 265 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3459366 ns/op 3151780 B/op 269 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3555785 ns/op 3151707 B/op 269 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3338994 ns/op 3151718 B/op 269 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3393252 ns/op 3151755 B/op 269 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3035989 ns/op 3151707 B/op 269 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1433079 ns/op 1575587 B/op 132 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1465220 ns/op 1575587 B/op 132 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1587430 ns/op 1575587 B/op 132 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1668259 ns/op 1575587 B/op 132 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1625507 ns/op 1575587 B/op 132 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 5038118 ns/op 3466193 B/op 304 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4926609 ns/op 3466193 B/op 304 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4965933 ns/op 3466196 B/op 304 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4991180 ns/op 3466193 B/op 304 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 5174867 ns/op 3466193 B/op 304 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4592110 ns/op 1891049 B/op 159 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4689758 ns/op 1891049 B/op 159 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4579822 ns/op 1891049 B/op 159 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4893416 ns/op 1891052 B/op 159 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4657032 ns/op 1891049 B/op 159 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4617374 ns/op 1891281 B/op 163 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4434153 ns/op 1891281 B/op 163 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4425112 ns/op 1891281 B/op 163 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4362413 ns/op 1891281 B/op 163 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4649669 ns/op 1891281 B/op 163 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2756954 ns/op 315161 B/op 26 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2987364 ns/op 315161 B/op 26 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2747530 ns/op 315161 B/op 26 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 3033686 ns/op 315161 B/op 26 allocs/op +BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2735607 ns/op 315161 B/op 26 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4829304 ns/op 6301427 B/op 659 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4554119 ns/op 6301427 B/op 659 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4920607 ns/op 6301427 B/op 659 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4624988 ns/op 6301427 B/op 659 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4587116 ns/op 6301427 B/op 659 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3398182 ns/op 3151859 B/op 283 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3626107 ns/op 3151859 B/op 283 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3743889 ns/op 3151859 B/op 283 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 4029608 ns/op 3151861 B/op 283 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 4958660 ns/op 3151859 B/op 283 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5919178 ns/op 3152091 B/op 287 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5632307 ns/op 3152091 B/op 287 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5473052 ns/op 3152091 B/op 287 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5915974 ns/op 3152091 B/op 287 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 6688551 ns/op 3152091 B/op 287 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2205247 ns/op 1575779 B/op 141 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2480043 ns/op 1575779 B/op 141 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2205044 ns/op 1575779 B/op 141 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 1927871 ns/op 1575779 B/op 141 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2036015 ns/op 1575779 B/op 141 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 7297434 ns/op 3465532 B/op 357 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5048127 ns/op 3465532 B/op 357 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5224367 ns/op 3465532 B/op 357 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5277320 ns/op 3465532 B/op 357 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 4971670 ns/op 3465535 B/op 357 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 5041342 ns/op 1891228 B/op 169 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4491682 ns/op 1891228 B/op 169 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4966235 ns/op 1891228 B/op 169 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4644107 ns/op 1891228 B/op 169 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4479055 ns/op 1891228 B/op 169 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4914007 ns/op 1891511 B/op 173 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 5031889 ns/op 1891508 B/op 173 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 5046504 ns/op 1891460 B/op 173 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4496234 ns/op 1891460 B/op 173 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4949162 ns/op 1891460 B/op 173 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2603068 ns/op 315148 B/op 27 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2859800 ns/op 315148 B/op 27 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2858236 ns/op 315148 B/op 27 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 3113986 ns/op 315148 B/op 27 allocs/op +BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2788538 ns/op 315148 B/op 27 allocs/op +PASS diff --git a/docs/results/status-publication/2026-09-15/native-build.log b/docs/results/status-publication/2026-09-15/native-build.log new file mode 100644 index 000000000..e69de29bb diff --git a/docs/results/status-publication/2026-09-15/native-final-run.log b/docs/results/status-publication/2026-09-15/native-final-run.log new file mode 100644 index 000000000..c6d28bfba --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-final-run.log @@ -0,0 +1,12 @@ +Running as unit: mithril-status-publication-final-20260915.service +race 0 +vet 0 +build 0 +benchmark-build 0 +benchmark 0 +overlap-final 0 + Finished with result: success +Main processes terminated with: code=exited, status=0/SUCCESS + Service runtime: 37.879s + CPU time consumed: 45.681s + Memory peak: 634.8M (swap: 0B) diff --git a/docs/results/status-publication/2026-09-15/native-overlap-final.log b/docs/results/status-publication/2026-09-15/native-overlap-final.log new file mode 100644 index 000000000..a1e297aaf --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-overlap-final.log @@ -0,0 +1,95 @@ +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusExecutionOverlap/legacy 5 24692874 ns/op 6513086 commit-with-wait-ns/op 18179262 execution-ns/op 23276118 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 22834712 ns/op 5772168 commit-with-wait-ns/op 17062174 execution-ns/op 23276118 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 23370663 ns/op 6486422 commit-with-wait-ns/op 16883798 execution-ns/op 23276136 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 23246803 ns/op 5956655 commit-with-wait-ns/op 17289566 execution-ns/op 23276145 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 24078611 ns/op 6062517 commit-with-wait-ns/op 18015548 execution-ns/op 23276059 B/op 242221 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20677847 ns/op 5607605 commit-with-wait-ns/op 15069600 execution-ns/op 23276630 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20222041 ns/op 5394296 commit-with-wait-ns/op 14827324 execution-ns/op 23276627 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19971750 ns/op 5078614 commit-with-wait-ns/op 14892735 execution-ns/op 23276532 B/op 242224 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 22424538 ns/op 5368235 commit-with-wait-ns/op 17055698 execution-ns/op 23276712 B/op 242226 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20975910 ns/op 5790017 commit-with-wait-ns/op 15184788 execution-ns/op 23276624 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 21296975 ns/op 3860111 commit-with-wait-ns/op 17436468 execution-ns/op 20125587 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 21108482 ns/op 4260022 commit-with-wait-ns/op 16848123 execution-ns/op 20125592 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 6 20802767 ns/op 4299800 commit-with-wait-ns/op 16502635 execution-ns/op 20125558 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 6 22758328 ns/op 5033044 commit-with-wait-ns/op 17724954 execution-ns/op 20125533 B/op 241931 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 23246396 ns/op 5467937 commit-with-wait-ns/op 17778082 execution-ns/op 20125614 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 19020036 ns/op 4526523 commit-with-wait-ns/op 14493163 execution-ns/op 20126028 B/op 241935 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18422586 ns/op 4167392 commit-with-wait-ns/op 14254792 execution-ns/op 20126009 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18573776 ns/op 4299496 commit-with-wait-ns/op 14273834 execution-ns/op 20126012 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18332367 ns/op 3962772 commit-with-wait-ns/op 14369209 execution-ns/op 20126006 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 19268730 ns/op 4381208 commit-with-wait-ns/op 14886998 execution-ns/op 20126038 B/op 241935 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 6 21422337 ns/op 4616295 commit-with-wait-ns/op 16804831 execution-ns/op 20125514 B/op 241931 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 21450447 ns/op 3882218 commit-with-wait-ns/op 17567431 execution-ns/op 20125593 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 23556986 ns/op 5081649 commit-with-wait-ns/op 18474472 execution-ns/op 20125593 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 22188363 ns/op 4492746 commit-with-wait-ns/op 17694970 execution-ns/op 20125587 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 6 22888561 ns/op 4725085 commit-with-wait-ns/op 18162634 execution-ns/op 20125536 B/op 241931 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17090947 ns/op 2119131 commit-with-wait-ns/op 14966507 execution-ns/op 20126253 B/op 241938 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17313847 ns/op 2127423 commit-with-wait-ns/op 15181536 execution-ns/op 20126212 B/op 241938 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 17116780 ns/op 1984282 commit-with-wait-ns/op 15127492 execution-ns/op 20126437 B/op 241939 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 16485753 ns/op 1917207 commit-with-wait-ns/op 14563704 execution-ns/op 20126176 B/op 241938 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16709920 ns/op 2210525 commit-with-wait-ns/op 14495070 execution-ns/op 20126125 B/op 241937 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1299566 94.83 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1284991 93.05 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1247547 96.51 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1211845 96.60 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1238605 95.12 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1596568 76.21 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1614400 74.16 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1581337 76.14 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1574422 74.71 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1589880 74.16 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 123.0 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 119.6 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 118.8 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 119.4 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 117.4 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1000000 100.4 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1231928 97.12 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1000000 101.3 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1200760 99.52 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1206331 100.0 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 189308 586.1 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 200305 607.5 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 188314 597.3 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 196060 586.6 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 197784 593.0 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 227380 470.5 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 240021 461.5 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 231192 472.9 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 254556 466.8 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 239902 443.4 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167852 716.6 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167379 708.2 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167563 708.3 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 169233 699.7 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 164168 735.9 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 190630 609.1 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 190263 615.6 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 176863 604.2 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 196165 632.1 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 185583 588.6 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 17916 6009 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19801 6085 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19857 6155 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19308 6126 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 20053 6002 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23298 5038 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23206 5158 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 22210 5145 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23406 5305 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23084 5114 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26672 4343 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26752 4456 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 27342 4471 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26960 4312 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 27775 4592 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 31856 3644 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32778 3717 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32239 3784 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 30372 3818 ns/op 3616 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32263 3770 ns/op 3616 B/op 13 allocs/op +PASS diff --git a/docs/results/status-publication/2026-09-15/native-overlap.log b/docs/results/status-publication/2026-09-15/native-overlap.log new file mode 100644 index 000000000..e347eed1c --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-overlap.log @@ -0,0 +1,102 @@ +Running as unit: mithril-status-overlap-bench-20260915.service +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusExecutionOverlap/legacy 6 22659962 ns/op 5699140 commit-with-wait-ns/op 16960453 execution-ns/op 23276112 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 22853486 ns/op 5584087 commit-with-wait-ns/op 17269100 execution-ns/op 23276118 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 23170780 ns/op 6502884 commit-with-wait-ns/op 16667603 execution-ns/op 23276136 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 22833095 ns/op 6053272 commit-with-wait-ns/op 16779392 execution-ns/op 23276116 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy 5 22058768 ns/op 5848843 commit-with-wait-ns/op 16209621 execution-ns/op 23276112 B/op 242222 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19456453 ns/op 5114189 commit-with-wait-ns/op 14341895 execution-ns/op 23276694 B/op 242226 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20721459 ns/op 5463814 commit-with-wait-ns/op 15257144 execution-ns/op 23276633 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 18883924 ns/op 5031514 commit-with-wait-ns/op 13851997 execution-ns/op 23276650 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19163664 ns/op 5128481 commit-with-wait-ns/op 14034884 execution-ns/op 23276584 B/op 242225 allocs/op +BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 18648556 ns/op 4971799 commit-with-wait-ns/op 13676368 execution-ns/op 23276654 B/op 242226 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 6 20730994 ns/op 3673079 commit-with-wait-ns/op 17057597 execution-ns/op 20125561 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 21164557 ns/op 4762666 commit-with-wait-ns/op 16401568 execution-ns/op 20125556 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 20803959 ns/op 3673651 commit-with-wait-ns/op 17130021 execution-ns/op 20125587 B/op 241932 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 21585622 ns/op 4745710 commit-with-wait-ns/op 16839541 execution-ns/op 20125537 B/op 241931 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized 5 22061011 ns/op 4786467 commit-with-wait-ns/op 17274047 execution-ns/op 20125526 B/op 241931 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18481807 ns/op 4332538 commit-with-wait-ns/op 14148935 execution-ns/op 20126009 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18233386 ns/op 4143998 commit-with-wait-ns/op 14088971 execution-ns/op 20125942 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 17926782 ns/op 3900186 commit-with-wait-ns/op 14026293 execution-ns/op 20126008 B/op 241935 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 17641308 ns/op 3846237 commit-with-wait-ns/op 13794796 execution-ns/op 20126009 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18874807 ns/op 4403265 commit-with-wait-ns/op 14471150 execution-ns/op 20126009 B/op 241934 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 6 21101004 ns/op 2063760 commit-with-wait-ns/op 19033430 execution-ns/op 20125788 B/op 241936 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 21351433 ns/op 3481211 commit-with-wait-ns/op 17866676 execution-ns/op 20125816 B/op 241936 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 21378252 ns/op 3800343 commit-with-wait-ns/op 17574134 execution-ns/op 20125816 B/op 241936 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 20630833 ns/op 1688017 commit-with-wait-ns/op 18939909 execution-ns/op 20125819 B/op 241936 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap 5 22262745 ns/op 3971910 commit-with-wait-ns/op 18286875 execution-ns/op 20125816 B/op 241936 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16698061 ns/op 2155589 commit-with-wait-ns/op 14538161 execution-ns/op 20126252 B/op 241938 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 16586794 ns/op 2151713 commit-with-wait-ns/op 14431196 execution-ns/op 20126259 B/op 241938 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16814463 ns/op 2297246 commit-with-wait-ns/op 14512878 execution-ns/op 20126390 B/op 241939 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17545014 ns/op 2416401 commit-with-wait-ns/op 15124436 execution-ns/op 20126122 B/op 241937 allocs/op +BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17149268 ns/op 2323257 commit-with-wait-ns/op 14821130 execution-ns/op 20126128 B/op 241937 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1273947 96.94 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1294056 91.35 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1271734 92.04 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1300807 94.81 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1272133 96.92 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1589078 76.19 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1577804 77.30 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1563583 77.34 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1626802 74.57 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1591627 75.31 ns/op 112 B/op 2 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 721636 185.7 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 704294 179.3 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 725799 182.1 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 725127 181.7 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 700783 176.9 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 726256 138.7 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 728829 139.5 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 861189 144.8 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 765130 143.7 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 751879 142.9 ns/op 264 B/op 5 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 194295 653.8 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 176815 648.9 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 174601 882.8 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 149839 829.7 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy 162600 810.9 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 242948 633.3 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 140822 720.2 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 156864 754.5 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 235440 462.3 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 215985 471.1 ns/op 960 B/op 9 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 66573 1705 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 71836 1640 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 70480 1655 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 71564 1641 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 73791 1645 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 77616 1489 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 78930 1520 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 75838 1511 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 74458 1491 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 79232 1551 ns/op 1192 B/op 13 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19609 6019 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 18758 6100 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19924 6118 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19128 6067 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19795 6000 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23222 5027 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23133 5064 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23482 5243 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 22717 5013 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 24062 5630 ns/op 6320 B/op 23 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 20682 6564 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 18136 5879 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 21844 5837 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 19086 5635 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 19254 5724 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 20040 5546 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 18378 5860 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 19862 5961 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 19912 6906 ns/op 3848 B/op 17 allocs/op +BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 21711 7037 ns/op 3848 B/op 17 allocs/op +PASS +ok github.com/Overclock-Validator/mithril/pkg/replay 17.258s + Finished with result: success +Main processes terminated with: code=exited, status=0/SUCCESS + Service runtime: 19.605s + CPU time consumed: 23.467s + Memory peak: 387.6M (swap: 0B) diff --git a/docs/results/status-publication/2026-09-15/native-race.log b/docs/results/status-publication/2026-09-15/native-race.log new file mode 100644 index 000000000..00504d333 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/native-race.log @@ -0,0 +1,3 @@ +ok github.com/Overclock-Validator/mithril/pkg/replay 2.267s +ok github.com/Overclock-Validator/mithril/pkg/block 1.133s +? github.com/Overclock-Validator/mithril/pkg/metrics [no test files] diff --git a/docs/results/status-publication/2026-09-15/native-vet.log b/docs/results/status-publication/2026-09-15/native-vet.log new file mode 100644 index 000000000..e69de29bb diff --git a/docs/results/status-publication/2026-09-15/recheck-abba.log b/docs/results/status-publication/2026-09-15/recheck-abba.log new file mode 100644 index 000000000..9720620a0 --- /dev/null +++ b/docs/results/status-publication/2026-09-15/recheck-abba.log @@ -0,0 +1,34 @@ +Running as unit: mithril-status-publication-recheck-20260915.service; invocation ID: 8079e456482443308a68c660fb8f6b29 +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 50 4400299 ns/op 6301401 B/op 659 allocs/op +PASS + +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 50 2984879 ns/op 3152087 B/op 287 allocs/op +PASS + +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 50 3131031 ns/op 3152091 B/op 287 allocs/op +PASS + +goos: linux +goarch: amd64 +pkg: github.com/Overclock-Validator/mithril/pkg/replay +cpu: AMD Ryzen 7 9700X 8-Core Processor +BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 50 4564884 ns/op 6301399 B/op 659 allocs/op +PASS + + Finished with result: success +Main processes terminated with: code=exited, status=0/SUCCESS + Service runtime: 1.278s + CPU time consumed: 1.456s + Memory peak: 73.9M (swap: 0B) From fe54faa2af45c00954b154bcf8fd65607dd1b0cd Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:42:51 -0500 Subject: [PATCH 008/111] Reuse immutable node encodings across status checkpoints Share lazy encoding state through capture and pruning without retaining parent links or copying synchronization primitives. Keep MTS2 bytes and caller-owned output unchanged; allocate exactly sized node and checkpoint buffers. Check the original wire encoder, concurrent pruning and encoding, output ownership and recovery. Benchmark moving windows including the default 128-root cadence; document retained-memory costs and the all-new-window limit. --- docs/status-checkpoint-capture.md | 53 ++++++- pkg/replay/transaction_status_cache.go | 138 +++++++++++------ .../transaction_status_capture_bench_test.go | 49 +++++- pkg/replay/transaction_status_capture_test.go | 141 +++++++++++++++++- 4 files changed, 331 insertions(+), 50 deletions(-) diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md index e47e67e2b..7f096d6b5 100644 --- a/docs/status-checkpoint-capture.md +++ b/docs/status-checkpoint-capture.md @@ -1,7 +1,52 @@ -# Transaction-status checkpoint capture +# Transaction-status checkpoint capture and encoding -Replay used to encode and sort the full status checkpoint while preparing a promotion. Capture now pins immutable lineage and coverage metadata; the existing promotion worker encodes the same checkpoint later. Capture occurs before pruning and publication ordering stays unchanged. Pinned snapshots retain their nodes across pruning/unwind. +Replay captures immutable lineage and coverage metadata before submitting a +checkpoint to the promotion worker. Sorting, encoding and writing happen on +that worker. Capture does not retain parent links outside the selected window. +Publication and durable-root ordering are unchanged. -The native Zen5 benchmark captured roughly30MB of status data: original202–207ms versus5.92–6.06microseconds on replay. Encoding moved to the worker and was not eliminated. Ten live captures took29–34microseconds; worker encoding179–367ms. These are historical stage measurements, not a promise of total validator speedup; see docs/results/status-cache/2026-09-14. +Each node memoizes its canonical encoded body on first serialization. Capture +and pruning share the same cache object when copying a node header; they never +copy a used synchronization primitive. Encoding depends on the immutable slot, +block-ID presence/value and status delta, not its parent link. Concurrent +encoders synchronize through `sync.Once` without taking the live cache lock. +Each snapshot still constructs its own coverage header and returns an owned +output buffer. The MTS2 format and restore validation are unchanged. -This PR also includes batched expiry; see transaction-status-expiry.md. The proposed transaction-status publication optimization has not been implemented or included. Fresh standalone replay race and vet checks are under docs/results/pr-split-2026-09-15/status. +The cache retains roughly one extra encoded window (30 MB for 1.5 million +keys), plus any nodes pinned by older views. There is no global encoding map: +caches become collectible with their last node/view. A completely new window +still pays for all sorting. Output copying and checkpoint I/O remain necessary. + +## Encoding benchmark + +`BenchmarkTransactionStatusCheckpointEncoding` uses a 300-root window with +5,000 keys per root (1.5 million keys, roughly 30 MB encoded). Each iteration +replaces the specified number of roots. Fixture creation and initial warming +are excluded; new node headers, sorting and output allocations are included. +The baseline is the original uncached wire encoder retained in tests. + +Apple M4 Pro, Go 1.26.4, one caller, GOMAXPROCS=12; medians of three runs: + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 158.05 ms | 1.35 ms | +| 8 | 157.14 ms | 5.09 ms | +| 32 | 155.62 ms | 17.51 ms | +| 128 (default fold cadence) | 153.74 ms | 67.11 ms | +| 300 (entirely new) | 155.90 ms | 156.29 ms | + +At the default cadence, allocated bytes per encoding fell from 99.12 MB to +57.33 MB; this excludes retained heap. These are encoding measurements, not +end-to-end fold/replay timings or live FAST improvements. Data distribution +matters: newly rooted large blocks can account for most keys in the window. + +Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusCheckpointEncoding$' -benchmem -benchtime=1s -count=3`. + +Tests compare exact bytes with the original encoder across coverage flags, +block IDs and sorted groups; check concurrent encoding during pruning/unwind; +verify cache sharing before and after warming; and restore checkpoints after +callers mutate their own output buffers. The replay race suite and vet pass. + +Related behavior: [status expiry](transaction-status-expiry.md) and +[status publication](transaction-status-publication.md). diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 2703c4199..f299f8a02 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -10,6 +10,7 @@ import ( "path/filepath" "sort" "sync" + "sync/atomic" b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/state" @@ -48,6 +49,38 @@ type transactionStatusNode struct { hasBlockID bool parent *transactionStatusNode delta transactionStatusDelta + + // Shared by parentless checkpoint copies and relinked retained nodes. + // Only the memoized encoding changes after publication; lineage and delta + // remain immutable. Never copy the atomic field after its first use. + encoding atomic.Pointer[transactionStatusNodeEncoding] +} + +type transactionStatusNodeEncoding struct { + once sync.Once + data []byte +} + +func (n *transactionStatusNode) encodingCache() *transactionStatusNodeEncoding { + if cache := n.encoding.Load(); cache != nil { + return cache + } + cache := new(transactionStatusNodeEncoding) + if n.encoding.CompareAndSwap(nil, cache) { + return cache + } + return n.encoding.Load() +} + +// copyInto initializes a fresh node, sharing its encoding without retaining +// excluded ancestry or copying a used synchronization primitive. The encoding excludes +// parent links and depends only on the immutable slot, block ID and delta. +func (n *transactionStatusNode) copyInto(copy *transactionStatusNode, parent *transactionStatusNode) { + *copy = transactionStatusNode{ + slot: n.slot, blockID: n.blockID, hasBlockID: n.hasBlockID, + parent: parent, delta: n.delta, + } + copy.encoding.Store(n.encodingCache()) } type visibleTransactionStatusGroup struct { @@ -733,10 +766,8 @@ func (c *TransactionStatusCache) CaptureSnapshotThrough(through uint64) (Transac owned := make([]transactionStatusNode, len(nodes)) pinned := make([]*transactionStatusNode, len(nodes)) for i, node := range nodes { - owned[i] = *node - // The encoder consumes only these node deltas. Do not keep the old - // parent chain, which could retain roots excluded from this snapshot. - owned[i].parent = nil + // Do not keep the old parent chain or copy its atomic field. + node.copyInto(&owned[i], nil) pinned[i] = &owned[i] } return &transactionStatusSnapshot{ @@ -895,10 +926,9 @@ func (c *TransactionStatusCache) pruneLocked(through uint64) { c.expireVisibleLocked(nodes[:drop], retained) var parent *transactionStatusNode for _, old := range retained { - parent = &transactionStatusNode{ - slot: old.slot, blockID: old.blockID, hasBlockID: old.hasBlockID, - parent: parent, delta: old.delta, - } + next := new(transactionStatusNode) + old.copyInto(next, parent) + parent = next } c.tip = parent } @@ -976,7 +1006,16 @@ func sliceTransactionStatusKey(messageHash [32]byte, keyIndex uint8) transaction } func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { - var buf bytes.Buffer + encoded := make([][]byte, len(nodes)) + size := 9 // magic, flags, rooted count and node count + for i, node := range nodes { + cache := node.encodingCache() + cache.once.Do(func() { cache.data = marshalTransactionStatusNode(node) }) + encoded[i] = cache.data + size += len(cache.data) + } + // Every caller owns its result. Never return or append into a cached slice. + buf := bytes.NewBuffer(make([]byte, 0, size)) buf.Write(transactionStatusSnapshotMagic[:]) flags := byte(0) if complete { @@ -986,44 +1025,57 @@ func marshalTransactionStatusNodes(nodes []*transactionStatusNode, rootedSinceSe flags |= 2 } buf.WriteByte(flags) - _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) - _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) - for _, node := range nodes { - _ = binary.Write(&buf, binary.LittleEndian, node.slot) - nodeFlags := byte(0) - if node.hasBlockID { - nodeFlags = 1 - } - buf.WriteByte(nodeFlags) - if node.hasBlockID { - buf.Write(node.blockID[:]) - } - blockhashes := make([]solana.Hash, 0, len(node.delta)) - for blockhash := range node.delta { - blockhashes = append(blockhashes, blockhash) + _ = binary.Write(buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(buf, binary.LittleEndian, uint16(len(nodes))) + for _, data := range encoded { + buf.Write(data) + } + return buf.Bytes(), nil +} + +func marshalTransactionStatusNode(node *transactionStatusNode) []byte { + size := 8 + 1 + 4 // slot, flags and group count + if node.hasBlockID { + size += len(node.blockID) + } + for _, group := range node.delta { + size += 32 + 1 + 4 + transactionStatusKeySize*len(group.keys) + } + buf := bytes.NewBuffer(make([]byte, 0, size)) + _ = binary.Write(buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) } - sort.Slice(blockhashes, func(i, j int) bool { - return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) - for _, blockhash := range blockhashes { - group := node.delta[blockhash] - buf.Write(blockhash[:]) - buf.WriteByte(group.keyIndex) - keys := make([]transactionStatusKey, 0, len(group.keys)) - for key := range group.keys { - keys = append(keys, key) - } - sort.Slice(keys, func(i, j int) bool { - return bytes.Compare(keys[i][:], keys[j][:]) < 0 - }) - _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) - for _, key := range keys { - buf.Write(key[:]) - } + _ = binary.Write(buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) } } - return buf.Bytes(), nil + return buf.Bytes() } func (c *TransactionStatusCache) restore(data []byte) error { diff --git a/pkg/replay/transaction_status_capture_bench_test.go b/pkg/replay/transaction_status_capture_bench_test.go index 64bf7a7eb..234743222 100644 --- a/pkg/replay/transaction_status_capture_bench_test.go +++ b/pkg/replay/transaction_status_capture_bench_test.go @@ -3,6 +3,7 @@ package replay import ( "crypto/sha256" "encoding/binary" + "fmt" "testing" "github.com/gagliardetto/solana-go" @@ -11,7 +12,7 @@ import ( var checkpointBenchmarkPayload []byte var checkpointBenchmarkCapture TransactionStatusSnapshot -func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { +func checkpointEncodingFixture() *TransactionStatusCache { // A private, not-yet-published fixture with the same complete 300-root // metadata as an imported cache. 1.5 million keys encode to roughly 30 MB. c := newTransactionStatusCache(true) @@ -32,6 +33,11 @@ func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { c.tip = &transactionStatusNode{slot: slot, parent: c.tip, delta: transactionStatusDelta{solana.Hash{1}: {keyIndex: 7, keys: keys}}} } + return c +} + +func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { + c := checkpointEncodingFixture() view, err := c.CaptureSnapshotThrough(maxTransactionStatusRoots) if err != nil { b.Fatal(err) @@ -64,3 +70,44 @@ func BenchmarkTransactionStatusCheckpointCapture(b *testing.B) { } }) } + +// Moving 300-root windows at several checkpoint cadences. Fixtures and initial +// cache warming are excluded; allocating new node headers and encoding/output +// allocation are included. This measures encoding only, not fsync or account +// checkpoint work. Cold encodes represent first startup/all-new windows. +func BenchmarkTransactionStatusCheckpointEncoding(b *testing.B) { + c := checkpointEncodingFixture() + view, err := c.CaptureSnapshotThrough(300) + if err != nil { + b.Fatal(err) + } + seed := view.(*transactionStatusSnapshot).nodes + for _, advance := range []int{0, 1, 8, 32, defaultFoldBatchSlots, 300} { + for _, cached := range []bool{false, true} { + b.Run(fmt.Sprintf("new=%d/cached=%t", advance, cached), func(b *testing.B) { + nodes := append([]*transactionStatusNode(nil), seed...) + if cached { + _, _ = marshalTransactionStatusNodes(nodes, 300, true, false) + } + b.ReportAllocs() + b.ResetTimer() + for n := 0; n < b.N; n++ { + copy(nodes, nodes[advance:]) + for i := 300 - advance; i < 300; i++ { + nodes[i] = &transactionStatusNode{slot: uint64(301 + n*advance + i), delta: seed[i].delta} + } + var err error + if cached { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodes(nodes, 300, true, false) + } else { + checkpointBenchmarkPayload, err = marshalTransactionStatusNodesUncached(nodes, 300, true, false) + } + if err != nil { + b.Fatal(err) + } + } + b.SetBytes(int64(len(checkpointBenchmarkPayload))) + }) + } + } +} diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go index 755fa9dfd..8e1344c6f 100644 --- a/pkg/replay/transaction_status_capture_test.go +++ b/pkg/replay/transaction_status_capture_test.go @@ -1,7 +1,10 @@ package replay import ( + "bytes" "encoding/binary" + "fmt" + "sort" "sync" "testing" "time" @@ -9,11 +12,12 @@ import ( b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/txstatus" "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) // Keep the pre-split selection/metadata calculation as a differential oracle. -// The wire encoder itself did not change. +// Use the original uncached wire encoder to check byte-for-byte compatibility. func legacyStatusSnapshotForTest(c *TransactionStatusCache, through uint64) ([]byte, error) { c.mu.RLock() defer c.mu.RUnlock() @@ -26,7 +30,7 @@ func legacyStatusSnapshotForTest(c *TransactionStatusCache, through uint64) ([]b if rooted > maxTransactionStatusRoots { rooted = maxTransactionStatusRoots } - return marshalTransactionStatusNodes(nodes, uint16(rooted), complete, c.coverageFromGenesis) + return marshalTransactionStatusNodesUncached(nodes, uint16(rooted), complete, c.coverageFromGenesis) } func importedStatusCacheForTest(t *testing.T) *TransactionStatusCache { @@ -160,3 +164,136 @@ func TestTransactionStatusCaptureEncodingDoesNotLockLiveCache(t *testing.T) { t.Fatal("checkpoint encoding waited for the live cache lock") } } + +func marshalTransactionStatusNodesUncached(nodes []*transactionStatusNode, rootedSinceSeed uint16, complete bool, coverageFromGenesis bool) ([]byte, error) { + var buf bytes.Buffer + buf.Write(transactionStatusSnapshotMagic[:]) + flags := byte(0) + if complete { + flags = 1 + } + if coverageFromGenesis { + flags |= 2 + } + buf.WriteByte(flags) + _ = binary.Write(&buf, binary.LittleEndian, rootedSinceSeed) + _ = binary.Write(&buf, binary.LittleEndian, uint16(len(nodes))) + for _, node := range nodes { + _ = binary.Write(&buf, binary.LittleEndian, node.slot) + nodeFlags := byte(0) + if node.hasBlockID { + nodeFlags = 1 + } + buf.WriteByte(nodeFlags) + if node.hasBlockID { + buf.Write(node.blockID[:]) + } + blockhashes := make([]solana.Hash, 0, len(node.delta)) + for blockhash := range node.delta { + blockhashes = append(blockhashes, blockhash) + } + sort.Slice(blockhashes, func(i, j int) bool { + return bytes.Compare(blockhashes[i][:], blockhashes[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(blockhashes))) + for _, blockhash := range blockhashes { + group := node.delta[blockhash] + buf.Write(blockhash[:]) + buf.WriteByte(group.keyIndex) + keys := make([]transactionStatusKey, 0, len(group.keys)) + for key := range group.keys { + keys = append(keys, key) + } + sort.Slice(keys, func(i, j int) bool { + return bytes.Compare(keys[i][:], keys[j][:]) < 0 + }) + _ = binary.Write(&buf, binary.LittleEndian, uint32(len(keys))) + for _, key := range keys { + buf.Write(key[:]) + } + } + } + return buf.Bytes(), nil +} + +func TestTransactionStatusEncodingSharedAcrossCaptureAndPrune(t *testing.T) { + for _, warmBeforePrune := range []bool{false, true} { + t.Run(fmt.Sprintf("warm=%t", warmBeforePrune), func(t *testing.T) { + c := importedStatusCacheForTest(t) + for slot := uint64(301); slot <= 305; slot++ { + require.NoError(t, c.CommitBlock(captureTestBlock(slot, 1))) + } + first, err := c.CaptureSnapshotThrough(304) + require.NoError(t, err) + pinned := first.(*transactionStatusSnapshot) + want, err := legacyStatusSnapshotForTest(c, 304) + require.NoError(t, err) + if warmBeforePrune { + got, err := first.MarshalBinary() + require.NoError(t, err) + require.Equal(t, want, got) + } + // Force relinking of retained nodes after the snapshot has copied + // their headers, including the still-unencoded case. + c.Root(305) + second, err := c.CaptureSnapshotThrough(305) + require.NoError(t, err) + current := second.(*transactionStatusSnapshot) + caches := make(map[uint64]*transactionStatusNodeEncoding) + for _, node := range pinned.nodes { + caches[node.slot] = node.encodingCache() + } + for _, node := range current.nodes { + if prior := caches[node.slot]; prior != nil { + require.Same(t, prior, node.encodingCache()) + } + } + var wg sync.WaitGroup + for i := 0; i < 8; i++ { + wg.Add(1) + go func() { + defer wg.Done() + got, err := first.MarshalBinary() + assert.NoError(t, err) + assert.Equal(t, want, got) + // Mutate the node body as well as the header; neither may + // alias the memoized node data or another caller's result. + clear(got) + }() + } + wg.Wait() + currentWant, err := legacyStatusSnapshotForTest(c, 305) + require.NoError(t, err) + got, err := second.MarshalBinary() + require.NoError(t, err) + require.Equal(t, currentWant, got) + restored, err := NewTransactionStatusCacheFromSnapshot(got) + require.NoError(t, err) + roundTrip, err := restored.SnapshotThrough(305) + require.NoError(t, err) + require.Equal(t, got, roundTrip) + }) + } +} + +func TestTransactionStatusEncodingMatchesOriginalWireFormat(t *testing.T) { + // Deliberately unsorted groups/keys, nonzero offsets, empty deltas and + // mixed block-ID presence exercise every independently cached field. + nodes := []*transactionStatusNode{ + {slot: 3, hasBlockID: true, blockID: solana.Hash{9}, delta: transactionStatusDelta{ + solana.Hash{7}: {keyIndex: 11, keys: map[transactionStatusKey]struct{}{{8}: {}, {1}: {}, {4}: {}}}, + solana.Hash{1}: {keyIndex: 2, keys: map[transactionStatusKey]struct{}{{9}: {}, {2}: {}}}, + }}, + {slot: 5}, + {slot: 8, delta: transactionStatusDelta{solana.Hash{3}: {keyIndex: 0, keys: map[transactionStatusKey]struct{}{}}}}, + } + for _, complete := range []bool{false, true} { + for _, genesis := range []bool{false, true} { + want, err := marshalTransactionStatusNodesUncached(nodes, 3, complete, genesis) + require.NoError(t, err) + got, err := marshalTransactionStatusNodes(nodes, 3, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got) + } + } +} From f9f2f64fd0247d595e8f460e2afe61a19ca3504c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Mon, 14 Sep 2026 22:51:15 -0500 Subject: [PATCH 009/111] Document native checkpoint encoding measurements Record moving-window results at the real 128-root cadence, retained-memory tradeoffs and unchanged cold-window costs. Keep raw benchmark artifacts outside the source tree; do not infer live voting gains from staging measurements. --- docs/status-checkpoint-capture.md | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md index 7f096d6b5..ff11bc1fc 100644 --- a/docs/status-checkpoint-capture.md +++ b/docs/status-checkpoint-capture.md @@ -50,3 +50,24 @@ callers mutate their own output buffers. The replay race suite and vet pass. Related behavior: [status expiry](transaction-status-expiry.md) and [status publication](transaction-status-publication.md). + +## Native Zen 5 validation + +AMD Ryzen 7 9700X, Go 1.26.4, GOMAXPROCS=2, Nice 15 and a two-core CPU quota, +while the validator continued its normal workload. Same moving-window fixture; +three samples per case, medians below. This compares the original uncached +encoder with memoization, not the whole status-publication change against dev. + +| New roots per checkpoint | Original encoding | Cached encoding | +| --- | ---: | ---: | +| 1 | 195.34 ms | 3.73 ms | +| 8 | 195.15 ms | 8.56 ms | +| 32 | 194.77 ms | 23.54 ms | +| 128 (default fold cadence) | 194.52 ms | 85.02 ms | +| 300 (entirely new) | 201.88 ms | 198.55 ms | + +The default-cadence result is approximately 2.3x, with the same 99.12 → 57.33 MB +allocation reduction. Cold/all-new windows remain roughly unchanged. Native +combined race suites, vet and the validator build passed. These are staging +measurements: the encoding cache has not been deployed, so a live reduction in +durable-root lag or missed FAST votes has not yet been established. From b8a3aedf11122ed59dc5b5d87d50c87e6f85b802 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 00:32:25 -0500 Subject: [PATCH 010/111] replay: preflight fold batch eligibility before copying account writes --- docs/status-checkpoint-capture.md | 26 ++++++++++++++++++++ pkg/accounts/overlay_test.go | 39 ++++++++++++++++++++++++++++++ pkg/accounts/working_set.go | 32 +++++++++++++++++++----- pkg/replay/async_promotion_test.go | 23 ++++++++++++++++++ pkg/replay/promotion.go | 13 ++++------ 5 files changed, 119 insertions(+), 14 deletions(-) diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md index ff11bc1fc..62a4dc0e8 100644 --- a/docs/status-checkpoint-capture.md +++ b/docs/status-checkpoint-capture.md @@ -71,3 +71,29 @@ allocation reduction. Cold/all-new windows remain roughly unchanged. Native combined race suites, vet and the validator build passed. These are staging measurements: the encoding cache has not been deployed, so a live reduction in durable-root lag or missed FAST votes has not yet been established. + +## Fold admission before collecting account writes + +Replay checks for a checkpoint batch on every iteration, including skipped +slots. `WorkingSet.PromotionChunk` first counts eligible held slots under its +read lock. If fewer than the configured batch size are available, ordinary +admission returns nil without allocating account-pointer lists. When ready, +it collects only the oldest batch, not the entire eligible suffix. Forced +partial folds still collect the available prefix. + +This preflight is not a finality shortcut or a new recovery policy. Replay's +existing finality/verification gates supply the upper bound. Selection and +collection hold the same lock; account pointers retain their existing ownership +contract. Preparation does not prune the suffix or advance the durable root. +The worker's write/commit order, required resume context, checkpoint reference +validation, completion bookkeeping, and forced shutdown/epoch-boundary paths +are unchanged. + +`BenchmarkBuildFoldJobWaitingForBatch` holds 127 slots with 512 account writes +each while waiting for the default 128-slot batch. On Ryzen 9700X, +GOMAXPROCS=8, three 300 ms runs, median admission-check time fell from 426 µs +to 31.8 ns; 627,008 bytes and 134 allocations per rejected preparation became +zero. This measures an ineligible batch check, not encoding, disk I/O, or a +ready checkpoint. Boundary tests cover gaps, the finality upper bound, a full +batch, forced partial batches, and selection after promotion; existing replay +checkpoint/recovery tests cover the unchanged durable path. diff --git a/pkg/accounts/overlay_test.go b/pkg/accounts/overlay_test.go index f40a26dfc..682e47217 100644 --- a/pkg/accounts/overlay_test.go +++ b/pkg/accounts/overlay_test.go @@ -411,3 +411,42 @@ func TestOverlayDeltaAccountsIncludesOverride(t *testing.T) { assert.Equal(t, pk(1), delta[0].Key) assert.Equal(t, uint64(99), delta[0].Lamports) } + +func TestWorkingSetPromotionChunkBoundaries(t *testing.T) { + w := NewWorkingSet() + for _, slot := range []uint64{5, 7, 9, 11} { + w.Add(slot, []*Account{uoAcct(1, slot), uoAcct(2, slot+100)}) + } + for _, tc := range []struct { + through uint64 + limit int + partial bool + slots []uint64 + }{ + {4, 2, true, nil}, {5, 2, false, nil}, {7, 2, false, []uint64{5, 7}}, + {11, 2, false, []uint64{5, 7}}, {9, 4, true, []uint64{5, 7, 9}}, + {9, 4, false, nil}, {11, 0, true, nil}, {11, -1, false, nil}, + } { + got := w.PromotionChunk(tc.through, tc.limit, tc.partial) + var slots []uint64 + for _, sd := range got { + slots = append(slots, sd.Slot) + require.Len(t, sd.Delta, 2) + for _, acct := range sd.Delta { + require.True(t, acct.Lamports == sd.Slot || acct.Lamports == sd.Slot+100) + } + } + require.Equal(t, tc.slots, slots) + } + // Preparing a job leaves the live suffix intact. Once the caller commits + // and promotes a prefix, the next chunk must start at the surviving slot. + require.Equal(t, 4, w.HeldSlots()) + w.PromotePrefix(7) + chunk := w.PromotionChunk(11, 2, false) + require.Equal(t, []uint64{9, 11}, []uint64{chunk[0].Slot, chunk[1].Slot}) + require.Zero(t, testing.AllocsPerRun(100, func() { + if w.PromotionChunk(9, 2, false) != nil { + panic("partial chunk escaped") + } + })) +} diff --git a/pkg/accounts/working_set.go b/pkg/accounts/working_set.go index d88725f24..083a4405d 100644 --- a/pkg/accounts/working_set.go +++ b/pkg/accounts/working_set.go @@ -134,17 +134,37 @@ func (w *WorkingSet) PromotionPrefix(through uint64) []SlotDelta { w.mu.RLock() defer w.mu.RUnlock() - var batch []SlotDelta - for _, slot := range w.order { // ascending - if slot > through { - break - } + return w.promotionChunkLocked(through, len(w.order), true) +} + +// PromotionChunk returns at most maxSlots oldest held slots through the caller's +// verified promotion bound. Unless allowPartial is set, an incomplete chunk +// returns nil before allocating or collecting account writes. Selection and +// collection share one read lock, so pruning cannot change the selected prefix. +// This only prepares borrowed account pointers; it does not commit, prune, or +// advance durability. Callers still own finality checks and durable commit order. +func (w *WorkingSet) PromotionChunk(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + w.mu.RLock() + defer w.mu.RUnlock() + return w.promotionChunkLocked(through, maxSlots, allowPartial) +} + +func (w *WorkingSet) promotionChunkLocked(through uint64, maxSlots int, allowPartial bool) []SlotDelta { + count := 0 + for count < len(w.order) && count < maxSlots && w.order[count] <= through { + count++ + } + if count == 0 || (!allowPartial && count < maxSlots) { + return nil + } + batch := make([]SlotDelta, count) + for i, slot := range w.order[:count] { layer := w.bySlot[slot] delta := make([]*Account, 0, len(layer.writes)) for _, a := range layer.writes { delta = append(delta, a) } - batch = append(batch, SlotDelta{Slot: slot, Delta: delta}) + batch[i] = SlotDelta{Slot: slot, Delta: delta} } return batch } diff --git a/pkg/replay/async_promotion_test.go b/pkg/replay/async_promotion_test.go index eff0de206..2780e425d 100644 --- a/pkg/replay/async_promotion_test.go +++ b/pkg/replay/async_promotion_test.go @@ -444,3 +444,26 @@ func TestShutdownFlushCannotFoldPastGateTarget(t *testing.T) { target = safePromoteTarget(9, true, 7, 6) assert.Equal(t, uint64(5), target, "persisted-divergence floor holds promotion below the disputed slot") } + +// Model repeated replay/skip iterations while a nearly full checkpoint batch +// waits for one more held bank. Account writes must not be copied on this path. +func BenchmarkBuildFoldJobWaitingForBatch(b *testing.B) { + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 128, "") + writes := make([]*accounts.Account, 512) + for i := range writes { + var key [32]byte + key[0], key[1] = byte(i), byte(i>>8) + writes[i] = &accounts.Account{Key: key, Lamports: 1} + } + for slot := uint64(1); slot <= 127; slot++ { + tail.Add(slot, writes, nil) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + job, err := tail.buildFoldJob(127, false) + if err != nil || job != nil { + b.Fatalf("unexpected fold admission: job=%v err=%v", job, err) + } + } +} diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index 752e0cddc..5472ec4ed 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -400,16 +400,13 @@ func (t *unrootedTail) buildFoldJob(through uint64, force bool, hookOverrides .. if err != nil { return nil, err } - prefix := t.overlay.PromotionPrefix(through) - if len(prefix) == 0 { + // Check chunk eligibility before materializing account-write lists. Replay + // calls this on every iteration, including skipped slots; a partial batch + // remains in RAM without rescanning all of its accounts each time. + chunk := t.overlay.PromotionChunk(through, t.batchSlots, force) + if len(chunk) == 0 { return nil, nil } - chunk := prefix - if len(chunk) > t.batchSlots { - chunk = chunk[:t.batchSlots] - } else if len(chunk) < t.batchSlots && !force { - return nil, nil // trailing partial chunk stays in RAM - } through = chunk[len(chunk)-1].Slot ctx := t.contexts[through] From 3f97e9c0d8700364f1406dd16af2a124d488a746 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:21:18 -0500 Subject: [PATCH 011/111] replay: retire completed rewards bookkeeping after durable promotion --- docs/rewards-unwind-retirement.md | 58 +++++++++++++ pkg/replay/block.go | 7 ++ pkg/replay/rewards_retirement.go | 49 +++++++++++ pkg/replay/rewards_retirement_test.go | 119 ++++++++++++++++++++++++++ 4 files changed, 233 insertions(+) create mode 100644 docs/rewards-unwind-retirement.md create mode 100644 pkg/replay/rewards_retirement.go create mode 100644 pkg/replay/rewards_retirement_test.go diff --git a/docs/rewards-unwind-retirement.md b/docs/rewards-unwind-retirement.md new file mode 100644 index 000000000..edf207be8 --- /dev/null +++ b/docs/rewards-unwind-retirement.md @@ -0,0 +1,58 @@ +# Retiring durable rewards bookkeeping + +A completed partitioned-rewards distribution used to leave its in-memory +descriptor alive for the rest of the replay attempt. The fork-switch guard +rejects any such descriptor because account-overlay unwind cannot restore the +consumed spool or its distribution counters. This is necessary while completion +is speculative, but unnecessarily forces checkpoint replay after completion +has become durable. + +Replay now observes the inactive EpochRewards sysvar in a successfully executed +bank's immutable snapshot, with zero partitions remaining. It remembers that +bank's slot and the exact distribution descriptor. Only applying a successful +durable fold through that slot retires the descriptor. Later bank observations +do not move the completion slot forward. A new descriptor/epoch invalidates the +old evidence; missing sysvars or unknown completion retain the old fallback. + +## Safety and recovery contract + +- Completion in memory, certificate finality, and submitting a fold do not + authorize retirement. Failed folds leave the durable watermark unchanged. +- Active distribution and completed-but-not-durable distribution retain the + existing rewards guard. No spool reconstruction or rewards rollback is added. +- After retirement, in-memory switches still require the existing epoch, + vote/stake-cache, parent-context, sysvar and transaction-status checks. + Switches at/below the durable watermark still require durable recovery. +- Completion evidence is replay-thread-owned and process-local. It does not + change checkpoint formats, signing reservations, persisted vote history, + clean-shutdown rules or restart authorization. Restart retains the existing + persisted EpochRewards validation. No extra file or disk sync is introduced. + +## Incident motivating the change + +On Zen 5, distribution completed at slot 3,942,001. At a later parent-linked +switch, the durable checkpoint was already 3,944,067; child 3,944,076 selected +parent 3,944,073, abandoning the suffix from 3,944,074. The remaining descriptor +forced the rewards-window fallback even though completion was below the root. +Checkpoint recovery re-fetched previously received blocks, with logged waits +of 2.739 seconds and 0.967 seconds. A buffered 665-transaction block waited +3,613.510 ms for replay admission and then executed in 7.520 ms. + +These are incident observations, not a before/after benchmark or a measurement +of checkpoint encoding/fsync time. Thirteen observed FAST aggregates omitted +our vote during the recovery interval; that does not prove absence from every +FAST aggregate or a single cause for all thirteen omissions. No live latency +improvement is established until a comparable switch exercises the new path. + +## Validation + +`rewards_retirement_test.go` covers active/missing bank state, unknown completion, +the exact durable boundary, later-bank observations, generation changes, failed +and successful folds, and an exact-parent unwind after retirement (including +account values, resume state and immutable rewards sysvars). Existing unwind +tests still require fallback for zero-remaining bookkeeping without retirement, +cross-epoch switches, dirty vote/stake caches and invalid parent snapshots. + +Full replay/rewards race suites passed locally and in the combined native +build; native node recovery/checkpoint race tests, vet and validator build also +passed. These are software tests, not mainnet power-loss qualification. diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 22cecfc18..1b1f23d74 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1747,6 +1747,7 @@ func ReplayBlocks( var unwoundParentBankSysvars *sealevel.BankSysvars var partitionedEpochRewardsEnabled bool var partitionedRewardsInfo *rewards.PartitionedRewardDistributionInfo + var rewardsCompletion partitionedRewardsCompletion var featuresActivatedInFirstSlot []*accounts.Account var parentFeaturesActivatedInFirstSlot []*accounts.Account @@ -2061,6 +2062,10 @@ func ReplayBlocks( mithrilState.LastRootedSlot = promotedThrough mithrilState.LastRootedBankhash = rootedCtx.Bankhash mithrilState.LastRootedContext = rootedCtx + if rewardsCompletion.retire(&partitionedRewardsInfo, promotedThrough) { + rewardsHoldBelowSlot = 0 + mlog.Log.Infof("epoch rewards bookkeeping retired through durable slot %d; later fork switches may unwind in memory", promotedThrough) + } if transactionStatuses.Root(promotedThrough) { mlog.Log.Infof("transaction status cache reconstructed complete %d-root coverage through durable slot %d", maxTransactionStatusRoots, promotedThrough) @@ -2873,6 +2878,7 @@ func ReplayBlocks( boundaryParentCtx = epochBoundaryParentCtx(acctsDb, block, currentEpoch, replayCtx.CurrentFeatures) } partitionedRewardsInfo = handleEpochTransition(acctsDb, partitionedEpochRewardsEnabled, boundaryParentCtx, replayCtx, epochSchedule, replayCtx.CurrentFeatures, block, currentEpoch, rpcc, dbgOpts) + rewardsCompletion = partitionedRewardsCompletion{} currentEpoch = block.Epoch justCrossedEpochBoundary = true // While partitioned rewards are distributing, hold durable promotion @@ -2985,6 +2991,7 @@ func ReplayBlocks( } // The successful child now owns its derived snapshot. Any later bank uses // lastSlotCtx; the one-shot retained unwind bridge is no longer needed. + rewardsCompletion.observeBank(partitionedRewardsInfo, lastSlotCtx.BankSysvars()) unwoundParentBankSysvars = nil postProcessBlockStart := processBlockEnd statusViewStart := time.Now() diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go new file mode 100644 index 000000000..a03a00cbf --- /dev/null +++ b/pkg/replay/rewards_retirement.go @@ -0,0 +1,49 @@ +package replay + +import ( + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" +) + +// partitionedRewardsCompletion is replay-thread-owned, process-local evidence +// that a successfully executed bank contains all effects of this distribution. +// It is not a checkpoint or signing authority. Until that bank is durable, +// tryInLoopUnwind must still reject even a zero-remaining distribution: its +// spool has been consumed and cannot be rolled back with the account overlay. +type partitionedRewardsCompletion struct { + info *rewards.PartitionedRewardDistributionInfo + slot uint64 +} + +// observeBank must run only after successful block execution/publication, using +// that bank's immutable sysvars (never the speculative global sysvar cache). +// If the first completed bank lacks evidence, recording a later descendant is +// conservative: retirement then waits for that later bank to become durable. +func (c *partitionedRewardsCompletion) observeBank(info *rewards.PartitionedRewardDistributionInfo, bank *sealevel.BankSysvars) { + if c.info != info { + *c = partitionedRewardsCompletion{info: info} + } + if info == nil || c.slot != 0 || info.NumRewardPartitionsRemaining != 0 || bank == nil || bank.Slot() == 0 { + return + } + epochRewards, ok := bank.EpochRewards() + if ok && !epochRewards.Active { + c.slot = bank.Slot() + } +} + +// retire is called only when replay applies a successfully committed fold and +// advances LastRootedSlot. Finality, an enqueued/in-flight fold, and a failed +// commit do not acknowledge durability. At this boundary every rewards effect +// is in AccountsDB; in-memory switches above it cannot undo distribution. +// Switches at/below it still take durable recovery, whose persisted +// EpochRewards validation remains unchanged. Restart loses this optional +// evidence and reconstructs state through the existing recovery path. +func (c *partitionedRewardsCompletion) retire(info **rewards.PartitionedRewardDistributionInfo, durableSlot uint64) bool { + if *info == nil || *info != c.info || c.slot == 0 || durableSlot < c.slot || (*info).NumRewardPartitionsRemaining != 0 { + return false + } + *info = nil + *c = partitionedRewardsCompletion{} + return true +} diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go new file mode 100644 index 000000000..460584eee --- /dev/null +++ b/pkg/replay/rewards_retirement_test.go @@ -0,0 +1,119 @@ +package replay + +import ( + "bytes" + "encoding/base64" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/state" + bin "github.com/gagliardetto/binary" + "github.com/mr-tron/base58" + "github.com/stretchr/testify/require" +) + +func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { + active := &sealevel.SysvarEpochRewards{Active: true} + var raw bytes.Buffer + require.NoError(t, active.MarshalWithEncoder(bin.NewBinEncoder(&raw))) + activeBank, err := sealevel.NewBankSysvars(5, &accounts.Account{Key: sealevel.SysvarEpochRewardsAddr, Data: raw.Bytes()}) + require.NoError(t, err) + missingBank, err := sealevel.NewBankSysvars(5) + require.NoError(t, err) + for _, tc := range []struct { + name string + remaining uint64 + bank *sealevel.BankSysvars + }{ + {"active distribution", 1, testUnwindBankSysvars(t, 5, 50)}, + {"active bank", 0, activeBank}, + {"missing bank", 0, nil}, + {"missing rewards", 0, missingBank}, + {"unknown slot", 0, testUnwindBankSysvars(t, 0, 50)}, + } { + t.Run(tc.name, func(t *testing.T) { + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: tc.remaining} + var completed partitionedRewardsCompletion + completed.observeBank(info, tc.bank) + require.False(t, completed.retire(&info, 100)) + require.NotNil(t, info) + }) + } + + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + require.False(t, completed.retire(&info, 100), "zero remaining without observed completion is insufficient") + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + require.False(t, completed.retire(&info, 4), "uncommitted completion must retain the guard") + require.True(t, completed.retire(&info, 5), "later observations must not postpone recorded completion") + require.Nil(t, info) + require.False(t, completed.retire(&info, 100), "retirement is one-shot") +} + +func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { + old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} + next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} + var completed partitionedRewardsCompletion + completed.observeBank(old, testUnwindBankSysvars(t, 5, 50)) + require.False(t, completed.retire(&next, 100), "old completion cannot retire new bookkeeping") + completed.observeBank(next, testUnwindBankSysvars(t, 11, 60)) + require.False(t, completed.retire(&next, 10)) + require.True(t, completed.retire(&next, 11)) +} + +func TestRewardsRetirementWaitsForSuccessfulFold(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 5} + tail := asyncTestTail(fc, 5, 6) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + job, err := tail.buildFoldJob(6, true) + require.NoError(t, err) + require.NotNil(t, job) + root := uint64(4) + require.False(t, completed.retire(&info, root), "capturing a job does not make its bank durable") + require.Error(t, runFoldJob(fc, job)) + require.False(t, completed.retire(&info, root), "a failed fold leaves the old durable root") + fc.failOn = 0 + require.NoError(t, runFoldJob(fc, job)) + ctx := tail.applyFoldJob(job) + require.NotNil(t, ctx) + root = job.through + require.True(t, completed.retire(&info, root)) +} + +func TestRewardsRetirementAllowsExactParentUnwind(t *testing.T) { + resetVoteStakeDirty() + t.Cleanup(resetVoteStakeDirty) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + parent := &state.ResumeContext{Slot: 7, Bankhash: base58.Encode(make([]byte, 32)), AcctsLtHash: base64.StdEncoding.EncodeToString(make([]byte, 2048)), Capitalization: 700} + bank := testUnwindBankSysvars(t, 7, 50) + tail.Add(7, []*accounts.Account{testAccount(1, 71)}, testHashBytes(7)) + tail.SetContext(7, parent, bank) + tail.Add(8, []*accounts.Account{testAccount(1, 81)}, testHashBytes(8)) + tail.SetContext(8, &state.ResumeContext{Slot: 8}, testUnwindBankSysvars(t, 8, 999)) + sw := &CertifiedSwitch{Slot: 8} + ms := &state.MithrilState{LastRootedSlot: 4} + sched := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000} + rs, _, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Nil(t, rs) + require.Equal(t, unwindFallbackRewardsWindow, reason) + ms.LastRootedSlot = 5 + markVoteStakeDirty(5) // completed reward writes are also below the durable root + require.True(t, completed.retire(&info, ms.LastRootedSlot)) + rs, restored, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Empty(t, reason) + require.Same(t, bank, restored, "use the surviving bank, never abandoned reward sysvars") + want, err := ResumeStateFromRootedContext(parent, nil) + require.NoError(t, err) + require.Equal(t, want, rs, "resume state must match rebuilding the exact retained parent") + acct, err := tail.GetAccount(8, testAccount(1, 0).Key) + require.NoError(t, err) + require.Equal(t, uint64(71), acct.Lamports, "abandoned account writes must be removed") +} From cb34ddcf6ac5179edc3a89e0f76ee2ba47519a54 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:38:35 -0500 Subject: [PATCH 012/111] replay: reuse ancestor validation for unchanged status cache --- docs/transaction-status-publication.md | 23 ++- pkg/replay/block.go | 4 +- pkg/replay/transaction_status_cache.go | 45 ++++-- .../transaction_status_prepared_test.go | 15 +- ...ction_status_publication_benchmark_test.go | 19 ++- pkg/replay/transaction_status_validation.go | 37 +++++ .../transaction_status_validation_test.go | 151 ++++++++++++++++++ 7 files changed, 271 insertions(+), 23 deletions(-) create mode 100644 pkg/replay/transaction_status_validation.go create mode 100644 pkg/replay/transaction_status_validation_test.go diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index d182bc6fb..d20ddc2d4 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -4,7 +4,7 @@ Replay previously built the immutable per-bank transaction-status delta and grew Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. -The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit still checks exact block/identity binding, complete coverage, parent lineage and all ancestor duplicates under the publication lock. A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. @@ -50,3 +50,24 @@ GOMAXPROCS=2 go build -p 2 ./cmd/mithril GOMAXPROCS=2 go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusPublication$' -benchtime=10x -count=5 go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatus(ExecutionOverlap|SmallPublication)$' -benchtime=100ms -count=5 -cpu=1,2 ``` + +## Reusing pre-execution ancestor validation + +`ProcessBlock` now carries a private validation receipt from its successful ancestor scan to status publication. Under the commit lock, an unchanged receipt avoids scanning all transaction messages again. Publication still checks block binding, complete coverage and parent lineage every time; a missing, foreign or stale receipt performs the full ancestor scan. Direct `CommitBlock` callers retain the full scan. + +The receipt is bound to the cache instance and exact immutable prepared-identity pointer. Visible-index insertion/removal, tip binding, root/prune and restore invalidate the version, including empty commits. Committing and then unwinding back to an identical parent cannot revive a receipt. Version saturation disables reuse permanently rather than wrapping. Snapshot/Agave recovery creates a new cache instance. Receipts are never persisted, and no checkpoint format, durability, voting-resume or crash-recovery guarantee changes. + +The publication benchmark adds `validated_commit` and `invalidated_commit` alongside `prepared_commit`. All three exclude delta preparation and the pre-execution scan. The first reuses that scan; the second calls `Root` between validation and publication, forcing revalidation. Each iteration unwinds and obtains a fresh receipt outside the timer. These are incremental publication comparisons, not the full PR against alpenglow-dev or per-block tail latency. Tests exercise fork replacement introducing duplicates, concurrent sibling commits, cross-cache and cross-identity misuse, snapshot replacement, pruning/root invalidation, binding changes, transaction replacement and version saturation. + +Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five samples × 20 iterations, Nice=19 / 200% CPU quota on the running validator host): + +| Recent blockhash groups | Existing ancestor groups | Full recheck | Reused validation | Invalidated validation | +|---|---|---|---|---| +| 1 | yes | 2.510 ms | 1.364 ms | 2.536 ms | +| 4 | yes | 2.412 ms | 1.283 ms | 2.440 ms | +| 1 | no | 1.461 ms | 1.472 ms | 1.769 ms | +| 4 | no | 1.323 ms | 1.395 ms | 1.303 ms | + +Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. + +A separate 180-second pre-change live trace observed 728 publications. In the 145 publications taking at least 1 ms, the repeated scan measured 1.805 ms median / 3.180 ms maximum; insertion 2.198 / 6.774 ms. Lock acquisition was at most 0.0058 ms across all publications, and the preparation join at most 0.0010 ms. This latency-selected cohort is not a fixed transaction-size sample or a before/after p99 comparison. Probe overhead is included. These measurements identify removable work; they do not establish a sustained FAST improvement. Raw traces, native test windows and exact combined source stay on the validator host at `/srv/mithril-status-validation-20260915`. diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 1b1f23d74..e4ad3b366 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -4117,7 +4117,7 @@ func ProcessBlock( return nil, fmt.Errorf("validate transaction messages for slot %d: %w", block.Slot, err) } statusValidationStart := time.Now() - statusValidationErr := transactionStatuses.validateBlockWithPlan(block, executionPlan) + statusValidation, statusValidationErr := transactionStatuses.validateBlockForPublication(block, executionPlan) metrics.GlobalBlockReplay.TransactionStatusValidation.AddTimingSince(statusValidationStart) if statusValidationErr != nil { return nil, fmt.Errorf("validate transaction statuses for slot %d: %w", block.Slot, statusValidationErr) @@ -4393,7 +4393,7 @@ func ProcessBlock( statusWaitStart := time.Now() preparedStatuses := statusPreparation.wait() metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) - statusErr := transactionStatuses.commitBlockWithPreparedDelta(block, executionPlan, preparedStatuses) + statusErr := transactionStatuses.commitBlockWithValidation(block, executionPlan, preparedStatuses, statusValidation) metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) if statusErr != nil { return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index f299f8a02..bb4a157fb 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -105,6 +105,9 @@ type TransactionStatusCache struct { // from a known-empty genesis cache. Without this bit, completeness requires // the full 300 retained roots; a serialized boolean alone is not evidence. coverageFromGenesis bool + + // Protected by mu; see transactionStatusValidation. Never serialized. + validationVersion uint64 } // TransactionStatusView is an immutable view of one bank lineage. It lazily @@ -445,6 +448,7 @@ func (c *TransactionStatusCache) BindTipBlockID(slot uint64, blockID solana.Hash if c.tip.hasBlockID && c.tip.blockID != blockID { return fmt.Errorf("transaction status tip at slot %d has block id %s, cannot bind %s", slot, c.tip.blockID, blockID) } + c.invalidateValidationLocked() c.tip = &transactionStatusNode{ slot: slot, blockID: blockID, hasBlockID: true, parent: c.tip.parent, delta: c.tip.delta, @@ -548,25 +552,33 @@ func (c *TransactionStatusCache) ValidateBlock(block *b.Block) error { // validateBlockWithPlan preserves the status-cache checks while letting // replay reuse the exact immutable identities used for execution planning. func (c *TransactionStatusCache) validateBlockWithPlan(block *b.Block, plan blockTransactionExecutionPlan) error { + _, err := c.validateBlockForPublication(block, plan) + return err +} + +func (c *TransactionStatusCache) validateBlockForPublication(block *b.Block, plan blockTransactionExecutionPlan) (transactionStatusValidation, error) { if block == nil { - return errors.New("nil block") + return transactionStatusValidation{}, errors.New("nil block") } if plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { - return errors.New("prepared transaction message identities do not match block") + return transactionStatusValidation{}, errors.New("prepared transaction message identities do not match block") } if c == nil { - return &IncompleteTransactionStatusCoverageError{} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{} } c.mu.RLock() defer c.mu.RUnlock() if !c.coverageComplete { - return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + return transactionStatusValidation{}, &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } if err := c.validateParentLocked(block); err != nil { - return err + return transactionStatusValidation{}, err } - return c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities) + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return transactionStatusValidation{}, err + } + return transactionStatusValidation{cache: c, identities: plan.messageIdentities, version: c.validationVersion}, nil } func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { @@ -623,6 +635,10 @@ func (c *TransactionStatusCache) commitBlockWithPlan(block *b.Block, plan blockT } func (c *TransactionStatusCache) commitBlockWithPreparedDelta(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta) error { + return c.commitBlockWithValidation(block, plan, prepared, transactionStatusValidation{}) +} + +func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan blockTransactionExecutionPlan, prepared *preparedTransactionStatusDelta, validation transactionStatusValidation) error { if block == nil || plan.messageIdentities == nil || !plan.messageIdentities.MatchesBlock(block) { return errors.New("prepared transaction message identities do not match block") } @@ -631,14 +647,17 @@ func (c *TransactionStatusCache) commitBlockWithPreparedDelta(block *b.Block, pl if !c.coverageComplete { return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} } - // Parent lineage and ancestor status are mutable, so both remain under the - // publication lock even when hashing and same-bank deduplication happened - // earlier. This keeps commit safe across a concurrent branch transition. + // Always check coverage, block binding and parent lineage. Reuse the earlier + // ancestor scan only under this lock and only for the same unchanged cache + // and immutable identities. A branch transition (including away and back) + // or root/prune invalidates it, requiring a fresh scan before publication. if err := c.validateParentLocked(block); err != nil { return err } - if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { - return err + if !validation.reusableForLocked(c, plan.messageIdentities) { + if err := c.validateAncestorTransactionsLocked(block.Slot, plan.messageIdentities); err != nil { + return err + } } delta := transactionStatusDelta(nil) @@ -704,6 +723,7 @@ func (c *TransactionStatusCache) Root(through uint64) bool { } c.mu.Lock() defer c.mu.Unlock() + c.invalidateValidationLocked() wasComplete := c.coverageComplete newlyRooted := c.countNodesBetweenLocked(c.rootedThrough, through) if through > c.rootedThrough { @@ -832,6 +852,7 @@ func (c *TransactionStatusCache) validateParentLocked(block *b.Block) error { } func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDelta) error { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { return fmt.Errorf("transaction status blockhash %s uses inconsistent key indexes %d and %d", @@ -855,6 +876,7 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe } func (c *TransactionStatusCache) removeDeltaVisibleLocked(delta transactionStatusDelta) { + c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { group := c.visible[blockhash] if group == nil { @@ -1079,6 +1101,7 @@ func marshalTransactionStatusNode(node *transactionStatusNode) []byte { } func (c *TransactionStatusCache) restore(data []byte) error { + c.invalidateValidationLocked() reader := bytes.NewReader(data) var magic [4]byte if _, err := io.ReadFull(reader, magic[:]); err != nil { diff --git a/pkg/replay/transaction_status_prepared_test.go b/pkg/replay/transaction_status_prepared_test.go index fd71b6857..0172fffa6 100644 --- a/pkg/replay/transaction_status_prepared_test.go +++ b/pkg/replay/transaction_status_prepared_test.go @@ -25,7 +25,8 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { candidate := statusCacheTestBlock(12, retried, unique) plan, err := planBlockTransactionExecution(candidate) requireNoError(err) - requireNoError(cache.validateBlockWithPlan(candidate, plan)) + validation, err := cache.validateBlockForPublication(candidate, plan) + requireNoError(err) prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) requireNoError(cache.Unwind(11)) @@ -35,7 +36,7 @@ func TestPreparedCommitRechecksAncestorAfterForkSwitch(t *testing.T) { ) requireNoError(cache.CommitBlock(replacement)) - err = cache.commitBlockWithPreparedDelta(candidate, plan, prepared) + err = cache.commitBlockWithValidation(candidate, plan, prepared, validation) var ancestorErr *AncestorAlreadyProcessedTransactionMessagesError if !errors.As(err, &ancestorErr) { t.Fatalf("prepared commit error = %v, want ancestor AlreadyProcessed", err) @@ -77,10 +78,12 @@ func TestConcurrentPreparedSiblingCommitsPublishExactlyOne(t *testing.T) { if err != nil { t.Fatal(err) } - if err := cache.validateBlockWithPlan(left, leftPlan); err != nil { + leftValidation, err := cache.validateBlockForPublication(left, leftPlan) + if err != nil { t.Fatalf("prevalidate left sibling: %v", err) } - if err := cache.validateBlockWithPlan(right, rightPlan); err != nil { + rightValidation, err := cache.validateBlockForPublication(right, rightPlan) + if err != nil { t.Fatalf("prevalidate right sibling: %v", err) } @@ -90,11 +93,11 @@ func TestConcurrentPreparedSiblingCommitsPublishExactlyOne(t *testing.T) { results := make(chan error, 2) go func() { <-start - results <- cache.commitBlockWithPreparedDelta(left, leftPlan, leftPrepared) + results <- cache.commitBlockWithValidation(left, leftPlan, leftPrepared, leftValidation) }() go func() { <-start - results <- cache.commitBlockWithPreparedDelta(right, rightPlan, rightPrepared) + results <- cache.commitBlockWithValidation(right, rightPlan, rightPrepared, rightValidation) }() close(start) diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go index 1bb19d66c..2fc7d0f45 100644 --- a/pkg/replay/transaction_status_publication_benchmark_test.go +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -90,6 +90,8 @@ func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionSt // prepared message identities. No execution, disk I/O, signing or networking. // prepared_commit excludes delta preparation; prepared_total includes it and // goroutine dispatch/join, with no execution overlap. Neither measures replay. +// validated_commit also excludes the successful pre-execution ancestor scan. +// invalidated_commit roots between validation and commit, forcing a full recheck. // Each iteration restores the same ancestor contents; existing maps retain // steady-state capacity. Fixture creation, seeding and unwind are not timed. func BenchmarkTransactionStatusPublication(tb *testing.B) { @@ -110,7 +112,7 @@ func BenchmarkTransactionStatusPublication(tb *testing.B) { if err != nil { tb.Fatal(err) } - for _, name := range []string{"legacy", "sized", "prepared_total", "prepared_commit"} { + for _, name := range []string{"legacy", "sized", "prepared_total", "prepared_commit", "validated_commit", "invalidated_commit"} { tb.Run(name, func(tb *testing.B) { cache := NewTransactionStatusCache() if err := cache.CommitBlock(parent); err != nil { @@ -120,11 +122,22 @@ func BenchmarkTransactionStatusPublication(tb *testing.B) { tb.ResetTimer() for range tb.N { var err error - if name == "prepared_commit" { + if name == "prepared_commit" || name == "validated_commit" || name == "invalidated_commit" { tb.StopTimer() prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + validation, validationErr := cache.validateBlockForPublication(blk, plan) + if validationErr != nil { + tb.Fatal(validationErr) + } + if name == "invalidated_commit" { + cache.Root(10) + } tb.StartTimer() - err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + if name == "prepared_commit" { + err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) + } else { + err = cache.commitBlockWithValidation(blk, plan, prepared, validation) + } } else if name == "prepared_total" { prepared := cache.startStatusPreparation(plan).wait() err = cache.commitBlockWithPreparedDelta(blk, plan, prepared) diff --git a/pkg/replay/transaction_status_validation.go b/pkg/replay/transaction_status_validation.go new file mode 100644 index 000000000..4413b8b8f --- /dev/null +++ b/pkg/replay/transaction_status_validation.go @@ -0,0 +1,37 @@ +package replay + +import ( + "math" + + b "github.com/Overclock-Validator/mithril/pkg/block" +) + +// transactionStatusValidation records a successful ancestor scan under cache.mu. +// It authorizes skipping only that scan, never the coverage, parent or exact +// block-identity checks. The receipt is private, bound to one cache instance and +// one immutable identity set, and checked under the publication lock. +// +// Mutating the visible index, binding the tip, rooting/pruning or restoring +// invalidates earlier receipts. In particular, committing and then unwinding to +// the same tip cannot resurrect one. Snapshot/Agave constructors create a new +// cache instance; this receipt is neither persisted nor usable after recovery. +// This optimization changes no crash-recovery or durable-checkpoint guarantee. +type transactionStatusValidation struct { + cache *TransactionStatusCache + identities *b.PreparedTransactionMessageIdentities + version uint64 +} + +func (v transactionStatusValidation) reusableForLocked(c *TransactionStatusCache, identities *b.PreparedTransactionMessageIdentities) bool { + return v.cache == c && v.identities == identities && + v.version == c.validationVersion && c.validationVersion != math.MaxUint64 +} + +// invalidateValidationLocked requires exclusive access (mu, or an unpublished +// constructor). Saturation permanently disables reuse instead of wrapping into +// an old generation. Even empty commits invalidate, because they change lineage. +func (c *TransactionStatusCache) invalidateValidationLocked() { + if c.validationVersion != math.MaxUint64 { + c.validationVersion++ + } +} diff --git a/pkg/replay/transaction_status_validation_test.go b/pkg/replay/transaction_status_validation_test.go new file mode 100644 index 000000000..9a1346125 --- /dev/null +++ b/pkg/replay/transaction_status_validation_test.go @@ -0,0 +1,151 @@ +package replay + +import ( + "errors" + "math" + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestStatusValidationInvalidation(t *testing.T) { + for _, action := range []string{"commit", "unwind", "round_trip", "root", "bind", "restore"} { + t.Run(action, func(t *testing.T) { + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if !receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("unchanged receipt not reusable") + } + switch action { + case "commit", "round_trip": + err = cache.CommitBlock(statusCacheTestBlock(11)) + if err == nil && action == "round_trip" { + err = cache.Unwind(11) + } + case "unwind": + err = cache.Unwind(10) + case "root": + cache.Root(10) + case "bind": + err = cache.BindTipBlockID(10, solana.Hash{1}) + case "restore": + var data []byte + data, err = cache.SnapshotThrough(10) + if err == nil { + cache, err = NewTransactionStatusCacheFromSnapshot(data) + } + } + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("receipt survived " + action) + } + }) + } +} + +func TestStatusValidationCannotCrossCacheOrIdentity(t *testing.T) { + good := NewTransactionStatusCache() + bad := NewTransactionStatusCache() + tx := statusCacheTestTransaction(1, 2, 3) + if err := good.CommitBlock(statusCacheTestBlock(10)); err != nil { + t.Fatal(err) + } + if err := bad.CommitBlock(statusCacheTestBlock(10, tx)); err != nil { + t.Fatal(err) + } + blk := statusCacheTestBlock(11, tx) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := good.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + // Both caches have the same version and parent slot, but different contents. + if good.validationVersion != bad.validationVersion { + t.Fatal("fixture must have equal versions") + } + prepared := bad.prepareTransactionStatusDelta(plan.messageIdentities) + var already *AncestorAlreadyProcessedTransactionMessagesError + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign cache: %v", err) + } + + unique := statusCacheTestBlock(11, statusCacheTestTransaction(4, 5, 6)) + uniquePlan, err := planBlockTransactionExecution(unique) + if err != nil { + t.Fatal(err) + } + receipt, err = bad.validateBlockForPublication(unique, uniquePlan) + if err != nil { + t.Fatal(err) + } + if err := bad.commitBlockWithValidation(blk, plan, prepared, receipt); !errors.As(err, &already) { + t.Fatalf("foreign identities: %v", err) + } + failed, err := bad.validateBlockForPublication(blk, plan) + if err == nil || failed.cache != nil { + t.Fatalf("failed validation returned a receipt: %+v, %v", failed, err) + } +} + +func TestStatusValidationSaturation(t *testing.T) { + cache := NewTransactionStatusCache() + cache.validationVersion = math.MaxUint64 - 1 + blk := statusCacheTestBlock(1) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + cache.Root(0) + cache.Root(0) + if cache.validationVersion != math.MaxUint64 || receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("generation wrapped or old receipt reusable") + } + receipt, err = cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + if receipt.reusableForLocked(cache, plan.messageIdentities) { + t.Fatal("saturated cache allowed reuse") + } + if err := cache.commitBlockWithValidation(blk, plan, nil, receipt); err != nil { + t.Fatal(err) + } +} + +func TestStatusValidationStillChecksBlockBinding(t *testing.T) { + cache := NewTransactionStatusCache() + blk := statusCacheTestBlock(1, statusCacheTestTransaction(1, 2, 3)) + plan, err := planBlockTransactionExecution(blk) + if err != nil { + t.Fatal(err) + } + receipt, err := cache.validateBlockForPublication(blk, plan) + if err != nil { + t.Fatal(err) + } + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + blk.Transactions[0] = statusCacheTestTransaction(4, 5, 6) + if err := cache.commitBlockWithValidation(blk, plan, prepared, receipt); err == nil { + t.Fatal("replaced transaction accepted") + } +} From c1f864fbfde9efa0a2e395d9cf35069efd17c756 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:44:04 -0500 Subject: [PATCH 013/111] replay: partition large transaction status index updates --- docs/transaction-status-publication.md | 70 ++++++++- pkg/replay/transaction_status_cache.go | 38 +++-- pkg/replay/transaction_status_index.go | 147 ++++++++++++++++++ ...transaction_status_index_benchmark_test.go | 62 ++++++++ pkg/replay/transaction_status_index_test.go | 104 +++++++++++++ pkg/replay/transaction_status_publication.go | 35 ++++- ...ction_status_publication_benchmark_test.go | 8 +- 7 files changed, 439 insertions(+), 25 deletions(-) create mode 100644 pkg/replay/transaction_status_index.go create mode 100644 pkg/replay/transaction_status_index_benchmark_test.go create mode 100644 pkg/replay/transaction_status_index_test.go diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index d20ddc2d4..7078d81e5 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -14,7 +14,7 @@ AMD Ryzen 7 9700X (Zen 5), Go 1.26.4, GOMAXPROCS=2. Tests ran in a separate proc Each block has 33,760 unique prepared message identities spread across one or four recent blockhashes. Existing-group cases seed 33,760 different ancestor transactions. Fixture creation, hashing, seeding and unwind are untimed. Existing maps retain capacity after unwind: the first timed commit's growth is amortized across the ten iterations. This does not model an index growing indefinitely across live blocks. -The frozen baseline functions exactly match alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions use the same prepared identities, parent/duplicate checks and fixtures. +These historical measurements used the benchmark at `23e18d81`, whose baseline functions matched alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions used the same prepared identities, parent/duplicate checks and fixtures. The current `legacy` helper uses the current visible-index representation; reproduce this historical comparison at that commit, not by treating today’s helper as a frozen index baseline. | Recent blockhash groups | Parent has keys in these groups | Baseline commit | Sized maps, inline | Preparation + commit, no overlap | Commit after preparation | |---|---|---:|---:|---:|---:| @@ -71,3 +71,71 @@ Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five sample Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. A separate 180-second pre-change live trace observed 728 publications. In the 145 publications taking at least 1 ms, the repeated scan measured 1.805 ms median / 3.180 ms maximum; insertion 2.198 / 6.774 ms. Lock acquisition was at most 0.0058 ms across all publications, and the preparation join at most 0.0010 ms. This latency-selected cohort is not a fixed transaction-size sample or a before/after p99 comparison. Probe overhead is included. These measurements identify removable work; they do not establish a sustained FAST improvement. Raw traces, native test windows and exact combined source stay on the validator host at `/srv/mithril-status-validation-20260915`. + +## Partitioned visible status index + +Large blockhash groups use 64 smaller reference-count maps, selected by the low +six bits of the stored message key's first byte. During delta preparation, unique +keys are grouped by partition in private scratch space; publication then updates +one partition at a time. Groups starting below 1,024 keys retain one map for their +lifetime, avoiding a full-index copy when they grow. All visible-index reads and +writes retain the existing cache lock. No extra publication worker is introduced. + +The immutable per-bank deltas and MTS2 checkpoint bytes are unchanged. Restore +reconstructs this derived index from those deltas; unwind removes the same bank +references. Preparation does not authorize a block, persist a vote, or extend a +checkpoint's durable coverage. Commit still checks identity binding, complete +coverage, parent lineage and the ancestor-validation receipt. Changed key-slice +offsets discard both the prepared delta and its partition batches. These are the +same duplicate-prevention and crash-recovery guarantees as before this change. + +Incremental Zen 5 comparison against the previously deployed combined validator +(binary SHA256 `2c3628ad62a313a9dcf13878cafcb6fbd8f5fa12cd8637038de762758489c651`), +not the full PR against alpenglow-dev: Go 1.26.4, GOMAXPROCS=2, Nice=19, +200% CPU quota on the active validator host, three samples of 30 iterations. +Each block contains 33,760 unique identities. Values are medians of sample means. + +| Blockhash groups | Existing groups | Validated commit before → after | Preparation + commit, without overlap | +|---|---|---|---| +| 1 | No | 1.492 → 0.852 ms | 3.422 → 3.479 ms | +| 1 | Yes | 1.425 → 1.126 ms | 4.676 → 4.977 ms | +| 4 | No | 1.151 → 0.869 ms | 3.072 → 4.306 ms | +| 4 | Yes | 1.281 → 0.984 ms | 4.346 → 5.812 ms | + +Publication improves in these samples, but total preparation work increases, +especially with multiple blockhashes. Scratch costs roughly 20 bytes per unique +key plus partition metadata and is not retained in published bank nodes. The +benefit depends on execution hiding preparation without excessive contention. + +`BenchmarkStatusMapCriticalTail` measures individual validated commits with +preparation and unwind excluded. Run the identical benchmark file on both source +revisions: three samples of 150 iterations, one blockhash, nearest-rank p99. The +median of each run's p99 fell from 3.142 to 1.488 ms for new groups, but rose from +2.549 to 2.869 ms for warmed existing groups. This is not a consistent component +p99 win, and neither benchmark predicts live FAST inclusion. + +Native targeted replay/block-production race tests, vet and the validator build +passed. Coverage includes a randomized reference-count oracle with concentrated +keys, compact-group growth, expiry, snapshot restore, fork unwind, stale identity +binding and concurrent publication. Source copies, native results, exclusions for +test load and deployment metadata are retained on Zen 5 under +`/srv/mithril-status-index-20260915`. + + +Initial live trial: 958 baseline versus 182 candidate received blocks with at +least 30,000 transactions, excluding startup/native-test windows. Publication +median/p99 measured **3.607/6.901 → 2.875/6.343 ms**. All 182 candidates had +controls matched by leader, position, sender overlap and transaction/CU within +10%; the median per-block difference was **−0.737 ms publication**, **+1.100 ms +preparation**, **+0.498 ms execution**, and **−0.298 ms full assembly-to-local +serialization**. Preparation-wait p99 remained 0.001 ms. Controls are reused and +windows are unequal, so this is observational evidence, not isolated causation. + +Overall large-block p99 was **119.050 → 123.871 ms**; an overall tail improvement +is not established. Five candidate admission outliers (four empty blocks) spent +31.823 ms median / 39.907 ms maximum between spool-completion entry and beginning +delivery, before status publication. Their deeper cause is not yet established +on this binary. Two initial five-minute captures contained 1,911 inclusions in +1,933 unique observed FAST proofs (98.86%); startup is included in this operational +score, and it is not a before/after FAST comparison. Keep the candidate under +monitoring; the status-stage gain alone does not establish the final p99 goal. diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index bb4a157fb..07a66e00a 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -85,7 +85,7 @@ func (n *transactionStatusNode) copyInto(copy *transactionStatusNode, parent *tr type visibleTransactionStatusGroup struct { keyIndex uint8 - keys map[transactionStatusKey]uint16 + keys transactionStatusIndex } // TransactionStatusCache is replay's authoritative, fork-aware @@ -590,7 +590,7 @@ func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, continue } key := sliceTransactionStatusKey(identity.MessageHash, group.keyIndex) - if group.keys[key] == 0 { + if group.keys.count(key) == 0 { continue } if already == nil { @@ -661,13 +661,16 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan } delta := transactionStatusDelta(nil) + var indexBatches map[solana.Hash]*transactionStatusIndexBatch if prepared != nil && prepared.identities == plan.messageIdentities { delta = prepared.delta + indexBatches = prepared.indexBatches // A restore or branch transition can change a blockhash's slice offset. // Rebuild from full identities if any current group uses another offset. for blockhash, group := range delta { if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { delta = nil + indexBatches = nil break } } @@ -683,7 +686,7 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan delta = buildTransactionStatusDelta(plan.messageIdentities, counts, indexes) } - if err := c.addDeltaVisibleLocked(delta); err != nil { + if err := c.addDeltaVisibleBatchesLocked(delta, indexBatches); err != nil { return err } c.tip = &transactionStatusNode{ @@ -852,6 +855,10 @@ func (c *TransactionStatusCache) validateParentLocked(block *b.Block) error { } func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDelta) error { + return c.addDeltaVisibleBatchesLocked(delta, nil) +} + +func (c *TransactionStatusCache) addDeltaVisibleBatchesLocked(delta transactionStatusDelta, batches map[solana.Hash]*transactionStatusIndexBatch) error { c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { @@ -864,12 +871,16 @@ func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDe if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, - keys: make(map[transactionStatusKey]uint16, len(deltaGroup.keys)), } + group.keys.init(len(deltaGroup.keys)) c.visible[blockhash] = group } - for key := range deltaGroup.keys { - group.keys[key]++ + if batch := batches[blockhash]; batch != nil { + group.keys.addBatch(batch) + } else { + for key := range deltaGroup.keys { + group.keys.add(key) + } } } return nil @@ -883,13 +894,9 @@ func (c *TransactionStatusCache) removeDeltaVisibleLocked(delta transactionStatu continue } for key := range deltaGroup.keys { - if group.keys[key] <= 1 { - delete(group.keys, key) - } else { - group.keys[key]-- - } + group.keys.remove(key) } - if len(group.keys) == 0 { + if group.keys.empty() { delete(c.visible, blockhash) } } @@ -989,14 +996,15 @@ func (c *TransactionStatusCache) expireVisibleLocked(expired, retained []*transa if len(g.survivors) == 0 { delete(c.visible, hash) } else if g.retainedKeys < g.expiredKeys { - rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex, keys: make(map[transactionStatusKey]uint16)} + rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex} + rebuilt.keys.init(g.retainedKeys) for _, delta := range g.survivors { for key := range delta.keys { - rebuilt.keys[key]++ + rebuilt.keys.add(key) } } c.visible[hash] = rebuilt - if len(rebuilt.keys) == 0 { + if rebuilt.keys.empty() { delete(c.visible, hash) } } diff --git a/pkg/replay/transaction_status_index.go b/pkg/replay/transaction_status_index.go new file mode 100644 index 000000000..a7c044b0e --- /dev/null +++ b/pkg/replay/transaction_status_index.go @@ -0,0 +1,147 @@ +package replay + +// Partition only the mutable lookup index, never the immutable bank deltas or +// checkpoint format. Message-hash bytes select a partition; even adversarially +// concentrated keys retain exactly the same membership/reference-count rules. +// Grouping prepared updates by partition keeps a smaller working set hot while +// publishing. All access is still protected by TransactionStatusCache.mu; this +// introduces neither background publication nor additional mutation workers. +// Recovery guarantee: this is a derived in-memory index. Snapshots still store +// the same immutable per-bank keys; restore rebuilds the counts from those keys. +// No checkpoint coverage, duplicate-check, vote persistence, or unwind rule is +// relaxed, and no prepared batch becomes authoritative before commit succeeds. +const transactionStatusIndexPartitions = 64 + +type transactionStatusIndex []map[transactionStatusKey]uint16 + +// Keep small groups in one map. The chosen layout remains fixed for the group +// lifetime: growing an existing group never forces a full-index copy on replay. +func (index *transactionStatusIndex) init(expected int) { + if len(*index) != 0 { + return + } + partitions := 1 + if expected >= 1024 { + partitions = transactionStatusIndexPartitions + } + *index = make(transactionStatusIndex, partitions) +} + +func statusIndexPartition(key transactionStatusKey) int { + return int(key[0]) & (transactionStatusIndexPartitions - 1) +} + +func (index *transactionStatusIndex) count(key transactionStatusKey) uint16 { + if len(*index) == 0 { + return 0 + } + return (*index)[int(key[0])&(len(*index)-1)][key] +} + +func (index *transactionStatusIndex) add(key transactionStatusKey) { + index.init(1) + partition := int(key[0]) & (len(*index) - 1) + if (*index)[partition] == nil { + (*index)[partition] = make(map[transactionStatusKey]uint16) + } + (*index)[partition][key]++ +} + +func (index *transactionStatusIndex) remove(key transactionStatusKey) { + if len(*index) == 0 { + return + } + number := int(key[0]) & (len(*index) - 1) + partition := (*index)[number] + if partition[key] <= 1 { + delete(partition, key) + } else { + partition[key]-- + } + if len(partition) == 0 { + (*index)[number] = nil + } +} + +func (index *transactionStatusIndex) empty() bool { + for _, partition := range *index { + if len(partition) != 0 { + return false + } + } + return true +} + +// A batch is private preparation scratch, not retained in a bank node or +// serialized. Build it from the deduplicated immutable delta so collisions in +// the stored 20-byte key still contribute only once per bank, as before. +type transactionStatusIndexBatch struct { + keys []transactionStatusKey + ends [transactionStatusIndexPartitions]int +} + +func prepareStatusIndexBatch(group *transactionStatusGroup) *transactionStatusIndexBatch { + batch := &transactionStatusIndexBatch{keys: make([]transactionStatusKey, 0, len(group.keys))} + for key := range group.keys { + batch.append(key) + } + batch.partition() + return batch +} + +func (batch *transactionStatusIndexBatch) append(key transactionStatusKey) { + batch.keys = append(batch.keys, key) + batch.ends[statusIndexPartition(key)]++ +} + +// Counting partition in place: each swap fills one destination position. This +// avoids a second key array and repeated iteration over the immutable key map. +func (batch *transactionStatusIndexBatch) partition() { + var positions [transactionStatusIndexPartitions]int + for i := 1; i < len(batch.ends); i++ { + batch.ends[i] += batch.ends[i-1] + positions[i] = batch.ends[i-1] + } + for bucket, end := range batch.ends { + for positions[bucket] < end { + at := positions[bucket] + key := batch.keys[at] + destination := statusIndexPartition(key) + if destination == bucket { + positions[bucket]++ + continue + } + to := positions[destination] + batch.keys[at], batch.keys[to] = batch.keys[to], key + positions[destination]++ + } + } +} + +func (index *transactionStatusIndex) addBatch(batch *transactionStatusIndexBatch) { + index.init(len(batch.keys)) + if len(*index) == 1 { + if (*index)[0] == nil && len(batch.keys) > 0 { + (*index)[0] = make(map[transactionStatusKey]uint16, len(batch.keys)) + } + for _, key := range batch.keys { + (*index)[0][key]++ + } + return + } + start := 0 + for i, end := range batch.ends { + if start == end { + continue + } + partition := (*index)[i] + if partition == nil { + partition = make(map[transactionStatusKey]uint16, end-start) + (*index)[i] = partition + } + for _, key := range batch.keys[start:end] { + partition[key]++ + } + start = end + } +} diff --git a/pkg/replay/transaction_status_index_benchmark_test.go b/pkg/replay/transaction_status_index_benchmark_test.go new file mode 100644 index 000000000..f8f11358b --- /dev/null +++ b/pkg/replay/transaction_status_index_benchmark_test.go @@ -0,0 +1,62 @@ +package replay + +import ( + "fmt" + "sort" + "testing" + "time" +) + +// Measure the publication tail separately from preparation and ancestor checks. +// Use the identical benchmark file on both source revisions. Includes binding +// checks and index publication; excludes fixture creation, preparation and unwind. +// A warmed existing blockhash index is kept across iterations. This is an +// isolated component benchmark, not a prediction of live voting percentiles. +func BenchmarkStatusMapCriticalTail(b *testing.B) { + for _, existing := range []bool{false, true} { + b.Run(fmt.Sprintf("existing_%t", existing), func(b *testing.B) { + txs := benchmarkUniqueTransactions(67520) + parent := statusCacheTestBlock(10, txs[:33760]...) + if !existing { + parent.Transactions = nil + } + block := statusCacheTestBlock(11, txs[33760:]...) + plan, err := planBlockTransactionExecution(block) + if err != nil { + b.Fatal(err) + } + cache := NewTransactionStatusCache() + if err := cache.CommitBlock(parent); err != nil { + b.Fatal(err) + } + durations := make([]int64, 0, b.N) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + b.StopTimer() + prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) + receipt, err := cache.validateBlockForPublication(block, plan) + if err != nil { + b.Fatal(err) + } + b.StartTimer() + start := time.Now() + err = cache.commitBlockWithValidation(block, plan, prepared, receipt) + duration := time.Since(start).Nanoseconds() + b.StopTimer() + if err != nil { + b.Fatal(err) + } + durations = append(durations, duration) + if err := cache.Unwind(11); err != nil { + b.Fatal(err) + } + b.StartTimer() + } + b.StopTimer() + sort.Slice(durations, func(i, j int) bool { return durations[i] < durations[j] }) + b.ReportMetric(float64(durations[(len(durations)-1)/2]), "p50-ns") + b.ReportMetric(float64(durations[(99*len(durations)+99)/100-1]), "p99-ns") + }) + } +} diff --git a/pkg/replay/transaction_status_index_test.go b/pkg/replay/transaction_status_index_test.go new file mode 100644 index 000000000..7fba9c88d --- /dev/null +++ b/pkg/replay/transaction_status_index_test.go @@ -0,0 +1,104 @@ +package replay + +import ( + "math/rand" + "testing" +) + +func TestTransactionStatusPartitionedIndexReferenceCounts(t *testing.T) { + for _, concentrated := range []bool{false, true} { + rng := rand.New(rand.NewSource(71)) + var index transactionStatusIndex + index.init(33760) + reference := make(map[transactionStatusKey]uint16) + keys := make([]transactionStatusKey, 1024) + for i := range keys { + rng.Read(keys[i][:]) + if concentrated { + keys[i][0] = 255 + } + } + var banks []*transactionStatusGroup + for step := 0; step < 600; step++ { + if len(banks) > 0 && (step%3 == 0 || len(banks) == 300) { + group := banks[0] + banks = banks[1:] + for k := range group.keys { + index.remove(k) + reference[k]-- + if reference[k] == 0 { + delete(reference, k) + } + } + } else { + group := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} + for i := 0; i < 100; i++ { + group.keys[keys[rng.Intn(len(keys))]] = struct{}{} + } + index.addBatch(prepareStatusIndexBatch(group)) + banks = append(banks, group) + for k := range group.keys { + reference[k]++ + } + } + for _, k := range keys { + if got := index.count(k); got != reference[k] { + t.Fatalf("concentrated=%t step=%d count=%d want=%d", concentrated, step, got, reference[k]) + } + } + } + for _, group := range banks { + for k := range group.keys { + index.remove(k) + } + } + if !index.empty() { + t.Fatal("index retained keys after all bank references removed") + } + } +} + +func TestTransactionStatusIndexBatchPartitionEdges(t *testing.T) { + for _, first := range []byte{0, 63, 64, 127, 255} { + group := &transactionStatusGroup{keys: map[transactionStatusKey]struct{}{{first, 1}: {}, {first, 2}: {}}} + batch := prepareStatusIndexBatch(group) + var index transactionStatusIndex + index.addBatch(batch) + index.addBatch(batch) + for k := range group.keys { + if index.count(k) != 2 { + t.Fatal("lost overlapping bank reference") + } + index.remove(k) + if index.count(k) != 1 { + t.Fatal("removed key still required by another bank") + } + index.remove(k) + } + if !index.empty() { + t.Fatal("partition did not empty") + } + index.addBatch(prepareStatusIndexBatch(&transactionStatusGroup{})) + if !index.empty() { + t.Fatal("empty batch introduced keys") + } + } +} + +func TestTransactionStatusIndexSmallGroupDoesNotRepartition(t *testing.T) { + var index transactionStatusIndex + index.add(transactionStatusKey{0, 1}) + group := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} + for i := 0; i < 4096; i++ { + group.keys[transactionStatusKey{byte(i), byte(i >> 8)}] = struct{}{} + } + index.addBatch(prepareStatusIndexBatch(group)) + if len(index) != 1 { + t.Fatal("existing small group copied into a new layout during publication") + } + for k := range group.keys { + if index.count(k) == 0 { + t.Fatal("batch lost a key when using compact layout") + } + } +} diff --git a/pkg/replay/transaction_status_publication.go b/pkg/replay/transaction_status_publication.go index e821071ad..5203fa17c 100644 --- a/pkg/replay/transaction_status_publication.go +++ b/pkg/replay/transaction_status_publication.go @@ -11,8 +11,9 @@ import ( // Preparation owns private, immutable maps. It never publishes a status or // authorizes a bank: commit still checks coverage, lineage, and duplicates. type preparedTransactionStatusDelta struct { - identities *b.PreparedTransactionMessageIdentities - delta transactionStatusDelta + identities *b.PreparedTransactionMessageIdentities + delta transactionStatusDelta + indexBatches map[solana.Hash]*transactionStatusIndexBatch } type transactionStatusPreparation struct { @@ -56,14 +57,33 @@ func countTransactionStatusGroups(identities *b.PreparedTransactionMessageIdenti } func buildTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8) transactionStatusDelta { + return buildTransactionStatusDeltaWithBatches(identities, counts, indexes, nil) +} + +func buildTransactionStatusDeltaWithBatches(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8, batches map[solana.Hash]*transactionStatusIndexBatch) transactionStatusDelta { delta := make(transactionStatusDelta, len(counts)) for blockhash, count := range counts { delta[blockhash] = &transactionStatusGroup{keyIndex: indexes[blockhash], keys: make(map[transactionStatusKey]struct{}, count)} + if batches != nil && count >= 1024 { + batches[blockhash] = &transactionStatusIndexBatch{keys: make([]transactionStatusKey, 0, count)} + } } + var previous solana.Hash + var group *transactionStatusGroup + var batch *transactionStatusIndexBatch for i := 0; i < identities.Len(); i++ { identity := identities.Identity(i) - group := delta[identity.RecentBlockhash] - group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} + if group == nil || identity.RecentBlockhash != previous { + previous = identity.RecentBlockhash + group = delta[previous] + batch = batches[previous] + } + key := sliceTransactionStatusKey(identity.MessageHash, group.keyIndex) + before := len(group.keys) + group.keys[key] = struct{}{} + if batch != nil && len(group.keys) != before { + batch.append(key) + } } return delta } @@ -79,5 +99,10 @@ func (c *TransactionStatusCache) prepareTransactionStatusDelta(identities *b.Pre } } c.mu.RUnlock() - return &preparedTransactionStatusDelta{identities: identities, delta: buildTransactionStatusDelta(identities, counts, indexes)} + batches := make(map[solana.Hash]*transactionStatusIndexBatch, len(counts)) + delta := buildTransactionStatusDeltaWithBatches(identities, counts, indexes, batches) + for _, batch := range batches { + batch.partition() + } + return &preparedTransactionStatusDelta{identities: identities, delta: delta, indexBatches: batches} } diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go index 2fc7d0f45..b6b73685c 100644 --- a/pkg/replay/transaction_status_publication_benchmark_test.go +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -61,8 +61,9 @@ func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, return nil } -// Frozen production commit algorithm before publication optimization. This is -// an independent baseline, including its original visible-index allocation. +// Historical unprepared delta construction, using the current visible index. +// For before/after index comparisons run the same benchmark at both commits; +// this helper is not a frozen baseline for the mutable index implementation. func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { @@ -75,12 +76,11 @@ func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionSt if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, - keys: make(map[transactionStatusKey]uint16), } c.visible[blockhash] = group } for key := range deltaGroup.keys { - group.keys[key]++ + group.keys.add(key) } } return nil From b8df19d61ae02670f364a655242bbc4b261a6960 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:32:17 -0500 Subject: [PATCH 014/111] review: defer status-map partitioning and archive investigation artifacts --- .gitattributes | 7 - .../pr-split-2026-09-15/status/README.md | 7 - .../status/status-tests.log | 1 - .../pr-split-2026-09-15/status/status-vet.log | 0 .../baseline-failure-comparison.json | 26 ---- .../2026-09-14/checkpoint-live.json | 118 -------------- .../checkpoint-native-benchmark.log | 15 -- .../2026-09-14/zen5-benchmark.log | 24 --- .../status-cache/2026-09-14/zen5-race.log | 1 - .../status-cache/2026-09-14/zen5-vet.log | 0 .../2026-09-15/local-race-final.log | 3 - .../2026-09-15/native-benchmark-summary.json | 82 ---------- .../2026-09-15/native-benchmark.log | 85 ---------- .../2026-09-15/native-build.log | 0 .../2026-09-15/native-final-run.log | 12 -- .../native-overlap-final-summary.json | 104 ------------- .../2026-09-15/native-overlap-final.log | 95 ----------- .../2026-09-15/native-overlap-summary.json | 104 ------------- .../2026-09-15/native-overlap.log | 102 ------------ .../2026-09-15/native-race.log | 3 - .../2026-09-15/native-vet.log | 0 .../2026-09-15/native-window.json | 4 - .../2026-09-15/post-test-health.json | 1 - .../2026-09-15/recheck-abba.log | 34 ---- .../2026-09-15/tested-source.json | 16 -- docs/status-checkpoint-capture.md | 5 +- docs/status-checkpoint-expiry-evidence.md | 19 +++ docs/transaction-status-expiry.md | 19 +-- docs/transaction-status-publication.md | 72 +-------- pkg/replay/transaction_status_cache.go | 38 ++--- pkg/replay/transaction_status_index.go | 147 ------------------ ...transaction_status_index_benchmark_test.go | 62 -------- pkg/replay/transaction_status_index_test.go | 104 ------------- pkg/replay/transaction_status_publication.go | 35 +---- ...ction_status_publication_benchmark_test.go | 8 +- 35 files changed, 51 insertions(+), 1302 deletions(-) delete mode 100644 .gitattributes delete mode 100644 docs/results/pr-split-2026-09-15/status/README.md delete mode 100644 docs/results/pr-split-2026-09-15/status/status-tests.log delete mode 100644 docs/results/pr-split-2026-09-15/status/status-vet.log delete mode 100644 docs/results/status-cache/2026-09-14/baseline-failure-comparison.json delete mode 100644 docs/results/status-cache/2026-09-14/checkpoint-live.json delete mode 100644 docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log delete mode 100644 docs/results/status-cache/2026-09-14/zen5-benchmark.log delete mode 100644 docs/results/status-cache/2026-09-14/zen5-race.log delete mode 100644 docs/results/status-cache/2026-09-14/zen5-vet.log delete mode 100644 docs/results/status-publication/2026-09-15/local-race-final.log delete mode 100644 docs/results/status-publication/2026-09-15/native-benchmark-summary.json delete mode 100644 docs/results/status-publication/2026-09-15/native-benchmark.log delete mode 100644 docs/results/status-publication/2026-09-15/native-build.log delete mode 100644 docs/results/status-publication/2026-09-15/native-final-run.log delete mode 100644 docs/results/status-publication/2026-09-15/native-overlap-final-summary.json delete mode 100644 docs/results/status-publication/2026-09-15/native-overlap-final.log delete mode 100644 docs/results/status-publication/2026-09-15/native-overlap-summary.json delete mode 100644 docs/results/status-publication/2026-09-15/native-overlap.log delete mode 100644 docs/results/status-publication/2026-09-15/native-race.log delete mode 100644 docs/results/status-publication/2026-09-15/native-vet.log delete mode 100644 docs/results/status-publication/2026-09-15/native-window.json delete mode 100644 docs/results/status-publication/2026-09-15/post-test-health.json delete mode 100644 docs/results/status-publication/2026-09-15/recheck-abba.log delete mode 100644 docs/results/status-publication/2026-09-15/tested-source.json create mode 100644 docs/status-checkpoint-expiry-evidence.md delete mode 100644 pkg/replay/transaction_status_index.go delete mode 100644 pkg/replay/transaction_status_index_benchmark_test.go delete mode 100644 pkg/replay/transaction_status_index_test.go diff --git a/.gitattributes b/.gitattributes deleted file mode 100644 index d64360d9a..000000000 --- a/.gitattributes +++ /dev/null @@ -1,7 +0,0 @@ -# Keep raw benchmark evidence available without overwhelming the review. -# Retain raw Go CPU padding and test-framework space/tab indentation. -docs/results/**/*.json linguist-generated=true -docs/results/**/*.jsonl linguist-generated=true -docs/results/**/*.txt linguist-generated=true whitespace=-blank-at-eol -docs/results/**/*.log linguist-generated=true -whitespace -docs/results/**/*.tar.gz linguist-generated=true diff --git a/docs/results/pr-split-2026-09-15/status/README.md b/docs/results/pr-split-2026-09-15/status/README.md deleted file mode 100644 index 134adf562..000000000 --- a/docs/results/pr-split-2026-09-15/status/README.md +++ /dev/null @@ -1,7 +0,0 @@ -# Review branch validation, September 15 - -This PR is split from #279 plus the later working-tree improvements. The tested code commit before this documentation commit was `a111ba892fb88ea95768930f56dd2f9a5091e6ce`. Tests ran locally on Apple M4 Pro, Go1.26.4, GOMAXPROCS3 and package parallelism2. Logs beside this file are the fresh split-branch checks, not native measurements. Existing Zen5 benchmark documents retain their original baselines and scope. No validator restart or deployment occurred during this reorganization. - -Four independent branches start at current alpenglow-dev33dde405. Voting is based on the certificate-processing PR; leader packing is based on the streaming-preparation PR. Runtime changes and status-cache changes are independent. Shared CLI/configuration additions need an ordinary three-file merge reconciliation when combining leader packing and voting. A separate audit checkout reconciled these additions and matched the preserved full implementation exactly across Go sources, module files, TOML configuration and CI. - -The branch-specific race suites and vet passed. The combined audit has a separately documented pre-existing intermittent peer reconnect timeout; this is not reported as an entirely green combined race run. diff --git a/docs/results/pr-split-2026-09-15/status/status-tests.log b/docs/results/pr-split-2026-09-15/status/status-tests.log deleted file mode 100644 index b6b40ef23..000000000 --- a/docs/results/pr-split-2026-09-15/status/status-tests.log +++ /dev/null @@ -1 +0,0 @@ -ok github.com/Overclock-Validator/mithril/pkg/replay 4.657s diff --git a/docs/results/pr-split-2026-09-15/status/status-vet.log b/docs/results/pr-split-2026-09-15/status/status-vet.log deleted file mode 100644 index e69de29bb..000000000 diff --git a/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json b/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json deleted file mode 100644 index 5353cc6eb..000000000 --- a/docs/results/status-cache/2026-09-14/baseline-failure-comparison.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "failed_tests": [ - "TestExecute_Tx_BpfLoader_Write_Success", - "TestExecute_Tx_BpfLoader_Write_Offset_Too_Large_Failure", - "TestExecute_Tx_BpfLoader_Write_Buffer_Authority_Didnt_Sign_Failure", - "TestExecute_Tx_BpfLoader_Write_Incorrect_Authority_Failure", - "TestExecute_Tx_BpfLoader_SetAuthority_Not_Enough_Instr_Accts_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Not_Enough_Instr_Accts_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Success", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Immutable_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Wrong_Upgrade_Authority_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Authority_Didnt_Sign_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_New_Authority_Didnt_Sign_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Uninitialized_Account_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Immutable_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Authority_Didnt_Sign_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_New_Authority_Didnt_Sign_Failure", - "TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Wrong_Authority_Failure", - "TestExecute_Tx_BpfLoader_Close_Buffer_Not_Enough_Accounts", - "TestExecute_Tx_BpfLoader_Close_ProgramData_Success" - ], - "same_failed_test_names": true, - "same_panic_function": "UpgradeableLoaderClose", - "same_failed_test_names_on_native_base": true -} diff --git a/docs/results/status-cache/2026-09-14/checkpoint-live.json b/docs/results/status-cache/2026-09-14/checkpoint-live.json deleted file mode 100644 index 0072e9189..000000000 --- a/docs/results/status-cache/2026-09-14/checkpoint-live.json +++ /dev/null @@ -1,118 +0,0 @@ -{ - "utc": "2026-09-14T03:34:45Z", - "count": 10, - "log_bytes": 1607896, - "rows": [ - { - "slots": 128, - "through": 3427284, - "worker_ms": 331.0, - "capture_us": 34.174, - "encode_ms": 236.108854, - "bytes": 31488298, - "line": "(+ 28s) async fold: committed 128 slots through 3427284 in 331ms checkpoint_capture=34.174\u00b5s checkpoint_encode=236.108854ms checkpoint_bytes=31488298" - }, - { - "slots": 128, - "through": 3427428, - "worker_ms": 264.0, - "capture_us": 30.366, - "encode_ms": 182.148884, - "bytes": 25219543, - "line": "(+ 59s) async fold: committed 128 slots through 3427428 in 264ms checkpoint_capture=30.366\u00b5s checkpoint_encode=182.148884ms checkpoint_bytes=25219543" - }, - { - "slots": 128, - "through": 3427569, - "worker_ms": 318.0, - "capture_us": 30.417, - "encode_ms": 225.601155, - "bytes": 29886172, - "line": "(+ 1m29s) async fold: committed 128 slots through 3427569 in 318ms checkpoint_capture=30.417\u00b5s checkpoint_encode=225.601155ms checkpoint_bytes=29886172" - }, - { - "slots": 128, - "through": 3427701, - "worker_ms": 258.0, - "capture_us": 29.495, - "encode_ms": 179.07318000000004, - "bytes": 24530250, - "line": "(+ 1m58s) async fold: committed 128 slots through 3427701 in 258ms checkpoint_capture=29.495\u00b5s checkpoint_encode=179.07318ms checkpoint_bytes=24530250" - }, - { - "slots": 128, - "through": 3427841, - "worker_ms": 305.0, - "capture_us": 30.888, - "encode_ms": 216.143473, - "bytes": 29469092, - "line": "(+ 2m28s) async fold: committed 128 slots through 3427841 in 305ms checkpoint_capture=30.888\u00b5s checkpoint_encode=216.143473ms checkpoint_bytes=29469092" - }, - { - "slots": 128, - "through": 3427989, - "worker_ms": 286.0, - "capture_us": 28.694, - "encode_ms": 201.186937, - "bytes": 27782921, - "line": "(+ 2m59s) async fold: committed 128 slots through 3427989 in 286ms checkpoint_capture=28.694\u00b5s checkpoint_encode=201.186937ms checkpoint_bytes=27782921" - }, - { - "slots": 128, - "through": 3428133, - "worker_ms": 376.0, - "capture_us": 28.874, - "encode_ms": 278.613922, - "bytes": 35692638, - "line": "(+ 3m29s) async fold: committed 128 slots through 3428133 in 376ms checkpoint_capture=28.874\u00b5s checkpoint_encode=278.613922ms checkpoint_bytes=35692638" - }, - { - "slots": 128, - "through": 3428269, - "worker_ms": 431.0, - "capture_us": 32.671, - "encode_ms": 329.865906, - "bytes": 37615609, - "line": "(+ 3m58s) async fold: committed 128 slots through 3428269 in 431ms checkpoint_capture=32.671\u00b5s checkpoint_encode=329.865906ms checkpoint_bytes=37615609" - }, - { - "slots": 128, - "through": 3428409, - "worker_ms": 403.0, - "capture_us": 31.108, - "encode_ms": 301.569555, - "bytes": 38403988, - "line": "(+ 4m28s) async fold: committed 128 slots through 3428409 in 403ms checkpoint_capture=31.108\u00b5s checkpoint_encode=301.569555ms checkpoint_bytes=38403988" - }, - { - "slots": 128, - "through": 3428549, - "worker_ms": 477.0, - "capture_us": 34.464, - "encode_ms": 366.581817, - "bytes": 43443732, - "line": "(+ 4m58s) async fold: committed 128 slots through 3428549 in 477ms checkpoint_capture=34.464\u00b5s checkpoint_encode=366.581817ms checkpoint_bytes=43443732" - } - ], - "capture_us": { - "min": 28.694, - "median": 30.652500000000003, - "max": 34.464 - }, - "encode_ms": { - "min": 179.07318000000004, - "median": 230.8550045, - "max": 366.581817 - }, - "bytes": { - "min": 24530250, - "median": 30687235.0, - "max": 43443732 - }, - "worker_ms": { - "min": 258.0, - "median": 324.5, - "max": 477.0 - }, - "error_lines": [] -} diff --git a/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log b/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log deleted file mode 100644 index 33f4c2cd6..000000000 --- a/docs/results/status-cache/2026-09-14/checkpoint-native-benchmark.log +++ /dev/null @@ -1,15 +0,0 @@ -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 206554597 ns/op 99120032 B/op 2731 allocs/op -BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 207352232 ns/op 99120072 B/op 2732 allocs/op -BenchmarkTransactionStatusCheckpointCapture/SynchronousBaseline 1 202442401 ns/op 99120048 B/op 2731 allocs/op -BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 47412 5936 ns/op 35168 B/op 11 allocs/op -BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 39506 6057 ns/op 35168 B/op 11 allocs/op -BenchmarkTransactionStatusCheckpointCapture/CaptureOnReplay 37850 5922 ns/op 35168 B/op 11 allocs/op -BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 2 202522908 ns/op 99108084 B/op 2723 allocs/op -BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 2 201841085 ns/op 99108076 B/op 2723 allocs/op -BenchmarkTransactionStatusCheckpointCapture/EncodeOnWorker 1 200834149 ns/op 99108104 B/op 2724 allocs/op -PASS -ok github.com/Overclock-Validator/mithril/pkg/replay 3.133s diff --git a/docs/results/status-cache/2026-09-14/zen5-benchmark.log b/docs/results/status-cache/2026-09-14/zen5-benchmark.log deleted file mode 100644 index 0229c19af..000000000 --- a/docs/results/status-cache/2026-09-14/zen5-benchmark.log +++ /dev/null @@ -1,24 +0,0 @@ -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 308302685 ns/op -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 311104834 ns/op -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=true-2 1 306059722 ns/op -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 56685 ns/op -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 49743 ns/op -BenchmarkTransactionStatusBatchExpiry/four-bank-groups/legacy=false-2 1 48581 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 700015562 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 718445604 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=true-2 1 716785553 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 30758 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 26119 ns/op -BenchmarkTransactionStatusBatchExpiry/one-expired-group/legacy=false-2 1 24436 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 734956959 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 722327226 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=true-2 1 717188987 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 2045462 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 1845859 ns/op -BenchmarkTransactionStatusBatchExpiry/crossing-group/legacy=false-2 1 1983167 ns/op -PASS -ok github.com/Overclock-Validator/mithril/pkg/replay 22.725s diff --git a/docs/results/status-cache/2026-09-14/zen5-race.log b/docs/results/status-cache/2026-09-14/zen5-race.log deleted file mode 100644 index 100c86a11..000000000 --- a/docs/results/status-cache/2026-09-14/zen5-race.log +++ /dev/null @@ -1 +0,0 @@ -ok github.com/Overclock-Validator/mithril/pkg/replay 2.189s diff --git a/docs/results/status-cache/2026-09-14/zen5-vet.log b/docs/results/status-cache/2026-09-14/zen5-vet.log deleted file mode 100644 index e69de29bb..000000000 diff --git a/docs/results/status-publication/2026-09-15/local-race-final.log b/docs/results/status-publication/2026-09-15/local-race-final.log deleted file mode 100644 index 17b7d4d2a..000000000 --- a/docs/results/status-publication/2026-09-15/local-race-final.log +++ /dev/null @@ -1,3 +0,0 @@ -ok github.com/Overclock-Validator/mithril/pkg/replay 4.618s -ok github.com/Overclock-Validator/mithril/pkg/block 1.774s -? github.com/Overclock-Validator/mithril/pkg/metrics [no test files] diff --git a/docs/results/status-publication/2026-09-15/native-benchmark-summary.json b/docs/results/status-publication/2026-09-15/native-benchmark-summary.json deleted file mode 100644 index 3936921d1..000000000 --- a/docs/results/status-publication/2026-09-15/native-benchmark-summary.json +++ /dev/null @@ -1,82 +0,0 @@ -{ - "BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2": { - "ns/op": 4990269.0, - "B/op": 6302003.0, - "allocs/op": 555.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2": { - "ns/op": 3331600.0, - "B/op": 3151475.0, - "allocs/op": 265.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2": { - "ns/op": 3393252.0, - "B/op": 3151718.0, - "allocs/op": 269.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2": { - "ns/op": 1587430.0, - "B/op": 1575587.0, - "allocs/op": 132.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2": { - "ns/op": 4991180.0, - "B/op": 3466193.0, - "allocs/op": 304.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2": { - "ns/op": 4657032.0, - "B/op": 1891049.0, - "allocs/op": 159.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2": { - "ns/op": 4434153.0, - "B/op": 1891281.0, - "allocs/op": 163.0 - }, - "BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2": { - "ns/op": 2756954.0, - "B/op": 315161.0, - "allocs/op": 26.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2": { - "ns/op": 4624988.0, - "B/op": 6301427.0, - "allocs/op": 659.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2": { - "ns/op": 3743889.0, - "B/op": 3151859.0, - "allocs/op": 283.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2": { - "ns/op": 5915974.0, - "B/op": 3152091.0, - "allocs/op": 287.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2": { - "ns/op": 2205044.0, - "B/op": 1575779.0, - "allocs/op": 141.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2": { - "ns/op": 5224367.0, - "B/op": 3465532.0, - "allocs/op": 357.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2": { - "ns/op": 4644107.0, - "B/op": 1891228.0, - "allocs/op": 169.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2": { - "ns/op": 4949162.0, - "B/op": 1891460.0, - "allocs/op": 173.0 - }, - "BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2": { - "ns/op": 2858236.0, - "B/op": 315148.0, - "allocs/op": 27.0 - } -} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-benchmark.log b/docs/results/status-publication/2026-09-15/native-benchmark.log deleted file mode 100644 index ea4b94603..000000000 --- a/docs/results/status-publication/2026-09-15/native-benchmark.log +++ /dev/null @@ -1,85 +0,0 @@ -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4971792 ns/op 6302003 B/op 555 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 5003752 ns/op 6302003 B/op 555 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4990269 ns/op 6302003 B/op 555 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 5067117 ns/op 6302003 B/op 555 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/legacy-2 10 4855132 ns/op 6302003 B/op 555 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3869972 ns/op 3151475 B/op 265 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3593875 ns/op 3151477 B/op 265 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3116164 ns/op 3151475 B/op 265 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3292134 ns/op 3151475 B/op 265 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/sized-2 10 3331600 ns/op 3151475 B/op 265 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3459366 ns/op 3151780 B/op 269 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3555785 ns/op 3151707 B/op 269 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3338994 ns/op 3151718 B/op 269 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3393252 ns/op 3151755 B/op 269 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_total-2 10 3035989 ns/op 3151707 B/op 269 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1433079 ns/op 1575587 B/op 132 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1465220 ns/op 1575587 B/op 132 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1587430 ns/op 1575587 B/op 132 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1668259 ns/op 1575587 B/op 132 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_false/prepared_commit-2 10 1625507 ns/op 1575587 B/op 132 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 5038118 ns/op 3466193 B/op 304 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4926609 ns/op 3466193 B/op 304 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4965933 ns/op 3466196 B/op 304 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 4991180 ns/op 3466193 B/op 304 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/legacy-2 10 5174867 ns/op 3466193 B/op 304 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4592110 ns/op 1891049 B/op 159 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4689758 ns/op 1891049 B/op 159 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4579822 ns/op 1891049 B/op 159 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4893416 ns/op 1891052 B/op 159 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/sized-2 10 4657032 ns/op 1891049 B/op 159 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4617374 ns/op 1891281 B/op 163 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4434153 ns/op 1891281 B/op 163 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4425112 ns/op 1891281 B/op 163 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4362413 ns/op 1891281 B/op 163 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_total-2 10 4649669 ns/op 1891281 B/op 163 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2756954 ns/op 315161 B/op 26 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2987364 ns/op 315161 B/op 26 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2747530 ns/op 315161 B/op 26 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 3033686 ns/op 315161 B/op 26 allocs/op -BenchmarkTransactionStatusPublication/groups_1/existing_true/prepared_commit-2 10 2735607 ns/op 315161 B/op 26 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4829304 ns/op 6301427 B/op 659 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4554119 ns/op 6301427 B/op 659 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4920607 ns/op 6301427 B/op 659 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4624988 ns/op 6301427 B/op 659 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 10 4587116 ns/op 6301427 B/op 659 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3398182 ns/op 3151859 B/op 283 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3626107 ns/op 3151859 B/op 283 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 3743889 ns/op 3151859 B/op 283 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 4029608 ns/op 3151861 B/op 283 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/sized-2 10 4958660 ns/op 3151859 B/op 283 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5919178 ns/op 3152091 B/op 287 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5632307 ns/op 3152091 B/op 287 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5473052 ns/op 3152091 B/op 287 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 5915974 ns/op 3152091 B/op 287 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 10 6688551 ns/op 3152091 B/op 287 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2205247 ns/op 1575779 B/op 141 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2480043 ns/op 1575779 B/op 141 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2205044 ns/op 1575779 B/op 141 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 1927871 ns/op 1575779 B/op 141 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_commit-2 10 2036015 ns/op 1575779 B/op 141 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 7297434 ns/op 3465532 B/op 357 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5048127 ns/op 3465532 B/op 357 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5224367 ns/op 3465532 B/op 357 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 5277320 ns/op 3465532 B/op 357 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/legacy-2 10 4971670 ns/op 3465535 B/op 357 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 5041342 ns/op 1891228 B/op 169 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4491682 ns/op 1891228 B/op 169 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4966235 ns/op 1891228 B/op 169 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4644107 ns/op 1891228 B/op 169 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/sized-2 10 4479055 ns/op 1891228 B/op 169 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4914007 ns/op 1891511 B/op 173 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 5031889 ns/op 1891508 B/op 173 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 5046504 ns/op 1891460 B/op 173 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4496234 ns/op 1891460 B/op 173 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_total-2 10 4949162 ns/op 1891460 B/op 173 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2603068 ns/op 315148 B/op 27 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2859800 ns/op 315148 B/op 27 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2858236 ns/op 315148 B/op 27 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 3113986 ns/op 315148 B/op 27 allocs/op -BenchmarkTransactionStatusPublication/groups_4/existing_true/prepared_commit-2 10 2788538 ns/op 315148 B/op 27 allocs/op -PASS diff --git a/docs/results/status-publication/2026-09-15/native-build.log b/docs/results/status-publication/2026-09-15/native-build.log deleted file mode 100644 index e69de29bb..000000000 diff --git a/docs/results/status-publication/2026-09-15/native-final-run.log b/docs/results/status-publication/2026-09-15/native-final-run.log deleted file mode 100644 index c6d28bfba..000000000 --- a/docs/results/status-publication/2026-09-15/native-final-run.log +++ /dev/null @@ -1,12 +0,0 @@ -Running as unit: mithril-status-publication-final-20260915.service -race 0 -vet 0 -build 0 -benchmark-build 0 -benchmark 0 -overlap-final 0 - Finished with result: success -Main processes terminated with: code=exited, status=0/SUCCESS - Service runtime: 37.879s - CPU time consumed: 45.681s - Memory peak: 634.8M (swap: 0B) diff --git a/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json b/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json deleted file mode 100644 index 7f2b767e3..000000000 --- a/docs/results/status-publication/2026-09-15/native-overlap-final-summary.json +++ /dev/null @@ -1,104 +0,0 @@ -{ - "BenchmarkTransactionStatusExecutionOverlap/legacy": { - "ns/op": 23370663.0, - "commit-with-wait-ns/op": 6062517.0, - "execution-ns/op": 17289566.0, - "B/op": 23276118.0, - "allocs/op": 242222.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/legacy-2": { - "ns/op": 20677847.0, - "commit-with-wait-ns/op": 5394296.0, - "execution-ns/op": 15069600.0, - "B/op": 23276627.0, - "allocs/op": 242225.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/sized": { - "ns/op": 21296975.0, - "commit-with-wait-ns/op": 4299800.0, - "execution-ns/op": 17436468.0, - "B/op": 20125587.0, - "allocs/op": 241932.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/sized-2": { - "ns/op": 18573776.0, - "commit-with-wait-ns/op": 4299496.0, - "execution-ns/op": 14369209.0, - "B/op": 20126012.0, - "allocs/op": 241934.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/overlap": { - "ns/op": 22188363.0, - "commit-with-wait-ns/op": 4616295.0, - "execution-ns/op": 17694970.0, - "B/op": 20125587.0, - "allocs/op": 241932.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/overlap-2": { - "ns/op": 17090947.0, - "commit-with-wait-ns/op": 2119131.0, - "execution-ns/op": 14966507.0, - "B/op": 20126212.0, - "allocs/op": 241938.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/legacy": { - "ns/op": 95.12, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2": { - "ns/op": 74.71, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total": { - "ns/op": 119.4, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2": { - "ns/op": 100.0, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/legacy": { - "ns/op": 593.0, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2": { - "ns/op": 466.8, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total": { - "ns/op": 708.3, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2": { - "ns/op": 609.1, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/legacy": { - "ns/op": 6085.0, - "B/op": 6320.0, - "allocs/op": 23.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2": { - "ns/op": 5145.0, - "B/op": 6320.0, - "allocs/op": 23.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total": { - "ns/op": 4456.0, - "B/op": 3616.0, - "allocs/op": 13.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2": { - "ns/op": 3770.0, - "B/op": 3616.0, - "allocs/op": 13.0 - } -} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-overlap-final.log b/docs/results/status-publication/2026-09-15/native-overlap-final.log deleted file mode 100644 index a1e297aaf..000000000 --- a/docs/results/status-publication/2026-09-15/native-overlap-final.log +++ /dev/null @@ -1,95 +0,0 @@ -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusExecutionOverlap/legacy 5 24692874 ns/op 6513086 commit-with-wait-ns/op 18179262 execution-ns/op 23276118 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 22834712 ns/op 5772168 commit-with-wait-ns/op 17062174 execution-ns/op 23276118 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 23370663 ns/op 6486422 commit-with-wait-ns/op 16883798 execution-ns/op 23276136 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 23246803 ns/op 5956655 commit-with-wait-ns/op 17289566 execution-ns/op 23276145 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 24078611 ns/op 6062517 commit-with-wait-ns/op 18015548 execution-ns/op 23276059 B/op 242221 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20677847 ns/op 5607605 commit-with-wait-ns/op 15069600 execution-ns/op 23276630 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20222041 ns/op 5394296 commit-with-wait-ns/op 14827324 execution-ns/op 23276627 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19971750 ns/op 5078614 commit-with-wait-ns/op 14892735 execution-ns/op 23276532 B/op 242224 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 22424538 ns/op 5368235 commit-with-wait-ns/op 17055698 execution-ns/op 23276712 B/op 242226 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20975910 ns/op 5790017 commit-with-wait-ns/op 15184788 execution-ns/op 23276624 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 21296975 ns/op 3860111 commit-with-wait-ns/op 17436468 execution-ns/op 20125587 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 21108482 ns/op 4260022 commit-with-wait-ns/op 16848123 execution-ns/op 20125592 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 6 20802767 ns/op 4299800 commit-with-wait-ns/op 16502635 execution-ns/op 20125558 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 6 22758328 ns/op 5033044 commit-with-wait-ns/op 17724954 execution-ns/op 20125533 B/op 241931 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 23246396 ns/op 5467937 commit-with-wait-ns/op 17778082 execution-ns/op 20125614 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 19020036 ns/op 4526523 commit-with-wait-ns/op 14493163 execution-ns/op 20126028 B/op 241935 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18422586 ns/op 4167392 commit-with-wait-ns/op 14254792 execution-ns/op 20126009 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18573776 ns/op 4299496 commit-with-wait-ns/op 14273834 execution-ns/op 20126012 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18332367 ns/op 3962772 commit-with-wait-ns/op 14369209 execution-ns/op 20126006 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 19268730 ns/op 4381208 commit-with-wait-ns/op 14886998 execution-ns/op 20126038 B/op 241935 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 6 21422337 ns/op 4616295 commit-with-wait-ns/op 16804831 execution-ns/op 20125514 B/op 241931 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 21450447 ns/op 3882218 commit-with-wait-ns/op 17567431 execution-ns/op 20125593 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 23556986 ns/op 5081649 commit-with-wait-ns/op 18474472 execution-ns/op 20125593 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 22188363 ns/op 4492746 commit-with-wait-ns/op 17694970 execution-ns/op 20125587 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 6 22888561 ns/op 4725085 commit-with-wait-ns/op 18162634 execution-ns/op 20125536 B/op 241931 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17090947 ns/op 2119131 commit-with-wait-ns/op 14966507 execution-ns/op 20126253 B/op 241938 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17313847 ns/op 2127423 commit-with-wait-ns/op 15181536 execution-ns/op 20126212 B/op 241938 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 17116780 ns/op 1984282 commit-with-wait-ns/op 15127492 execution-ns/op 20126437 B/op 241939 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 16485753 ns/op 1917207 commit-with-wait-ns/op 14563704 execution-ns/op 20126176 B/op 241938 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16709920 ns/op 2210525 commit-with-wait-ns/op 14495070 execution-ns/op 20126125 B/op 241937 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1299566 94.83 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1284991 93.05 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1247547 96.51 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1211845 96.60 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1238605 95.12 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1596568 76.21 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1614400 74.16 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1581337 76.14 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1574422 74.71 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1589880 74.16 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 123.0 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 119.6 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 118.8 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 119.4 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 1000000 117.4 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1000000 100.4 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1231928 97.12 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1000000 101.3 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1200760 99.52 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 1206331 100.0 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 189308 586.1 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 200305 607.5 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 188314 597.3 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 196060 586.6 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 197784 593.0 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 227380 470.5 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 240021 461.5 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 231192 472.9 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 254556 466.8 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 239902 443.4 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167852 716.6 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167379 708.2 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 167563 708.3 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 169233 699.7 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 164168 735.9 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 190630 609.1 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 190263 615.6 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 176863 604.2 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 196165 632.1 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 185583 588.6 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 17916 6009 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19801 6085 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19857 6155 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19308 6126 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 20053 6002 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23298 5038 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23206 5158 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 22210 5145 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23406 5305 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23084 5114 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26672 4343 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26752 4456 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 27342 4471 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 26960 4312 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 27775 4592 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 31856 3644 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32778 3717 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32239 3784 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 30372 3818 ns/op 3616 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 32263 3770 ns/op 3616 B/op 13 allocs/op -PASS diff --git a/docs/results/status-publication/2026-09-15/native-overlap-summary.json b/docs/results/status-publication/2026-09-15/native-overlap-summary.json deleted file mode 100644 index efa0017aa..000000000 --- a/docs/results/status-publication/2026-09-15/native-overlap-summary.json +++ /dev/null @@ -1,104 +0,0 @@ -{ - "BenchmarkTransactionStatusExecutionOverlap/legacy": { - "ns/op": 22833095.0, - "commit-with-wait-ns/op": 5848843.0, - "execution-ns/op": 16779392.0, - "B/op": 23276116.0, - "allocs/op": 242222.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/legacy-2": { - "ns/op": 19163664.0, - "commit-with-wait-ns/op": 5114189.0, - "execution-ns/op": 14034884.0, - "B/op": 23276650.0, - "allocs/op": 242225.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/sized": { - "ns/op": 21164557.0, - "commit-with-wait-ns/op": 4745710.0, - "execution-ns/op": 17057597.0, - "B/op": 20125556.0, - "allocs/op": 241932.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/sized-2": { - "ns/op": 18233386.0, - "commit-with-wait-ns/op": 4143998.0, - "execution-ns/op": 14088971.0, - "B/op": 20126009.0, - "allocs/op": 241934.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/overlap": { - "ns/op": 21351433.0, - "commit-with-wait-ns/op": 3481211.0, - "execution-ns/op": 18286875.0, - "B/op": 20125816.0, - "allocs/op": 241936.0 - }, - "BenchmarkTransactionStatusExecutionOverlap/overlap-2": { - "ns/op": 16814463.0, - "commit-with-wait-ns/op": 2297246.0, - "execution-ns/op": 14538161.0, - "B/op": 20126252.0, - "allocs/op": 241938.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/legacy": { - "ns/op": 94.81, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2": { - "ns/op": 76.19, - "B/op": 112.0, - "allocs/op": 2.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total": { - "ns/op": 181.7, - "B/op": 264.0, - "allocs/op": 5.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2": { - "ns/op": 142.9, - "B/op": 264.0, - "allocs/op": 5.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/legacy": { - "ns/op": 810.9, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2": { - "ns/op": 633.3, - "B/op": 960.0, - "allocs/op": 9.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total": { - "ns/op": 1645.0, - "B/op": 1192.0, - "allocs/op": 13.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2": { - "ns/op": 1511.0, - "B/op": 1192.0, - "allocs/op": 13.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/legacy": { - "ns/op": 6067.0, - "B/op": 6320.0, - "allocs/op": 23.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2": { - "ns/op": 5064.0, - "B/op": 6320.0, - "allocs/op": 23.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total": { - "ns/op": 5837.0, - "B/op": 3848.0, - "allocs/op": 17.0 - }, - "BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2": { - "ns/op": 5961.0, - "B/op": 3848.0, - "allocs/op": 17.0 - } -} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/native-overlap.log b/docs/results/status-publication/2026-09-15/native-overlap.log deleted file mode 100644 index e347eed1c..000000000 --- a/docs/results/status-publication/2026-09-15/native-overlap.log +++ /dev/null @@ -1,102 +0,0 @@ -Running as unit: mithril-status-overlap-bench-20260915.service -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusExecutionOverlap/legacy 6 22659962 ns/op 5699140 commit-with-wait-ns/op 16960453 execution-ns/op 23276112 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 22853486 ns/op 5584087 commit-with-wait-ns/op 17269100 execution-ns/op 23276118 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 23170780 ns/op 6502884 commit-with-wait-ns/op 16667603 execution-ns/op 23276136 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 22833095 ns/op 6053272 commit-with-wait-ns/op 16779392 execution-ns/op 23276116 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy 5 22058768 ns/op 5848843 commit-with-wait-ns/op 16209621 execution-ns/op 23276112 B/op 242222 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19456453 ns/op 5114189 commit-with-wait-ns/op 14341895 execution-ns/op 23276694 B/op 242226 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 5 20721459 ns/op 5463814 commit-with-wait-ns/op 15257144 execution-ns/op 23276633 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 18883924 ns/op 5031514 commit-with-wait-ns/op 13851997 execution-ns/op 23276650 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 19163664 ns/op 5128481 commit-with-wait-ns/op 14034884 execution-ns/op 23276584 B/op 242225 allocs/op -BenchmarkTransactionStatusExecutionOverlap/legacy-2 6 18648556 ns/op 4971799 commit-with-wait-ns/op 13676368 execution-ns/op 23276654 B/op 242226 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 6 20730994 ns/op 3673079 commit-with-wait-ns/op 17057597 execution-ns/op 20125561 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 21164557 ns/op 4762666 commit-with-wait-ns/op 16401568 execution-ns/op 20125556 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 20803959 ns/op 3673651 commit-with-wait-ns/op 17130021 execution-ns/op 20125587 B/op 241932 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 21585622 ns/op 4745710 commit-with-wait-ns/op 16839541 execution-ns/op 20125537 B/op 241931 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized 5 22061011 ns/op 4786467 commit-with-wait-ns/op 17274047 execution-ns/op 20125526 B/op 241931 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18481807 ns/op 4332538 commit-with-wait-ns/op 14148935 execution-ns/op 20126009 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18233386 ns/op 4143998 commit-with-wait-ns/op 14088971 execution-ns/op 20125942 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 17926782 ns/op 3900186 commit-with-wait-ns/op 14026293 execution-ns/op 20126008 B/op 241935 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 17641308 ns/op 3846237 commit-with-wait-ns/op 13794796 execution-ns/op 20126009 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/sized-2 6 18874807 ns/op 4403265 commit-with-wait-ns/op 14471150 execution-ns/op 20126009 B/op 241934 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 6 21101004 ns/op 2063760 commit-with-wait-ns/op 19033430 execution-ns/op 20125788 B/op 241936 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 21351433 ns/op 3481211 commit-with-wait-ns/op 17866676 execution-ns/op 20125816 B/op 241936 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 21378252 ns/op 3800343 commit-with-wait-ns/op 17574134 execution-ns/op 20125816 B/op 241936 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 20630833 ns/op 1688017 commit-with-wait-ns/op 18939909 execution-ns/op 20125819 B/op 241936 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap 5 22262745 ns/op 3971910 commit-with-wait-ns/op 18286875 execution-ns/op 20125816 B/op 241936 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16698061 ns/op 2155589 commit-with-wait-ns/op 14538161 execution-ns/op 20126252 B/op 241938 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 7 16586794 ns/op 2151713 commit-with-wait-ns/op 14431196 execution-ns/op 20126259 B/op 241938 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 16814463 ns/op 2297246 commit-with-wait-ns/op 14512878 execution-ns/op 20126390 B/op 241939 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17545014 ns/op 2416401 commit-with-wait-ns/op 15124436 execution-ns/op 20126122 B/op 241937 allocs/op -BenchmarkTransactionStatusExecutionOverlap/overlap-2 6 17149268 ns/op 2323257 commit-with-wait-ns/op 14821130 execution-ns/op 20126128 B/op 241937 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1273947 96.94 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1294056 91.35 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1271734 92.04 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1300807 94.81 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy 1272133 96.92 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1589078 76.19 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1577804 77.30 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1563583 77.34 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1626802 74.57 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/legacy-2 1591627 75.31 ns/op 112 B/op 2 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 721636 185.7 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 704294 179.3 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 725799 182.1 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 725127 181.7 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total 700783 176.9 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 726256 138.7 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 728829 139.5 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 861189 144.8 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 765130 143.7 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_0/prepared_total-2 751879 142.9 ns/op 264 B/op 5 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 194295 653.8 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 176815 648.9 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 174601 882.8 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 149839 829.7 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy 162600 810.9 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 242948 633.3 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 140822 720.2 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 156864 754.5 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 235440 462.3 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/legacy-2 215985 471.1 ns/op 960 B/op 9 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 66573 1705 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 71836 1640 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 70480 1655 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 71564 1641 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total 73791 1645 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 77616 1489 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 78930 1520 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 75838 1511 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 74458 1491 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_1/prepared_total-2 79232 1551 ns/op 1192 B/op 13 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19609 6019 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 18758 6100 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19924 6118 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19128 6067 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy 19795 6000 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23222 5027 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23133 5064 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 23482 5243 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 22717 5013 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/legacy-2 24062 5630 ns/op 6320 B/op 23 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 20682 6564 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 18136 5879 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 21844 5837 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 19086 5635 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total 19254 5724 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 20040 5546 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 18378 5860 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 19862 5961 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 19912 6906 ns/op 3848 B/op 17 allocs/op -BenchmarkTransactionStatusSmallPublication/txs_32/prepared_total-2 21711 7037 ns/op 3848 B/op 17 allocs/op -PASS -ok github.com/Overclock-Validator/mithril/pkg/replay 17.258s - Finished with result: success -Main processes terminated with: code=exited, status=0/SUCCESS - Service runtime: 19.605s - CPU time consumed: 23.467s - Memory peak: 387.6M (swap: 0B) diff --git a/docs/results/status-publication/2026-09-15/native-race.log b/docs/results/status-publication/2026-09-15/native-race.log deleted file mode 100644 index 00504d333..000000000 --- a/docs/results/status-publication/2026-09-15/native-race.log +++ /dev/null @@ -1,3 +0,0 @@ -ok github.com/Overclock-Validator/mithril/pkg/replay 2.267s -ok github.com/Overclock-Validator/mithril/pkg/block 1.133s -? github.com/Overclock-Validator/mithril/pkg/metrics [no test files] diff --git a/docs/results/status-publication/2026-09-15/native-vet.log b/docs/results/status-publication/2026-09-15/native-vet.log deleted file mode 100644 index e69de29bb..000000000 diff --git a/docs/results/status-publication/2026-09-15/native-window.json b/docs/results/status-publication/2026-09-15/native-window.json deleted file mode 100644 index f283937ee..000000000 --- a/docs/results/status-publication/2026-09-15/native-window.json +++ /dev/null @@ -1,4 +0,0 @@ -{ - "start": "2026-09-15T02:57:29.581963+00:00", - "end": "2026-09-15T02:58:07.245339+00:00" -} \ No newline at end of file diff --git a/docs/results/status-publication/2026-09-15/post-test-health.json b/docs/results/status-publication/2026-09-15/post-test-health.json deleted file mode 100644 index df730357e..000000000 --- a/docs/results/status-publication/2026-09-15/post-test-health.json +++ /dev/null @@ -1 +0,0 @@ -{"utc": "2026-09-15T02:58:13.874813+00:00", "health": {"utc": "2026-09-15T02:58:13.949166+00:00", "pid": 568418, "rpc_slot": 3824460, "local_slot": 3824460, "last_vote": 3824459, "vote_lag": 1}, "pid": "MainPID=568418", "services": ["active", "active", "active", "active"]} diff --git a/docs/results/status-publication/2026-09-15/recheck-abba.log b/docs/results/status-publication/2026-09-15/recheck-abba.log deleted file mode 100644 index 9720620a0..000000000 --- a/docs/results/status-publication/2026-09-15/recheck-abba.log +++ /dev/null @@ -1,34 +0,0 @@ -Running as unit: mithril-status-publication-recheck-20260915.service; invocation ID: 8079e456482443308a68c660fb8f6b29 -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 50 4400299 ns/op 6301401 B/op 659 allocs/op -PASS - -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 50 2984879 ns/op 3152087 B/op 287 allocs/op -PASS - -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusPublication/groups_4/existing_false/prepared_total-2 50 3131031 ns/op 3152091 B/op 287 allocs/op -PASS - -goos: linux -goarch: amd64 -pkg: github.com/Overclock-Validator/mithril/pkg/replay -cpu: AMD Ryzen 7 9700X 8-Core Processor -BenchmarkTransactionStatusPublication/groups_4/existing_false/legacy-2 50 4564884 ns/op 6301399 B/op 659 allocs/op -PASS - - Finished with result: success -Main processes terminated with: code=exited, status=0/SUCCESS - Service runtime: 1.278s - CPU time consumed: 1.456s - Memory peak: 73.9M (swap: 0B) diff --git a/docs/results/status-publication/2026-09-15/tested-source.json b/docs/results/status-publication/2026-09-15/tested-source.json deleted file mode 100644 index c4889f6a2..000000000 --- a/docs/results/status-publication/2026-09-15/tested-source.json +++ /dev/null @@ -1,16 +0,0 @@ -{ - "base": "33dde4050d9250557583395810799aaac2f54017", - "branch_before_change": "31e0d8c0", - "files": { - "pkg/replay/transaction_status_overlap_benchmark_test.go": "6310ce381e6e88151b14a7f0e5f56e7370a8e1d94af466911dd69707379d8cdb", - "pkg/replay/transaction_status_publication.go": "a94c1227ce0b0e329dfff390c2303bb36525493e0cde3fce7d3fe177eaa1681b", - "pkg/replay/transaction_status_publication_benchmark_test.go": "0ff3a1f0c2a8d0ceafb5d7ba300c6cb5358082fd019f75a4b6fe0c35fde1575e", - "pkg/replay/transaction_status_publication_test.go": "eae775d2d10e6ca5b13b1e6f201cff15d8d0174b40ff20a69aa4279004799ca1", - "pkg/metrics/metrics.go": "17005056d872f9b8acf75fee15dc2c175bc4addad1b2dafee61445124dfbe307", - "pkg/replay/block.go": "19e7e931ec5e8aaab2e910808cb2f6e19f5721ff2ad953541f42828a88076a0d", - "pkg/replay/transaction_status_cache.go": "731febda7fdfe09b84d4281c53c3d0f4381e8f7d67d9d47f084c8a10c1523935", - "pkg/replay/transaction_status_plan_binding_test.go": "6c1fe9c92af2011402025ca96601847a1313bf2ffb564082eb06aa89088e250b", - "pkg/replay/transaction_status_prepared_test.go": "1944b443624d1075f6ea3e89630cdca42a6245a95a3116bbd47bd670311ad876" - }, - "baseline_check": "Frozen commitBlockWithPlan and addDeltaVisibleLocked exactly match alpenglow-dev at the recorded base after renaming benchmark helper methods." -} diff --git a/docs/status-checkpoint-capture.md b/docs/status-checkpoint-capture.md index 62a4dc0e8..8ba391cb5 100644 --- a/docs/status-checkpoint-capture.md +++ b/docs/status-checkpoint-capture.md @@ -68,9 +68,8 @@ encoder with memoization, not the whole status-publication change against dev. The default-cadence result is approximately 2.3x, with the same 99.12 → 57.33 MB allocation reduction. Cold/all-new windows remain roughly unchanged. Native -combined race suites, vet and the validator build passed. These are staging -measurements: the encoding cache has not been deployed, so a live reduction in -durable-root lag or missed FAST votes has not yet been established. +combined race suites, vet and the validator build passed. These are historical staging measurements. They do not establish an isolated +live reduction in durable-root lag or missed FAST votes. ## Fold admission before collecting account writes diff --git a/docs/status-checkpoint-expiry-evidence.md b/docs/status-checkpoint-expiry-evidence.md new file mode 100644 index 000000000..22661d482 --- /dev/null +++ b/docs/status-checkpoint-expiry-evidence.md @@ -0,0 +1,19 @@ +# Status Checkpoint Expiry: benchmark evidence + +The maintained subsystem documentation and reusable Go benchmarks describe the +implementation and reproduction method. Historical raw results and session +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/blob/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf) +(tag `review-evidence-20260916-status-checkpoint-expiry`). They are omitted from this proposed merge. + +[Historical result files](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results) + +Measurements retain their original baselines. Rebasing onto PR #278 does not +turn an intermediate-version benchmark into a comparison with the new base. +Component timings and short live observations do not establish sustained FAST +inclusion gains. The final review description records validation of the rebased +source separately from historical benchmark results. + +The tagged snapshot also preserves the later 64-partition visible-status-map +experiment. That experiment is deliberately excluded from this review: it added +preparation work and did not demonstrate an overall large-block p99 benefit. +Earlier publication preparation and immutable-node encoding reuse remain. diff --git a/docs/transaction-status-expiry.md b/docs/transaction-status-expiry.md index 179202360..538acb0f8 100644 --- a/docs/transaction-status-expiry.md +++ b/docs/transaction-status-expiry.md @@ -42,18 +42,9 @@ measurements, not end-to-end Root/replay or a prediction of live FAST scores. Run `go test ./pkg/replay -run '^$' -bench '^BenchmarkTransactionStatusBatchExpiry$' -benchtime=1x -count=2`. -Prepared on `7layer/status-expiry-performance` above the isolated Votor fix. -This source is not the exact live FEC-integrated source. No deployment or public -PR change is implied by these local results. - -## Zen 5 validation — 20:05 UTC - -Native AMD Ryzen 7 9700X tests used an isolated copy of the preserved live -FEC-integrated source at `/srv/mithril-status-expiry-test-20260914/source`. -The original status-cache file was byte-identical to the change's parent. -Only the reviewed cache source and new tests were overlaid, with SHA256 checks. -The full replay race suite passed (2.189 s), and vet passed. No binary was deployed. +## Native benchmark +Ryzen 7 9700X, Go 1.26.4, original per-key expiry versus batched expiry. Benchmarks ran with GOMAXPROCS=2, nice=15, one caller and three iterations per case, while the validator and loader remained active. Setup and later GC are excluded from the expiry timer. Each case expires 4,321,280 entries (128 banks @@ -66,7 +57,5 @@ live stall samples and are not an end-to-end replay or FAST-score comparison. | One fully expired blockhash group | 700–718 ms | 0.024–0.031 ms | | Group crossing the retention boundary | 717–735 ms | 1.85–2.05 ms | -At 20:05:13 UTC the enrolled validator PID 291548 was at RPC/local slot 3,708,093, -last vote 3,708,092. Loader unpaused; validator, loader, FAST and Titan services -all active. Source/implementation and deployment status remain unchanged. -See zen5-benchmark.log, zen5-race.log, zen5-vet.log and zen5-health.json. +[Historical evidence](status-checkpoint-expiry-evidence.md) preserves the original +source revisions, raw measurements and validation. diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index 7078d81e5..a099af906 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -14,7 +14,7 @@ AMD Ryzen 7 9700X (Zen 5), Go 1.26.4, GOMAXPROCS=2. Tests ran in a separate proc Each block has 33,760 unique prepared message identities spread across one or four recent blockhashes. Existing-group cases seed 33,760 different ancestor transactions. Fixture creation, hashing, seeding and unwind are untimed. Existing maps retain capacity after unwind: the first timed commit's growth is amortized across the ten iterations. This does not model an index growing indefinitely across live blocks. -These historical measurements used the benchmark at `23e18d81`, whose baseline functions matched alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions used the same prepared identities, parent/duplicate checks and fixtures. The current `legacy` helper uses the current visible-index representation; reproduce this historical comparison at that commit, not by treating today’s helper as a frozen index baseline. +The frozen baseline functions exactly match alpenglow-dev commit `33dde4050d9250557583395810799aaac2f54017`. Both versions use the same prepared identities, parent/duplicate checks and fixtures. | Recent blockhash groups | Parent has keys in these groups | Baseline commit | Sized maps, inline | Preparation + commit, no overlap | Commit after preparation | |---|---|---:|---:|---:|---:| @@ -39,7 +39,7 @@ The initial unrestricted version showed no additional total-time benefit from ov Full replay and block race suites passed on both Zen 5 and M4 Pro. Metrics has no tests. Native vet for replay/metrics and the validator build passed. Tests cover fork replacement introducing a duplicate after preparation, concurrent sibling publication, stale identity binding, changed snapshot slice offsets, rejected/incomplete banks, mismatched preparation, pinned views, snapshot restore, unwind, empty banks and scheduling boundaries. -Raw logs, source hashes, summaries and the alternating recheck are in [results/status-publication/2026-09-15](results/status-publication/2026-09-15). The baseline comparison covers only status publication. No live replay or FAST improvement is claimed. The staging binary was not deployed; the existing validator remained active and voting throughout the tests. +Raw logs, source hashes, summaries and the alternating recheck are in [results/status-publication/2026-09-15](https://github.com/Overclock-Validator/mithril/blob/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results/status-publication/2026-09-15). The baseline comparison covers only status publication. No live replay or FAST improvement is claimed. The staging binary was not deployed; the existing validator remained active and voting throughout the tests. Reproduce from this branch: @@ -71,71 +71,3 @@ Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five sample Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. A separate 180-second pre-change live trace observed 728 publications. In the 145 publications taking at least 1 ms, the repeated scan measured 1.805 ms median / 3.180 ms maximum; insertion 2.198 / 6.774 ms. Lock acquisition was at most 0.0058 ms across all publications, and the preparation join at most 0.0010 ms. This latency-selected cohort is not a fixed transaction-size sample or a before/after p99 comparison. Probe overhead is included. These measurements identify removable work; they do not establish a sustained FAST improvement. Raw traces, native test windows and exact combined source stay on the validator host at `/srv/mithril-status-validation-20260915`. - -## Partitioned visible status index - -Large blockhash groups use 64 smaller reference-count maps, selected by the low -six bits of the stored message key's first byte. During delta preparation, unique -keys are grouped by partition in private scratch space; publication then updates -one partition at a time. Groups starting below 1,024 keys retain one map for their -lifetime, avoiding a full-index copy when they grow. All visible-index reads and -writes retain the existing cache lock. No extra publication worker is introduced. - -The immutable per-bank deltas and MTS2 checkpoint bytes are unchanged. Restore -reconstructs this derived index from those deltas; unwind removes the same bank -references. Preparation does not authorize a block, persist a vote, or extend a -checkpoint's durable coverage. Commit still checks identity binding, complete -coverage, parent lineage and the ancestor-validation receipt. Changed key-slice -offsets discard both the prepared delta and its partition batches. These are the -same duplicate-prevention and crash-recovery guarantees as before this change. - -Incremental Zen 5 comparison against the previously deployed combined validator -(binary SHA256 `2c3628ad62a313a9dcf13878cafcb6fbd8f5fa12cd8637038de762758489c651`), -not the full PR against alpenglow-dev: Go 1.26.4, GOMAXPROCS=2, Nice=19, -200% CPU quota on the active validator host, three samples of 30 iterations. -Each block contains 33,760 unique identities. Values are medians of sample means. - -| Blockhash groups | Existing groups | Validated commit before → after | Preparation + commit, without overlap | -|---|---|---|---| -| 1 | No | 1.492 → 0.852 ms | 3.422 → 3.479 ms | -| 1 | Yes | 1.425 → 1.126 ms | 4.676 → 4.977 ms | -| 4 | No | 1.151 → 0.869 ms | 3.072 → 4.306 ms | -| 4 | Yes | 1.281 → 0.984 ms | 4.346 → 5.812 ms | - -Publication improves in these samples, but total preparation work increases, -especially with multiple blockhashes. Scratch costs roughly 20 bytes per unique -key plus partition metadata and is not retained in published bank nodes. The -benefit depends on execution hiding preparation without excessive contention. - -`BenchmarkStatusMapCriticalTail` measures individual validated commits with -preparation and unwind excluded. Run the identical benchmark file on both source -revisions: three samples of 150 iterations, one blockhash, nearest-rank p99. The -median of each run's p99 fell from 3.142 to 1.488 ms for new groups, but rose from -2.549 to 2.869 ms for warmed existing groups. This is not a consistent component -p99 win, and neither benchmark predicts live FAST inclusion. - -Native targeted replay/block-production race tests, vet and the validator build -passed. Coverage includes a randomized reference-count oracle with concentrated -keys, compact-group growth, expiry, snapshot restore, fork unwind, stale identity -binding and concurrent publication. Source copies, native results, exclusions for -test load and deployment metadata are retained on Zen 5 under -`/srv/mithril-status-index-20260915`. - - -Initial live trial: 958 baseline versus 182 candidate received blocks with at -least 30,000 transactions, excluding startup/native-test windows. Publication -median/p99 measured **3.607/6.901 → 2.875/6.343 ms**. All 182 candidates had -controls matched by leader, position, sender overlap and transaction/CU within -10%; the median per-block difference was **−0.737 ms publication**, **+1.100 ms -preparation**, **+0.498 ms execution**, and **−0.298 ms full assembly-to-local -serialization**. Preparation-wait p99 remained 0.001 ms. Controls are reused and -windows are unequal, so this is observational evidence, not isolated causation. - -Overall large-block p99 was **119.050 → 123.871 ms**; an overall tail improvement -is not established. Five candidate admission outliers (four empty blocks) spent -31.823 ms median / 39.907 ms maximum between spool-completion entry and beginning -delivery, before status publication. Their deeper cause is not yet established -on this binary. Two initial five-minute captures contained 1,911 inclusions in -1,933 unique observed FAST proofs (98.86%); startup is included in this operational -score, and it is not a before/after FAST comparison. Keep the candidate under -monitoring; the status-stage gain alone does not establish the final p99 goal. diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index 07a66e00a..bb4a157fb 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -85,7 +85,7 @@ func (n *transactionStatusNode) copyInto(copy *transactionStatusNode, parent *tr type visibleTransactionStatusGroup struct { keyIndex uint8 - keys transactionStatusIndex + keys map[transactionStatusKey]uint16 } // TransactionStatusCache is replay's authoritative, fork-aware @@ -590,7 +590,7 @@ func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, continue } key := sliceTransactionStatusKey(identity.MessageHash, group.keyIndex) - if group.keys.count(key) == 0 { + if group.keys[key] == 0 { continue } if already == nil { @@ -661,16 +661,13 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan } delta := transactionStatusDelta(nil) - var indexBatches map[solana.Hash]*transactionStatusIndexBatch if prepared != nil && prepared.identities == plan.messageIdentities { delta = prepared.delta - indexBatches = prepared.indexBatches // A restore or branch transition can change a blockhash's slice offset. // Rebuild from full identities if any current group uses another offset. for blockhash, group := range delta { if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { delta = nil - indexBatches = nil break } } @@ -686,7 +683,7 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan delta = buildTransactionStatusDelta(plan.messageIdentities, counts, indexes) } - if err := c.addDeltaVisibleBatchesLocked(delta, indexBatches); err != nil { + if err := c.addDeltaVisibleLocked(delta); err != nil { return err } c.tip = &transactionStatusNode{ @@ -855,10 +852,6 @@ func (c *TransactionStatusCache) validateParentLocked(block *b.Block) error { } func (c *TransactionStatusCache) addDeltaVisibleLocked(delta transactionStatusDelta) error { - return c.addDeltaVisibleBatchesLocked(delta, nil) -} - -func (c *TransactionStatusCache) addDeltaVisibleBatchesLocked(delta transactionStatusDelta, batches map[solana.Hash]*transactionStatusIndexBatch) error { c.invalidateValidationLocked() for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { @@ -871,16 +864,12 @@ func (c *TransactionStatusCache) addDeltaVisibleBatchesLocked(delta transactionS if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, + keys: make(map[transactionStatusKey]uint16, len(deltaGroup.keys)), } - group.keys.init(len(deltaGroup.keys)) c.visible[blockhash] = group } - if batch := batches[blockhash]; batch != nil { - group.keys.addBatch(batch) - } else { - for key := range deltaGroup.keys { - group.keys.add(key) - } + for key := range deltaGroup.keys { + group.keys[key]++ } } return nil @@ -894,9 +883,13 @@ func (c *TransactionStatusCache) removeDeltaVisibleLocked(delta transactionStatu continue } for key := range deltaGroup.keys { - group.keys.remove(key) + if group.keys[key] <= 1 { + delete(group.keys, key) + } else { + group.keys[key]-- + } } - if group.keys.empty() { + if len(group.keys) == 0 { delete(c.visible, blockhash) } } @@ -996,15 +989,14 @@ func (c *TransactionStatusCache) expireVisibleLocked(expired, retained []*transa if len(g.survivors) == 0 { delete(c.visible, hash) } else if g.retainedKeys < g.expiredKeys { - rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex} - rebuilt.keys.init(g.retainedKeys) + rebuilt := &visibleTransactionStatusGroup{keyIndex: g.survivors[0].keyIndex, keys: make(map[transactionStatusKey]uint16)} for _, delta := range g.survivors { for key := range delta.keys { - rebuilt.keys.add(key) + rebuilt.keys[key]++ } } c.visible[hash] = rebuilt - if rebuilt.keys.empty() { + if len(rebuilt.keys) == 0 { delete(c.visible, hash) } } diff --git a/pkg/replay/transaction_status_index.go b/pkg/replay/transaction_status_index.go deleted file mode 100644 index a7c044b0e..000000000 --- a/pkg/replay/transaction_status_index.go +++ /dev/null @@ -1,147 +0,0 @@ -package replay - -// Partition only the mutable lookup index, never the immutable bank deltas or -// checkpoint format. Message-hash bytes select a partition; even adversarially -// concentrated keys retain exactly the same membership/reference-count rules. -// Grouping prepared updates by partition keeps a smaller working set hot while -// publishing. All access is still protected by TransactionStatusCache.mu; this -// introduces neither background publication nor additional mutation workers. -// Recovery guarantee: this is a derived in-memory index. Snapshots still store -// the same immutable per-bank keys; restore rebuilds the counts from those keys. -// No checkpoint coverage, duplicate-check, vote persistence, or unwind rule is -// relaxed, and no prepared batch becomes authoritative before commit succeeds. -const transactionStatusIndexPartitions = 64 - -type transactionStatusIndex []map[transactionStatusKey]uint16 - -// Keep small groups in one map. The chosen layout remains fixed for the group -// lifetime: growing an existing group never forces a full-index copy on replay. -func (index *transactionStatusIndex) init(expected int) { - if len(*index) != 0 { - return - } - partitions := 1 - if expected >= 1024 { - partitions = transactionStatusIndexPartitions - } - *index = make(transactionStatusIndex, partitions) -} - -func statusIndexPartition(key transactionStatusKey) int { - return int(key[0]) & (transactionStatusIndexPartitions - 1) -} - -func (index *transactionStatusIndex) count(key transactionStatusKey) uint16 { - if len(*index) == 0 { - return 0 - } - return (*index)[int(key[0])&(len(*index)-1)][key] -} - -func (index *transactionStatusIndex) add(key transactionStatusKey) { - index.init(1) - partition := int(key[0]) & (len(*index) - 1) - if (*index)[partition] == nil { - (*index)[partition] = make(map[transactionStatusKey]uint16) - } - (*index)[partition][key]++ -} - -func (index *transactionStatusIndex) remove(key transactionStatusKey) { - if len(*index) == 0 { - return - } - number := int(key[0]) & (len(*index) - 1) - partition := (*index)[number] - if partition[key] <= 1 { - delete(partition, key) - } else { - partition[key]-- - } - if len(partition) == 0 { - (*index)[number] = nil - } -} - -func (index *transactionStatusIndex) empty() bool { - for _, partition := range *index { - if len(partition) != 0 { - return false - } - } - return true -} - -// A batch is private preparation scratch, not retained in a bank node or -// serialized. Build it from the deduplicated immutable delta so collisions in -// the stored 20-byte key still contribute only once per bank, as before. -type transactionStatusIndexBatch struct { - keys []transactionStatusKey - ends [transactionStatusIndexPartitions]int -} - -func prepareStatusIndexBatch(group *transactionStatusGroup) *transactionStatusIndexBatch { - batch := &transactionStatusIndexBatch{keys: make([]transactionStatusKey, 0, len(group.keys))} - for key := range group.keys { - batch.append(key) - } - batch.partition() - return batch -} - -func (batch *transactionStatusIndexBatch) append(key transactionStatusKey) { - batch.keys = append(batch.keys, key) - batch.ends[statusIndexPartition(key)]++ -} - -// Counting partition in place: each swap fills one destination position. This -// avoids a second key array and repeated iteration over the immutable key map. -func (batch *transactionStatusIndexBatch) partition() { - var positions [transactionStatusIndexPartitions]int - for i := 1; i < len(batch.ends); i++ { - batch.ends[i] += batch.ends[i-1] - positions[i] = batch.ends[i-1] - } - for bucket, end := range batch.ends { - for positions[bucket] < end { - at := positions[bucket] - key := batch.keys[at] - destination := statusIndexPartition(key) - if destination == bucket { - positions[bucket]++ - continue - } - to := positions[destination] - batch.keys[at], batch.keys[to] = batch.keys[to], key - positions[destination]++ - } - } -} - -func (index *transactionStatusIndex) addBatch(batch *transactionStatusIndexBatch) { - index.init(len(batch.keys)) - if len(*index) == 1 { - if (*index)[0] == nil && len(batch.keys) > 0 { - (*index)[0] = make(map[transactionStatusKey]uint16, len(batch.keys)) - } - for _, key := range batch.keys { - (*index)[0][key]++ - } - return - } - start := 0 - for i, end := range batch.ends { - if start == end { - continue - } - partition := (*index)[i] - if partition == nil { - partition = make(map[transactionStatusKey]uint16, end-start) - (*index)[i] = partition - } - for _, key := range batch.keys[start:end] { - partition[key]++ - } - start = end - } -} diff --git a/pkg/replay/transaction_status_index_benchmark_test.go b/pkg/replay/transaction_status_index_benchmark_test.go deleted file mode 100644 index f8f11358b..000000000 --- a/pkg/replay/transaction_status_index_benchmark_test.go +++ /dev/null @@ -1,62 +0,0 @@ -package replay - -import ( - "fmt" - "sort" - "testing" - "time" -) - -// Measure the publication tail separately from preparation and ancestor checks. -// Use the identical benchmark file on both source revisions. Includes binding -// checks and index publication; excludes fixture creation, preparation and unwind. -// A warmed existing blockhash index is kept across iterations. This is an -// isolated component benchmark, not a prediction of live voting percentiles. -func BenchmarkStatusMapCriticalTail(b *testing.B) { - for _, existing := range []bool{false, true} { - b.Run(fmt.Sprintf("existing_%t", existing), func(b *testing.B) { - txs := benchmarkUniqueTransactions(67520) - parent := statusCacheTestBlock(10, txs[:33760]...) - if !existing { - parent.Transactions = nil - } - block := statusCacheTestBlock(11, txs[33760:]...) - plan, err := planBlockTransactionExecution(block) - if err != nil { - b.Fatal(err) - } - cache := NewTransactionStatusCache() - if err := cache.CommitBlock(parent); err != nil { - b.Fatal(err) - } - durations := make([]int64, 0, b.N) - b.ReportAllocs() - b.ResetTimer() - for i := 0; i < b.N; i++ { - b.StopTimer() - prepared := cache.prepareTransactionStatusDelta(plan.messageIdentities) - receipt, err := cache.validateBlockForPublication(block, plan) - if err != nil { - b.Fatal(err) - } - b.StartTimer() - start := time.Now() - err = cache.commitBlockWithValidation(block, plan, prepared, receipt) - duration := time.Since(start).Nanoseconds() - b.StopTimer() - if err != nil { - b.Fatal(err) - } - durations = append(durations, duration) - if err := cache.Unwind(11); err != nil { - b.Fatal(err) - } - b.StartTimer() - } - b.StopTimer() - sort.Slice(durations, func(i, j int) bool { return durations[i] < durations[j] }) - b.ReportMetric(float64(durations[(len(durations)-1)/2]), "p50-ns") - b.ReportMetric(float64(durations[(99*len(durations)+99)/100-1]), "p99-ns") - }) - } -} diff --git a/pkg/replay/transaction_status_index_test.go b/pkg/replay/transaction_status_index_test.go deleted file mode 100644 index 7fba9c88d..000000000 --- a/pkg/replay/transaction_status_index_test.go +++ /dev/null @@ -1,104 +0,0 @@ -package replay - -import ( - "math/rand" - "testing" -) - -func TestTransactionStatusPartitionedIndexReferenceCounts(t *testing.T) { - for _, concentrated := range []bool{false, true} { - rng := rand.New(rand.NewSource(71)) - var index transactionStatusIndex - index.init(33760) - reference := make(map[transactionStatusKey]uint16) - keys := make([]transactionStatusKey, 1024) - for i := range keys { - rng.Read(keys[i][:]) - if concentrated { - keys[i][0] = 255 - } - } - var banks []*transactionStatusGroup - for step := 0; step < 600; step++ { - if len(banks) > 0 && (step%3 == 0 || len(banks) == 300) { - group := banks[0] - banks = banks[1:] - for k := range group.keys { - index.remove(k) - reference[k]-- - if reference[k] == 0 { - delete(reference, k) - } - } - } else { - group := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} - for i := 0; i < 100; i++ { - group.keys[keys[rng.Intn(len(keys))]] = struct{}{} - } - index.addBatch(prepareStatusIndexBatch(group)) - banks = append(banks, group) - for k := range group.keys { - reference[k]++ - } - } - for _, k := range keys { - if got := index.count(k); got != reference[k] { - t.Fatalf("concentrated=%t step=%d count=%d want=%d", concentrated, step, got, reference[k]) - } - } - } - for _, group := range banks { - for k := range group.keys { - index.remove(k) - } - } - if !index.empty() { - t.Fatal("index retained keys after all bank references removed") - } - } -} - -func TestTransactionStatusIndexBatchPartitionEdges(t *testing.T) { - for _, first := range []byte{0, 63, 64, 127, 255} { - group := &transactionStatusGroup{keys: map[transactionStatusKey]struct{}{{first, 1}: {}, {first, 2}: {}}} - batch := prepareStatusIndexBatch(group) - var index transactionStatusIndex - index.addBatch(batch) - index.addBatch(batch) - for k := range group.keys { - if index.count(k) != 2 { - t.Fatal("lost overlapping bank reference") - } - index.remove(k) - if index.count(k) != 1 { - t.Fatal("removed key still required by another bank") - } - index.remove(k) - } - if !index.empty() { - t.Fatal("partition did not empty") - } - index.addBatch(prepareStatusIndexBatch(&transactionStatusGroup{})) - if !index.empty() { - t.Fatal("empty batch introduced keys") - } - } -} - -func TestTransactionStatusIndexSmallGroupDoesNotRepartition(t *testing.T) { - var index transactionStatusIndex - index.add(transactionStatusKey{0, 1}) - group := &transactionStatusGroup{keys: make(map[transactionStatusKey]struct{})} - for i := 0; i < 4096; i++ { - group.keys[transactionStatusKey{byte(i), byte(i >> 8)}] = struct{}{} - } - index.addBatch(prepareStatusIndexBatch(group)) - if len(index) != 1 { - t.Fatal("existing small group copied into a new layout during publication") - } - for k := range group.keys { - if index.count(k) == 0 { - t.Fatal("batch lost a key when using compact layout") - } - } -} diff --git a/pkg/replay/transaction_status_publication.go b/pkg/replay/transaction_status_publication.go index 5203fa17c..e821071ad 100644 --- a/pkg/replay/transaction_status_publication.go +++ b/pkg/replay/transaction_status_publication.go @@ -11,9 +11,8 @@ import ( // Preparation owns private, immutable maps. It never publishes a status or // authorizes a bank: commit still checks coverage, lineage, and duplicates. type preparedTransactionStatusDelta struct { - identities *b.PreparedTransactionMessageIdentities - delta transactionStatusDelta - indexBatches map[solana.Hash]*transactionStatusIndexBatch + identities *b.PreparedTransactionMessageIdentities + delta transactionStatusDelta } type transactionStatusPreparation struct { @@ -57,33 +56,14 @@ func countTransactionStatusGroups(identities *b.PreparedTransactionMessageIdenti } func buildTransactionStatusDelta(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8) transactionStatusDelta { - return buildTransactionStatusDeltaWithBatches(identities, counts, indexes, nil) -} - -func buildTransactionStatusDeltaWithBatches(identities *b.PreparedTransactionMessageIdentities, counts map[solana.Hash]int, indexes map[solana.Hash]uint8, batches map[solana.Hash]*transactionStatusIndexBatch) transactionStatusDelta { delta := make(transactionStatusDelta, len(counts)) for blockhash, count := range counts { delta[blockhash] = &transactionStatusGroup{keyIndex: indexes[blockhash], keys: make(map[transactionStatusKey]struct{}, count)} - if batches != nil && count >= 1024 { - batches[blockhash] = &transactionStatusIndexBatch{keys: make([]transactionStatusKey, 0, count)} - } } - var previous solana.Hash - var group *transactionStatusGroup - var batch *transactionStatusIndexBatch for i := 0; i < identities.Len(); i++ { identity := identities.Identity(i) - if group == nil || identity.RecentBlockhash != previous { - previous = identity.RecentBlockhash - group = delta[previous] - batch = batches[previous] - } - key := sliceTransactionStatusKey(identity.MessageHash, group.keyIndex) - before := len(group.keys) - group.keys[key] = struct{}{} - if batch != nil && len(group.keys) != before { - batch.append(key) - } + group := delta[identity.RecentBlockhash] + group.keys[sliceTransactionStatusKey(identity.MessageHash, group.keyIndex)] = struct{}{} } return delta } @@ -99,10 +79,5 @@ func (c *TransactionStatusCache) prepareTransactionStatusDelta(identities *b.Pre } } c.mu.RUnlock() - batches := make(map[solana.Hash]*transactionStatusIndexBatch, len(counts)) - delta := buildTransactionStatusDeltaWithBatches(identities, counts, indexes, batches) - for _, batch := range batches { - batch.partition() - } - return &preparedTransactionStatusDelta{identities: identities, delta: delta, indexBatches: batches} + return &preparedTransactionStatusDelta{identities: identities, delta: buildTransactionStatusDelta(identities, counts, indexes)} } diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go index b6b73685c..2fc7d0f45 100644 --- a/pkg/replay/transaction_status_publication_benchmark_test.go +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -61,9 +61,8 @@ func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, return nil } -// Historical unprepared delta construction, using the current visible index. -// For before/after index comparisons run the same benchmark at both commits; -// this helper is not a frozen baseline for the mutable index implementation. +// Frozen production commit algorithm before publication optimization. This is +// an independent baseline, including its original visible-index allocation. func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { @@ -76,11 +75,12 @@ func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionSt if group == nil { group = &visibleTransactionStatusGroup{ keyIndex: deltaGroup.keyIndex, + keys: make(map[transactionStatusKey]uint16), } c.visible[blockhash] = group } for key := range deltaGroup.keys { - group.keys.add(key) + group.keys[key]++ } } return nil From f164274b871498a916d2f5fa5d8f855f7ae42369 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:33:09 -0500 Subject: [PATCH 015/111] docs: keep review specifications and archive operational notes --- docs/transaction-status-publication.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index a099af906..2ee1d38c9 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -70,4 +70,4 @@ Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five sample Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. -A separate 180-second pre-change live trace observed 728 publications. In the 145 publications taking at least 1 ms, the repeated scan measured 1.805 ms median / 3.180 ms maximum; insertion 2.198 / 6.774 ms. Lock acquisition was at most 0.0058 ms across all publications, and the preparation join at most 0.0010 ms. This latency-selected cohort is not a fixed transaction-size sample or a before/after p99 comparison. Probe overhead is included. These measurements identify removable work; they do not establish a sustained FAST improvement. Raw traces, native test windows and exact combined source stay on the validator host at `/srv/mithril-status-validation-20260915`. +Original run artifacts are retained in the [evidence archive](status-checkpoint-expiry-evidence.md). \ No newline at end of file From ced869c8d964c39d1159150780935da3eefbe73b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:39:57 -0500 Subject: [PATCH 016/111] docs: validate archive links and formatting --- docs/status-checkpoint-expiry-evidence.md | 2 +- docs/transaction-status-publication.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/status-checkpoint-expiry-evidence.md b/docs/status-checkpoint-expiry-evidence.md index 22661d482..978180120 100644 --- a/docs/status-checkpoint-expiry-evidence.md +++ b/docs/status-checkpoint-expiry-evidence.md @@ -2,7 +2,7 @@ The maintained subsystem documentation and reusable Go benchmarks describe the implementation and reproduction method. Historical raw results and session -notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/blob/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf) +notes are retained at [the tested source snapshot](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf) (tag `review-evidence-20260916-status-checkpoint-expiry`). They are omitted from this proposed merge. [Historical result files](https://github.com/Overclock-Validator/mithril/tree/a511ad3b0bc77cf8b5ae4ee16359ac6b453fc7bf/docs/results) diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index 2ee1d38c9..c74be8335 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -70,4 +70,4 @@ Zen 5 incremental measurement (Ryzen 9700X, Go 1.26.4, GOMAXPROCS=2, five sample Values are medians of sample means. Existing-group cases remove approximately 1.1 ms of repeated lookup work; new-group cases show no clear gain and shared-host variation. Full native replay/block race suites, targeted node recovery race tests, vet and the combined build passed. Local replay race tests and vet also passed. -Original run artifacts are retained in the [evidence archive](status-checkpoint-expiry-evidence.md). \ No newline at end of file +Original run artifacts are retained in the [evidence archive](status-checkpoint-expiry-evidence.md). From 939aa261933133a47b45f5867f7f2eabadc69e3c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:57:11 -0500 Subject: [PATCH 017/111] replay: rebind disappeared status groups and randomize encoding checks --- docs/transaction-status-publication.md | 2 +- pkg/replay/transaction_status_cache.go | 9 +++-- pkg/replay/transaction_status_capture_test.go | 35 +++++++++++++++++++ ...ction_status_publication_benchmark_test.go | 3 ++ .../transaction_status_publication_test.go | 21 +++++++++++ 5 files changed, 67 insertions(+), 3 deletions(-) diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index c74be8335..68ca309fa 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -4,7 +4,7 @@ Replay previously built the immutable per-bank transaction-status delta and grew Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. -The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset, disappearance of a previously nonzero-offset group, or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index bb4a157fb..d17110491 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -664,9 +664,14 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan if prepared != nil && prepared.identities == plan.messageIdentities { delta = prepared.delta // A restore or branch transition can change a blockhash's slice offset. - // Rebuild from full identities if any current group uses another offset. + // Rebuild from full identities if a group changed offset or disappeared; + // a missing group uses the same zero offset as fresh preparation. for blockhash, group := range delta { - if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { + index := uint8(0) + if visible := c.visible[blockhash]; visible != nil { + index = visible.keyIndex + } + if index != group.keyIndex { delta = nil break } diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go index 8e1344c6f..34b8273af 100644 --- a/pkg/replay/transaction_status_capture_test.go +++ b/pkg/replay/transaction_status_capture_test.go @@ -4,6 +4,7 @@ import ( "bytes" "encoding/binary" "fmt" + "math/rand" "sort" "sync" "testing" @@ -297,3 +298,37 @@ func TestTransactionStatusEncodingMatchesOriginalWireFormat(t *testing.T) { } } } + +func TestTransactionStatusEncodingRandomizedByteIdentity(t *testing.T) { + rng := rand.New(rand.NewSource(20260916)) + for trial := 0; trial < 200; trial++ { + nodes := make([]*transactionStatusNode, rng.Intn(13)) + var slot uint64 + for i := range nodes { + slot += uint64(1 + rng.Intn(5)) + node := &transactionStatusNode{slot: slot, hasBlockID: rng.Intn(2) == 1, delta: make(transactionStatusDelta)} + _, _ = rng.Read(node.blockID[:]) + for g := rng.Intn(8); g > 0; g-- { + var hash solana.Hash + _, _ = rng.Read(hash[:]) + group := &transactionStatusGroup{keyIndex: uint8(rng.Intn(int(txstatus.MaxCachedKeyIndex) + 1)), keys: make(map[transactionStatusKey]struct{})} + for k := rng.Intn(13); k > 0; k-- { + var key transactionStatusKey + _, _ = rng.Read(key[:]) + group.keys[key] = struct{}{} + } + node.delta[hash] = group + } + nodes[i] = node + } + rooted := uint16(rng.Intn(301)) + complete, genesis := rng.Intn(2) == 1, rng.Intn(2) == 1 + want, err := marshalTransactionStatusNodesUncached(nodes, rooted, complete, genesis) + require.NoError(t, err) + for pass := 0; pass < 2; pass++ { + got, err := marshalTransactionStatusNodes(nodes, rooted, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got, "trial=%d pass=%d", trial, pass) + } + } +} diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go index 2fc7d0f45..2151ef330 100644 --- a/pkg/replay/transaction_status_publication_benchmark_test.go +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -63,6 +63,9 @@ func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, // Frozen production commit algorithm before publication optimization. This is // an independent baseline, including its original visible-index allocation. +// Benchmark fixtures never reuse validation receipts on this path. This frozen +// helper deliberately omits validation-version bumps and must not be used by +// production callers or copied as a model for mutating the live cache. func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { diff --git a/pkg/replay/transaction_status_publication_test.go b/pkg/replay/transaction_status_publication_test.go index 58a24b06f..3b9b92885 100644 --- a/pkg/replay/transaction_status_publication_test.go +++ b/pkg/replay/transaction_status_publication_test.go @@ -131,3 +131,24 @@ func TestPreparedStatusDeltaScheduling(t *testing.T) { } } } + +func TestPreparedStatusDeltaRebindsMissingSnapshotGroup(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, 7)}}, + }, 0) + require.NoError(t, err) + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed.prepareTransactionStatusDelta(plan.messageIdentities) + // Model restore/branch replacement removing the group after preparation. + cache := NewTransactionStatusCache() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + inline := NewTransactionStatusCache() + require.NoError(t, inline.commitBlockWithPlan(candidate, plan)) + require.Equal(t, inline.tip.delta, cache.tip.delta) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} From 065ee9eabd38d0709efab85ea8e408581900aee5 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:57:11 -0500 Subject: [PATCH 018/111] replay: rebind disappeared status groups and randomize encoding checks --- docs/transaction-status-publication.md | 2 +- pkg/replay/transaction_status_cache.go | 9 +++-- pkg/replay/transaction_status_capture_test.go | 35 +++++++++++++++++++ ...ction_status_publication_benchmark_test.go | 3 ++ .../transaction_status_publication_test.go | 21 +++++++++++ 5 files changed, 67 insertions(+), 3 deletions(-) diff --git a/docs/transaction-status-publication.md b/docs/transaction-status-publication.md index c74be8335..68ca309fa 100644 --- a/docs/transaction-status-publication.md +++ b/docs/transaction-status-publication.md @@ -4,7 +4,7 @@ Replay previously built the immutable per-bank transaction-status delta and grew Count identities by recent blockhash and allocate each delta map at its final capacity. Pre-size newly created visible maps too. For banks with more than 32 transactions and GOMAXPROCS greater than one, prepare the immutable delta during account loading and execution. Smaller banks and single-thread configurations keep the work inline. There is at most one preparation task per ProcessBlock call, and every return joins it, including rejected banks. No status becomes visible during preparation. -The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. +The worker reads immutable prepared message identities and briefly snapshots only blockhash slice offsets under the cache read lock. It builds its private maps outside the lock. Commit checks exact block/identity binding, complete coverage and parent lineage under the publication lock. It rechecks ancestor duplicates unless the successful pre-execution validation belongs to the same cache instance, immutable identity set and unchanged cache version (see below). A changed slice offset, disappearance of a previously nonzero-offset group, or mismatched preparation triggers a rebuild from the actual block's identities. Publication still happens only after successful bank-state commit. Failed instructions within an accepted bank remain processed; rejected banks publish nothing. Pinned views, snapshots, reference counts and unwind keep their existing semantics. TransactionStatusPreparation measures worker wall time, which overlaps execution; it is not additive with replay wall time. TransactionStatusPreparationWait measures the residual join and is nested inside TransactionStatusCommit. The latter still includes waiting, final checks, visible-index updates and node publication. Preparation time excludes initial goroutine scheduling delay; any residual scheduling delay remains in the join/commit timer. diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index bb4a157fb..d17110491 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -664,9 +664,14 @@ func (c *TransactionStatusCache) commitBlockWithValidation(block *b.Block, plan if prepared != nil && prepared.identities == plan.messageIdentities { delta = prepared.delta // A restore or branch transition can change a blockhash's slice offset. - // Rebuild from full identities if any current group uses another offset. + // Rebuild from full identities if a group changed offset or disappeared; + // a missing group uses the same zero offset as fresh preparation. for blockhash, group := range delta { - if visible := c.visible[blockhash]; visible != nil && visible.keyIndex != group.keyIndex { + index := uint8(0) + if visible := c.visible[blockhash]; visible != nil { + index = visible.keyIndex + } + if index != group.keyIndex { delta = nil break } diff --git a/pkg/replay/transaction_status_capture_test.go b/pkg/replay/transaction_status_capture_test.go index 8e1344c6f..34b8273af 100644 --- a/pkg/replay/transaction_status_capture_test.go +++ b/pkg/replay/transaction_status_capture_test.go @@ -4,6 +4,7 @@ import ( "bytes" "encoding/binary" "fmt" + "math/rand" "sort" "sync" "testing" @@ -297,3 +298,37 @@ func TestTransactionStatusEncodingMatchesOriginalWireFormat(t *testing.T) { } } } + +func TestTransactionStatusEncodingRandomizedByteIdentity(t *testing.T) { + rng := rand.New(rand.NewSource(20260916)) + for trial := 0; trial < 200; trial++ { + nodes := make([]*transactionStatusNode, rng.Intn(13)) + var slot uint64 + for i := range nodes { + slot += uint64(1 + rng.Intn(5)) + node := &transactionStatusNode{slot: slot, hasBlockID: rng.Intn(2) == 1, delta: make(transactionStatusDelta)} + _, _ = rng.Read(node.blockID[:]) + for g := rng.Intn(8); g > 0; g-- { + var hash solana.Hash + _, _ = rng.Read(hash[:]) + group := &transactionStatusGroup{keyIndex: uint8(rng.Intn(int(txstatus.MaxCachedKeyIndex) + 1)), keys: make(map[transactionStatusKey]struct{})} + for k := rng.Intn(13); k > 0; k-- { + var key transactionStatusKey + _, _ = rng.Read(key[:]) + group.keys[key] = struct{}{} + } + node.delta[hash] = group + } + nodes[i] = node + } + rooted := uint16(rng.Intn(301)) + complete, genesis := rng.Intn(2) == 1, rng.Intn(2) == 1 + want, err := marshalTransactionStatusNodesUncached(nodes, rooted, complete, genesis) + require.NoError(t, err) + for pass := 0; pass < 2; pass++ { + got, err := marshalTransactionStatusNodes(nodes, rooted, complete, genesis) + require.NoError(t, err) + require.Equal(t, want, got, "trial=%d pass=%d", trial, pass) + } + } +} diff --git a/pkg/replay/transaction_status_publication_benchmark_test.go b/pkg/replay/transaction_status_publication_benchmark_test.go index 2fc7d0f45..2151ef330 100644 --- a/pkg/replay/transaction_status_publication_benchmark_test.go +++ b/pkg/replay/transaction_status_publication_benchmark_test.go @@ -63,6 +63,9 @@ func (c *TransactionStatusCache) legacyCommitStatusForBenchmark(block *b.Block, // Frozen production commit algorithm before publication optimization. This is // an independent baseline, including its original visible-index allocation. +// Benchmark fixtures never reuse validation receipts on this path. This frozen +// helper deliberately omits validation-version bumps and must not be used by +// production callers or copied as a model for mutating the live cache. func (c *TransactionStatusCache) legacyAddStatusForBenchmark(delta transactionStatusDelta) error { for blockhash, deltaGroup := range delta { if group := c.visible[blockhash]; group != nil && group.keyIndex != deltaGroup.keyIndex { diff --git a/pkg/replay/transaction_status_publication_test.go b/pkg/replay/transaction_status_publication_test.go index 58a24b06f..3b9b92885 100644 --- a/pkg/replay/transaction_status_publication_test.go +++ b/pkg/replay/transaction_status_publication_test.go @@ -131,3 +131,24 @@ func TestPreparedStatusDeltaScheduling(t *testing.T) { } } } + +func TestPreparedStatusDeltaRebindsMissingSnapshotGroup(t *testing.T) { + ancestor := statusCacheTestTransaction(1, 2, 3) + seed, err := NewTransactionStatusCacheFromAgaveSnapshot([]txstatus.SnapshotSlotDelta{ + {Slot: 0, IsRoot: true, Statuses: []txstatus.SnapshotStatus{snapshotStatusCacheStatusForTx(t, ancestor, 7)}}, + }, 0) + require.NoError(t, err) + candidate := statusCacheTestBlock(1, statusCacheTestTransaction(1, 4, 5)) + plan, err := planBlockTransactionExecution(candidate) + require.NoError(t, err) + prepared := seed.prepareTransactionStatusDelta(plan.messageIdentities) + // Model restore/branch replacement removing the group after preparation. + cache := NewTransactionStatusCache() + require.NoError(t, cache.commitBlockWithPreparedDelta(candidate, plan, prepared)) + inline := NewTransactionStatusCache() + require.NoError(t, inline.commitBlockWithPlan(candidate, plan)) + require.Equal(t, inline.tip.delta, cache.tip.delta) + found, err := cache.View().ContainsTransaction(candidate.Transactions[0]) + require.NoError(t, err) + require.True(t, found) +} From 59253380036dbb896ef76303cc9ec977cfd7f781 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 21:20:30 -0500 Subject: [PATCH 019/111] turbine: authenticate recovered shreds against the signed FEC tree --- docs/fec-recovery-authentication.md | 63 ++++++ pkg/turbine/assembler.go | 10 +- pkg/turbine/assembler_test.go | 9 + pkg/turbine/recovery_authentication.go | 111 ++++++++++ pkg/turbine/recovery_authentication_test.go | 226 ++++++++++++++++++++ 5 files changed, 418 insertions(+), 1 deletion(-) create mode 100644 docs/fec-recovery-authentication.md create mode 100644 pkg/turbine/recovery_authentication.go create mode 100644 pkg/turbine/recovery_authentication_test.go diff --git a/docs/fec-recovery-authentication.md b/docs/fec-recovery-authentication.md new file mode 100644 index 000000000..a1b7b752b --- /dev/null +++ b/docs/fec-recovery-authentication.md @@ -0,0 +1,63 @@ +# Authenticating recovered FEC data + +Received shreds must pass leader-signature verification before entering the slot +assembler. Reed–Solomon reconstruction alone does not authenticate missing data: +a leader can sign a Merkle tree containing inconsistent data and coding shards. +Structural validation of recovered headers does not reject that case. + +Both the specialized one-missing-data decoder and the general decoder now require +`authenticateRecoveredFEC` to succeed before returning any recovered data. It +reconstructs missing coding shards too, builds the complete data/coding Merkle +tree, and compares its root with a received coding shred's signed root. Checking +only recovered-data proofs would not detect an inconsistent commitment to a +missing coding shard. Existing received packets are read-only throughout recovery. + +On success, recovered data receives complete Merkle proofs and the coding +template's chained root and, where applicable, retransmitter signature. The latter +is a hop signature copied from a received packet, not a reconstruction of a lost +relay's signature. On failure, no recovered data is published. The all-data-present +path does not reconstruct or authenticate another tree; it relies on ingress +verification of the received data. + +This follows [Agave's recovery algorithm](https://github.com/anza-xyz/alpenglow/blob/9f284c913f3c78b36179ae2461fa91286a616fb9/ledger/src/shred/merkle.rs#L670): +reconstruct all missing shards, validate recovered headers, compare the full root, +and populate proofs. Signature verification is not repeated after the root match. + +## Validation + +`TestRecoveredFECAuthentication` exercises one, three and all 32 missing data +shreds, chained and resigned packets, altered recovered bytes, and leader-signed +inconsistent parity both received and absent. Invalid sets return no recovered +shreds. Valid recovered packets verify with the leader's public key. + +`TestRecoveredFECAgaveSignedCapture` uses four FEC sets from the existing captured +Agave slot 1,752,420. It regenerates parity without signing a new root, checks the +original committed roots, and compares recovered authenticated bytes and proofs +with the capture. Transport repair nonces and relay-specific signatures are +handled separately. This is a captured-wire compatibility test, not a run of a +Rust recovery oracle. The older localnet recovery test also covers an unchained +1+17 layout, with its 2022 proofs replaced by current signed proofs. + +## Cost and reproduction + +Apple M4 Pro, `GOMAXPROCS=2`, five 200 ms samples, warmed encoder cache. The baseline +is `039ebd67`; times are medians for one FEC recovery, excluding ingress signature +verification and block execution. + +| Received / recovered shape | Before | Authenticated recovery | +|---|---:|---:| +| 31 data + 32 coding; recover one data | 2.47 µs | 30.17 µs | +| 31 data + 1 coding; recover one data and missing parity | — | 104.52 µs | +| 29 data + 3 coding; recover three data and missing parity | — | 117.93 µs | +| 32 coding; recover all data | — | 118.01 µs | + +The first case increases from 4,264 bytes / 5 allocations to 8,360 bytes / 6 +allocations per operation. Threshold-arrival cases also pay to reconstruct missing +parity and assemble its leaf bytes. These measurements establish the cost of the +correctness check, not a live FAST-score or whole-block performance result. + +``` +GOMAXPROCS=2 go test ./pkg/turbine -run '^$' \ + -bench 'Benchmark(RecoverFECOneMissingBoundary|AuthenticatedFECRecovery)$' \ + -benchmem -benchtime=200ms -count=5 +``` diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index d2bb4acee..0dcff2fa6 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -1414,7 +1414,12 @@ func (a *SlotAssembler) recoverFEC(state *slotState, fecSetIndex uint32) ([]*Shr if err != nil { return nil, err } - return []*Shred{shred}, nil + shards[missingDataIndex] = dst + recovered := []*Shred{shred} + if err := a.authenticateRecoveredFEC(fec, shards, recovered); err != nil { + return nil, err + } + return recovered, nil } required := make([]bool, int(layout.dataShreds)+int(layout.codingShreds)) @@ -1447,6 +1452,9 @@ func (a *SlotAssembler) recoverFEC(state *slotState, fecSetIndex uint32) ([]*Shr } recovered = append(recovered, shred) } + if err := a.authenticateRecoveredFEC(fec, shards, recovered); err != nil { + return nil, err + } return recovered, nil } diff --git a/pkg/turbine/assembler_test.go b/pkg/turbine/assembler_test.go index f70f3d695..e38b468b3 100644 --- a/pkg/turbine/assembler_test.go +++ b/pkg/turbine/assembler_test.go @@ -622,6 +622,15 @@ func TestDecodeAlpenglowParentMarkers(t *testing.T) { func TestSlotAssemblerRecoversMissingMerkleDataShredFromCodingShreds(t *testing.T) { dataShreds := localnetMerkleShreds(t, "d") codeShreds := localnetMerkleShreds(t, "c") + // These 2022 fixtures supply a useful non-power-of-two, unchained 1+17 + // erasure layout, but their old proofs do not yield a common root under + // today's Merkle hashing. Re-sign each complete tree using current proofs; + // recovery must now authenticate the tree, not just reconstruct the data. + for i, data := range dataShreds { + packets := append([][]byte{data}, codeShreds[i*17:(i+1)*17]...) + resignRecoveryFixture(t, packets) + } + if len(dataShreds) < 2 || len(codeShreds) == 0 { t.Fatalf("fixture needs data and coding shreds") } diff --git a/pkg/turbine/recovery_authentication.go b/pkg/turbine/recovery_authentication.go new file mode 100644 index 000000000..6f1e0c91b --- /dev/null +++ b/pkg/turbine/recovery_authentication.go @@ -0,0 +1,111 @@ +package turbine + +import ( + "encoding/binary" + "fmt" + "math" + "math/bits" + + "github.com/gagliardetto/solana-go" +) + +// authenticateRecoveredFEC follows Agave's merkle::recover: reconstruct every +// missing shard, rebuild the full tree (including coding shreds), and compare +// its root with a received shred's authenticated root before publishing data. +// Signature verification of received shreds is the ingress caller's contract; +// Reed-Solomon reconstruction alone is never evidence of authenticity. +func (a *SlotAssembler) authenticateRecoveredFEC(f *fecState, shards [][]byte, recovered []*Shred) error { + // Agave takes its template from the highest-position received coding shred. + var code *Shred + for _, s := range f.coding { + if code == nil || s.Position > code.Position { + code = s + } + } + if code == nil { + return fmt.Errorf("recover FEC: missing coding template") + } + proofSize, chained, _, ok := merkleVariantInfo(code.Variant) + if !ok || int(proofSize) != bits.Len(uint(len(shards)-1)) { + return fmt.Errorf("recover FEC: invalid Merkle proof depth") + } + if code.Index < uint32(code.Position) { + return fmt.Errorf("recover FEC: invalid coding index") + } + codeBase := code.Index - uint32(code.Position) + if uint64(codeBase)+uint64(f.layout.codingShreds)-1 > math.MaxUint32 { + return fmt.Errorf("recover FEC: coding index overflow") + } + expected, err := code.MerkleRoot() + if err != nil { + return err + } + encoder, err := a.fecEncoder(f.layout) + if err != nil { + return err + } + // Present shards are read-only. Only absent coding shards remain after the + // caller's data recovery, so this also checks commitments to missing parity. + if err = encoder.Reconstruct(shards); err != nil { + return fmt.Errorf("recover FEC authentication: %w", err) + } + dataCount := int(f.layout.dataShreds) + data := make(map[uint32]*Shred, len(recovered)) + for _, s := range recovered { + // Chained root and retransmitter signature are outside the erasure region. + // Copy the template suffix now, then replace its proof after authenticating. + copy(s.Payload[shredSignatureSize+f.layout.shardSize:], code.Payload[codingHeaderSize+f.layout.shardSize:]) + data[s.Index-f.fecSetIndex] = s + } + nodes := make([]solana.Hash, 0, merkleTreeSize(len(shards))) + for i, shard := range shards { + var s *Shred + if i < dataCount { + s = f.data[uint32(i)] + if s == nil { + s = data[uint32(i)] + } + } else { + pos := uint16(i - dataCount) + s = f.coding[pos] + if s == nil { + payload := append([]byte(nil), code.Payload...) + binary.LittleEndian.PutUint32(payload[shredIndexOffset:], codeBase+uint32(pos)) + binary.LittleEndian.PutUint16(payload[codingPositionOffset:], pos) + copy(payload[codingHeaderSize:], shard) + // Only header fields and bytes consumed by merkleLeaf are needed here. + copyOfCode := *code + copyOfCode.Payload = payload + copyOfCode.Index = codeBase + uint32(pos) + copyOfCode.Position = pos + s = ©OfCode + } + } + if s == nil { + return fmt.Errorf("recover FEC: missing reconstructed data %d", i) + } + leaf, err := s.merkleLeaf() + if err != nil { + return err + } + nodes = append(nodes, leaf) + } + for size := len(shards); size > 1; size = (size + 1) >> 1 { + offset := len(nodes) - size + for i := 0; i < size; i += 2 { + right := min(i+1, size-1) + nodes = append(nodes, merkleHashNode(nodes[offset+i][:merkleProofEntrySize], nodes[offset+right][:merkleProofEntrySize])) + } + } + if nodes[len(nodes)-1] != expected { + return fmt.Errorf("%w: recovered FEC Merkle root mismatch slot=%d fec_set=%d", ErrInvalidSignature, f.slot, f.fecSetIndex) + } + proofOffset := shredSignatureSize + f.layout.shardSize + if chained { + proofOffset += merkleRootSize + } + for _, s := range recovered { + writeMerkleProof(s.Payload[proofOffset:], nodes, int(s.Index-f.fecSetIndex), len(shards)) + } + return nil +} diff --git a/pkg/turbine/recovery_authentication_test.go b/pkg/turbine/recovery_authentication_test.go new file mode 100644 index 000000000..626742850 --- /dev/null +++ b/pkg/turbine/recovery_authentication_test.go @@ -0,0 +1,226 @@ +package turbine + +import ( + "crypto/ed25519" + "encoding/binary" + "fmt" + "sort" + "testing" + + "github.com/gagliardetto/solana-go" + "github.com/klauspost/reedsolomon" + "github.com/stretchr/testify/require" +) + +func resignRecoveryFixture(t *testing.T, packets [][]byte) { + t.Helper() + nodes, err := buildMerkleTree(packets) + require.NoError(t, err) + root := nodes[len(nodes)-1] + sig := ed25519.Sign(ed25519.PrivateKey(benchmarkLeaderKey()), root[:]) + for i, p := range packets { + s, err := ParseShred(p) + require.NoError(t, err) + shard, err := s.erasureShard() + require.NoError(t, err) + start := shredSignatureSize + if s.Type == ShredTypeCode { + start = codingHeaderSize + } + _, chained, _, _ := merkleVariantInfo(s.Variant) + offset := start + len(shard) + if chained { + offset += merkleRootSize + } + copy(p[:shredSignatureSize], sig) + writeMerkleProof(p[offset:], nodes, i, len(packets)) + } +} + +func recoveryFixtureState(t testing.TB, packets [][]byte, missing int) (*SlotAssembler, *slotState) { + t.Helper() + state := &slotState{slot: 10, shreds: make(map[uint32]*Shred), fecSets: make(map[uint32]*fecState), lastIndex: ^uint32(0)} + // Exactly 32 received shreds: also force reconstruction of missing parity. + for i, p := range packets { + if i < missing || i >= 32+missing { + continue + } + s, err := ParseShred(p) + require.NoError(t, err) + state.slot = s.Slot + state.shredVer = s.Version + if s.Type == ShredTypeData { + require.NoError(t, state.addDataShred(s)) + } else { + require.NoError(t, state.addCodingShred(s)) + } + } + return NewSlotAssembler(), state +} + +func TestRecoveredFECAuthentication(t *testing.T) { + for _, resigned := range []bool{false, true} { + for _, missing := range []int{1, 3, 32} { + for _, attack := range []string{"valid", "received_parity", "missing_parity", "recovered_bytes"} { + t.Run(fmt.Sprintf("resigned=%t/missing=%d/%s", resigned, missing, attack), func(t *testing.T) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + packets, _, _, _, err := gen.MakeShredsFromData(benchmarkLeaderKey(), benchmarkPayload(128), resigned, solana.Hash{9}, 0, 0) + require.NoError(t, err) + require.Len(t, packets, 64) + if attack == "received_parity" { + packets[32][codingHeaderSize+128] ^= 1 + resignRecoveryFixture(t, packets) + } + if attack == "missing_parity" { + if missing == 32 { + t.Skip("no missing coding shard") + } + packets[63][codingHeaderSize+128] ^= 1 + resignRecoveryFixture(t, packets) + } + // The adversarial fixtures have valid leader signatures, not corrupted + // network packets: only RS/tree consistency is wrong. + for _, p := range packets { + s, e := ParseShred(p) + require.NoError(t, e) + require.NoError(t, s.VerifySignature(benchmarkLeaderKey().PublicKey())) + } + a, state := recoveryFixtureState(t, packets, missing) + recovered, err := a.recoverFEC(state, 0) + if attack == "received_parity" || attack == "missing_parity" { + require.ErrorIs(t, err, ErrInvalidSignature) + require.Empty(t, recovered) + return + } + require.NoError(t, err) + require.Len(t, recovered, missing) + for _, s := range recovered { + require.Equal(t, packets[s.Index], s.Payload) + require.NoError(t, s.VerifySignature(benchmarkLeaderKey().PublicKey())) + } + if attack == "recovered_bytes" { + f := state.fecSets[0] + shards := make([][]byte, 64) + for i, s := range f.data { + shards[i], err = s.erasureShard() + require.NoError(t, err) + } + for i, s := range f.coding { + shards[32+int(i)], err = s.erasureShard() + require.NoError(t, err) + } + for _, s := range recovered { + shards[s.Index], err = s.erasureShard() + require.NoError(t, err) + } + recovered[0].Payload[dataHeaderSize+1] ^= 1 + require.ErrorIs(t, a.authenticateRecoveredFEC(f, shards, recovered), ErrInvalidSignature) + } + }) + } + } + } +} + +// Recreate coding packets from immutable Agave data packets, without signing a +// new root. Equality to the root committed by Agave checks erasure layout, +// parity coefficients, coding headers, chain roots and Merkle construction. +func TestRecoveredFECAgaveSignedCapture(t *testing.T) { + all := agavePaddedSlot1752420Packets(t) + sort.Slice(all, func(i, j int) bool { + return binary.LittleEndian.Uint32(all[i][shredIndexOffset:]) < binary.LittleEndian.Uint32(all[j][shredIndexOffset:]) + }) + for group := 0; group < 4; group++ { + packets := make([][]byte, 64) + shards := make([][]byte, 64) + var first *Shred + for i := 0; i < 32; i++ { + // Captured repair packets may append a four-byte nonce, which is + // transport metadata rather than part of the authenticated shred. + packets[i] = all[group*32+i][:dataPayloadSize] + s, err := ParseShred(packets[i]) + require.NoError(t, err) + if i == 0 { + first = s + } + shards[i], err = s.erasureShard() + require.NoError(t, err) + } + codeVariant, ok := merkleCounterpartVariant(first.Variant, ShredTypeCode) + require.True(t, ok) + for i := 0; i < 32; i++ { + p := make([]byte, codingPayloadSize) + copy(p[:codingNumDataOffset], first.Payload[:codingNumDataOffset]) + p[shredVariantOffset] = codeVariant + binary.LittleEndian.PutUint32(p[shredIndexOffset:], first.FECSetIndex+uint32(i)) + binary.LittleEndian.PutUint16(p[codingNumDataOffset:], 32) + binary.LittleEndian.PutUint16(p[codingNumCodingOffset:], 32) + binary.LittleEndian.PutUint16(p[codingPositionOffset:], uint16(i)) + copy(p[codingHeaderSize+len(shards[0]):], first.Payload[shredSignatureSize+len(shards[0]):]) + packets[32+i] = p + shards[32+i] = p[codingHeaderSize : codingHeaderSize+len(shards[0])] + } + enc, err := reedsolomon.New(32, 32) + require.NoError(t, err) + require.NoError(t, enc.Encode(shards)) + nodes, err := buildMerkleTree(packets) + require.NoError(t, err) + root, err := first.MerkleRoot() + require.NoError(t, err) + require.Equal(t, root, nodes[len(nodes)-1]) + proofSize, chained, _, _ := merkleVariantInfo(codeVariant) + offset := codingHeaderSize + len(shards[0]) + if chained { + offset += merkleRootSize + } + for i := 32; i < 64; i++ { + require.Equal(t, int(proofSize), writeMerkleProof(packets[i][offset:], nodes, i, 64)) + } + for _, missing := range []int{1, 3, 32} { + a, state := recoveryFixtureState(t, packets, missing) + got, err := a.recoverFEC(state, first.FECSetIndex) + require.NoError(t, err) + require.Len(t, got, missing) + for _, s := range got { + end := dataPayloadSize + _, _, resigned, _ := merkleVariantInfo(s.Variant) + if resigned { + // Hop signatures can differ by relay; Agave copies the + // received coding template's signature, not a lost one. + end -= shredSignatureSize + want, err := first.RetransmitterSignature() + require.NoError(t, err) + got, err := s.RetransmitterSignature() + require.NoError(t, err) + require.Equal(t, want, got) + } + require.Equal(t, packets[s.Index-first.FECSetIndex][:end], s.Payload[:end]) + actual, err := s.MerkleRoot() + require.NoError(t, err) + require.Equal(t, root, actual) + } + } + } +} + +func BenchmarkAuthenticatedFECRecovery(b *testing.B) { + for _, missing := range []int{1, 3, 32} { + b.Run(fmt.Sprintf("missing_data=%d", missing), func(b *testing.B) { + gen := ShredGenerator{Slot: 10, ParentSlot: 9, Version: 1} + packets, _, _, _, err := gen.MakeShredsFromData(benchmarkLeaderKey(), benchmarkPayload(128), false, solana.Hash{9}, 0, 0) + require.NoError(b, err) + a, state := recoveryFixtureState(b, packets, missing) + _, err = a.recoverFEC(state, 0) + require.NoError(b, err) + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + recovered, err := a.recoverFEC(state, 0) + if err != nil { + b.Fatal(err) + } + benchmarkRecoveredShredsSink = recovered + } + }) + } +} From 25a8b69c7bb63a6b96c778800c899d54bf34825e Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:09:49 -0500 Subject: [PATCH 020/111] leader: precheck buffer admission before transaction preparation --- pkg/blockprod/scheduler/buffer.go | 40 +++++++++++++++++--------- pkg/blockprod/scheduler/buffer_test.go | 25 ++++++++++++++++ pkg/blockprod/scheduler/scheduler.go | 8 ++++-- pkg/costmodel/limits.go | 3 +- 4 files changed, 60 insertions(+), 16 deletions(-) diff --git a/pkg/blockprod/scheduler/buffer.go b/pkg/blockprod/scheduler/buffer.go index 38bb57a95..2aa5a70c5 100644 --- a/pkg/blockprod/scheduler/buffer.go +++ b/pkg/blockprod/scheduler/buffer.go @@ -129,27 +129,41 @@ const ( InsertRejectedCapacity ) -// Insert adds e when not a duplicate. At capacity, the lowest-reward entry is -// evicted if e has a strictly higher reward; otherwise e is rejected. -func (b *Buffer) Insert(e *entry) (InsertResult, *entry) { - if e == nil || e.tx == nil { - return InsertRejectedCapacity, nil - } +// precheck rejects entries that cannot be admitted now, without reserving space +// or evicting anything. Preparation runs outside mu; Insert must recheck because +// concurrent arrivals or draining can change both duplicates and the floor. +func (b *Buffer) precheck(e *entry) InsertResult { b.mu.Lock() defer b.mu.Unlock() + return b.admissionLocked(e) +} +func (b *Buffer) admissionLocked(e *entry) InsertResult { + if e == nil || e.tx == nil { + return InsertRejectedCapacity + } if _, exists := b.byHash[e.messageHash]; exists { - return InsertDuplicate, nil + return InsertDuplicate } - var evicted *entry if b.alive >= b.capacity { min := b.peekMinAliveLocked() - if min == nil { - return InsertRejectedCapacity, nil - } - if e.reward <= min.reward { - return InsertRejectedCapacity, nil + if min == nil || e.reward <= min.reward { + return InsertRejectedCapacity } + } + return InsertAccepted +} + +// Insert adds e when not a duplicate. At capacity, the lowest-reward entry is +// evicted if e has a strictly higher reward; otherwise e is rejected. +func (b *Buffer) Insert(e *entry) (InsertResult, *entry) { + b.mu.Lock() + defer b.mu.Unlock() + if result := b.admissionLocked(e); result != InsertAccepted { + return result, nil + } + var evicted *entry + if b.alive >= b.capacity { evicted = b.popMinAliveLocked() } b.pushAliveLocked(e) diff --git a/pkg/blockprod/scheduler/buffer_test.go b/pkg/blockprod/scheduler/buffer_test.go index 9b85554ce..6ec46415d 100644 --- a/pkg/blockprod/scheduler/buffer_test.go +++ b/pkg/blockprod/scheduler/buffer_test.go @@ -79,3 +79,28 @@ func TestBufferCleanup(t *testing.T) { require.Equal(t, 1, b.Len()) require.Equal(t, byte(2), b.PopMax().messageHash[0]) } + +func TestBufferPrecheckDoesNotReserveAndInsertRechecks(t *testing.T) { + b := NewBuffer(1) + first := testEntry(10, 1, 1) + require.Equal(t, InsertAccepted, b.precheck(first)) + require.Zero(t, b.Len()) + b.Insert(first) + require.Equal(t, InsertDuplicate, b.precheck(testEntry(20, 2, 1))) + require.Equal(t, InsertRejectedCapacity, b.precheck(testEntry(10, 2, 2))) + candidate := testEntry(20, 3, 3) + require.Equal(t, InsertAccepted, b.precheck(candidate)) + // A precheck must not evict the existing entry, and a later higher-priority + // arrival can reject the candidate even though its precheck succeeded. + require.Equal(t, first, b.PopMax()) + b.Insert(testEntry(30, 4, 4)) + result, evicted := b.Insert(candidate) + require.Equal(t, InsertRejectedCapacity, result) + require.Nil(t, evicted) + b.PopMax() + require.Equal(t, InsertAccepted, b.precheck(candidate)) + b.Insert(testEntry(20, 5, 3)) + result, evicted = b.Insert(candidate) + require.Equal(t, InsertDuplicate, result) + require.Nil(t, evicted) +} diff --git a/pkg/blockprod/scheduler/scheduler.go b/pkg/blockprod/scheduler/scheduler.go index 88e6435bf..9b0864d27 100644 --- a/pkg/blockprod/scheduler/scheduler.go +++ b/pkg/blockprod/scheduler/scheduler.go @@ -173,7 +173,6 @@ func (s *Scheduler) Receive(pkt packet.Packet) { e := &entry{ tx: tx, - prepared: s.preparer.Load().Prepare(tx), wire: wire, wireSize: len(wire), messageHash: messageHash, @@ -181,7 +180,12 @@ func (s *Scheduler) Receive(pkt packet.Packet) { reward: reward, seq: s.seq.Add(1), } - result, evicted := s.buffer.Insert(e) + result := s.buffer.precheck(e) + var evicted *entry + if result == InsertAccepted { + e.prepared = s.preparer.Load().Prepare(tx) + result, evicted = s.buffer.Insert(e) + } s.mu.Lock() switch result { case InsertAccepted: diff --git a/pkg/costmodel/limits.go b/pkg/costmodel/limits.go index 4e3f726ff..26091a3d0 100644 --- a/pkg/costmodel/limits.go +++ b/pkg/costmodel/limits.go @@ -90,7 +90,8 @@ func LimitsForFeatures(feats *features.Features) Limits { return limits } -// LimitsForSlot mirrors Agave v4.3.0-rc.1 runtime/slot_params.rs. A slot-time +// LimitsForSlot mirrors Agave v4.3.0-rc.1 runtime/src/slot_params.rs. +// Reference: https://github.com/anza-xyz/agave/blob/v4.3.0-rc.1/runtime/src/slot_params.rs A slot-time // gate takes effect in the epoch after activation; among effective gates the // shortest duration wins, even if longer-duration gates activate later. func LimitsForSlot(feats *features.Features, schedule *sealevel.SysvarEpochSchedule, slot uint64) (Limits, error) { From f2acfb48e6bd7376ebbd1281b1bce25856a4460d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:09:49 -0500 Subject: [PATCH 021/111] turbine: warn when completion fencing prevents spool deletion --- pkg/turbine/shredspool.go | 3 +++ 1 file changed, 3 insertions(+) diff --git a/pkg/turbine/shredspool.go b/pkg/turbine/shredspool.go index 964a98ce3..0170b0f76 100644 --- a/pkg/turbine/shredspool.go +++ b/pkg/turbine/shredspool.go @@ -12,6 +12,8 @@ import ( "strconv" "strings" "sync" + + "github.com/Overclock-Validator/mithril/pkg/mlog" ) // ShredSpool is a disposable on-disk cache of VERIFIED raw shreds, one @@ -385,6 +387,7 @@ func (s *ShredSpool) ensureRoomLocked(slot uint64, additional int64) bool { func (s *ShredSpool) dropSlotLocked(slot uint64) bool { if err := s.invalidateCompleteLocked(slot); err != nil { + mlog.Log.Warnf("shred spool: retaining slot %d after completion invalidation failed: %v", slot, err) return false } s.closeSlotLocked(slot) From b552ee9340caeb0f22049b7804aed3a88226b066 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:12:58 -0500 Subject: [PATCH 022/111] sbpf: reject overflow in contiguous virtual memory ranges --- pkg/sbpf/interpreter.go | 6 +++--- pkg/sbpf/translate_overflow_test.go | 25 +++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 3 deletions(-) create mode 100644 pkg/sbpf/translate_overflow_test.go diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 4fa0f784a..5f351a5d6 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1186,7 +1186,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.ro)) { + if lo+size < lo || lo+size > uint64(len(ip.ro)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds program read") } return unsafe.Pointer(&ip.ro[lo]), nil @@ -1203,7 +1203,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.heap)) { + if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } return unsafe.Pointer(&ip.heap[lo]), nil @@ -1214,7 +1214,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if len(ip.inputRegions) != 0 { return ip.translateInputRegion(lo, size, write) } - if lo+size > uint64(len(ip.input)) { + if lo+size < lo || lo+size > uint64(len(ip.input)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds input access") } return unsafe.Pointer(&ip.input[lo]), nil diff --git a/pkg/sbpf/translate_overflow_test.go b/pkg/sbpf/translate_overflow_test.go new file mode 100644 index 000000000..98d8ee2d0 --- /dev/null +++ b/pkg/sbpf/translate_overflow_test.go @@ -0,0 +1,25 @@ +package sbpf + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestTranslateRejectsOverflowingRange(t *testing.T) { + ip := &Interpreter{ro: make([]byte, 32), heap: make([]byte, 32), input: make([]byte, 32)} + for _, base := range []uint64{VaddrProgram, VaddrHeap, VaddrInput} { + for _, write := range []bool{false, true} { + for _, offset := range []uint64{1, 31, 33, 0xffffffff} { + for _, size := range []uint64{^uint64(0), ^uint64(0) - 15} { + _, err := ip.Translate(base+offset, size, write) + require.Error(t, err, "base=%x offset=%d size=%d write=%v", base, offset, size, write) + } + } + } + got, err := ip.Translate(base+31, 1, false) + require.NoError(t, err) + require.Len(t, got, 1) + _, err = ip.Translate(base+31, 2, false) + require.Error(t, err) + } +} From ace6e268d2f3d5e231d0a6f3af1a903fcd8b5e6d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:14:24 -0500 Subject: [PATCH 023/111] sealevel: remove SHA-256 syscall descriptor and digest allocations --- docs/sha256-syscall.md | 42 ++++ pkg/sealevel/sealevel_test.go | 12 +- pkg/sealevel/syscalls_hash.go | 15 +- pkg/sealevel/syscalls_sha256_bench_test.go | 268 +++++++++++++++++++++ 4 files changed, 325 insertions(+), 12 deletions(-) create mode 100644 docs/sha256-syscall.md create mode 100644 pkg/sealevel/syscalls_sha256_bench_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md new file mode 100644 index 000000000..58a83abca --- /dev/null +++ b/docs/sha256-syscall.md @@ -0,0 +1,42 @@ +# SHA-256 syscall overhead + +The syscall decodes the already-translated slice descriptor array directly and +writes the final digest into the translated output buffer. It retains streaming +SHA-256, slice order, memory translations, CU charges and validation order. Output +is written only after all inputs have been read, preserving overlapping-buffer +behavior. No special case for a particular on-chain program is introduced. + +A bounded 55-byte input-buffer prototype was slower than this simpler path and +is retained only as a benchmark comparison. The baseline reference is copied +from combined review commit `bd17683a`. + +Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; +medians below. Each benchmark runs serially through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. This does +not include VM instruction dispatch, a complete program, or block replay. + +| Input | Original | Direct decoding/output | Buffered prototype | +|---|---:|---:|---:| +| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | +| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | +| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | +| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | + +The two-slice case removes four allocations (112 bytes) per call. This is an +ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native +Zen 5 and captured heavy-block replay before deployment decisions. + +Reproduce with: + +```sh +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +``` + +Differential tests compare hashes, return/error values, remaining CU and all +input/output memory over valid inputs, invalid descriptors/addresses, depleted +budgets and output aliasing. The existing SHA program fixture also executes. +Testing exposed pre-existing overflow in contiguous VM region bounds checks; +that correction and its regression test are a separate preceding commit. Both +benchmark variants use the corrected VM. The old SHA fixture also needed its +compute-meter pointer initialized for the current interpreter API. diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ddfe230a..6ad64f3a9 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -416,13 +416,15 @@ func TestInterpreter_Sha256(t *testing.T) { syscalls.Register("my_memcmp", SyscallMemcmp) var log LogRecorder + ctx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: ctx, + ComputeMeter: &ctx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_hash.go b/pkg/sealevel/syscalls_hash.go index b73a1cd0c..33edd31d0 100644 --- a/pkg/sealevel/syscalls_hash.go +++ b/pkg/sealevel/syscalls_hash.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" "crypto/sha256" + "encoding/binary" "fmt" "math/big" @@ -49,15 +50,13 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 } var data []byte - reader := bytes.NewReader(vals) + // Translate validated the complete descriptor array above. Decode directly + // to avoid allocating a reader and a temporary buffer for each slice. for count := uint64(0); count < valsLen; count++ { - var vec VectorDescrC - err = vec.Unmarshal(reader) - if err != nil { - return syscallErr(err) - } + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} data, err = vm.Translate(vec.Addr, vec.Len, false) if err != nil { @@ -73,7 +72,9 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 hasher.Write(data) } } - copy(hashResult[:], hasher.Sum(nil)) + // All inputs have been read before writing, including when output aliases + // input memory. Append into the translated destination without allocating. + hasher.Sum(hashResult[:0]) return syscallSuccess(0) } diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go new file mode 100644 index 000000000..b4240f54e --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -0,0 +1,268 @@ +package sealevel + +import ( + "bytes" + "crypto/sha256" + "encoding/binary" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +// Frozen syscall implementation from bd17683a; keep independent for differential tests. +func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256BaselineReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + reader := bytes.NewReader(vals) + + for count := uint64(0); count < valsLen; count++ { + + var vec VectorDescrC + err = vec.Unmarshal(reader) + if err != nil { + return syscallErr(err) + } + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + hasher.Write(data) + } + } + copy(hashResult[:], hasher.Sum(nil)) + return syscallSuccess(0) +} + +// Experimental bounded-buffer variant retained only for benchmark comparison. +func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256SmallInputReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only + // this bounded case; larger inputs retain streaming hashing. + var small [55]byte + buffered := 0 + streaming := false + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + + for count := uint64(0); count < valsLen; count++ { + + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + if !streaming && len(data) <= len(small)-buffered { + buffered += copy(small[buffered:], data) + } else { + if !streaming { + hasher.Write(small[:buffered]) + streaming = true + } + hasher.Write(data) + } + } + } + if streaming { + hasher.Sum(hashResult[:0]) + } else { + digest := sha256.Sum256(small[:buffered]) + copy(hashResult, digest[:]) + } + return syscallSuccess(0) +} + +type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) + +func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { + mem := make([]byte, 32768) + pos := 8192 + for i, n := range sizes { + binary.LittleEndian.PutUint64(mem[i*16:], sbpf.VaddrInput+uint64(pos)) + binary.LittleEndian.PutUint64(mem[i*16+8:], uint64(n)) + for j := 0; j < n; j++ { + mem[pos+j] = byte(i + j) + } + pos += n + } + return mem, sbpf.VaddrInput, uint64(len(sizes)), sbpf.VaddrInput + 4096 +} + +func sha256VM(mem []byte, budget uint64) (*sbpf.Interpreter, *ExecutionCtx) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(budget)} + vm := sbpf.NewInterpreter(&sbpf.Program{}, &sbpf.VMOpts{Input: mem, HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + return vm, ctx +} + +func TestSha256SyscallDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(1234)) + for i := 0; i < 700; i++ { + sizes := make([]int, rng.Intn(8)) + for j := range sizes { + sizes[j] = rng.Intn(80) + } + if i < 8 { + sizes = [][]int{nil, {0}, {32, 4}, {55}, {56}, {32, 24}, {4096}, {0, 32, 0, 4}}[i] + } + mem, a, n, out := sha256Fixture(sizes) + budget := uint64(100000) + switch i % 11 { + case 1: + budget = uint64(rng.Intn(250)) + case 2: + out = 1 + case 3: + a = 1 + case 4: + if n > 0 { + binary.LittleEndian.PutUint64(mem, 1) + } + case 5: + if n > 0 { + binary.LittleEndian.PutUint64(mem[8:], ^uint64(0)) + } + case 6: + out = sbpf.VaddrInput + 8192 // output overlaps the input + case 7: + out = a // output overlaps descriptors + case 8: + n = cu.CUSha256MaxSlices + 1 + case 9: + n = cu.CUSha256MaxSlices + case 10: + a = sbpf.VaddrInput + uint64(len(mem)-1) + } + var wantMem []byte + var wantRet, wantCU uint64 + var wantErr string + for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + buf := append([]byte(nil), mem...) + vm, ctx := sha256VM(buf, budget) + ret, err := fn(vm, a, n, out) + remaining := ctx.ComputeMeter.Remaining() + vm.Finish() + if k == 0 { + wantMem = buf + wantRet = ret + wantCU = remaining + wantErr = fmt.Sprint(err) + continue + } + require.Equal(t, wantRet, ret, "case %d variant %d", i, k) + require.Equal(t, wantErr, fmt.Sprint(err), "case %d variant %d", i, k) + require.Equal(t, wantCU, remaining, "case %d variant %d", i, k) + require.Equal(t, wantMem, buf, "case %d variant %d", i, k) + } + } +} + +var sha256BenchDigest [32]byte + +func BenchmarkSha256Syscall(b *testing.B) { + for _, tc := range []struct { + name string + sizes []int + }{{"empty", nil}, {"36_contiguous", []int{36}}, {"32_plus_4", []int{32, 4}}, {"55", []int{55}}, {"56", []int{56}}, {"1232", []int{1232}}, {"4096", []int{4096}}} { + for _, variant := range []struct { + name string + fn sha256Call + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + b.Run(tc.name+"/"+variant.name, func(b *testing.B) { + mem, a, n, out := sha256Fixture(tc.sizes) + vm, ctx := sha256VM(mem, ^uint64(0)) + defer vm.Finish() + b.ReportAllocs() + b.ResetTimer() + for j := 0; j < b.N; j++ { + ctx.ComputeMeter = cu.NewComputeMeter(100000) + if _, err := variant.fn(vm, a, n, out); err != nil { + b.Fatal(err) + } + } + }) + } + } + b.Run("raw36", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + sha256BenchDigest = sha256.Sum256(data[:]) + } + }) +} From d85514147cfc7944799e342bb2d8a81d889ed1d7 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:20:08 -0500 Subject: [PATCH 024/111] test: measure captured SHA loop and document Zen 5 acceleration --- docs/sha256-syscall.md | 47 ++++++++++ pkg/sealevel/syscalls_sha256_loop_test.go | 102 ++++++++++++++++++++++ 2 files changed, 149 insertions(+) create mode 100644 pkg/sealevel/syscalls_sha256_loop_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 58a83abca..10c1b59b2 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -40,3 +40,50 @@ Testing exposed pre-existing overflow in contiguous VM region bounds checks; that correction and its regression test are a separate preceding commit. Both benchmark variants use the corrected VM. The old SHA fixture also needed its compute-meter pointer initialized for the current interpreter API. + + +## Zen 5 acceleration and captured loop + +The live Go 1.26.4 validator on Ryzen 7 9700X had its actual +`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is +already active. No validator runtime setting was changed. + +The isolated loop harness copies text slots 518–545 from the captured SBF v0 +program, resolves the SHA syscall relocation, and supplies 1,000 iterations and +zero initial state. It retains descriptor setup, stack accesses, digest copying, +counter update and loop branching. A test checks its result against a Go hash +chain and checks equal CU consumption for both syscall implementations. This +excludes transaction loading, account dependencies, CPI and the remaining program. + +A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, +affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. +Three 150 ms samples (medians, per hash iteration): + +| Isolated loop | Time | +|---|---:| +| Original syscall | 234.5 ns | +| Optimized syscall | 165.9 ns | +| Dispatch-only diagnostic control | 109.2 ns | +| Go hash chain without VM | 54.69 ns | + +The optimized loop takes about 29% less time. The dispatch-only control replaces +the syscall with a no-op: it omits hashing, translations and syscall CU charging, +and is only an overhead diagnostic, never a valid execution implementation. +The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; +the buffered-input prototype remained slower at 71–73 ns. These short tests +share a host with other processes; they are not isolated-core latency guarantees. + +A separate short CPU profile of the optimized loop attributed 36.2% cumulative +sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI +compression routine. Most remaining sampled work was VM execution: instruction +dispatch/decoding, stack address translation, loads/stores and compute metering. +Cumulative and flat percentages overlap and must not be added. This profile is +of the harness, not of full-block replay or the live validator. + +The next execution experiment should target measured VM overhead and then replay +captured blocks; these results do not justify a claimed 29% block-time improvement. + +```sh +go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +``` diff --git a/pkg/sealevel/syscalls_sha256_loop_test.go b/pkg/sealevel/syscalls_sha256_loop_test.go new file mode 100644 index 000000000..32c22d9f0 --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_loop_test.go @@ -0,0 +1,102 @@ +package sealevel + +import ( + "crypto/sha256" + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +// This is text slots 518..545 from the captured AogGeA81 program's hash loop. +// Captured ELF SHA-256: b3286f96f5611ee7db62dc11ea9aab1afe80908c08a0fbd3fdd783879d70ed88. +// The captured ELF uses SBF v0. Only the unresolved syscall relocation is replaced. The harness supplies a +// loop bound and zero initial state; transaction loading, CPI and the rest of +// the program are deliberately excluded. +func sha256LoopProgram(iterations uint32) *sbpf.Program { + text := []sbpf.Slot{ + sbpf.Slot(sbpf.OpMov64Imm) | 7<<8 | sbpf.Slot(iterations)<<32, + 0xa1bf, 0xffffffb000000107, 0xfe701a7b, 0xa1bf, 0xfffffe1000000107, + 0xfe601a7b, 0xffb08a63, 0x4fe780a7a, 0x20fe680a7a, 0xa1bf, + 0xfffffe6000000107, 0xa3bf, 0xffffffe000000307, 0x2000002b7, + sbpf.Slot(sbpf.OpCall) | sbpf.Slot(hash_sol_sha256)<<32, + 0xffe0a179, 0xfe101a7b, 0xffe8a179, 0xfe181a7b, 0xfff0a179, + 0xfe201a7b, 0xfff8a179, 0xfe281a7b, 0x100000807, 0x81bf, + 0x2000000167, 0x2000000177, 0xffe471ad, + // Return the first digest word so the harness can check the computation. + 0xfe10a079, sbpf.Slot(sbpf.OpExit), + } + return &sbpf.Program{Text: text} +} + +func runSha256Loop(p *sbpf.Program, fn sha256Call) (uint64, uint64, error) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(10000000)} + vm := sbpf.NewInterpreter(p, &sbpf.VMOpts{Context: ctx, ComputeMeter: &ctx.ComputeMeter, Syscalls: func(hash uint32) (sbpf.Syscall, bool) { return sbpf.SyscallFunc3(fn), hash == hash_sol_sha256 }}) + ret, _, err := vm.Run() + used := ctx.ComputeMeter.Used() + vm.Finish() + return ret, used, err +} + +func TestSha256CapturedLoop(t *testing.T) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + require.NoError(t, p.Verify()) + var data [36]byte + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + want := binary.LittleEndian.Uint64(data[:8]) + var wantCU uint64 + for _, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { + got, used, err := runSha256Loop(p, fn) + require.NoError(t, err) + require.Equal(t, want, got) + if wantCU == 0 { + wantCU = used + } + require.Equal(t, wantCU, used) + } +} + +func BenchmarkSha256CapturedLoop(b *testing.B) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + for _, v := range []struct { + name string + fn sha256Call + }{ + {"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, + // Diagnostic lower bound: no hashing, translations or syscall CU charging. + // It is not a valid implementation and must never be used in replay. + {"dispatch_only", func(sbpf.VM, uint64, uint64, uint64) (uint64, error) { return 0, nil }}, + } { + b.Run(v.name, func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + if _, _, err := runSha256Loop(p, v.fn); err != nil { + b.Fatal(err) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) + } + b.Run("raw_chain", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + clear(data[:]) + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + } + sha256BenchDigest = sha256.Sum256(data[:]) + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) +} From ef418c7c8881029c757f773f1060c211b35635f5 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 025/111] sbpf: cut per-instruction overhead in the interpreter (~2x on SPL Token transfer) Consensus-neutral performance changes to pkg/sbpf, validated against the unmodified interpreter with a 100k-program differential corpus (identical return values, errors/PCs, CU consumed, meter remaining, memory contents, input-region state) plus the package's unit tests: - meter instructions with a local due/budget pair synced around syscalls and on exit (Agave's due_insn_count scheme) instead of calling ComputeMeter.Consume per instruction - move cold opcodes to executeCold so Run drops below the compiler's "big function" threshold and Consume/Read*/Push/Pop/fast paths inline - zero only the dirty range of the pooled stack/heap in Finish (page bitmap on the fast path, byte range on the translate path) instead of 256 KiB + HeapMax per execution - per-window fast-path address translation table (Agave aligned mapping layout, branch-free v0 frame gaps, one-entry cache for VASA input regions) - 16-wide register file (no bounds checks on r[dst]/r[src]), in-place call-frame Push/Pop, precomputed internal call targets per Program pooling_test writes through the VM's translation layer now, since the pool only re-zeroes memory the VM saw written (all production writes go through translation). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/cu/cu.go | 5 + pkg/sbpf/fastmem.go | 59 ++ pkg/sbpf/interpreter.go | 1236 +++++++++++++++++++++---------------- pkg/sbpf/loader/loader.go | 4 +- pkg/sbpf/pooling_test.go | 26 +- pkg/sbpf/program.go | 23 + pkg/sbpf/stack.go | 74 ++- 7 files changed, 873 insertions(+), 554 deletions(-) create mode 100644 pkg/sbpf/fastmem.go diff --git a/pkg/cu/cu.go b/pkg/cu/cu.go index 79d337199..f90aa1e2c 100644 --- a/pkg/cu/cu.go +++ b/pkg/cu/cu.go @@ -33,6 +33,11 @@ func (cm *ComputeMeter) Consume(cost uint64) error { return nil } +// Disabled reports whether metering is currently switched off (Consume is a no-op). +func (cm *ComputeMeter) Disabled() bool { + return cm.disable +} + func (cm *ComputeMeter) Used() uint64 { return cm.startingBalance - cm.computeMeter } diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go new file mode 100644 index 000000000..421f3ed42 --- /dev/null +++ b/pkg/sbpf/fastmem.go @@ -0,0 +1,59 @@ +package sbpf + +import "unsafe" + +// memRegion describes one of the fixed 4 GiB virtual address windows +// (rodata, stack, heap, input) as a contiguous host buffer, so that the +// interpreter can translate the common case without a function call. +// +// rlen / wlen are the readable / writable byte lengths (wlen == 0 for +// read-only windows). gapShift/gapMask implement SBPF v0 stack frame gaps +// exactly like Agave's MemoryRegion::vm_gap_shift (gapShift = 63 and +// gapMask = 0 for windows without gaps, which makes the gap logic a no-op). +// A window that needs special handling (VASA input regions, ...) has +// rlen = wlen = 0 and falls back to translateInternal. +type memRegion struct { + base unsafe.Pointer // host address of window offset `start` + start uint64 // offset of the region within its 4 GiB window + rlen uint64 + wlen uint64 + gapShift uint64 + gapMask uint64 + dirty uint64 // 4 KiB page bitmap of fast-path writes (stack/heap only matter) +} + +// emptyRegion never matches any access. +var emptyRegion = memRegion{gapShift: 63} + +const numFastRegions = 6 // index 5 is a permanently empty catch-all + +// fastRead returns a host pointer for a size-byte read at vma, or nil if the +// access is not covered by the fast path (caller falls back to Read*). +func (ip *Interpreter) fastRead(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + // Truncating to 32 bits makes lo < start wrap to a value >= 2^32-start, + // which is always > rlen (start+rlen < 2^32), so one compare suffices. + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.rlen || inGap != 0 { + return nil + } + return unsafe.Add(reg.base, off) +} + +// fastWrite is the write counterpart of fastRead; it also records the dirty +// range so Finish only needs to zero what was touched. +func (ip *Interpreter) fastWrite(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.wlen || inGap != 0 { + return nil + } + // Mark the 4 KiB page (and, conservatively, the next one, since an access + // is at most 8 bytes and may straddle a page boundary) as dirty. + reg.dirty |= 3 << (off >> 12) + return unsafe.Add(reg.base, off) +} diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 5f351a5d6..e7e28fda0 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1,6 +1,7 @@ package sbpf import ( + "errors" "fmt" "math" "math/bits" @@ -46,6 +47,31 @@ type Interpreter struct { sbpfVersion sbpfver.SbpfVersion programId solana.PublicKey txSignature solana.Signature + + // callTargets[pc] is the resolved internal-function target of the `call imm` + // at pc (or -1). Computed once per Program at load time. + callTargets []int64 + + // Fast-path translation table indexed by (vaddr >> 32); see fastmem.go. + regions [numFastRegions]memRegion + // dirtyLo/dirtyHi: byte range written through translateInternal; + // memRegion.dirty: 4 KiB page bitmap of writes through the fast path. + dirtyLo [numFastRegions]uint64 + dirtyHi [numFastRegions]uint64 +} + +// dirtyRange returns the union of the byte ranges that may have been written +// in window idx (stack or heap), as [lo, hi). +func (ip *Interpreter) dirtyRange(idx uint64, size uint64) (lo, hi uint64) { + lo, hi = ip.dirtyLo[idx], ip.dirtyHi[idx] + if pages := ip.regions[idx].dirty; pages != 0 { + plo := uint64(bits.TrailingZeros64(pages)) << 12 + phi := uint64(bits.Len64(pages)) << 12 + lo = min(lo, plo) + hi = max(hi, phi) + } + hi = min(hi, size) + return lo, hi } type TraceSink interface { @@ -76,12 +102,13 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { heap = slices.Grow(heap, opts.HeapMax-len(heap)) } heap = heap[:opts.HeapMax] - clear(heap) + // Buffers in the pool are zeroed (for their dirty range) in Finish, so + // no clear is needed here. } else { heap = newHeap() } - return &Interpreter{ + ip := &Interpreter{ textVA: p.TextVA, textBytes: p.TextBytes, text: p.Text, @@ -103,17 +130,56 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { sbpfVersion: p.SbpfVersion, programId: opts.ProgramId, txSignature: opts.TxSignature, + callTargets: p.CallTargets, + } + ip.initRegions() + return ip +} + +// initRegions fills the fast-path translation table. Windows that need the +// full logic in translateInternal are left empty (rlen = wlen = 0). +func (ip *Interpreter) initRegions() { + for i := range ip.dirtyLo { + ip.dirtyLo[i] = math.MaxUint64 + ip.dirtyHi[i] = 0 + ip.regions[i].gapShift = 63 + } + if len(ip.ro) != 0 { + idx := VaddrProgram >> 32 + if ip.sbpfVersion.EnableLowerRodataVaddr() { + idx = 0 + } + ip.regions[idx] = memRegion{base: unsafe.Pointer(&ip.ro[0]), rlen: uint64(len(ip.ro)), gapShift: 63} + } + if len(ip.stack.mem) != 0 { + r := memRegion{base: unsafe.Pointer(&ip.stack.mem[0]), rlen: StackMax, wlen: StackMax, gapShift: 63} + if ip.stack.stackFrameGaps { + r.gapShift = 12 // log2(StackFrameSize) + r.gapMask = GapMask + } + ip.regions[VaddrStack>>32] = r + } + if len(ip.heap) != 0 { + ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} + } + if len(ip.inputRegions) == 0 && len(ip.input) != 0 { + ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } } func (ip *Interpreter) Finish() { if UsePool { + lo, hi := ip.dirtyRange(VaddrHeap>>32, uint64(len(ip.heap))) + if hi > lo { + clear(ip.heap[lo:hi]) + } heapPool.Put(ip.heap) } + ip.stack.MarkDirty(ip.dirtyRange(VaddrStack>>32, StackMax)) ip.stack.Finish() } -func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, error) { +func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[16]uint64) (int64, error) { var taken bool dst := uint32(r[ins.Dst()]) src := uint32(r[ins.Src()]) @@ -200,7 +266,7 @@ func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, e // // This function may panic given code that doesn't pass the static verifier. func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { - var r [11]uint64 + var r [16]uint64 // 16 (not 11) so that r[ins.Dst()] (4-bit field) needs no bounds check r[1] = VaddrInput r[2] = ip.inputDataVaddr @@ -216,137 +282,228 @@ func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { // initialize pc to program entry point pc := int64(ip.entry) + // Loop-invariant state hoisted into locals so the compiler can keep them in + // registers (fields of ip may alias with the unsafe stores in the loop and + // would otherwise be reloaded on every instruction). + text := ip.text + tracing := ip.enableTracing + jmp32 := ip.sbpfVersion.EnableJmp32() + moveMem := ip.sbpfVersion.MoveMemoryInstructionClasses() + pqr := ip.sbpfVersion.EnablePqr() + staticSyscalls := ip.sbpfVersion.EnableStaticSyscalls() + callTargets := ip.callTargets + + // Instruction metering (mirrors Agave's due_insn_count / previous_instruction_meter): + // count executed instructions locally and only sync with the shared compute + // meter around syscalls and on exit. `budget` is the number of instructions + // we may still execute before the meter would be exhausted. + meter := ip.computeMeter + var budget, due uint64 + reloadBudget := func() { + if meter.Disabled() { + budget = math.MaxUint64 + } else { + budget = meter.Remaining() + } + due = 0 + // A syscall (CPI in particular) may have changed the input regions; + // drop the cached input-region fast path entry, it is re-populated on + // the next slow-path translation. (With a plain, region-less input the + // entry is static and stays.) + if len(ip.inputRegions) != 0 { + ip.regions[VaddrInput>>32] = emptyRegion + } + } + flushDue := func() { + if due != 0 { + _ = meter.Consume(due) + due = 0 + } + } + reloadBudget() + mainLoop: for i := 0; true; i++ { // Fetch - if pc < 0 || pc >= int64(len(ip.text)) { + if pc < 0 || pc >= int64(len(text)) { + flushDue() return 0, 0, &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, ExcExecutionOverrun), } } - ins := ip.getSlot(pc) - if ip.enableTracing { + ins := text[pc] + if tracing { regsDump := fmt.Sprintf("%016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x", r[0], r[1], r[2], r[3], r[4], r[5], r[6], r[7], r[8], r[9], r[10]) fmt.Printf("% 5d [%s]: %s\n", i, strings.ToUpper(regsDump), ip.disassemble(ins, 0)) } - err = ip.computeMeter.Consume(1) - if err != nil { + // Meter: identical semantics to Consume(1) before each instruction. + if due == budget { + err = cu.ErrComputeExceeded break mainLoop } + due++ // Execute - if ip.sbpfVersion.EnableJmp32() && ins.Op()&0x07 == ClassPqr { + if jmp32 && ins.Op()&0x07 == ClassPqr { pc, err = ip.executeJmp32(ins, pc, &r) goto postExecute } switch ins.Op() { case OpLdxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 1); p != nil { + r[ins.Dst()] = uint64(*(*uint8)(p)) + } else { + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 2); p != nil { + r[ins.Dst()] = uint64(*(*uint16)(p)) + } else { + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 4); p != nil { + r[ins.Dst()] = uint64(*(*uint32)(p)) + } else { + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v + if p := ip.fastRead(vma, 8); p != nil { + r[ins.Dst()] = uint64(*(*uint64)(p)) + } else { + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpStb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(ins.Uimm()) + } else { + err = ip.Write8(vma, uint8(ins.Uimm())) + } pc++ case OpSth: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(ins.Uimm()) + } else { + err = ip.Write16(vma, uint16(ins.Uimm())) + } pc++ case OpStw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(ins.Uimm()) + } else { + err = ip.Write32(vma, uint32(ins.Uimm())) + } pc++ case OpStdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(ins.Imm()) + } else { + err = ip.Write64(vma, uint64(ins.Imm())) + } pc++ case OpStxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(r[ins.Src()]) + } else { + err = ip.Write8(vma, uint8(r[ins.Src()])) + } pc++ case OpStxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(r[ins.Src()]) + } else { + err = ip.Write16(vma, uint16(r[ins.Src()])) + } pc++ case OpStxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(r[ins.Src()]) + } else { + err = ip.Write32(vma, uint32(r[ins.Src()])) + } pc++ case OpStxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(r[ins.Src()]) + } else { + err = ip.Write64(vma, uint64(r[ins.Src()])) + } pc++ case OpAdd32Imm: r[ins.Dst()] = ip.signExtension(int32(r[ins.Dst()]) + ins.Imm()) @@ -383,343 +540,6 @@ mainLoop: case OpMul32Imm: r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * ins.Imm()) pc++ - case OpMul32Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd1BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpMul64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) - pc++ - } - case OpMul64Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= r[ins.Src()] - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) - pc++ - } - case OpDiv32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpDiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd2BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpLd4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) - pc++ - case OpDiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] /= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) - pc++ - } - case OpDiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) - pc++ - } - case OpSt4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) - pc++ - case OpLmul32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) - pc++ - case OpLmul32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) - pc++ - case OpLmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= uint64(int64(ins.Imm())) - pc++ - case OpLmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= r[ins.Src()] - pc++ - case OpUhmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) - r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() - pc++ - case OpUhmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) - r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() - pc++ - case OpShmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - imm128 := wide.Int128FromInt64(int64(ins.Imm())) - r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() - pc++ - case OpShmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - src128 := wide.Int128FromInt64(int64(r[ins.Src()])) - r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() - pc++ - case OpUdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpUdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] /= uint64(ins.Uimm()) - pc++ - case OpUdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpUrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] = uint64(r[ins.Dst()] % src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] %= uint64(ins.Uimm()) - pc++ - case OpUrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= r[ins.Src()] - } else { - err = ExcDivideByZero - } - pc++ - case OpSdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) - pc++ - case OpSdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) - pc++ - case OpSdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) - pc++ - case OpSrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) - pc++ - case OpSrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) - } else { - err = ExcDivideByZero - break - } - pc++ case OpOr32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) | ins.Uimm()) pc++ @@ -768,66 +588,6 @@ mainLoop: case OpRsh64Reg: r[ins.Dst()] >>= r[ins.Src()] & 0x3f pc++ - case OpNeg32: - if ip.sbpfVersion.DisableNeg() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) - pc++ - case OpNeg64: - if !ip.sbpfVersion.DisableNeg() { - r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt4BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) - pc++ - } - case OpMod32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpMod32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd8BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v - pc++ - } - case OpMod64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] %= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) - pc++ - } - case OpMod64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) - pc++ - } case OpXor32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) ^ ins.Uimm()) pc++ @@ -856,53 +616,6 @@ mainLoop: case OpMov64Reg: r[ins.Dst()] = r[ins.Src()] pc++ - case OpArsh32Imm: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) - pc++ - case OpArsh32Reg: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) - pc++ - case OpArsh64Imm: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) - pc++ - case OpArsh64Reg: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) - pc++ - case OpHor64Imm: - if !ip.sbpfVersion.DisableLddw() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] |= uint64(ins.Uimm()) << 32 - pc++ - case OpLe: - if ip.sbpfVersion.DisableLe() { - err = ExcInvalidInstr - break - } - switch ins.Uimm() { - case 16: - r[ins.Dst()] &= math.MaxUint16 - case 32: - r[ins.Dst()] &= math.MaxUint32 - case 64: - r[ins.Dst()] &= math.MaxUint64 - default: - err = ExcUnsupportedInstruction - } - pc++ - case OpBe: - switch ins.Uimm() { - case 16: - r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) - case 32: - r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) - case 64: - r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) - default: - err = ExcUnsupportedInstruction - } - pc++ case OpLddw: if ip.sbpfVersion.DisableLddw() { err = ExcInvalidInstr @@ -1024,14 +737,16 @@ mainLoop: } pc++ case OpCall: - if ip.sbpfVersion.EnableStaticSyscalls() { + if staticSyscalls { if ins.Src() == 0 { sc, ok := ip.syscalls(ins.Uimm()) if !ok { err = ExcCallDest{ins.Uimm()} break } + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } @@ -1042,7 +757,7 @@ mainLoop: err = ExcCallDest{uint32(targetPC)} break } - if ok := ip.stack.Push(r[:], pc+1); !ok { + if ok := ip.stack.Push(&r, pc+1); !ok { err = ExcCallDepth } pc = targetPC @@ -1051,60 +766,147 @@ mainLoop: } } else { if sc, ok := ip.syscalls(ins.Uimm()); ok { + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } pc++ - } else if target, ok := ip.funcs[ins.Uimm()]; ok { - ok = ip.stack.Push(r[:], pc+1) + } else { + var target int64 + var ok bool + if callTargets != nil { + target = callTargets[pc] + ok = target >= 0 + } else { + target, ok = ip.funcs[ins.Uimm()] + } if !ok { + err = ExcCallDest{ins.Uimm()} + break + } + if !ip.stack.Push(&r, pc+1) { err = ExcCallDepth } pc = target - } else { - err = ExcCallDest{ins.Uimm()} } } - case OpCallx: - var target uint64 - if ip.sbpfVersion.CallXUsesSrcReg() { - target = r[ins.Src()] - } else if ip.sbpfVersion.CallXUsesDstReg() { - target = r[ins.Dst()] + case OpExit: + var ok bool + pc, ok = ip.stack.Pop(&r) + if !ok { + ret = r[0] + break mainLoop + } + case OpMul32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) + pc++ + case OpMul64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= uint64(ins.Imm()) + pc++ + case OpMul64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpDiv32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) } else { - target = r[ins.Uimm()] + err = ExcDivideByZero } - - if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { - err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + pc++ + case OpDiv64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - targetPC := int64((target - ip.textVA) / 8) - if ok := ip.stack.Push(r[:], pc+1); !ok { - err = ExcCallDepth + r[ins.Dst()] /= uint64(ins.Imm()) + pc++ + case OpDiv64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - pc = targetPC - case OpExit: - var ok bool - pc, ok = ip.stack.Pop(r[:]) - if !ok { - ret = r[0] - break mainLoop + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero } + pc++ + case OpMod32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpMod64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] %= uint64(ins.Imm()) + pc++ + case OpMod64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= src + } else { + err = ExcDivideByZero + } + pc++ + case OpArsh32Imm: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) + pc++ + case OpArsh32Reg: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) + pc++ + case OpArsh64Imm: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) + pc++ + case OpArsh64Reg: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) + pc++ default: - err = ExcUnsupportedInstruction - return + pc, err = ip.executeCold(ins, pc, &r) + if err == errUnknownOpcode { + // Preserve the original behaviour for an unknown opcode: + // a bare (unwrapped) ExcUnsupportedInstruction. + flushDue() + return 0, 0, ExcUnsupportedInstruction + } } // Post execute postExecute: - if err == cu.ErrComputeExceeded { - err = ExcOutOfCU - } - if err != nil { + flushDue() + if err == cu.ErrComputeExceeded { + err = ExcOutOfCU + } exc := &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, err), @@ -1117,6 +919,9 @@ mainLoop: } } + flushDue() + // NB: when the loop exits because the meter is exhausted, err is the bare + // cu.ErrComputeExceeded (not wrapped in an Exception), as before. cuConsumed = ip.initialInstrMeter - ip.computeMeter.Remaining() return @@ -1198,6 +1003,11 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } + if write { + off := StackMax - uint64(len(mem)) + ip.dirtyLo[VaddrStack>>32] = min(ip.dirtyLo[VaddrStack>>32], off) + ip.dirtyHi[VaddrStack>>32] = max(ip.dirtyHi[VaddrStack>>32], off+size) + } return unsafe.Pointer(&mem[0]), nil case VaddrHeap >> 32: if size == 0 { @@ -1206,6 +1016,10 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } + if write { + ip.dirtyLo[VaddrHeap>>32] = min(ip.dirtyLo[VaddrHeap>>32], lo) + ip.dirtyHi[VaddrHeap>>32] = max(ip.dirtyHi[VaddrHeap>>32], lo+size) + } return unsafe.Pointer(&ip.heap[lo]), nil case VaddrInput >> 32: if size == 0 { @@ -1255,6 +1069,8 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } if write && (!region.Writable || requestedLen > region.RegionSize) && region.OnWrite != nil { + // The callback may replace region.Data / grow the region: drop the cache. + ip.regions[VaddrInput>>32] = emptyRegion if err := region.OnWrite(region, requestedLen); err != nil { return nil, err } @@ -1263,23 +1079,37 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un if !write || !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } + ip.regions[VaddrInput>>32] = emptyRegion region.RegionSize = region.AddressSpaceReserved } if write && !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "write to readonly input region") } + var base unsafe.Pointer if region.Data != nil { if requestedLen > uint64(len(region.Data)) { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } - return unsafe.Pointer(®ion.Data[regionOffset]), nil + base = unsafe.Pointer(unsafe.SliceData(region.Data)) + } else { + hostOffset := region.HostOffset + regionOffset + if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { + return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + } + base = unsafe.Pointer(&ip.input[region.HostOffset]) } - - hostOffset := region.HostOffset + regionOffset - if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { - return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + // Cache this region for the interpreter's fast path (one-entry cache, + // same idea as Agave's MappingCache). Only the currently mapped + // RegionSize bytes are exposed; anything beyond takes the slow path + // again so that OnWrite / growth semantics are preserved. + if region.RegionSize != 0 && (region.Data == nil || uint64(len(region.Data)) >= region.RegionSize) { + cached := memRegion{base: base, start: region.Offset, rlen: region.RegionSize, gapShift: 63} + if region.Writable { + cached.wlen = region.RegionSize + } + ip.regions[VaddrInput>>32] = cached } - return unsafe.Pointer(&ip.input[hostOffset]), nil + return unsafe.Add(base, regionOffset), nil } func (ip *Interpreter) TranslateInput(addr uint64, size uint64) ([]byte, error) { @@ -1335,6 +1165,7 @@ func (ip *Interpreter) SetInputRegionData(addr uint64, data []byte, length uint6 } region.RegionSize = length region.Writable = writable + ip.regions[VaddrInput>>32] = emptyRegion return true } @@ -1461,3 +1292,360 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { *(*uint64)(ptr) = x return nil } + +// executeCold handles the less frequently executed opcodes. Keeping them out of +// Run keeps that function below the compiler's "big function" threshold so the +// hot helpers (metering, fast memory translation, stack push/pop) stay inlinable. +func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { + var err error + switch ins.Op() { + case OpDiv32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpLd4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, uint32(r[ins.Src()])) + pc++ + case OpLmul32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) + pc++ + case OpLmul32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) + pc++ + case OpLmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= uint64(int64(ins.Imm())) + pc++ + case OpLmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpUhmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) + r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() + pc++ + case OpUhmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) + r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() + pc++ + case OpShmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + imm128 := wide.Int128FromInt64(int64(ins.Imm())) + r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() + pc++ + case OpShmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + src128 := wide.Int128FromInt64(int64(r[ins.Src()])) + r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() + pc++ + case OpUdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpUdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] /= uint64(ins.Uimm()) + pc++ + case OpUdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpUrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] = uint64(r[ins.Dst()] % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] %= uint64(ins.Uimm()) + pc++ + case OpUrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= r[ins.Src()] + } else { + err = ExcDivideByZero + } + pc++ + case OpSdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) + pc++ + case OpSdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) + pc++ + case OpSdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) + pc++ + case OpSrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) + pc++ + case OpSrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpNeg32: + if ip.sbpfVersion.DisableNeg() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) + pc++ + case OpNeg64: + if !ip.sbpfVersion.DisableNeg() { + r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) + pc++ + } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { + // OpSt4BImm + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, ins.Uimm()) + pc++ + } + case OpMod32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpHor64Imm: + if !ip.sbpfVersion.DisableLddw() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] |= uint64(ins.Uimm()) << 32 + pc++ + case OpLe: + if ip.sbpfVersion.DisableLe() { + err = ExcInvalidInstr + break + } + switch ins.Uimm() { + case 16: + r[ins.Dst()] &= math.MaxUint16 + case 32: + r[ins.Dst()] &= math.MaxUint32 + case 64: + r[ins.Dst()] &= math.MaxUint64 + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpBe: + switch ins.Uimm() { + case 16: + r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) + case 32: + r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) + case 64: + r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpCallx: + var target uint64 + if ip.sbpfVersion.CallXUsesSrcReg() { + target = r[ins.Src()] + } else if ip.sbpfVersion.CallXUsesDstReg() { + target = r[ins.Dst()] + } else { + target = r[ins.Uimm()] + } + + if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { + err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + break + } + targetPC := int64((target - ip.textVA) / 8) + if ok := ip.stack.Push(r, pc+1); !ok { + err = ExcCallDepth + break + } + pc = targetPC + default: + err = errUnknownOpcode + } + return pc, err +} + +// errUnknownOpcode is an internal sentinel returned by executeCold for an +// opcode that is not handled by either switch; Run turns it into the bare +// ExcUnsupportedInstruction return of the original implementation. +var errUnknownOpcode = errors.New("unknown opcode") diff --git a/pkg/sbpf/loader/loader.go b/pkg/sbpf/loader/loader.go index 7b9dc52ca..23a26ec76 100644 --- a/pkg/sbpf/loader/loader.go +++ b/pkg/sbpf/loader/loader.go @@ -162,7 +162,7 @@ func parseSlots(bs []byte) []sbpf.Slot { } func (l *Loader) getProgram() *sbpf.Program { - return &sbpf.Program{ + p := &sbpf.Program{ RO: l.program, TextBytes: l.text, Text: parseSlots(l.text), @@ -171,4 +171,6 @@ func (l *Loader) getProgram() *sbpf.Program { Funcs: l.funcs, SbpfVersion: l.sbpfVersion(), } + p.ResolveCallTargets() + return p } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index 38d1028f8..c4bc5eb0d 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -30,14 +30,28 @@ func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { t.Error("pooled VM exposed data from an earlier execution") } - parent.heap[0] = byte(worker + 1) - parent.stack.mem[0] = byte(worker + 1) + // Write through the VM's translation layer, as programs and + // syscalls do: the pool only re-zeroes memory the VM saw written. + if err := parent.Write8(VaddrHeap, byte(worker+1)); err != nil { + t.Error(err) + } + if err := parent.Write8(VaddrStack, byte(worker+1)); err != nil { + t.Error(err) + } child := poolingInterpreter(256 * 1024) - for j := range child.heap { - child.heap[j] = 0xab + childHeap, err := child.Translate(VaddrHeap, uint64(len(child.heap)), true) + if err != nil { + t.Error(err) + } + for j := range childHeap { + childHeap[j] = 0xab + } + childStack, err := child.Translate(VaddrStack, StackMax, true) + if err != nil { + t.Error(err) } - for j := range child.stack.mem { - child.stack.mem[j] = 0xcd + for j := range childStack { + childStack[j] = 0xcd } child.Finish() if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index c112be14d..969a15469 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -13,6 +13,29 @@ type Program struct { Entrypoint uint64 // PC Funcs map[uint32]int64 SbpfVersion sbpfver.SbpfVersion + + // CallTargets[pc] holds the resolved internal function target for a + // `call imm` slot at pc (non-static-syscall versions), or -1. + CallTargets []int64 +} + +// ResolveCallTargets precomputes CallTargets from Funcs so the interpreter +// does not need a map lookup per call instruction. +func (p *Program) ResolveCallTargets() { + if p.SbpfVersion.EnableStaticSyscalls() { + p.CallTargets = nil + return + } + targets := make([]int64, len(p.Text)) + for pc, slot := range p.Text { + targets[pc] = -1 + if slot.Op() == OpCall { + if t, ok := p.Funcs[slot.Uimm()]; ok { + targets[pc] = t + } + } + } + p.CallTargets = targets } func (p *Program) MemoryBytes() uint64 { diff --git a/pkg/sbpf/stack.go b/pkg/sbpf/stack.go index 3f3331b6e..edc662c18 100644 --- a/pkg/sbpf/stack.go +++ b/pkg/sbpf/stack.go @@ -36,6 +36,12 @@ type Stack struct { shadow []Frame dynamicStackFrames bool stackFrameGaps bool + // dirtyLo/dirtyHi bound the physical byte range of mem that may have been + // written during this execution. Finish only has to zero this range before + // returning the buffer to the pool. + dirtyLo uint64 + dirtyHi uint64 + maxDepth int } // Frame is an entry on the shadow stack. @@ -92,9 +98,11 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { } s := Stack{ - mem: m, - sp: VaddrStack, - shadow: sh, + mem: m, + sp: VaddrStack, + shadow: sh, + dirtyLo: StackMax, + dirtyHi: 0, } var sz uint64 @@ -115,10 +123,12 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { func (s *Stack) Finish() { if UsePool { s.mem = s.mem[:StackMax] - clear(s.mem) + if s.dirtyHi > s.dirtyLo { + clear(s.mem[s.dirtyLo:s.dirtyHi]) + } stackMemPool.Put(s.mem) s.shadow = s.shadow[:StackDepth] - clear(s.shadow) + clear(s.shadow[:max(s.maxDepth, 1)]) s.shadow = s.shadow[:1] stackShadowPool.Put(s.shadow) } @@ -153,21 +163,38 @@ func (s *Stack) GetFrame(addr uint32) []byte { } } +// MarkDirty records that physical stack bytes [off, off+size) may be written. +func (s *Stack) MarkDirty(lo, hi uint64) { + if hi <= lo { + return + } + s.dirtyLo = min(s.dirtyLo, lo) + s.dirtyHi = max(s.dirtyHi, hi) +} + // Push allocates a new call frame. // // Saves the given nonvolatile regs, return address, // and current frame pointer. // Returns the new frame pointer. -func (s *Stack) Push(regs []uint64, ret int64) bool { - if ok := len(s.shadow) < cap(s.shadow); !ok { +func (s *Stack) Push(regs *[16]uint64, ret int64) bool { + n := len(s.shadow) + if n >= cap(s.shadow) { return false } - - frame := Frame{RetAddr: ret} - copy(frame.NVRegs[:], regs[6:10]) - frame.FramePtr = regs[10] - - s.shadow = append(s.shadow, frame) + // Write the frame in place (no temporary Frame value / copy) to avoid + // store-forwarding stalls in this very hot path. + s.shadow = s.shadow[:n+1] + f := &s.shadow[n] + f.RetAddr = ret + f.NVRegs[0] = regs[6] + f.NVRegs[1] = regs[7] + f.NVRegs[2] = regs[8] + f.NVRegs[3] = regs[9] + f.FramePtr = regs[10] + if n+1 > s.maxDepth { + s.maxDepth = n + 1 + } if !s.dynamicStackFrames { if s.stackFrameGaps { @@ -185,16 +212,17 @@ func (s *Stack) Push(regs []uint64, ret int64) bool { // Restores saved nonvolatile regs into provided slice. // Returns saved return address and returns true upon success, // and returns false if no call frames are left. -func (s *Stack) Pop(regs []uint64) (int64, bool) { - if len(s.shadow) <= 1 { +func (s *Stack) Pop(regs *[16]uint64) (int64, bool) { + n := len(s.shadow) + if n <= 1 { return 0, false } - - var frame Frame - frame, s.shadow = s.shadow[len(s.shadow)-1], s.shadow[:len(s.shadow)-1] - - copy(regs[6:10], frame.NVRegs[:]) - regs[10] = frame.FramePtr - - return frame.RetAddr, true + f := &s.shadow[n-1] + regs[6] = f.NVRegs[0] + regs[7] = f.NVRegs[1] + regs[8] = f.NVRegs[2] + regs[9] = f.NVRegs[3] + regs[10] = f.FramePtr + s.shadow = s.shadow[:n-1] + return f.RetAddr, true } From b9e723af1e8cd50a062d279d1be99b04c6038193 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 026/111] sbpf: add interpreter benchmarks and a differential test corpus - perf_bench_test.go: synthetic ALU / load-store / call loops and interpreter setup+teardown - loader/token_perf_bench_test.go: real SPL Token Transfer through the loader/verifier/interpreter with sealevel-equivalent syscalls, in the aligned and VASA input layouts - perf_differential_test.go: deterministic random program corpus; run on two builds with SBPF_DIFF_OUT= and diff the outputs; SBPF_CHECK_POOL_ZERO=1 asserts pooled buffers come back zeroed Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sbpf/loader/token_perf_bench_test.go | 425 +++++++++++++++++++++++ pkg/sbpf/perf_bench_test.go | 193 ++++++++++ pkg/sbpf/perf_differential_test.go | 338 ++++++++++++++++++ 3 files changed, 956 insertions(+) create mode 100644 pkg/sbpf/loader/token_perf_bench_test.go create mode 100644 pkg/sbpf/perf_bench_test.go create mode 100644 pkg/sbpf/perf_differential_test.go diff --git a/pkg/sbpf/loader/token_perf_bench_test.go b/pkg/sbpf/loader/token_perf_bench_test.go new file mode 100644 index 000000000..c2534e586 --- /dev/null +++ b/pkg/sbpf/loader/token_perf_bench_test.go @@ -0,0 +1,425 @@ +package loader_test + +import ( + "encoding/binary" + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" +) + +// ---- minimal syscall set mirroring pkg/sealevel semantics (CU costs + memory behaviour) ---- + +const ( + cuSyscallBase = 100 + cuMemOpBase = 10 + cuCpiBytesPerCU = 250 +) + +type stats struct { + logs int + memcpy int + memset int + memcmp int + memmove int + bytes uint64 +} + +var st stats + +func memOpConsume(vm sbpf.VM, n uint64) error { + cost := max(uint64(cuMemOpBase), n/cuCpiBytesPerCU) + return vm.ComputeMeter().Consume(cost) +} + +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { + srcBuf := make([]byte, n) // same allocation pattern as sealevel/syscalls_mem.go + err = vm.Read(src, srcBuf) + if err != nil { + return + } + err = vm.Write(dst, srcBuf) + return +} + +func isNonOverlapping(src, srcLen, dst, dstLen uint64) bool { + if src > dst { + return src-dst >= dstLen + } + return dst-src >= srcLen +} + +var syscallMemcpy = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memcpy++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + if !isNonOverlapping(src, n, dst, n) { + return 0, errors.New("overlapping") + } + if n == 0 { + return 0, nil + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemmove = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memmove++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemcmp = sbpf.SyscallFunc4(func(vm sbpf.VM, a1, a2, n, res uint64) (uint64, error) { + st.memcmp++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + s1, err := vm.Translate(a1, n, false) + if err != nil { + return 0, err + } + s2, err := vm.Translate(a2, n, false) + if err != nil { + return 0, err + } + r := int32(0) + for i := uint64(0); i < n; i++ { + if s1[i] != s2[i] { + r = int32(s1[i]) - int32(s2[i]) + break + } + } + out, err := vm.Translate(res, 4, true) + if err != nil { + return 0, err + } + binary.LittleEndian.PutUint32(out, uint32(r)) + return 0, nil +}) + +var syscallMemset = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, c, n uint64) (uint64, error) { + st.memset++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + mem, err := vm.Translate(dst, n, true) + if err != nil { + return 0, err + } + for i := uint64(0); i < n; i++ { + mem[i] = byte(c) + } + return 0, nil +}) + +var syscallLog = sbpf.SyscallFunc2(func(vm sbpf.VM, ptr, strlen uint64) (uint64, error) { + st.logs++ + if err := vm.ComputeMeter().Consume(max(uint64(cuSyscallBase), strlen)); err != nil { + return 0, err + } + buf := make([]byte, strlen) + if err := vm.Read(ptr, buf); err != nil { + return 0, err + } + _ = string(buf) + return 0, nil +}) + +var syscallLog64 = sbpf.SyscallFunc5(func(vm sbpf.VM, a, b, c, d, e uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogPubkey = sbpf.SyscallFunc1(func(vm sbpf.VM, a uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogCUs = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallAbort = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { return 0, errors.New("abort") }) +var syscallPanic = sbpf.SyscallFunc4(func(vm sbpf.VM, f, l, line, col uint64) (uint64, error) { + return 0, errors.New("panic") +}) +var syscallAllocFree = sbpf.SyscallFunc2(func(vm sbpf.VM, size, free uint64) (uint64, error) { + if free != 0 { + return 0, nil + } + hs := (vm.HeapSize() + 7) &^ 7 + addr := sbpf.VaddrHeap + hs + hs += size + if hs > vm.HeapMax() { + return 0, nil + } + vm.UpdateHeapSize(hs) + return addr, nil +}) + +var registry = map[uint32]sbpf.Syscall{ + sbpf.SymbolHash("abort"): syscallAbort, + sbpf.SymbolHash("sol_panic_"): syscallPanic, + sbpf.SymbolHash("sol_log_"): syscallLog, + sbpf.SymbolHash("sol_log_64_"): syscallLog64, + sbpf.SymbolHash("sol_log_pubkey"): syscallLogPubkey, + sbpf.SymbolHash("sol_log_compute_units_"): syscallLogCUs, + sbpf.SymbolHash("sol_memcpy_"): syscallMemcpy, + sbpf.SymbolHash("sol_memmove_"): syscallMemmove, + sbpf.SymbolHash("sol_memcmp_"): syscallMemcmp, + sbpf.SymbolHash("sol_memset_"): syscallMemset, + sbpf.SymbolHash("sol_alloc_free_"): syscallAllocFree, +} + +var syscalls = sbpf.SyscallRegistry(func(h uint32) (sbpf.Syscall, bool) { + s, ok := registry[h] + return s, ok +}) + +// ---- aligned input serialization (BPF loader v2/v3 format, no direct mapping) ---- + +const maxPermittedDataIncrease = 10 * 1024 + +type acct struct { + key, owner [32]byte + lamports uint64 + data []byte + signer, writable bool +} + +func serializeAligned(accts []acct, instrData []byte, programId [32]byte) []byte { + out := binary.LittleEndian.AppendUint64(nil, uint64(len(accts))) + for _, a := range accts { + out = append(out, 0xff) + out = append(out, b2u8(a.signer), b2u8(a.writable), 0) + out = append(out, 0, 0, 0, 0) // original_data_len + out = append(out, a.key[:]...) + out = append(out, a.owner[:]...) + out = binary.LittleEndian.AppendUint64(out, a.lamports) + out = binary.LittleEndian.AppendUint64(out, uint64(len(a.data))) + out = append(out, a.data...) + pad := maxPermittedDataIncrease + ((8 - len(a.data)%8) % 8) + out = append(out, make([]byte, pad)...) + out = binary.LittleEndian.AppendUint64(out, ^uint64(0)) // rent epoch + } + out = binary.LittleEndian.AppendUint64(out, uint64(len(instrData))) + out = append(out, instrData...) + out = append(out, programId[:]...) + return out +} + +func b2u8(b bool) byte { + if b { + return 1 + } + return 0 +} + +// SPL token account layout (165 bytes) +func tokenAccount(mint, owner [32]byte, amount uint64) []byte { + d := make([]byte, 165) + copy(d[0:32], mint[:]) + copy(d[32:64], owner[:]) + binary.LittleEndian.PutUint64(d[64:72], amount) + // delegate: COption none (4 bytes 0) + 32 + d[108] = 1 // state = Initialized + // is_native COption none, delegated_amount 0, close_authority none + return d +} + +func key(b byte) [32]byte { + var k [32]byte + for i := range k { + k[i] = b + } + return k +} + +func loadTokenProgram(tb testing.TB) *sbpf.Program { + elfBytes := fixtures.Load(tb, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + tb.Fatal(err) + } + p, err := l.Load() + if err != nil { + tb.Fatal(err) + } + if err := p.Verify(); err != nil { + tb.Fatal(err) + } + return p +} + +func transferInput(programId [32]byte) ([]byte, []acct) { + mint := key(0x11) + authority := key(0x22) + src := key(0x33) + dst := key(0x44) + accts := []acct{ + {key: src, owner: programId, lamports: 2039280, data: tokenAccount(mint, authority, 1_000_000), writable: true}, + {key: dst, owner: programId, lamports: 2039280, data: tokenAccount(mint, key(0x55), 5), writable: true}, + {key: authority, owner: key(0), lamports: 1_000_000_000, data: nil, signer: true}, + } + instr := append([]byte{3}, binary.LittleEndian.AppendUint64(nil, 1000)...) + return serializeAligned(accts, instr, programId), accts +} + +func runTransfer(tb testing.TB, p *sbpf.Program, input []byte) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransfer(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, _ := transferInput(programId) + st = stats{} + ret, used := runTransfer(t, p, input) + t.Logf("ret=%d cuUsed=%d stats=%+v", ret, used, st) + // verify balances changed in the serialized input + // account 0 data starts at 8 + 8 + 32+32+8+8 = 96 + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + off1 := 8 + (8 + 32 + 32 + 8 + 8 + 165 + maxPermittedDataIncrease + 3 + 8) + dstAmt := binary.LittleEndian.Uint64(input[off1+88+64:]) + t.Logf("src=%d dst=%d", srcAmt, dstAmt) + if ret != 0 || srcAmt != 999_000 || dstAmt != 1005 { + t.Fatalf("unexpected result ret=%d src=%d dst=%d", ret, srcAmt, dstAmt) + } +} + +func BenchmarkTokenTransfer(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, _ := transferInput(programId) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + var used uint64 + for i := 0; i < b.N; i++ { + copy(input, orig) + _, used = runTransfer(b, p, input) + } + b.StopTimer() + b.ReportMetric(float64(used), "cu/op") + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/float64(used), "ns/cu") +} + +func BenchmarkTokenLoadVerify(b *testing.B) { + elfBytes := fixtures.Load(b, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + b.ReportAllocs() + for i := 0; i < b.N; i++ { + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + b.Fatal(err) + } + p, err := l.Load() + if err != nil { + b.Fatal(err) + } + if err := p.Verify(); err != nil { + b.Fatal(err) + } + } +} + +// ---- VASA layout (VirtualAddressSpaceAdjustments active, direct mapping off) ---- +// Mirrors serializeParametersAligned with vasa=true, directMapping=false: +// same bytes as the aligned layout, but the input window is split into +// metadata regions and per-account data regions. + +func vasaRegions(accts []acct, instrLen int) []sbpf.InputRegion { + var regions []sbpf.InputRegion + var regionStart, hostRegionStart uint64 + vmOff := uint64(8) + for i, a := range accts { + l := vmOff // host offset == vm offset in this layout + dataLen := uint64(len(a.data)) + align := (8 - dataLen%8) % 8 + reserved := dataLen + maxPermittedDataIncrease + dataStart := vmOff + 88 + if dataStart > regionStart { + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: dataStart - regionStart, AddressSpaceReserved: dataStart - regionStart, Writable: true, AccountIndex: -1}) + } + regions = append(regions, sbpf.InputRegion{Offset: dataStart, HostOffset: l + 88, RegionSize: dataLen, + AddressSpaceReserved: reserved, Writable: a.writable, AccountIndex: i}) + hostRegionStart = l + 88 + reserved + regionStart = dataStart + reserved + vmOff += 88 + reserved + align + 8 + } + end := vmOff + 8 + uint64(instrLen) + 32 + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: end - regionStart, AddressSpaceReserved: end - regionStart, Writable: true, AccountIndex: -1}) + return regions +} + +func runTransferVasa(tb testing.TB, p *sbpf.Program, input []byte, regions []sbpf.InputRegion) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + // regions are mutated by the VM (RegionSize on growth), so copy per run + rc := append([]sbpf.InputRegion(nil), regions...) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + InputRegions: rc, + DisableStackFrameGaps: true, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransferVasa(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + if regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize != uint64(len(input)) { + t.Fatalf("region layout mismatch: %d vs %d", regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize, len(input)) + } + ret, used := runTransferVasa(t, p, input, regions) + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + if ret != 0 || srcAmt != 999_000 { + t.Fatalf("unexpected ret=%d src=%d", ret, srcAmt) + } + t.Logf("ret=%d cu=%d regions=%d", ret, used, len(regions)) +} + +func BenchmarkTokenTransferVasa(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + copy(input, orig) + runTransferVasa(b, p, input, regions) + } +} diff --git a/pkg/sbpf/perf_bench_test.go b/pkg/sbpf/perf_bench_test.go new file mode 100644 index 000000000..1d34d4d7a --- /dev/null +++ b/pkg/sbpf/perf_bench_test.go @@ -0,0 +1,193 @@ +package sbpf + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +func slot(op uint8, dst uint8, src uint8, off int16, imm uint32) Slot { + return Slot(op) | Slot(dst)<<8 | Slot(src)<<12 | Slot(uint16(off))<<16 | Slot(imm)<<32 +} + +func slotsToBytes(slots []Slot) []byte { + out := make([]byte, len(slots)*SlotSize) + for i, s := range slots { + binary.LittleEndian.PutUint64(out[i*SlotSize:], uint64(s)) + } + return out +} + +func mkProgram(text []Slot, ver uint32) *Program { + return &Program{ + TextBytes: slotsToBytes(text), + Text: text, + TextVA: VaddrProgram, + Entrypoint: 0, + Funcs: map[uint32]int64{}, + SbpfVersion: sbpfver.SbpfVersion{Version: ver}, + } +} + +var noSyscalls = SyscallRegistry(func(uint32) (Syscall, bool) { return nil, false }) + +// resolveCallTargetsIfSupported precomputes internal call targets on trees +// that have Program.ResolveCallTargets (the loader does this at load time); +// it is a no-op on the baseline tree so the same benchmark code runs on both. +func resolveCallTargetsIfSupported(p *Program) { + if r, ok := any(p).(interface{ ResolveCallTargets() }); ok { + r.ResolveCallTargets() + } +} + +// aluLoop: r1 = N; loop: r2 += r1; r2 ^= r3; r3 = r2; r3 *= 7; r3 >>= 3; r1 -= 1; jne r1,0 loop; exit +// 7 instructions per iteration. +func aluLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Imm, 3, 0, 0, 3), + // loop @3 + slot(OpAdd64Reg, 2, 1, 0, 0), + slot(OpXor64Reg, 2, 3, 0, 0), + slot(OpMov64Reg, 3, 2, 0, 0), + slot(OpMul64Imm, 3, 0, 0, 7), + slot(OpRsh64Imm, 3, 0, 0, 3), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -7, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// memLoop: writes and reads 8 byte values on the stack frame and heap. +// r1 = N; r4 = r10 - 4096 (frame base); r5 = heap base +// loop: stxdw [r4+0], r1; ldxdw r6, [r4+0]; add r2, r6; stxdw [r5+8], r2; ldxdw r7,[r5+8]; xor r2,r7 ; r1 -= 1; jne +func memLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Reg, 4, 10, 0, 0), + slot(OpAdd64Imm, 4, 0, 0, uint32(0xfffff000)), // r4 = r10 - 4096 + slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), + slot(0, 0, 0, 0, uint32(VaddrHeap>>32)), + // loop @6 + slot(OpStxdw, 4, 1, 0, 0), + slot(OpLdxdw, 6, 4, 0, 0), + slot(OpAdd64Reg, 2, 6, 0, 0), + slot(OpStxdw, 5, 2, 8, 0), + slot(OpLdxdw, 7, 5, 8, 0), + slot(OpXor64Reg, 2, 7, 0, 0), + slot(OpStxw, 5, 2, 16, 0), + slot(OpLdxb, 8, 5, 16, 0), + slot(OpAdd64Reg, 2, 8, 0, 0), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -11, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// callLoop: calls a tiny function N times (tests Push/Pop + call resolution) +func callLoopProgram(n uint32, ver uint32) *Program { + fnPC := int64(7) + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + // loop @2 + slot(OpCall, 0, 0, 0, 0), // patched below + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -3, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + // fn @7 + slot(OpAdd64Imm, 2, 0, 0, 3), + slot(OpXor64Reg, 2, 1, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + p := mkProgram(text, ver) + if ver >= sbpfver.SbpfVersionV3 { + // relative call: target = pc + imm + 1 ; pc=2 -> imm = 7-2-1 = 4 + text[2] = slot(OpCall, 0, 1, 0, uint32(fnPC-2-1)) + } else { + h := PCHash(uint64(fnPC)) + p.Funcs[h] = fnPC + text[2] = slot(OpCall, 0, 0, 0, h) + } + p.TextBytes = slotsToBytes(text) + return p +} + +func runProgram(b *testing.B, p *Program, input []byte, syscalls SyscallRegistry, budget uint64) uint64 { + cm := cu.NewComputeMeter(budget) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, _, err := ip.Run() + ip.Finish() + if err != nil { + b.Fatal(err) + } + return ret +} + +func benchLoop(b *testing.B, p *Program, insnsPerRun uint64) { + if err := p.Verify(); err != nil { + b.Fatal(err) + } + resolveCallTargetsIfSupported(p) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1<<40) + } + b.StopTimer() + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(uint64(b.N)*insnsPerRun), "ns/insn") +} + +const loopN = 200_000 + +func BenchmarkAluLoopV0(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 0), 3+7*loopN+2) } +func BenchmarkAluLoopV3(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 3), 3+7*loopN+2) } +func BenchmarkMemLoopV0(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 0), 5+11*loopN+2) } +func BenchmarkMemLoopV3(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 3), 5+11*loopN+2) } +func BenchmarkCallLoopV0(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 0), 2+6*loopN+2) +} +func BenchmarkCallLoopV3(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 3), 2+6*loopN+2) +} + +// Interpreter setup/teardown cost only (tiny program). +func BenchmarkNewInterpreterAndExit(b *testing.B) { + p := mkProgram([]Slot{slot(OpExit, 0, 0, 0, 0)}, 0) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1000) + } +} + +func TestSyntheticProgramsRun(t *testing.T) { + for _, ver := range []uint32{0, 3} { + for _, p := range []*Program{aluLoopProgram(1000, ver), memLoopProgram(1000, ver), callLoopProgram(1000, ver)} { + if err := p.Verify(); err != nil { + t.Fatal(err) + } + cm := cu.NewComputeMeter(1 << 30) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32 * 1024, Syscalls: noSyscalls, ComputeMeter: &cm}) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + t.Fatal(err) + } + t.Logf("ver=%d ret=%d cu=%d", ver, ret, used) + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go new file mode 100644 index 000000000..1d679d4e4 --- /dev/null +++ b/pkg/sbpf/perf_differential_test.go @@ -0,0 +1,338 @@ +package sbpf + +import ( + "bufio" + "encoding/binary" + "fmt" + "hash/fnv" + "math/rand" + "os" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +// Differential test: generates deterministic pseudo-random programs and dumps +// (return value, error string, CU consumed, memory hash) per program to the +// file named by SBPF_DIFF_OUT. Running it against the baseline and the +// optimized interpreter and diffing the two files checks that observable +// behaviour is identical. + +type diffSyscall struct { + fn func(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) +} + +func (s diffSyscall) Invoke(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) { + return s.fn(vm, r1, r2, r3, r4, r5) +} + +var ( + hashPoke = SymbolHash("poke") // write r3 bytes of value r2 at r1 via vm.Write + hashPeek = SymbolHash("peek") // read 8 bytes at r1 -> r0 + hashCopy = SymbolHash("copy") // copy r3 bytes from r2 to r1 (Translate based) + hashBurn = SymbolHash("burn") // consume r1 CU + hashSetLen = SymbolHash("setlen") // SetInputRegionLength(r1, r2, r3!=0) +) + +func diffRegistry(h uint32) (Syscall, bool) { + switch h { + case hashPoke: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + buf := make([]byte, r3) + for i := range buf { + buf[i] = byte(r2 + uint64(i)) + } + return 0, vm.Write(r1, buf) + }}, true + case hashPeek: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + return vm.Read64(r1) + }}, true + case hashCopy: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + src, err := vm.Translate(r2, r3, false) + if err != nil { + return 0, err + } + dst, err := vm.Translate(r1, r3, true) + if err != nil { + return 0, err + } + copy(dst, src) + return 0, nil + }}, true + case hashBurn: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + return 0, vm.ComputeMeter().Consume(r1 & 0xff) + }}, true + case hashSetLen: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + ip := vm.(*Interpreter) + ok := ip.SetInputRegionLength(r1, r2, r3 != 0) + if ok { + return 1, nil + } + return 0, nil + }}, true + } + return nil, false +} + +func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { + reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 + imm := func() uint32 { + switch rng.Intn(4) { + case 0: + return uint32(rng.Intn(16)) + case 1: + return uint32(int32(-rng.Intn(16))) + case 2: + return rng.Uint32() + default: + return uint32(rng.Intn(4096)) + } + } + alu64 := []uint8{OpAdd64Imm, OpAdd64Reg, OpSub64Imm, OpSub64Reg, OpMul64Imm, OpMul64Reg, OpDiv64Imm, OpDiv64Reg, + OpOr64Imm, OpOr64Reg, OpAnd64Imm, OpAnd64Reg, OpLsh64Imm, OpLsh64Reg, OpRsh64Imm, OpRsh64Reg, OpMod64Imm, OpMod64Reg, + OpXor64Imm, OpXor64Reg, OpMov64Imm, OpMov64Reg, OpArsh64Imm, OpArsh64Reg, OpNeg64, + OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, + OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, + OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, + OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} + switch rng.Intn(10) { + case 0, 1, 2, 3: // alu + op := alu64[rng.Intn(len(alu64))] + i := imm() + if op == OpLe || op == OpBe { + i = []uint32{16, 32, 64}[rng.Intn(3)] + } + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + i = 3 + } + switch op { + case OpLsh32Imm, OpRsh32Imm, OpArsh32Imm: + i = uint32(rng.Intn(32)) + case OpLsh64Imm, OpRsh64Imm, OpArsh64Imm: + i = uint32(rng.Intn(64)) + } + return []Slot{slot(op, reg(), reg(), 0, i)} + case 4: // load + ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random + var base uint8 + var off int16 + switch rng.Intn(4) { + case 0: + base, off = 10, int16(-rng.Intn(4096)) + case 1: + base, off = 5, int16(rng.Intn(1024)) + case 2: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} + case 5: // store + ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + var base uint8 + var off int16 + switch rng.Intn(5) { + case 0, 1: + base, off = 10, int16(-rng.Intn(4096)) + case 2: + base, off = 5, int16(rng.Intn(1024)) + case 3: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(8)], base, reg(), off, imm())} + case 6: // forward conditional jump (never backwards: guarantees termination) + maxOff := n - pc - 2 + if maxOff <= 0 { + return []Slot{slot(OpMov64Imm, reg(), 0, 0, imm())} + } + return []Slot{slot(jmp[rng.Intn(len(jmp))], reg(), reg(), int16(rng.Intn(min(maxOff, 8))), imm())} + case 7: // syscall + hs := []uint32{hashPoke, hashPeek, hashCopy, hashBurn, hashSetLen} + return []Slot{slot(OpCall, 0, 0, 0, hs[rng.Intn(len(hs))])} + case 8: // internal call + if ver >= sbpfver.SbpfVersionV3 { + return []Slot{slot(OpCall, 0, 1, 0, uint32(fnPC-int64(pc)-1))} + } + return []Slot{slot(OpCall, 0, 0, 0, PCHash(uint64(fnPC)))} + default: // set up pointer registers + switch rng.Intn(3) { + case 0: // r5 = heap + return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + case 1: // r1 = input + small + return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + default: // r9 = random 64-bit + return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + } + } +} + +func genProgram(rng *rand.Rand, ver uint32) *Program { + n := 8 + rng.Intn(120) + // layout: [0, n) main body then exit; fn at fnPC: a few ALU ops + exit + body := make([]Slot, 0, n+16) + fnPC := int64(n + 1) + for len(body) < n { + body = append(body, randSlot(rng, len(body), n, ver, fnPC)...) + } + if len(body) > n { + body = body[:n-1] // drop a cut lddw pair + } + body = append(body, slot(OpExit, 0, 0, 0, 0)) + // fix up forward jumps that would land on the second slot of an lddw + for pc := range body { + if body[pc].Op()&0x07 == ClassJmp && body[pc].Op() != OpCall && body[pc].Op() != OpExit { + dst := pc + int(body[pc].Off()) + 1 + if dst < len(body) && body[dst].Op() == 0 { + body[pc] = body[pc]&^(Slot(0xffff)<<16) | Slot(uint16(body[pc].Off()+1))<<16 + } + } + } + // function + body = append(body, + slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), + slot(OpXor64Reg, 7, 6, 0, 0), + slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(body, ver) + if ver < sbpfver.SbpfVersionV3 { + p.Funcs[PCHash(uint64(fnPC))] = fnPC + } + p.RO = make([]byte, 256) + for i := range p.RO { + p.RO[i] = byte(i * 7) + } + return p +} + +func memHash(bs ...[]byte) uint64 { + h := fnv.New64a() + for _, b := range bs { + h.Write(b) + } + return h.Sum64() +} + +func TestDifferentialDump(t *testing.T) { + out := os.Getenv("SBPF_DIFF_OUT") + if out == "" { + t.Skip("SBPF_DIFF_OUT not set") + } + f, err := os.Create(out) + if err != nil { + t.Fatal(err) + } + defer f.Close() + w := bufio.NewWriter(f) + defer w.Flush() + + rng := rand.New(rand.NewSource(12345)) + const N = 100000 + generated, verified := 0, 0 + for i := 0; i < N; i++ { + ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + p := genProgram(rng, ver) + generated++ + if err := p.Verify(); err != nil { + fmt.Fprintf(w, "%d ver=%d VERIFY_FAIL %v\n", i, ver, err) + continue + } + verified++ + resolveCallTargetsIfSupported(p) + input := make([]byte, 700) + for j := range input { + input[j] = byte(j) + } + var regions []InputRegion + useRegions := rng.Intn(2) == 0 + if useRegions { + regions = []InputRegion{ + {Offset: 0, HostOffset: 0, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}, + {Offset: 100, HostOffset: 100, RegionSize: 150, AddressSpaceReserved: 300, Writable: rng.Intn(2) == 0, AccountIndex: 0}, + {Offset: 400, HostOffset: 400, RegionSize: 300, AddressSpaceReserved: 300, Writable: true, AccountIndex: -1}, + } + } + budget := uint64(1 + rng.Intn(400)) + if rng.Intn(4) == 0 { + budget = 100000 + } + cm := cu.NewComputeMeter(budget) + heapMax := 4096 * (1 + rng.Intn(4)) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: heapMax, + Syscalls: diffRegistry, + ComputeMeter: &cm, + Input: input, + InputRegions: regions, + DisableStackFrameGaps: rng.Intn(3) == 0, + }) + var ret, cuUsed uint64 + var runErr error + func() { + defer func() { + if r := recover(); r != nil { + runErr = fmt.Errorf("PANIC: %v", r) + } + }() + ret, cuUsed, runErr = ip.Run() + }() + errStr := "" + if runErr != nil { + errStr = runErr.Error() + } + h := memHash(ip.stack.mem, ip.heap, input) + regionSizes := "" + for _, r := range ip.inputRegions { + regionSizes += fmt.Sprintf("%d/%v,", r.RegionSize, r.Writable) + } + fmt.Fprintf(w, "%d ver=%d budget=%d ret=%d cu=%d remaining=%d err=%q mem=%x regions=%s\n", + i, ver, budget, ret, cuUsed, cm.Remaining(), errStr, h, regionSizes) + ip.Finish() + if os.Getenv("SBPF_CHECK_POOL_ZERO") != "" { + // The buffers just returned to the pool must be all-zero. + st := stackMemPool.Get().([]byte) + hp := heapPool.Get().([]byte) + for j, b := range st[:StackMax] { + if b != 0 { + t.Fatalf("program %d: pooled stack not zeroed at %d", i, j) + } + } + for j, b := range hp[:cap(hp)] { + if b != 0 { + t.Fatalf("program %d: pooled heap not zeroed at %d", i, j) + } + } + stackMemPool.Put(st) + heapPool.Put(hp) + } + } + t.Logf("generated=%d verified=%d", generated, verified) +} + +var _ = binary.LittleEndian From be48dcc12b250941480746ed2033661e389c94fa Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 027/111] sealevel: add native program micro-benchmarks (System transfer, Vote TowerSync) Measured through ExecutionCtx.ProcessInstruction so instruction-context push/pop, lamport-sum checks and timing metrics are included; each has a NoTiming variant (SkipTimingMetrics) to quantify instrumentation cost, plus a vote-state (de)serialization round trip. NOTE: written without a local build of pkg/sealevel (sandbox cannot fetch its dependencies); expect to fix compile errors on first run. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/native_perf_bench_test.go | 287 +++++++++++++++++++++++++ 1 file changed, 287 insertions(+) create mode 100644 pkg/sealevel/native_perf_bench_test.go diff --git a/pkg/sealevel/native_perf_bench_test.go b/pkg/sealevel/native_perf_bench_test.go new file mode 100644 index 000000000..447931695 --- /dev/null +++ b/pkg/sealevel/native_perf_bench_test.go @@ -0,0 +1,287 @@ +package sealevel + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" +) + +// Micro-benchmarks for the natively implemented programs that are still on the +// mainnet hot path (System transfer, Vote TowerSync), measured through the same +// ExecutionCtx.ProcessInstruction entry point replay uses, so the per-instruction +// plumbing (instruction context push/pop, lamport-sum checks, timing metrics) +// is included. Run with: +// +// go test ./pkg/sealevel/ -run XXX -bench 'Native|VoteState|Timing' -benchmem -cpu 1 -count 5 + +func benchPubkey(b byte) solana.PublicKey { + var pk solana.PublicKey + for i := range pk { + pk[i] = b + } + return pk +} + +// newBenchExecCtx mirrors newSystemProgramTestExecCtx without testing.T. +func newBenchExecCtx(txAccts *TransactionAccounts, clockSlot uint64, enabled ...features.FeatureGate) *ExecutionCtx { + txCtx := NewTransactionCtx(*txAccts, 5, 64) + execCtx := &ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(1 << 62)} + execCtx.Accounts = accounts.NewMemAccounts() + + clockAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct); err != nil { + panic(err) + } + WriteClockSysvar(&execCtx.Accounts, SysvarClock{Slot: clockSlot, Epoch: 0}) + + rentAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct); err != nil { + panic(err) + } + WriteRentSysvar(&execCtx.Accounts, SysvarRent{LamportsPerUint8Year: 3480, ExemptionThreshold: 2, BurnPercent: 50}) + + f := features.NewFeaturesDefault() + for _, gate := range enabled { + f.EnableFeature(gate, 0) + } + execCtx.Features = *f + return execCtx +} + +// resetTxCtx gives the execution context a fresh instruction trace/stack for +// the next instruction (each ProcessInstruction consumes one trace slot). +func resetTxCtx(execCtx *ExecutionCtx, txAccts *TransactionAccounts) { + execCtx.TransactionContext = NewTransactionCtx(*txAccts, 5, 64) +} + +// ---------------------------------------------------------------- System + +func encodeSystemTransfer(lamports uint64) []byte { + out := binary.LittleEndian.AppendUint32(nil, uint32(SystemProgramInstrTypeTransfer)) + return binary.LittleEndian.AppendUint64(out, lamports) +} + +func benchmarkSystemTransfer(b *testing.B, skipTiming bool) { + systemProgramAcct := accounts.Account{Key: a.SystemProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + from := accounts.Account{Key: benchPubkey(0x11), Lamports: 1 << 60, Data: []byte{}, Owner: a.SystemProgramAddr} + to := accounts.Account{Key: benchPubkey(0x22), Lamports: 1_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{systemProgramAcct, from, to}) + metas := []AccountMeta{ + {Pubkey: from.Key, IsSigner: true, IsWritable: true}, + {Pubkey: to.Key, IsSigner: false, IsWritable: true}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + instr := encodeSystemTransfer(1) + + execCtx := newBenchExecCtx(txAccts, 1234) + execCtx.SkipTimingMetrics = skipTiming + + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + resetTxCtx(execCtx, txAccts) + if err := execCtx.ProcessInstruction(instr, instrAccts, []uint64{0}); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + if got := txAccts.Accounts[2].Lamports; got != 1_000_000+uint64(b.N) { + b.Fatalf("destination lamports %d, want %d", got, 1_000_000+uint64(b.N)) + } +} + +func BenchmarkNativeSystemTransfer(b *testing.B) { benchmarkSystemTransfer(b, false) } +func BenchmarkNativeSystemTransferNoTiming(b *testing.B) { benchmarkSystemTransfer(b, true) } + +// ---------------------------------------------------------------- Vote + +const ( + benchVoteRoot = uint64(1000) + benchVoteLastSlot = benchVoteRoot + MaxLockoutHistory // 1031 + benchClockSlot = benchVoteLastSlot + 2 +) + +func benchSlotHash(slot uint64) [32]byte { + var h [32]byte + binary.LittleEndian.PutUint64(h[:], slot*0x9e3779b97f4a7c15) + binary.LittleEndian.PutUint64(h[8:], ^slot) + h[31] = 0xa5 + return h +} + +// benchSlotHashes builds a 512-entry SlotHashes sysvar (newest first) that +// covers every slot the benchmark tower refers to. +func benchSlotHashes() SysvarSlotHashes { + const n = 512 + newest := benchVoteLastSlot + 1 + sh := make(SysvarSlotHashes, 0, n) + for i := uint64(0); i < n; i++ { + s := newest - i + sh = append(sh, SlotHash{Slot: s, Hash: benchSlotHash(s)}) + } + return sh +} + +// benchInitialVoteState is a fully populated current-version vote state with a +// full 31-entry tower ending at benchVoteLastSlot. +func benchInitialVoteState(voter solana.PublicKey) *VoteState { + vs := &VoteState{ + NodePubkey: voter, + AuthorizedWithdrawer: voter, + Commission: 10, + PriorVoters: PriorVoters{Index: 31, IsEmpty: true}, + EpochCredits: []EpochCredits{{Epoch: 0, Credits: 1000, PrevCredits: 0}}, + LastTimestamp: BlockTimestamp{Slot: benchVoteLastSlot, Timestamp: 1_700_000_000}, + } + vs.AuthorizedVoters.AuthorizedVoters.Set(0, voter) + root := benchVoteRoot + vs.RootSlot = &root + for i := uint64(0); i < MaxLockoutHistory; i++ { + vs.Votes.PushBack(LandedVote{ + Latency: 1, + Lockout: VoteLockout{Slot: benchVoteRoot + 1 + i, ConfirmationCount: uint32(MaxLockoutHistory - i)}, + }) + } + return vs +} + +func benchSerializedVoteState(vs *VoteState) []byte { + versioned := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + data := make([]byte, VoteStateV3Size) + if err := WriteVersionedVoteStateInPlace(data, versioned); err != nil { + panic(err) + } + return data +} + +// encodeTowerSync encodes a TowerSync that advances the tower by one slot: +// root = old root + 1, lockouts = old lockouts shifted by one slot plus the +// new slot, i.e. exactly what a validator sends every slot. +func encodeTowerSync() []byte { + root := benchVoteRoot + 1 + out := binary.LittleEndian.AppendUint32(nil, uint32(VoteProgramInstrTypeTowerSync)) + out = binary.LittleEndian.AppendUint64(out, root) + out = append(out, byte(MaxLockoutHistory)) // compact-u16, < 0x80 + prev := root + for i := uint64(0); i < MaxLockoutHistory; i++ { + slot := root + 1 + i + out = binary.AppendUvarint(out, slot-prev) + out = append(out, byte(MaxLockoutHistory-i)) + prev = slot + } + last := root + MaxLockoutHistory // benchVoteLastSlot + 1 + h := benchSlotHash(last) + out = append(out, h[:]...) + out = append(out, 1) // Some(timestamp) + out = binary.LittleEndian.AppendUint64(out, uint64(1_700_000_001)) + var blockID [32]byte + out = append(out, blockID[:]...) + return out +} + +type voteBench struct { + execCtx *ExecutionCtx + txAccts *TransactionAccounts + instr []byte + instrAccts []InstructionAccount + initial []byte +} + +func newVoteBench(skipTiming bool) *voteBench { + voter := benchPubkey(0x33) + votePk := benchPubkey(0x44) + initial := benchSerializedVoteState(benchInitialVoteState(voter)) + + voteProgramAcct := accounts.Account{Key: a.VoteProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + voteAcct := accounts.Account{Key: votePk, Lamports: 1_000_000_000, Data: append([]byte(nil), initial...), Owner: a.VoteProgramAddr} + voterAcct := accounts.Account{Key: voter, Lamports: 1_000_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{voteProgramAcct, voteAcct, voterAcct}) + metas := []AccountMeta{ + {Pubkey: votePk, IsSigner: false, IsWritable: true}, + {Pubkey: voter, IsSigner: true, IsWritable: false}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + + execCtx := newBenchExecCtx(txAccts, benchClockSlot, + features.EnableTowerSyncIx, + features.VoteStateAddVoteLatency, + features.TimelyVoteCredits, + features.DeprecateUnusedLegacyVotePlumbing, + ) + execCtx.SkipTimingMetrics = skipTiming + shAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarSlotHashesAddr, &shAcct); err != nil { + panic(err) + } + WriteSlotHashesSysvar(&execCtx.Accounts, benchSlotHashes()) + + return &voteBench{execCtx: execCtx, txAccts: txAccts, instr: encodeTowerSync(), instrAccts: instrAccts, initial: initial} +} + +// step runs one TowerSync against the initial vote state (the account data is +// rewound to the initial state first so every iteration performs identical work). +func (vb *voteBench) step() error { + copy(vb.txAccts.Accounts[1].Data, vb.initial) + resetTxCtx(vb.execCtx, vb.txAccts) + return vb.execCtx.ProcessInstruction(vb.instr, vb.instrAccts, []uint64{0}) +} + +func TestNativeVoteTowerSyncBenchSetup(t *testing.T) { + vb := newVoteBench(false) + if err := vb.step(); err != nil { + t.Fatalf("TowerSync failed: %v", err) + } + versioned, err := UnmarshalVersionedVoteState(vb.txAccts.Accounts[1].Data) + if err != nil { + t.Fatal(err) + } + vs := versioned.ConvertToCurrent() + if vs.Votes.Len() != MaxLockoutHistory { + t.Fatalf("tower length %d, want %d", vs.Votes.Len(), MaxLockoutHistory) + } + if last := vs.Votes.Back().Lockout.Slot; last != benchVoteLastSlot+1 { + t.Fatalf("last voted slot %d, want %d", last, benchVoteLastSlot+1) + } + if vs.RootSlot == nil || *vs.RootSlot != benchVoteRoot+1 { + t.Fatalf("root %v, want %d", vs.RootSlot, benchVoteRoot+1) + } +} + +func benchmarkVoteTowerSync(b *testing.B, skipTiming bool) { + vb := newVoteBench(skipTiming) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := vb.step(); err != nil { + b.Fatal(err) + } + } +} + +func BenchmarkNativeVoteTowerSync(b *testing.B) { benchmarkVoteTowerSync(b, false) } +func BenchmarkNativeVoteTowerSyncNoTiming(b *testing.B) { benchmarkVoteTowerSync(b, true) } + +// BenchmarkVoteStateRoundTrip isolates vote-state (de)serialization: decode +// the account, convert to current, re-encode — the fixed cost of every vote. +func BenchmarkVoteStateRoundTrip(b *testing.B) { + data := benchSerializedVoteState(benchInitialVoteState(benchPubkey(0x33))) + out := make([]byte, VoteStateV3Size) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + versioned, err := UnmarshalVersionedVoteState(data) + if err != nil { + b.Fatal(err) + } + vs := versioned.ConvertToCurrent() + cur := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + if err := WriteVersionedVoteStateInPlace(out, cur); err != nil { + b.Fatal(err) + } + } +} From 9589709a304b7a96b08b6010a42948b231c24fc1 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 028/111] test: update VASA stack registers for fixed-size interpreter state --- pkg/sbpf/vasa_test.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/pkg/sbpf/vasa_test.go b/pkg/sbpf/vasa_test.go index e7d64034c..70296ea43 100644 --- a/pkg/sbpf/vasa_test.go +++ b/pkg/sbpf/vasa_test.go @@ -73,7 +73,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { gapped := NewStack(version, false) defer gapped.Finish() - gappedRegs := make([]uint64, 11) + gappedRegs := new([16]uint64) gappedRegs[10] = VaddrStack + StackFrameSize require.True(t, gapped.Push(gappedRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*3, gappedRegs[10]) @@ -81,7 +81,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { contiguous := NewStack(version, true) defer contiguous.Finish() - contiguousRegs := make([]uint64, 11) + contiguousRegs := new([16]uint64) contiguousRegs[10] = VaddrStack + StackFrameSize require.True(t, contiguous.Push(contiguousRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, contiguousRegs[10]) @@ -94,7 +94,7 @@ func TestStackFrameGapsAreLegacyOnly(t *testing.T) { stack := NewStack(version, false) defer stack.Finish() - regs := make([]uint64, 11) + regs := new([16]uint64) regs[10] = VaddrStack + StackFrameSize require.True(t, stack.Push(regs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, regs[10]) From ee5234bf040fc58406f9cbbbb43ecb911ee1cc9e Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 029/111] test: benchmark verified token arithmetic and CPI workloads with warm programs --- pkg/sealevel/program_workloads_bench_test.go | 195 +++++++++++++++++++ 1 file changed, 195 insertions(+) create mode 100644 pkg/sealevel/program_workloads_bench_test.go diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go new file mode 100644 index 000000000..cc6b83e78 --- /dev/null +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -0,0 +1,195 @@ +package sealevel + +import ( + "encoding/binary" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" +) + +// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// No live account writes or network calls occur in this harness. Each invocation +// gets fresh account data; the program cache is warm and shared between runs. +type programWorkload struct { + name string + elf []byte + program solana.PublicKey + accts []accounts.Account + metas []AccountMeta + instruction []byte + check func(testing.TB, *ExecutionCtx) +} + +func expandedProgramWorkloads(t testing.TB) []programWorkload { + t.Helper() + program := benchPubkey(0x91) + from := accounts.Account{Key: benchPubkey(0x31), Owner: program, Lamports: 10000} + to := accounts.Account{Key: benchPubkey(0x32), Owner: program, Lamports: 10000} + cases := []programWorkload{{name: "BPF_LamportTransfer", elf: fixtures.Load(t, "sbpf", "cpi_c_to_bpf.so"), program: program, accts: []accounts.Account{from, to}, metas: []AccountMeta{{Pubkey: program}, {Pubkey: from.Key, IsSigner: true, IsWritable: true}, {Pubkey: to.Key, IsSigner: true, IsWritable: true}}, instruction: []byte{0}, check: func(t testing.TB, ctx *ExecutionCtx) { + src, err := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, err) + dst, err := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, err) + require.Equal(t, uint64(9000), src.Lamports) + require.Equal(t, uint64(11000), dst.Lamports) + }}} + + pda, bump, err := solana.FindProgramAddress([][]byte{[]byte("You pass butter")}, program) + require.NoError(t, err) + cases = append(cases, programWorkload{name: "CPI_Rust_SystemAllocate", elf: fixtures.Load(t, "sbpf", "cpi_rust_to_system_program_allocate.so"), program: program, + accts: []accounts.Account{{Key: a.SystemProgramAddr, Owner: a.NativeLoaderAddr, Executable: true, Lamports: 10000}, {Key: pda, Owner: a.SystemProgramAddr, Lamports: 10000}}, + metas: []AccountMeta{{Pubkey: a.SystemProgramAddr}, {Pubkey: pda, IsSigner: true, IsWritable: true}}, instruction: []byte{bump}, check: func(t testing.TB, ctx *ExecutionCtx) { + acct, e := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, e) + require.Len(t, acct.Data, 1337) + require.NotEmpty(t, ctx.InnerInstrs) + }}) + dir := os.Getenv("MITHRIL_PROGRAM_BENCH_DIR") + if dir == "" { + return cases + } + arithmetic, err := os.ReadFile(filepath.Join(dir, "rotation_compute.so")) + require.NoError(t, err) + for _, iterations := range []uint32{500, 5000} { + n := iterations + data := append([]byte("RC01"), 0, 0, 0, 0) + binary.LittleEndian.PutUint32(data[4:], n) + name := "Arithmetic_500" + if n == 5000 { + name = "Arithmetic_5000" + } + cases = append(cases, programWorkload{name: name, elf: arithmetic, program: program, instruction: data, check: func(t testing.TB, ctx *ExecutionCtx) { + x := uint64(0x9e3779b97f4a7c15) + for i := uint32(0); i < n; i++ { + x = ((x << 7) | (x >> 57)) ^ (uint64(i) + 0x517cc1b727220a95) + } + _, got := ctx.TransactionContext.ReturnData() + require.Len(t, got, 8) + require.Equal(t, x, binary.LittleEndian.Uint64(got)) + }}) + } + token, err := os.ReadFile(filepath.Join(dir, "token2022.so")) + require.NoError(t, err) + mint, auth := benchPubkey(0x51), benchPubkey(0x52) + tokenData := func(amount uint64) []byte { + d := make([]byte, 165) + copy(d, mint[:]) + copy(d[32:], auth[:]) + binary.LittleEndian.PutUint64(d[64:], amount) + d[108] = 1 + return d + } + mintData := make([]byte, 82) + mintData[44] = 6 + mintData[45] = 1 + src := accounts.Account{Key: benchPubkey(0x53), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(1000000)} + dst := accounts.Account{Key: benchPubkey(0x54), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(5)} + instr := append([]byte{12}, binary.LittleEndian.AppendUint64(nil, 1000)...) + instr = append(instr, 6) + cases = append(cases, programWorkload{name: "Token2022_TransferChecked", elf: token, program: solana.Token2022ProgramID, + accts: []accounts.Account{src, {Key: mint, Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: mintData}, dst, {Key: auth, Owner: a.SystemProgramAddr, Lamports: 10000000}}, + metas: []AccountMeta{{Pubkey: src.Key, IsWritable: true}, {Pubkey: mint}, {Pubkey: dst.Key, IsWritable: true}, {Pubkey: auth, IsSigner: true}}, instruction: instr, + check: func(t testing.TB, ctx *ExecutionCtx) { + s, e := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, e) + d, e := ctx.TransactionContext.Accounts.GetAccount(3) + require.NoError(t, e) + require.Equal(t, uint64(999000), binary.LittleEndian.Uint64(s.Data[64:])) + require.Equal(t, uint64(1005), binary.LittleEndian.Uint64(d.Data[64:])) + }}) + return cases +} +func workloadRunner(t testing.TB, w programWorkload, vasa bool) func() (*ExecutionCtx, error) { + t.Helper() + f := features.NewFeaturesDefault() + if vasa { + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + } + l, err := loader.NewLoaderWithSyscalls(w.elf, func(h uint32) (sbpf.Syscall, bool) { return Syscalls(f, false, h) }, false, f) + require.NoError(t, err) + prog, err := l.Load() + require.NoError(t, err) + require.NoError(t, prog.Verify()) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + db := &accountsdb.AccountsDb{ProgramCache: cache} + db.AddProgramToCache(w.program, &accountsdb.ProgramCacheEntry{Program: prog}) + _, cached := db.MaybeGetProgramFromCache(w.program) + require.True(t, cached, "warm program must be cached") + return func() (*ExecutionCtx, error) { + list := make([]accounts.Account, len(w.accts)+1) + list[0] = accounts.Account{Key: w.program, Owner: a.BpfLoader2Addr, Lamports: 10000000, Executable: true, Data: w.elf} + for i, acct := range w.accts { + list[i+1] = acct + list[i+1].Data = append([]byte(nil), acct.Data...) + } + tx := NewTransactionAccounts(list) + ctx := newBenchExecCtx(tx, 1337) + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} + ctx.Features = *f + ctx.ComputeMeter = cu.NewComputeMeter(1400000) + ctx.SlotCtx = &SlotCtx{Slot: 1337, AccountsDb: db} + ctx.Log = &LogRecorder{} + ctx.RecordInnerInstructions = true + err := ctx.ProcessInstruction(w.instruction, InstructionAcctsFromAccountMetas(w.metas, *tx), []uint64{0}) + return ctx, err + } +} +func TestProgramWorkloadResults(t *testing.T) { + for _, w := range expandedProgramWorkloads(t) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + t.Run(name, func(t *testing.T) { + run := workloadRunner(t, w, vasa) + ctx, err := run() + require.NoError(t, err) + w.check(t, ctx) + t.Logf("cu=%d inner=%d", ctx.ComputeMeter.Used(), len(ctx.InnerInstrs)) + }) + } + } +} +func BenchmarkProgramWorkloads(b *testing.B) { + for _, w := range expandedProgramWorkloads(b) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + b.Run(name, func(b *testing.B) { + run := workloadRunner(b, w, vasa) + ctx, err := run() + require.NoError(b, err) + w.check(b, ctx) + used := ctx.ComputeMeter.Used() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + ctx, err = run() + if err != nil { + b.Fatal(err) + } + } + b.StopTimer() + w.check(b, ctx) + b.ReportMetric(float64(used), "cu/op") + }) + } + } +} From 1ca61123e717f912b96bee5bba98c255b4545dd2 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:21:44 -0500 Subject: [PATCH 030/111] docs: record individual-program and matched Alpenglow replay measurements --- docs/sbpf-interpreter-benchmarks.md | 103 ++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 docs/sbpf-interpreter-benchmarks.md diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md new file mode 100644 index 000000000..73eb970e5 --- /dev/null +++ b/docs/sbpf-interpreter-benchmarks.md @@ -0,0 +1,103 @@ +# Interpreter performance validation + +This experiment compares `9db7dcb` plus identical benchmark code with the same +base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift +semantics patch is excluded. Three VASA tests were updated to pass the new +`*[16]uint64` register type; no further production optimization was added during +this measurement pass. + +## Individual programs + +Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five +one-second samples per variant with alternating run order. The existing validator +continued running. CPU 15 shares a physical core with CPU 7; these are shared-host +measurements rather than an isolated-machine throughput ceiling. + +The harness has a preloaded program cache and asserts a cache hit before timing. +Every invocation gets fresh account data and execution context. It measures +instruction setup, serialization, execution and publication together; it is not +just the interpreter loop. Timing instrumentation and instruction recording are +enabled identically on both versions. Tests check arithmetic return values, +post-transfer balances, allocation results, and recorded CPI. Requested budgets +are equal and the measured CU charges match between variants. + +| Workload | CU | Median before → after | Speedup | +|---|---:|---:|---:| +| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | +| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | +| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | +| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | +| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | +| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | +| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | + +The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. +The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from +the earlier SPL Token loader-only benchmark: they use different program binaries, +instructions, and include execution-context setup. + +Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` +and `token2022.so` to enable those external fixtures. Without it, the in-repository +BPF/CPI fixtures still run. Pinned input SHA-256 values: + +- Arithmetic ELF: `db7c55d6563c879e35dfe2b24edb0fe0515d5a5ae627fe00e3e247001c786441`. + Source: `ag-transaction-bench` at `7e5a263fa5a1c72088f191daf5b7c5d2484c997c`, + `transaction-bench/program/src/rotation_compute.c`. +- Token-2022 ELF: `a794161408080f690dac00832f45b3c3e2b71f1339586667ad1f979cf91d5b68`. + Public Alpenglow program `TokenzQdBNbLqP5VEhdkAS6EPFLC1PHnBqCXEpPxuEb`, + fetched at RPC context slot 4,231,444, program-data account + `DoU57AYuPFu2QU514RktNPG22QhApEjnKxnBcu4BHDTY`. Strip its 45-byte upgradeable + loader metadata before saving the ELF. Verify the hash; do not silently replace + it with a later deployment. + +``` +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^TestProgramWorkloadResults$' -v +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^$' -bench '^BenchmarkProgramWorkloads$' \ + -benchtime=1s -count=5 -benchmem +``` + +## Recorded Alpenglow blocks + +Each run bootstrapped a fresh isolated AccountsDB from the same public full +snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay +covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and +142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, +nice 19. Two paired runs used baseline/candidate then candidate/baseline order. +The live validator's AccountsDB and configuration were not used or changed. + +All 233 per-slot bank hashes matched between baseline and candidate in both +pairs, excluding the run-specific comment header in `bankhash.log`. + +| Work measured | Pair 1 before → after | Pair 2 before → after | +|---|---:|---:| +| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | +| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | +| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | +| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | +| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | + +The sample includes blocks around 40–46 million CU. Three inspected non-empty +blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and +Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries +attribute whole-transaction CU to every participating program and must not be +summed as if they were exclusive per-program execution costs. + +Whole-block p95 did not improve, and the p99 changes are small/variable. The +slowest candidate blocks in the first pair contained no sBPF execution; their +large timers were dispatch and signature verification. These single-CPU replay +results do not establish a live voting/FAST improvement or production parallel +replay latency. They exclude network wait from ProcessBlock and are not elapsed +end-to-end catch-up times. + +## Correctness and limits + +- Native Zen 5 baseline and candidate differential outputs match for 100,000 + deterministic generated programs; candidate pool-zero checks pass. +- Baseline/candidate workload effects and CU charges match. Targeted race tests + for the interpreter, loader, and workload harness pass; vet passes. +- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; + therefore no complete sealevel test-suite pass is claimed. Broader conformance + testing remains separate from this performance experiment. +- No candidate was deployed and no validator restart was needed. From cbef9646492827ceb7b714cad341c880505186f3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:39:11 +0000 Subject: [PATCH 031/111] sealevel: copy, compare and fill VM memory without temporaries or byte loops sol_memcpy_/sol_memmove_ read the source into a fresh heap buffer and wrote it back; the copy now goes directly between the two translated slices with Go's memmove-semantics copy (overlap handled, source translated first so error precedence is unchanged, and a copy-on-write/growth of the destination region still reads the pre-write bytes because the source slice keeps the previous backing buffer alive). sol_memcmp_ uses bytes.Equal for the common equal case and word-skips to the first differing byte otherwise; sol_memset_ uses clear for zero and a doubling copy for other values. An SPL Token transfer issues two memcpy and four memcmp calls, so this is a small, allocation-free win rather than a large one. Tests: memcmpResult against the previous byte loop on 100k random inputs, memsetBytes over sizes and values, and VM-level memmove/memcpy overlap, error-ordering, copy-on-write-region and memcmp/memset checks. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/syscalls_mem.go | 77 +++++++++--- pkg/sealevel/syscalls_mem_test.go | 199 ++++++++++++++++++++++++++++++ 2 files changed, 256 insertions(+), 20 deletions(-) create mode 100644 pkg/sealevel/syscalls_mem_test.go diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 7bea5331e..29834a9ce 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -1,6 +1,7 @@ package sealevel import ( + "bytes" "encoding/binary" //"github.com/Overclock-Validator/mithril/pkg/mlog" @@ -15,14 +16,24 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { return execCtx.ComputeMeter.Consume(cost) } -func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { - srcBuf := make([]byte, n) - err = vm.Read(src, srcBuf) +// memmoveImplInternal copies n bytes from src to dst inside the VM without a +// temporary buffer. The source is translated first so a bad source address is +// reported before a bad destination, as before. Go's copy has memmove +// semantics, so overlapping ranges within one region are handled; and when +// the destination translation grows or copy-on-writes an account region, the +// source slice still refers to the previous backing buffer, whose bytes are +// exactly what the old read-then-write sequence would have copied. +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + srcMem, err := vm.Translate(src, n, false) if err != nil { - return + return err + } + dstMem, err := vm.Translate(dst, n, true) + if err != nil { + return err } - err = vm.Write(dst, srcBuf) - return + copy(dstMem, srcMem) + return nil } // SyscallMemcpyImpl is the implementation of the memcpy (sol_memcpy_) syscall. @@ -76,6 +87,26 @@ func SyscallMemmoveImpl(vm sbpf.VM, dst, src, n uint64) (uint64, error) { var SyscallMemmove = sbpf.SyscallFunc3(SyscallMemmoveImpl) +// memcmpResult returns the C memcmp result of two equal-length slices: zero +// when they are equal, otherwise the difference of the first differing bytes +// as unsigned values, matching Agave's `(b1 as i32) - (b2 as i32)`. +func memcmpResult(a, b []byte) int32 { + if bytes.Equal(a, b) { + return 0 + } + // The slices differ: skip equal 8-byte words, then locate the byte. + i := 0 + for i+8 <= len(a) && binary.LittleEndian.Uint64(a[i:]) == binary.LittleEndian.Uint64(b[i:]) { + i += 8 + } + for ; i < len(a); i++ { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + // SyscallMemcmpImpl is the implementation for the memcmp (sol_memcmp_) syscall. func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemcmp") @@ -96,15 +127,7 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, return syscallErr(err) } - cmpResult := int32(0) - for count := uint64(0); count < n; count++ { - b1 := slice1[count] - b2 := slice2[count] - if b1 != b2 { - cmpResult = int32(b1) - int32(b2) - break - } - } + cmpResult := memcmpResult(slice1, slice2) resultSlice, err := vm.Translate(resultAddr, 4, true) if err != nil { @@ -118,7 +141,23 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, var SyscallMemcmp = sbpf.SyscallFunc4(SyscallMemcmpImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// memsetBytes fills mem with c using the runtime's block clear for zero and a +// doubling copy otherwise, instead of a byte-at-a-time loop. +func memsetBytes(mem []byte, c byte) { + if len(mem) == 0 { + return + } + if c == 0 { + clear(mem) + return + } + mem[0] = c + for filled := 1; filled < len(mem); filled *= 2 { + copy(mem[filled:], mem[:filled]) + } +} + +// SyscallMemsetImpl is the implementation for the memset (sol_memset_) syscall. func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemset") @@ -133,16 +172,14 @@ func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { return syscallErr(err) } - for i := uint64(0); i < n; i++ { - mem[i] = byte(c) - } + memsetBytes(mem, byte(c)) return syscallSuccess(0) } var SyscallMemset = sbpf.SyscallFunc3(SyscallMemsetImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// SyscallAllocFreeImpl is the implementation for the alloc/free (sol_alloc_free_) syscall. func SyscallAllocFreeImpl(vm sbpf.VM, size, freeAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallAllocFreeImpl") diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go new file mode 100644 index 000000000..3b02daea0 --- /dev/null +++ b/pkg/sealevel/syscalls_mem_test.go @@ -0,0 +1,199 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + "math/rand" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + feat "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func newMemSyscallVM(t *testing.T, input []byte, regions []sbpf.InputRegion) (*sbpf.Interpreter, *ExecutionCtx) { + t.Helper() + features := feat.NewFeaturesDefault() + execCtx := &ExecutionCtx{Features: *features, ComputeMeter: cu.NewComputeMeter(1_000_000)} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram, Funcs: map[uint32]int64{}}, &sbpf.VMOpts{ + Input: input, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, + InputRegions: regions, + }) + t.Cleanup(vm.Finish) + return vm, execCtx +} + +func TestSyscallMemmoveOverlapping(t *testing.T) { + for _, test := range []struct { + name string + dst, src, n uint64 + expectMemcpyE bool + }{ + {name: "forward-overlap", dst: 4, src: 0, n: 16, expectMemcpyE: true}, + {name: "backward-overlap", dst: 0, src: 4, n: 16, expectMemcpyE: true}, + {name: "disjoint", dst: 40, src: 0, n: 16}, + {name: "adjacent", dst: 16, src: 0, n: 16}, + {name: "empty", dst: 0, src: 0, n: 0}, + } { + t.Run(test.name, func(t *testing.T) { + input := make([]byte, 64) + for i := range input { + input[i] = byte(i + 1) + } + want := append([]byte(nil), input...) + copy(want[test.dst:test.dst+test.n], want[test.src:test.src+test.n]) + + vm, _ := newMemSyscallVM(t, input, nil) + ret, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, want, input, "memmove must have Go copy (memmove) semantics") + + // memcpy: the same result for disjoint ranges, an error for overlap. + input2 := make([]byte, 64) + for i := range input2 { + input2[i] = byte(i + 1) + } + vm2, _ := newMemSyscallVM(t, input2, nil) + _, err = SyscallMemcpyImpl(vm2, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + if test.expectMemcpyE { + require.ErrorIs(t, err, SyscallErrCopyOverlapping) + } else { + require.NoError(t, err) + require.Equal(t, want, input2) + } + }) + } +} + +func TestSyscallMemmoveBadAddressOrder(t *testing.T) { + input := make([]byte, 32) + vm, _ := newMemSyscallVM(t, input, nil) + // Unreadable source is reported before an unwritable destination. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput+100, 8) + require.Error(t, err) + var badAccess sbpf.ExcBadAccess + require.ErrorAs(t, err, &badAccess) + require.False(t, badAccess.Write, "the source translation must fail first") + // Readable source, write to a read-only region. + _, err = SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput, 8) + require.Error(t, err) + require.ErrorAs(t, err, &badAccess) + require.True(t, badAccess.Write) +} + +func TestSyscallMemmoveIntoGrowingInputRegion(t *testing.T) { + // The destination region copy-on-writes and grows on first write; the + // source slice taken before that must still yield the original bytes. + original := []byte{1, 2, 3, 4, 5, 6, 7, 8} + var replaced []byte + region := sbpf.InputRegion{ + Offset: 0, + RegionSize: uint64(len(original)), + AddressSpaceReserved: 32, + Writable: false, + AccountIndex: 0, + Data: original, + OnWrite: func(region *sbpf.InputRegion, requestedLen uint64) error { + replaced = make([]byte, 32) + copy(replaced, region.Data) + region.Data = replaced + region.RegionSize = 32 + region.Writable = true + return nil + }, + } + vm, _ := newMemSyscallVM(t, nil, []sbpf.InputRegion{region}) + // Copy the first 4 bytes over bytes 4..8 within the same region: the + // source translation sees the original buffer, the destination the clone. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+4, sbpf.VaddrInput, 4) + require.NoError(t, err) + require.NotNil(t, replaced, "the first write must trigger the copy-on-write hook") + require.Equal(t, []byte{1, 2, 3, 4, 1, 2, 3, 4}, replaced[:8]) + require.Equal(t, []byte{1, 2, 3, 4, 5, 6, 7, 8}, original, "the shared buffer must stay untouched") +} + +func TestSyscallMemcmpAndMemset(t *testing.T) { + input := make([]byte, 128) + for i := range input { + input[i] = byte(i) + } + vm, _ := newMemSyscallVM(t, input, nil) + + // memcmp of equal and differing 32-byte keys, result written at 96. + copy(input[32:64], input[0:32]) + _, err := SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0), int32(binary.LittleEndian.Uint32(input[96:]))) + input[63] = 0xff + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(31)-int32(0xff), int32(binary.LittleEndian.Uint32(input[96:]))) + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput+32, sbpf.VaddrInput, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) + + // memset 0xab over 33 bytes, then zero over 9 bytes. + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) + require.Equal(t, byte(97), input[97]) + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) + require.Equal(t, make([]byte, 9), input[70:79]) + require.Equal(t, bytes.Repeat([]byte{0xab}, 18), input[79:97]) +} + +// The byte loop the syscalls used before memcmpResult; kept as the reference. +func referenceMemcmp(a, b []byte) int32 { + for i := range a { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + +func TestMemcmpResultMatchesByteLoop(t *testing.T) { + rng := rand.New(rand.NewSource(7)) + for iter := 0; iter < 100000; iter++ { + n := rng.Intn(70) + a := make([]byte, n) + rng.Read(a) + b := append([]byte(nil), a...) + if n > 0 && rng.Intn(4) != 0 { + b[rng.Intn(n)] = byte(rng.Intn(256)) + if rng.Intn(2) == 0 { + b[rng.Intn(n)] ^= byte(1 + rng.Intn(255)) + } + } + if got, want := memcmpResult(a, b), referenceMemcmp(a, b); got != want { + t.Fatalf("n=%d a=%x b=%x: got %d want %d", n, a, b, got, want) + } + } + if memcmpResult([]byte{0xff}, []byte{0x00}) != 255 || memcmpResult([]byte{0x00}, []byte{0xff}) != -255 { + t.Fatal("memcmp must return the unsigned byte difference") + } + if memcmpResult(nil, nil) != 0 { + t.Fatal("empty compare must be 0") + } +} + +func TestMemsetBytes(t *testing.T) { + for _, n := range []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 100, 1023, 4096, 10001} { + for _, c := range []byte{0, 1, 0x7f, 0xff} { + mem := make([]byte, n) + for i := range mem { + mem[i] = byte(i) + } + memsetBytes(mem, c) + if !bytes.Equal(mem, bytes.Repeat([]byte{c}, n)) { + t.Fatalf("n=%d c=%d: %x", n, c, mem) + } + } + } +} From ff5b10a68b989d2a3a5e7fb02f9be11900056ce0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:41:52 +0000 Subject: [PATCH 032/111] lthash: vectorize MixIn/MixOut with AVX2 on amd64 LtHash.MixIn/MixOut are 1024-lane uint16 add/subtract loops that run twice per modified account in the accounts delta hash (once for the old value, once for the new). The scalar loop costs ~560 ns per call in the sandbox and roughly 350 ns on Zen 5, so a block with ~10k modified accounts spends several milliseconds of worker CPU on lane arithmetic alone. On amd64 with AVX2 the lanes are now mixed with VPADDW/VPSUBW, 16 lanes per instruction, four vectors per iteration, unaligned loads and stores (28 ns per call here, 20x). Dispatch is a package variable set from cpu.X86.HasAVX2 (golang.org/x/sys is already a direct dependency); other architectures, CPUs without AVX2 and the purego build tag keep the portable loops, which remain the reference. Equals now compares the two arrays directly (runtime memequal) instead of a lane loop. Tests compare the assembly and the dispatched functions against the portable loops on random lanes including wrap-around values, check that MixOut inverts MixIn, that aliased operands behave, and that the generic fallback is selectable; go vet's asmdecl check passes and the package builds under -tags purego and GOARCH=arm64. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/lthash/lthash.go | 18 ++---- pkg/lthash/mix.go | 16 +++++ pkg/lthash/mix_amd64.go | 32 ++++++++++ pkg/lthash/mix_amd64.s | 56 +++++++++++++++++ pkg/lthash/mix_amd64_test.go | 51 +++++++++++++++ pkg/lthash/mix_generic.go | 6 ++ pkg/lthash/mix_test.go | 116 +++++++++++++++++++++++++++++++++++ 7 files changed, 283 insertions(+), 12 deletions(-) create mode 100644 pkg/lthash/mix.go create mode 100644 pkg/lthash/mix_amd64.go create mode 100644 pkg/lthash/mix_amd64.s create mode 100644 pkg/lthash/mix_amd64_test.go create mode 100644 pkg/lthash/mix_generic.go create mode 100644 pkg/lthash/mix_test.go diff --git a/pkg/lthash/lthash.go b/pkg/lthash/lthash.go index 9c4aaee1b..faa09a8a5 100644 --- a/pkg/lthash/lthash.go +++ b/pkg/lthash/lthash.go @@ -120,16 +120,15 @@ func (ltHash *LtHash) Clone() *LtHash { return new } +// MixIn adds other's 1024 lanes to ltHash, lane-wise modulo 2^16. The +// lane arithmetic is vectorized where the platform supports it (see mix.go). func (ltHash *LtHash) MixIn(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] + other.value[i] - } + mixIn(<Hash.value, &other.value) } +// MixOut subtracts other's lanes from ltHash, the inverse of MixIn. func (ltHash *LtHash) MixOut(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] - other.value[i] - } + mixOut(<Hash.value, &other.value) } func (ltHash *LtHash) Add(other *LtHash) *LtHash { @@ -143,12 +142,7 @@ func (ltHash *LtHash) Sub(other *LtHash) *LtHash { } func (ltHash *LtHash) Equals(other *LtHash) bool { - for i, element := range ltHash.value { - if element != other.value[i] { - return false - } - } - return true + return ltHash.value == other.value } func (ltHash *LtHash) Checksum() []byte { diff --git a/pkg/lthash/mix.go b/pkg/lthash/mix.go new file mode 100644 index 000000000..6cce9a4d9 --- /dev/null +++ b/pkg/lthash/mix.go @@ -0,0 +1,16 @@ +package lthash + +// mixInGeneric and mixOutGeneric are the portable lane loops. Every +// architecture-specific implementation must produce identical results: the +// lanes are independent uint16 additions and subtractions modulo 2^16. +func mixInGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] += src[i] + } +} + +func mixOutGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] -= src[i] + } +} diff --git a/pkg/lthash/mix_amd64.go b/pkg/lthash/mix_amd64.go new file mode 100644 index 000000000..f4818245c --- /dev/null +++ b/pkg/lthash/mix_amd64.go @@ -0,0 +1,32 @@ +//go:build amd64 && !purego + +package lthash + +import "golang.org/x/sys/cpu" + +// useAVX2 selects the vector lane loops. cpu.X86.HasAVX2 already includes +// the operating-system XSAVE/YMM-state check. Tests flip it to compare the +// two implementations on the same machine. +var useAVX2 = cpu.X86.HasAVX2 + +func mixIn(dst, src *[numElements]uint16) { + if useAVX2 { + mixInAVX2(dst, src) + return + } + mixInGeneric(dst, src) +} + +func mixOut(dst, src *[numElements]uint16) { + if useAVX2 { + mixOutAVX2(dst, src) + return + } + mixOutGeneric(dst, src) +} + +//go:noescape +func mixInAVX2(dst, src *[numElements]uint16) + +//go:noescape +func mixOutAVX2(dst, src *[numElements]uint16) diff --git a/pkg/lthash/mix_amd64.s b/pkg/lthash/mix_amd64.s new file mode 100644 index 000000000..2233de57e --- /dev/null +++ b/pkg/lthash/mix_amd64.s @@ -0,0 +1,56 @@ +//go:build amd64 && !purego + +#include "textflag.h" + +// The LtHash value is 1024 uint16 lanes = 2048 bytes = 16 iterations of +// four 32-byte YMM vectors. VPADDW/VPSUBW operate on 16-bit lanes modulo +// 2^16, exactly like the generic Go loop. Loads and stores are unaligned +// (VMOVDQU): LtHash values live inside Go structs with 2-byte alignment. + +// func mixInAVX2(dst, src *[1024]uint16) +TEXT ·mixInAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixin_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPADDW (SI)(AX*1), Y0, Y0 + VPADDW 32(SI)(AX*1), Y1, Y1 + VPADDW 64(SI)(AX*1), Y2, Y2 + VPADDW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixin_loop + VZEROUPPER + RET + +// func mixOutAVX2(dst, src *[1024]uint16) +TEXT ·mixOutAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixout_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPSUBW (SI)(AX*1), Y0, Y0 + VPSUBW 32(SI)(AX*1), Y1, Y1 + VPSUBW 64(SI)(AX*1), Y2, Y2 + VPSUBW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixout_loop + VZEROUPPER + RET diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go new file mode 100644 index 000000000..0cd378ea2 --- /dev/null +++ b/pkg/lthash/mix_amd64_test.go @@ -0,0 +1,51 @@ +//go:build amd64 && !purego + +package lthash + +import ( + "math/rand" + "testing" +) + +// TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has +// AVX2) against the portable loops so the comparison does not depend on the +// dispatch variable. +func TestMixAVX2AgainstGeneric(t *testing.T) { + if !useAVX2 { + t.Skip("no AVX2 on this machine") + } + rng := rand.New(rand.NewSource(6)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + want, got := *dst, *dst + mixInGeneric(&want, src) + mixInAVX2(&got, src) + if got != want { + t.Fatalf("mixInAVX2 diverges (iteration %d)", iter) + } + want, got = *dst, *dst + mixOutGeneric(&want, src) + mixOutAVX2(&got, src) + if got != want { + t.Fatalf("mixOutAVX2 diverges (iteration %d)", iter) + } + } +} + +// TestMixGenericFallbackSelectable makes sure the dispatch honours the flag, +// so a machine without AVX2 takes the portable path. +func TestMixGenericFallbackSelectable(t *testing.T) { + saved := useAVX2 + defer func() { useAVX2 = saved }() + useAVX2 = false + rng := rand.New(rand.NewSource(8)) + dst := randomLanes(rng) + src := randomLanes(rng) + want := *dst + mixInGeneric(&want, src) + mixIn(dst, src) + if *dst != want { + t.Fatal("generic fallback must be used when AVX2 is disabled") + } +} diff --git a/pkg/lthash/mix_generic.go b/pkg/lthash/mix_generic.go new file mode 100644 index 000000000..79e7b0bb4 --- /dev/null +++ b/pkg/lthash/mix_generic.go @@ -0,0 +1,6 @@ +//go:build !amd64 || purego + +package lthash + +func mixIn(dst, src *[numElements]uint16) { mixInGeneric(dst, src) } +func mixOut(dst, src *[numElements]uint16) { mixOutGeneric(dst, src) } diff --git a/pkg/lthash/mix_test.go b/pkg/lthash/mix_test.go new file mode 100644 index 000000000..272051689 --- /dev/null +++ b/pkg/lthash/mix_test.go @@ -0,0 +1,116 @@ +package lthash + +import ( + "math/rand" + "testing" +) + +func randomLanes(rng *rand.Rand) *[numElements]uint16 { + var lanes [numElements]uint16 + for i := range lanes { + switch rng.Intn(8) { + case 0: + lanes[i] = 0 + case 1: + lanes[i] = 0xffff + case 2: + lanes[i] = 0x8000 + default: + lanes[i] = uint16(rng.Uint32()) + } + } + return &lanes +} + +// TestMixMatchesGeneric checks the platform mixIn/mixOut against the +// portable loops, including wrap-around lanes, and that MixOut inverts MixIn. +func TestMixMatchesGeneric(t *testing.T) { + rng := rand.New(rand.NewSource(3)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + wantIn := *dst + mixInGeneric(&wantIn, src) + gotIn := *dst + mixIn(&gotIn, src) + if gotIn != wantIn { + t.Fatalf("mixIn diverges from the generic loop (iteration %d)", iter) + } + wantOut := *dst + mixOutGeneric(&wantOut, src) + gotOut := *dst + mixOut(&gotOut, src) + if gotOut != wantOut { + t.Fatalf("mixOut diverges from the generic loop (iteration %d)", iter) + } + roundTrip := gotIn + mixOut(&roundTrip, src) + if roundTrip != *dst { + t.Fatalf("mixOut does not invert mixIn (iteration %d)", iter) + } + } + // In-place: mixing a value into itself doubles every lane. + dst := randomLanes(rng) + want := *dst + for i := range want { + want[i] *= 2 + } + mixIn(dst, dst) + if *dst != want { + t.Fatal("mixIn with aliased operands must double every lane") + } + mixOut(dst, dst) + if *dst != [numElements]uint16{} { + t.Fatal("mixOut with aliased operands must clear every lane") + } +} + +func TestLtHashMixInMixOutAndEquals(t *testing.T) { + rng := rand.New(rand.NewSource(4)) + var a, b, c LtHash + a.value = *randomLanes(rng) + b.value = *randomLanes(rng) + c = *a.Clone() + c.MixIn(&b) + if c.Equals(&a) { + t.Fatal("mixing in a random value must change the hash") + } + c.MixOut(&b) + if !c.Equals(&a) { + t.Fatal("MixOut must undo MixIn") + } + c.value[numElements-1]++ + if c.Equals(&a) { + t.Fatal("Equals must see a last-lane difference") + } +} + +func BenchmarkMixIn(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixIn(dst, src) + } +} + +func BenchmarkMixInGeneric(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixInGeneric(dst, src) + } +} + +func BenchmarkMixOut(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixOut(dst, src) + } +} From 6c2490ee1e89c138c72aca45f03f4ca948d35d5c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 033/111] test: preserve zero-length memory validation and repair VM fixtures --- pkg/sealevel/sealevel_test.go | 12 +++++++----- pkg/sealevel/syscalls_mem.go | 8 ++++++++ pkg/sealevel/syscalls_mem_test.go | 21 ++++++++++++++++++++- 3 files changed, 35 insertions(+), 6 deletions(-) diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ad64f3a9..18309927b 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -65,12 +65,14 @@ func TestInterpreter_Noop(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 29834a9ce..73f96fc67 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,6 +24,14 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + // Translate intentionally bypasses address validation for zero-length slices; + // Read/Write did not. Preserve the old syscall validation and error order. + if n == 0 { + if err := vm.Read(src, nil); err != nil { + return err + } + return vm.Write(dst, nil) + } srcMem, err := vm.Translate(src, n, false) if err != nil { return err diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 3b02daea0..6cf5b5ba9 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -137,10 +137,11 @@ func TestSyscallMemcmpAndMemset(t *testing.T) { require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) // memset 0xab over 33 bytes, then zero over 9 bytes. + sentinel := input[97] // The preceding memcmp result overwrote bytes 96..99. _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) - require.Equal(t, byte(97), input[97]) + require.Equal(t, sentinel, input[97]) _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) @@ -197,3 +198,21 @@ func TestMemsetBytes(t *testing.T) { } } } + +func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { + for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { + for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { + vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) + want := vm.Read(src, nil) + if want == nil { + want = vm.Write(dst, nil) + } + got := memmoveImplInternal(vm, dst, src, 0) + if want == nil { + require.NoError(t, got) + } else { + require.EqualError(t, got, want.Error()) + } + } + } +} From 8764903c811207d1c9ee51dc221a12f0f27dca55 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 034/111] test: cover unaligned and aliased AVX2 hash lanes --- pkg/lthash/mix_amd64_test.go | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go index 0cd378ea2..110c1b39a 100644 --- a/pkg/lthash/mix_amd64_test.go +++ b/pkg/lthash/mix_amd64_test.go @@ -5,6 +5,7 @@ package lthash import ( "math/rand" "testing" + "unsafe" ) // TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has @@ -49,3 +50,25 @@ func TestMixGenericFallbackSelectable(t *testing.T) { t.Fatal("generic fallback must be used when AVX2 is disabled") } } + +func TestMixAVX2UnalignedAndAliased(t *testing.T) { + if !useAVX2 { + t.Skip("AVX2 unavailable") + } + rng := rand.New(rand.NewSource(19)) + for off := 0; off < 32; off += 2 { + storage := make([]byte, numElements*2+32) + dst := (*[numElements]uint16)(unsafe.Pointer(&storage[off])) + *dst = *randomLanes(rng) + want := *dst + mixInGeneric(&want, &want) + mixInAVX2(dst, dst) + if *dst != want { + t.Fatalf("aliased addition offset %d", off) + } + mixOutAVX2(dst, dst) + if *dst != ([numElements]uint16{}) { + t.Fatalf("aliased subtraction offset %d", off) + } + } +} From b03afadcfdf9161f2686e7ca13a985d6e97de155 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:15:39 -0500 Subject: [PATCH 035/111] sbpf: account for resolved call targets in program cache cost --- pkg/sbpf/program.go | 2 +- pkg/sbpf/program_memory_test.go | 12 ++++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) create mode 100644 pkg/sbpf/program_memory_test.go diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index 969a15469..a353a4594 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -42,7 +42,7 @@ func (p *Program) MemoryBytes() uint64 { if p == nil { return 0 } - total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + uint64(len(p.CallTargets))*8 if len(p.RO) == 0 { total += uint64(len(p.TextBytes)) } diff --git a/pkg/sbpf/program_memory_test.go b/pkg/sbpf/program_memory_test.go new file mode 100644 index 000000000..3360cf45d --- /dev/null +++ b/pkg/sbpf/program_memory_test.go @@ -0,0 +1,12 @@ +package sbpf + +import "testing" + +func TestProgramMemoryIncludesResolvedCalls(t *testing.T) { + p := &Program{Text: make([]Slot, 20)} + before := p.MemoryBytes() + p.ResolveCallTargets() + if got := p.MemoryBytes() - before; got != 20*8 { + t.Fatalf("resolved-call cache bytes = %d, want 160", got) + } +} From e53c1f333aa666b6b72ed949ca50278fc5a909dc Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:25:52 -0500 Subject: [PATCH 036/111] test: retain syscall differential coverage and execution reproduction guide --- docs/sbpf-interpreter-benchmarks.md | 80 +++++++++++++---------------- pkg/sealevel/syscalls_mem_test.go | 31 +++++++++++ 2 files changed, 68 insertions(+), 43 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 73eb970e5..d78653991 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -58,46 +58,40 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ -benchtime=1s -count=5 -benchmem ``` -## Recorded Alpenglow blocks - -Each run bootstrapped a fresh isolated AccountsDB from the same public full -snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay -covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and -142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, -nice 19. Two paired runs used baseline/candidate then candidate/baseline order. -The live validator's AccountsDB and configuration were not used or changed. - -All 233 per-slot bank hashes matched between baseline and candidate in both -pairs, excluding the run-specific comment header in `bankhash.log`. - -| Work measured | Pair 1 before → after | Pair 2 before → after | -|---|---:|---:| -| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | -| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | -| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | -| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | -| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | - -The sample includes blocks around 40–46 million CU. Three inspected non-empty -blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and -Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries -attribute whole-transaction CU to every participating program and must not be -summed as if they were exclusive per-program execution costs. - -Whole-block p95 did not improve, and the p99 changes are small/variable. The -slowest candidate blocks in the first pair contained no sBPF execution; their -large timers were dispatch and signature verification. These single-CPU replay -results do not establish a live voting/FAST improvement or production parallel -replay latency. They exclude network wait from ProcessBlock and are not elapsed -end-to-end catch-up times. - -## Correctness and limits - -- Native Zen 5 baseline and candidate differential outputs match for 100,000 - deterministic generated programs; candidate pool-zero checks pass. -- Baseline/candidate workload effects and CU charges match. Targeted race tests - for the interpreter, loader, and workload harness pass; vet passes. -- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; - therefore no complete sealevel test-suite pass is claimed. Broader conformance - testing remains separate from this performance experiment. -- No candidate was deployed and no validator restart was needed. +## Correctness and comparison boundaries + +The generated-program harness compares return values, errors, CU usage and memory +for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate +and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and +verifier semantics changes are excluded from this performance work. + +For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, +identical transaction parallelism, and the same slot interval. Compare normalized +per-slot bank hashes and slot sets before interpreting timings. Compare exact +`ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate +run order and retain raw outputs plus commit IDs outside the merge diff. + +The PR description links the recorded Alpenglow replay results and raw evidence. +Single-core shared-host results do not establish multicore contention or live FAST +inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe +a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now +supplies its execution context's compute meter. + +## Memory syscalls and LtHash + +Memory syscalls retain CU charges, source-before-destination error order, +zero-length behavior, memcpy overlap rejection and memmove overlap support. +Tests cover copy-on-write/growing regions and differential memory/CU results. + +LtHash uses AVX2 only when supported by both CPU and OS; other architectures and +`-tags purego` use portable loops. The vector path preserves 16-bit wraparound and +in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cover +both paths. Component speedups are not block-latency speedups. + +```sh +go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test -tags purego ./pkg/lthash +go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' +SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 +go test ./pkg/lthash -run '^$' -bench BenchmarkMix -benchmem -count=5 +``` diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 6cf5b5ba9..9fe7d22b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -216,3 +216,34 @@ func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { } } } + +// Compare the syscall with the old read-then-write path, including charged CU. +func TestMemoryCopyDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(127)) + for i := 0; i < 300; i++ { + data := make([]byte, 128) + rng.Read(data) + actual, want := append([]byte(nil), data...), append([]byte(nil), data...) + vm, ctx := newMemSyscallVM(t, actual, nil) + ref, refCtx := newMemSyscallVM(t, want, nil) + src, dst, n := uint64(rng.Intn(145)), uint64(rng.Intn(145)), uint64(rng.Intn(80)) + src += sbpf.VaddrInput + dst += sbpf.VaddrInput + _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) + wantErr := MemOpConsume(refCtx, n) + if wantErr == nil { + buf := make([]byte, n) + wantErr = ref.Read(src, buf) + if wantErr == nil { + wantErr = ref.Write(dst, buf) + } + } + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + require.Equal(t, want, actual) + require.Equal(t, refCtx.ComputeMeter.Remaining(), ctx.ComputeMeter.Remaining()) + } +} From 05ca7aed0276a5995a0f4f7cfc951f9f812cd95f Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:00:42 -0500 Subject: [PATCH 037/111] test: trim execution benchmark experiments and documentation --- docs/sbpf-interpreter-benchmarks.md | 49 ++------- docs/sha256-syscall.md | 107 ++++++------------- pkg/sealevel/program_workloads_bench_test.go | 2 +- pkg/sealevel/syscalls_sha256_bench_test.go | 81 +------------- 4 files changed, 43 insertions(+), 196 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index d78653991..095eb0b06 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -1,40 +1,11 @@ -# Interpreter performance validation +# Execution performance validation -This experiment compares `9db7dcb` plus identical benchmark code with the same -base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift -semantics patch is excluded. Three VASA tests were updated to pass the new -`*[16]uint64` register type; no further production optimization was added during -this measurement pass. +## Program workloads -## Individual programs - -Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five -one-second samples per variant with alternating run order. The existing validator -continued running. CPU 15 shares a physical core with CPU 7; these are shared-host -measurements rather than an isolated-machine throughput ceiling. - -The harness has a preloaded program cache and asserts a cache hit before timing. -Every invocation gets fresh account data and execution context. It measures -instruction setup, serialization, execution and publication together; it is not -just the interpreter loop. Timing instrumentation and instruction recording are -enabled identically on both versions. Tests check arithmetic return values, -post-transfer balances, allocation results, and recorded CPI. Requested budgets -are equal and the measured CU charges match between variants. - -| Workload | CU | Median before → after | Speedup | -|---|---:|---:|---:| -| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | -| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | -| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | -| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | -| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | -| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | -| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | - -The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. -The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from -the earlier SPL Token loader-only benchmark: they use different program binaries, -instructions, and include execution-context setup. +The program harness measures instruction setup, serialization, execution and +publication with a warm program cache and fresh account data per invocation. +It checks return values, account updates, CPI and CU consumption. Loader-only +benchmarks separately measure VM execution and program loading. Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` and `token2022.so` to enable those external fixtures. Without it, the in-repository @@ -71,11 +42,9 @@ per-slot bank hashes and slot sets before interpreting timings. Compare exact `ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate run order and retain raw outputs plus commit IDs outside the merge diff. -The PR description links the recorded Alpenglow replay results and raw evidence. +Record tested commit IDs, hardware, Go version, affinity and parallelism with results. Single-core shared-host results do not establish multicore contention or live FAST -inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe -a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now -supplies its execution context's compute meter. +inclusion gains. ## Memory syscalls and LtHash @@ -89,7 +58,7 @@ in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cov both paths. Component speedups are not block-latency speedups. ```sh -go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader go test -tags purego ./pkg/lthash go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 10c1b59b2..4045b2bac 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -1,89 +1,44 @@ -# SHA-256 syscall overhead +# SHA-256 syscall validation -The syscall decodes the already-translated slice descriptor array directly and -writes the final digest into the translated output buffer. It retains streaming -SHA-256, slice order, memory translations, CU charges and validation order. Output -is written only after all inputs have been read, preserving overlapping-buffer -behavior. No special case for a particular on-chain program is introduced. +The syscall decodes the translated slice descriptors directly and writes the +final digest into the translated output buffer. It retains streaming SHA-256, +slice order, memory translations, CU charges and validation order. Output is +written only after all inputs have been read, preserving overlapping-buffer +behavior. -A bounded 55-byte input-buffer prototype was slower than this simpler path and -is retained only as a benchmark comparison. The baseline reference is copied -from combined review commit `bd17683a`. - -Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; -medians below. Each benchmark runs serially through a real interpreter's memory -translation and CU meter, with VM creation outside the timed region. This does -not include VM instruction dispatch, a complete program, or block replay. - -| Input | Original | Direct decoding/output | Buffered prototype | -|---|---:|---:|---:| -| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | -| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | -| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | -| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | - -The two-slice case removes four allocations (112 bytes) per call. This is an -ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native -Zen 5 and captured heavy-block replay before deployment decisions. - -Reproduce with: - -```sh -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 -go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' -``` +The frozen reference implementation in the test harness supports differential +checks and before/after benchmarks. Both variants use the same VM, including +its contiguous-region bounds-overflow fix. Differential tests compare hashes, return/error values, remaining CU and all input/output memory over valid inputs, invalid descriptors/addresses, depleted budgets and output aliasing. The existing SHA program fixture also executes. -Testing exposed pre-existing overflow in contiguous VM region bounds checks; -that correction and its regression test are a separate preceding commit. Both -benchmark variants use the corrected VM. The old SHA fixture also needed its -compute-meter pointer initialized for the current interpreter API. +## Benchmarks -## Zen 5 acceleration and captured loop +`BenchmarkSha256Syscall` measures the syscall through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. It covers +empty, single-slice, multiple-slice and larger inputs; it excludes instruction +dispatch and transaction execution. -The live Go 1.26.4 validator on Ryzen 7 9700X had its actual -`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is -already active. No validator runtime setting was changed. +`BenchmarkSha256CapturedLoop` uses a captured SBF v0 hash loop with 1,000 +iterations and zero initial state. It retains descriptor setup, stack accesses, +digest copying, counter updates and branching. Its test checks the result against +a Go hash chain and checks equal CU consumption for both syscall implementations. +The captured ELF hash and extracted instruction range are recorded in the test. +Transaction loading, CPI and the rest of the original program are excluded. -The isolated loop harness copies text slots 518–545 from the captured SBF v0 -program, resolves the SHA syscall relocation, and supplies 1,000 iterations and -zero initial state. It retains descriptor setup, stack accesses, digest copying, -counter update and loop branching. A test checks its result against a Go hash -chain and checks equal CU consumption for both syscall implementations. This -excludes transaction loading, account dependencies, CPI and the remaining program. - -A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, -affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. -Three 150 ms samples (medians, per hash iteration): - -| Isolated loop | Time | -|---|---:| -| Original syscall | 234.5 ns | -| Optimized syscall | 165.9 ns | -| Dispatch-only diagnostic control | 109.2 ns | -| Go hash chain without VM | 54.69 ns | - -The optimized loop takes about 29% less time. The dispatch-only control replaces -the syscall with a no-op: it omits hashing, translations and syscall CU charging, -and is only an overhead diagnostic, never a valid execution implementation. -The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; -the buffered-input prototype remained slower at 71–73 ns. These short tests -share a host with other processes; they are not isolated-core latency guarantees. - -A separate short CPU profile of the optimized loop attributed 36.2% cumulative -sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI -compression routine. Most remaining sampled work was VM execution: instruction -dispatch/decoding, stack address translation, loads/stores and compute metering. -Cumulative and flat percentages overlap and must not be added. This profile is -of the harness, not of full-block replay or the live validator. - -The next execution experiment should target measured VM overhead and then replay -captured blocks; these results do not justify a claimed 29% block-time improvement. +The dispatch-only control omits hashing, translations and syscall CU charging; +it is an overhead diagnostic, not a valid execution implementation. The raw Go +hash chain provides another comparison outside the VM. Neither control can +establish a full-block speedup. ```sh -go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +go test ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256|TestSha256CapturedLoop' +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256(Syscall|CapturedLoop)$' -benchtime=1s -count=5 -benchmem ``` + +Alternate baseline/candidate run order and record commit IDs, Go version, +hardware and CPU affinity. Report component timings separately from full-program +and replay timings; hardware SHA acceleration also affects these results. diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go index cc6b83e78..649d35ddd 100644 --- a/pkg/sealevel/program_workloads_bench_test.go +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -19,7 +19,7 @@ import ( "github.com/stretchr/testify/require" ) -// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// External ELF inputs are pinned by SHA-256 in docs/sbpf-interpreter-benchmarks.md. // No live account writes or network calls occur in this harness. Each invocation // gets fresh account data; the program cache is warm and shared between runs. type programWorkload struct { diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go index b4240f54e..6c972f37f 100644 --- a/pkg/sealevel/syscalls_sha256_bench_test.go +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -14,8 +14,6 @@ import ( // Frozen syscall implementation from bd17683a; keep independent for differential tests. func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256BaselineReference") - if valsLen > cu.CUSha256MaxSlices { return syscallErr(SyscallErrTooManySlices) } @@ -73,81 +71,6 @@ func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) return syscallSuccess(0) } -// Experimental bounded-buffer variant retained only for benchmark comparison. -func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256SmallInputReference") - - if valsLen > cu.CUSha256MaxSlices { - return syscallErr(SyscallErrTooManySlices) - } - - execCtx := executionCtx(vm) - err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) - if err != nil { - return syscallCuErr() - } - - hashResult, err := vm.Translate(resultsAddr, 32, true) - if err != nil { - return syscallErr(err) - } - - hasher := sha256.New() - // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only - // this bounded case; larger inputs retain streaming hashing. - var small [55]byte - buffered := 0 - streaming := false - if valsLen > 0 { - var vals []byte - - // The data at 'valsAddr' consists of an array of 'slice references', which consists - // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that - // refers to an input value to hash. - // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above - vals, err = vm.Translate(valsAddr, valsLen*16, false) - if err != nil { - return syscallErr(err) - } - - var data []byte - - for count := uint64(0); count < valsLen; count++ { - - offset := count * 16 - vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} - - data, err = vm.Translate(vec.Addr, vec.Len, false) - if err != nil { - return syscallErr(err) - } - - cost := max(vec.Len/2, cu.CUMemOpBaseCost) - err = execCtx.ComputeMeter.Consume(cost) - if err != nil { - return syscallCuErr() - } - - if !streaming && len(data) <= len(small)-buffered { - buffered += copy(small[buffered:], data) - } else { - if !streaming { - hasher.Write(small[:buffered]) - streaming = true - } - hasher.Write(data) - } - } - } - if streaming { - hasher.Sum(hashResult[:0]) - } else { - digest := sha256.Sum256(small[:buffered]) - copy(hashResult, digest[:]) - } - return syscallSuccess(0) -} - type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { @@ -211,7 +134,7 @@ func TestSha256SyscallDifferential(t *testing.T) { var wantMem []byte var wantRet, wantCU uint64 var wantErr string - for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + for k, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { buf := append([]byte(nil), mem...) vm, ctx := sha256VM(buf, budget) ret, err := fn(vm, a, n, out) @@ -242,7 +165,7 @@ func BenchmarkSha256Syscall(b *testing.B) { for _, variant := range []struct { name string fn sha256Call - }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}} { b.Run(tc.name+"/"+variant.name, func(b *testing.B) { mem, a, n, out := sha256Fixture(tc.sizes) vm, ctx := sha256VM(mem, ^uint64(0)) From 8cdb0cf28138ec2276487e3f31e0802b028d3202 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 05:59:03 +0000 Subject: [PATCH 038/111] replay: record FullToReplayed, the vote-path latency replay controls For a turbine block the notarize vote fires from the replayed block, so the time from the last shred being assembled to OnReplayResult is the part of vote latency that execution determines. Record it per block as BlockReplay.FullToReplayed (from the assembler's ShredFullNanos) and as the replay_full_to_replayed_duration_seconds statsd timing, and reserve a StreamingExecution record for execution that overlaps shred reception. No behaviour change. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/metrics/metrics.go | 27 +++++++++++++++++++++++++++ pkg/replay/block.go | 17 +++++++++++++++++ pkg/statsd/statsd.go | 5 +++++ 3 files changed, 49 insertions(+) diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 3af188d16..8628093b0 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -145,6 +145,24 @@ type VoteRewardDetails struct { VoteAccountsUpdated uint64 } +// StreamingExecution records execution that ran while a block's shreds were +// still arriving. Groups are the contiguous ready-batch sets executed per +// wake-up; Transactions counts what they executed. TxLoopBeforeFull is the +// group execution wall time that finished before the slot was fully +// assembled, i.e. the work hidden behind reception. OpenDelay runs from the +// header batch being decoded to the stream opening (it measures how long the +// parent's tail held the child). Discarded is 1 when a stream for this slot +// was thrown away and the block was executed whole; DiscardReason names why. +type StreamingExecution struct { + Opened uint64 + Groups uint64 + Transactions uint64 + TxLoopBeforeFull Timing + OpenDelay Timing + Discarded uint64 + DiscardReason string +} + // Metrics for replaying a single block type BlockReplay struct { Slot uint64 @@ -215,6 +233,15 @@ type BlockReplay struct { ChainTipUpdate Timing ResumeContext Timing + // FullToReplayed is the vote-path latency Mithril controls: wall time from + // the last shred of a turbine block being assembled (the assembler's fullAt) + // to the replay result being handed to consensus. Absent for blocks that + // did not arrive as shreds. It is the number streaming execution reduces. + FullToReplayed Timing + // StreamingExecution summarizes any execution that overlapped shred + // reception for this block; all zero when the block was executed whole. + StreamingExecution StreamingExecution + LtHashInputAccounts uint64 LtHashUniqueAccounts uint64 LtHashUnchangedAccounts uint64 diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 646554962..2b41560dd 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -3089,6 +3089,7 @@ func ReplayBlocks( break } } + recordFullToReplayed(block) if rpcServer != nil { rpcServer.SetSlotCtx(lastSlotCtx) @@ -4445,3 +4446,19 @@ func ProcessBlock( setReplayStage("done") return slotCtx, err } + +// recordFullToReplayed measures the vote-path latency replay controls for a +// turbine block: from the assembler's full-assembly instant (the last shred, +// carried as ShredFullNanos) to the replay result reaching consensus. Blocks +// that did not arrive as shreds carry no full instant and record nothing. +func recordFullToReplayed(block *b.Block) { + if block == nil || block.ShredFullNanos <= 0 { + return + } + fullToReplayed := time.Since(time.Unix(0, block.ShredFullNanos)) + if fullToReplayed <= 0 { + return + } + metrics.GlobalBlockReplay.FullToReplayed.AddTiming(fullToReplayed) + _ = statsd.Duration(statsd.ReplayFullToReplayed, fullToReplayed, nil) +} diff --git a/pkg/statsd/statsd.go b/pkg/statsd/statsd.go index 2418cc025..5a9eda429 100644 --- a/pkg/statsd/statsd.go +++ b/pkg/statsd/statsd.go @@ -132,6 +132,8 @@ var ( TurbineEarlyPreparationWait = Metric{"turbine_early_preparation_wait_seconds"} TurbineEarlyVerifiedTransactions = Metric{"turbine_early_verified_transactions_total"} TurbineFullToReady = Metric{"turbine_full_to_ready_duration_seconds"} + // ReplayFullToReplayed: last shred assembled -> replay result handed to consensus. + ReplayFullToReplayed = Metric{"replay_full_to_replayed_duration_seconds"} // ReplaySigverifyGroup times one drained group of transaction signatures // and ReplaySigverifyGroupSignatures counts how many signatures were in it. // The pair is what tells an operator whether batching is actually happening: @@ -259,6 +261,7 @@ var MetricToType = map[Metric]metricType{ TurbineEarlyPreparationWait: TimingT, TurbineEarlyVerifiedTransactions: CountT, TurbineFullToReady: TimingT, + ReplayFullToReplayed: TimingT, ReplaySigverifyGroup: TimingT, ReplaySigverifyGroupSignatures: CountT, TurbineReplayAdmission: TimingT, @@ -373,6 +376,7 @@ var MetricToLabels = map[Metric][]string{ TurbineEarlyPreparationWait: {}, TurbineEarlyVerifiedTransactions: {}, TurbineFullToReady: {}, + ReplayFullToReplayed: {}, ReplaySigverifyGroup: {}, ReplaySigverifyGroupSignatures: {}, TurbineReplayAdmission: {}, @@ -414,6 +418,7 @@ var MetricToBuckets = map[Metric][]float64{ TurbineEarlyTransactionSigverify: turbinePipelineDurationBuckets, TurbineEarlyPreparationWait: turbinePipelineDurationBuckets, TurbineFullToReady: turbinePipelineDurationBuckets, + ReplayFullToReplayed: turbinePipelineDurationBuckets, ReplaySigverifyGroup: turbinePipelineDurationBuckets, TurbineReplayAdmission: turbinePipelineDurationBuckets, AlpenglowVoteRewards: turbinePipelineDurationBuckets, From 411e8b06f65672a3b1b5ef243cc9aa395d35ea6c Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 06:08:06 +0000 Subject: [PATCH 039/111] replay: make block execution resumable (open / execute group / finalize) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Split ProcessBlock's state into blockExecution so a bank can be opened on the parent state, fed transaction groups in block order, and finalized later. ProcessBlock keeps its exact sequence — whole-block validation, prepared planner overlapping the account load, TxLoop — and now calls the shared opening (installSlotCtx) and the shared tail (finalize: fees to the leader, rent, incinerator, Alpenglow footer clock and vote rewards, bank sysvar finalization, bank hash, footer verification, state publication, status commit), which moved verbatim into block_execution.go. Its trace task, stage watchdog and signature-verification join live on the same object with the same cleanup order. executeTransactionGroup is the incremental path streaming execution will drive: per group it checks message versions, duplicate messages across all groups so far (a duplicate makes the whole block invalid, reported exactly as planBlockTransactionExecution reports it), the ancestor already-processed status check, resolves address-table lookups, loads every account the group references into the same pristine parent snapshot (accounts already present keep their first image), and executes with a dependency plan over the group and up to txParallelism workers, or sequentially. Groups run one after another, so cross-group ordering is block order. Nothing calls it yet. Tests: the group executor against the sequential ProcessTransaction reference and against itself at random split points (shared payer and destination across groups, failing transfers that still pay fees, txpar 0, 1 and 4), duplicate detection across groups, V1 rejection, the first parent image being kept, and refusal after close. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/block.go | 246 +-------- pkg/replay/block_execution.go | 683 +++++++++++++++++++++++++ pkg/replay/block_execution_test.go | 304 +++++++++++ pkg/replay/transaction_status_cache.go | 19 + 4 files changed, 1027 insertions(+), 225 deletions(-) create mode 100644 pkg/replay/block_execution.go create mode 100644 pkg/replay/block_execution_test.go diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 2b41560dd..2f35831e0 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -25,7 +25,6 @@ import ( "github.com/Overclock-Validator/mithril/pkg/accountsdb" a "github.com/Overclock-Validator/mithril/pkg/addresses" "github.com/Overclock-Validator/mithril/pkg/arena" - "github.com/Overclock-Validator/mithril/pkg/bankhash" "github.com/Overclock-Validator/mithril/pkg/base58" b "github.com/Overclock-Validator/mithril/pkg/block" "github.com/Overclock-Validator/mithril/pkg/blockstream" @@ -37,7 +36,6 @@ import ( "github.com/Overclock-Validator/mithril/pkg/lthash" "github.com/Overclock-Validator/mithril/pkg/metrics" "github.com/Overclock-Validator/mithril/pkg/mlog" - "github.com/Overclock-Validator/mithril/pkg/rent" "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/rpcclient" "github.com/Overclock-Validator/mithril/pkg/sealevel" @@ -4174,66 +4172,17 @@ func ProcessBlock( metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) } }() - ctx, task := trace.NewTask(context.Background(), "ProcessBlock") - defer task.End() - trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) - trace.Log(ctx, "txCount", fmt.Sprintf("%d", len(block.Transactions))) - var replayStage atomic.Value - var replayStageSince atomic.Int64 - setReplayStage := func(stage string) { - replayStage.Store(stage) - replayStageSince.Store(time.Now().UnixNano()) - } - setReplayStage("prepare_dependency_planner") - - replayWatchdogDone := make(chan struct{}) - go func() { - ticker := time.NewTicker(5 * time.Second) - defer ticker.Stop() - - var lastLoggedStage string - var lastLoggedSince int64 - for { - select { - case <-replayWatchdogDone: - return - case <-ticker.C: - stageVal := replayStage.Load() - stage, ok := stageVal.(string) - if !ok || stage == "" { - continue - } - sinceUnix := replayStageSince.Load() - if sinceUnix == 0 { - continue - } - if stage == lastLoggedStage && sinceUnix == lastLoggedSince { - continue - } - stageDuration := time.Since(time.Unix(0, sinceUnix)) - if stageDuration < 10*time.Second { - continue - } - mlog.Log.Warnf("REPLAY WATCHDOG: slot %d stuck in stage %s for %s | txs=%d | lightbringer=%t", - block.Slot, stage, stageDuration.Round(time.Second), len(block.Transactions), block.FromLiveStream) - lastLoggedStage = stage - lastLoggedSince = sinceUnix - } - } - }() - defer close(replayWatchdogDone) + // The resumable execution state carries the trace task, stage watchdog and + // the SlotCtx; streaming execution drives the same object group by group. + exec := newBlockExecution(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, tail, transactionStatuses, alpenglowClock, parentBankSysvars) + defer exec.close() + exec.setReplayStage("prepare_dependency_planner") if SerializedParameterArena != nil { SerializedParameterArena.Reset() } - var sigverifyWg sync.WaitGroup - defer func() { - sigverifyJoinStart := time.Now() - sigverifyWg.Wait() - metrics.GlobalBlockReplay.SignatureVerificationJoin.AddTimingSince(sigverifyJoinStart) - }() plannerPreparationStart := time.Now() var planner *preparedDependencyPlanner if txParallelism > 0 { @@ -4246,15 +4195,9 @@ func ProcessBlock( metrics.GlobalBlockReplay.DependencyPlannerPreparation.AddTimingSince(plannerPreparationStart) start := time.Now() - setReplayStage("load_accounts") - loadAcctsRegion := trace.StartRegion(ctx, "LoadBlockAccounts") - // In rooted-durable mode, block accounts/sysvars load through the unrooted - // tail (overlay→durable) so execution sees confirmed-but-unrooted state. - var blockSrc blockAccountSource = acctsDb - if tail != nil { - blockSrc = tail - } - accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(blockSrc, block, epochSchedule, alpenglowClock, parentBankSysvars, planner) + exec.setReplayStage("load_accounts") + loadAcctsRegion := trace.StartRegion(exec.ctx, "LoadBlockAccounts") + accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(exec.blockSrc, block, epochSchedule, alpenglowClock, parentBankSysvars, planner) loadAcctsRegion.End() if err != nil { panic(fmt.Sprintf("unable to load slot accounts and update sysvars: %s", err)) @@ -4265,186 +4208,39 @@ func ProcessBlock( metrics.GlobalBlockReplay.LoadBlockAccounts.AddTimingSince(start) slotCtxSetupStart := time.Now() - slotCtx := newSlotCtx(block, accts, parentAccts, acctsDb, tail, accountMapCapacity) - if err := slotCtx.PublishBankSysvars(bankSysvars); err != nil { - return nil, fmt.Errorf("publish bank sysvars at slot %d: %w", block.Slot, err) + if err := exec.installSlotCtx(accts, parentAccts, accountMapCapacity, bankSysvars); err != nil { + return nil, err } - bankEpochScheduleValue, ok := bankSysvars.EpochSchedule() - if !ok { - return nil, fmt.Errorf("bank-local EpochSchedule sysvar unavailable at slot %d", block.Slot) - } - bankEpochSchedule := &bankEpochScheduleValue + slotCtx := exec.slotCtx if requireAlpenglowBlockFooter(block, slotCtx, alpenglowClock) { if err := validateAlpenglowFooterNanosecondClock(slotCtx, block); err != nil { return nil, err } } - slotCtx.TraceCtx = ctx slotCtx.NumSignatures = executionPlan.processedSignatures metrics.GlobalBlockReplay.SlotCtxSetup.AddTimingSince(slotCtxSetupStart) var txFeeAccumulator fees.TxFeeInfoAccumulator var totalComputeUnitsConsumed uint64 start = time.Now() - setReplayStage("tx_loop") - txLoopRegion := trace.StartRegion(ctx, "TxLoop") + exec.setReplayStage("tx_loop") + txLoopRegion := trace.StartRegion(exec.ctx, "TxLoop") shouldVerifySignatures := !block.TransactionSignaturesVerified() if txParallelism > 0 { - txFeeAccumulator, totalComputeUnitsConsumed = parallelTxLoop(slotCtx, &sigverifyWg, planner, block, executionPlan, txParallelism, dbgOpts, shouldVerifySignatures) + txFeeAccumulator, totalComputeUnitsConsumed = parallelTxLoop(slotCtx, &exec.sigverifyWg, planner, block, executionPlan, txParallelism, dbgOpts, shouldVerifySignatures) } else { - txFeeAccumulator, totalComputeUnitsConsumed = sequentialTxLoop(slotCtx, &sigverifyWg, block, executionPlan, dbgOpts, shouldVerifySignatures) + txFeeAccumulator, totalComputeUnitsConsumed = sequentialTxLoop(slotCtx, &exec.sigverifyWg, block, executionPlan, dbgOpts, shouldVerifySignatures) } slotCtx.TotalComputeUnitsConsumed = totalComputeUnitsConsumed txLoopRegion.End() metrics.GlobalBlockReplay.TxLoop.AddTimingSince(start) - start = time.Now() - setReplayStage("distribute_fees") - - // distribute tx fees to the slot leader - // skip leader handling if there are zero transactions in this block - if !global.ManageLeaderSchedule() && block.BlockReward != nil && len(block.Transactions) > 0 { - slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.BlockReward.Leader, &txFeeAccumulator) - slotCtx.RecordModifiedAcct(block.BlockReward.Leader) - } else if global.ManageLeaderSchedule() && len(block.Transactions) > 0 { - slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.Leader, &txFeeAccumulator) - slotCtx.RecordModifiedAcct(block.Leader) - } - metrics.GlobalBlockReplay.Reward.AddTimingSince(start) - - start = time.Now() - setReplayStage("collect_rent") - bankRent, ok := slotCtx.BankSysvars().Rent() - if !ok { - return nil, fmt.Errorf("bank-local Rent sysvar unavailable at slot %d", block.Slot) - } - rentAccts := rent.CollectRentEagerly(slotCtx, &bankRent, bankEpochSchedule) - metrics.GlobalBlockReplay.Rent.AddTimingSince(start) - - start = time.Now() - setReplayStage("run_incinerator") - runIncinerator(slotCtx) - metrics.GlobalBlockReplay.RunIncinerator.AddTimingSince(start) - - // Alpenglow banks set the Clock timestamp from the block footer after execution. - if alpenglowClock { - footerClockStart := time.Now() - if err := applyAlpenglowFooterClock(slotCtx, block, bankEpochSchedule); err != nil { - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - return nil, fmt.Errorf("apply alpenglow footer clock at slot %d: %w", block.Slot, err) - } - if err := updateAlpenglowNanosecondClockAccount(slotCtx, block); err != nil { - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - return nil, err - } - metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) - voteRewardsStart := time.Now() - voteRewardsErr := ApplyAlpenglowVoteRewards(slotCtx, block, bankEpochSchedule, block.SkipRewardCert, block.NotarRewardCert, block.BlockFinalCert, block.AlpenglowShredVersion) - metrics.GlobalBlockReplay.AlpenglowVoteRewards.AddTimingSince(voteRewardsStart) - if voteRewardsErr != nil { - return nil, voteRewardsErr - } - } - if err := finalizeBankSysvars(slotCtx); err != nil { - return nil, fmt.Errorf("finalize bank sysvars at slot %d: %w", block.Slot, err) - } - - setReplayStage("compile_accounts") - start = time.Now() - writableAccts, modifiedAccts := compileWritableAndModifiedAccts(slotCtx, block, rentAccts) - metrics.GlobalBlockReplay.CompileWritableAndModifiedAccts.AddTimingSince(start) - start = time.Now() - ensureParentsErr := ensureParentAccountsForModified(slotCtx, modifiedAccts) - metrics.GlobalBlockReplay.EnsureParentAccountsForModified.AddTimingSince(start) - if ensureParentsErr != nil { - return nil, ensureParentsErr - } - - start = time.Now() - setReplayStage("bankhash") - slotCtx.FinalBankhash = bankhash.CalculateBankHash(slotCtx, writableAccts, modifiedAccts, block.ParentBankhash, slotCtx.NumSignatures, block.Blockhash) - metrics.GlobalBlockReplay.BankHash.AddTimingSince(start) - if alpenglowClock { - footerVerificationStart := time.Now() - footerVerificationErr := verifyAlpenglowBlockFooter(slotCtx, block, alpenglowClock) - metrics.GlobalBlockReplay.AlpenglowFooterVerification.AddTimingSince(footerVerificationStart) - if footerVerificationErr != nil { - writeFooterBankhashMismatchArtifact(footerVerificationErr, block, slotCtx, writableAccts, modifiedAccts) - return nil, footerVerificationErr - } - } - - // Bankhash consensus enforcement is handled in the replay loop (not here) - // because forkchoice is fed after ProcessBlock returns — checking here would - // never see votes from recently submitted blocks and could deadlock. - - // Enter critical commit window - panics here may leave AccountsDB inconsistent - commitSlot.Store(slotCtx.Slot) - commitInProgress.Store(true) - blockUpdateStart := time.Now() - setReplayStage("store_accounts") - persistedSlot := slotCtx.Slot - persistedBankhash := append([]byte(nil), slotCtx.FinalBankhash...) - persistedBlockSlot := block.Slot - stakeIndexDir := filepath.Join(acctsDb.AcctsDir, "..") - afterStoreAccounts := func() { - if tail != nil { - // Rooted-durable: accounts + bankhash are buffered in the overlay and - // become durable only on promotion; nothing written here (rooted-only). - } else { - if berr := acctsDb.StoreBankHashForSlot(persistedSlot, persistedBankhash); berr != nil { - mlog.Log.Infof("unable to store bankhash for slot %d", persistedSlot) - } - } - if tail == nil { - // Legacy/verify modes (no fork ambiguity): flush per block as before. - // Rooted-durable replay flushes at FOLD time instead — entries stay - // slot-scoped in RAM so a fork unwind can drop them, and scans merge - // the pending set (StreamStakeAccounts) for completeness meanwhile. - flushed, err := global.FlushPendingStakePubkeys(stakeIndexDir) - if err != nil { - mlog.Log.Errorf("failed to flush stake pubkey index: %v", err) - } else if flushed > 0 { - mlog.Log.Debugf("flushed %d new stake pubkeys to index", flushed) - } - } - - persistedHashes.Set(persistedBlockSlot, persistedBankhash) - - // Exit critical commit window - AccountsDB is now consistent - commitInProgress.Store(false) - commitSlot.Store(0) - } - - if tail != nil { - // Rooted-durable: buffer this slot's writes + bankhash in the RAM overlay - // (always, even when empty, so the bankhash is recorded); no durable write. - tail.Add(slotCtx.Slot, modifiedAccts, persistedBankhash) - afterStoreAccounts() - } else if len(modifiedAccts) > 0 { - err = acctsDb.StoreAccounts(modifiedAccts, slotCtx.Slot, afterStoreAccounts) - } - // In rooted-durable mode the callback above is synchronous, so this includes - // the complete critical-path overlay publication. Legacy StoreAccounts only - // enqueues here; its asynchronous disk work deliberately belongs to no slot's - // replay wall time and must never update a later slot's metrics record. - metrics.GlobalBlockReplay.BlockUpdateAccounts.AddTimingSince(blockUpdateStart) - if err != nil { - return slotCtx, err - } - statusCommitStart := time.Now() - statusWaitStart := time.Now() - preparedStatuses := statusPreparation.wait() - metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) - statusErr := transactionStatuses.commitBlockWithValidation(block, executionPlan, preparedStatuses, statusValidation) - metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) - if statusErr != nil { - return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) - } - - global.IncrTransactionCount(executionPlan.processedTxCount) - setReplayStage("done") - return slotCtx, err + exec.txFeeAccumulator = txFeeAccumulator + exec.totalCU = totalComputeUnitsConsumed + exec.executionPlan = executionPlan + exec.statusPreparation = statusPreparation + exec.statusValidation = statusValidation + return exec.finalize() } // recordFullToReplayed measures the vote-path latency replay controls for a diff --git a/pkg/replay/block_execution.go b/pkg/replay/block_execution.go new file mode 100644 index 000000000..854590479 --- /dev/null +++ b/pkg/replay/block_execution.go @@ -0,0 +1,683 @@ +package replay + +import ( + "context" + "errors" + "fmt" + "path/filepath" + "runtime/trace" + "sync" + "sync/atomic" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/arena" + "github.com/Overclock-Validator/mithril/pkg/bankhash" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/fees" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/rent" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) + +// blockExecution is the resumable state of one bank's execution. ProcessBlock +// drives it in one pass (plan, load, execute every transaction, finalize). +// Streaming execution opens it before the block is complete, feeds +// transaction groups as their shreds arrive, and finalizes against the +// complete block. Both paths share the opening (bank sysvars, SlotCtx) and the +// tail (fees, rent, footer, bank hash, commit), so a bank produced either way +// runs the same end-of-block code over the same SlotCtx. +type blockExecution struct { + acctsDb *accountsdb.AccountsDb + block *b.Block + epochSchedule *sealevel.SysvarEpochSchedule + txParallelism int + dbgOpts *DebugOptions + persistedHashes *persistedTracker + tail unrootedState + transactionStatuses *TransactionStatusCache + alpenglowClock bool + parentBankSysvars *sealevel.BankSysvars + + blockSrc blockAccountSource + slotCtx *sealevel.SlotCtx + parentAccts accounts.MemAccounts + accts accounts.Accounts + bankSysvars *sealevel.BankSysvars + bankEpochSchedule *sealevel.SysvarEpochSchedule + + ctx context.Context + task *trace.Task + setReplayStage func(string) + watchdogDone chan struct{} + sigverifyWg sync.WaitGroup + closed bool + + // Whole-block inputs to the tail, set by ProcessBlock. + executionPlan blockTransactionExecutionPlan + statusPreparation *transactionStatusPreparation + statusValidation transactionStatusValidation + + // Incremental transaction bookkeeping in block order, maintained by + // executeTransactionGroup. ProcessBlock does not use it. + transactions []*solana.Transaction + identities []txstatus.TransactionMessageIdentity + execute []bool + seenMessages map[[32]byte]int + processedTxCount uint64 + processedSignatures uint64 + groups int + + txFeeAccumulator fees.TxFeeInfoAccumulator + totalCU uint64 +} + +// newBlockExecution installs the per-bank trace task, the stage watchdog and +// the account source; it does not touch bank state. +func newBlockExecution( + acctsDb *accountsdb.AccountsDb, + block *b.Block, + epochSchedule *sealevel.SysvarEpochSchedule, + txParallelism int, + dbgOpts *DebugOptions, + persistedHashes *persistedTracker, + tail unrootedState, + transactionStatuses *TransactionStatusCache, + alpenglowClock bool, + parentBankSysvars *sealevel.BankSysvars, +) *blockExecution { + exec := &blockExecution{ + acctsDb: acctsDb, + block: block, + epochSchedule: epochSchedule, + txParallelism: txParallelism, + dbgOpts: dbgOpts, + persistedHashes: persistedHashes, + tail: tail, + transactionStatuses: transactionStatuses, + alpenglowClock: alpenglowClock, + parentBankSysvars: parentBankSysvars, + seenMessages: make(map[[32]byte]int), + } + // In rooted-durable mode, block accounts/sysvars load through the unrooted + // tail (overlay→durable) so execution sees confirmed-but-unrooted state. + exec.blockSrc = acctsDb + if tail != nil { + exec.blockSrc = tail + } + + ctx, task := trace.NewTask(context.Background(), "ProcessBlock") + exec.ctx, exec.task = ctx, task + trace.Log(ctx, "slot", fmt.Sprintf("%d", block.Slot)) + trace.Log(ctx, "txCount", fmt.Sprintf("%d", len(block.Transactions))) + + var replayStage atomic.Value + var replayStageSince atomic.Int64 + exec.setReplayStage = func(stage string) { + replayStage.Store(stage) + replayStageSince.Store(time.Now().UnixNano()) + } + + exec.watchdogDone = make(chan struct{}) + go func() { + ticker := time.NewTicker(5 * time.Second) + defer ticker.Stop() + + var lastLoggedStage string + var lastLoggedSince int64 + for { + select { + case <-exec.watchdogDone: + return + case <-ticker.C: + stageVal := replayStage.Load() + stage, ok := stageVal.(string) + if !ok || stage == "" { + continue + } + sinceUnix := replayStageSince.Load() + if sinceUnix == 0 { + continue + } + if stage == lastLoggedStage && sinceUnix == lastLoggedSince { + continue + } + stageDuration := time.Since(time.Unix(0, sinceUnix)) + if stageDuration < 10*time.Second { + continue + } + mlog.Log.Warnf("REPLAY WATCHDOG: slot %d stuck in stage %s for %s | txs=%d | lightbringer=%t", + block.Slot, stage, stageDuration.Round(time.Second), len(block.Transactions), block.FromLiveStream) + lastLoggedStage = stage + lastLoggedSince = sinceUnix + } + } + }() + return exec +} + +// close joins outstanding signature verification, stops the watchdog and ends +// the trace task, in the order ProcessBlock's deferred cleanup always used. It +// is idempotent so a discarded stream and a finalized bank both call it. +func (exec *blockExecution) close() { + if exec == nil || exec.closed { + return + } + exec.closed = true + sigverifyJoinStart := time.Now() + exec.sigverifyWg.Wait() + metrics.GlobalBlockReplay.SignatureVerificationJoin.AddTimingSince(sigverifyJoinStart) + if exec.watchdogDone != nil { + close(exec.watchdogDone) + } + if exec.task != nil { + exec.task.End() + } +} + +// installSlotCtx publishes the loaded parent snapshot and derived bank sysvars +// as this bank's SlotCtx. The overlay/parent pair comes from +// loadBlockAccountsAndUpdateSysvars; streaming grows the parent snapshot +// afterwards, group by group, through loadTransactionAccounts. +func (exec *blockExecution) installSlotCtx(accts accounts.Accounts, parentAccts accounts.Accounts, accountMapCapacity int, bankSysvars *sealevel.BankSysvars) error { + block := exec.block + slotCtx := newSlotCtx(block, accts, parentAccts, exec.acctsDb, exec.tail, accountMapCapacity) + if err := slotCtx.PublishBankSysvars(bankSysvars); err != nil { + return fmt.Errorf("publish bank sysvars at slot %d: %w", block.Slot, err) + } + bankEpochScheduleValue, ok := bankSysvars.EpochSchedule() + if !ok { + return fmt.Errorf("bank-local EpochSchedule sysvar unavailable at slot %d", block.Slot) + } + slotCtx.TraceCtx = exec.ctx + exec.slotCtx = slotCtx + exec.accts = accts + if mem, ok := parentAccts.(accounts.MemAccounts); ok { + exec.parentAccts = mem + } + exec.bankSysvars = bankSysvars + exec.bankEpochSchedule = &bankEpochScheduleValue + return nil +} + +// open performs the bank-start work that needs only the parent state: the +// parent sysvar pin and this bank's Clock/SlotHashes derivation, the parent +// snapshot for whatever transactions the block currently carries (none for a +// streaming shell), the overlay, and the SlotCtx. It is ProcessBlock's opening +// without the whole-block planner, so a streaming caller can start executing +// groups before any transaction of the block is known. +func (exec *blockExecution) open() error { + block := exec.block + exec.setReplayStage("prepare_dependency_planner") + if SerializedParameterArena != nil { + SerializedParameterArena.Reset() + } + + start := time.Now() + exec.setReplayStage("load_accounts") + loadAcctsRegion := trace.StartRegion(exec.ctx, "LoadBlockAccounts") + accts, parentAccts, accountMapCapacity, bankSysvars, err := loadBlockAccountsAndUpdateSysvars(exec.blockSrc, block, exec.epochSchedule, exec.alpenglowClock, exec.parentBankSysvars, nil) + loadAcctsRegion.End() + if err != nil { + return fmt.Errorf("load slot accounts and update sysvars at slot %d: %w", block.Slot, err) + } + if err := bankSysvars.ValidateForExecution(); err != nil { + return fmt.Errorf("invalid bank sysvar snapshot at slot %d: %w", block.Slot, err) + } + metrics.GlobalBlockReplay.LoadBlockAccounts.AddTimingSince(start) + + slotCtxSetupStart := time.Now() + if err := exec.installSlotCtx(accts, parentAccts, accountMapCapacity, bankSysvars); err != nil { + return err + } + metrics.GlobalBlockReplay.SlotCtxSetup.AddTimingSince(slotCtxSetupStart) + return nil +} + +// errBlockExecutionClosed reports a group offered after close or finalize. +var errBlockExecutionClosed = errors.New("block execution is closed") + +// groupIdentitiesFor returns prepared identities for a transaction group, +// hashing the messages when the caller has none from signature verification. +func groupIdentitiesFor(txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities) (*b.PreparedTransactionMessageIdentities, error) { + if identities != nil { + if identities.Len() != len(txs) { + return nil, fmt.Errorf("group identities cover %d transactions, group has %d", identities.Len(), len(txs)) + } + return identities, nil + } + view := &b.Block{Transactions: txs} + return view.PrepareTransactionMessageIdentities() +} + +// executeTransactionGroup executes the next transactions of the block, in +// block order, against the open SlotCtx. Every check ProcessBlock applies to a +// whole block is applied incrementally: message versions against the bank's +// features, duplicate messages across every group so far (a duplicate makes +// the whole block invalid, exactly as planBlockTransactionExecution reports +// it), ancestor status-cache validation, address-table resolution, account +// loading into the same parent snapshot, and a dependency plan over the group +// executed by up to txParallelism workers. Groups run strictly one after +// another, so cross-group ordering is the sequential block order. +// +// identities may carry the verifier's message identities for exactly these +// transactions; nil hashes them here. shouldVerifySignatures is passed to +// ProcessTransaction unchanged. +func (exec *blockExecution) executeTransactionGroup(txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error { + if exec == nil || exec.slotCtx == nil { + return errors.New("block execution is not open") + } + if exec.closed { + return errBlockExecutionClosed + } + if len(txs) == 0 { + return nil + } + block := exec.block + slot := block.Slot + base := len(exec.transactions) + + view := &b.Block{Slot: slot, Transactions: txs, Features: block.Features} + if err := validateBlockTransactionVersions(view); err != nil { + return fmt.Errorf("validate transaction versions for slot %d: %w", slot, err) + } + + prepared, err := groupIdentitiesFor(txs, identities) + if err != nil { + return fmt.Errorf("validate transaction messages for slot %d: %w", slot, err) + } + execute := make([]bool, len(txs)) + var duplicates *DuplicateTransactionMessagesError + for idx, tx := range txs { + if tx == nil { + return fmt.Errorf("validate transaction messages for slot %d: transaction %d is nil", slot, base+idx) + } + identity := prepared.Identity(idx) + if firstIndex, duplicate := exec.seenMessages[identity.MessageHash]; duplicate { + if duplicates == nil { + duplicates = &DuplicateTransactionMessagesError{Slot: slot} + } + duplicates.DuplicateCount++ + if len(duplicates.Occurrences) < maxDuplicateTransactionOccurrences { + duplicates.Occurrences = append(duplicates.Occurrences, DuplicateTransactionOccurrence{ + Index: base + idx, FirstIndex: firstIndex, + }) + } + continue + } + exec.seenMessages[identity.MessageHash] = base + idx + execute[idx] = true + } + if duplicates != nil { + return fmt.Errorf("validate transaction messages for slot %d: %w", slot, duplicates) + } + if exec.transactionStatuses != nil { + if err := exec.transactionStatuses.validateTransactionsAgainstAncestors(slot, prepared); err != nil { + return fmt.Errorf("validate transaction statuses for slot %d: %w", slot, err) + } + } + + // Record the group before executing so a failure after this point still + // leaves the block-order view consistent for finalize's prefix proof. + for idx, tx := range txs { + exec.transactions = append(exec.transactions, tx) + exec.identities = append(exec.identities, prepared.Identity(idx)) + exec.execute = append(exec.execute, execute[idx]) + if execute[idx] { + exec.processedTxCount++ + exec.processedSignatures += uint64(tx.Message.Header.NumRequiredSignatures) + } + } + exec.slotCtx.NumSignatures = exec.processedSignatures + + exec.setReplayStage("load_accounts") + if err := exec.loadTransactionAccounts(view); err != nil { + return err + } + + exec.setReplayStage("tx_loop") + start := time.Now() + txLoopRegion := trace.StartRegion(exec.ctx, "TxLoop") + feeInfos, computeUnits, err := exec.runTransactionGroup(txs, execute, shouldVerifySignatures) + txLoopRegion.End() + metrics.GlobalBlockReplay.TxLoop.AddTimingSince(start) + if err != nil { + return err + } + for idx, txFeeInfo := range feeInfos { + if !execute[idx] { + continue + } + exec.totalCU += computeUnits[idx] + if txFeeInfo == nil { + reportNilFeeInfo(exec.slotCtx, txs[idx], slot) + } + exec.txFeeAccumulator.Add(txFeeInfo) + } + exec.slotCtx.TotalComputeUnitsConsumed = exec.totalCU + exec.groups++ + metrics.GlobalBlockReplay.StreamingExecution.Groups++ + metrics.GlobalBlockReplay.StreamingExecution.Transactions += uint64(len(txs)) + return nil +} + +// loadTransactionAccounts resolves the group's address-table lookups and adds +// the pristine parent image of every account the group can touch to the +// parent snapshot, exactly as the whole-block loader does for a block, except +// that accounts already present keep their earlier image: the batch read at +// block.Slot through the same source returns parent state regardless of the +// overlay, so the first image is the right one and later groups must not +// replace it. +func (exec *blockExecution) loadTransactionAccounts(view *b.Block) error { + phaseStart := time.Now() + if err := resolveAddrTableLookups(exec.blockSrc, view); err != nil { + return fmt.Errorf("resolve address table lookups at slot %d: %w", view.Slot, err) + } + metrics.GlobalBlockReplay.AccountLoader.AddressTableLookups.AddTimingSince(phaseStart) + + phaseStart = time.Now() + dedupedAccts, _ := extractAndDedupeBlockAccts(view) + if exec.parentAccts.Map != nil { + filtered := dedupedAccts[:0] + for _, key := range dedupedAccts { + if _, loaded := exec.parentAccts.Map[key]; !loaded { + filtered = append(filtered, key) + } + } + dedupedAccts = filtered + } + metrics.GlobalBlockReplay.AccountLoader.DedupeBlockAccounts.AddTimingSince(phaseStart) + if len(dedupedAccts) == 0 { + return nil + } + + phaseStart = time.Now() + slotAccts, batchStats, err := getAccountsBatchSharedWithStats(context.Background(), exec.blockSrc, view.Slot, dedupedAccts) + metrics.GlobalBlockReplay.AccountLoader.SourceBatch.AddTimingSince(phaseStart) + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, batchStats) + if err != nil { + return fmt.Errorf("load transaction accounts at slot %d: %w", view.Slot, err) + } + if exec.parentAccts.Map == nil { + return fmt.Errorf("load transaction accounts at slot %d: parent snapshot is not a memory account set", view.Slot) + } + phaseStart = time.Now() + for _, acct := range slotAccts { + if acct == nil { + continue + } + if _, loaded := exec.parentAccts.Map[acct.Key]; loaded { + continue + } + key := [32]byte(acct.Key) + if err := exec.parentAccts.SetAccount(&key, acct); err != nil { + return err + } + } + metrics.GlobalBlockReplay.AccountLoader.ParentAccounts += uint64(len(slotAccts)) + metrics.GlobalBlockReplay.AccountLoader.ParentMapBuild.AddTimingSince(phaseStart) + return nil +} + +// runTransactionGroup is parallelTxLoop over a transaction slice with a plan +// built for the group alone (indices are group-local). Without the planner +// (txParallelism == 0, or an unresolvable lookup) it runs sequentially, which +// is always correct because groups are consumed in block order. +func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execute []bool, shouldVerifySignatures bool) ([]*fees.TxFeeInfo, []uint64, error) { + slotCtx := exec.slotCtx + feeInfos := make([]*fees.TxFeeInfo, len(txs)) + computeUnits := make([]uint64, len(txs)) + dbgOpts := exec.dbgOpts + + workers := exec.txParallelism + if workers > len(txs) { + workers = len(txs) + } + var plan *dependencyPlan + if workers > 1 { + plannerBuildStart := time.Now() + plannerAccounts, available := plannerAccountsForBlock(&b.Block{Transactions: txs}) + if available { + plan = buildDependencyPlan(plannerAccounts) + } + metrics.GlobalBlockReplay.DependencyPlannerBuild.AddTimingSince(plannerBuildStart) + } + if plan == nil { + for idx, tx := range txs { + if !execute[idx] { + continue + } + feeInfos[idx], computeUnits[idx], _ = ProcessTransaction(slotCtx, &exec.sigverifyWg, tx, nil, dbgOpts, nil, shouldVerifySignatures) + } + return feeInfos, computeUnits, nil + } + + metrics.GlobalBlockReplay.DependencyPlannerPrepared = 1 + do := make(chan int, len(txs)) + done := make(chan int, len(txs)) + plannerDone := make(chan struct{}) + go func() { + defer close(plannerDone) + plannerDispatchStart := time.Now() + dispatchDependencyPlan(plan, do, done) + metrics.GlobalBlockReplay.DependencyPlannerDispatch.AddTimingSince(plannerDispatchStart) + }() + + wg := &sync.WaitGroup{} + wg.Add(workers) + for i := 0; i < workers; i++ { + go func(workerIdx int) { + defer wg.Done() + var workerArena *arena.Arena[sealevel.BorrowedAccount] + if workerIdx < len(sealevel.BorrowedAccountArenas) { + workerArena = sealevel.BorrowedAccountArenas[workerIdx] + } + for idx := range do { + if !execute[idx] { + done <- idx + continue + } + feeInfos[idx], computeUnits[idx], _ = ProcessTransaction(slotCtx, &exec.sigverifyWg, txs[idx], nil, dbgOpts, workerArena, shouldVerifySignatures) + done <- idx + } + }(i) + } + wg.Wait() + close(done) + <-plannerDone + return feeInfos, computeUnits, nil +} + +// reportNilFeeInfo reproduces ProcessBlock's diagnostic for a transaction whose +// fee information is missing, which only happens when blockhash validation +// failed for a transaction the block claims to have processed. +func reportNilFeeInfo(slotCtx *sealevel.SlotCtx, tx *solana.Transaction, slot uint64) { + var recentBlockhashes sealevel.SysvarRecentBlockhashes + if bankSysvars := slotCtx.BankSysvars(); bankSysvars != nil { + recentBlockhashes, _ = bankSysvars.RecentBlockhashes() + } + mlog.Log.Errorf("txFeeInfo is nil for tx %s in slot %d", tx.Signatures[0], slot) + mlog.Log.Errorf(" tx blockhash: %s", tx.Message.RecentBlockhash) + mlog.Log.Errorf(" LatestEvictedBlockhash: %x", slotCtx.LatestEvictedBlockhash[:8]) + if len(recentBlockhashes) > 0 { + mlog.Log.Errorf(" RecentBlockhashes: %d entries, newest=%x, oldest=%x", + len(recentBlockhashes), recentBlockhashes[0].Blockhash[:8], recentBlockhashes[len(recentBlockhashes)-1].Blockhash[:8]) + } else { + mlog.Log.Errorf(" RecentBlockhashes: nil or empty!") + } + panic(fmt.Sprintf("txFeeInfo is nil - blockhash validation failed for tx %s", tx.Signatures[0])) +} + +// finalize runs the end-of-block phases over the open SlotCtx: fees to the +// leader, rent, incinerator, the Alpenglow footer clock and vote rewards, bank +// sysvar finalization, bank hash, footer verification, state publication and +// transaction status commit. It is the unchanged tail of ProcessBlock and is +// shared by streaming execution, which calls it once the complete block has +// been matched against the executed prefix. +func (exec *blockExecution) finalize() (*sealevel.SlotCtx, error) { + block := exec.block + slotCtx := exec.slotCtx + acctsDb := exec.acctsDb + tail := exec.tail + setReplayStage := exec.setReplayStage + alpenglowClock := exec.alpenglowClock + bankEpochSchedule := exec.bankEpochSchedule + txFeeAccumulator := exec.txFeeAccumulator + executionPlan := exec.executionPlan + transactionStatuses := exec.transactionStatuses + persistedHashes := exec.persistedHashes + var err error + + start := time.Now() + setReplayStage("distribute_fees") + + // distribute tx fees to the slot leader + // skip leader handling if there are zero transactions in this block + if !global.ManageLeaderSchedule() && block.BlockReward != nil && len(block.Transactions) > 0 { + slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.BlockReward.Leader, &txFeeAccumulator) + slotCtx.RecordModifiedAcct(block.BlockReward.Leader) + } else if global.ManageLeaderSchedule() && len(block.Transactions) > 0 { + slotCtx.LamportsBurnt = fees.DistributeTxFeesToSlotLeader(acctsDb, slotCtx, block.Leader, &txFeeAccumulator) + slotCtx.RecordModifiedAcct(block.Leader) + } + metrics.GlobalBlockReplay.Reward.AddTimingSince(start) + + start = time.Now() + setReplayStage("collect_rent") + bankRent, ok := slotCtx.BankSysvars().Rent() + if !ok { + return nil, fmt.Errorf("bank-local Rent sysvar unavailable at slot %d", block.Slot) + } + rentAccts := rent.CollectRentEagerly(slotCtx, &bankRent, bankEpochSchedule) + metrics.GlobalBlockReplay.Rent.AddTimingSince(start) + + start = time.Now() + setReplayStage("run_incinerator") + runIncinerator(slotCtx) + metrics.GlobalBlockReplay.RunIncinerator.AddTimingSince(start) + + // Alpenglow banks set the Clock timestamp from the block footer after execution. + if alpenglowClock { + footerClockStart := time.Now() + if err := applyAlpenglowFooterClock(slotCtx, block, bankEpochSchedule); err != nil { + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + return nil, fmt.Errorf("apply alpenglow footer clock at slot %d: %w", block.Slot, err) + } + if err := updateAlpenglowNanosecondClockAccount(slotCtx, block); err != nil { + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + return nil, err + } + metrics.GlobalBlockReplay.AlpenglowFooterClock.AddTimingSince(footerClockStart) + voteRewardsStart := time.Now() + voteRewardsErr := ApplyAlpenglowVoteRewards(slotCtx, block, bankEpochSchedule, block.SkipRewardCert, block.NotarRewardCert, block.BlockFinalCert, block.AlpenglowShredVersion) + metrics.GlobalBlockReplay.AlpenglowVoteRewards.AddTimingSince(voteRewardsStart) + if voteRewardsErr != nil { + return nil, voteRewardsErr + } + } + if err := finalizeBankSysvars(slotCtx); err != nil { + return nil, fmt.Errorf("finalize bank sysvars at slot %d: %w", block.Slot, err) + } + + setReplayStage("compile_accounts") + start = time.Now() + writableAccts, modifiedAccts := compileWritableAndModifiedAccts(slotCtx, block, rentAccts) + metrics.GlobalBlockReplay.CompileWritableAndModifiedAccts.AddTimingSince(start) + start = time.Now() + ensureParentsErr := ensureParentAccountsForModified(slotCtx, modifiedAccts) + metrics.GlobalBlockReplay.EnsureParentAccountsForModified.AddTimingSince(start) + if ensureParentsErr != nil { + return nil, ensureParentsErr + } + + start = time.Now() + setReplayStage("bankhash") + slotCtx.FinalBankhash = bankhash.CalculateBankHash(slotCtx, writableAccts, modifiedAccts, block.ParentBankhash, slotCtx.NumSignatures, block.Blockhash) + metrics.GlobalBlockReplay.BankHash.AddTimingSince(start) + if alpenglowClock { + footerVerificationStart := time.Now() + footerVerificationErr := verifyAlpenglowBlockFooter(slotCtx, block, alpenglowClock) + metrics.GlobalBlockReplay.AlpenglowFooterVerification.AddTimingSince(footerVerificationStart) + if footerVerificationErr != nil { + writeFooterBankhashMismatchArtifact(footerVerificationErr, block, slotCtx, writableAccts, modifiedAccts) + return nil, footerVerificationErr + } + } + + // Bankhash consensus enforcement is handled in the replay loop (not here) + // because forkchoice is fed after ProcessBlock returns — checking here would + // never see votes from recently submitted blocks and could deadlock. + + // Enter critical commit window - panics here may leave AccountsDB inconsistent + commitSlot.Store(slotCtx.Slot) + commitInProgress.Store(true) + blockUpdateStart := time.Now() + setReplayStage("store_accounts") + persistedSlot := slotCtx.Slot + persistedBankhash := append([]byte(nil), slotCtx.FinalBankhash...) + persistedBlockSlot := block.Slot + stakeIndexDir := filepath.Join(acctsDb.AcctsDir, "..") + afterStoreAccounts := func() { + if tail != nil { + // Rooted-durable: accounts + bankhash are buffered in the overlay and + // become durable only on promotion; nothing written here (rooted-only). + } else { + if berr := acctsDb.StoreBankHashForSlot(persistedSlot, persistedBankhash); berr != nil { + mlog.Log.Infof("unable to store bankhash for slot %d", persistedSlot) + } + } + if tail == nil { + // Legacy/verify modes (no fork ambiguity): flush per block as before. + // Rooted-durable replay flushes at FOLD time instead — entries stay + // slot-scoped in RAM so a fork unwind can drop them, and scans merge + // the pending set (StreamStakeAccounts) for completeness meanwhile. + flushed, err := global.FlushPendingStakePubkeys(stakeIndexDir) + if err != nil { + mlog.Log.Errorf("failed to flush stake pubkey index: %v", err) + } else if flushed > 0 { + mlog.Log.Debugf("flushed %d new stake pubkeys to index", flushed) + } + } + + persistedHashes.Set(persistedBlockSlot, persistedBankhash) + + // Exit critical commit window - AccountsDB is now consistent + commitInProgress.Store(false) + commitSlot.Store(0) + } + + if tail != nil { + // Rooted-durable: buffer this slot's writes + bankhash in the RAM overlay + // (always, even when empty, so the bankhash is recorded); no durable write. + tail.Add(slotCtx.Slot, modifiedAccts, persistedBankhash) + afterStoreAccounts() + } else if len(modifiedAccts) > 0 { + err = acctsDb.StoreAccounts(modifiedAccts, slotCtx.Slot, afterStoreAccounts) + } + // In rooted-durable mode the callback above is synchronous, so this includes + // the complete critical-path overlay publication. Legacy StoreAccounts only + // enqueues here; its asynchronous disk work deliberately belongs to no slot's + // replay wall time and must never update a later slot's metrics record. + metrics.GlobalBlockReplay.BlockUpdateAccounts.AddTimingSince(blockUpdateStart) + if err != nil { + return slotCtx, err + } + statusCommitStart := time.Now() + statusWaitStart := time.Now() + preparedStatuses := exec.statusPreparation.wait() + metrics.GlobalBlockReplay.TransactionStatusPreparationWait.AddTimingSince(statusWaitStart) + statusErr := transactionStatuses.commitBlockWithValidation(block, executionPlan, preparedStatuses, exec.statusValidation) + metrics.GlobalBlockReplay.TransactionStatusCommit.AddTimingSince(statusCommitStart) + if statusErr != nil { + return nil, fmt.Errorf("commit transaction statuses for slot %d after bank state commit: %w", block.Slot, statusErr) + } + + global.IncrTransactionCount(executionPlan.processedTxCount) + setReplayStage("done") + return slotCtx, err +} diff --git a/pkg/replay/block_execution_test.go b/pkg/replay/block_execution_test.go new file mode 100644 index 000000000..93ca05a23 --- /dev/null +++ b/pkg/replay/block_execution_test.go @@ -0,0 +1,304 @@ +package replay + +import ( + "context" + "errors" + "math" + "math/rand" + "sort" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// memBlockSource serves a fixed parent state to the group loader the way the +// batch loader sees AccountsDB/the unrooted tail: a missing key yields a +// zero-lamport placeholder, never an error. +type memBlockSource struct { + mem accounts.MemAccounts +} + +func (s *memBlockSource) GetAccount(_ uint64, pubkey solana.PublicKey) (*accounts.Account, error) { + if acct, err := s.mem.GetAccountWithoutLock(pubkey); err == nil { + return acct.Clone(), nil + } + return &accounts.Account{Key: pubkey}, nil +} + +func (s *memBlockSource) GetAccountsBatch(_ context.Context, slot uint64, pks []solana.PublicKey) ([]*accounts.Account, error) { + out := make([]*accounts.Account, len(pks)) + for i, pk := range pks { + out[i], _ = s.GetAccount(slot, pk) + } + return out, nil +} + +// groupExecutionEnv is a block execution over a MemAccounts parent snapshot +// and overlay, mirroring what ProcessBlock builds through the account loader, +// with the process-wide sysvar cache set the way newCommitTestSlotCtx sets it. +type groupExecutionEnv struct { + exec *blockExecution + parent accounts.MemAccounts + source *memBlockSource + cleanup func() +} + +func newGroupExecutionEnv(t *testing.T, txParallelism int, payerLamports uint64) *groupExecutionEnv { + t.Helper() + feats := features.NewFeaturesDefault() + feats.EnableFeature(features.FormalizeLoadedTransactionDataSize, 0) + + durable := accounts.NewMemAccounts() + _ = durable.SetAccountWithoutLock(addresses.SystemProgramAddr, &accounts.Account{ + Key: addresses.SystemProgramAddr, Lamports: 1, Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: math.MaxUint64, + }) + _ = durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: payerLamports, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + }) + _ = durable.SetAccountWithoutLock(txfixture.DestPubkey(), &accounts.Account{ + Key: txfixture.DestPubkey(), Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: math.MaxUint64, + }) + + prevRBH := sealevel.SysvarCache.RecentBlockHashes.Sysvar + rbh := sealevel.SysvarRecentBlockhashes{{Blockhash: txfixture.TestBlockhash(), FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5000}}} + sealevel.SysvarCache.RecentBlockHashes.Sysvar = &rbh + prevRent := sealevel.SysvarCache.Rent.Sysvar + rentSysvar := sealevel.NewDefaultRentSysvar() + sealevel.SysvarCache.Rent.Sysvar = &rentSysvar + + parent := accounts.NewMemAccounts() + overlay := accounts.NewOverlayAccounts(parent) + block := &b.Block{Slot: 42, Features: feats} + slotCtx := &sealevel.SlotCtx{ + Accounts: overlay, + ParentAccts: parent, + Slot: block.Slot, + Features: feats, + FeeRateGovernor: &sealevel.FeeRateGovernor{PrevLamportsPerSignature: 5000}, + LastBlockhash: txfixture.TestBlockhash(), + AcctMapsMu: &sync.Mutex{}, + ModifiedAccts: make(map[solana.PublicKey]bool), + WritableAccts: make(map[solana.PublicKey]bool), + VoteTimestampMu: &sync.Mutex{}, + VoteTimestamps: make(map[solana.PublicKey]sealevel.BlockTimestamp), + Replay: true, + } + source := &memBlockSource{mem: durable} + exec := &blockExecution{ + block: block, + txParallelism: txParallelism, + blockSrc: source, + slotCtx: slotCtx, + parentAccts: parent, + accts: overlay, + ctx: context.Background(), + setReplayStage: func(string) {}, + seenMessages: make(map[[32]byte]int), + } + return &groupExecutionEnv{ + exec: exec, + parent: parent, + source: source, + cleanup: func() { + sealevel.SysvarCache.RecentBlockHashes.Sysvar = prevRBH + sealevel.SysvarCache.Rent.Sysvar = prevRent + }, + } +} + +func (env *groupExecutionEnv) lamports(t *testing.T, key solana.PublicKey) uint64 { + t.Helper() + acct, err := env.exec.slotCtx.GetAccountShared(key) + require.NoError(t, err) + return acct.Lamports +} + +func (env *groupExecutionEnv) modifiedKeys() []string { + keys := make([]string, 0, len(env.exec.slotCtx.ModifiedAccts)) + for key := range env.exec.slotCtx.ModifiedAccts { + keys = append(keys, key.String()) + } + sort.Strings(keys) + return keys +} + +func transferTransactions(t *testing.T, n int, firstSeq uint64) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, n) + for i := range txs { + tx, err := solana.TransactionFromBytes(txfixture.MustSignedTransferWire(firstSeq + uint64(i))) + require.NoError(t, err) + txs[i] = tx + } + return txs +} + +type groupExecutionOutcome struct { + payer, dest uint64 + fees uint64 + cu uint64 + processed, sigs uint64 + executeMask []bool + modified []string + parentPayerLamports uint64 +} + +func runGroups(t *testing.T, txParallelism int, payerLamports uint64, txs []*solana.Transaction, splits []int) groupExecutionOutcome { + t.Helper() + env := newGroupExecutionEnv(t, txParallelism, payerLamports) + defer env.cleanup() + start := 0 + for _, end := range append(append([]int(nil), splits...), len(txs)) { + if end < start { + end = start + } + require.NoError(t, env.exec.executeTransactionGroup(txs[start:end], nil, false)) + start = end + } + parentPayer, err := env.parent.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + return groupExecutionOutcome{ + payer: env.lamports(t, txfixture.PayerPubkey()), + dest: env.lamports(t, txfixture.DestPubkey()), + fees: env.exec.txFeeAccumulator.TotalFees, + cu: env.exec.totalCU, + processed: env.exec.processedTxCount, + sigs: env.exec.processedSignatures, + executeMask: append([]bool(nil), env.exec.execute...), + modified: env.modifiedKeys(), + parentPayerLamports: parentPayer.Lamports, + } +} + +// The reference is the pre-existing sequential path: ProcessTransaction over +// the block order on a slot context whose accounts were preloaded whole. +func runSequentialReference(t *testing.T, payerLamports uint64, txs []*solana.Transaction) groupExecutionOutcome { + t.Helper() + env := newGroupExecutionEnv(t, 0, payerLamports) + defer env.cleanup() + keys := []solana.PublicKey{addresses.SystemProgramAddr, txfixture.PayerPubkey(), txfixture.DestPubkey()} + for _, key := range keys { + acct, _ := env.source.GetAccount(42, key) + pk := [32]byte(key) + require.NoError(t, env.parent.SetAccount(&pk, acct)) + } + var sigverify sync.WaitGroup + var out groupExecutionOutcome + for _, tx := range txs { + feeInfo, cu, _ := ProcessTransaction(env.exec.slotCtx, &sigverify, tx, nil, nil, nil, false) + require.NotNil(t, feeInfo) + out.fees += feeInfo.TotalFee + out.cu += cu + out.processed++ + out.sigs += uint64(tx.Message.Header.NumRequiredSignatures) + out.executeMask = append(out.executeMask, true) + } + sigverify.Wait() + out.payer = env.lamports(t, txfixture.PayerPubkey()) + out.dest = env.lamports(t, txfixture.DestPubkey()) + out.modified = env.modifiedKeys() + out.parentPayerLamports = payerLamports + return out +} + +// TestExecuteTransactionGroupMatchesWholeBlock runs the same block-ordered +// transfers as one group, as random groups, and through the sequential +// reference. Every transfer shares the payer and destination, so every group +// boundary is a cross-group write dependency, and the payer balance is small +// enough that later transfers fail for insufficient funds, which exercises +// fee charging on failed transactions and makes outcomes order-dependent. +func TestExecuteTransactionGroupMatchesWholeBlock(t *testing.T) { + // Amounts are 999,001+ lamports each (seq%1e6+1), so with a 5 SOL-ish + // payer of 5,000,000 lamports the first few transfers succeed and the + // rest fail inside the System program while still paying their fee. + const payerLamports = 5_000_000 + txs := transferTransactions(t, 48, 999_000) + reference := runSequentialReference(t, payerLamports, txs) + require.Less(t, reference.payer, uint64(payerLamports)) + + for _, txParallelism := range []int{0, 1, 4} { + single := runGroups(t, txParallelism, payerLamports, txs, nil) + require.Equal(t, reference.payer, single.payer, "txpar %d single group payer", txParallelism) + require.Equal(t, reference.dest, single.dest, "txpar %d single group dest", txParallelism) + require.Equal(t, reference.fees, single.fees) + require.Equal(t, reference.cu, single.cu) + require.Equal(t, reference.processed, single.processed) + require.Equal(t, reference.sigs, single.sigs) + require.Equal(t, reference.executeMask, single.executeMask) + require.Equal(t, reference.modified, single.modified) + require.Equal(t, uint64(payerLamports), single.parentPayerLamports, "parent image must stay pristine") + + rng := rand.New(rand.NewSource(int64(7 + txParallelism))) + for iter := 0; iter < 8; iter++ { + splitCount := 1 + rng.Intn(6) + splits := make([]int, splitCount) + for i := range splits { + splits[i] = rng.Intn(len(txs) + 1) + } + sort.Ints(splits) + grouped := runGroups(t, txParallelism, payerLamports, txs, splits) + require.Equal(t, single, grouped, "txpar %d splits %v", txParallelism, splits) + } + } +} + +func TestExecuteTransactionGroupRejectsDuplicatesAcrossGroups(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + txs := transferTransactions(t, 3, 100) + require.NoError(t, env.exec.executeTransactionGroup(txs[:2], nil, false)) + err := env.exec.executeTransactionGroup([]*solana.Transaction{txs[2], txs[0]}, nil, false) + var duplicateErr *DuplicateTransactionMessagesError + require.Error(t, err) + require.True(t, errors.As(err, &duplicateErr)) + require.Equal(t, uint64(42), duplicateErr.Slot) + require.Equal(t, uint64(1), duplicateErr.DuplicateCount) + require.Equal(t, []DuplicateTransactionOccurrence{{Index: 3, FirstIndex: 0}}, duplicateErr.Occurrences) + // The rejected group must not have been recorded or executed. + require.Len(t, env.exec.transactions, 2) + require.Equal(t, uint64(2), env.exec.processedTxCount) +} + +func TestExecuteTransactionGroupRejectsV1BeforeActivation(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + tx, err := solana.TransactionFromBytes(txfixture.MustSignedV1Wire(9, 8)) + require.NoError(t, err) + err = env.exec.executeTransactionGroup([]*solana.Transaction{tx}, nil, false) + require.ErrorIs(t, err, TxErrUnsupportedVersion) + require.Empty(t, env.exec.transactions) +} + +func TestExecuteTransactionGroupKeepsFirstParentImage(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + txs := transferTransactions(t, 4, 200) + require.NoError(t, env.exec.executeTransactionGroup(txs[:2], nil, false)) + afterFirst := env.lamports(t, txfixture.PayerPubkey()) + require.Less(t, afterFirst, uint64(10_000_000_000)) + // A later group touching the same accounts must not reload the payer's + // parent image over the pristine one, nor see stale overlay state. + require.NoError(t, env.exec.executeTransactionGroup(txs[2:], nil, false)) + parentPayer, err := env.parent.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000), parentPayer.Lamports) + require.Less(t, env.lamports(t, txfixture.PayerPubkey()), afterFirst) + require.Equal(t, 2, env.exec.groups) + require.Len(t, env.exec.transactions, 4) +} + +func TestExecuteTransactionGroupRefusesClosedExecution(t *testing.T) { + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + defer env.cleanup() + env.exec.closed = true + err := env.exec.executeTransactionGroup(transferTransactions(t, 1, 300), nil, false) + require.ErrorIs(t, err, errBlockExecutionClosed) +} diff --git a/pkg/replay/transaction_status_cache.go b/pkg/replay/transaction_status_cache.go index d17110491..85068b084 100644 --- a/pkg/replay/transaction_status_cache.go +++ b/pkg/replay/transaction_status_cache.go @@ -581,6 +581,25 @@ func (c *TransactionStatusCache) validateBlockForPublication(block *b.Block, pla return transactionStatusValidation{cache: c, identities: plan.messageIdentities, version: c.validationVersion}, nil } +// validateTransactionsAgainstAncestors is the per-group form of the ancestor +// already-processed check, for execution that starts before the complete +// block exists. It does not validate the parent link; the complete block is +// validated again in full, with validateBlockForPublication, before commit. +func (c *TransactionStatusCache) validateTransactionsAgainstAncestors(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { + if identities == nil { + return errors.New("nil transaction message identities") + } + if c == nil { + return &IncompleteTransactionStatusCoverageError{} + } + c.mu.RLock() + defer c.mu.RUnlock() + if !c.coverageComplete { + return &IncompleteTransactionStatusCoverageError{CachedRoot: c.rootedThrough} + } + return c.validateAncestorTransactionsLocked(slot, identities) +} + func (c *TransactionStatusCache) validateAncestorTransactionsLocked(slot uint64, identities *b.PreparedTransactionMessageIdentities) error { var already *AncestorAlreadyProcessedTransactionMessagesError for index := 0; index < identities.Len(); index++ { From 400dc659ea90c79bce9e7913d1235e8fd75781b5 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 06:40:38 +0000 Subject: [PATCH 040/111] turbine: streaming feed of decoded batches for replay The entry prefetch already decodes and signature-verifies every closed DATA_COMPLETE range while a slot's shreds arrive. Expose those batches to one subscriber as immutable views (StreamBatch: slot, opaque generation, range, marker kind with parent slot/ID for header and UpdateParent, the batch's transactions, and WaitVerification returning the verifier's raw message identities) through StreamEvents: BatchReady when a batch is decoded, Completed when the generation's block was accepted by finalizeCompletion, Cancelled with a reason from every other release path (reset, eviction, retention sweep, non-canonical identity, shutdown). Events are wake-ups only: a full subscriber drops them and counts the drop; StreamStatusOf and PendingStreamBatches read the assembler's own state under its lock and are the authoritative recovery path. A generation is the slotState identity, so a re-assembled slot is a new generation. The transactions in a view are the very objects the complete block will reference when completion reuses the batch (byte-identical shreds), which is what lets a streaming consumer prove an executed prefix by pointer identity. The last component of a slot is decoded by completion, never by the prefetch, so it never streams. Nothing about completion, identity attachment or emission changes. BlockSource gains TurbineStreamingExecution (off by default) which sizes the feed channel and subscribes each active receiver, plus StreamEvents, StreamStatusOf, PendingStreamBatches and PrioritizeStreamRepair for the replay side. pkg/block gains PrepareVerifiedTransactionMessageIdentities (now used by CacheVerifiedTransactionMessageIdentities) and PreparedTransactionMessageIdentities.Slice. NewDetachedStreamGeneration, NewDetachedStreamBatch and NewDetachedStreamMarker let a consumer unit-test its state machine against a fake feed. Tests: batch and completion events with a header marker, verification identities bound to the batch's transactions, pointer identity with the completed block; cancellation on reset with a renewed generation; dropped wake-ups under a full subscriber. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/block/message_identity.go | 13 + pkg/block/verified_message_identity.go | 38 ++- pkg/blockstream/block_source.go | 33 ++- pkg/blockstream/turbine_stream.go | 55 +++++ pkg/turbine/assembler.go | 18 ++ pkg/turbine/entry_prefetch.go | 11 + pkg/turbine/receiver.go | 18 ++ pkg/turbine/stream.go | 316 +++++++++++++++++++++++++ pkg/turbine/stream_test.go | 174 ++++++++++++++ 9 files changed, 654 insertions(+), 22 deletions(-) create mode 100644 pkg/turbine/stream.go create mode 100644 pkg/turbine/stream_test.go diff --git a/pkg/block/message_identity.go b/pkg/block/message_identity.go index 067ac4dec..df5da3a5a 100644 --- a/pkg/block/message_identity.go +++ b/pkg/block/message_identity.go @@ -34,3 +34,16 @@ func (prepared *PreparedTransactionMessageIdentities) Identity(index int) txstat func (prepared *PreparedTransactionMessageIdentities) MatchesBlock(block *Block) bool { return block != nil && prepared.matches(block.Transactions) } + +// Slice returns the prepared identities for transactions [from, to) as an +// independent prepared set bound to that sub-slice. +func (prepared *PreparedTransactionMessageIdentities) Slice(from, to int) *PreparedTransactionMessageIdentities { + if prepared == nil || from < 0 || to > len(prepared.identities) || from > to { + return nil + } + return &PreparedTransactionMessageIdentities{ + transactions: prepared.transactions[from:to:to], + versions: prepared.versions[from:to:to], + identities: prepared.identities[from:to:to], + } +} diff --git a/pkg/block/verified_message_identity.go b/pkg/block/verified_message_identity.go index 9eff1d264..001067513 100644 --- a/pkg/block/verified_message_identity.go +++ b/pkg/block/verified_message_identity.go @@ -8,27 +8,45 @@ import ( "github.com/gagliardetto/solana-go" ) -// CacheVerifiedTransactionMessageIdentities publishes identities from joined -// signature-verification requests. Every result must cover the exact ordered -// transaction slice. Failed/partial requests and obsolete prefetch generations -// cannot seed the cache. It does not replace block-wide duplicate/status checks. -func (b *Block) CacheVerifiedTransactionMessageIdentities(identities []txverify.VerifiedMessageIdentity) error { - if b == nil || len(identities) != len(b.Transactions) { - return fmt.Errorf("verified message identities do not cover block transactions") +// PrepareVerifiedTransactionMessageIdentities binds identities from joined +// signature-verification requests to the exact ordered transaction slice they +// were computed for. Every identity must be a verified result for the +// transaction at the same index; failed/partial requests cannot seed a set. +func PrepareVerifiedTransactionMessageIdentities(transactions []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) (*PreparedTransactionMessageIdentities, error) { + if len(identities) != len(transactions) { + return nil, fmt.Errorf("verified message identities do not cover transactions") } prepared := &PreparedTransactionMessageIdentities{ - transactions: append([]*solana.Transaction(nil), b.Transactions...), + transactions: append([]*solana.Transaction(nil), transactions...), versions: make([]solana.MessageVersion, len(identities)), identities: make([]txstatus.TransactionMessageIdentity, len(identities)), } - for i, tx := range b.Transactions { + for i, tx := range transactions { identity, ok := identities[i].ForTransaction(tx) if !ok { - return fmt.Errorf("verified message identity does not match transaction %d", i) + return nil, fmt.Errorf("verified message identity does not match transaction %d", i) } prepared.versions[i] = tx.Message.GetVersion() prepared.identities[i] = identity } + return prepared, nil +} + +// CacheVerifiedTransactionMessageIdentities publishes identities from joined +// signature-verification requests. Every result must cover the exact ordered +// transaction slice. Failed/partial requests and obsolete prefetch generations +// cannot seed the cache. It does not replace block-wide duplicate/status checks. +func (b *Block) CacheVerifiedTransactionMessageIdentities(identities []txverify.VerifiedMessageIdentity) error { + if b == nil { + return fmt.Errorf("verified message identities do not cover block transactions") + } + prepared, err := PrepareVerifiedTransactionMessageIdentities(b.Transactions, identities) + if err != nil { + if len(identities) != len(b.Transactions) { + return fmt.Errorf("verified message identities do not cover block transactions") + } + return err + } state := b.transactionState() state.mu.Lock() defer state.mu.Unlock() diff --git a/pkg/blockstream/block_source.go b/pkg/blockstream/block_source.go index 44991cbe3..e81416f14 100644 --- a/pkg/blockstream/block_source.go +++ b/pkg/blockstream/block_source.go @@ -47,18 +47,21 @@ type BlockSourceOpts struct { // Enables Alpenglow/Votor block-id hints for the Turbine assembler. Classic // Solana clusters leave this off even when blocks are sourced from Turbine. TurbineAlpenglowBlockIDHints bool - TurbineIdentity ed25519.PrivateKey - LeaderForSlot func(slot uint64) (solana.PublicKey, bool) - TurbineStakesForSlot func(slot uint64) map[solana.PublicKey]uint64 - TurbineEpochForSlot func(slot uint64) uint64 - TurbineRootSlot func() uint64 - TurbineUseChaCha8 bool - TurbineDedupAddrs bool - LocalLeaderForSlot func(slot uint64) bool - GossipClient *gossip.Client - AlpenglowDecisionSource func(anchorSlot uint64) (alpenglow.ChainDecision, bool) - AlpenglowCandidateBlockSink func(alpenglow.ReplayBlockObservation) - AlpenglowInvalidBlockSink func(alpenglow.BlockID, string) error + // TurbineStreamingExecution subscribes replay's streaming executor to the + // assembler's decoded-batch feed (StreamEvents). Off by default. + TurbineStreamingExecution bool + TurbineIdentity ed25519.PrivateKey + LeaderForSlot func(slot uint64) (solana.PublicKey, bool) + TurbineStakesForSlot func(slot uint64) map[solana.PublicKey]uint64 + TurbineEpochForSlot func(slot uint64) uint64 + TurbineRootSlot func() uint64 + TurbineUseChaCha8 bool + TurbineDedupAddrs bool + LocalLeaderForSlot func(slot uint64) bool + GossipClient *gossip.Client + AlpenglowDecisionSource func(anchorSlot uint64) (alpenglow.ChainDecision, bool) + AlpenglowCandidateBlockSink func(alpenglow.ReplayBlockObservation) + AlpenglowInvalidBlockSink func(alpenglow.BlockID, string) error // AlpenglowCandidateValidator prevents objectively invalid assembled blocks // from polluting the early ancestry tracker. Replay independently validates // again at the consensus boundary before observing or executing the block. @@ -446,6 +449,11 @@ type BlockSource struct { knownAlpenglowBlockIDs map[uint64]solana.Hash knownAlpenglowBlockIDOrder []uint64 activeTurbineReceiver *turbine.UDPReceiver + // streamEvents carries the turbine streaming feed to replay; nil unless + // TurbineStreamingExecution was requested. Sized for several blocks of + // batches; a full channel drops wake-ups, which the consumer tolerates by + // polling PendingStreamBatches. + streamEvents chan turbine.StreamEvent // Repair-first catchup: gap slots [repairCatchupFrom, repairCatchupUntil] // fill via turbine repair; RPC never fetches at/above the gate while // pending or active. The pending hold persists from construction until @@ -765,6 +773,7 @@ func NewBlockSource(opts *BlockSourceOpts) *BlockSource { turbineShredVersion: opts.TurbineShredVersion, turbineAlpenglowAddr: opts.TurbineAlpenglowAddr, turbineAlpenglowBlockIDHints: opts.TurbineAlpenglowBlockIDHints, + streamEvents: newStreamEventChannel(opts), turbineIdentity: clonePrivateKey(opts.TurbineIdentity), leaderForSlot: opts.LeaderForSlot, turbineStakesForSlot: opts.TurbineStakesForSlot, diff --git a/pkg/blockstream/turbine_stream.go b/pkg/blockstream/turbine_stream.go index 1d1228c45..01585d6f0 100644 --- a/pkg/blockstream/turbine_stream.go +++ b/pkg/blockstream/turbine_stream.go @@ -221,6 +221,9 @@ func (bs *BlockSource) attachAlpenglowBlockIDHintsToReceiver(receiver *turbine.U // would then make its slot permanently unfetchable. bs.alpenglowMu.Lock() bs.activeTurbineReceiver = receiver + if bs.streamEvents != nil { + receiver.SubscribeStream(bs.streamEvents) + } if !bs.turbineAlpenglowBlockIDHints { bs.alpenglowMu.Unlock() return @@ -568,3 +571,55 @@ func (bs *BlockSource) runTurbineStream() { } } } + +// streamEventBuffer bounds the streaming feed: a heavy block is a few hundred +// DATA_COMPLETE ranges, and the consumer drains between groups. +const streamEventBuffer = 4096 + +func newStreamEventChannel(opts *BlockSourceOpts) chan turbine.StreamEvent { + if opts == nil || opts.SourceType != BlockSourceTurbine || !opts.TurbineStreamingExecution { + return nil + } + return make(chan turbine.StreamEvent, streamEventBuffer) +} + +// StreamEvents is the turbine streaming feed for replay's streaming executor; +// nil when streaming execution is not enabled for this source. +func (bs *BlockSource) StreamEvents() <-chan turbine.StreamEvent { + if bs.streamEvents == nil { + return nil + } + return bs.streamEvents +} + +// StreamStatusOf reports the assembler's view of a streaming generation +// through the active receiver; a generation is gone when no receiver is +// active. +func (bs *BlockSource) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver == nil { + return turbine.StreamGone + } + return receiver.StreamStatusOf(g) +} + +// PendingStreamBatches returns the generation's decoded batches at or after +// fromStart, in shred-index order, from the active receiver. +func (bs *BlockSource) PendingStreamBatches(g turbine.StreamGeneration, fromStart uint32) []*turbine.StreamBatch { + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver == nil { + return nil + } + return receiver.PendingStreamBatches(g, fromStart) +} + +// PrioritizeStreamRepair keeps a slot that replay is executing while its +// shreds arrive pinned for repair, since the emitter pins the head only when +// it observes a gap. +func (bs *BlockSource) PrioritizeStreamRepair(slot uint64) { + bs.prioritizeTurbineRepairRange(slot, slot) +} diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 0dcff2fa6..2769adf10 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -80,6 +80,10 @@ type SlotAssembler struct { // Production uses the process-wide bounded transaction verifier. verifyTransactions func(context.Context, *block.Block) error entryPrefetch *entryPrefetchPool + // Streaming feed subscriber (see stream.go); nil when nothing consumes + // batches before completion. + streamSubscriber chan<- StreamEvent + streamDroppedEvents uint64 } type SlotRepairRequest struct { @@ -124,6 +128,11 @@ type slotState struct { // flow is usually poisoned state — the latest error names the poison. errCount int lastErr string + // streamCompleted marks a generation whose complete block was accepted, so + // the feed's release event says "completed" rather than "cancelled"; + // streamCancelReason names the discard path otherwise. + streamCompleted bool + streamCancelReason string } func (s *slotState) noteError(err error) { @@ -496,6 +505,7 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p if !a.acceptAlpenglowBlockIDLocked(blk) { a.trackNonCanonicalBlockIDLocked(blk) a.recordPartialObsLocked(state) + state.streamCancelReason = "non_canonical" a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.mu.Unlock() @@ -505,6 +515,7 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p return nil, nil } + state.streamCompleted = true a.releasePrefetchLocked(state) delete(a.slots, state.slot) a.completedSlots[state.slot] = struct{}{} @@ -611,6 +622,9 @@ func (a *SlotAssembler) ResetSlot(slot uint64) { a.retentionDirty = true a.recordPartialObsLocked(a.slots[slot]) + if state := a.slots[slot]; state != nil { + state.streamCancelReason = "reset" + } a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) delete(a.completedSlots, slot) @@ -724,6 +738,9 @@ func (a *SlotAssembler) pruneOldSlotsLocked() { return } a.recordPartialObsLocked(a.slots[victim]) + if state := a.slots[victim]; state != nil { + state.streamCancelReason = "evicted" + } a.releasePrefetchLocked(a.slots[victim]) delete(a.slots, victim) a.evictedSlots++ @@ -739,6 +756,7 @@ func (a *SlotAssembler) sweepRetentionMapsLocked() { for slot, state := range a.slots { if slot < minSlot && !state.completing { a.recordPartialObsLocked(state) + state.streamCancelReason = "retention" a.releasePrefetchLocked(a.slots[slot]) delete(a.slots, slot) a.evictedSlots++ diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go index 2b082e52b..1de831111 100644 --- a/pkg/turbine/entry_prefetch.go +++ b/pkg/turbine/entry_prefetch.go @@ -189,6 +189,9 @@ func (p *entryPrefetchPool) run() { p.a.mu.Lock() f.queued = false close(f.queueDone) + if !f.released && p.a.slots[s.slot] == s { + p.a.publishStreamBatchReadyLocked(s, batch) + } p.enqueueLocked(s) p.a.mu.Unlock() } @@ -219,6 +222,11 @@ func (a *SlotAssembler) releasePrefetchLocked(s *slotState) { f := s.prefetch f.released = true f.cancel() + reason := s.streamCancelReason + if reason == "" { + reason = "released" + } + a.publishStreamReleaseLocked(s, reason) p := f.pool queueDone := f.queueDone p.cleanup.Add(1) @@ -250,6 +258,9 @@ func (p *entryPrefetchPool) closeAndWait() { } for _, s := range p.a.slots { if s.prefetch != nil && s.prefetch.pool == p { + if s.streamCancelReason == "" { + s.streamCancelReason = "shutdown" + } p.a.releasePrefetchLocked(s) } } diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index f5588bf6a..0080cd417 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -389,6 +389,24 @@ func (r *UDPReceiver) PrioritizeRepairRange(start, end uint64) { r.assembler.PrioritizeRepairRange(start, end) } +// SubscribeStream installs the streaming-execution feed subscriber on this +// receiver's assembler (see stream.go). Only one subscriber is supported. +func (r *UDPReceiver) SubscribeStream(ch chan<- StreamEvent) { + r.assembler.SubscribeStream(ch) +} + +// StreamStatusOf reports whether a streaming generation is still the slot's +// current assembly, completed into a block, or gone. +func (r *UDPReceiver) StreamStatusOf(g StreamGeneration) StreamStatus { + return r.assembler.StreamStatusOf(g) +} + +// PendingStreamBatches returns the generation's decoded batches starting at +// or after fromStart, in shred-index order. +func (r *UDPReceiver) PendingStreamBatches(g StreamGeneration, fromStart uint32) []*StreamBatch { + return r.assembler.PendingStreamBatches(g, fromStart) +} + func (r *UDPReceiver) Blocks() <-chan *block.Block { return r.blocks } diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go new file mode 100644 index 000000000..2a6139749 --- /dev/null +++ b/pkg/turbine/stream.go @@ -0,0 +1,316 @@ +package turbine + +import ( + "context" + "errors" + "sort" + "time" + + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// Streaming feed: the assembler already decodes and signature-verifies every +// closed DATA_COMPLETE range of a slot while its shreds arrive (the entry +// prefetch). The feed exposes those batches, in the order they become ready, +// to one subscriber — replay's streaming executor — as immutable views, so the +// block can be executed while the rest of it is still in flight. +// +// The feed is advisory. Events are wake-ups: a full subscriber channel drops +// the event, and the subscriber recovers by asking PendingStreamBatches and +// StreamStatus, which read the assembler's own state under its lock. Nothing +// here changes how a slot completes, how its identity is attached, or how the +// complete block is emitted; the complete block remains the authority. + +// StreamGeneration identifies one assembly of a slot. A slot that is reset +// and assembled again is a different generation. It is opaque: consumers +// compare it for equality and pass it back to the assembler. +type StreamGeneration struct { + slot uint64 + state *slotState +} + +// Slot returns the generation's slot. +func (g StreamGeneration) Slot() uint64 { return g.slot } + +// IsZero reports whether the generation was never set. +func (g StreamGeneration) IsZero() bool { return g.state == nil } + +// NewDetachedStreamGeneration returns a non-zero generation for slot that no +// assembler knows about. It exists so consumers (replay's streaming executor) +// can unit-test their state machines with a fake feed; a real assembler +// reports it as StreamGone. +func NewDetachedStreamGeneration(slot uint64) StreamGeneration { + return StreamGeneration{slot: slot, state: &slotState{slot: slot}} +} + +// NewDetachedStreamBatch builds a ready entry-batch view for consumers' unit +// tests: txs are its transactions and identities, when non-nil, is a +// completed verification result for exactly those transactions (as +// txverify.BatchVerifier.VerifyWithMessageIdentities produces). With nil +// identities the batch reports itself unverified. The assembler never builds +// batches this way. +func NewDetachedStreamBatch(g StreamGeneration, start, end uint32, txs []*solana.Transaction, identities []txverify.VerifiedMessageIdentity) *StreamBatch { + ready := make(chan struct{}) + close(ready) + batch := &prefetchedShredBatch{start: start, end: end, ready: ready} + if identities != nil { + batch.verification = &transactionVerification{done: ready, cancel: func() {}, identities: identities, finishedAt: time.Now()} + } + view := newStreamBatch(g, batch) + view.Transactions = txs + return view +} + +// NewDetachedStreamMarker builds a marker batch view (header, update-parent +// or footer) for consumers' unit tests. +func NewDetachedStreamMarker(g StreamGeneration, start, end uint32, kind StreamMarkerKind, parentSlot uint64, parentBlockID solana.Hash) *StreamBatch { + ready := make(chan struct{}) + close(ready) + view := newStreamBatch(g, &prefetchedShredBatch{start: start, end: end, ready: ready, marker: true}) + view.Marker = kind + view.ParentSlot = parentSlot + view.ParentBlockID = parentBlockID + return view +} + +// StreamMarkerKind classifies a batch that carries an Alpenglow block +// component instead of entries. +type StreamMarkerKind uint8 + +const ( + // StreamMarkerNone is an ordinary entry batch. + StreamMarkerNone StreamMarkerKind = iota + // StreamMarkerHeader is the block header (FEC set 0): parent slot and ID. + StreamMarkerHeader + // StreamMarkerUpdateParent selects an older parent and abandons every + // batch before ReplayFECSetIndex (the optimistic prefix). + StreamMarkerUpdateParent + // StreamMarkerFooter is the block footer (certificates, bank hash, clock). + StreamMarkerFooter +) + +// StreamBatch is an immutable view of one decoded DATA_COMPLETE range. Its +// transactions are the same objects the complete block will reference when +// completion reuses this batch (byte-identical shreds), which is what lets a +// streaming consumer prove its executed prefix is the block by identity. +type StreamBatch struct { + Slot uint64 + Generation StreamGeneration + Start, End uint32 + Marker StreamMarkerKind + // Parent fields are set for header and UpdateParent markers. + ParentSlot uint64 + ParentBlockID solana.Hash + ReplayFECSetIndex uint32 + // Transactions is empty for markers and for batches that failed to decode. + Transactions []*solana.Transaction + // Err is the decode error; a batch with Err makes the whole slot invalid. + Err error + ReadyAt time.Time + + batch *prefetchedShredBatch +} + +// ErrStreamBatchUnverified reports that no signature-verification result is +// attached to the batch (no transactions, or admission was refused); the +// consumer must verify signatures itself. +var ErrStreamBatchUnverified = errors.New("stream batch has no verification result") + +// WaitVerification joins the batch's asynchronous signature verification and +// returns the verifier's message identities, one per transaction, bound to +// Transactions (see block.PrepareVerifiedTransactionMessageIdentities). A +// nil error with verified == false means no result is attached and the +// caller must verify itself; any other error means a signature failed (the +// slot is invalid) or ctx ended. +func (sb *StreamBatch) WaitVerification(ctx context.Context) (identities []txverify.VerifiedMessageIdentity, verified bool, err error) { + if sb == nil || sb.batch == nil { + return nil, false, ErrStreamBatchUnverified + } + if sb.batch.verification == nil { + return nil, false, nil + } + if _, err := sb.batch.verification.waitContext(ctx); err != nil { + return nil, false, err + } + if len(sb.batch.verification.identities) != len(sb.Transactions) { + return nil, false, nil + } + return sb.batch.verification.identities, true, nil +} + +// StreamEventKind is the kind of a feed wake-up. +type StreamEventKind uint8 + +const ( + // StreamBatchReady: Batch was decoded (and its verification submitted). + StreamBatchReady StreamEventKind = iota + // StreamCancelled: the generation's state is gone without a complete + // block (reset, eviction, invalid identity, shutdown). Reason says why. + StreamCancelled + // StreamCompleted: the generation assembled and the complete block is on + // its way through the normal emission path. + StreamCompleted +) + +// StreamEvent is one feed wake-up. +type StreamEvent struct { + Kind StreamEventKind + Slot uint64 + Generation StreamGeneration + Batch *StreamBatch + Reason string +} + +// StreamStatus is the assembler's view of a generation. +type StreamStatus uint8 + +const ( + // StreamActive: the generation is the slot's current assembly. + StreamActive StreamStatus = iota + // StreamDone: the generation completed and its block was (or is being) + // emitted. + StreamDone + // StreamGone: the generation was discarded without a block. + StreamGone +) + +// SubscribeStream installs the single feed subscriber. Events are sent +// without blocking; a full channel drops the event and counts it. +func (a *SlotAssembler) SubscribeStream(ch chan<- StreamEvent) { + a.mu.Lock() + defer a.mu.Unlock() + a.streamSubscriber = ch +} + +// StreamDroppedEvents reports wake-ups dropped because the subscriber was +// full; the subscriber polls PendingStreamBatches after any wake-up, so a +// non-zero count is a sizing hint, not a correctness problem. +func (a *SlotAssembler) StreamDroppedEvents() uint64 { + a.mu.Lock() + defer a.mu.Unlock() + return a.streamDroppedEvents +} + +func (a *SlotAssembler) publishStreamLocked(event StreamEvent) { + if a.streamSubscriber == nil { + return + } + select { + case a.streamSubscriber <- event: + default: + a.streamDroppedEvents++ + } +} + +// StreamStatusOf reports whether a generation is still the slot's current +// assembly, completed into a block, or gone. +func (a *SlotAssembler) StreamStatusOf(g StreamGeneration) StreamStatus { + if g.state == nil { + return StreamGone + } + a.mu.Lock() + defer a.mu.Unlock() + return a.streamStatusLocked(g) +} + +func (a *SlotAssembler) streamStatusLocked(g StreamGeneration) StreamStatus { + if a.slots[g.slot] == g.state { + return StreamActive + } + if g.state.streamCompleted { + return StreamDone + } + return StreamGone +} + +// PendingStreamBatches returns every decoded batch of the generation whose +// range starts at or after fromStart, in shred-index order. It reads the +// prefetch state directly, so it is the authoritative recovery path after a +// dropped wake-up. The result is empty once the generation is no longer +// active (its batches may still be used by a completed block, but the feed +// has nothing more to offer). +func (a *SlotAssembler) PendingStreamBatches(g StreamGeneration, fromStart uint32) []*StreamBatch { + if g.state == nil { + return nil + } + a.mu.Lock() + defer a.mu.Unlock() + if a.streamStatusLocked(g) != StreamActive || g.state.prefetch == nil || g.state.prefetch.released { + return nil + } + var out []*StreamBatch + for start, batch := range g.state.prefetch.batches { + if start < fromStart { + continue + } + select { + case <-batch.ready: + default: + continue + } + out = append(out, newStreamBatch(g, batch)) + } + sort.Slice(out, func(i, j int) bool { return out[i].Start < out[j].Start }) + return out +} + +// newStreamBatch builds the immutable view; it must only be called after the +// batch's ready channel closed (its fields are immutable from then on). +func newStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatch { + view := &StreamBatch{ + Slot: g.slot, + Generation: g, + Start: batch.start, + End: batch.end, + Err: batch.err, + ReadyAt: time.Now(), + batch: batch, + } + switch { + case batch.err != nil: + case batch.marker && batch.parent != nil: + view.ParentSlot = batch.parent.ParentSlot + view.ParentBlockID = batch.parent.ParentBlockID + view.ReplayFECSetIndex = batch.parent.ReplayFECSetIndex + if batch.parent.FromUpdateParent { + view.Marker = StreamMarkerUpdateParent + } else { + view.Marker = StreamMarkerHeader + } + case batch.marker && batch.footer != nil: + view.Marker = StreamMarkerFooter + case batch.marker: + // A marker without decoded content is treated like a footer-less + // component boundary: nothing to execute, nothing to select. + view.Marker = StreamMarkerFooter + default: + view.Transactions = entryBatchTransactions(batch.entries) + } + return view +} + +// publishStreamBatchReady is called by the prefetch worker, under the +// assembler lock, after the batch's ready channel closed. +func (a *SlotAssembler) publishStreamBatchReadyLocked(s *slotState, batch *prefetchedShredBatch) { + if a.streamSubscriber == nil || s == nil || batch == nil { + return + } + g := StreamGeneration{slot: s.slot, state: s} + a.publishStreamLocked(StreamEvent{Kind: StreamBatchReady, Slot: s.slot, Generation: g, Batch: newStreamBatch(g, batch)}) +} + +// publishStreamReleaseLocked is called from releasePrefetchLocked, i.e. from +// every path that drops a slot generation, and tells the subscriber whether a +// complete block follows (finalizeCompletion marked it) or the state is gone. +func (a *SlotAssembler) publishStreamReleaseLocked(s *slotState, reason string) { + if a.streamSubscriber == nil || s == nil { + return + } + g := StreamGeneration{slot: s.slot, state: s} + if s.streamCompleted { + a.publishStreamLocked(StreamEvent{Kind: StreamCompleted, Slot: s.slot, Generation: g}) + return + } + a.publishStreamLocked(StreamEvent{Kind: StreamCancelled, Slot: s.slot, Generation: g, Reason: reason}) +} diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go new file mode 100644 index 000000000..72aea908a --- /dev/null +++ b/pkg/turbine/stream_test.go @@ -0,0 +1,174 @@ +package turbine + +import ( + "context" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +func nextStreamEvent(t *testing.T, ch <-chan StreamEvent, kind StreamEventKind) StreamEvent { + t.Helper() + deadline := time.After(3 * time.Second) + for { + select { + case event := <-ch: + if event.Kind == kind { + return event + } + case <-deadline: + t.Fatalf("no stream event of kind %d", kind) + } + } +} + +// The feed publishes each prefetched batch once it is decoded — the header +// marker with its parent identity, then entry batches whose transactions are +// the very objects the completed block references — and ends with a +// completion event for the same generation. The final component (the ending +// tick) is decoded by completion, never by the prefetch, so it is not fed. +func TestStreamFeedPublishesBatchesAndCompletion(t *testing.T) { + v := newTransactionVerifier(2, 16, func(tx *solana.Transaction) error { return txverify.VerifyTransaction(tx) }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent, 64) + a.SubscribeStream(events) + + const slot = 300 + parentID := solana.Hash{9, 9, 9} + batches := prefetchTestShreds(t, slot, + testAlpenglowParentMarkerBytes(blockMarkerVariantHeader, slot-1, parentID), + prefetchTestPayload(t, verifierSignedTransactions(t, 3)), + prefetchTestPayload(t, verifierSignedTransactions(t, 4)), + buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + header := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, uint64(slot), header.Slot) + require.False(t, header.Generation.IsZero()) + require.Equal(t, StreamMarkerHeader, header.Batch.Marker) + require.Equal(t, uint64(slot-1), header.Batch.ParentSlot) + require.Equal(t, parentID, header.Batch.ParentBlockID) + require.Empty(t, header.Batch.Transactions) + _, verified, err := header.Batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.False(t, verified, "markers carry no verification") + + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + first := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, header.Generation, first.Generation) + require.Equal(t, batches[1][0].Index, first.Batch.Start) + require.Equal(t, StreamMarkerNone, first.Batch.Marker) + require.Len(t, first.Batch.Transactions, 3) + require.NoError(t, first.Batch.Err) + require.Equal(t, StreamActive, a.StreamStatusOf(first.Generation)) + + identities, verified, err := first.Batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, verified) + require.Len(t, identities, 3) + prepared, err := block.PrepareVerifiedTransactionMessageIdentities(first.Batch.Transactions, identities) + require.NoError(t, err) + require.Equal(t, 3, prepared.Len()) + + // Recovery path: the pending list must show the same batches by range. + pending := a.PendingStreamBatches(first.Generation, 0) + require.Len(t, pending, 2) + require.Equal(t, header.Batch.Start, pending[0].Start) + require.Equal(t, first.Batch.Start, pending[1].Start) + require.Equal(t, first.Batch.End, pending[1].End) + require.Empty(t, a.PendingStreamBatches(first.Generation, first.Batch.End+1)) + + require.Nil(t, feedPrefetchShreds(t, a, batches[2])) + second := nextStreamEvent(t, events, StreamBatchReady) + require.Equal(t, first.Generation, second.Generation) + require.Len(t, second.Batch.Transactions, 4) + // Make sure the prefetch has retained both entry batches before the last + // component completes the slot; completion then reuses them by identity. + waitPrefetchedBatch(t, a, slot, second.Batch.Start) + + blk := feedPrefetchShreds(t, a, batches[3]) + require.NotNil(t, blk) + done := nextStreamEvent(t, events, StreamCompleted) + require.Equal(t, first.Generation, done.Generation) + require.Equal(t, StreamDone, a.StreamStatusOf(first.Generation)) + require.Empty(t, a.PendingStreamBatches(first.Generation, 0), "a completed generation has nothing pending") + + // Pointer identity: the prefix a streaming consumer executed is the block. + require.Len(t, blk.Transactions, 7) + for i, tx := range first.Batch.Transactions { + require.Same(t, tx, blk.Transactions[i]) + } + for i, tx := range second.Batch.Transactions { + require.Same(t, tx, blk.Transactions[3+i]) + } + require.Equal(t, uint64(slot-1), blk.SourceParentSlot) + require.True(t, blk.HasAlpenglowParentBlockID) + require.Equal(t, parentID, solana.Hash(blk.AlpenglowParentBlockID)) + require.Zero(t, a.StreamDroppedEvents()) +} + +// A reset while a slot is streaming cancels the generation; re-assembling the +// slot produces a different generation. +func TestStreamFeedCancelsOnResetAndRenewsGeneration(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent, 64) + a.SubscribeStream(events) + + const slot = 301 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 2)), + prefetchTestPayload(t, verifierSignedTransactions(t, 2))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + first := nextStreamEvent(t, events, StreamBatchReady) + + a.ResetSlot(slot) + cancelled := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, first.Generation, cancelled.Generation) + require.Equal(t, "reset", cancelled.Reason) + require.Equal(t, StreamGone, a.StreamStatusOf(first.Generation)) + require.Empty(t, a.PendingStreamBatches(first.Generation, 0)) + + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + renewed := nextStreamEvent(t, events, StreamBatchReady) + require.NotEqual(t, first.Generation, renewed.Generation) + require.Equal(t, StreamActive, a.StreamStatusOf(renewed.Generation)) +} + +// Dropped wake-ups are counted and never lose state: the batches remain +// discoverable through PendingStreamBatches. +func TestStreamFeedDropsWakeupsWhenSubscriberIsFull(t *testing.T) { + v := newTransactionVerifier(2, 16, func(*solana.Transaction) error { return nil }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + events := make(chan StreamEvent) // unbuffered and never drained: every send drops + a.SubscribeStream(events) + + const slot = 302 + batches := prefetchTestShreds(t, slot, + prefetchTestPayload(t, verifierSignedTransactions(t, 2)), + prefetchTestPayload(t, verifierSignedTransactions(t, 2))) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + cached := waitPrefetchedBatch(t, a, slot, 0) + require.NotNil(t, cached) + require.Eventually(t, func() bool { return a.StreamDroppedEvents() >= 1 }, 3*time.Second, time.Millisecond) + + a.mu.Lock() + state := a.slots[slot] + a.mu.Unlock() + g := StreamGeneration{slot: slot, state: state} + pending := a.PendingStreamBatches(g, 0) + require.Len(t, pending, 1) + require.Len(t, pending[0].Transactions, 2) +} From ad9f7346fbd779f58aa3ae687d0fcad9d2647a64 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 06:40:38 +0000 Subject: [PATCH 041/111] sealevel/replay: let a speculative bank defer its process-global side effects A bank executed before its block is complete must leave nothing behind if it is thrown away. Account writes already live in the bank's own overlay; this makes the remaining process-global effects of transaction execution deferrable on the SlotCtx: - DeferVoteCachePublication buffers global vote-cache puts and deletes in per-SlotCtx pending maps (reads consult them first so a bank sees its own changes) and turns the vote/stake dirty marker into VoteStakeDirty; publishDeferredVoteCache applies them once the bank is accepted. Banks that do not defer publish immediately, exactly as before. This is the replacement for the plan's "reset the dirty-slot marker", which Codex correctly flagged as not undoing writes already made to the caches. - TrackProgramCacheAdds records every program-cache insertion made while executing the bank (bpf_loader and loader_v4 report through RecordProgramCacheAdd) so a discarded bank can evict them again; eviction only forces a reload from account data, so it is always safe. - Stake index entries were already slot-keyed (DropPendingStakePubkeysFrom). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/transaction.go | 95 +++++++++++++++++++++++++++++++++-- pkg/sealevel/bpf_loader.go | 1 + pkg/sealevel/execution_ctx.go | 44 ++++++++++++++++ pkg/sealevel/loader_v4.go | 1 + 4 files changed, 136 insertions(+), 5 deletions(-) diff --git a/pkg/replay/transaction.go b/pkg/replay/transaction.go index ab5a61534..39af9c6e7 100644 --- a/pkg/replay/transaction.go +++ b/pkg/replay/transaction.go @@ -297,21 +297,106 @@ func recordVoteTimestampAndSlot(slotCtx *sealevel.SlotCtx, acct *accounts.Accoun func recordStakeAndVoteAccount(slotCtx *sealevel.SlotCtx, execCtx *sealevel.ExecutionCtx, acct *accounts.Account, modifiedVoteAccts bool) { if acct.Lamports == 0 || acct.Owner != a.VoteProgramAddr { - if global.VoteCacheItem(acct.Key) != nil { - global.DeleteVoteCacheItem(acct.Key) - markVoteStakeDirty(slotCtx.Slot) // global cache mutated — gates in-loop unwind + if voteCacheHas(slotCtx, acct.Key) { + deleteVoteCacheItem(slotCtx, acct.Key) + markSlotVoteStakeDirty(slotCtx) // global cache mutated — gates in-loop unwind } } else if modifiedVoteAccts { recordVoteTimestampAndSlot(slotCtx, acct) newVersionedVoteState, wasModified := execCtx.ModifiedVoteStates[acct.Key] if wasModified { - global.PutVoteCacheItem(acct.Key, newVersionedVoteState) + putVoteCacheItem(slotCtx, acct.Key, newVersionedVoteState) } - markVoteStakeDirty(slotCtx.Slot) + markSlotVoteStakeDirty(slotCtx) } if acct.Owner == a.StakeProgramAddr { recordStakeDelegation(slotCtx.Slot, acct) + markSlotVoteStakeDirty(slotCtx) + } +} + +// The vote cache is process-global. A bank executed speculatively (streaming +// replay) defers its puts/deletes into the SlotCtx so a discard leaves the +// cache untouched; publishDeferredVoteCache applies them once the bank is +// accepted. Non-deferring banks publish immediately, exactly as before. + +func voteCacheHas(slotCtx *sealevel.SlotCtx, key solana.PublicKey) bool { + if slotCtx.DeferVoteCachePublication { + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if _, deleted := slotCtx.PendingVoteCacheDeletes[key]; deleted { + return false + } + if _, pending := slotCtx.PendingVoteCache[key]; pending { + return true + } + } + return global.VoteCacheItem(key) != nil +} + +func putVoteCacheItem(slotCtx *sealevel.SlotCtx, key solana.PublicKey, state *sealevel.VoteStateVersions) { + if !slotCtx.DeferVoteCachePublication { + global.PutVoteCacheItem(key, state) + return + } + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if slotCtx.PendingVoteCache == nil { + slotCtx.PendingVoteCache = make(map[solana.PublicKey]*sealevel.VoteStateVersions) + } + slotCtx.PendingVoteCache[key] = state + delete(slotCtx.PendingVoteCacheDeletes, key) +} + +func deleteVoteCacheItem(slotCtx *sealevel.SlotCtx, key solana.PublicKey) { + if !slotCtx.DeferVoteCachePublication { + global.DeleteVoteCacheItem(key) + return + } + slotCtx.PendingVoteCacheMu.Lock() + defer slotCtx.PendingVoteCacheMu.Unlock() + if slotCtx.PendingVoteCacheDeletes == nil { + slotCtx.PendingVoteCacheDeletes = make(map[solana.PublicKey]struct{}) + } + slotCtx.PendingVoteCacheDeletes[key] = struct{}{} + delete(slotCtx.PendingVoteCache, key) +} + +func markSlotVoteStakeDirty(slotCtx *sealevel.SlotCtx) { + if slotCtx.DeferVoteCachePublication { + slotCtx.PendingVoteCacheMu.Lock() + slotCtx.VoteStakeDirty = true + slotCtx.PendingVoteCacheMu.Unlock() + return + } + markVoteStakeDirty(slotCtx.Slot) +} + +// publishDeferredVoteCache applies a speculative bank's buffered vote-cache +// changes and dirty marker. It is called by the streaming finalize step after +// the complete block has been matched to the executed prefix and is a no-op +// for banks that published immediately. +func publishDeferredVoteCache(slotCtx *sealevel.SlotCtx) { + if slotCtx == nil || !slotCtx.DeferVoteCachePublication { + return + } + slotCtx.PendingVoteCacheMu.Lock() + puts := slotCtx.PendingVoteCache + deletes := slotCtx.PendingVoteCacheDeletes + dirty := slotCtx.VoteStakeDirty + slotCtx.PendingVoteCache = nil + slotCtx.PendingVoteCacheDeletes = nil + slotCtx.VoteStakeDirty = false + slotCtx.DeferVoteCachePublication = false + slotCtx.PendingVoteCacheMu.Unlock() + for key := range deletes { + global.DeleteVoteCacheItem(key) + } + for key, state := range puts { + global.PutVoteCacheItem(key, state) + } + if dirty { markVoteStakeDirty(slotCtx.Slot) } } diff --git a/pkg/sealevel/bpf_loader.go b/pkg/sealevel/bpf_loader.go index b80ab8853..845accd60 100644 --- a/pkg/sealevel/bpf_loader.go +++ b/pkg/sealevel/bpf_loader.go @@ -1366,6 +1366,7 @@ func addProgramToCache(execCtx *ExecutionCtx, programAddr solana.PublicKey, entr return } execCtx.SlotCtx.AccountsDb.AddProgramToCache(programAddr, entry) + execCtx.SlotCtx.RecordProgramCacheAdd(programAddr) } func mapVirtualAddressSpaceRunErr(execCtx *ExecutionCtx, err error, inputRegions []sbpf.InputRegion) error { diff --git a/pkg/sealevel/execution_ctx.go b/pkg/sealevel/execution_ctx.go index ba54cc267..326da6762 100644 --- a/pkg/sealevel/execution_ctx.go +++ b/pkg/sealevel/execution_ctx.go @@ -107,6 +107,50 @@ type SlotCtx struct { bankSysvars atomic.Pointer[BankSysvars] TraceCtx context.Context + + // Speculative execution support (streaming replay). A bank executed before + // its block is complete must not publish to process-global state until + // the block is accepted, so a discard leaves nothing behind. + // + // DeferVoteCachePublication buffers global vote-cache puts and deletes in + // the pending maps (protected by PendingVoteCacheMu) and turns the + // vote/stake dirty marker into VoteStakeDirty; the replay finalize step + // publishes them once the bank is accepted. + DeferVoteCachePublication bool + PendingVoteCacheMu sync.Mutex + PendingVoteCache map[solana.PublicKey]*VoteStateVersions + PendingVoteCacheDeletes map[solana.PublicKey]struct{} + VoteStakeDirty bool + // TrackProgramCacheAdds records every program-cache insertion made while + // executing this bank so a discarded bank can evict them again (eviction + // only forces a reload from account data, so it is always safe). + TrackProgramCacheAdds bool + ProgramCacheAddsMu sync.Mutex + ProgramCacheAdds []solana.PublicKey +} + +// RecordProgramCacheAdd notes a program-cache insertion for later undo when +// TrackProgramCacheAdds is set. +func (slotCtx *SlotCtx) RecordProgramCacheAdd(key solana.PublicKey) { + if slotCtx == nil || !slotCtx.TrackProgramCacheAdds { + return + } + slotCtx.ProgramCacheAddsMu.Lock() + slotCtx.ProgramCacheAdds = append(slotCtx.ProgramCacheAdds, key) + slotCtx.ProgramCacheAddsMu.Unlock() +} + +// TakeProgramCacheAdds returns and clears the recorded program-cache +// insertions. +func (slotCtx *SlotCtx) TakeProgramCacheAdds() []solana.PublicKey { + if slotCtx == nil { + return nil + } + slotCtx.ProgramCacheAddsMu.Lock() + defer slotCtx.ProgramCacheAddsMu.Unlock() + adds := slotCtx.ProgramCacheAdds + slotCtx.ProgramCacheAdds = nil + return adds } // BankSysvars returns the immutable sysvar snapshot owned by this bank. diff --git a/pkg/sealevel/loader_v4.go b/pkg/sealevel/loader_v4.go index 6bb264b21..942aaf892 100644 --- a/pkg/sealevel/loader_v4.go +++ b/pkg/sealevel/loader_v4.go @@ -627,6 +627,7 @@ func LoaderV4ProcessDeploy(execCtx *ExecutionCtx) error { entry := &accountsdb.ProgramCacheEntry{Program: programObj, DeploymentSlot: currentSlot} if !execCtx.IsSimulation { execCtx.SlotCtx.AccountsDb.AddProgramToCache(program.Key(), entry) + execCtx.SlotCtx.RecordProgramCacheAdd(program.Key()) } return nil From 607724bd52bcaf5aa36bde14bcf15618095716b4 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 06:41:17 +0000 Subject: [PATCH 042/111] replay: streaming execution of turbine blocks while their shreds arrive MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Behind --streaming-execution (tuning.streaming_execution; default off), replay opens a speculative bank for frontier+1 as soon as its header component arrives and its parent is the executed frontier (same parent slot, same executed block ID, no fork switch pending, same epoch, no partitioned rewards in flight), then executes the slot's decoded entry batches as contiguous groups while the rest of the block is still in flight. The complete block from the ordinary emission path stays the authority: at arrival the executor proves that its executed prefix is the block — pointer identity of every executed transaction (completion reuses the prefetched batches by byte identity), same parent slot/ID, same configured parent bank hash, same feature-set pointer, same parent sysvar snapshot, same epoch, no epoch account updates, generation not gone — and only then executes the unexecuted suffix and runs the unchanged end-of-block tail (fees, rent, footer clock, vote rewards, bank hash, footer verification, commit). Any mismatch, or any doubt, discards the speculative bank and the block executes whole exactly as before. A discard restores the world: the overlay dies with the SlotCtx, deferred vote-cache changes are dropped, recorded program-cache insertions are evicted, the slot's pending stake-index entries are dropped, the parent's VoteTimestamps map was cloned at open, and the shell is configured without publishing the process-global current slot. Batches the assembler's verifier did not admit are never self-verified (that path halts the process on a bad signature); the stream is discarded instead and the block-level verification at completion covers them. The loop discards the stream on every fork switch, quarantine, skip, local-production block and other-slot emission, and at exit. The wait for the next replay input (waitForReplayInput, BlockSource. NextReplayInput) now also selects on the feed and on the executor's 5 ms poll timer (nil while idle); feed wake-ups and ticks are handled on the replay goroutine and never end the wait, so the loop body still only sees a block, a switch or an ended wait. Without streaming the wait is unchanged (waitForAlpenglowReplayInput delegates to it). Metrics: StreamingExecution{Opened, Groups, Transactions, TxLoopBeforeFull, OpenDelay, Discarded, DiscardReason} per block. Flags: --streaming-execution, --streaming-workers (0 = min(txpar, 4)), --streaming-min-group-batches, --streaming-max-open-ms (2000). Tests (fake feed, real group execution): contiguous grouping in shred order with gaps held and duplicate wake-ups ignored; tick recovery after dropped wake-ups; the group minimum with completion override; discards on UpdateParent, decode error, unverified batch, group failure, cancellation, gone generation and open-age bounds; global side-effect undo; eligibility reasons; the handshake's every binding; finalize falling back on a prefix mismatch; the wait loop dispatching wake-ups and preserving the legacy paths. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- cmd/mithril/node/node.go | 18 + config.example.toml | 15 + pkg/blockstream/block_source.go | 38 +- pkg/replay/alpenglow_switch.go | 82 +++- pkg/replay/block.go | 93 +++- pkg/replay/streaming.go | 749 ++++++++++++++++++++++++++++++++ pkg/replay/streaming_test.go | 611 ++++++++++++++++++++++++++ 7 files changed, 1588 insertions(+), 18 deletions(-) create mode 100644 pkg/replay/streaming.go create mode 100644 pkg/replay/streaming_test.go diff --git a/cmd/mithril/node/node.go b/cmd/mithril/node/node.go index 57c8f76a0..c9c964064 100644 --- a/cmd/mithril/node/node.go +++ b/cmd/mithril/node/node.go @@ -126,6 +126,9 @@ var ( pprofPort int64 blockstorePath string txParallelism int64 + // streamingMaxOpenMs is --streaming-max-open-ms; resolved into + // replay.StreamingExecutionCfg.MaxOpenAge with the other [replay] keys. + streamingMaxOpenMs int debugTxs []string debugAcctWrites []string @@ -538,6 +541,14 @@ func init() { // [replay] section flags Run.Flags().Int64Var(&txParallelism, "txpar", 0, "Transaction execution workers (>0 enables topsort parallelism; explicit 0 is sequential; unset validator mode defaults to 2x CPU cores)") Run.Flags().Int64Var(&numReplaySlots, "num-slots", 0, "Number of slots to replay (0 = run continuously)") + Run.Flags().BoolVar(&replay.StreamingExecutionCfg.Enabled, "streaming-execution", false, + "Execute Turbine blocks while their shreds arrive (Alpenglow validator/verifying modes only; the complete block remains authoritative and any mismatch falls back to whole-block execution)") + Run.Flags().IntVar(&replay.StreamingExecutionCfg.Workers, "streaming-workers", 0, + "Streaming execution workers per transaction group (0 = min(txpar, 4))") + Run.Flags().IntVar(&replay.StreamingExecutionCfg.MinGroupBatches, "streaming-min-group-batches", 0, + "Contiguous decoded batches to accumulate before a streaming group executes (0 or 1 = execute as batches arrive)") + Run.Flags().IntVar(&streamingMaxOpenMs, "streaming-max-open-ms", 0, + "Discard a streaming bank whose block has not completed after this many milliseconds (0 = 2000)") Run.Flags().Int64VarP(&endSlot, "end-slot", "e", -1, "Block at which to stop replaying, inclusive (-1 = run continuously)") // [consensus] section flags @@ -1165,6 +1176,13 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { } resolvedSigverifyBackend = resolved sbpf.UsePool = getBool("use-pool", "tuning.use_pool") + // [tuning] streaming execution (off by default; Alpenglow turbine only). + replay.StreamingExecutionCfg.Enabled = getBool("streaming-execution", "tuning.streaming_execution") + replay.StreamingExecutionCfg.Workers = getInt("streaming-workers", "tuning.streaming_workers") + replay.StreamingExecutionCfg.MinGroupBatches = getInt("streaming-min-group-batches", "tuning.streaming_min_group_batches") + if ms := getInt("streaming-max-open-ms", "tuning.streaming_max_open_ms"); ms > 0 { + replay.StreamingExecutionCfg.MaxOpenAge = time.Duration(ms) * time.Millisecond + } accountsdb.StoreAccountsWorkers = getInt("store-accounts-workers", "tuning.store_accounts_workers") accountsdb.ProgramCacheMaxMB = getInt("program-cache-max-mb", "tuning.program_cache_max_mb") if accountsdb.ProgramCacheMaxMB <= 0 { diff --git a/config.example.toml b/config.example.toml index 7106f737a..4211c3871 100644 --- a/config.example.toml +++ b/config.example.toml @@ -511,6 +511,21 @@ name = "mithril" # Zstd decoder concurrency (defaults to NumCPU) # zstd_decoder_concurrency = 16 + # Streaming execution (Alpenglow, native turbine only): execute a block's + # entry batches while its remaining shreds arrive, on a speculative bank + # over the executed parent. The complete block stays authoritative — the + # executed prefix must be the block's own transactions (pointer identity) + # or the bank is discarded and the block executes whole. Off by default. + # streaming_execution = false + # Execution workers per streaming group (0 = min(txpar, 4)). + # streaming_workers = 0 + # Contiguous decoded batches to accumulate before a group executes + # (0 or 1 = execute as each batch arrives). + # streaming_min_group_batches = 0 + # Discard a speculative bank whose block has not completed after this + # many milliseconds (0 = 2000). + # streaming_max_open_ms = 0 + # Snapshot bootstrap I/O tuning. # These defaults deliberately avoid flooding a single NVMe with hundreds of # concurrent writes. Increase cautiously on very fast multi-disk systems. diff --git a/pkg/blockstream/block_source.go b/pkg/blockstream/block_source.go index e81416f14..72e5bb530 100644 --- a/pkg/blockstream/block_source.go +++ b/pkg/blockstream/block_source.go @@ -3803,20 +3803,48 @@ func (bs *BlockSource) NextBlock() *b.Block { // may be nil to disable decision wakeups, but must never be closed. The third // result distinguishes a decision wakeup from a closed stream or cancellation. func (bs *BlockSource) NextBlockOrAlpenglowEvent(ctx context.Context, decisionChanges <-chan struct{}) (block *b.Block, parentSwitch *AlpenglowParentSwitch, decisionChanged bool) { + in := bs.NextReplayInput(ctx, decisionChanges, nil, nil) + return in.Block, in.ParentSwitch, in.DecisionChanged +} + +// ReplayInput is one wake-up of replay's wait for the next thing to do. At +// most one field is set; the zero value means the wait context ended or the +// source closed. +type ReplayInput struct { + Block *b.Block + ParentSwitch *AlpenglowParentSwitch + DecisionChanged bool + // StreamEvent is a turbine streaming-feed wake-up (see StreamEvents). + StreamEvent *turbine.StreamEvent + // StreamTick is the streaming executor's poll timer. + StreamTick bool +} + +// NextReplayInput is NextBlockOrAlpenglowEvent extended with the streaming +// feed and the executor's poll timer, both of which may be nil (never fire). +// A queued parent switch keeps its priority; among the remaining inputs the +// choice is the runtime's, which is fine because every stream wake-up is +// idempotent and the complete block is processed the same way whether or not +// the feed was drained first. +func (bs *BlockSource) NextReplayInput(ctx context.Context, decisionChanges <-chan struct{}, streamEvents <-chan turbine.StreamEvent, streamTick <-chan time.Time) ReplayInput { select { case event := <-bs.alpenglowParentSwitchCh: - return nil, &event, false + return ReplayInput{ParentSwitch: &event} default: } select { case event := <-bs.alpenglowParentSwitchCh: - return nil, &event, false + return ReplayInput{ParentSwitch: &event} case block := <-bs.streamChan: - return block, nil, false + return ReplayInput{Block: block} case <-decisionChanges: - return nil, nil, true + return ReplayInput{DecisionChanged: true} + case event := <-streamEvents: + return ReplayInput{StreamEvent: &event} + case <-streamTick: + return ReplayInput{StreamTick: true} case <-ctx.Done(): - return nil, nil, false + return ReplayInput{} } } diff --git a/pkg/replay/alpenglow_switch.go b/pkg/replay/alpenglow_switch.go index 3dba13703..e19c8d763 100644 --- a/pkg/replay/alpenglow_switch.go +++ b/pkg/replay/alpenglow_switch.go @@ -13,6 +13,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/state" + "github.com/Overclock-Validator/mithril/pkg/turbine" "github.com/gagliardetto/solana-go" ) @@ -176,23 +177,86 @@ func waitForAlpenglowReplayInput( decisionChanges <-chan struct{}, pollInterval time.Duration, ) (*b.Block, *blockstream.AlpenglowParentSwitch, *CertifiedSwitch) { + adapted := func(ctx context.Context, decisionChanges <-chan struct{}, _ <-chan turbine.StreamEvent, _ <-chan time.Time) blockstream.ReplayInput { + block, parentSwitch, decisionChanged := next(ctx, decisionChanges) + return blockstream.ReplayInput{Block: block, ParentSwitch: parentSwitch, DecisionChanged: decisionChanged} + } + return waitForReplayInput(ctx, adapted, sweep, decisionChanges, pollInterval, nil) +} + +// replayStreamer is the streaming executor as the wait loop drives it: the +// feed it listens to, its poll timer (nil while idle), and the two handlers, +// which execute transaction groups on the replay goroutine. +type replayStreamer interface { + events() <-chan turbine.StreamEvent + tick() <-chan time.Time + handleEvent(turbine.StreamEvent) + handleTick() +} + +// replayInputSource is BlockSource.NextReplayInput. +type replayInputSource func(ctx context.Context, decisionChanges <-chan struct{}, streamEvents <-chan turbine.StreamEvent, streamTick <-chan time.Time) blockstream.ReplayInput + +// waitForReplayInput is waitForAlpenglowReplayInput with the streaming +// executor folded into the same wait: feed wake-ups and poll ticks are +// handled here, on the replay goroutine, and never end the wait, so the loop +// body only ever sees a block, a switch, or an ended wait exactly as before. +// streamer may be nil (no streaming); sweep may be nil (no certificate +// correction), in which case decision notifications stay disabled and the +// wait blocks without polling, as it always did. +func waitForReplayInput( + ctx context.Context, + next replayInputSource, + sweep func() *CertifiedSwitch, + decisionChanges <-chan struct{}, + pollInterval time.Duration, + streamer replayStreamer, +) (*b.Block, *blockstream.AlpenglowParentSwitch, *CertifiedSwitch) { + var streamEvents <-chan turbine.StreamEvent + if streamer != nil { + // A slot may have become eligible while the previous block executed + // (its header arrived mid-execution); open it before blocking. + streamer.handleTick() + streamEvents = streamer.events() + } if sweep == nil { - block, parentSwitch, _ := next(ctx, nil) - return block, parentSwitch, nil + decisionChanges = nil + if streamer == nil { + in := next(ctx, nil, nil, nil) + return in.Block, in.ParentSwitch, nil + } } for { if ctx.Err() != nil { return nil, nil, nil } - if sw := sweep(); sw != nil { - return nil, nil, sw + if sweep != nil { + if sw := sweep(); sw != nil { + return nil, nil, sw + } + } + waitCtx, cancel := ctx, func() {} + if sweep != nil { + waitCtx, cancel = context.WithTimeout(ctx, pollInterval) } - waitCtx, cancel := context.WithTimeout(ctx, pollInterval) - block, parentSwitch, decisionChanged := next(waitCtx, decisionChanges) - timedOut := waitCtx.Err() == context.DeadlineExceeded + var tick <-chan time.Time + if streamer != nil { + tick = streamer.tick() + } + in := next(waitCtx, decisionChanges, streamEvents, tick) + timedOut := sweep != nil && waitCtx.Err() == context.DeadlineExceeded cancel() - if block != nil || parentSwitch != nil || (!decisionChanged && !timedOut) { - return block, parentSwitch, nil + switch { + case in.Block != nil || in.ParentSwitch != nil: + return in.Block, in.ParentSwitch, nil + case in.StreamEvent != nil: + streamer.handleEvent(*in.StreamEvent) + case in.StreamTick: + streamer.handleTick() + case in.DecisionChanged || timedOut: + // Re-sweep, then wait again. + default: + return nil, nil, nil } } } diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 2f35831e0..e84277fc7 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1293,6 +1293,18 @@ func reconstructFeeRateGovernor(s *state.MithrilState) *sealevel.FeeRateGovernor func configureBlock(block *b.Block, lastSlotCtx *sealevel.SlotCtx, epochSchedule *sealevel.SysvarEpochSchedule) error { + return configureBlockFromParent(block, lastSlotCtx, epochSchedule, true) +} + +// configureBlockFromParent derives the block's parent-dependent fields from +// the executed parent context. publishGlobal also publishes the block as the +// process-wide current slot (configureGlobalCtx); a speculative streaming +// shell passes false so the global view keeps describing the executed +// frontier until the complete block is configured. +func configureBlockFromParent(block *b.Block, + lastSlotCtx *sealevel.SlotCtx, + epochSchedule *sealevel.SysvarEpochSchedule, + publishGlobal bool) error { copy(block.ParentBankhash[:], lastSlotCtx.FinalBankhash) block.AcctsLtHash = lastSlotCtx.AcctsLtHash @@ -1308,7 +1320,9 @@ func configureBlock(block *b.Block, block.LastBlockhash = lastSlotCtx.Blockhash } - configureGlobalCtx(block) + if publishGlobal { + configureGlobalCtx(block) + } if global.ManageLeaderSchedule() { // epoch boundary. do not set leader @@ -2330,6 +2344,9 @@ func ReplayBlocks( opts.InitialAlpenglowBlockID = resumeState.ParentAlpenglowBlockID opts.HasInitialAlpenglowBlockID = true } + // Streaming execution needs the turbine batch feed; it is only ever + // eligible under Alpenglow with the unrooted tail (see streamingExecutor). + opts.TurbineStreamingExecution = StreamingExecutionCfg.Enabled && useTurbine && alpenglowMode && unrootedTailState != nil // Apply block fetching options if provided if blockFetchOpts != nil { @@ -2500,6 +2517,47 @@ func ReplayBlocks( } } + // Streaming execution: while the loop waits for the next complete block, + // the executor runs the entry batches of frontier+1 as turbine decodes + // them, against a speculative bank on lastSlotCtx. The closures read the + // loop's state at call time. streamInput is nil when the feed is off so + // the wait keeps its exact pre-streaming behaviour. + var streamer *streamingExecutor + var streamInput replayStreamer + if blockStream.StreamEvents() != nil { + streamer = newStreamingExecutor(streamingDeps{ + acctsDb: acctsDb, + feed: blockStream, + epochSchedule: epochSchedule, + txParallelism: txParallelism, + dbgOpts: dbgOpts, + persistedHashes: persistedHashes, + tail: unrootedTailState, + transactionStatuses: transactionStatuses, + alpenglowClock: alpenglowMode, + alpenglowMode: alpenglowMode, + unrootedTailUsed: unrootedTailState != nil, + lastSlotCtx: func() *sealevel.SlotCtx { return lastSlotCtx }, + frontier: func() uint64 { return replayFrontier }, + currentFeatures: func() *features.Features { return replayCtx.CurrentFeatures }, + currentEpoch: func() uint64 { return currentEpoch }, + rewardsInFlight: func() bool { + return partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 + }, + switchPending: func() bool { return sweepWhileWaiting != nil && sweepWhileWaiting() != nil }, + executedBlockID: func(slot uint64) (solana.Hash, bool) { + if id, ok := alpenglowExecutedBlockIDs[slot]; ok { + return id, true + } + return global.AlpenglowBlockID(slot) + }, + }) + streamInput = streamer + defer streamer.shutdown() + mlog.Log.Infof("streaming execution enabled: workers=%d min_group_batches=%d max_open=%s", + StreamingExecutionCfg.workers(txParallelism), StreamingExecutionCfg.MinGroupBatches, StreamingExecutionCfg.maxOpenAge()) + } + for { // The collector is per replay attempt. Discarded candidates, skipped // slots, and typed-recovery exits must never leak timings into the next @@ -2551,8 +2609,8 @@ func ReplayBlocks( } neededAt = time.Now() - block, parentSwitch, certifiedSwitch = waitForAlpenglowReplayInput(ctx, - blockStream.NextBlockOrAlpenglowEvent, sweepWhileWaiting, decisionChanges, alpenglowSwitchPollInterval) + block, parentSwitch, certifiedSwitch = waitForReplayInput(ctx, + blockStream.NextReplayInput, sweepWhileWaiting, decisionChanges, alpenglowSwitchPollInterval, streamInput) if ingress, ok := block.CompleteTurbineReplayAdmission(time.Now()); ok { _ = statsd.Duration(statsd.TurbineReplayAdmission, ingress.ReplayAdmission, nil) ingressTimings = &ingress @@ -2568,6 +2626,11 @@ func ReplayBlocks( result.WasCancelled = true break } + // Any fork switch invalidates a speculative bank above the frontier: + // its parent chain is about to be unwound or re-served. + if certifiedSwitch != nil || parentSwitch != nil { + streamer.discard("fork_switch") + } if certifiedSwitch != nil { if handleAlpenglowSwitch(certifiedSwitch, func() bool { blockStream.RewindForAlpenglowSwitch(certifiedSwitch.Slot, certifiedSwitch.Certified) @@ -2655,6 +2718,7 @@ func ReplayBlocks( // emitted suffix IDs are hard-tombstoned before that send, so discard // any leaked descendant before it reaches consensus observation. if blockStream.IsObjectivelyInvalidAlpenglowBlock(block) { + streamer.discardSlot(block.Slot, "quarantined") mlog.Log.Warnf("replay: discarding quarantined Alpenglow block %s at slot %d before consensus observation", solana.Hash(block.AlpenglowBlockID), block.Slot) continue @@ -2664,6 +2728,7 @@ func ReplayBlocks( if validationErr := validatePreConsensusTransactionStatuses( transactionStatuses, block, currentExecutedAnchorSlot(), ); validationErr != nil { + streamer.discardSlot(block.Slot, "status_validation") if !IsAlreadyProcessedTransactionError(validationErr) { result.Error = fmt.Errorf("pre-consensus block validation failed at slot %d: %w", block.Slot, validationErr) mlog.Log.Errorf("%v", result.Error) @@ -2685,6 +2750,7 @@ func ReplayBlocks( // or any bank changes, while the selected parent is still untouched. if alpenglowMode && !block.IsSkipped { if validationErr := validatePreConsensusRewardCertificates(block, epochSchedule, block.AlpenglowShredVersion); validationErr != nil { + streamer.discardSlot(block.Slot, "reward_certificates") if !IsInvalidRewardCertificateError(validationErr) { result.Error = fmt.Errorf("pre-consensus reward validation failed at slot %d: %w", block.Slot, validationErr) mlog.Log.Errorf("%v", result.Error) @@ -2734,6 +2800,7 @@ func ReplayBlocks( // re-fetches the certified version either way). if unrootedTailState != nil { if sw := switchSweeper.sweep(alpenglowExecutedBlockIDs, mithrilState.LastRootedSlot, replayFrontier); sw != nil { + streamer.discard("fork_switch") if handleAlpenglowSwitch(sw, func() bool { blockStream.RewindForAlpenglowSwitch(sw.Slot, sw.Certified) return true @@ -2789,6 +2856,7 @@ func ReplayBlocks( // Handle skipped slots - log and continue without execution if block.IsSkipped { + streamer.discardSlot(block.Slot, "skipped") // Zero is the explicit locally executed outcome for a skip. Parent-ID // gap inference is provisional; recording it lets a later certificate // naming a real block trigger the same in-RAM switch as a wrong sibling. @@ -3015,9 +3083,26 @@ func ReplayBlocks( parentBankSysvars = lastSlotCtx.BankSysvars() } if block.FromLocalProduction { + streamer.discardSlot(block.Slot, "local_production") lastSlotCtx, err = adoptLocalLeaderBlock(block, unrootedTailState, transactionStatuses, persistedHashes) } else { - lastSlotCtx, err = ProcessBlock(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, unrootedTailState, transactionStatuses, alpenglowClock, parentBankSysvars) + // A stream open on this slot finishes the block on its speculative + // bank when the block proves to be what it executed; otherwise the + // stream is discarded and the block executes whole, exactly as + // without streaming. + streamed := false + if streamer.matches(block.Slot) { + var streamedCtx *sealevel.SlotCtx + streamedCtx, streamed, err = streamer.finalize(block, parentBankSysvars) + if streamed { + lastSlotCtx = streamedCtx + } + } else { + streamer.discard("other_block") + } + if !streamed { + lastSlotCtx, err = ProcessBlock(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, unrootedTailState, transactionStatuses, alpenglowClock, parentBankSysvars) + } } processBlockEnd := time.Now() metrics.GlobalBlockReplay.ProcessBlock.AddTiming(processBlockEnd.Sub(processBlockStart)) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go new file mode 100644 index 000000000..bce9ba920 --- /dev/null +++ b/pkg/replay/streaming.go @@ -0,0 +1,749 @@ +package replay + +import ( + "context" + "errors" + "fmt" + "maps" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/blockstream" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/mlog" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" +) + +// Streaming execution executes a turbine block's entry batches while the rest +// of its shreds are still arriving, so that only the last batch and the +// end-of-block tail remain after the final shred. The complete block from the +// ordinary emission path stays the authority: the executor only pre-computes +// the bank overlay for a prefix of it, proves at finalize that the prefix is +// the block (pointer identity of every executed transaction, same parent slot +// and ID, same generation, same features), and otherwise throws the prefix +// away and lets the whole-block path execute the block from scratch. +// +// Nothing a stream does reaches process-global state until finalize: the +// overlay lives in the SlotCtx, vote-cache publication is deferred on the +// SlotCtx, program-cache insertions are recorded for undo, pending stake index +// entries are slot-keyed and dropped, the parent's VoteTimestamps map is +// cloned at open, the process-wide current slot is not published for the +// shell, and the legacy sysvar cache written by bank open is snapshotted and +// restored. Discard therefore restores the world to the state before the +// stream opened. + +// StreamingExecutionConfig is set from the node flags before replay starts. +type StreamingExecutionConfig struct { + // Enabled turns streaming execution on for turbine-sourced blocks. + Enabled bool + // Workers bounds the executor goroutines per group (0 = min(txpar, 4)). + Workers int + // MinGroupBatches delays a group until this many contiguous batches are + // ready, unless the slot is already complete (0 or 1 = execute as soon as + // one batch is ready). + MinGroupBatches int + // MaxOpenAge discards a stream that has been open this long without its + // block completing (0 = 2 s). + MaxOpenAge time.Duration +} + +// StreamingExecutionCfg is the process-wide streaming configuration. +var StreamingExecutionCfg StreamingExecutionConfig + +const ( + defaultStreamingWorkers = 4 + defaultStreamingMaxAge = 2 * time.Second + streamingPollInterval = 5 * time.Millisecond + // streamingHardOpenAgeFactor bounds a completed-but-not-yet-emitted stream + // to this multiple of MaxOpenAge. + streamingHardOpenAgeFactor = 10 +) + +func (cfg StreamingExecutionConfig) workers(txParallelism int) int { + workers := cfg.Workers + if workers <= 0 { + workers = defaultStreamingWorkers + } + if txParallelism > 0 && workers > txParallelism { + workers = txParallelism + } + return workers +} + +func (cfg StreamingExecutionConfig) maxOpenAge() time.Duration { + if cfg.MaxOpenAge <= 0 { + return defaultStreamingMaxAge + } + return cfg.MaxOpenAge +} + +// streamingFeed is the block source's view of the turbine feed +// (*blockstream.BlockSource implements it; tests substitute a fake). +type streamingFeed interface { + StreamEvents() <-chan turbine.StreamEvent + StreamStatusOf(turbine.StreamGeneration) turbine.StreamStatus + PendingStreamBatches(turbine.StreamGeneration, uint32) []*turbine.StreamBatch + PrioritizeStreamRepair(uint64) +} + +var _ streamingFeed = (*blockstream.BlockSource)(nil) + +// streamingDeps is what the executor needs from the replay loop. The closures +// read loop-local state (last slot context, frontier, features, switch +// status) at call time so the executor never caches a stale view. +type streamingDeps struct { + acctsDb *accountsdb.AccountsDb + feed streamingFeed + epochSchedule *sealevel.SysvarEpochSchedule + txParallelism int + dbgOpts *DebugOptions + persistedHashes *persistedTracker + tail unrootedState + transactionStatuses *TransactionStatusCache + alpenglowClock bool + + lastSlotCtx func() *sealevel.SlotCtx + frontier func() uint64 + currentFeatures func() *features.Features + currentEpoch func() uint64 + rewardsInFlight func() bool + switchPending func() bool + executedBlockID func(slot uint64) (solana.Hash, bool) + alpenglowMode bool + unrootedTailUsed bool +} + +type streamingGroup struct { + startedAt, finishedAt time.Time + transactions int +} + +// streamingSlot is one in-progress stream. +type streamingSlot struct { + slot uint64 + generation turbine.StreamGeneration + parentSlot uint64 + parentID solana.Hash + exec *blockExecution + nextStart uint32 + pending map[uint32]*turbine.StreamBatch + footerSeen bool + completed bool + openedAt time.Time + headerAt time.Time + groups []streamingGroup + // restoreSysvarCache puts the legacy process-global sysvar cache back to + // its state before the bank opened; nil when nothing was published. + restoreSysvarCache func() +} + +// streamingExecutor is owned by the replay loop and driven from its select. +type streamingExecutor struct { + deps streamingDeps + current *streamingSlot + // headers remembers header batches for slots ahead of the frontier so the + // next slot can open as soon as its parent finishes, even when its header + // wake-up arrived earlier. + headers map[uint64]*turbine.StreamBatch + ticker *time.Ticker + // executeFn runs one group on the open execution; tests substitute it. + executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error +} + +func newStreamingExecutor(deps streamingDeps) *streamingExecutor { + return &streamingExecutor{ + deps: deps, + headers: make(map[uint64]*turbine.StreamBatch), + executeFn: (*blockExecution).executeTransactionGroup, + } +} + +// tick returns the polling channel, which is nil (never fires) while no +// stream is open, so the replay loop's select stays quiet when idle. +func (s *streamingExecutor) tick() <-chan time.Time { + if s == nil || s.current == nil { + return nil + } + if s.ticker == nil { + s.ticker = time.NewTicker(streamingPollInterval) + } + return s.ticker.C +} + +func (s *streamingExecutor) stopTicker() { + if s.ticker != nil { + s.ticker.Stop() + s.ticker = nil + } +} + +// events is the feed channel the wait loop selects on; nil when the feed is +// off, which never fires. +func (s *streamingExecutor) events() <-chan turbine.StreamEvent { + if s == nil || s.deps.feed == nil { + return nil + } + return s.deps.feed.StreamEvents() +} + +// matches reports whether a stream is open for slot. +func (s *streamingExecutor) matches(slot uint64) bool { + return s != nil && s.current != nil && s.current.slot == slot +} + +// shutdown discards any open stream; the replay loop defers it so an exiting +// attempt never leaves a speculative bank (and its watchdog) behind. +func (s *streamingExecutor) shutdown() { + if s == nil { + return + } + s.discard("shutdown") + s.stopTicker() + s.headers = make(map[uint64]*turbine.StreamBatch) +} + +// handleEvent consumes one feed wake-up. +func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { + if s == nil { + return + } + switch event.Kind { + case turbine.StreamBatchReady: + if event.Batch == nil { + return + } + if s.current != nil && event.Generation == s.current.generation { + s.offer(event.Batch) + s.consume() + return + } + if event.Batch.Marker == turbine.StreamMarkerHeader && event.Batch.Start == 0 { + s.rememberHeader(event.Batch) + } + s.tryOpen() + case turbine.StreamCancelled: + if s.current != nil && event.Generation == s.current.generation { + s.discard("cancelled:" + event.Reason) + } + if header, ok := s.headers[event.Slot]; ok && header.Generation == event.Generation { + delete(s.headers, event.Slot) + } + s.tryOpen() + case turbine.StreamCompleted: + if s.current != nil && event.Generation == s.current.generation { + s.current.completed = true + // Wake-ups for the slot's batches precede this event in the + // channel, so everything decoded is already pending; run it now + // (the group minimum no longer applies). Anything a dropped + // wake-up missed runs in the finalize suffix: the prefetch state + // is released at completion, so there is nothing left to pull. + s.consume() + return + } + if header, ok := s.headers[event.Slot]; ok && header.Generation == event.Generation { + delete(s.headers, event.Slot) + } + } +} + +// handleTick polls the assembler for batches (recovery after dropped +// wake-ups), enforces the open-age bound, and opens the next slot if idle. +func (s *streamingExecutor) handleTick() { + if s == nil { + return + } + if s.current == nil { + s.tryOpen() + return + } + cur := s.current + switch s.deps.feed.StreamStatusOf(cur.generation) { + case turbine.StreamGone: + s.discard("gone") + s.tryOpen() + return + case turbine.StreamDone: + cur.completed = true + } + // An incomplete slot is bounded by MaxOpenAge (its shreds stopped + // arriving); a completed one may legitimately wait longer in the emitter + // (ancestry decisions) and is only bounded to cap the overlay's lifetime. + age := time.Since(cur.openedAt) + if (!cur.completed && age > StreamingExecutionCfg.maxOpenAge()) || age > streamingHardOpenAgeFactor*StreamingExecutionCfg.maxOpenAge() { + s.discard("timeout") + s.tryOpen() + return + } + s.pull() + s.consume() +} + +func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { + frontier := s.deps.frontier() + if header.Slot <= frontier { + return + } + s.headers[header.Slot] = header + s.pruneHeaders(frontier) +} + +// pruneHeaders bounds the header map: anything at or below the frontier can +// never open. +func (s *streamingExecutor) pruneHeaders(frontier uint64) { + for slot := range s.headers { + if slot <= frontier { + delete(s.headers, slot) + } + } +} + +// tryOpen opens a stream for frontier+1 when its header is known and every +// eligibility condition holds. +func (s *streamingExecutor) tryOpen() { + if s == nil || s.current != nil || !StreamingExecutionCfg.Enabled { + return + } + frontier := s.deps.frontier() + s.pruneHeaders(frontier) + next := frontier + 1 + header, ok := s.headers[next] + if !ok { + return + } + if reason := s.eligibility(header); reason != "" { + mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", next, reason) + delete(s.headers, next) + return + } + delete(s.headers, next) + s.openStream(header) +} + +// eligibility returns an empty string when a stream may open on header, or +// the reason it may not. +func (s *streamingExecutor) eligibility(header *turbine.StreamBatch) string { + d := s.deps + if !d.alpenglowMode || !d.unrootedTailUsed || d.tail == nil { + return "requires alpenglow rooted-durable replay" + } + if d.feed.StreamStatusOf(header.Generation) != turbine.StreamActive { + return "generation no longer active" + } + last := d.lastSlotCtx() + if last == nil { + return "no executed parent context" + } + if frontier := d.frontier(); header.Slot != frontier+1 || header.ParentSlot != last.Slot { + return fmt.Sprintf("slot %d on parent %d does not extend the executed frontier %d (parent context %d)", header.Slot, header.ParentSlot, frontier, last.Slot) + } + executedID, ok := d.executedBlockID(last.Slot) + if !ok || executedID == (solana.Hash{}) || executedID != header.ParentBlockID { + return "parent block id does not match the executed parent" + } + if d.switchPending() { + return "fork switch pending" + } + if d.epochSchedule == nil || d.epochSchedule.GetEpoch(header.Slot) != d.currentEpoch() { + return "epoch boundary" + } + if d.rewardsInFlight() { + return "partitioned rewards in flight" + } + if d.currentFeatures() == nil { + return "no feature set" + } + return "" +} + +func (s *streamingExecutor) openStream(header *turbine.StreamBatch) { + d := s.deps + last := d.lastSlotCtx() + shell := &b.Block{ + Slot: header.Slot, + SourceParentSlot: header.ParentSlot, + FromLiveStream: true, + AlpenglowParentBlockID: header.ParentBlockID, + HasAlpenglowParentBlockID: true, + } + shell.Epoch = d.epochSchedule.GetEpoch(shell.Slot) + // Same derivation the loop applies to the complete block, minus the + // process-global "current slot" publication, which stays at the frontier + // until the complete block is configured. + if err := configureBlockFromParent(shell, last, d.epochSchedule, false); err != nil { + mlog.Log.Warnf("streaming: slot %d not opened: %v", shell.Slot, err) + return + } + // The parent's VoteTimestamps map is shared by reference through + // configureBlock; a speculative bank must mutate its own copy. + shell.VoteTimestamps = maps.Clone(last.VoteTimestamps) + shell.Features = d.currentFeatures() + + // Bank open publishes the child's derived Clock/SlotHashes to the legacy + // process-global sysvar cache (Alpenglow banks never read it back — they + // pin from parentBankSysvars — but RPC simulation may). Snapshot it so a + // discard restores the parent's view; the accepted bank leaves it as a + // whole-block open would have. + sysvarCacheAtOpen := sealevel.SysvarCache + exec := newBlockExecution(d.acctsDb, shell, d.epochSchedule, StreamingExecutionCfg.workers(d.txParallelism), d.dbgOpts, d.persistedHashes, d.tail, d.transactionStatuses, d.alpenglowClock, last.BankSysvars()) + if err := exec.open(); err != nil { + exec.close() + sealevel.SysvarCache = sysvarCacheAtOpen + mlog.Log.Warnf("streaming: slot %d not opened: %v", shell.Slot, err) + return + } + exec.slotCtx.DeferVoteCachePublication = true + exec.slotCtx.TrackProgramCacheAdds = true + exec.setReplayStage("streaming_wait") + + s.current = &streamingSlot{ + slot: shell.Slot, + generation: header.Generation, + parentSlot: header.ParentSlot, + parentID: header.ParentBlockID, + exec: exec, + pending: make(map[uint32]*turbine.StreamBatch), + openedAt: time.Now(), + headerAt: header.ReadyAt, + restoreSysvarCache: func() { sealevel.SysvarCache = sysvarCacheAtOpen }, + } + metrics.GlobalBlockReplay.StreamingExecution.Opened = 1 + d.feed.PrioritizeStreamRepair(shell.Slot) + mlog.Log.FileOnlyf("streaming: opened slot %d on parent %d", shell.Slot, header.ParentSlot) + s.offer(header) + s.pull() + s.consume() +} + +func (s *streamingExecutor) offer(batch *turbine.StreamBatch) { + cur := s.current + if cur == nil || batch == nil || batch.Start < cur.nextStart { + return + } + if _, seen := cur.pending[batch.Start]; !seen { + cur.pending[batch.Start] = batch + } +} + +// pull asks the assembler for everything decoded since nextStart; it is the +// authoritative path after a dropped wake-up. +func (s *streamingExecutor) pull() { + cur := s.current + if cur == nil { + return + } + for _, batch := range s.deps.feed.PendingStreamBatches(cur.generation, cur.nextStart) { + s.offer(batch) + } +} + +// consume executes every contiguous ready batch from nextStart as one group. +// Nothing is removed from pending until the group is committed, so holding +// for the group minimum leaves markers and batches exactly where they were. +func (s *streamingExecutor) consume() { + cur := s.current + if cur == nil { + return + } + var group []*turbine.StreamBatch + next := cur.nextStart + footer := false + for { + batch, ok := cur.pending[next] + if !ok { + break + } + if batch.Err != nil { + s.discard("decode_error") + return + } + switch batch.Marker { + case turbine.StreamMarkerHeader: + // The header opened the stream; nothing to execute. + case turbine.StreamMarkerUpdateParent: + // The leader abandoned the optimistic prefix we executed. + s.discard("update_parent") + return + case turbine.StreamMarkerFooter: + footer = true + default: + group = append(group, batch) + } + next = batch.End + 1 + } + if minBatches := StreamingExecutionCfg.MinGroupBatches; minBatches > 1 && len(group) > 0 && len(group) < minBatches && !cur.completed { + return // not enough ready work yet; everything stays pending + } + for start := cur.nextStart; start < next; { + batch := cur.pending[start] + delete(cur.pending, start) + start = batch.End + 1 + } + cur.nextStart = next + if footer { + cur.footerSeen = true + } + if len(group) == 0 { + return + } + if err := s.executeGroup(group); err != nil { + s.discard(err.Error()) + } +} + +// executeGroup joins verification for every batch in the group and executes +// the group's transactions as one unit. +func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { + cur := s.current + var txs []*solana.Transaction + var verified []txverify.VerifiedMessageIdentity + allVerified := true + for _, batch := range group { + identities, ok, err := batch.WaitVerification(context.Background()) + if errors.Is(err, turbine.ErrStreamBatchUnverified) { + ok, err = false, nil + } + if err != nil { + return fmt.Errorf("sigverify: %w", err) + } + if !ok { + allVerified = false + } + txs = append(txs, batch.Transactions...) + verified = append(verified, identities...) + } + if len(txs) == 0 { + return nil + } + if !allVerified || len(verified) != len(txs) { + // The assembler's verifier refused the batch (admission), so the + // block-level verification at completion will cover it. Verifying here + // through ProcessTransaction is not an option: that path halts the + // process on an invalid signature, which a speculative bank on an + // unauthenticated prefix must never do. + return errors.New("unverified_batch") + } + prepared, err := b.PrepareVerifiedTransactionMessageIdentities(txs, verified) + if err != nil { + return fmt.Errorf("identities: %w", err) + } + started := time.Now() + err = s.executeFn(cur.exec, txs, prepared, false) + cur.exec.setReplayStage("streaming_wait") + if err != nil { + var duplicates *DuplicateTransactionMessagesError + if errors.As(err, &duplicates) { + return errors.New("duplicate_message") + } + if IsAlreadyProcessedTransactionError(err) { + return errors.New("already_processed") + } + return fmt.Errorf("group: %w", err) + } + cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(txs)}) + return nil +} + +// discard throws the in-progress stream away and undoes every side effect it +// may have had outside its own SlotCtx. +func (s *streamingExecutor) discard(reason string) { + if s == nil || s.current == nil { + return + } + cur := s.current + s.current = nil + s.stopTicker() + exec := cur.exec + if exec != nil { + exec.close() + if exec.slotCtx != nil { + if s.deps.acctsDb != nil { + for _, key := range exec.slotCtx.TakeProgramCacheAdds() { + s.deps.acctsDb.RemoveProgramFromCache(key) + } + } + // Deferred vote-cache changes die with the SlotCtx; the stake index + // entries are keyed by slot and the stream is the only bank above + // the frontier. + exec.slotCtx.PendingVoteCache = nil + exec.slotCtx.PendingVoteCacheDeletes = nil + exec.slotCtx.VoteStakeDirty = false + } + } + global.DropPendingStakePubkeysFrom(cur.slot) + if cur.restoreSysvarCache != nil { + cur.restoreSysvarCache() + } + metrics.GlobalBlockReplay.StreamingExecution.Discarded = 1 + metrics.GlobalBlockReplay.StreamingExecution.DiscardReason = reason + mlog.Log.FileOnlyf("streaming: discarded slot %d after %d groups (%s)", cur.slot, len(cur.groups), reason) +} + +// discardSlot discards the stream if it is open for slot. +func (s *streamingExecutor) discardSlot(slot uint64, reason string) { + if s.matches(slot) { + s.discard(reason) + } +} + +// finalizeErr marks a failure after the handshake passed; the block is as +// invalid as it would have been for the whole-block path. +type streamingFinalizeError struct{ err error } + +func (e *streamingFinalizeError) Error() string { return e.err.Error() } +func (e *streamingFinalizeError) Unwrap() error { return e.err } + +// finalize completes execution of block on the open stream. ok reports +// whether the stream matched the block; when it did not, the stream has been +// discarded and the caller must execute the block whole. A non-nil error with +// ok == true is a failure after the handshake and is final. +func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel.BankSysvars) (slotCtx *sealevel.SlotCtx, ok bool, err error) { + if s == nil || s.current == nil || block == nil { + return nil, false, nil + } + cur := s.current + if reason := s.handshake(block, parentBankSysvars); reason != "" { + s.discard("prefix_mismatch:" + reason) + return nil, false, nil + } + exec := cur.exec + fullAt := time.Time{} + if block.ShredFullNanos > 0 { + fullAt = time.Unix(0, block.ShredFullNanos) + } + + // Whole-block plan and status validation, exactly as ProcessBlock does + // them, now that the authoritative block exists. + if err := validateBlockTransactionVersions(block); err != nil { + s.discard("versions") + return nil, false, nil + } + executionPlanStart := time.Now() + executionPlan, err := planBlockTransactionExecution(block) + metrics.GlobalBlockReplay.TransactionExecutionPlan.AddTimingSince(executionPlanStart) + if err != nil { + s.discard("plan") + return nil, false, nil + } + executed := len(exec.transactions) + for i := 0; i < executed; i++ { + if executionPlan.execute[i] != exec.execute[i] { + s.discard("execution_mask") + return nil, false, nil + } + } + statusValidationStart := time.Now() + statusValidation, statusValidationErr := s.deps.transactionStatuses.validateBlockForPublication(block, executionPlan) + metrics.GlobalBlockReplay.TransactionStatusValidation.AddTimingSince(statusValidationStart) + if statusValidationErr != nil { + s.discard("status_validation") + return nil, false, nil + } + statusPreparation := s.deps.transactionStatuses.startStatusPreparation(executionPlan) + defer func() { + statusPreparation.wait() + if statusPreparation != nil { + metrics.GlobalBlockReplay.TransactionStatusPreparation.AddTiming(statusPreparation.duration) + } + }() + + // From here on the stream is committed to this block: switch the + // execution to the complete block object, execute the unexecuted suffix + // with the same machinery, publish what was deferred, and run the tail. + s.current = nil + s.stopTicker() + defer exec.close() + block.FeeRateGovernor = exec.block.FeeRateGovernor + block.VoteTimestamps = exec.slotCtx.VoteTimestamps + exec.block = block + exec.slotCtx.Blockhash = block.Blockhash + exec.slotCtx.Epoch = block.Epoch + if requireAlpenglowBlockFooter(block, exec.slotCtx, s.deps.alpenglowClock) { + if err := validateAlpenglowFooterNanosecondClock(exec.slotCtx, block); err != nil { + return nil, true, &streamingFinalizeError{err: err} + } + } + if suffix := block.Transactions[executed:]; len(suffix) > 0 { + started := time.Now() + if err := exec.executeTransactionGroup(suffix, executionPlan.messageIdentities.Slice(executed, len(block.Transactions)), !block.TransactionSignaturesVerified()); err != nil { + return nil, true, &streamingFinalizeError{err: fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)} + } + cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(suffix)}) + } + if exec.processedSignatures != executionPlan.processedSignatures || exec.processedTxCount != executionPlan.processedTxCount { + return nil, true, &streamingFinalizeError{err: fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", + block.Slot, exec.processedTxCount, exec.processedSignatures, executionPlan.processedTxCount, executionPlan.processedSignatures)} + } + exec.slotCtx.NumSignatures = executionPlan.processedSignatures + exec.slotCtx.TrackProgramCacheAdds = false + exec.slotCtx.TakeProgramCacheAdds() + publishDeferredVoteCache(exec.slotCtx) + + exec.executionPlan = executionPlan + exec.statusPreparation = statusPreparation + exec.statusValidation = statusValidation + slotCtx, err = exec.finalize() + if err != nil { + return nil, true, &streamingFinalizeError{err: err} + } + + // The per-block record is rebuilt from the stream's own bookkeeping: the + // loop resets the collector between waits, so counters accumulated while + // executing groups may or may not have survived to this point. + record := &metrics.GlobalBlockReplay.StreamingExecution + discarded, discardReason := record.Discarded, record.DiscardReason + *record = metrics.StreamingExecution{Opened: 1, Discarded: discarded, DiscardReason: discardReason} + record.Groups = uint64(len(cur.groups)) + for _, group := range cur.groups { + record.Transactions += uint64(group.transactions) + // Work that finished before the last shred arrived is the latency the + // stream took off the vote path. + if !fullAt.IsZero() && group.finishedAt.Before(fullAt) { + record.TxLoopBeforeFull.AddTiming(group.finishedAt.Sub(group.startedAt)) + } + } + record.OpenDelay.AddTiming(cur.openedAt.Sub(cur.headerAt)) + return slotCtx, true, nil +} + +// handshake proves the executed prefix is the block. It returns the mismatch +// reason, or "" when every binding holds. +func (s *streamingExecutor) handshake(block *b.Block, parentBankSysvars *sealevel.BankSysvars) string { + cur := s.current + exec := cur.exec + switch { + case block.Slot != cur.slot: + return "slot" + case block.IsSkipped || !block.FromLiveStream: + return "not a live block" + case !block.HasAlpenglowParentBlockID || block.AlpenglowParentBlockID != cur.parentID || block.SourceParentSlot != cur.parentSlot: + return "parent" + case block.ParentSlot != exec.block.ParentSlot || block.ParentBankhash != exec.block.ParentBankhash: + return "configured parent" + case block.Features != exec.block.Features: + return "features" + case parentBankSysvars == nil || parentBankSysvars != exec.parentBankSysvars: + return "parent sysvars" + case block.Epoch != exec.block.Epoch: + return "epoch" + case len(block.EpochUpdatedAccts) != 0: + return "epoch account updates" + case len(block.Transactions) < len(exec.transactions): + return "shorter than executed prefix" + } + status := s.deps.feed.StreamStatusOf(cur.generation) + if status == turbine.StreamGone { + return "generation gone" + } + for i, tx := range exec.transactions { + if block.Transactions[i] != tx { + return fmt.Sprintf("transaction %d identity", i) + } + } + return "" +} diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go new file mode 100644 index 000000000..d3730cec7 --- /dev/null +++ b/pkg/replay/streaming_test.go @@ -0,0 +1,611 @@ +package replay + +import ( + "context" + "errors" + "sort" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/blockstream" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// fakeStreamFeed stands in for the block source's view of the turbine feed. +type fakeStreamFeed struct { + events chan turbine.StreamEvent + status map[turbine.StreamGeneration]turbine.StreamStatus + pending map[turbine.StreamGeneration][]*turbine.StreamBatch + prioritized []uint64 +} + +func newFakeStreamFeed() *fakeStreamFeed { + return &fakeStreamFeed{ + events: make(chan turbine.StreamEvent, 64), + status: make(map[turbine.StreamGeneration]turbine.StreamStatus), + pending: make(map[turbine.StreamGeneration][]*turbine.StreamBatch), + } +} + +func (f *fakeStreamFeed) StreamEvents() <-chan turbine.StreamEvent { return f.events } + +func (f *fakeStreamFeed) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + if status, ok := f.status[g]; ok { + return status + } + return turbine.StreamGone +} + +func (f *fakeStreamFeed) PendingStreamBatches(g turbine.StreamGeneration, fromStart uint32) []*turbine.StreamBatch { + var out []*turbine.StreamBatch + for _, batch := range f.pending[g] { + if batch.Start >= fromStart { + out = append(out, batch) + } + } + sort.Slice(out, func(i, j int) bool { return out[i].Start < out[j].Start }) + return out +} + +func (f *fakeStreamFeed) PrioritizeStreamRepair(slot uint64) { + f.prioritized = append(f.prioritized, slot) +} + +// fakeUnrootedState satisfies the tail interface for eligibility checks; no +// method is ever called on it. +type fakeUnrootedState struct{ unrootedState } + +// verifiedIdentities runs the real batch verifier so the batches carry the +// identities the assembler's verifier would attach. +func verifiedIdentities(t *testing.T, txs []*solana.Transaction) []txverify.VerifiedMessageIdentity { + t.Helper() + var verifier txverify.BatchVerifier + errs := make([]error, len(txs)) + identities := make([]txverify.VerifiedMessageIdentity, len(txs)) + verifier.VerifyWithMessageIdentities(txs, errs, identities) + for i, err := range errs { + require.NoError(t, err, "fixture transaction %d must verify", i) + } + return identities +} + +// streamingTestHarness is an executor with a stream already open on the group +// execution environment's bank (slot 42 on parent 41), which is what +// openStream would have produced without the bank machinery. +type streamingTestHarness struct { + env *groupExecutionEnv + feed *fakeStreamFeed + exec *streamingExecutor + gen turbine.StreamGeneration + parentID solana.Hash + frontier uint64 + lastCtx *sealevel.SlotCtx +} + +func newStreamingTestHarness(t *testing.T) *streamingTestHarness { + t.Helper() + env := newGroupExecutionEnv(t, 2, 10_000_000_000) + t.Cleanup(env.cleanup) + feed := newFakeStreamFeed() + h := &streamingTestHarness{env: env, feed: feed, parentID: solana.Hash{7, 7, 7}, frontier: 41} + h.gen = turbine.NewDetachedStreamGeneration(env.exec.block.Slot) + feed.status[h.gen] = turbine.StreamActive + h.lastCtx = &sealevel.SlotCtx{Slot: 41, Epoch: env.exec.block.Epoch} + epochSchedule := sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000, LeaderScheduleSlotOffset: 432000, FirstNormalEpoch: 0, FirstNormalSlot: 0} + h.exec = newStreamingExecutor(streamingDeps{ + feed: feed, + epochSchedule: &epochSchedule, + tail: fakeUnrootedState{}, + alpenglowMode: true, + unrootedTailUsed: true, + lastSlotCtx: func() *sealevel.SlotCtx { return h.lastCtx }, + frontier: func() uint64 { return h.frontier }, + currentFeatures: func() *features.Features { return env.exec.block.Features }, + currentEpoch: func() uint64 { return env.exec.block.Epoch }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(slot uint64) (solana.Hash, bool) { + if slot == 41 { + return h.parentID, true + } + return solana.Hash{}, false + }, + }) + env.exec.parentBankSysvars = &sealevel.BankSysvars{} + h.open() + return h +} + +// open installs the stream the way openStream does after its bank opened. +func (h *streamingTestHarness) open() { + h.exec.current = &streamingSlot{ + slot: h.env.exec.block.Slot, + generation: h.gen, + parentSlot: 41, + parentID: h.parentID, + exec: h.env.exec, + pending: make(map[uint32]*turbine.StreamBatch), + openedAt: time.Now(), + headerAt: time.Now(), + } + h.exec.handleEvent(h.event(h.header())) +} + +func (h *streamingTestHarness) header() *turbine.StreamBatch { + return turbine.NewDetachedStreamMarker(h.gen, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) +} + +func (h *streamingTestHarness) batch(t *testing.T, start, end uint32, txs []*solana.Transaction) *turbine.StreamBatch { + t.Helper() + return turbine.NewDetachedStreamBatch(h.gen, start, end, txs, verifiedIdentities(t, txs)) +} + +func (h *streamingTestHarness) event(batch *turbine.StreamBatch) turbine.StreamEvent { + return turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: batch.Slot, Generation: batch.Generation, Batch: batch} +} + +func (h *streamingTestHarness) executed() []*solana.Transaction { return h.env.exec.transactions } + +func (h *streamingTestHarness) discardReason() string { + return metrics.GlobalBlockReplay.StreamingExecution.DiscardReason +} + +func sameTransactions(t *testing.T, want, got []*solana.Transaction) { + t.Helper() + require.Len(t, got, len(want)) + for i := range want { + require.Same(t, want[i], got[i], "transaction %d", i) + } +} + +func TestStreamingConsumeExecutesContiguousGroupsInOrder(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 500) + a, bb, c := h.batch(t, 1, 3, txs[0:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:9]) + + require.Equal(t, uint32(1), h.exec.current.nextStart, "header consumed") + h.exec.handleEvent(h.event(bb)) + require.Empty(t, h.executed(), "a gap before the batch holds it") + h.exec.handleEvent(h.event(a)) + sameTransactions(t, txs[0:6], h.executed()) + require.Len(t, h.exec.current.groups, 1, "contiguous batches execute as one group") + require.Equal(t, uint32(7), h.exec.current.nextStart) + + // Duplicate wake-ups for consumed or pending ranges are ignored. + h.exec.handleEvent(h.event(a)) + h.exec.handleEvent(h.event(bb)) + sameTransactions(t, txs[0:6], h.executed()) + + h.exec.handleEvent(h.event(c)) + sameTransactions(t, txs, h.executed()) + require.Len(t, h.exec.current.groups, 2) + require.Equal(t, uint32(10), h.exec.current.nextStart) + require.Equal(t, uint64(9), h.env.exec.processedTxCount) +} + +func TestStreamingTickPullsBatchesMissedByDroppedWakeups(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: time.Minute} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 6, 600) + h.feed.pending[h.gen] = []*turbine.StreamBatch{h.batch(t, 4, 6, txs[3:6]), h.batch(t, 1, 3, txs[0:3])} + h.exec.handleTick() + sameTransactions(t, txs, h.executed()) + require.Len(t, h.exec.current.groups, 1) +} + +func TestStreamingConsumeHoldsUntilMinGroupBatches(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MinGroupBatches: 2} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 700) + a, bb, c := h.batch(t, 1, 3, txs[0:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:9]) + + h.exec.handleEvent(h.event(a)) + require.Empty(t, h.executed(), "one batch is below the group minimum") + require.Equal(t, uint32(1), h.exec.current.nextStart, "held batch is put back") + h.exec.handleEvent(h.event(bb)) + sameTransactions(t, txs[0:6], h.executed()) + + h.exec.handleEvent(h.event(c)) + require.Len(t, h.executed(), 6, "a lone trailing batch waits") + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: c.Slot, Generation: h.gen}) + require.True(t, h.exec.current.completed) + sameTransactions(t, txs, h.executed()) +} + +func TestStreamingDiscardsOnUpdateParentDecodeErrorAndUnverified(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 800) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs))) + sameTransactions(t, txs, h.executed()) + update := turbine.NewDetachedStreamMarker(h.gen, 4, 4, turbine.StreamMarkerUpdateParent, 40, solana.Hash{1}) + h.exec.handleEvent(h.event(update)) + require.Nil(t, h.exec.current) + require.True(t, h.env.exec.closed, "discard closes the execution") + require.Equal(t, "update_parent", h.discardReason()) + require.Equal(t, uint64(1), metrics.GlobalBlockReplay.StreamingExecution.Discarded) + + h = newStreamingTestHarness(t) + broken := h.batch(t, 1, 3, txs) + broken.Err = errors.New("bad entry") + h.exec.handleEvent(h.event(broken)) + require.Nil(t, h.exec.current) + require.Equal(t, "decode_error", h.discardReason()) + + h = newStreamingTestHarness(t) + unverified := turbine.NewDetachedStreamBatch(h.gen, 1, 3, txs, nil) + _, verified, err := unverified.WaitVerification(context.Background()) + require.NoError(t, err) + require.False(t, verified) + h.exec.handleEvent(h.event(unverified)) + require.Nil(t, h.exec.current, "a batch the verifier did not admit is never self-verified") + require.Equal(t, "unverified_batch", h.discardReason()) + require.Empty(t, h.executed()) +} + +func TestStreamingGroupFailureDiscards(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 4, 900) + h.exec.executeFn = func(exec *blockExecution, group []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerify bool) error { + require.False(t, shouldVerify, "verified batches never re-verify") + return &DuplicateTransactionMessagesError{Slot: exec.block.Slot, DuplicateCount: 1} + } + h.exec.handleEvent(h.event(h.batch(t, 1, 2, txs[:2]))) + require.Nil(t, h.exec.current) + require.Equal(t, "duplicate_message", h.discardReason()) +} + +func TestStreamingIgnoresOtherGenerationsAndHonoursCancellation(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 1000) + other := turbine.NewDetachedStreamGeneration(h.env.exec.block.Slot) + foreign := turbine.NewDetachedStreamBatch(other, 1, 3, txs, verifiedIdentities(t, txs)) + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: foreign.Slot, Generation: other, Batch: foreign}) + require.Empty(t, h.executed(), "another generation's batches are not this stream's") + require.NotNil(t, h.exec.current) + + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCancelled, Slot: foreign.Slot, Generation: other, Reason: "reset"}) + require.NotNil(t, h.exec.current, "another generation's cancellation is ignored") + + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCancelled, Slot: h.env.exec.block.Slot, Generation: h.gen, Reason: "reset"}) + require.Nil(t, h.exec.current) + require.Equal(t, "cancelled:reset", h.discardReason()) +} + +func TestStreamingTickEnforcesStatusAndAge(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: 50 * time.Millisecond} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + + h := newStreamingTestHarness(t) + h.feed.status[h.gen] = turbine.StreamGone + h.exec.handleTick() + require.Nil(t, h.exec.current) + require.Equal(t, "gone", h.discardReason()) + + h = newStreamingTestHarness(t) + h.exec.current.openedAt = time.Now().Add(-time.Second) + h.exec.handleTick() + require.Nil(t, h.exec.current) + require.Equal(t, "timeout", h.discardReason()) + + h = newStreamingTestHarness(t) + h.feed.status[h.gen] = turbine.StreamDone + h.exec.current.openedAt = time.Now().Add(-100 * time.Millisecond) + h.exec.handleTick() + require.NotNil(t, h.exec.current, "a completed stream outlives MaxOpenAge while its block is emitted") + require.True(t, h.exec.current.completed) + h.exec.current.openedAt = time.Now().Add(-time.Minute) + h.exec.handleTick() + require.Nil(t, h.exec.current, "but not the hard cap") + require.Equal(t, "timeout", h.discardReason()) +} + +func TestStreamingDiscardUndoesGlobalSideEffects(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + slot := h.env.exec.block.Slot + slotCtx := h.env.exec.slotCtx + slotCtx.DeferVoteCachePublication = true + voteKey := solana.PublicKey{9} + putVoteCacheItem(slotCtx, voteKey, &sealevel.VoteStateVersions{}) + markSlotVoteStakeDirty(slotCtx) + require.Nil(t, global.VoteCacheItem(voteKey), "deferred put stays off the global cache") + before := len(global.PendingStakeEntriesSnapshot()) + global.EnqueuePendingStakePubkey(slot, solana.PublicKey{8}) + require.Len(t, global.PendingStakeEntriesSnapshot(), before+1) + + h.exec.discard("test") + require.Nil(t, slotCtx.PendingVoteCache) + require.False(t, slotCtx.VoteStakeDirty) + require.Nil(t, global.VoteCacheItem(voteKey)) + require.Len(t, global.PendingStakeEntriesSnapshot(), before, "the stream's stake index entries are dropped") + require.False(t, h.exec.matches(slot)) + require.Nil(t, h.exec.tick(), "no poll timer while idle") +} + +func TestStreamingEligibility(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + h.exec.discard("reset for eligibility") + header := h.header() + require.Equal(t, "", h.exec.eligibility(header)) + + cases := []struct { + name string + mutate func() + want string + }{ + {"generation gone", func() { h.feed.status[h.gen] = turbine.StreamGone }, "generation no longer active"}, + {"generation done", func() { h.feed.status[h.gen] = turbine.StreamDone }, "generation no longer active"}, + {"no parent context", func() { h.lastCtx = nil }, "no executed parent context"}, + {"frontier moved", func() { h.frontier = 42 }, "slot 42 on parent 41 does not extend the executed frontier 42 (parent context 41)"}, + {"parent is not the executed slot", func() { h.lastCtx = &sealevel.SlotCtx{Slot: 40} }, "slot 42 on parent 41 does not extend the executed frontier 41 (parent context 40)"}, + {"parent id mismatch", func() { h.parentID = solana.Hash{1} }, "parent block id does not match the executed parent"}, + {"switch pending", func() { h.exec.deps.switchPending = func() bool { return true } }, "fork switch pending"}, + {"epoch boundary", func() { h.exec.deps.currentEpoch = func() uint64 { return 99 } }, "epoch boundary"}, + {"rewards", func() { h.exec.deps.rewardsInFlight = func() bool { return true } }, "partitioned rewards in flight"}, + {"no features", func() { h.exec.deps.currentFeatures = func() *features.Features { return nil } }, "no feature set"}, + {"no tail", func() { h.exec.deps.tail = nil }, "requires alpenglow rooted-durable replay"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + saved := *h + savedDeps := h.exec.deps + savedStatus := h.feed.status[h.gen] + tc.mutate() + require.Equal(t, tc.want, h.exec.eligibility(header)) + *h = saved + h.exec.deps = savedDeps + h.feed.status[h.gen] = savedStatus + }) + } +} + +func TestStreamingRememberHeaderAndTryOpenBounds(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + h.exec.discard("idle") + + stale := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(41), 0, 0, turbine.StreamMarkerHeader, 40, solana.Hash{}) + h.exec.handleEvent(h.event(stale)) + require.Empty(t, h.exec.headers, "headers at or below the frontier are not kept") + + ahead := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(44), 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{}) + h.exec.handleEvent(h.event(ahead)) + require.Contains(t, h.exec.headers, uint64(44), "a header ahead of the frontier waits for its parent") + require.Nil(t, h.exec.current) + + // The next slot's header is ineligible (its generation is unknown to the + // feed), so it is dropped rather than opened; nothing else changes. + next := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(42), 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.exec.handleEvent(h.event(next)) + require.Nil(t, h.exec.current) + require.NotContains(t, h.exec.headers, uint64(42)) + require.Contains(t, h.exec.headers, uint64(44)) + + h.frontier = 44 + h.exec.handleTick() + require.Empty(t, h.exec.headers, "advancing the frontier past a remembered header drops it") + + StreamingExecutionCfg.Enabled = false + h.frontier = 41 + h.exec.headers[42] = next + h.exec.tryOpen() + require.Contains(t, h.exec.headers, uint64(42), "disabled: nothing opens") +} + +func (h *streamingTestHarness) matchingBlock(t *testing.T, txs []*solana.Transaction) *b.Block { + t.Helper() + shell := h.env.exec.block + block := &b.Block{ + Slot: shell.Slot, + Epoch: shell.Epoch, + ParentSlot: shell.ParentSlot, + ParentBankhash: shell.ParentBankhash, + Features: shell.Features, + FromLiveStream: true, + SourceParentSlot: 41, + AlpenglowParentBlockID: h.parentID, + HasAlpenglowParentBlockID: true, + Transactions: txs, + } + return block +} + +func TestStreamingHandshake(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 4, 1100) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs[:3]))) + sameTransactions(t, txs[:3], h.executed()) + sysvars := h.env.exec.parentBankSysvars + + require.Equal(t, "", h.exec.handshake(h.matchingBlock(t, txs), sysvars), "the block extends the executed prefix") + require.Equal(t, "", h.exec.handshake(h.matchingBlock(t, txs[:3]), sysvars), "the block may end exactly at the prefix") + + cases := []struct { + name string + mutate func(block *b.Block) + want string + }{ + {"slot", func(block *b.Block) { block.Slot++ }, "slot"}, + {"skipped", func(block *b.Block) { block.IsSkipped = true }, "not a live block"}, + {"rpc block", func(block *b.Block) { block.FromLiveStream = false }, "not a live block"}, + {"parent id", func(block *b.Block) { block.AlpenglowParentBlockID = [32]byte{1} }, "parent"}, + {"parent slot", func(block *b.Block) { block.SourceParentSlot = 40 }, "parent"}, + {"configured parent", func(block *b.Block) { block.ParentBankhash = [32]byte{2} }, "configured parent"}, + {"features", func(block *b.Block) { block.Features = features.NewFeaturesDefault() }, "features"}, + {"epoch", func(block *b.Block) { block.Epoch++ }, "epoch"}, + {"epoch accounts", func(block *b.Block) { block.EpochUpdatedAccts = append(block.EpochUpdatedAccts, nil) }, "epoch account updates"}, + {"shorter", func(block *b.Block) { block.Transactions = block.Transactions[:2] }, "shorter than executed prefix"}, + {"different transaction", func(block *b.Block) { + replacement := transferTransactions(t, 1, 1101)[0] // same bytes as txs[1], different object + block.Transactions[1] = replacement + }, "transaction 1 identity"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + block := h.matchingBlock(t, append([]*solana.Transaction(nil), txs...)) + tc.mutate(block) + require.Equal(t, tc.want, h.exec.handshake(block, sysvars)) + }) + } + require.Equal(t, "parent sysvars", h.exec.handshake(h.matchingBlock(t, txs), &sealevel.BankSysvars{})) + require.Equal(t, "parent sysvars", h.exec.handshake(h.matchingBlock(t, txs), nil)) + h.feed.status[h.gen] = turbine.StreamGone + require.Equal(t, "generation gone", h.exec.handshake(h.matchingBlock(t, txs), sysvars)) +} + +func TestStreamingFinalizeFallsBackOnMismatch(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 1200) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, txs))) + block := h.matchingBlock(t, txs) + block.Transactions[0] = transferTransactions(t, 1, 1200)[0] + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.NoError(t, err) + require.False(t, ok, "the caller must execute the block whole") + require.Nil(t, slotCtx) + require.Nil(t, h.exec.current) + require.Equal(t, "prefix_mismatch:transaction 0 identity", h.discardReason()) + require.True(t, h.env.exec.closed) + + _, ok, err = h.exec.finalize(block, nil) + require.NoError(t, err) + require.False(t, ok, "no stream, nothing to finalize") + require.False(t, h.exec.matches(block.Slot)) +} + +// recordingStreamer is the wait loop's view of an executor. +type recordingStreamer struct { + ch chan turbine.StreamEvent + tickCh chan time.Time + events []turbine.StreamEvent + ticks int +} + +func (r *recordingStreamer) events() <-chan turbine.StreamEvent { return r.ch } +func (r *recordingStreamer) tick() <-chan time.Time { return r.tickCh } +func (r *recordingStreamer) handleEvent(e turbine.StreamEvent) { r.events = append(r.events, e) } +func (r *recordingStreamer) handleTick() { r.ticks++ } + +func TestWaitForReplayInputDispatchesStreamWakeups(t *testing.T) { + streamer := &recordingStreamer{ch: make(chan turbine.StreamEvent, 4), tickCh: make(chan time.Time, 4)} + block := &b.Block{Slot: 5} + script := []blockstream.ReplayInput{ + {StreamEvent: &turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: 6}}, + {StreamTick: true}, + {DecisionChanged: true}, + {StreamEvent: &turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: 6}}, + {Block: block}, + } + var seenEvents []<-chan turbine.StreamEvent + var seenTicks []<-chan time.Time + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + seenEvents = append(seenEvents, events) + seenTicks = append(seenTicks, tick) + in := script[0] + script = script[1:] + return in + } + sweeps := 0 + sweep := func() *CertifiedSwitch { sweeps++; return nil } + got, parentSwitch, sw := waitForReplayInput(context.Background(), next, sweep, make(chan struct{}), time.Second, streamer) + require.Same(t, block, got) + require.Nil(t, parentSwitch) + require.Nil(t, sw) + require.Len(t, streamer.events, 2, "feed wake-ups are handled and never end the wait") + require.Equal(t, turbine.StreamCompleted, streamer.events[1].Kind) + require.Equal(t, 2, streamer.ticks, "one tick on entry, one from the timer") + require.Equal(t, 5, sweeps, "every wait is preceded by a sweep") + for _, ch := range seenEvents { + require.Equal(t, (<-chan turbine.StreamEvent)(streamer.ch), ch) + } + for _, ch := range seenTicks { + require.Equal(t, (<-chan time.Time)(streamer.tickCh), ch) + } +} + +func TestWaitForReplayInputStreamerWithoutSweepBlocksWithoutPolling(t *testing.T) { + streamer := &recordingStreamer{ch: make(chan turbine.StreamEvent, 1)} + calls := 0 + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + calls++ + require.Nil(t, decisionChanges, "decision wake-ups stay disabled without a sweep") + _, hasDeadline := ctx.Deadline() + require.False(t, hasDeadline, "no poll timeout without a sweep") + if calls == 1 { + return blockstream.ReplayInput{StreamTick: true} + } + return blockstream.ReplayInput{} + } + block, parentSwitch, sw := waitForReplayInput(context.Background(), next, nil, make(chan struct{}), time.Second, streamer) + require.Nil(t, block) + require.Nil(t, parentSwitch) + require.Nil(t, sw) + require.Equal(t, 2, calls) + require.Equal(t, 2, streamer.ticks) +} + +func TestWaitForReplayInputWithoutStreamerIsUnchanged(t *testing.T) { + block := &b.Block{Slot: 9} + next := func(ctx context.Context, decisionChanges <-chan struct{}, events <-chan turbine.StreamEvent, tick <-chan time.Time) blockstream.ReplayInput { + require.Nil(t, events) + require.Nil(t, tick) + require.Nil(t, decisionChanges) + return blockstream.ReplayInput{Block: block} + } + got, _, sw := waitForReplayInput(context.Background(), next, nil, make(chan struct{}), time.Second, nil) + require.Same(t, block, got) + require.Nil(t, sw) + + var nilStreamer *streamingExecutor + require.Nil(t, nilStreamer.tick()) + require.Nil(t, nilStreamer.events()) + require.False(t, nilStreamer.matches(1)) + nilStreamer.handleTick() + nilStreamer.discard("noop") + nilStreamer.discardSlot(1, "noop") + nilStreamer.shutdown() + _, ok, err := nilStreamer.finalize(block, nil) + require.False(t, ok) + require.NoError(t, err) +} + +func TestStreamingConfigDefaults(t *testing.T) { + var cfg StreamingExecutionConfig + require.Equal(t, defaultStreamingWorkers, cfg.workers(0)) + require.Equal(t, 2, cfg.workers(2)) + require.Equal(t, defaultStreamingWorkers, cfg.workers(64)) + cfg.Workers = 8 + require.Equal(t, 8, cfg.workers(0)) + require.Equal(t, 3, cfg.workers(3)) + require.Equal(t, defaultStreamingMaxAge, cfg.maxOpenAge()) + cfg.MaxOpenAge = time.Second + require.Equal(t, time.Second, cfg.maxOpenAge()) +} From 14677781980335810e4eff1b2c8ca97bfed005be Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 07:41:04 +0000 Subject: [PATCH 043/111] replay: keep every transaction of the group-equivalence fixture processable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex's phase-1 run failed TestExecuteTransactionGroupMatchesWholeBlock in the sequential reference: with a 5,000,000-lamport payer and 48 transfers, the failed transfers' fees eventually drop the payer below its rent-exempt minimum, after which ProcessTransaction reports InsufficientFundsForFee with a nil fee — an unprocessable transaction a valid block never contains, and which the group executor (like the whole-block loops) treats as a replay invariant violation. Fund the payer so two transfers succeed and the other 46 fail while every fee stays payable, and report the ProcessTransaction error when the reference hits a nil fee so the next such failure is diagnosable from the test output. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/block_execution_test.go | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/pkg/replay/block_execution_test.go b/pkg/replay/block_execution_test.go index 93ca05a23..5bcff38ba 100644 --- a/pkg/replay/block_execution_test.go +++ b/pkg/replay/block_execution_test.go @@ -192,9 +192,12 @@ func runSequentialReference(t *testing.T, payerLamports uint64, txs []*solana.Tr } var sigverify sync.WaitGroup var out groupExecutionOutcome - for _, tx := range txs { - feeInfo, cu, _ := ProcessTransaction(env.exec.slotCtx, &sigverify, tx, nil, nil, nil, false) - require.NotNil(t, feeInfo) + for i, tx := range txs { + feeInfo, cu, err := ProcessTransaction(env.exec.slotCtx, &sigverify, tx, nil, nil, nil, false) + // A failed transfer still returns its fee; a nil fee means the + // transaction was unprocessable (fee payer below rent exemption after + // the fee, bad blockhash...), which a valid block never contains. + require.NotNil(t, feeInfo, "transaction %d unprocessable: %v", i, err) out.fees += feeInfo.TotalFee out.cu += cu out.processed++ @@ -216,10 +219,14 @@ func runSequentialReference(t *testing.T, payerLamports uint64, txs []*solana.Tr // enough that later transfers fail for insufficient funds, which exercises // fee charging on failed transactions and makes outcomes order-dependent. func TestExecuteTransactionGroupMatchesWholeBlock(t *testing.T) { - // Amounts are 999,001+ lamports each (seq%1e6+1), so with a 5 SOL-ish - // payer of 5,000,000 lamports the first few transfers succeed and the - // rest fail inside the System program while still paying their fee. - const payerLamports = 5_000_000 + // Amounts are 999,001+ lamports each (seq%1e6+1) at a 5,000-lamport fee. + // With 3,200,000 lamports the first two transfers succeed (leaving + // 1,191,997) and the remaining 46 fail — the third would drop the payer + // below its 890,880-lamport rent-exempt minimum — while still paying + // their fee, ending at 961,997: every transaction stays processable (the + // fee payer never falls below rent exemption after the fee), which is + // what a valid block guarantees and what the loaders assert. + const payerLamports = 3_200_000 txs := transferTransactions(t, 48, 999_000) reference := runSequentialReference(t, payerLamports, txs) require.Less(t, reference.payer, uint64(payerLamports)) From 580d5294770c2f354c6f1a60324fcf64a3e1057f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 07:41:04 +0000 Subject: [PATCH 044/111] turbine: stream feed test verifies with the identity-producing verifier TestStreamFeedPublishesBatchesAndCompletion asserted that WaitVerification returns the verifier's message identities, but built the verifier with a per-transaction hook, and only the production (nil-hook) path computes identities alongside signature verification. Use the production path; the assertion stands. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/turbine/stream_test.go | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index 72aea908a..79dc0ce25 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -6,7 +6,6 @@ import ( "time" "github.com/Overclock-Validator/mithril/pkg/block" - "github.com/Overclock-Validator/mithril/pkg/txverify" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" ) @@ -32,7 +31,9 @@ func nextStreamEvent(t *testing.T, ch <-chan StreamEvent, kind StreamEventKind) // completion event for the same generation. The final component (the ending // tick) is decoded by completion, never by the prefetch, so it is not fed. func TestStreamFeedPublishesBatchesAndCompletion(t *testing.T) { - v := newTransactionVerifier(2, 16, func(tx *solana.Transaction) error { return txverify.VerifyTransaction(tx) }) + // The production verifier (nil hook) is the one that attaches message + // identities; a per-transaction hook verifies without producing them. + v := newTransactionVerifier(2, 16, nil) defer v.closeAndWait() a := NewSlotAssembler() p := newEntryPrefetchPool(context.Background(), a, v) From b85caf96d7c0fb9cc0d2e9c5decc442183d61f87 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 07:41:04 +0000 Subject: [PATCH 045/111] replay: streaming finalize keeps ownership of its bank until the tail commits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex's phase-2 review: finalize cleared the stream before the footer check and the suffix, so a failure there closed the execution but could no longer evict tracked program-cache insertions, drop the slot's stake entries or restore the legacy sysvar cache — the discard contract did not hold on the post-handshake error paths. finalize now keeps s.current until exec.finalize() has returned; every failure after the handshake (footer clock, suffix execution, processed counts, tail) goes through discard with a "finalize:" reason and returns the final error. The deferred vote cache is published at the same point whole-block execution would have written it — after the suffix, just before the tail — so a tail failure leaves the footprint a whole-block tail failure leaves (dirty marker set, rooted-checkpoint re-replay on recovery) and nothing else. discard now always consumes the tracked program-cache adds (evicting them when an AccountsDb is attached) and clears tracking. The suffix runs through the same execute hook as groups so tests can fail it. Tests: suffix failure, footer rejection and processed-count mismatch after a valid executed prefix each assert the full undo (execution closed, sysvar cache restored once, vote-cache entry never published, program cache adds consumed, stake entries dropped, no poll timer); the wait-loop test double's recorded-events field no longer collides with the interface method. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/streaming.go | 62 ++++++++++----- pkg/replay/streaming_test.go | 143 ++++++++++++++++++++++++++++++----- 2 files changed, 167 insertions(+), 38 deletions(-) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index bce9ba920..a8697f1c6 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -562,8 +562,9 @@ func (s *streamingExecutor) discard(reason string) { if exec != nil { exec.close() if exec.slotCtx != nil { - if s.deps.acctsDb != nil { - for _, key := range exec.slotCtx.TakeProgramCacheAdds() { + exec.slotCtx.TrackProgramCacheAdds = false + for _, key := range exec.slotCtx.TakeProgramCacheAdds() { + if s.deps.acctsDb != nil { s.deps.acctsDb.RemoveProgramFromCache(key) } } @@ -591,8 +592,8 @@ func (s *streamingExecutor) discardSlot(slot uint64, reason string) { } } -// finalizeErr marks a failure after the handshake passed; the block is as -// invalid as it would have been for the whole-block path. +// streamingFinalizeError marks a failure after the handshake passed; the +// block is as invalid as it would have been for the whole-block path. type streamingFinalizeError struct{ err error } func (e *streamingFinalizeError) Error() string { return e.err.Error() } @@ -601,7 +602,19 @@ func (e *streamingFinalizeError) Unwrap() error { return e.err } // finalize completes execution of block on the open stream. ok reports // whether the stream matched the block; when it did not, the stream has been // discarded and the caller must execute the block whole. A non-nil error with -// ok == true is a failure after the handshake and is final. +// ok == true is a failure after the handshake and is final for the block, +// exactly as a ProcessBlock error is. +// +// Ownership: the stream keeps owning its bank (s.current) until the tail has +// committed, so every failure path after the handshake goes through the same +// discard as a pre-handshake mismatch — program-cache insertions evicted, +// unpublished vote-cache entries dropped, slot-keyed stake entries dropped, +// the legacy sysvar cache restored, the execution closed. The one publication +// that precedes the tail is the deferred vote cache, applied at the point +// where whole-block execution would already have written it (before fees, +// rent, footer and bank hash); a failure inside the tail therefore leaves the +// same footprint a whole-block tail failure leaves, and the dirty marker it +// sets is what forces the rooted-checkpoint re-replay on recovery. func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel.BankSysvars) (slotCtx *sealevel.SlotCtx, ok bool, err error) { if s == nil || s.current == nil || block == nil { return nil, false, nil @@ -618,7 +631,8 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel } // Whole-block plan and status validation, exactly as ProcessBlock does - // them, now that the authoritative block exists. + // them, now that the authoritative block exists. A failure here is not yet + // a verdict on the block: the whole-block path re-derives it. if err := validateBlockTransactionVersions(block); err != nil { s.discard("versions") return nil, false, nil @@ -652,12 +666,13 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel } }() - // From here on the stream is committed to this block: switch the - // execution to the complete block object, execute the unexecuted suffix - // with the same machinery, publish what was deferred, and run the tail. - s.current = nil + // From here on the stream is committed to this block: any failure is the + // block's failure. fail undoes the stream's side effects and reports it. s.stopTicker() - defer exec.close() + fail := func(reason string, err error) (*sealevel.SlotCtx, bool, error) { + s.discard("finalize:" + reason) + return nil, true, &streamingFinalizeError{err: err} + } block.FeeRateGovernor = exec.block.FeeRateGovernor block.VoteTimestamps = exec.slotCtx.VoteTimestamps exec.block = block @@ -665,32 +680,39 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel exec.slotCtx.Epoch = block.Epoch if requireAlpenglowBlockFooter(block, exec.slotCtx, s.deps.alpenglowClock) { if err := validateAlpenglowFooterNanosecondClock(exec.slotCtx, block); err != nil { - return nil, true, &streamingFinalizeError{err: err} + return fail("footer_clock", err) } } if suffix := block.Transactions[executed:]; len(suffix) > 0 { started := time.Now() - if err := exec.executeTransactionGroup(suffix, executionPlan.messageIdentities.Slice(executed, len(block.Transactions)), !block.TransactionSignaturesVerified()); err != nil { - return nil, true, &streamingFinalizeError{err: fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)} + err := s.executeFn(exec, suffix, executionPlan.messageIdentities.Slice(executed, len(block.Transactions)), !block.TransactionSignaturesVerified()) + if err != nil { + return fail("suffix", fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)) } cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(suffix)}) } if exec.processedSignatures != executionPlan.processedSignatures || exec.processedTxCount != executionPlan.processedTxCount { - return nil, true, &streamingFinalizeError{err: fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", - block.Slot, exec.processedTxCount, exec.processedSignatures, executionPlan.processedTxCount, executionPlan.processedSignatures)} + return fail("counts", fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", + block.Slot, exec.processedTxCount, exec.processedSignatures, executionPlan.processedTxCount, executionPlan.processedSignatures)) } exec.slotCtx.NumSignatures = executionPlan.processedSignatures - exec.slotCtx.TrackProgramCacheAdds = false - exec.slotCtx.TakeProgramCacheAdds() - publishDeferredVoteCache(exec.slotCtx) + // Acceptance of the executed transactions: publish what execution would + // have published as it ran, then run the unchanged tail. Program-cache + // insertions stay tracked until the tail commits so a tail failure can + // still evict them. + publishDeferredVoteCache(exec.slotCtx) exec.executionPlan = executionPlan exec.statusPreparation = statusPreparation exec.statusValidation = statusValidation slotCtx, err = exec.finalize() if err != nil { - return nil, true, &streamingFinalizeError{err: err} + return fail("tail", err) } + exec.slotCtx.TrackProgramCacheAdds = false + exec.slotCtx.TakeProgramCacheAdds() + s.current = nil + exec.close() // The per-block record is rebuilt from the stream's own bookkeeping: the // loop resets the collector between waits, so counters accumulated while diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index d3730cec7..439cee3db 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -101,17 +101,18 @@ func newStreamingTestHarness(t *testing.T) *streamingTestHarness { h.lastCtx = &sealevel.SlotCtx{Slot: 41, Epoch: env.exec.block.Epoch} epochSchedule := sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000, LeaderScheduleSlotOffset: 432000, FirstNormalEpoch: 0, FirstNormalSlot: 0} h.exec = newStreamingExecutor(streamingDeps{ - feed: feed, - epochSchedule: &epochSchedule, - tail: fakeUnrootedState{}, - alpenglowMode: true, - unrootedTailUsed: true, - lastSlotCtx: func() *sealevel.SlotCtx { return h.lastCtx }, - frontier: func() uint64 { return h.frontier }, - currentFeatures: func() *features.Features { return env.exec.block.Features }, - currentEpoch: func() uint64 { return env.exec.block.Epoch }, - rewardsInFlight: func() bool { return false }, - switchPending: func() bool { return false }, + feed: feed, + epochSchedule: &epochSchedule, + tail: fakeUnrootedState{}, + transactionStatuses: NewTransactionStatusCache(), + alpenglowMode: true, + unrootedTailUsed: true, + lastSlotCtx: func() *sealevel.SlotCtx { return h.lastCtx }, + frontier: func() uint64 { return h.frontier }, + currentFeatures: func() *features.Features { return env.exec.block.Features }, + currentEpoch: func() uint64 { return env.exec.block.Epoch }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, executedBlockID: func(slot uint64) (solana.Hash, bool) { if slot == 41 { return h.parentID, true @@ -501,17 +502,123 @@ func TestStreamingFinalizeFallsBackOnMismatch(t *testing.T) { require.False(t, h.exec.matches(block.Slot)) } +// finalizeFailureHarness prepares a stream whose executed prefix matches the +// block (three of four transfers executed) and instruments every undo hook, +// so a post-handshake failure can be checked for the discard contract. +type finalizeFailureHarness struct { + *streamingTestHarness + txs []*solana.Transaction + restores int + voteKey solana.PublicKey + stakeBefore int + openedPending int +} + +func newFinalizeFailureHarness(t *testing.T) *finalizeFailureHarness { + t.Helper() + h := &finalizeFailureHarness{streamingTestHarness: newStreamingTestHarness(t), voteKey: solana.PublicKey{3, 3, 3}} + h.txs = transferTransactions(t, 4, 1300) + h.exec.handleEvent(h.event(h.batch(t, 1, 3, h.txs[:3]))) + sameTransactions(t, h.txs[:3], h.executed()) + h.exec.current.restoreSysvarCache = func() { h.restores++ } + slotCtx := h.env.exec.slotCtx + slotCtx.DeferVoteCachePublication = true + slotCtx.TrackProgramCacheAdds = true + putVoteCacheItem(slotCtx, h.voteKey, &sealevel.VoteStateVersions{}) + slotCtx.RecordProgramCacheAdd(solana.PublicKey{4}) + h.stakeBefore = len(global.PendingStakeEntriesSnapshot()) + global.EnqueuePendingStakePubkey(h.env.exec.block.Slot, solana.PublicKey{5}) + return h +} + +func (h *finalizeFailureHarness) assertUndone(t *testing.T, reason string) { + t.Helper() + require.Nil(t, h.exec.current, "the stream no longer owns a bank") + require.True(t, h.env.exec.closed, "the execution is closed") + require.Equal(t, reason, h.discardReason()) + require.Equal(t, 1, h.restores, "the legacy sysvar cache is restored once") + require.Nil(t, global.VoteCacheItem(h.voteKey), "unpublished vote-cache entries never reach the global cache") + require.Nil(t, h.env.exec.slotCtx.PendingVoteCache) + require.Empty(t, h.env.exec.slotCtx.TakeProgramCacheAdds(), "tracked program-cache adds were consumed by the undo") + require.Len(t, global.PendingStakeEntriesSnapshot(), h.stakeBefore, "the slot's stake index entries are dropped") + require.Nil(t, h.exec.tick()) +} + +func TestStreamingFinalizeSuffixFailureUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + suffixErr := errors.New("suffix exploded") + calls := 0 + h.exec.executeFn = func(exec *blockExecution, group []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerify bool) error { + calls++ + sameTransactions(t, h.txs[3:], group) + require.Equal(t, 1, identities.Len()) + require.True(t, shouldVerify, "an unmarked block's suffix is verified like any whole block") + return suffixErr + } + block := h.matchingBlock(t, h.txs) + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok, "the handshake passed: the failure is the block's") + require.Nil(t, slotCtx) + require.ErrorIs(t, err, suffixErr) + var final *streamingFinalizeError + require.ErrorAs(t, err, &final) + require.Equal(t, 1, calls) + h.assertUndone(t, "finalize:suffix") +} + +func TestStreamingFinalizeFooterRejectionUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + // An Alpenglow bank requires the footer before it executes the suffix; a + // live block without one is rejected exactly as ProcessBlock rejects it. + h.exec.deps.alpenglowClock = true + h.env.exec.block.Features.EnableFeature(features.AlpenglowDevContext, 0) + h.exec.executeFn = func(*blockExecution, []*solana.Transaction, *b.PreparedTransactionMessageIdentities, bool) error { + t.Fatal("the suffix must not execute after a footer rejection") + return nil + } + block := h.matchingBlock(t, h.txs) + require.False(t, block.HasAlpenglowFooter) + + slotCtx, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok) + require.Nil(t, slotCtx) + require.ErrorContains(t, err, "missing block footer") + h.assertUndone(t, "finalize:footer_clock") +} + +func TestStreamingFinalizeProcessedCountMismatchUndoesTheStream(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newFinalizeFailureHarness(t) + // A suffix that "succeeds" without recording its transactions leaves the + // executed counts short of the whole-block plan. + h.exec.executeFn = func(*blockExecution, []*solana.Transaction, *b.PreparedTransactionMessageIdentities, bool) error { + return nil + } + block := h.matchingBlock(t, h.txs) + + _, ok, err := h.exec.finalize(block, h.env.exec.parentBankSysvars) + require.True(t, ok) + require.ErrorContains(t, err, "processed 3 transactions") + h.assertUndone(t, "finalize:counts") +} + // recordingStreamer is the wait loop's view of an executor. type recordingStreamer struct { - ch chan turbine.StreamEvent - tickCh chan time.Time - events []turbine.StreamEvent - ticks int + ch chan turbine.StreamEvent + tickCh chan time.Time + handled []turbine.StreamEvent + ticks int } func (r *recordingStreamer) events() <-chan turbine.StreamEvent { return r.ch } func (r *recordingStreamer) tick() <-chan time.Time { return r.tickCh } -func (r *recordingStreamer) handleEvent(e turbine.StreamEvent) { r.events = append(r.events, e) } +func (r *recordingStreamer) handleEvent(e turbine.StreamEvent) { r.handled = append(r.handled, e) } func (r *recordingStreamer) handleTick() { r.ticks++ } func TestWaitForReplayInputDispatchesStreamWakeups(t *testing.T) { @@ -539,8 +646,8 @@ func TestWaitForReplayInputDispatchesStreamWakeups(t *testing.T) { require.Same(t, block, got) require.Nil(t, parentSwitch) require.Nil(t, sw) - require.Len(t, streamer.events, 2, "feed wake-ups are handled and never end the wait") - require.Equal(t, turbine.StreamCompleted, streamer.events[1].Kind) + require.Len(t, streamer.handled, 2, "feed wake-ups are handled and never end the wait") + require.Equal(t, turbine.StreamCompleted, streamer.handled[1].Kind) require.Equal(t, 2, streamer.ticks, "one tick on entry, one from the timer") require.Equal(t, 5, sweeps, "every wait is preceded by a sweep") for _, ch := range seenEvents { From ad5051111846f5b2649d73ee6e9b37bfa43327d3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 08:01:17 +0000 Subject: [PATCH 046/111] =?UTF-8?q?replay:=20lifecycle=20equivalence=20tes?= =?UTF-8?q?t=20=E2=80=94=20streamed=20bank=20equals=20whole-block=20bank?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex asked for the full open -> groups -> finalize gate rather than the group-executor comparison alone. Over a self-contained bank (an in-memory unrooted tail, an explicit parent BankSysvars snapshot with all eight lifecycle sysvars, rent rewrites skipped, an AccountsDB referenced only for its directory), the same 12-transfer block — two succeed, ten fail for rent and still pay — is executed by ProcessBlock (sequential and with four workers) and by the streaming path: bank opened on a transaction-less shell, transactions fed through the executor as one group, three groups, twelve single-transaction groups, two groups plus a three-transaction suffix, and entirely as the finalize suffix. Every run must produce the same bank hash, the same committed account delta (keys, lamports, owner, data), the same signature/compute/burn totals and the same transaction count increment. A second test executes a prefix on a stream, discards it, and shows the tail and durable view untouched and the whole-block result unchanged. The test installs the per-worker borrowed-account arena slots the node normally creates at startup (parallelTxLoop indexes them unguarded) and restores the process-global sysvar cache afterwards. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/streaming_lifecycle_test.go | 385 +++++++++++++++++++++++++ 1 file changed, 385 insertions(+) create mode 100644 pkg/replay/streaming_lifecycle_test.go diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go new file mode 100644 index 000000000..a50aa9697 --- /dev/null +++ b/pkg/replay/streaming_lifecycle_test.go @@ -0,0 +1,385 @@ +package replay + +import ( + "context" + "sort" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/arena" + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// The lifecycle equivalence gate: the same block executed whole by +// ProcessBlock and executed as a stream (bank opened on a transaction-less +// shell, groups fed through the executor, finalize against the complete +// block) must produce the same bank hash, the same committed account delta, +// the same signature/fee/compute totals and the same status publication. +// The bank is self-contained: an in-memory tail stands in for the unrooted +// working set, the parent bank sysvars are an explicit snapshot, rent +// rewrites are skipped (the rent scan needs a real AccountsDB), and the +// AccountsDB is only referenced for its directory. + +// lifecycleTail is an unrooted working set over a fixed durable memory: reads +// resolve to clones (or a zero-lamport placeholder), commits are recorded. +type lifecycleTail struct { + unrootedState + durable accounts.MemAccounts + added []lifecycleCommit +} + +type lifecycleCommit struct { + slot uint64 + delta []*accounts.Account + bankhash []byte +} + +func (t *lifecycleTail) GetAccount(_ uint64, pubkey solana.PublicKey) (*accounts.Account, error) { + if acct, err := t.durable.GetAccountWithoutLock(pubkey); err == nil { + return acct.Clone(), nil + } + return &accounts.Account{Key: pubkey}, nil +} + +func (t *lifecycleTail) GetAccountsBatch(_ context.Context, slot uint64, pks []solana.PublicKey) ([]*accounts.Account, error) { + out := make([]*accounts.Account, len(pks)) + for i, pk := range pks { + out[i], _ = t.GetAccount(slot, pk) + } + return out, nil +} + +func (t *lifecycleTail) Add(slot uint64, delta []*accounts.Account, bankhash []byte) { + t.added = append(t.added, lifecycleCommit{slot: slot, delta: delta, bankhash: append([]byte(nil), bankhash...)}) +} + +func (t *lifecycleTail) OverCap() bool { return false } + +type lifecycleEnv struct { + feats *features.Features + durable accounts.MemAccounts + acctsDb *accountsdb.AccountsDb + epochSchedule *sealevel.SysvarEpochSchedule + parent *sealevel.BankSysvars +} + +const ( + lifecycleParentSlot = uint64(7) + lifecycleSlot = uint64(8) +) + +var lifecycleParentBlockID = solana.Hash{0xAA, 0xBB} + +// ensureBorrowedAccountArenas gives parallelTxLoop the per-worker arena slots +// the node installs at startup (nil arenas are accepted by ProcessTransaction). +func ensureBorrowedAccountArenas(t *testing.T, n int) { + t.Helper() + if len(sealevel.BorrowedAccountArenas) >= n { + return + } + prev := sealevel.BorrowedAccountArenas + sealevel.BorrowedAccountArenas = make([]*arena.Arena[sealevel.BorrowedAccount], n) + t.Cleanup(func() { sealevel.BorrowedAccountArenas = prev }) +} + +func newLifecycleEnv(t *testing.T) *lifecycleEnv { + t.Helper() + ensureBorrowedAccountArenas(t, 4) + // Bank open publishes derived sysvars to the legacy process-global cache; + // leave it as we found it for the rest of the package. + sysvarCacheBefore := sealevel.SysvarCache + t.Cleanup(func() { sealevel.SysvarCache = sysvarCacheBefore }) + feats := features.NewFeaturesDefault() + feats.EnableFeature(features.FormalizeLoadedTransactionDataSize, 0) + feats.EnableFeature(features.SkipRentRewrites, 0) + + durable := accounts.NewMemAccounts() + _ = durable.SetAccountWithoutLock(addresses.SystemProgramAddr, &accounts.Account{ + Key: addresses.SystemProgramAddr, Lamports: 1, Owner: addresses.NativeLoaderAddr, Executable: true, RentEpoch: ^uint64(0), + }) + _ = durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: 3_200_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = durable.SetAccountWithoutLock(txfixture.DestPubkey(), &accounts.Account{ + Key: txfixture.DestPubkey(), Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + + // The parent bank's sysvar snapshot, as the retained parent context would + // hold it. RecentBlockhashes carries the fixture blockhash so the transfers + // are age-valid. + clock := sealevel.SysvarClock{Slot: lifecycleParentSlot, EpochStartTimestamp: 111, UnixTimestamp: 222} + slotHashes := sealevel.SysvarSlotHashes{{Slot: lifecycleParentSlot - 1, Hash: [32]byte{0x61}}} + recent := sealevel.SysvarRecentBlockhashes{{ + Blockhash: txfixture.TestBlockhash(), + FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5_000}, + }} + slotHistory := sealevel.SysvarSlotHistory{ + Bits: sealevel.SlotHistoryBitvec{ + Bits: sealevel.SlotHistoryInner{BlocksLen: 1, Blocks: []uint64{0x81}}, + Len: 64, + }, + NextSlot: lifecycleSlot, + } + stakeHistory := sealevel.SysvarStakeHistory{{Epoch: 0, Entry: sealevel.StakeHistoryEntry{Effective: 91}}} + lastRestart := sealevel.SysvarLastRestartSlot{LastRestartSlot: 3} + epochSchedule := sealevel.SysvarEpochSchedule{SlotsPerEpoch: 100, LeaderScheduleSlotOffset: 100} + rent := sealevel.NewDefaultRentSysvar() + parent, err := sealevel.NewBankSysvars(lifecycleParentSlot, + &accounts.Account{Key: sealevel.SysvarClockAddr, Lamports: 1, Data: clock.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarSlotHashesAddr, Lamports: 1, Data: slotHashes.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarRecentBlockHashesAddr, Lamports: 1, Data: recent.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarSlotHistoryAddr, Lamports: 1, Data: slotHistory.MustMarshal()}, + &accounts.Account{Key: sealevel.SysvarStakeHistoryAddr, Lamports: 1, Data: marshalStakeHistoryForParentLoader(t, &stakeHistory)}, + &accounts.Account{Key: sealevel.SysvarLastRestartSlotAddr, Lamports: 1, Data: marshalLastRestartSlotForParentLoader(t, lastRestart)}, + &accounts.Account{Key: sealevel.SysvarEpochScheduleAddr, Lamports: 1, Data: marshalEpochScheduleForParentLoader(t, epochSchedule)}, + &accounts.Account{Key: sealevel.SysvarRentAddr, Lamports: 1, Data: rent.MustMarshal()}, + ) + require.NoError(t, err) + require.NoError(t, parent.ValidateForExecution()) + + return &lifecycleEnv{ + feats: feats, + durable: durable, + acctsDb: &accountsdb.AccountsDb{AcctsDir: t.TempDir()}, + epochSchedule: &epochSchedule, + parent: parent, + } +} + +// block builds the slot's block as the loop would have configured it on the +// executed parent; txs nil is the streaming shell. +func (env *lifecycleEnv) block(txs []*solana.Transaction) *b.Block { + return &b.Block{ + Slot: lifecycleSlot, + Epoch: 0, + ParentSlot: lifecycleParentSlot, + ParentBankhash: [32]byte{0x88}, + Blockhash: [32]byte{0x99}, + LastBlockhash: [32]byte{0x77}, + Features: env.feats, + Transactions: txs, + PrevFeeRateGovernor: &sealevel.FeeRateGovernor{TargetLamportsPerSignature: 5_000, LamportsPerSignature: 5_000}, + FromLiveStream: true, + SourceParentSlot: lifecycleParentSlot, + AlpenglowParentBlockID: lifecycleParentBlockID, + HasAlpenglowParentBlockID: true, + } +} + +type lifecycleOutcome struct { + bankhash []byte + numSignatures uint64 + computeUnits uint64 + lamportsBurnt uint64 + delta map[solana.PublicKey]*accounts.Account +} + +func lifecycleOutcomeOf(t *testing.T, slotCtx *sealevel.SlotCtx, tail *lifecycleTail) lifecycleOutcome { + t.Helper() + require.NotNil(t, slotCtx) + require.Len(t, tail.added, 1, "the bank commits exactly once") + require.Equal(t, lifecycleSlot, tail.added[0].slot) + require.Equal(t, slotCtx.FinalBankhash, tail.added[0].bankhash) + delta := make(map[solana.PublicKey]*accounts.Account, len(tail.added[0].delta)) + for _, acct := range tail.added[0].delta { + delta[acct.Key] = acct + } + require.Contains(t, delta, txfixture.PayerPubkey()) + require.Contains(t, delta, txfixture.DestPubkey()) + return lifecycleOutcome{ + bankhash: append([]byte(nil), slotCtx.FinalBankhash...), + numSignatures: slotCtx.NumSignatures, + computeUnits: slotCtx.TotalComputeUnitsConsumed, + lamportsBurnt: slotCtx.LamportsBurnt, + delta: delta, + } +} + +func requireSameLifecycleOutcome(t *testing.T, want, got lifecycleOutcome) { + t.Helper() + require.NotEmpty(t, want.bankhash) + require.Equal(t, want.bankhash, got.bankhash, "bank hash") + require.Equal(t, want.numSignatures, got.numSignatures) + require.Equal(t, want.computeUnits, got.computeUnits) + require.Equal(t, want.lamportsBurnt, got.lamportsBurnt) + wantKeys := make([]string, 0, len(want.delta)) + for key := range want.delta { + wantKeys = append(wantKeys, key.String()) + } + gotKeys := make([]string, 0, len(got.delta)) + for key := range got.delta { + gotKeys = append(gotKeys, key.String()) + } + sort.Strings(wantKeys) + sort.Strings(gotKeys) + require.Equal(t, wantKeys, gotKeys, "committed account set") + for key, acct := range want.delta { + other := got.delta[key] + require.Equal(t, acct.Lamports, other.Lamports, "%s lamports", key) + require.Equal(t, acct.Owner, other.Owner, "%s owner", key) + require.Equal(t, acct.Data, other.Data, "%s data", key) + require.Equal(t, acct.Executable, other.Executable, "%s executable", key) + } +} + +func lifecycleWholeBlock(t *testing.T, env *lifecycleEnv, txs []*solana.Transaction, txParallelism int) lifecycleOutcome { + t.Helper() + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + block := env.block(txs) + block.MarkTransactionSignaturesVerified() + slotCtx, err := ProcessBlock(env.acctsDb, block, env.epochSchedule, txParallelism, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, err) + return lifecycleOutcomeOf(t, slotCtx, tail) +} + +// lifecycleStream opens the bank on a transaction-less shell, feeds txs in +// the given batch splits through the executor, and finalizes against the +// complete block; suffixTxs of the transactions are never streamed and +// execute at finalize. +func lifecycleStream(t *testing.T, env *lifecycleEnv, txs []*solana.Transaction, splits []int, suffixTxs int, workers int) (lifecycleOutcome, *streamingExecutor) { + t.Helper() + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + shell := env.block(nil) + exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, workers, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, exec.open()) + exec.slotCtx.DeferVoteCachePublication = true + exec.slotCtx.TrackProgramCacheAdds = true + + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{ + feed: feed, + epochSchedule: env.epochSchedule, + tail: tail, + transactionStatuses: statuses, + alpenglowMode: true, + unrootedTailUsed: true, + frontier: func() uint64 { return lifecycleParentSlot }, + lastSlotCtx: func() *sealevel.SlotCtx { return nil }, + currentFeatures: func() *features.Features { return env.feats }, + currentEpoch: func() uint64 { return 0 }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(uint64) (solana.Hash, bool) { return lifecycleParentBlockID, true }, + }) + s.current = &streamingSlot{ + slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), + } + header := turbine.NewDetachedStreamMarker(gen, 0, 0, turbine.StreamMarkerHeader, lifecycleParentSlot, lifecycleParentBlockID) + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: header}) + + streamed := txs[:len(txs)-suffixTxs] + start, next := 0, uint32(1) + for _, end := range append(append([]int(nil), splits...), len(streamed)) { + if end <= start { + continue + } + group := streamed[start:end] + batch := turbine.NewDetachedStreamBatch(gen, next, next+uint32(len(group))-1, group, verifiedIdentities(t, group)) + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: batch}) + next += uint32(len(group)) + start = end + } + require.NotNil(t, s.current, "no group may have discarded the stream (%s)", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + sameTransactions(t, streamed, exec.transactions) + + block := env.block(txs) + block.MarkTransactionSignaturesVerified() + slotCtx, ok, err := s.finalize(block, env.parent) + require.NoError(t, err) + require.True(t, ok, "the stream must accept its own block") + require.Nil(t, s.current) + require.True(t, exec.closed) + require.Equal(t, uint64(1), metrics.GlobalBlockReplay.StreamingExecution.Opened) + require.Equal(t, uint64(len(txs)), metrics.GlobalBlockReplay.StreamingExecution.Transactions) + return lifecycleOutcomeOf(t, slotCtx, tail), s +} + +func TestStreamingLifecycleMatchesWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + // Transfers of 999,001+ lamports at a 5,000-lamport fee from a + // 3,200,000-lamport payer: the first two succeed, the rest fail for rent + // and still pay, so every group boundary is a write dependency on the + // payer and the outcome is order-dependent. + txs := transferTransactions(t, 12, 999_000) + txCountBefore := global.TransactionCount() + + whole := lifecycleWholeBlock(t, newLifecycleEnv(t), txs, 0) + require.Equal(t, txCountBefore+uint64(len(txs)), global.TransactionCount()) + wholeParallel := lifecycleWholeBlock(t, newLifecycleEnv(t), txs, 4) + requireSameLifecycleOutcome(t, whole, wholeParallel) + + cases := []struct { + name string + splits []int + suffixTxs int + workers int + }{ + {"one group, no suffix", nil, 0, 1}, + {"three groups, no suffix", []int{3, 7}, 0, 4}, + {"per-transaction groups", []int{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11}, 0, 2}, + {"two groups and a suffix", []int{4}, 3, 4}, + {"everything in the suffix", nil, 12, 4}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + before := global.TransactionCount() + streamed, _ := lifecycleStream(t, newLifecycleEnv(t), txs, tc.splits, tc.suffixTxs, tc.workers) + requireSameLifecycleOutcome(t, whole, streamed) + require.Equal(t, before+uint64(len(txs)), global.TransactionCount()) + }) + } +} + +// A stream discarded after executing groups leaves the durable view and the +// tail untouched, and the same block then executes whole to the same result. +func TestStreamingLifecycleDiscardThenWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + txs := transferTransactions(t, 8, 999_000) + env := newLifecycleEnv(t) + whole := lifecycleWholeBlock(t, env, txs, 2) + + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + shell := env.block(nil) + exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, exec.open()) + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: env.epochSchedule, tail: tail, transactionStatuses: statuses}) + s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, + Batch: turbine.NewDetachedStreamBatch(gen, 1, 5, txs[:5], verifiedIdentities(t, txs[:5]))}) + sameTransactions(t, txs[:5], exec.transactions) + payerNow, err := exec.slotCtx.GetAccountShared(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Less(t, payerNow.Lamports, uint64(3_200_000), "the overlay saw the executed prefix") + + s.discard("update_parent") + require.Empty(t, tail.added, "a discarded stream commits nothing") + durablePayer, err := env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(3_200_000), durablePayer.Lamports, "the durable view is untouched") + + again := lifecycleWholeBlock(t, env, txs, 2) + requireSameLifecycleOutcome(t, whole, again) +} From 88373b197bcf11dfef350bec1c17756530907d32 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 09:01:56 +0000 Subject: [PATCH 047/111] block: PreparedTransactionMessageIdentities.Rebind for stream-owned copies Streaming execution runs its own copies of a block's transactions so that address-table resolution never touches the block's objects; the copies must carry the identities the verifier bound to the originals. Rebind returns the same identities bound to copies of the same ordered transactions, checking each copy's message version and recent blockhash against the prepared entry, and leaves the original set untouched. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/block/message_identity.go | 30 ++++++++++++- pkg/block/message_identity_rebind_test.go | 54 +++++++++++++++++++++++ 2 files changed, 83 insertions(+), 1 deletion(-) create mode 100644 pkg/block/message_identity_rebind_test.go diff --git a/pkg/block/message_identity.go b/pkg/block/message_identity.go index df5da3a5a..ddcf75b47 100644 --- a/pkg/block/message_identity.go +++ b/pkg/block/message_identity.go @@ -1,6 +1,11 @@ package block -import "github.com/Overclock-Validator/mithril/pkg/txstatus" +import ( + "fmt" + + "github.com/Overclock-Validator/mithril/pkg/txstatus" + "github.com/gagliardetto/solana-go" +) // transactionState initializes the nonserialized holder under a short global // lock. Message hashing itself is protected only by the per-block state lock, @@ -35,6 +40,29 @@ func (prepared *PreparedTransactionMessageIdentities) MatchesBlock(block *Block) return block != nil && prepared.matches(block.Transactions) } +// Rebind returns the same identities bound to copies of the same ordered +// transactions: copies[i] must carry the message version and recent +// blockhash identity i was prepared for. Streaming execution runs +// stream-owned copies of a block's transactions (so address-table resolution +// never touches the block's own objects) while proving the block by the +// originals; the copies share the originals' identities. +func (prepared *PreparedTransactionMessageIdentities) Rebind(copies []*solana.Transaction) (*PreparedTransactionMessageIdentities, error) { + if prepared == nil || len(copies) != len(prepared.identities) || len(copies) != len(prepared.versions) { + return nil, fmt.Errorf("prepared identities do not cover %d transaction copies", len(copies)) + } + for index, tx := range copies { + if tx == nil || tx.Message.GetVersion() != prepared.versions[index] || + tx.Message.RecentBlockhash != prepared.identities[index].RecentBlockhash { + return nil, fmt.Errorf("transaction copy %d does not match its prepared identity", index) + } + } + return &PreparedTransactionMessageIdentities{ + transactions: append([]*solana.Transaction(nil), copies...), + versions: append([]solana.MessageVersion(nil), prepared.versions...), + identities: append([]txstatus.TransactionMessageIdentity(nil), prepared.identities...), + }, nil +} + // Slice returns the prepared identities for transactions [from, to) as an // independent prepared set bound to that sub-slice. func (prepared *PreparedTransactionMessageIdentities) Slice(from, to int) *PreparedTransactionMessageIdentities { diff --git a/pkg/block/message_identity_rebind_test.go b/pkg/block/message_identity_rebind_test.go new file mode 100644 index 000000000..0639b6109 --- /dev/null +++ b/pkg/block/message_identity_rebind_test.go @@ -0,0 +1,54 @@ +package block + +import ( + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestPreparedTransactionMessageIdentitiesRebindToCopies(t *testing.T) { + first, second := identityTestTransaction(1), identityTestTransaction(2) + prepared, err := (&Block{Transactions: []*solana.Transaction{first, second}}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare identities: %v", err) + } + copyOf := func(tx *solana.Transaction) *solana.Transaction { + message := tx.Message + message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) + return &solana.Transaction{Signatures: tx.Signatures, Message: message} + } + copies := []*solana.Transaction{copyOf(first), copyOf(second)} + + rebound, err := prepared.Rebind(copies) + if err != nil { + t.Fatalf("rebind: %v", err) + } + if rebound.Len() != 2 || rebound.Identity(0) != prepared.Identity(0) || rebound.Identity(1) != prepared.Identity(1) { + t.Fatalf("rebound identities differ: %+v vs %+v", rebound, prepared) + } + if !rebound.matches(copies) { + t.Fatal("rebound set is not bound to the copies") + } + if rebound.matches([]*solana.Transaction{first, second}) { + t.Fatal("rebound set must not claim the originals") + } + if !prepared.matches([]*solana.Transaction{first, second}) { + t.Fatal("rebinding must not alter the original set") + } + + if _, err := prepared.Rebind(copies[:1]); err == nil { + t.Fatal("a shorter copy list must be rejected") + } + wrongHash := copyOf(second) + wrongHash.Message.RecentBlockhash = solana.Hash{0x56} + if _, err := prepared.Rebind([]*solana.Transaction{copies[0], wrongHash}); err == nil { + t.Fatal("a copy with a different recent blockhash must be rejected") + } + if _, err := prepared.Rebind([]*solana.Transaction{copies[0], nil}); err == nil { + t.Fatal("a nil copy must be rejected") + } + var none *PreparedTransactionMessageIdentities + if _, err := none.Rebind(copies); err == nil { + t.Fatal("a nil prepared set must be rejected") + } +} From a64efd66571d1245108461838acd6eb4e2d1df53 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 09:01:57 +0000 Subject: [PATCH 048/111] replay: streams execute their own transaction copies; block objects stay pristine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex's confirmed blocker (design review point 4): executing a v0 block object resolves its address-table lookups in place — solana-go's SetAddressTables refuses a second call and ResolveLookups appends the looked-up keys to AccountKeys — so a prefix executed on a stream and then discarded would leave the block's objects resolved against the stream's parent, and whole-block execution would either fail ("address tables already set") or, if it skipped resolution, run with the wrong parent's addresses after a fork switch. The stream now executes stream-owned copies: executionCopy takes the message by value with its own account-key slice (signatures and instructions are shared and never mutated by execution) and refuses an input that is already resolved. The verifier's identities are checked against the originals, then rebound to the copies. The stream records the originals in block order (streamingSlot.origin) and the handshake proves the block by those; the finalize suffix still executes the block's own objects, as whole-block execution does, once the block is committed. Tests: executionCopy leaves the original unresolved and still resolvable; every group test now checks the bank ran copies of exactly the fed objects; lifecycle equivalence with v0 transfers whose destination is an address-table entry (two groups plus a suffix, equal to whole-block; the streamed originals remain unresolved, the suffix ones resolve as whole-block would); and the discard/fallback case Codex asked for — a v0 prefix executed against parent A's table, discarded, then the same authoritative objects executed whole against parent B whose table names a different destination: B's destination is credited, nothing of A's resolution survives, and the result equals a fresh-decoded reference on B. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/streaming.go | 66 +++++++- pkg/replay/streaming_lifecycle_test.go | 204 ++++++++++++++++++++++++- pkg/replay/streaming_test.go | 32 +++- 3 files changed, 293 insertions(+), 9 deletions(-) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index a8697f1c6..f67ae8df8 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -131,6 +131,10 @@ type streamingSlot struct { parentSlot uint64 parentID solana.Hash exec *blockExecution + // origin holds the block's own transaction objects in executed order; + // the bank executes stream-owned copies (see executionCopy), and the + // handshake proves the block by these originals. + origin []*solana.Transaction nextStart uint32 pending map[uint32]*turbine.StreamBatch footerSeen bool @@ -528,12 +532,22 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { // unauthenticated prefix must never do. return errors.New("unverified_batch") } - prepared, err := b.PrepareVerifiedTransactionMessageIdentities(txs, verified) + // The identities are bound to the block's objects by the verifier; that + // binding is checked here, on the originals, before the copies inherit it. + preparedForOriginals, err := b.PrepareVerifiedTransactionMessageIdentities(txs, verified) + if err != nil { + return fmt.Errorf("identities: %w", err) + } + copies, err := executionCopies(txs) + if err != nil { + return fmt.Errorf("copies: %w", err) + } + prepared, err := preparedForOriginals.Rebind(copies) if err != nil { return fmt.Errorf("identities: %w", err) } started := time.Now() - err = s.executeFn(cur.exec, txs, prepared, false) + err = s.executeFn(cur.exec, copies, prepared, false) cur.exec.setReplayStage("streaming_wait") if err != nil { var duplicates *DuplicateTransactionMessagesError @@ -545,10 +559,48 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { } return fmt.Errorf("group: %w", err) } + cur.origin = append(cur.origin, txs...) cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(txs)}) return nil } +// errStreamInputResolved reports a batch whose transactions already carry +// address-table resolution; the assembler never produces one, and a stream +// must not execute an object whose account keys were derived elsewhere. +var errStreamInputResolved = errors.New("stream input is already resolved") + +// executionCopy returns the object the stream executes in place of a block's +// own transaction. Execution resolves address-table lookups in place +// (SetAddressTables refuses a second call; ResolveLookups appends to +// AccountKeys), so running the block's object would leave it resolved +// against the stream's parent — and unusable, or worse, wrong, for the +// whole-block path after a discard. The copy takes the message by value with +// its own account-key slice; signatures and instructions are shared and never +// mutated by execution. +func executionCopy(tx *solana.Transaction) (*solana.Transaction, error) { + if tx == nil { + return nil, errors.New("nil transaction") + } + if tx.Message.GetVersion() == solana.MessageVersionV0 && tx.Message.IsResolved() { + return nil, errStreamInputResolved + } + message := tx.Message + message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) + return &solana.Transaction{Signatures: tx.Signatures, Message: message}, nil +} + +func executionCopies(txs []*solana.Transaction) ([]*solana.Transaction, error) { + copies := make([]*solana.Transaction, len(txs)) + for i, tx := range txs { + dup, err := executionCopy(tx) + if err != nil { + return nil, fmt.Errorf("transaction %d: %w", i, err) + } + copies[i] = dup + } + return copies, nil +} + // discard throws the in-progress stream away and undoes every side effect it // may have had outside its own SlotCtx. func (s *streamingExecutor) discard(reason string) { @@ -644,7 +696,11 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel s.discard("plan") return nil, false, nil } - executed := len(exec.transactions) + executed := len(cur.origin) + if len(exec.transactions) != executed { + s.discard("prefix_bookkeeping") + return nil, false, nil + } for i := 0; i < executed; i++ { if executionPlan.execute[i] != exec.execute[i] { s.discard("execution_mask") @@ -755,14 +811,14 @@ func (s *streamingExecutor) handshake(block *b.Block, parentBankSysvars *sealeve return "epoch" case len(block.EpochUpdatedAccts) != 0: return "epoch account updates" - case len(block.Transactions) < len(exec.transactions): + case len(block.Transactions) < len(cur.origin): return "shorter than executed prefix" } status := s.deps.feed.StreamStatusOf(cur.generation) if status == turbine.StreamGone { return "generation gone" } - for i, tx := range exec.transactions { + for i, tx := range cur.origin { if block.Transactions[i] != tx { return fmt.Sprintf("transaction %d identity", i) } diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go index a50aa9697..3c5e2dc48 100644 --- a/pkg/replay/streaming_lifecycle_test.go +++ b/pkg/replay/streaming_lifecycle_test.go @@ -1,7 +1,9 @@ package replay import ( + "bytes" "context" + "encoding/binary" "sort" "testing" "time" @@ -17,6 +19,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" "github.com/Overclock-Validator/mithril/pkg/turbine" + bin "github.com/gagliardetto/binary" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" ) @@ -196,7 +199,6 @@ func lifecycleOutcomeOf(t *testing.T, slotCtx *sealevel.SlotCtx, tail *lifecycle delta[acct.Key] = acct } require.Contains(t, delta, txfixture.PayerPubkey()) - require.Contains(t, delta, txfixture.DestPubkey()) return lifecycleOutcome{ bankhash: append([]byte(nil), slotCtx.FinalBankhash...), numSignatures: slotCtx.NumSignatures, @@ -296,7 +298,8 @@ func lifecycleStream(t *testing.T, env *lifecycleEnv, txs []*solana.Transaction, start = end } require.NotNil(t, s.current, "no group may have discarded the stream (%s)", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) - sameTransactions(t, streamed, exec.transactions) + sameTransactions(t, streamed, s.current.origin) + sameCopies(t, streamed, exec.transactions) block := env.block(txs) block.MarkTransactionSignaturesVerified() @@ -369,7 +372,8 @@ func TestStreamingLifecycleDiscardThenWholeBlock(t *testing.T) { exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: turbine.NewDetachedStreamBatch(gen, 1, 5, txs[:5], verifiedIdentities(t, txs[:5]))}) - sameTransactions(t, txs[:5], exec.transactions) + sameTransactions(t, txs[:5], s.current.origin) + sameCopies(t, txs[:5], exec.transactions) payerNow, err := exec.slotCtx.GetAccountShared(txfixture.PayerPubkey()) require.NoError(t, err) require.Less(t, payerNow.Lamports, uint64(3_200_000), "the overlay saw the executed prefix") @@ -383,3 +387,197 @@ func TestStreamingLifecycleDiscardThenWholeBlock(t *testing.T) { again := lifecycleWholeBlock(t, env, txs, 2) requireSameLifecycleOutcome(t, whole, again) } + +// V0 / address-lookup-table coverage. Resolving lookups mutates the message +// object (SetAddressTables refuses a second call; ResolveLookups appends to +// AccountKeys), so a stream must execute its own copies and leave the block's +// objects for the whole-block path — which may run them against a different +// parent, with a different table. + +var lifecycleTableKey = solana.PublicKey{0x7A, 0xB1, 0xE0} + +// lookupTableAccount is an active address lookup table whose only entry is +// dest, encoded the way the ALT program stores it. +func lookupTableAccount(t *testing.T, dest solana.PublicKey) *accounts.Account { + t.Helper() + var buf bytes.Buffer + enc := bin.NewBinEncoder(&buf) + require.NoError(t, enc.WriteUint32(sealevel.AddressLookupTableProgramStateLookupTable, bin.LE)) + authority := txfixture.PayerPubkey() + meta := sealevel.LookupTableMeta{DeactivationSlot: ^uint64(0), LastExtendedSlot: 1, Authority: &authority} + require.NoError(t, meta.MarshalWithEncoder(enc)) + require.NoError(t, enc.WriteBytes(dest[:], false)) + require.Equal(t, sealevel.AddressLookupTableMetaSize+32, buf.Len()) + return &accounts.Account{Key: lifecycleTableKey, Lamports: 1_000_000, Owner: addresses.AddressLookupTableAddr, Data: buf.Bytes(), RentEpoch: ^uint64(0)} +} + +func systemTransferData(lamports uint64) []byte { + data := make([]byte, 12) + binary.LittleEndian.PutUint32(data, 2) // SystemInstruction::Transfer + binary.LittleEndian.PutUint64(data[4:], lamports) + return data +} + +// signedV0TransferViaTableWire is a signed v0 transfer from the fixture payer +// to entry 0 of lifecycleTableKey (a writable lookup): static keys are the +// payer and the System program, so the destination is account index 2. +func signedV0TransferViaTableWire(t *testing.T, seq uint64) []byte { + t.Helper() + msg := solana.Message{ + Header: solana.MessageHeader{NumRequiredSignatures: 1, NumReadonlyUnsignedAccounts: 1}, + AccountKeys: solana.PublicKeySlice{txfixture.PayerPubkey(), solana.SystemProgramID}, + RecentBlockhash: txfixture.TestBlockhash(), + Instructions: []solana.CompiledInstruction{{ + ProgramIDIndex: 1, + Accounts: []uint16{0, 2}, + Data: systemTransferData(1_000 + seq), + }}, + AddressTableLookups: solana.MessageAddressTableLookupSlice{{AccountKey: lifecycleTableKey, WritableIndexes: []uint8{0}}}, + } + _, err := msg.SetVersion(solana.MessageVersionV0) + require.NoError(t, err) + tx := &solana.Transaction{Message: msg} + payerKey := txfixture.PayerPrivateKey() + _, err = tx.Sign(func(key solana.PublicKey) *solana.PrivateKey { + if key.Equals(txfixture.PayerPubkey()) { + return &payerKey + } + return nil + }) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + return wire +} + +// decodeTransactions decodes wires the way turbine does, so each call yields +// fresh, unresolved objects. +func decodeTransactions(t *testing.T, wires [][]byte) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + require.Equal(t, solana.MessageVersionV0, tx.Message.GetVersion()) + require.False(t, tx.Message.IsResolved()) + txs[i] = tx + } + return txs +} + +// newLifecycleEnvWithTable is newLifecycleEnv with a well-funded payer, the +// lookup table pointing at dest, and dest as an existing rent-exempt account, +// so every v0 transfer succeeds and the credited destination shows which +// table resolved it. +func newLifecycleEnvWithTable(t *testing.T, dest solana.PublicKey) *lifecycleEnv { + t.Helper() + env := newLifecycleEnv(t) + _ = env.durable.SetAccountWithoutLock(txfixture.PayerPubkey(), &accounts.Account{ + Key: txfixture.PayerPubkey(), Lamports: 10_000_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = env.durable.SetAccountWithoutLock(dest, &accounts.Account{ + Key: dest, Lamports: 10_000_000, Owner: addresses.SystemProgramAddr, RentEpoch: ^uint64(0), + }) + _ = env.durable.SetAccountWithoutLock(lifecycleTableKey, lookupTableAccount(t, dest)) + return env +} + +func TestExecutionCopyLeavesTheBlockObjectUnresolved(t *testing.T) { + original := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1)})[0] + staticKeys := len(original.Message.AccountKeys) + + dup, err := executionCopy(original) + require.NoError(t, err) + require.NotSame(t, original, dup) + require.Equal(t, original.Signatures, dup.Signatures) + require.NoError(t, dup.Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD1}}})) + require.NoError(t, dup.Message.ResolveLookups()) + require.True(t, dup.Message.IsResolved()) + require.Len(t, dup.Message.AccountKeys, staticKeys+1) + require.False(t, original.Message.IsResolved(), "resolving the copy must not resolve the original") + require.Len(t, original.Message.AccountKeys, staticKeys) + require.NoError(t, original.Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD2}}}), + "the original still accepts its own resolution") + + _, err = executionCopy(dup) + require.ErrorIs(t, err, errStreamInputResolved, "a resolved input is never accepted by a stream") + _, err = executionCopies([]*solana.Transaction{original, dup}) + require.ErrorIs(t, err, errStreamInputResolved) +} + +func TestStreamingLifecycleV0LookupsMatchWholeBlock(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + dest := solana.PublicKey{0xDA} + wires := make([][]byte, 6) + for i := range wires { + wires[i] = signedV0TransferViaTableWire(t, uint64(i)) + } + whole := lifecycleWholeBlock(t, newLifecycleEnvWithTable(t, dest), decodeTransactions(t, wires), 2) + require.Contains(t, whole.delta, dest) + require.Equal(t, uint64(10_000_000+6*1_000+0+1+2+3+4+5), whole.delta[dest].Lamports, "every transfer reached the table's entry") + + orig := decodeTransactions(t, wires) + streamed, _ := lifecycleStream(t, newLifecycleEnvWithTable(t, dest), orig, []int{2}, 2, 2) + requireSameLifecycleOutcome(t, whole, streamed) + for i := 0; i < 4; i++ { + require.False(t, orig[i].Message.IsResolved(), "streamed transaction %d ran as a copy", i) + } + for i := 4; i < 6; i++ { + require.True(t, orig[i].Message.IsResolved(), "suffix transaction %d ran as the block's own object, like whole-block execution", i) + } +} + +// A v0 prefix executed on a stream against parent A is discarded; the same +// authoritative objects then execute whole against parent B, whose table +// names a different destination. The block's objects must still resolve +// (they were never touched), B's destination must be the one credited, and +// the result must equal a fresh-decoded whole-block reference on B. +func TestStreamingLifecycleDiscardedV0PrefixResolvesAgainstTheNewParent(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + destA, destB := solana.PublicKey{0xA1}, solana.PublicKey{0xB2} + wires := make([][]byte, 6) + for i := range wires { + wires[i] = signedV0TransferViaTableWire(t, uint64(10+i)) + } + orig := decodeTransactions(t, wires) + + envA := newLifecycleEnvWithTable(t, destA) + tail := &lifecycleTail{durable: envA.durable} + statuses := NewTransactionStatusCache() + exec := newBlockExecution(envA.acctsDb, envA.block(nil), envA.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, envA.parent) + require.NoError(t, exec.open()) + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: envA.epochSchedule, tail: tail, transactionStatuses: statuses}) + s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, + Batch: turbine.NewDetachedStreamBatch(gen, 1, 4, orig[:4], verifiedIdentities(t, orig[:4]))}) + require.NotNil(t, s.current, "stream discarded: %s", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + sameTransactions(t, orig[:4], s.current.origin) + sameCopies(t, orig[:4], exec.transactions) + creditedA, err := exec.slotCtx.GetAccountShared(destA) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000+4*1_000+10+11+12+13), creditedA.Lamports, "the stream resolved through A's table") + for i, tx := range orig { + require.False(t, tx.Message.IsResolved(), "block object %d must be untouched by the stream", i) + } + + s.discard("fork_switch") + require.Empty(t, tail.added) + + envB := newLifecycleEnvWithTable(t, destB) + whole := lifecycleWholeBlock(t, envB, orig, 2) + require.Contains(t, whole.delta, destB, "the block's objects resolved through B's table") + require.NotContains(t, whole.delta, destA, "nothing of A's resolution survived") + require.Equal(t, uint64(10_000_000+6*1_000+10+11+12+13+14+15), whole.delta[destB].Lamports) + for i, tx := range orig { + require.True(t, tx.Message.IsResolved(), "block object %d was resolved by the whole-block path", i) + } + + reference := lifecycleWholeBlock(t, newLifecycleEnvWithTable(t, destB), decodeTransactions(t, wires), 2) + requireSameLifecycleOutcome(t, reference, whole) +} diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 439cee3db..692de3b47 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -153,7 +153,35 @@ func (h *streamingTestHarness) event(batch *turbine.StreamBatch) turbine.StreamE return turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: batch.Slot, Generation: batch.Generation, Batch: batch} } -func (h *streamingTestHarness) executed() []*solana.Transaction { return h.env.exec.transactions } +// executed returns the block objects the stream has executed (in order); the +// bank itself ran stream-owned copies, which are checked to be copies of +// exactly those objects. +func (h *streamingTestHarness) executed() []*solana.Transaction { + if h.exec.current == nil { + return nil + } + return h.exec.current.origin +} + +// sameCopies asserts that executed holds fresh copies of want, in order: the +// same signatures and static keys, never the same objects, and never sharing +// the originals' account-key storage. +func sameCopies(t *testing.T, want, executed []*solana.Transaction) { + t.Helper() + require.Len(t, executed, len(want)) + for i := range want { + require.NotSame(t, want[i], executed[i], "transaction %d must be a copy", i) + require.Equal(t, want[i].Signatures, executed[i].Signatures, "transaction %d signatures", i) + require.Equal(t, want[i].Message.GetVersion(), executed[i].Message.GetVersion()) + require.Equal(t, want[i].Message.RecentBlockhash, executed[i].Message.RecentBlockhash) + if want[i].Message.GetVersion() == solana.MessageVersionV0 { + require.False(t, want[i].Message.IsResolved(), "transaction %d: the block's object must stay untouched", i) + } + if len(want[i].Message.AccountKeys) > 0 { + require.NotSame(t, &want[i].Message.AccountKeys[0], &executed[i].Message.AccountKeys[0], "transaction %d shares account-key storage", i) + } + } +} func (h *streamingTestHarness) discardReason() string { return metrics.GlobalBlockReplay.StreamingExecution.DiscardReason @@ -179,6 +207,7 @@ func TestStreamingConsumeExecutesContiguousGroupsInOrder(t *testing.T) { require.Empty(t, h.executed(), "a gap before the batch holds it") h.exec.handleEvent(h.event(a)) sameTransactions(t, txs[0:6], h.executed()) + sameCopies(t, txs[0:6], h.env.exec.transactions) require.Len(t, h.exec.current.groups, 1, "contiguous batches execute as one group") require.Equal(t, uint32(7), h.exec.current.nextStart) @@ -189,6 +218,7 @@ func TestStreamingConsumeExecutesContiguousGroupsInOrder(t *testing.T) { h.exec.handleEvent(h.event(c)) sameTransactions(t, txs, h.executed()) + sameCopies(t, txs, h.env.exec.transactions) require.Len(t, h.exec.current.groups, 2) require.Equal(t, uint32(10), h.exec.current.nextStart) require.Equal(t, uint64(9), h.env.exec.processedTxCount) From 1347b87c3fccb049d32e53234d628e42f346d853 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 09:37:41 +0000 Subject: [PATCH 049/111] block: make execution copies and their identities one operation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex's review of 0010: Rebind checked only version and recent blockhash, which does not authenticate the rest of the signed message, and as an exported method it let a future caller bind identities to unrelated same-blockhash transactions. Replace it with PreparedTransactionMessageIdentities.ExecutionCopies, which makes the copies itself from the originals the set is bound to — sharing signatures and instructions, taking the message by value with its own account-key slice — and returns the identities bound to exactly those copies. The identity is the original's by construction; the doc comment states that nothing re-authenticates the signed message and that callers must not treat the copies as independently verified. A bound v0 transaction that is already resolved is refused (ErrTransactionAlreadyResolved). The replay executor's own copy helpers are removed in favour of this; a stream now reports "resolved_input" and discards when a batch carries resolution. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/block/message_identity.go | 56 ++++++++--- .../message_identity_execution_copies_test.go | 95 +++++++++++++++++++ pkg/block/message_identity_rebind_test.go | 54 ----------- pkg/replay/streaming.go | 53 ++--------- pkg/replay/streaming_lifecycle_test.go | 37 ++++---- 5 files changed, 161 insertions(+), 134 deletions(-) create mode 100644 pkg/block/message_identity_execution_copies_test.go delete mode 100644 pkg/block/message_identity_rebind_test.go diff --git a/pkg/block/message_identity.go b/pkg/block/message_identity.go index ddcf75b47..4796b90f8 100644 --- a/pkg/block/message_identity.go +++ b/pkg/block/message_identity.go @@ -1,6 +1,7 @@ package block import ( + "errors" "fmt" "github.com/Overclock-Validator/mithril/pkg/txstatus" @@ -40,24 +41,49 @@ func (prepared *PreparedTransactionMessageIdentities) MatchesBlock(block *Block) return block != nil && prepared.matches(block.Transactions) } -// Rebind returns the same identities bound to copies of the same ordered -// transactions: copies[i] must carry the message version and recent -// blockhash identity i was prepared for. Streaming execution runs -// stream-owned copies of a block's transactions (so address-table resolution -// never touches the block's own objects) while proving the block by the -// originals; the copies share the originals' identities. -func (prepared *PreparedTransactionMessageIdentities) Rebind(copies []*solana.Transaction) (*PreparedTransactionMessageIdentities, error) { - if prepared == nil || len(copies) != len(prepared.identities) || len(copies) != len(prepared.versions) { - return nil, fmt.Errorf("prepared identities do not cover %d transaction copies", len(copies)) +// ErrTransactionAlreadyResolved reports a v0 transaction that already carries +// address-table resolution where an unresolved, wire-decoded object is +// required. +var ErrTransactionAlreadyResolved = errors.New("transaction address-table lookups are already resolved") + +// ExecutionCopies returns execution copies of the transactions this set is +// bound to, together with the same identities bound to those copies. +// +// Streaming replay executes a block's transactions before the block is +// complete. Execution resolves a v0 transaction's address-table lookups in +// place (solana-go's SetAddressTables refuses a second call and ResolveLookups +// appends the looked-up keys to AccountKeys), so a bank that later turns out +// not to be the block's — or the block's, on a different parent — must never +// have run the block's own objects. The copies are made here, from the +// originals this set was prepared for, so an identity can only ever be +// attached to a copy of the very transaction it was computed from: each copy +// shares the original's signatures and instructions and takes the message by +// value with its own account-key slice, which is all that resolution mutates. +// The identity itself (canonical message hash, recent blockhash) is therefore +// the original's by construction; nothing here re-authenticates the signed +// message, and callers must not treat the copies as independently verified. +// +// A bound transaction that already carries resolution is refused with +// ErrTransactionAlreadyResolved: its account keys were derived elsewhere, +// against a parent this bank cannot vouch for. +func (prepared *PreparedTransactionMessageIdentities) ExecutionCopies() ([]*solana.Transaction, *PreparedTransactionMessageIdentities, error) { + if prepared == nil || len(prepared.transactions) != len(prepared.identities) || len(prepared.transactions) != len(prepared.versions) { + return nil, nil, errors.New("prepared identities are not bound to their transactions") } - for index, tx := range copies { - if tx == nil || tx.Message.GetVersion() != prepared.versions[index] || - tx.Message.RecentBlockhash != prepared.identities[index].RecentBlockhash { - return nil, fmt.Errorf("transaction copy %d does not match its prepared identity", index) + copies := make([]*solana.Transaction, len(prepared.transactions)) + for index, tx := range prepared.transactions { + if tx == nil { + return nil, nil, fmt.Errorf("transaction %d is nil", index) + } + if tx.Message.GetVersion() == solana.MessageVersionV0 && tx.Message.IsResolved() { + return nil, nil, fmt.Errorf("transaction %d: %w", index, ErrTransactionAlreadyResolved) } + message := tx.Message + message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) + copies[index] = &solana.Transaction{Signatures: tx.Signatures, Message: message} } - return &PreparedTransactionMessageIdentities{ - transactions: append([]*solana.Transaction(nil), copies...), + return copies, &PreparedTransactionMessageIdentities{ + transactions: copies, versions: append([]solana.MessageVersion(nil), prepared.versions...), identities: append([]txstatus.TransactionMessageIdentity(nil), prepared.identities...), }, nil diff --git a/pkg/block/message_identity_execution_copies_test.go b/pkg/block/message_identity_execution_copies_test.go new file mode 100644 index 000000000..57c520c33 --- /dev/null +++ b/pkg/block/message_identity_execution_copies_test.go @@ -0,0 +1,95 @@ +package block + +import ( + "errors" + "testing" + + "github.com/gagliardetto/solana-go" +) + +func TestPreparedTransactionMessageIdentitiesExecutionCopies(t *testing.T) { + first, second := identityTestTransaction(1), identityTestTransaction(2) + originals := []*solana.Transaction{first, second} + prepared, err := (&Block{Transactions: originals}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare identities: %v", err) + } + + copies, rebound, err := prepared.ExecutionCopies() + if err != nil { + t.Fatalf("execution copies: %v", err) + } + if len(copies) != 2 || copies[0] == first || copies[1] == second { + t.Fatalf("copies must be distinct objects: %v", copies) + } + for i, tx := range copies { + if &tx.Message.AccountKeys[0] == &originals[i].Message.AccountKeys[0] { + t.Fatalf("copy %d shares the original's account-key storage", i) + } + if len(tx.Signatures) != 1 || tx.Signatures[0] != originals[i].Signatures[0] { + t.Fatalf("copy %d signatures differ", i) + } + if tx.Message.RecentBlockhash != originals[i].Message.RecentBlockhash || len(tx.Message.Instructions) != 1 { + t.Fatalf("copy %d message differs", i) + } + } + if rebound.Len() != 2 || rebound.Identity(0) != prepared.Identity(0) || rebound.Identity(1) != prepared.Identity(1) { + t.Fatalf("rebound identities differ: %+v vs %+v", rebound, prepared) + } + if !rebound.matches(copies) { + t.Fatal("rebound set is not bound to the copies") + } + if rebound.matches(originals) { + t.Fatal("rebound set must not claim the originals") + } + if !prepared.matches(originals) { + t.Fatal("making copies must not alter the original set") + } + + // Resolving a v0 copy leaves the original unresolved and still resolvable. + tableID := solana.PublicKey{0x70} + lookup := identityTestTransaction(3) + lookup.Message.SetAddressTableLookups([]solana.MessageAddressTableLookup{{AccountKey: tableID, WritableIndexes: []byte{0}}}) + if _, err := lookup.Message.SetVersion(solana.MessageVersionV0); err != nil { + t.Fatalf("set v0: %v", err) + } + staticKeys := len(lookup.Message.AccountKeys) + prepared, err = (&Block{Transactions: []*solana.Transaction{lookup}}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare v0 identity: %v", err) + } + copies, _, err = prepared.ExecutionCopies() + if err != nil { + t.Fatalf("v0 execution copy: %v", err) + } + tables := map[solana.PublicKey]solana.PublicKeySlice{tableID: {{0x71}}} + if err := copies[0].Message.SetAddressTables(tables); err != nil { + t.Fatalf("resolve copy: %v", err) + } + if err := copies[0].Message.ResolveLookups(); err != nil { + t.Fatalf("resolve copy: %v", err) + } + if !copies[0].Message.IsResolved() || len(copies[0].Message.AccountKeys) != staticKeys+1 { + t.Fatal("copy did not resolve") + } + if lookup.Message.IsResolved() || len(lookup.Message.AccountKeys) != staticKeys { + t.Fatal("resolving the copy touched the original") + } + if err := lookup.Message.SetAddressTables(tables); err != nil { + t.Fatalf("the original must still accept its own resolution: %v", err) + } + + // An already-resolved v0 transaction is refused. + prepared, err = (&Block{Transactions: []*solana.Transaction{copies[0]}}).PrepareTransactionMessageIdentities() + if err != nil { + t.Fatalf("prepare resolved identity: %v", err) + } + if _, _, err := prepared.ExecutionCopies(); !errors.Is(err, ErrTransactionAlreadyResolved) { + t.Fatalf("resolved input must be refused, got %v", err) + } + + var none *PreparedTransactionMessageIdentities + if _, _, err := none.ExecutionCopies(); err == nil { + t.Fatal("a nil prepared set must be rejected") + } +} diff --git a/pkg/block/message_identity_rebind_test.go b/pkg/block/message_identity_rebind_test.go deleted file mode 100644 index 0639b6109..000000000 --- a/pkg/block/message_identity_rebind_test.go +++ /dev/null @@ -1,54 +0,0 @@ -package block - -import ( - "testing" - - "github.com/gagliardetto/solana-go" -) - -func TestPreparedTransactionMessageIdentitiesRebindToCopies(t *testing.T) { - first, second := identityTestTransaction(1), identityTestTransaction(2) - prepared, err := (&Block{Transactions: []*solana.Transaction{first, second}}).PrepareTransactionMessageIdentities() - if err != nil { - t.Fatalf("prepare identities: %v", err) - } - copyOf := func(tx *solana.Transaction) *solana.Transaction { - message := tx.Message - message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) - return &solana.Transaction{Signatures: tx.Signatures, Message: message} - } - copies := []*solana.Transaction{copyOf(first), copyOf(second)} - - rebound, err := prepared.Rebind(copies) - if err != nil { - t.Fatalf("rebind: %v", err) - } - if rebound.Len() != 2 || rebound.Identity(0) != prepared.Identity(0) || rebound.Identity(1) != prepared.Identity(1) { - t.Fatalf("rebound identities differ: %+v vs %+v", rebound, prepared) - } - if !rebound.matches(copies) { - t.Fatal("rebound set is not bound to the copies") - } - if rebound.matches([]*solana.Transaction{first, second}) { - t.Fatal("rebound set must not claim the originals") - } - if !prepared.matches([]*solana.Transaction{first, second}) { - t.Fatal("rebinding must not alter the original set") - } - - if _, err := prepared.Rebind(copies[:1]); err == nil { - t.Fatal("a shorter copy list must be rejected") - } - wrongHash := copyOf(second) - wrongHash.Message.RecentBlockhash = solana.Hash{0x56} - if _, err := prepared.Rebind([]*solana.Transaction{copies[0], wrongHash}); err == nil { - t.Fatal("a copy with a different recent blockhash must be rejected") - } - if _, err := prepared.Rebind([]*solana.Transaction{copies[0], nil}); err == nil { - t.Fatal("a nil copy must be rejected") - } - var none *PreparedTransactionMessageIdentities - if _, err := none.Rebind(copies); err == nil { - t.Fatal("a nil prepared set must be rejected") - } -} diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index f67ae8df8..dde7ea771 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -132,7 +132,7 @@ type streamingSlot struct { parentID solana.Hash exec *blockExecution // origin holds the block's own transaction objects in executed order; - // the bank executes stream-owned copies (see executionCopy), and the + // the bank executes stream-owned copies (block.ExecutionCopies), and the // handshake proves the block by these originals. origin []*solana.Transaction nextStart uint32 @@ -533,19 +533,21 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { return errors.New("unverified_batch") } // The identities are bound to the block's objects by the verifier; that - // binding is checked here, on the originals, before the copies inherit it. + // binding is checked here, on the originals. The bank then executes + // copies made from those very originals (see block.ExecutionCopies): the + // block's objects are never resolved or otherwise mutated by a stream, + // so a discard leaves them exactly as turbine decoded them. preparedForOriginals, err := b.PrepareVerifiedTransactionMessageIdentities(txs, verified) if err != nil { return fmt.Errorf("identities: %w", err) } - copies, err := executionCopies(txs) + copies, prepared, err := preparedForOriginals.ExecutionCopies() if err != nil { + if errors.Is(err, b.ErrTransactionAlreadyResolved) { + return errors.New("resolved_input") + } return fmt.Errorf("copies: %w", err) } - prepared, err := preparedForOriginals.Rebind(copies) - if err != nil { - return fmt.Errorf("identities: %w", err) - } started := time.Now() err = s.executeFn(cur.exec, copies, prepared, false) cur.exec.setReplayStage("streaming_wait") @@ -564,43 +566,6 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { return nil } -// errStreamInputResolved reports a batch whose transactions already carry -// address-table resolution; the assembler never produces one, and a stream -// must not execute an object whose account keys were derived elsewhere. -var errStreamInputResolved = errors.New("stream input is already resolved") - -// executionCopy returns the object the stream executes in place of a block's -// own transaction. Execution resolves address-table lookups in place -// (SetAddressTables refuses a second call; ResolveLookups appends to -// AccountKeys), so running the block's object would leave it resolved -// against the stream's parent — and unusable, or worse, wrong, for the -// whole-block path after a discard. The copy takes the message by value with -// its own account-key slice; signatures and instructions are shared and never -// mutated by execution. -func executionCopy(tx *solana.Transaction) (*solana.Transaction, error) { - if tx == nil { - return nil, errors.New("nil transaction") - } - if tx.Message.GetVersion() == solana.MessageVersionV0 && tx.Message.IsResolved() { - return nil, errStreamInputResolved - } - message := tx.Message - message.AccountKeys = append(solana.PublicKeySlice(nil), tx.Message.AccountKeys...) - return &solana.Transaction{Signatures: tx.Signatures, Message: message}, nil -} - -func executionCopies(txs []*solana.Transaction) ([]*solana.Transaction, error) { - copies := make([]*solana.Transaction, len(txs)) - for i, tx := range txs { - dup, err := executionCopy(tx) - if err != nil { - return nil, fmt.Errorf("transaction %d: %w", i, err) - } - copies[i] = dup - } - return copies, nil -} - // discard throws the in-progress stream away and undoes every side effect it // may have had outside its own SlotCtx. func (s *streamingExecutor) discard(reason string) { diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go index 3c5e2dc48..972140334 100644 --- a/pkg/replay/streaming_lifecycle_test.go +++ b/pkg/replay/streaming_lifecycle_test.go @@ -482,27 +482,22 @@ func newLifecycleEnvWithTable(t *testing.T, dest solana.PublicKey) *lifecycleEnv return env } -func TestExecutionCopyLeavesTheBlockObjectUnresolved(t *testing.T) { - original := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1)})[0] - staticKeys := len(original.Message.AccountKeys) - - dup, err := executionCopy(original) - require.NoError(t, err) - require.NotSame(t, original, dup) - require.Equal(t, original.Signatures, dup.Signatures) - require.NoError(t, dup.Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD1}}})) - require.NoError(t, dup.Message.ResolveLookups()) - require.True(t, dup.Message.IsResolved()) - require.Len(t, dup.Message.AccountKeys, staticKeys+1) - require.False(t, original.Message.IsResolved(), "resolving the copy must not resolve the original") - require.Len(t, original.Message.AccountKeys, staticKeys) - require.NoError(t, original.Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD2}}}), - "the original still accepts its own resolution") - - _, err = executionCopy(dup) - require.ErrorIs(t, err, errStreamInputResolved, "a resolved input is never accepted by a stream") - _, err = executionCopies([]*solana.Transaction{original, dup}) - require.ErrorIs(t, err, errStreamInputResolved) +// A stream refuses a batch whose transactions already carry resolution: +// their account keys were derived against a parent the stream cannot vouch +// for. The assembler never produces one; this pins the refusal. +func TestStreamingRefusesResolvedInput(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + txs := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1), signedV0TransferViaTableWire(t, 2)}) + identities := verifiedIdentities(t, txs) + require.NoError(t, txs[1].Message.SetAddressTables(map[solana.PublicKey]solana.PublicKeySlice{lifecycleTableKey: {{0xD1}}})) + require.NoError(t, txs[1].Message.ResolveLookups()) + + h := newStreamingTestHarness(t) + h.exec.handleEvent(h.event(turbine.NewDetachedStreamBatch(h.gen, 1, 2, txs, identities))) + require.Nil(t, h.exec.current) + require.Equal(t, "resolved_input", h.discardReason()) + require.False(t, txs[0].Message.IsResolved(), "the unresolved sibling is untouched") } func TestStreamingLifecycleV0LookupsMatchWholeBlock(t *testing.T) { From ebc58062ab4a420847029e0f7119f69f22050a60 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 10:54:41 +0000 Subject: [PATCH 050/111] turbine/replay: real-feed lifecycle gate; recover a dropped header wake-up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The composition test Codex asked for (V5): a leader-side BroadcastSession shreds a header, a legacy entry batch, a v0/ALT entry batch, a footer carrying the whole-block reference bank hash, and the ending tick; the packets cross loopback UDP into a real UDPReceiver (assembler, entry prefetch, verifier); the receiver's streaming feed drives the real streamingExecutor on the lifecycle bank; the receiver emits the complete block; finalize matches it and the outcome must equal whole-block replay. Three tests: prefix executed before the last shred (asserted by the footer's ShredFullNanos and TxLoopBeforeFull), dropped wake-ups with a capacity-1 feed, and reset → discard → renewed generation → same result. The dropped-wake-up test exposed a gap: the prefetch publishes header and batch wake-ups as ranges decode (concurrently), so with a full channel the survivor may be a batch, and a dropped header wake-up was never recovered — the slot silently fell back to whole-block. recoverHeader now looks the header up in the assembler (PendingStreamBatches, the authoritative source after a drop) on any batch wake-up for the slot that could open next (frontier+1 while idle, the open stream's successor otherwise). A generation the executor discarded or declined is retired and never recovered, so a discard stays final. Unit test for the recovery rules; UDPReceiver.StreamDroppedEvents exposes the assembler's drop counter. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/replay/streaming.go | 60 +++- pkg/replay/streaming_realfeed_test.go | 420 ++++++++++++++++++++++++++ pkg/replay/streaming_test.go | 68 +++++ pkg/turbine/receiver.go | 6 + 4 files changed, 552 insertions(+), 2 deletions(-) create mode 100644 pkg/replay/streaming_realfeed_test.go diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index dde7ea771..8790f1dac 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -155,6 +155,10 @@ type streamingExecutor struct { // next slot can open as soon as its parent finishes, even when its header // wake-up arrived earlier. headers map[uint64]*turbine.StreamBatch + // retired is the last generation per slot that this executor discarded or + // declined; header recovery (recoverHeader) never reopens it. Pruned with + // headers. + retired map[uint64]turbine.StreamGeneration ticker *time.Ticker // executeFn runs one group on the open execution; tests substitute it. executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error @@ -164,6 +168,7 @@ func newStreamingExecutor(deps streamingDeps) *streamingExecutor { return &streamingExecutor{ deps: deps, headers: make(map[uint64]*turbine.StreamBatch), + retired: make(map[uint64]turbine.StreamGeneration), executeFn: (*blockExecution).executeTransactionGroup, } } @@ -210,6 +215,7 @@ func (s *streamingExecutor) shutdown() { s.discard("shutdown") s.stopTicker() s.headers = make(map[uint64]*turbine.StreamBatch) + s.retired = make(map[uint64]turbine.StreamGeneration) } // handleEvent consumes one feed wake-up. @@ -229,6 +235,8 @@ func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { } if event.Batch.Marker == turbine.StreamMarkerHeader && event.Batch.Start == 0 { s.rememberHeader(event.Batch) + } else { + s.recoverHeader(event.Slot, event.Generation) } s.tryOpen() case turbine.StreamCancelled: @@ -297,14 +305,60 @@ func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { s.pruneHeaders(frontier) } -// pruneHeaders bounds the header map: anything at or below the frontier can -// never open. +// recoverHeader handles a wake-up for a batch of a generation whose header +// this executor has not seen: the header's own wake-up may have been dropped +// (full channel), in which case the assembler is the authoritative source. +// Only the slot that could open next is worth the lookup — frontier+1 while +// idle, the open stream's successor otherwise — and a generation this +// executor already retired (discarded, or declined as ineligible) is never +// brought back: whole-block execution owns it from then on. A recovered +// header's ReadyAt is the lookup time, so OpenDelay reads as ~0 for it. +func (s *streamingExecutor) recoverHeader(slot uint64, g turbine.StreamGeneration) { + if g.IsZero() { + return + } + next := s.deps.frontier() + 1 + if s.current != nil { + next = s.current.slot + 1 + } + if slot != next { + return + } + if known, ok := s.headers[slot]; ok && known.Generation == g { + return + } + if retired, ok := s.retired[slot]; ok && retired == g { + return + } + // Sorted by start: the header is the first batch, at 0, or not decoded. + pending := s.deps.feed.PendingStreamBatches(g, 0) + if len(pending) > 0 && pending[0].Start == 0 && pending[0].Marker == turbine.StreamMarkerHeader { + s.rememberHeader(pending[0]) + } +} + +// retire records that generation g of slot must not open again through +// header recovery; discard and the ineligible path call it. +func (s *streamingExecutor) retire(slot uint64, g turbine.StreamGeneration) { + if g.IsZero() { + return + } + s.retired[slot] = g +} + +// pruneHeaders bounds the header and retired maps: anything at or below the +// frontier can never open. func (s *streamingExecutor) pruneHeaders(frontier uint64) { for slot := range s.headers { if slot <= frontier { delete(s.headers, slot) } } + for slot := range s.retired { + if slot <= frontier { + delete(s.retired, slot) + } + } } // tryOpen opens a stream for frontier+1 when its header is known and every @@ -323,6 +377,7 @@ func (s *streamingExecutor) tryOpen() { if reason := s.eligibility(header); reason != "" { mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", next, reason) delete(s.headers, next) + s.retire(next, header.Generation) return } delete(s.headers, next) @@ -575,6 +630,7 @@ func (s *streamingExecutor) discard(reason string) { cur := s.current s.current = nil s.stopTicker() + s.retire(cur.slot, cur.generation) exec := cur.exec if exec != nil { exec.close() diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go new file mode 100644 index 000000000..8880e2a91 --- /dev/null +++ b/pkg/replay/streaming_realfeed_test.go @@ -0,0 +1,420 @@ +package replay + +import ( + "context" + "net" + "testing" + "time" + + b "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/sigverify" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// The real-feed gate. A leader-side BroadcastSession shreds a header, entry +// batches (legacy and v0 transfers), a footer and the ending tick; the +// packets cross a loopback UDP socket into a real UDPReceiver (assembler, +// entry prefetch, signature verifier); the receiver's streaming feed drives +// the real streamingExecutor, which opens a bank on the lifecycle +// environment and executes the prefix while the slot is still incomplete; +// the receiver then emits the complete block, finalize matches it, and the +// result must equal whole-block replay of the same block — enforced twice: +// by the footer's expected bank hash inside finalize (the footer carries the +// whole-block reference hash) and by the explicit outcome comparison. + +const ( + realFeedSlot = lifecycleSlot + realFeedParentSlot = lifecycleParentSlot + realFeedShredVersion = uint16(7) + realFeedTickHashByte = 0xEE + realFeedDriveDeadline = 20 * time.Second +) + +// receiverFeed is the block source's view of the feed, backed by a receiver. +type receiverFeed struct { + r *turbine.UDPReceiver + events chan turbine.StreamEvent +} + +func (f *receiverFeed) StreamEvents() <-chan turbine.StreamEvent { return f.events } +func (f *receiverFeed) StreamStatusOf(g turbine.StreamGeneration) turbine.StreamStatus { + return f.r.StreamStatusOf(g) +} +func (f *receiverFeed) PendingStreamBatches(g turbine.StreamGeneration, from uint32) []*turbine.StreamBatch { + return f.r.PendingStreamBatches(g, from) +} +func (f *receiverFeed) PrioritizeStreamRepair(uint64) {} + +type realFeedRig struct { + t *testing.T + env *lifecycleEnv + receiver *turbine.UDPReceiver + feed *receiverFeed + broadcaster *turbine.UDPBroadcaster + leader solana.PrivateKey + lastCtx *sealevel.SlotCtx + tail *lifecycleTail + statuses *TransactionStatusCache + exec *streamingExecutor + block *b.Block + // the slot's content, as wire bytes, decoded fresh for every use + legacyWires, v0Wires [][]byte +} + +func newRealFeedRig(t *testing.T, dest solana.PublicKey, eventBuffer int) *realFeedRig { + t.Helper() + previousOverlap := sigverify.Cfg.DisableShredOverlap + sigverify.Cfg.DisableShredOverlap = false // the entry prefetch is what streams + t.Cleanup(func() { sigverify.Cfg.DisableShredOverlap = previousOverlap }) + // The open-age bound is the loop's protection against a stalled slot; a + // loaded CI host must not trip it between two drive calls. + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, Workers: 2, MaxOpenAge: realFeedDriveDeadline} + t.Cleanup(func() { StreamingExecutionCfg = StreamingExecutionConfig{} }) + // The per-block collector, as the loop resets it before every wait. + previousMetrics := metrics.GlobalBlockReplay + metrics.GlobalBlockReplay = metrics.BlockReplay{} + t.Cleanup(func() { metrics.GlobalBlockReplay = previousMetrics }) + + env := newLifecycleEnvWithTable(t, dest) + rig := &realFeedRig{t: t, env: env, leader: solana.NewWallet().PrivateKey} + for i := 0; i < 4; i++ { + rig.legacyWires = append(rig.legacyWires, txfixture.MustSignedTransferWire(uint64(2000+i))) + } + for i := 0; i < 3; i++ { + rig.v0Wires = append(rig.v0Wires, signedV0TransferViaTableWire(t, uint64(30+i))) + } + + // The executed parent, as the loop would hold it: its bank hash is the + // child's ParentBankhash and its sysvar snapshot the child's parent. + rig.lastCtx = &sealevel.SlotCtx{ + Slot: realFeedParentSlot, + Epoch: 0, + FinalBankhash: append([]byte{0x88}, make([]byte, 31)...), + FeeRateGovernor: &sealevel.FeeRateGovernor{TargetLamportsPerSignature: 5_000, LamportsPerSignature: 5_000}, + VoteTimestamps: map[solana.PublicKey]sealevel.BlockTimestamp{}, + } + require.NoError(t, rig.lastCtx.PublishBankSysvars(env.parent)) + + // A real receiver on a loopback port we pick ourselves (the receiver does + // not report an ephemeral bind), fed by a real broadcaster. + probe, err := net.ListenPacket("udp", "127.0.0.1:0") + require.NoError(t, err) + bindAddr := probe.LocalAddr().String() + require.NoError(t, probe.Close()) + receiver := turbine.NewUDPReceiver(bindAddr) + receiver.SetShredVersion(realFeedShredVersion) + leaderKey := rig.leader.PublicKey() + receiver.SetLeaderForSlot(func(uint64) (solana.PublicKey, bool) { return leaderKey, true }) + rig.feed = &receiverFeed{r: receiver, events: make(chan turbine.StreamEvent, eventBuffer)} + receiver.SubscribeStream(rig.feed.events) + ctx, cancel := context.WithCancel(context.Background()) + runDone := make(chan struct{}) + go func() { _ = receiver.Run(ctx); close(runDone) }() + t.Cleanup(func() { + cancel() + select { + case <-runDone: + case <-time.After(5 * time.Second): + t.Error("receiver did not stop") + } + }) + select { + case err := <-receiver.Ready(): + require.NoError(t, err) + case <-time.After(5 * time.Second): + t.Fatal("receiver did not become ready") + } + rig.receiver = receiver + udpAddr, err := net.ResolveUDPAddr("udp", bindAddr) + require.NoError(t, err) + rig.broadcaster, err = turbine.NewUDPBroadcaster("127.0.0.1:0") + require.NoError(t, err) + rig.broadcaster.AddPeer(udpAddr) + t.Cleanup(func() { _ = rig.broadcaster.Close() }) + + rig.tail = &lifecycleTail{durable: env.durable} + rig.statuses = NewTransactionStatusCache() + rig.exec = newStreamingExecutor(streamingDeps{ + acctsDb: env.acctsDb, + feed: rig.feed, + epochSchedule: env.epochSchedule, + txParallelism: 2, + persistedHashes: &persistedTracker{}, + tail: rig.tail, + transactionStatuses: rig.statuses, + alpenglowMode: true, + unrootedTailUsed: true, + lastSlotCtx: func() *sealevel.SlotCtx { return rig.lastCtx }, + frontier: func() uint64 { return realFeedParentSlot }, + currentFeatures: func() *features.Features { return env.feats }, + currentEpoch: func() uint64 { return 0 }, + rewardsInFlight: func() bool { return false }, + switchPending: func() bool { return false }, + executedBlockID: func(uint64) (solana.Hash, bool) { return lifecycleParentBlockID, true }, + }) + t.Cleanup(rig.exec.shutdown) + return rig +} + +// entries decodes the slot's content into the two entry batches the leader +// broadcasts: legacy transfers, then v0 transfers through the lookup table. +func (rig *realFeedRig) entries() (legacy, v0 []turbine.Entry) { + values := func(wires [][]byte) []solana.Transaction { + out := make([]solana.Transaction, len(wires)) + for i, tx := range decodeWires(rig.t, wires) { + out[i] = *tx + } + return out + } + return []turbine.Entry{{NumHashes: 1, Hash: solana.Hash{0x11}, Txns: values(rig.legacyWires)}}, + []turbine.Entry{{NumHashes: 1, Hash: solana.Hash{0x22}, Txns: values(rig.v0Wires)}} +} + +func (rig *realFeedRig) allWires() [][]byte { + return append(append([][]byte(nil), rig.legacyWires...), rig.v0Wires...) +} + +func decodeWires(t *testing.T, wires [][]byte) []*solana.Transaction { + t.Helper() + txs := make([]*solana.Transaction, len(wires)) + for i, wire := range wires { + tx, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + txs[i] = tx + } + return txs +} + +// configure applies what the replay loop applies to every block on the +// executed parent before execution. +func (rig *realFeedRig) configure(block *b.Block) { + require.NoError(rig.t, configureBlockFromParent(block, rig.lastCtx, rig.env.epochSchedule, false)) + block.Epoch = rig.env.epochSchedule.GetEpoch(block.Slot) + block.Features = rig.env.feats +} + +// reference executes the slot whole, from freshly decoded objects, exactly as +// the loop would: same parent, same content, same last-entry hash. +func (rig *realFeedRig) reference() lifecycleOutcome { + block := &b.Block{ + Slot: realFeedSlot, + SourceParentSlot: realFeedParentSlot, + FromLiveStream: true, + AlpenglowParentBlockID: lifecycleParentBlockID, + HasAlpenglowParentBlockID: true, + Transactions: decodeWires(rig.t, rig.allWires()), + Blockhash: solana.Hash{realFeedTickHashByte}, + } + rig.configure(block) + block.MarkTransactionSignaturesVerified() + tail := &lifecycleTail{durable: rig.env.durable} + slotCtx, err := ProcessBlock(rig.env.acctsDb, block, rig.env.epochSchedule, 2, nil, &persistedTracker{}, tail, NewTransactionStatusCache(), false, rig.env.parent) + require.NoError(rig.t, err) + return lifecycleOutcomeOf(rig.t, slotCtx, tail) +} + +func (rig *realFeedRig) session() *turbine.BroadcastSession { + return turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ + Leader: rig.leader, + Slot: realFeedSlot, + ParentSlot: realFeedParentSlot, + ParentBlockID: lifecycleParentBlockID, + ParentChainedMerkleRoot: solana.Hash{0xBB}, + Broadcaster: rig.broadcaster, + Version: realFeedShredVersion, + }) +} + +// broadcastPrefix sends the header and both entry batches; the slot stays +// incomplete (no footer, no ending tick). +func (rig *realFeedRig) broadcastPrefix(session *turbine.BroadcastSession) { + legacy, v0 := rig.entries() + require.NoError(rig.t, session.BroadcastHeader(lifecycleParentBlockID)) + require.NoError(rig.t, session.BroadcastEntryBatch(legacy)) + require.NoError(rig.t, session.BroadcastEntryBatch(v0)) +} + +// broadcastCompletion sends the footer carrying the expected bank hash and +// the ending tick, which completes the slot. +func (rig *realFeedRig) broadcastCompletion(session *turbine.BroadcastSession, expectedBankhash []byte) { + require.NoError(rig.t, session.BroadcastFooter(solana.HashFromBytes(expectedBankhash), 1_700_000_000_000_000_000, nil, nil)) + require.NoError(rig.t, session.BroadcastEndingTickLast(solana.Hash{realFeedTickHashByte})) +} + +// drive runs the replay loop's wait as the loop would: feed wake-ups and +// poll ticks go to the executor, a complete block ends the wait. +func (rig *realFeedRig) drive(until func() bool, what string) { + rig.t.Helper() + deadline := time.After(realFeedDriveDeadline) + for !until() { + select { + case event := <-rig.feed.events: + rig.exec.handleEvent(event) + case <-rig.exec.tick(): + rig.exec.handleTick() + case blk, ok := <-rig.receiver.Blocks(): + require.True(rig.t, ok, "receiver closed its block channel") + rig.receiver.AcknowledgeBlockDelivery(blk.Slot) + rig.block = blk + case <-deadline: + reason := metrics.GlobalBlockReplay.StreamingExecution.DiscardReason + rig.t.Fatalf("timed out waiting for %s (stream open: %v, discard reason %q, block: %v)", what, rig.exec.current != nil, reason, rig.block != nil) + } + } +} + +func (rig *realFeedRig) executedPrefix() int { + if rig.exec.current == nil { + return -1 + } + return len(rig.exec.current.origin) +} + +// finalizeAndCompare completes the streamed bank against the emitted block +// and checks it against the whole-block reference. +func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { + rig.t.Helper() + block := rig.block + require.NotNil(rig.t, block) + require.Equal(rig.t, realFeedSlot, block.Slot) + require.True(rig.t, block.HasAlpenglowParentBlockID) + require.Equal(rig.t, lifecycleParentBlockID, solana.Hash(block.AlpenglowParentBlockID)) + require.True(rig.t, block.HasExpectedBankhash, "the footer carried the reference bank hash") + require.Len(rig.t, block.Transactions, len(rig.allWires())) + require.Positive(rig.t, block.ShredFullNanos) + rig.configure(block) + + slotCtx, ok, err := rig.exec.finalize(block, rig.env.parent) + require.NoError(rig.t, err) + require.True(rig.t, ok, "the stream must accept its own block (discard reason %q)", metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + require.Nil(rig.t, rig.exec.current) + streamed := lifecycleOutcomeOf(rig.t, slotCtx, rig.tail) + requireSameLifecycleOutcome(rig.t, reference, streamed) + require.Equal(rig.t, reference.bankhash, block.ExpectedBankhash[:], "finalize verified the footer hash against the same value") + + record := metrics.GlobalBlockReplay.StreamingExecution + require.Equal(rig.t, uint64(1), record.Opened) + require.Equal(rig.t, uint64(len(rig.allWires())), record.Transactions) + require.Zero(rig.t, record.Discarded, "discard reason %q", record.DiscardReason) +} + +func TestStreamingRealFeedExecutesPrefixBeforeCompletionAndMatchesWholeBlock(t *testing.T) { + dest := solana.PublicKey{0xF1} + rig := newRealFeedRig(t, dest, 64) + reference := rig.reference() + require.Contains(t, reference.delta, dest) + total := len(rig.allWires()) + + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the prefix to execute") + require.False(t, rig.receiver.SlotCompleted(realFeedSlot), "the slot is still incomplete while the prefix executes") + require.Nil(t, rig.block) + prefixDone := time.Now() + for i, tx := range rig.exec.current.origin[len(rig.legacyWires):] { + require.Equal(t, solana.MessageVersionV0, tx.Message.GetVersion()) + require.False(t, tx.Message.IsResolved(), "block object %d ran as a stream-owned copy", i) + } + sameCopies(t, rig.exec.current.origin, rig.exec.current.exec.transactions) + + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + require.NotNil(t, rig.exec.current, "the stream survives completion") + require.Equal(t, turbine.StreamDone, rig.receiver.StreamStatusOf(rig.exec.current.generation)) + require.True(t, time.Unix(0, rig.block.ShredFullNanos).After(prefixDone), "the prefix executed before the last shred arrived") + for i, tx := range rig.exec.current.origin { + require.Same(t, rig.block.Transactions[i], tx, "the executed prefix is the block, by identity") + } + + rig.finalizeAndCompare(reference) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.TxLoopBeforeFull.Count, "transaction work finished before the slot was full") + require.Zero(t, rig.receiver.StreamDroppedEvents()) +} + +// Dropped wake-ups: with room for one event, whichever of the three prefix +// wake-ups (header, two batches) the prefetch publishes first is queued and +// the other two are dropped — the prefetch decodes ranges concurrently, so +// the survivor is not fixed. The executor must reach the same state from any +// of them: a surviving header opens the stream and pulls the batches; a +// surviving batch recovers the header from the assembler, then opens and +// pulls the same way. Nothing is read from the feed until every wake-up has +// been published, so exactly two are dropped. +func TestStreamingRealFeedRecoversDroppedWakeups(t *testing.T) { + dest := solana.PublicKey{0xF2} + rig := newRealFeedRig(t, dest, 1) + reference := rig.reference() + total := len(rig.allWires()) + + session := rig.session() + rig.broadcastPrefix(session) + require.Eventually(t, func() bool { + return rig.receiver.StreamDroppedEvents() == 2 + }, realFeedDriveDeadline, 5*time.Millisecond, "one wake-up queued, two dropped") + var survivor turbine.StreamEvent + select { + case survivor = <-rig.feed.events: + default: + t.Fatal("the surviving wake-up is not queued") + } + require.Zero(t, len(rig.feed.events), "nothing else was published") + require.Equal(t, turbine.StreamBatchReady, survivor.Kind) + require.Equal(t, uint64(realFeedSlot), survivor.Slot) + require.Len(t, rig.receiver.PendingStreamBatches(survivor.Generation, 0), 3, "header and both batches are decoded and discoverable") + require.Nil(t, rig.exec.current) + + rig.exec.handleEvent(survivor) + require.NotNil(t, rig.exec.current, "the surviving wake-up (marker %v at shred %d) opened the stream", survivor.Batch.Marker, survivor.Batch.Start) + require.Equal(t, total, rig.executedPrefix(), "opening pulled every decoded batch from the assembler") + require.Equal(t, uint64(2), rig.receiver.StreamDroppedEvents(), "recovery reads the assembler, it does not replay wake-ups") + require.False(t, rig.receiver.SlotCompleted(realFeedSlot)) + + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + rig.finalizeAndCompare(reference) +} + +// Reset while streaming: the generation is cancelled, the stream discards +// and leaves nothing behind; the slot re-broadcast is a new generation, which +// opens a new stream and finalizes to the same result. +func TestStreamingRealFeedResetDiscardsAndRenews(t *testing.T) { + dest := solana.PublicKey{0xF3} + rig := newRealFeedRig(t, dest, 64) + reference := rig.reference() + total := len(rig.allWires()) + + stakeBefore := len(global.PendingStakeEntriesSnapshot()) + rig.broadcastPrefix(rig.session()) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the first prefix to execute") + firstGeneration := rig.exec.current.generation + + rig.receiver.ResetSlot(realFeedSlot) + rig.drive(func() bool { return rig.exec.current == nil }, "the cancellation") + // The reset reaches the executor either as the feed's cancellation wake-up + // or, when the poll tick is selected first, as the generation reading gone. + require.Contains(t, []string{"cancelled:reset", "gone"}, metrics.GlobalBlockReplay.StreamingExecution.DiscardReason) + require.Equal(t, turbine.StreamGone, rig.receiver.StreamStatusOf(firstGeneration)) + require.Empty(t, rig.tail.added, "a discarded stream commits nothing") + require.Len(t, global.PendingStakeEntriesSnapshot(), stakeBefore) + durablePayer, err := rig.env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(10_000_000_000), durablePayer.Lamports, "the durable view is untouched") + // The loop starts the next replay attempt with a fresh collector. + metrics.GlobalBlockReplay = metrics.BlockReplay{} + + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == total }, "the renewed prefix to execute") + // Compared as values (a deep comparison would walk the assembler's live + // slot state without its lock). + require.False(t, firstGeneration == rig.exec.current.generation, "a re-assembled slot is a new generation") + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "the complete block") + rig.finalizeAndCompare(reference) +} diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 692de3b47..e4abef968 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -446,6 +446,74 @@ func TestStreamingRememberHeaderAndTryOpenBounds(t *testing.T) { require.Contains(t, h.exec.headers, uint64(42), "disabled: nothing opens") } +// A dropped header wake-up is recovered from the assembler by the next +// wake-up for the generation (the pending list is authoritative after a +// drop); the usual open path follows. The lookup is only made for the slot +// that could open next (frontier+1 while idle, the open stream's successor +// otherwise), never for a generation this executor retired, and only finds +// a header that is decoded. +func TestStreamingRecoversHeaderFromPendingAfterDroppedWakeup(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 3, 700) + g := turbine.NewDetachedStreamGeneration(42) + header := turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + batch := turbine.NewDetachedStreamBatch(g, 1, 3, txs, verifiedIdentities(t, txs)) + h.feed.pending[g] = []*turbine.StreamBatch{batch, header} + + h.exec.recoverHeader(42, g) + require.Empty(t, h.exec.headers, "the open stream's own slot is not looked up (its successor is; see below)") + + h.exec.discard("idle") + require.Equal(t, h.gen, h.exec.retired[42], "the discarded generation is retired") + h.exec.recoverHeader(42, g) + require.Same(t, header, h.exec.headers[42], "the batch wake-up recovered the decoded header") + + // Declined once (the feed does not know the generation, so it is + // ineligible), the generation is retired and no later wake-up brings it + // back; whole-block execution owns the slot. + h.exec.tryOpen() + require.Nil(t, h.exec.current) + require.Empty(t, h.exec.headers) + require.Equal(t, g, h.exec.retired[42]) + h.exec.recoverHeader(42, g) + require.Empty(t, h.exec.headers, "a retired generation is never recovered") + + // A new generation of the slot (after a reset) is recoverable again, but + // only once its header is decoded. + renewed := turbine.NewDetachedStreamGeneration(42) + renewedHeader := turbine.NewDetachedStreamMarker(renewed, 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.feed.pending[renewed] = []*turbine.StreamBatch{turbine.NewDetachedStreamBatch(renewed, 1, 3, txs, verifiedIdentities(t, txs))} + h.exec.recoverHeader(42, renewed) + require.Empty(t, h.exec.headers, "a header that is not decoded yet cannot be recovered") + h.feed.pending[renewed] = append(h.feed.pending[renewed], renewedHeader) + h.exec.recoverHeader(42, renewed) + require.Same(t, renewedHeader, h.exec.headers[42]) + delete(h.exec.headers, 42) + + ahead := turbine.NewDetachedStreamGeneration(44) + h.feed.pending[ahead] = []*turbine.StreamBatch{turbine.NewDetachedStreamMarker(ahead, 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{})} + h.exec.recoverHeader(44, ahead) + require.Empty(t, h.exec.headers, "only the next slot to open is looked up") + + h.exec.recoverHeader(42, turbine.StreamGeneration{}) + require.Empty(t, h.exec.headers, "a zero generation has nothing pending") + + // While a stream is open, its successor is the slot that could open next. + h.frontier = 42 + h.exec.pruneHeaders(h.frontier) + require.Empty(t, h.exec.retired, "retirements at or below the frontier are pruned") + h.frontier = 41 + h.open() + successor := turbine.NewDetachedStreamGeneration(43) + successorHeader := turbine.NewDetachedStreamMarker(successor, 0, 0, turbine.StreamMarkerHeader, 42, solana.Hash{1}) + h.feed.pending[successor] = []*turbine.StreamBatch{successorHeader} + h.exec.recoverHeader(43, successor) + require.Same(t, successorHeader, h.exec.headers[43], "the open stream's successor is recoverable") + require.NotNil(t, h.exec.current, "the open stream is untouched") +} + func (h *streamingTestHarness) matchingBlock(t *testing.T, txs []*solana.Transaction) *b.Block { t.Helper() shell := h.env.exec.block diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index 0080cd417..ce4cf03a7 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -401,6 +401,12 @@ func (r *UDPReceiver) StreamStatusOf(g StreamGeneration) StreamStatus { return r.assembler.StreamStatusOf(g) } +// StreamDroppedEvents reports feed wake-ups dropped because the subscriber +// was full; the subscriber recovers through PendingStreamBatches. +func (r *UDPReceiver) StreamDroppedEvents() uint64 { + return r.assembler.StreamDroppedEvents() +} + // PendingStreamBatches returns the generation's decoded batches starting at // or after fromStart, in shred-index order. func (r *UDPReceiver) PendingStreamBatches(g StreamGeneration, fromStart uint32) []*StreamBatch { From bc4ac81264909ec9dcd9da1a7c7e689c69078bab Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 09:40:15 -0500 Subject: [PATCH 051/111] replay: stream direct children across unresolved slot gaps --- pkg/replay/block.go | 5 ++ pkg/replay/streaming.go | 70 +++++++++++++++++++------- pkg/replay/streaming_lifecycle_test.go | 2 +- pkg/replay/streaming_realfeed_test.go | 36 +++++++++++-- pkg/replay/streaming_test.go | 41 +++++++++++++-- 5 files changed, 127 insertions(+), 27 deletions(-) diff --git a/pkg/replay/block.go b/pkg/replay/block.go index e84277fc7..2f70e9df8 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -2714,6 +2714,11 @@ func ReplayBlocks( continue } + // A speculative stream may span unresolved slots on the last bank. + // An actual intervening block invalidates that assumption before any + // validation or bank work can observe speculative global state. + streamer.beforeBlock(block) + // An in-flight source send can race the first quarantine drain. Exact // emitted suffix IDs are hard-tombstoned before that send, so discard // any leaked descendant before it reaches consensus observation. diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index 8790f1dac..d18e655bd 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -60,6 +60,9 @@ const ( defaultStreamingWorkers = 4 defaultStreamingMaxAge = 2 * time.Second streamingPollInterval = 5 * time.Millisecond + // Bound speculation across missing leaders; this never advances replay + // or establishes that the intervening slots are actually skipped. + streamingMaxSlotDistance = uint64(32) // streamingHardOpenAgeFactor bounds a completed-but-not-yet-emitted stream // to this multiple of MaxOpenAge. streamingHardOpenAgeFactor = 10 @@ -206,6 +209,14 @@ func (s *streamingExecutor) matches(slot uint64) bool { return s != nil && s.current != nil && s.current.slot == slot } +// beforeBlock restores speculative state before an intervening real bank is +// observed. A skip has no bank changes and must not throw away a later child. +func (s *streamingExecutor) beforeBlock(block *b.Block) { + if s != nil && s.current != nil && !block.IsSkipped && !s.matches(block.Slot) { + s.discard("other_block") + } +} + // shutdown discards any open stream; the replay loop defers it so an exiting // attempt never leaves a speculative bank (and its watchdog) behind. func (s *streamingExecutor) shutdown() { @@ -308,8 +319,7 @@ func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { // recoverHeader handles a wake-up for a batch of a generation whose header // this executor has not seen: the header's own wake-up may have been dropped // (full channel), in which case the assembler is the authoritative source. -// Only the slot that could open next is worth the lookup — frontier+1 while -// idle, the open stream's successor otherwise — and a generation this +// Only slots within the bounded lookahead are worth the lookup, and a generation this // executor already retired (discarded, or declined as ineligible) is never // brought back: whole-block execution owns it from then on. A recovered // header's ReadyAt is the lookup time, so OpenDelay reads as ~0 for it. @@ -317,11 +327,11 @@ func (s *streamingExecutor) recoverHeader(slot uint64, g turbine.StreamGeneratio if g.IsZero() { return } - next := s.deps.frontier() + 1 + anchor := s.deps.frontier() if s.current != nil { - next = s.current.slot + 1 + anchor = s.current.slot } - if slot != next { + if slot <= anchor || slot-anchor > streamingMaxSlotDistance { return } if known, ok := s.headers[slot]; ok && known.Generation == g { @@ -361,27 +371,49 @@ func (s *streamingExecutor) pruneHeaders(frontier uint64) { } } -// tryOpen opens a stream for frontier+1 when its header is known and every -// eligibility condition holds. +// nextHeader prefers the next slot, otherwise the earliest nearby child of +// the executed bank. A header is only a speculation hint: it does not prove +// skips, advance the frontier, or authorize publication or voting. +func (s *streamingExecutor) nextHeader(frontier uint64) *turbine.StreamBatch { + last := s.deps.lastSlotCtx() + var selected *turbine.StreamBatch + for slot, header := range s.headers { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { + continue + } + if slot-frontier != 1 && (last == nil || header.ParentSlot != last.Slot) { + continue + } + if selected == nil || slot < selected.Slot { + selected = header + } + } + return selected +} + +// tryOpen may speculate across unresolved slots only on the exact executed +// parent. Real intervening blocks and fork switches discard the overlay; +// skip records leave it alone. The complete-block handshake stays mandatory. func (s *streamingExecutor) tryOpen() { if s == nil || s.current != nil || !StreamingExecutionCfg.Enabled { return } frontier := s.deps.frontier() s.pruneHeaders(frontier) - next := frontier + 1 - header, ok := s.headers[next] - if !ok { - return - } - if reason := s.eligibility(header); reason != "" { - mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", next, reason) - delete(s.headers, next) - s.retire(next, header.Generation) + for { + header := s.nextHeader(frontier) + if header == nil { + return + } + delete(s.headers, header.Slot) + if reason := s.eligibility(header); reason != "" { + mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", header.Slot, reason) + s.retire(header.Slot, header.Generation) + continue + } + s.openStream(header) return } - delete(s.headers, next) - s.openStream(header) } // eligibility returns an empty string when a stream may open on header, or @@ -398,7 +430,7 @@ func (s *streamingExecutor) eligibility(header *turbine.StreamBatch) string { if last == nil { return "no executed parent context" } - if frontier := d.frontier(); header.Slot != frontier+1 || header.ParentSlot != last.Slot { + if frontier := d.frontier(); header.Slot <= frontier || header.Slot-frontier > streamingMaxSlotDistance || header.ParentSlot != last.Slot { return fmt.Sprintf("slot %d on parent %d does not extend the executed frontier %d (parent context %d)", header.Slot, header.ParentSlot, frontier, last.Slot) } executedID, ok := d.executedBlockID(last.Slot) diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go index 972140334..bbd4b747a 100644 --- a/pkg/replay/streaming_lifecycle_test.go +++ b/pkg/replay/streaming_lifecycle_test.go @@ -192,7 +192,7 @@ func lifecycleOutcomeOf(t *testing.T, slotCtx *sealevel.SlotCtx, tail *lifecycle t.Helper() require.NotNil(t, slotCtx) require.Len(t, tail.added, 1, "the bank commits exactly once") - require.Equal(t, lifecycleSlot, tail.added[0].slot) + require.Equal(t, slotCtx.Slot, tail.added[0].slot) require.Equal(t, slotCtx.FinalBankhash, tail.added[0].bankhash) delta := make(map[solana.PublicKey]*accounts.Account, len(tail.added[0].delta)) for _, acct := range tail.added[0].delta { diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index 8880e2a91..7c44db6c9 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -53,6 +53,7 @@ func (f *receiverFeed) PendingStreamBatches(g turbine.StreamGeneration, from uin func (f *receiverFeed) PrioritizeStreamRepair(uint64) {} type realFeedRig struct { + slot uint64 t *testing.T env *lifecycleEnv receiver *turbine.UDPReceiver @@ -83,7 +84,7 @@ func newRealFeedRig(t *testing.T, dest solana.PublicKey, eventBuffer int) *realF t.Cleanup(func() { metrics.GlobalBlockReplay = previousMetrics }) env := newLifecycleEnvWithTable(t, dest) - rig := &realFeedRig{t: t, env: env, leader: solana.NewWallet().PrivateKey} + rig := &realFeedRig{slot: realFeedSlot, t: t, env: env, leader: solana.NewWallet().PrivateKey} for i := 0; i < 4; i++ { rig.legacyWires = append(rig.legacyWires, txfixture.MustSignedTransferWire(uint64(2000+i))) } @@ -204,7 +205,7 @@ func (rig *realFeedRig) configure(block *b.Block) { // the loop would: same parent, same content, same last-entry hash. func (rig *realFeedRig) reference() lifecycleOutcome { block := &b.Block{ - Slot: realFeedSlot, + Slot: rig.slot, SourceParentSlot: realFeedParentSlot, FromLiveStream: true, AlpenglowParentBlockID: lifecycleParentBlockID, @@ -223,7 +224,7 @@ func (rig *realFeedRig) reference() lifecycleOutcome { func (rig *realFeedRig) session() *turbine.BroadcastSession { return turbine.NewBroadcastSession(turbine.BroadcastSessionConfig{ Leader: rig.leader, - Slot: realFeedSlot, + Slot: rig.slot, ParentSlot: realFeedParentSlot, ParentBlockID: lifecycleParentBlockID, ParentChainedMerkleRoot: solana.Hash{0xBB}, @@ -283,7 +284,7 @@ func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { rig.t.Helper() block := rig.block require.NotNil(rig.t, block) - require.Equal(rig.t, realFeedSlot, block.Slot) + require.Equal(rig.t, rig.slot, block.Slot) require.True(rig.t, block.HasAlpenglowParentBlockID) require.Equal(rig.t, lifecycleParentBlockID, solana.Hash(block.AlpenglowParentBlockID)) require.True(rig.t, block.HasExpectedBankhash, "the footer carried the reference bank hash") @@ -418,3 +419,30 @@ func TestStreamingRealFeedResetDiscardsAndRenews(t *testing.T) { rig.drive(func() bool { return rig.block != nil }, "the complete block") rig.finalizeAndCompare(reference) } + +// The child executes on parent 7 while slots 8..11 are unresolved. Consuming +// those skips advances only the frontier; the completed child must still +// produce exactly the whole-block bank hash, accounts, fees and CU. +func TestStreamingRealFeedAcrossSkippedSlots(t *testing.T) { + rig := newRealFeedRig(t, solana.PublicKey{0xF1}, 64) + rig.slot += 4 + frontier := uint64(realFeedParentSlot) + rig.exec.deps.frontier = func() uint64 { return frontier } + reference := rig.reference() + session := rig.session() + rig.broadcastPrefix(session) + rig.drive(func() bool { return rig.executedPrefix() == len(rig.allWires()) }, "prefix across unresolved skips") + require.Equal(t, uint64(realFeedParentSlot), frontier, "speculation does not certify skips or advance replay") + require.False(t, rig.receiver.SlotCompleted(rig.slot)) + cur := rig.exec.current + for slot := frontier + 1; slot < rig.slot; slot++ { + rig.exec.beforeBlock(&b.Block{Slot: slot, IsSkipped: true}) + rig.exec.discardSlot(slot, "skipped") + frontier = slot + rig.exec.handleTick() + require.Same(t, cur, rig.exec.current) + } + rig.broadcastCompletion(session, reference.bankhash) + rig.drive(func() bool { return rig.block != nil }, "completed child after skips") + rig.finalizeAndCompare(reference) +} diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index e4abef968..8e0ff9880 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -449,8 +449,7 @@ func TestStreamingRememberHeaderAndTryOpenBounds(t *testing.T) { // A dropped header wake-up is recovered from the assembler by the next // wake-up for the generation (the pending list is authoritative after a // drop); the usual open path follows. The lookup is only made for the slot -// that could open next (frontier+1 while idle, the open stream's successor -// otherwise), never for a generation this executor retired, and only finds +// within the bounded lookahead, never for a generation this executor retired, and only finds // a header that is decoded. func TestStreamingRecoversHeaderFromPendingAfterDroppedWakeup(t *testing.T) { StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} @@ -495,7 +494,8 @@ func TestStreamingRecoversHeaderFromPendingAfterDroppedWakeup(t *testing.T) { ahead := turbine.NewDetachedStreamGeneration(44) h.feed.pending[ahead] = []*turbine.StreamBatch{turbine.NewDetachedStreamMarker(ahead, 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{})} h.exec.recoverHeader(44, ahead) - require.Empty(t, h.exec.headers, "only the next slot to open is looked up") + require.Contains(t, h.exec.headers, uint64(44), "nearby header can be recovered before its parent is ready") + delete(h.exec.headers, 44) h.exec.recoverHeader(42, turbine.StreamGeneration{}) require.Empty(t, h.exec.headers, "a zero generation has nothing pending") @@ -814,3 +814,38 @@ func TestStreamingConfigDefaults(t *testing.T) { cfg.MaxOpenAge = time.Second require.Equal(t, time.Second, cfg.maxOpenAge()) } + +func TestStreamingGapSelectionAndSafety(t *testing.T) { + h := newStreamingTestHarness(t) + defer h.exec.shutdown() + makeHeader := func(slot, parent uint64, id solana.Hash) *turbine.StreamBatch { + g := turbine.NewDetachedStreamGeneration(slot) + h.feed.status[g] = turbine.StreamActive + return turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, parent, id) + } + gap := makeHeader(46, 41, h.parentID) + h.exec.headers[46] = gap + h.exec.headers[48] = makeHeader(48, 41, h.parentID) + h.exec.headers[44] = makeHeader(44, 43, h.parentID) + require.Same(t, gap, h.exec.nextHeader(h.frontier), "earliest direct child, not a grandchild") + require.Empty(t, h.exec.eligibility(gap)) + require.NotEmpty(t, h.exec.eligibility(makeHeader(46, 41, solana.Hash{99}))) + require.NotEmpty(t, h.exec.eligibility(makeHeader(46, 40, h.parentID))) + require.NotEmpty(t, h.exec.eligibility(makeHeader(74, 41, h.parentID)), "lookahead is bounded") + h.exec.deps.switchPending = func() bool { return true } + require.Equal(t, "fork switch pending", h.exec.eligibility(gap)) + h.exec.deps.switchPending = func() bool { return false } + h.exec.headers[42] = makeHeader(42, 41, h.parentID) + require.Equal(t, uint64(42), h.exec.nextHeader(h.frontier).Slot) + + // A late real bank in the unresolved gap must restore all speculative + // state before that bank is validated, configured or executed. + h.exec.current.slot = 46 + cur := h.exec.current + h.exec.beforeBlock(&b.Block{Slot: 42, IsSkipped: true}) + require.Same(t, cur, h.exec.current) + h.exec.beforeBlock(&b.Block{Slot: 42}) + require.Nil(t, h.exec.current) + require.True(t, cur.exec.closed) + require.Equal(t, h.gen, h.exec.retired[46]) +} From 188b90304b62a3527a64c67605c2c7e173b0900d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 10:16:48 -0500 Subject: [PATCH 052/111] turbine: wake repair promptly for new priority slots --- pkg/turbine/assembler.go | 10 ++- pkg/turbine/receiver.go | 6 +- pkg/turbine/repair.go | 74 ++++++++++++++++---- pkg/turbine/repair_wakeup_test.go | 111 ++++++++++++++++++++++++++++++ 4 files changed, 186 insertions(+), 15 deletions(-) create mode 100644 pkg/turbine/repair_wakeup_test.go diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 2769adf10..4ccec5f6e 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -635,8 +635,13 @@ func (a *SlotAssembler) PrioritizeRepairSlot(slot uint64) { } func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { + a.prioritizeRepairRange(start, end) +} + +// Report only newly installed pins, so repeated replay hints do not wake repair. +func (a *SlotAssembler) prioritizeRepairRange(start, end uint64) bool { if start == 0 { - return + return false } if end < start { end = start @@ -648,11 +653,13 @@ func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { a.mu.Lock() defer a.mu.Unlock() + changed := false for slot := start; ; slot++ { if _, completed := a.completedSlots[slot]; !completed { if _, exists := a.priorityRepairSlots[slot]; !exists { a.priorityRepairSlots[slot] = struct{}{} a.priorityRepairOrder = append(a.priorityRepairOrder, slot) + changed = true } } if slot == end { @@ -660,6 +667,7 @@ func (a *SlotAssembler) PrioritizeRepairRange(start, end uint64) { } } a.prunePriorityRepairSlotsLocked() + return changed } func (a *SlotAssembler) slotState(slot uint64, version uint16) *slotState { diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index ce4cf03a7..372e44000 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -379,14 +379,16 @@ func (r *UDPReceiver) PrioritizeRepairSlot(slot uint64) { if r == nil || r.assembler == nil { return } - r.assembler.PrioritizeRepairSlot(slot) + r.PrioritizeRepairRange(slot, slot) } func (r *UDPReceiver) PrioritizeRepairRange(start, end uint64) { if r == nil || r.assembler == nil { return } - r.assembler.PrioritizeRepairRange(start, end) + if r.assembler.prioritizeRepairRange(start, end) && r.repairClient != nil { + r.repairClient.wakePriority() + } } // SubscribeStream installs the streaming-execution feed subscriber on this diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index f9be294af..980f0b621 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -304,8 +304,9 @@ type RepairPeerReport struct { } type repairClient struct { - identity ed25519.PrivateKey - peerSource RepairPeerSource + priorityWake chan struct{} // initialized before the receiver starts; coalesced hints + identity ed25519.PrivateKey + peerSource RepairPeerSource mu sync.Mutex outstanding map[repairRequestKey]outstandingRepairRequest @@ -375,28 +376,77 @@ func newRepairClient(identity ed25519.PrivateKey, peerSource RepairPeerSource) ( return nil, fmt.Errorf("repair peer source is required") } c := &repairClient{ - identity: append(ed25519.PrivateKey(nil), identity...), - peerSource: peerSource, - outstanding: make(map[repairRequestKey]outstandingRepairRequest), - byResponse: make(map[repairResponseKey]repairRequestKey), - inflight: make(map[shredKey]*shredInflight), - perPeer: make(map[repairAddressKey]*peerRecord), - expiredCur: make(map[repairResponseKey]outstandingRepairRequest, repairExpiredGenMin), + priorityWake: make(chan struct{}, 1), + identity: append(ed25519.PrivateKey(nil), identity...), + peerSource: peerSource, + outstanding: make(map[repairRequestKey]outstandingRepairRequest), + byResponse: make(map[repairResponseKey]repairRequestKey), + inflight: make(map[shredKey]*shredInflight), + perPeer: make(map[repairAddressKey]*peerRecord), + expiredCur: make(map[repairResponseKey]outstandingRepairRequest, repairExpiredGenMin), } c.timeoutNanos.Store(int64(repairMinRequestTimeout)) return c, nil } +// wakePriority does not send requests or mint rate tokens. It only asks the +// single repair loop to reconsider newly prioritized work sooner. +func (c *repairClient) wakePriority() { + select { + case c.priorityWake <- struct{}{}: + default: + } +} + func (c *repairClient) run(ctx context.Context, conn *net.UDPConn, assembler *SlotAssembler) { - ticker := time.NewTicker(repairScanInterval) + runRepairSchedule(ctx, c.priorityWake, repairScanInterval, 20*time.Millisecond, func() { + c.expireOutstanding(time.Now()) + c.repairOnce(conn, assembler) + }) +} + +// Keep periodic scans for retries/freshness. Coalesce urgent hints and bound +// scan frequency; all sends still use the existing token bucket, admission, +// retry, fanout and peer budgets. The loop remains the sole sender. +func runRepairSchedule(ctx context.Context, wake <-chan struct{}, interval, minSpacing time.Duration, scan func()) { + ticker := time.NewTicker(interval) defer ticker.Stop() + var timer *time.Timer + var urgent <-chan time.Time + var last time.Time + stopTimer := func() { + if timer != nil { + timer.Stop() + } + urgent = nil + } + defer stopTimer() + run := func() { + stopTimer() + if ctx.Err() != nil { + return + } + scan() + last = time.Now() + } + schedule := func() { + if delay := minSpacing - time.Since(last); delay <= 0 { + run() + } else if urgent == nil { + timer = time.NewTimer(delay) + urgent = timer.C + } + } for { select { case <-ctx.Done(): return case <-ticker.C: - c.expireOutstanding(time.Now()) - c.repairOnce(conn, assembler) + schedule() + case <-urgent: + run() + case <-wake: + schedule() } } } diff --git a/pkg/turbine/repair_wakeup_test.go b/pkg/turbine/repair_wakeup_test.go new file mode 100644 index 000000000..d9669534a --- /dev/null +++ b/pkg/turbine/repair_wakeup_test.go @@ -0,0 +1,111 @@ +package turbine + +import ( + "context" + "testing" + "time" +) + +func TestPriorityRepairWakeOnlyForNewPins(t *testing.T) { + r := NewUDPReceiver("127.0.0.1:0") + r.repairClient = &repairClient{priorityWake: make(chan struct{}, 1)} + r.PrioritizeRepairSlot(10) + if len(r.repairClient.priorityWake) != 1 { + t.Fatal("new pin did not wake repair") + } + <-r.repairClient.priorityWake + for i := 0; i < 100; i++ { + r.PrioritizeRepairSlot(10) + } + if len(r.repairClient.priorityWake) != 0 { + t.Fatal("duplicate pins caused wakeups") + } + r.PrioritizeRepairRange(10, 12) + r.PrioritizeRepairSlot(13) + if len(r.repairClient.priorityWake) != 1 { + t.Fatal("new pins should coalesce") + } + <-r.repairClient.priorityWake + r.assembler.completedSlots[14] = struct{}{} + r.PrioritizeRepairSlot(14) + r.PrioritizeRepairSlot(0) + if len(r.repairClient.priorityWake) != 0 { + t.Fatal("completed/invalid slot woke repair") + } +} + +func TestRepairScheduleWakeCoalescingAndCancellation(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + wake := make(chan struct{}, 1) + calls := make(chan time.Time, 4) + entered := make(chan struct{}) + release := make(chan struct{}) + done := make(chan struct{}) + go func() { + defer close(done) + first := true + runRepairSchedule(ctx, wake, time.Hour, 20*time.Millisecond, func() { + calls <- time.Now() + if first { + first = false + close(entered) + <-release + } + }) + }() + wake <- struct{}{} + select { + case <-entered: + case <-time.After(time.Second): + t.Fatal("wake did not bypass periodic timer") + } + for i := 0; i < 100; i++ { + select { + case wake <- struct{}{}: + default: + } + } + first := <-calls + close(release) + select { + case second := <-calls: + if second.Sub(first) < 20*time.Millisecond { + t.Fatal("unbounded scan frequency") + } + case <-time.After(time.Second): + t.Fatal("pending wake lost") + } + cancel() + select { + case <-done: + case <-time.After(time.Second): + t.Fatal("scheduler did not stop") + } + if len(calls) != 0 { + t.Fatal("wake burst caused extra scans") + } +} + +func TestRepairSchedulePeriodicWithoutWake(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + calls := make(chan struct{}, 1) + done := make(chan struct{}) + go func() { + defer close(done) + runRepairSchedule(ctx, nil, time.Millisecond, time.Millisecond, func() { + select { + case calls <- struct{}{}: + default: + } + }) + }() + select { + case <-calls: + case <-time.After(time.Second): + t.Fatal("periodic repair stopped") + } + cancel() + <-done +} From 7a2386d2202a699abcad7ea1acb3adc4924468c3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 15:04:58 +0000 Subject: [PATCH 053/111] =?UTF-8?q?replay:=20streaming=20open=20timeline?= =?UTF-8?q?=20=E2=80=94=20what=20held=20a=20child's=20stream,=20and=20why?= =?UTF-8?q?=20a=20block=20ran=20whole?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The live tail (P99-FINDINGS.md): three large blocks opened their stream 104–181 ms after their header was decoded and gained nothing; the record had OpenDelay but nothing to attribute it with. Every streamed block now carries a timeline in its record (unix nanos: header ready, header seen, parent full, parent admitted, parent replayed, first wait entry after the parent, opened, first group start, full, finalize start) and OpenDelay decomposed into attributable waits, each zero when the timeline cannot support it: header ready ─(OpenWaitParentArrival)─▶ parent full ─(OpenWaitParentReplay = OpenWaitParentQueue + OpenWaitParentExec)─▶ parent replayed ─(OpenWaitLoop = OpenWaitPostReplay + OpenWaitDispatch)─▶ opened so "the parent's shreds came late", "the parent sat in the source behind its own ancestors", "the parent's execution", "the parent's post-replay tail (promotion, RPC, stats) held the child" and "the header was there and the loop still took its time" are separable per block. The loop records a mark for the last executed block (admission when the source handed it over, replayed at the instant FullToReplayed ends, and its first entry into the replay wait afterwards); the executor uses the mark only when it is the child's parent (a re-based frontier reads as unknown, never as the wrong slot). The arithmetic runs on the wall-clock nanos the record carries, so instants and components are exactly consistent for anyone joining them. A block executed whole now says why no stream opened for it (NotOpenedReason: header_not_seen, declined:, discarded:, waiting_for_parent:…), from a per-slot observation the executor keeps until the frontier passes the slot. The open log line prints the timeline relative to the header. summarize_replay_timings.py (shared folder) prints the decomposition and a --tail of the worst full->replayed blocks with their instants relative to their last shred. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/metrics/metrics.go | 79 +++++++++- pkg/replay/block.go | 20 ++- pkg/replay/streaming.go | 216 +++++++++++++++++++++++++- pkg/replay/streaming_realfeed_test.go | 29 ++++ pkg/replay/streaming_test.go | 163 ++++++++++++++++++- 5 files changed, 497 insertions(+), 10 deletions(-) diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 8628093b0..b9480c199 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -150,9 +150,11 @@ type VoteRewardDetails struct { // wake-up; Transactions counts what they executed. TxLoopBeforeFull is the // group execution wall time that finished before the slot was fully // assembled, i.e. the work hidden behind reception. OpenDelay runs from the -// header batch being decoded to the stream opening (it measures how long the -// parent's tail held the child). Discarded is 1 when a stream for this slot -// was thrown away and the block was executed whole; DiscardReason names why. +// header batch being decoded to the stream opening; the timeline fields and +// the OpenWait* timings below say what held the child (its parent's arrival, +// its parent's replay, or the loop itself). Discarded is 1 when a stream for +// this slot was thrown away and the block was executed whole; DiscardReason +// names why. type StreamingExecution struct { Opened uint64 Groups uint64 @@ -161,6 +163,77 @@ type StreamingExecution struct { OpenDelay Timing Discarded uint64 DiscardReason string + + // Timeline: wall-clock unix nanoseconds of the events that bound the + // stream's open, zero when unknown. They join with the parent's record + // (ParentFullNanos is the parent's FullNanos) and with external captures. + // + // HeaderReadyNanos the child's header batch was decoded + // HeaderSeenNanos the executor first handled that header (from its + // wake-up, or recovered from the assembler after a + // dropped wake-up, in which case ready is the + // lookup instant and seen follows it at once) + // ParentFullNanos the parent's last shred (0: parent was a skip or + // not a turbine block) + // ParentAdmittedNanos the source handed the parent to replay (its own + // ancestors replayed, the emitter released it) + // ParentReplayedNanos the parent's replay result reached consensus and + // the frontier advanced to it (0: unknown, e.g. the + // frontier was re-based by a fork switch) + // OpenedNanos the stream's bank opened + // FirstGroupStartNanos the first executed group started + // WaitEnteredNanos the loop first entered the replay wait after the + // parent was replayed (0: unknown) + // FullNanos this block's last shred + // FinalizeStartNanos the complete block reached the stream + HeaderReadyNanos int64 + HeaderSeenNanos int64 + ParentFullNanos int64 + ParentAdmittedNanos int64 + ParentReplayedNanos int64 + WaitEnteredNanos int64 + OpenedNanos int64 + FirstGroupStartNanos int64 + FullNanos int64 + FinalizeStartNanos int64 + + // OpenDelay decomposed into attributable waits (each zero when the + // timeline cannot support it): + // OpenWaitParentArrival header ready → parent's last shred: the child's + // header was decoded before its parent was even + // fully received (a leader/arrival gap, not ours) + // OpenWaitParentReplay parent's last shred (or header ready, whichever + // is later) → parent replayed: the parent's own + // post-full path held the child; split, when the + // parent's admission is known, into + // OpenWaitParentQueue … → the parent's admission: the parent's own + // post-full path in the source (completion, + // verification, and waiting for its ancestors — + // a leader window's earlier slots still replaying) + // OpenWaitParentExec admission → replayed: the parent's execution + // and tail + // OpenWaitLoop parent replayed (or header seen, whichever is + // later) → opened: the replay loop's own latency + // to open once nothing else stood in the way; + // split, when the wait entry is known, into + // OpenWaitPostReplay … → the loop's first wait entry after the + // parent: the parent's post-replay tail + // (promotion, RPC, stats) held the child + // OpenWaitDispatch wait entry (or header seen) → opened: events + // ahead of the header in the feed, the poll + OpenWaitParentArrival Timing + OpenWaitParentReplay Timing + OpenWaitParentQueue Timing + OpenWaitParentExec Timing + OpenWaitLoop Timing + OpenWaitPostReplay Timing + OpenWaitDispatch Timing + + // NotOpenedReason is set when the block was executed whole without a + // stream having opened for it: why the executor never opened one + // ("header_not_seen", "declined:", "waiting_for_parent:…"). + // Empty when a stream opened (see Discarded for the ones thrown away). + NotOpenedReason string } // Metrics for replaying a single block diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 2f70e9df8..8f2f84820 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -2524,6 +2524,11 @@ func ReplayBlocks( // the wait keeps its exact pre-streaming behaviour. var streamer *streamingExecutor var streamInput replayStreamer + // frontierMark is the timeline record of the last executed block (skips + // do not touch it: a child's parent is always a real block); the executor + // reads it to attribute a child's open delay to its parent's arrival, its + // parent's replay, or the loop itself. + var frontierMark streamingFrontierMark if blockStream.StreamEvents() != nil { streamer = newStreamingExecutor(streamingDeps{ acctsDb: acctsDb, @@ -2539,6 +2544,7 @@ func ReplayBlocks( unrootedTailUsed: unrootedTailState != nil, lastSlotCtx: func() *sealevel.SlotCtx { return lastSlotCtx }, frontier: func() uint64 { return replayFrontier }, + frontierMark: func() streamingFrontierMark { return frontierMark }, currentFeatures: func() *features.Features { return replayCtx.CurrentFeatures }, currentEpoch: func() uint64 { return currentEpoch }, rewardsInFlight: func() bool { @@ -2576,6 +2582,7 @@ func ReplayBlocks( ingressTimings *b.TurbineIngressTimings waitTime time.Duration neededAt time.Time // when replay asked the source for this slot + admittedAt time.Time // when the source handed replay this input ) { @@ -2609,6 +2616,11 @@ func ReplayBlocks( } neededAt = time.Now() + if frontierMark.waitEnteredAt.IsZero() { + // First wait after the last executed block: what precedes it is + // that block's post-replay tail (promotion, RPC, stats). + frontierMark.waitEnteredAt = neededAt + } block, parentSwitch, certifiedSwitch = waitForReplayInput(ctx, blockStream.NextReplayInput, sweepWhileWaiting, decisionChanges, alpenglowSwitchPollInterval, streamInput) if ingress, ok := block.CompleteTurbineReplayAdmission(time.Now()); ok { @@ -2616,7 +2628,8 @@ func ReplayBlocks( ingressTimings = &ingress } - waitTime = time.Since(neededAt) + admittedAt = time.Now() + waitTime = admittedAt.Sub(neededAt) if stallDone != nil { close(stallDone) @@ -3106,6 +3119,7 @@ func ReplayBlocks( streamer.discard("other_block") } if !streamed { + streamer.noteWholeBlock(block) lastSlotCtx, err = ProcessBlock(acctsDb, block, epochSchedule, txParallelism, dbgOpts, persistedHashes, unrootedTailState, transactionStatuses, alpenglowClock, parentBankSysvars) } } @@ -3178,6 +3192,10 @@ func ReplayBlocks( } } recordFullToReplayed(block) + // The same instant FullToReplayed ends at: from here to the next wait + // entry is this block's post-replay tail, which a child's open timeline + // reports as OpenWaitPostReplay. + frontierMark = streamingFrontierMark{slot: block.Slot, fullNanos: block.ShredFullNanos, admittedAt: admittedAt, replayedAt: time.Now()} if rpcServer != nil { rpcServer.SetSlotCtx(lastSlotCtx) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index d18e655bd..361a20b6a 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -111,8 +111,11 @@ type streamingDeps struct { transactionStatuses *TransactionStatusCache alpenglowClock bool - lastSlotCtx func() *sealevel.SlotCtx - frontier func() uint64 + lastSlotCtx func() *sealevel.SlotCtx + frontier func() uint64 + // frontierMark reports the last executed block's replay and full instants + // (timeline only; nil when the loop does not track it). + frontierMark func() streamingFrontierMark currentFeatures func() *features.Features currentEpoch func() uint64 rewardsInFlight func() bool @@ -127,6 +130,39 @@ type streamingGroup struct { transactions int } +// streamingFrontierMark is the replay loop's record of the last executed +// block: the slot, the instant its replay result reached consensus, and its +// last-shred instant (0 for a block that did not arrive as shreds). Skips do +// not update it. It only feeds the timeline; the executor never decides +// anything on it. +type streamingFrontierMark struct { + slot uint64 + fullNanos int64 + // admittedAt is when the source handed the block to replay (after its + // ancestors were replayed and the emitter released it); replayedAt when + // its replay result reached consensus. + admittedAt time.Time + replayedAt time.Time + // waitEnteredAt is the loop's first entry into the replay wait after + // replayedAt; what lies between is the executed block's post-replay tail. + waitEnteredAt time.Time +} + +// streamingObservation is what the executor knows about a slot's header. It +// outlives the header itself (pruned when the frontier passes the slot) so +// that a block executed whole can report why no stream opened for it, and a +// stream can report how long its header waited and on what. +type streamingObservation struct { + generation turbine.StreamGeneration + parentSlot uint64 + readyAt time.Time // header batch decoded (its wake-up's ReadyAt) + seenAt time.Time // executor first handled the header + frontierAtSeen uint64 + declined string // eligibility reason, when the header was declined + discarded string // discard reason, when a stream opened and was thrown away + openedAt time.Time +} + // streamingSlot is one in-progress stream. type streamingSlot struct { slot uint64 @@ -145,6 +181,14 @@ type streamingSlot struct { openedAt time.Time headerAt time.Time groups []streamingGroup + // timeline: what bounded the open (see metrics.StreamingExecution). Kept + // here rather than in the collector because the loop resets the collector + // before every wait and the block may arrive several waits after the open. + headerSeenAt time.Time + parentFullNanos int64 + parentAdmittedAt time.Time + parentReplayedAt time.Time + waitEnteredAt time.Time // restoreSysvarCache puts the legacy process-global sysvar cache back to // its state before the bank opened; nil when nothing was published. restoreSysvarCache func() @@ -162,7 +206,10 @@ type streamingExecutor struct { // declined; header recovery (recoverHeader) never reopens it. Pruned with // headers. retired map[uint64]turbine.StreamGeneration - ticker *time.Ticker + // observed is the per-slot header timeline (see streamingObservation), + // pruned with headers. + observed map[uint64]*streamingObservation + ticker *time.Ticker // executeFn runs one group on the open execution; tests substitute it. executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error } @@ -172,6 +219,7 @@ func newStreamingExecutor(deps streamingDeps) *streamingExecutor { deps: deps, headers: make(map[uint64]*turbine.StreamBatch), retired: make(map[uint64]turbine.StreamGeneration), + observed: make(map[uint64]*streamingObservation), executeFn: (*blockExecution).executeTransactionGroup, } } @@ -227,6 +275,7 @@ func (s *streamingExecutor) shutdown() { s.stopTicker() s.headers = make(map[uint64]*turbine.StreamBatch) s.retired = make(map[uint64]turbine.StreamGeneration) + s.observed = make(map[uint64]*streamingObservation) } // handleEvent consumes one feed wake-up. @@ -313,6 +362,15 @@ func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { return } s.headers[header.Slot] = header + if obs := s.observed[header.Slot]; obs == nil || obs.generation != header.Generation { + s.observed[header.Slot] = &streamingObservation{ + generation: header.Generation, + parentSlot: header.ParentSlot, + readyAt: header.ReadyAt, + seenAt: time.Now(), + frontierAtSeen: frontier, + } + } s.pruneHeaders(frontier) } @@ -369,6 +427,11 @@ func (s *streamingExecutor) pruneHeaders(frontier uint64) { delete(s.retired, slot) } } + for slot := range s.observed { + if slot <= frontier { + delete(s.observed, slot) + } + } } // nextHeader prefers the next slot, otherwise the earliest nearby child of @@ -409,6 +472,9 @@ func (s *streamingExecutor) tryOpen() { if reason := s.eligibility(header); reason != "" { mlog.Log.FileOnlyf("streaming: slot %d not opened (%s)", header.Slot, reason) s.retire(header.Slot, header.Generation) + if obs := s.observed[header.Slot]; obs != nil && obs.generation == header.Generation { + obs.declined = reason + } continue } s.openStream(header) @@ -492,7 +558,7 @@ func (s *streamingExecutor) openStream(header *turbine.StreamBatch) { exec.slotCtx.TrackProgramCacheAdds = true exec.setReplayStage("streaming_wait") - s.current = &streamingSlot{ + cur := &streamingSlot{ slot: shell.Slot, generation: header.Generation, parentSlot: header.ParentSlot, @@ -501,11 +567,30 @@ func (s *streamingExecutor) openStream(header *turbine.StreamBatch) { pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: header.ReadyAt, + headerSeenAt: header.ReadyAt, restoreSysvarCache: func() { sealevel.SysvarCache = sysvarCacheAtOpen }, } + if obs := s.observed[shell.Slot]; obs != nil && obs.generation == header.Generation { + obs.openedAt = cur.openedAt + if !obs.seenAt.IsZero() { + cur.headerSeenAt = obs.seenAt + } + } + if d.frontierMark != nil { + // The mark is only the child's parent when the last executed block is + // that very slot; after a fork-switch re-base it is not, and the + // timeline says "unknown" rather than blaming the wrong slot. + if mark := d.frontierMark(); mark.slot == header.ParentSlot && !mark.replayedAt.IsZero() { + cur.parentFullNanos = mark.fullNanos + cur.parentAdmittedAt = mark.admittedAt + cur.parentReplayedAt = mark.replayedAt + cur.waitEnteredAt = mark.waitEnteredAt + } + } + s.current = cur metrics.GlobalBlockReplay.StreamingExecution.Opened = 1 d.feed.PrioritizeStreamRepair(shell.Slot) - mlog.Log.FileOnlyf("streaming: opened slot %d on parent %d", shell.Slot, header.ParentSlot) + mlog.Log.FileOnlyf("streaming: opened slot %d on parent %d | %s", shell.Slot, header.ParentSlot, cur.openTimeline()) s.offer(header) s.pull() s.consume() @@ -663,6 +748,9 @@ func (s *streamingExecutor) discard(reason string) { s.current = nil s.stopTicker() s.retire(cur.slot, cur.generation) + if obs := s.observed[cur.slot]; obs != nil && obs.generation == cur.generation { + obs.discarded = reason + } exec := cur.exec if exec != nil { exec.close() @@ -725,6 +813,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel return nil, false, nil } cur := s.current + finalizeStart := time.Now() if reason := s.handshake(block, parentBankSysvars); reason != "" { s.discard("prefix_mismatch:" + reason) return nil, false, nil @@ -839,6 +928,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel } } record.OpenDelay.AddTiming(cur.openedAt.Sub(cur.headerAt)) + cur.recordTimeline(record, block, finalizeStart) return slotCtx, true, nil } @@ -878,3 +968,119 @@ func (s *streamingExecutor) handshake(block *b.Block, parentBankSysvars *sealeve } return "" } + +// nanosOf is a time as unix nanoseconds, 0 for the zero time. +func nanosOf(t time.Time) int64 { + if t.IsZero() { + return 0 + } + return t.UnixNano() +} + +// recordTimeline writes the stream's open timeline and its decomposition +// into the block's record. Every wait is attributed to exactly one thing: +// +// header ready ─(parent arrival)─▶ parent full ─(parent replay)─▶ parent +// replayed ─(loop: post-replay tail │ dispatch)─▶ opened +// +// with the header's own wake-up latency (ready → seen) reported alongside. +// Each component is zero when the timeline cannot support it (unknown +// instants, or an ordering that makes it empty). The arithmetic is done on +// the wall-clock nanos the record carries, so the components and the +// instants are exactly consistent for whoever joins them later. +func (cur *streamingSlot) recordTimeline(record *metrics.StreamingExecution, block *b.Block, finalizeStart time.Time) { + ready, seen, opened := nanosOf(cur.headerAt), nanosOf(cur.headerSeenAt), nanosOf(cur.openedAt) + if seen == 0 { + seen = ready + } + parentFull, parentAdmitted, parentReplayed, waitEntered := cur.parentFullNanos, nanosOf(cur.parentAdmittedAt), nanosOf(cur.parentReplayedAt), nanosOf(cur.waitEnteredAt) + record.HeaderReadyNanos = ready + record.HeaderSeenNanos = seen + record.OpenedNanos = opened + record.ParentFullNanos = parentFull + record.ParentAdmittedNanos = parentAdmitted + record.ParentReplayedNanos = parentReplayed + record.WaitEnteredNanos = waitEntered + if len(cur.groups) > 0 { + record.FirstGroupStartNanos = nanosOf(cur.groups[0].startedAt) + } + if block != nil && block.ShredFullNanos > 0 { + record.FullNanos = block.ShredFullNanos + } + record.FinalizeStartNanos = nanosOf(finalizeStart) + + add := func(timing *metrics.Timing, from, to int64) { + if from > 0 && to > from { + timing.AddTiming(time.Duration(to - from)) + } + } + add(&record.OpenWaitParentArrival, ready, parentFull) + if parentReplayed > 0 { + replayStart := max(ready, parentFull) + add(&record.OpenWaitParentReplay, replayStart, parentReplayed) + // The parent's replay splits at its admission: before it the parent + // was still in the source (completion, verification, waiting for its + // own ancestors); after it, its execution and tail. Queue + Exec == + // ParentReplay whatever the ordering. + if parentAdmitted > 0 { + add(&record.OpenWaitParentQueue, replayStart, min(parentAdmitted, parentReplayed)) + add(&record.OpenWaitParentExec, max(parentAdmitted, replayStart), parentReplayed) + } + } + loopStart := max(seen, parentReplayed) + add(&record.OpenWaitLoop, loopStart, opened) + // The loop's wait splits at its first entry into the replay wait after + // the parent: before it is the parent's post-replay tail, after it the + // dispatch of the child's header (queued events ahead of it, the poll). + if waitEntered > 0 && parentReplayed > 0 && waitEntered > parentReplayed { + add(&record.OpenWaitPostReplay, loopStart, min(waitEntered, opened)) + add(&record.OpenWaitDispatch, max(waitEntered, seen), opened) + } +} + +// openTimeline renders the open's timeline for the log, relative to the +// header's decode instant. +func (cur *streamingSlot) openTimeline() string { + base := nanosOf(cur.headerAt) + rel := func(nanos int64) string { + if nanos == 0 { + return "?" + } + return fmt.Sprintf("%+.1fms", float64(nanos-base)/1e6) + } + return fmt.Sprintf("header seen %s, parent full %s, parent admitted %s, parent replayed %s, wait entered %s, opened %s (vs header ready)", + rel(nanosOf(cur.headerSeenAt)), rel(cur.parentFullNanos), rel(nanosOf(cur.parentAdmittedAt)), rel(nanosOf(cur.parentReplayedAt)), rel(nanosOf(cur.waitEnteredAt)), rel(nanosOf(cur.openedAt))) +} + +// noteWholeBlock records, for a block about to execute whole, why no stream +// opened for it (the block's record otherwise only says Opened == 0). The +// header timeline is filled in when the header was seen, so the analysis can +// tell "never decoded a header" from "decoded one and could not use it". +func (s *streamingExecutor) noteWholeBlock(block *b.Block) { + if s == nil || block == nil || block.IsSkipped { + return + } + record := &metrics.GlobalBlockReplay.StreamingExecution + if block.ShredFullNanos > 0 { + record.FullNanos = block.ShredFullNanos + } + obs := s.observed[block.Slot] + switch { + case obs == nil: + record.NotOpenedReason = "header_not_seen" + return + case obs.discarded != "": + record.NotOpenedReason = "discarded:" + obs.discarded + case obs.declined != "": + record.NotOpenedReason = "declined:" + obs.declined + case !obs.openedAt.IsZero(): + // Unreachable in practice (an opened stream ends in finalize or in a + // discard, which records itself); kept so the record never lies. + record.NotOpenedReason = "opened_not_discarded" + default: + record.NotOpenedReason = fmt.Sprintf("waiting_for_parent:header_on_parent_%d_seen_at_frontier_%d", obs.parentSlot, obs.frontierAtSeen) + } + record.HeaderReadyNanos = nanosOf(obs.readyAt) + record.HeaderSeenNanos = nanosOf(obs.seenAt) + record.OpenedNanos = nanosOf(obs.openedAt) +} diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index 7c44db6c9..6780d713c 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -65,6 +65,9 @@ type realFeedRig struct { statuses *TransactionStatusCache exec *streamingExecutor block *b.Block + // mark is the loop's record of the executed parent (replayed before any + // shred of the child is broadcast), as the loop would hold it. + mark streamingFrontierMark // the slot's content, as wire bytes, decoded fresh for every use legacyWires, v0Wires [][]byte } @@ -142,6 +145,8 @@ func newRealFeedRig(t *testing.T, dest solana.PublicKey, eventBuffer int) *realF rig.tail = &lifecycleTail{durable: env.durable} rig.statuses = NewTransactionStatusCache() + now := time.Now() + rig.mark = streamingFrontierMark{slot: realFeedParentSlot, fullNanos: now.Add(-10 * time.Millisecond).UnixNano(), admittedAt: now.Add(-5 * time.Millisecond), replayedAt: now, waitEnteredAt: now.Add(time.Microsecond)} rig.exec = newStreamingExecutor(streamingDeps{ acctsDb: env.acctsDb, feed: rig.feed, @@ -154,6 +159,7 @@ func newRealFeedRig(t *testing.T, dest solana.PublicKey, eventBuffer int) *realF unrootedTailUsed: true, lastSlotCtx: func() *sealevel.SlotCtx { return rig.lastCtx }, frontier: func() uint64 { return realFeedParentSlot }, + frontierMark: func() streamingFrontierMark { return rig.mark }, currentFeatures: func() *features.Features { return env.feats }, currentEpoch: func() uint64 { return 0 }, rewardsInFlight: func() bool { return false }, @@ -304,6 +310,29 @@ func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { require.Equal(rig.t, uint64(1), record.Opened) require.Equal(rig.t, uint64(len(rig.allWires())), record.Transactions) require.Zero(rig.t, record.Discarded, "discard reason %q", record.DiscardReason) + require.Empty(rig.t, record.NotOpenedReason) + + // The open timeline, in order: the parent was replayed before the child's + // header was decoded, the header was seen no earlier than decoded, the + // stream opened after that, executed, and finalized after the last shred. + require.Equal(rig.t, rig.mark.replayedAt.UnixNano(), record.ParentReplayedNanos) + require.Equal(rig.t, rig.mark.admittedAt.UnixNano(), record.ParentAdmittedNanos) + require.Equal(rig.t, rig.mark.fullNanos, record.ParentFullNanos) + require.Less(rig.t, record.ParentReplayedNanos, record.HeaderReadyNanos) + require.LessOrEqual(rig.t, record.HeaderReadyNanos, record.HeaderSeenNanos) + require.LessOrEqual(rig.t, record.HeaderSeenNanos, record.OpenedNanos) + require.LessOrEqual(rig.t, record.OpenedNanos, record.FirstGroupStartNanos) + require.Equal(rig.t, block.ShredFullNanos, record.FullNanos) + require.LessOrEqual(rig.t, record.FullNanos, record.FinalizeStartNanos) + require.Zero(rig.t, record.OpenWaitParentArrival.Count, "the parent was fully received before the header") + require.Zero(rig.t, record.OpenWaitParentReplay.Count, "the parent was replayed before the header") + require.Zero(rig.t, record.OpenWaitParentQueue.Count) + require.Zero(rig.t, record.OpenWaitParentExec.Count) + require.Equal(rig.t, uint64(1), record.OpenWaitLoop.Count) + require.Equal(rig.t, uint64(record.OpenedNanos-record.HeaderSeenNanos), record.OpenWaitLoop.SumNanoseconds, "with nothing to wait for, the whole open delay is the loop's") + require.Equal(rig.t, rig.mark.waitEnteredAt.UnixNano(), record.WaitEnteredNanos) + require.Zero(rig.t, record.OpenWaitPostReplay.Count, "the header arrived after the loop was already waiting") + require.Equal(rig.t, record.OpenWaitLoop, record.OpenWaitDispatch, "…so the loop's delay is all dispatch") } func TestStreamingRealFeedExecutesPrefixBeforeCompletionAndMatchesWholeBlock(t *testing.T) { diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 8e0ff9880..24a35932f 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -134,8 +134,8 @@ func (h *streamingTestHarness) open() { parentID: h.parentID, exec: h.env.exec, pending: make(map[uint32]*turbine.StreamBatch), - openedAt: time.Now(), headerAt: time.Now(), + openedAt: time.Now(), } h.exec.handleEvent(h.event(h.header())) } @@ -849,3 +849,164 @@ func TestStreamingGapSelectionAndSafety(t *testing.T) { require.True(t, cur.exec.closed) require.Equal(t, h.gen, h.exec.retired[46]) } + +// The open timeline attributes every millisecond between the header's decode +// and the open to exactly one of: the parent's arrival (header decoded before +// the parent's last shred), the parent's replay (last shred → replay result), +// or the loop (nothing left to wait for, still not opened). +func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { + t0 := time.Unix(1_700_000_000, 0) + at := func(ms int) time.Time { return t0.Add(time.Duration(ms) * time.Millisecond) } + block := &b.Block{ShredFullNanos: at(300).UnixNano()} + record := func(cur *streamingSlot) metrics.StreamingExecution { + var out metrics.StreamingExecution + cur.recordTimeline(&out, block, at(320)) + return out + } + ms := func(timing metrics.Timing) float64 { return float64(timing.SumNanoseconds) / 1e6 } + + // Header decoded 50 ms before the parent's last shred, parent admitted + // 20 ms after that and replayed 10 ms later, opened 5 ms after; the + // header wake-up was handled 10 ms after decode. + cur := &streamingSlot{ + headerAt: at(0), headerSeenAt: at(10), openedAt: at(85), + parentFullNanos: at(50).UnixNano(), parentAdmittedAt: at(70), parentReplayedAt: at(80), + groups: []streamingGroup{{startedAt: at(90), finishedAt: at(120), transactions: 3}}, + } + r := record(cur) + require.Equal(t, at(0).UnixNano(), r.HeaderReadyNanos) + require.Equal(t, at(10).UnixNano(), r.HeaderSeenNanos) + require.Equal(t, at(50).UnixNano(), r.ParentFullNanos) + require.Equal(t, at(70).UnixNano(), r.ParentAdmittedNanos) + require.Equal(t, at(80).UnixNano(), r.ParentReplayedNanos) + require.Equal(t, at(85).UnixNano(), r.OpenedNanos) + require.Equal(t, at(90).UnixNano(), r.FirstGroupStartNanos) + require.Equal(t, at(300).UnixNano(), r.FullNanos) + require.Equal(t, at(320).UnixNano(), r.FinalizeStartNanos) + require.Equal(t, 50.0, ms(r.OpenWaitParentArrival)) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 20.0, ms(r.OpenWaitParentQueue), "the parent sat in the source 20 ms after its last shred") + require.Equal(t, 10.0, ms(r.OpenWaitParentExec)) + require.Equal(t, 5.0, ms(r.OpenWaitLoop)) + require.Equal(t, uint64(1), r.OpenWaitLoop.Count) + + // The parent was admitted before the child's header was decoded (its + // queueing cannot have held the child): the replay wait is all execution. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(1), openedAt: at(40), parentFullNanos: at(-30).UnixNano(), parentAdmittedAt: at(-5), parentReplayedAt: at(30)} + r = record(cur) + require.Zero(t, r.OpenWaitParentQueue.Count) + require.Equal(t, 30.0, ms(r.OpenWaitParentExec)) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + + // Parent fully received before the header was even decoded: no arrival + // wait; the parent's replay wait starts at the header. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(1), openedAt: at(40), parentFullNanos: at(-20).UnixNano(), parentReplayedAt: at(30)} + r = record(cur) + require.Zero(t, r.OpenWaitParentArrival.Count) + require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 10.0, ms(r.OpenWaitLoop)) + + // Header handled only after the parent was replayed (the wake-up sat in + // the channel): the loop wait runs from the header being seen. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(90), openedAt: at(95), parentFullNanos: at(20).UnixNano(), parentReplayedAt: at(60)} + r = record(cur) + require.Equal(t, 20.0, ms(r.OpenWaitParentArrival)) + require.Equal(t, 40.0, ms(r.OpenWaitParentReplay)) + require.Equal(t, 5.0, ms(r.OpenWaitLoop)) + + // The loop wait splits at the first wait entry after the parent: a header + // remembered while the parent executed waited through the parent's + // post-replay tail (replayed → wait entry) and then its dispatch (wait + // entry → opened). + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(5), openedAt: at(130), parentFullNanos: at(-100).UnixNano(), parentReplayedAt: at(60), waitEnteredAt: at(125)} + r = record(cur) + require.Equal(t, at(125).UnixNano(), r.WaitEnteredNanos) + require.Equal(t, 70.0, ms(r.OpenWaitLoop)) + require.Equal(t, 65.0, ms(r.OpenWaitPostReplay)) + require.Equal(t, 5.0, ms(r.OpenWaitDispatch)) + + // A header seen only after the wait entry (it arrived while the loop was + // already waiting) was not held by the tail: dispatch only, from seen. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(140), openedAt: at(141), parentFullNanos: at(-100).UnixNano(), parentReplayedAt: at(60), waitEnteredAt: at(125)} + r = record(cur) + require.Zero(t, r.OpenWaitPostReplay.Count) + require.Equal(t, 1.0, ms(r.OpenWaitDispatch)) + require.Equal(t, 1.0, ms(r.OpenWaitLoop)) + require.Contains(t, cur.openTimeline(), "wait entered +125.0ms") + + // Unknown parent instants (no mark, or a re-based frontier): only the + // loop wait, from the header being seen. + cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(0), openedAt: at(70)} + r = record(cur) + require.Zero(t, r.ParentFullNanos) + require.Zero(t, r.ParentReplayedNanos) + require.Zero(t, r.OpenWaitParentArrival.Count) + require.Zero(t, r.OpenWaitParentReplay.Count) + require.Zero(t, r.OpenWaitParentQueue.Count) + require.Zero(t, r.OpenWaitParentExec.Count) + require.Equal(t, 70.0, ms(r.OpenWaitLoop)) + require.Zero(t, r.OpenWaitPostReplay.Count) + require.Zero(t, r.OpenWaitDispatch.Count) + require.Zero(t, r.WaitEnteredNanos) + require.Zero(t, r.FirstGroupStartNanos, "no group ran") + require.Contains(t, cur.openTimeline(), "parent full ?") + require.Contains(t, cur.openTimeline(), "opened +70.0ms") +} + +// A block executed whole says why no stream opened for it, from the +// executor's per-slot observation of its header. +func TestStreamingNoteWholeBlockReasons(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + reset := func() { metrics.GlobalBlockReplay.StreamingExecution = metrics.StreamingExecution{} } + reason := func() string { return metrics.GlobalBlockReplay.StreamingExecution.NotOpenedReason } + + // Discarded stream, recorded on the observation (the record's own + // Discarded/DiscardReason may or may not have survived the loop's + // per-attempt reset; the observation always has it). + reset() + h.exec.observed[42] = &streamingObservation{generation: h.gen, parentSlot: 41, readyAt: time.Now(), seenAt: time.Now(), frontierAtSeen: 41} + h.exec.discard("timeout") + h.exec.noteWholeBlock(&b.Block{Slot: 42, ShredFullNanos: 123}) + require.Equal(t, "discarded:timeout", reason()) + require.Equal(t, int64(123), metrics.GlobalBlockReplay.StreamingExecution.FullNanos) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.HeaderReadyNanos) + require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.HeaderSeenNanos) + + // Never saw a header. + reset() + delete(h.exec.observed, 42) + h.exec.noteWholeBlock(&b.Block{Slot: 42}) + require.Equal(t, "header_not_seen", reason()) + + // Declined: the header's generation is unknown to the feed (gone). + reset() + declined := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(42), 0, 0, turbine.StreamMarkerHeader, 41, h.parentID) + h.exec.handleEvent(h.event(declined)) + require.Nil(t, h.exec.current) + h.exec.noteWholeBlock(&b.Block{Slot: 42}) + require.Contains(t, reason(), "declined:generation no longer active") + + // Waiting: a header whose parent is not the executed frontier stays + // remembered, and a whole-block execution of it reports what it waited on. + reset() + waiting := turbine.NewDetachedStreamMarker(turbine.NewDetachedStreamGeneration(44), 0, 0, turbine.StreamMarkerHeader, 43, solana.Hash{}) + h.exec.handleEvent(h.event(waiting)) + require.Contains(t, h.exec.headers, uint64(44)) + h.exec.noteWholeBlock(&b.Block{Slot: 44}) + require.Equal(t, "waiting_for_parent:header_on_parent_43_seen_at_frontier_41", reason()) + + // Skips never report; a nil executor is a no-op. + reset() + h.exec.noteWholeBlock(&b.Block{Slot: 44, IsSkipped: true}) + require.Empty(t, reason()) + var none *streamingExecutor + none.noteWholeBlock(&b.Block{Slot: 44}) + require.Empty(t, reason()) + + // Observations are pruned with the frontier. + h.frontier = 44 + h.exec.pruneHeaders(h.frontier) + require.Empty(t, h.exec.observed) +} From cb63e0c96a84c329e1f06455d3442a164d95a2d2 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 15:19:38 +0000 Subject: [PATCH 054/111] =?UTF-8?q?replay:=20streaming=20per-group=20recor?= =?UTF-8?q?d=20=E2=80=94=20verifier=20waits,=20execution=20after=20the=20l?= =?UTF-8?q?ast=20shred,=20group=20sizes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit REMAINING-OUTLIERS.md asked for the next measurement: per-group readiness, verification-wait, execution start/end and size, because consume() joins every contiguous decoded batch into one group and executeGroup waits for the verifier to finish all of them before executing any — after a late open that is the whole backlog, and the record could not say whether the tail was verification wait or execution. Each group now records when it was formed, when its last batch was verified, and its execution bounds and size; the block's record gains GroupVerifyWait (and the part after the last shred), TxLoopAfterFull (the execution FullToReplayed actually paid for: groups straddling or after the last shred, plus the suffix), GroupsStraddlingFull, the largest group and the last group's end. Blocks with more than 30 ms of execution or 5 ms of verifier wait after the last shred log the per-group timeline (bounded). summarize_replay_timings.py prints the group view and three more --tail columns. Nothing about grouping or dispatch changes. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/metrics/metrics.go | 21 ++++++ pkg/replay/streaming.go | 97 +++++++++++++++++++++++++-- pkg/replay/streaming_realfeed_test.go | 6 ++ pkg/replay/streaming_test.go | 43 ++++++++++++ 4 files changed, 163 insertions(+), 4 deletions(-) diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index b9480c199..f27177101 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -229,6 +229,27 @@ type StreamingExecution struct { OpenWaitPostReplay Timing OpenWaitDispatch Timing + // Groups, from the executor's per-group bookkeeping (each group is the + // contiguous set of decoded batches that was ready at one wake-up, or + // the finalize suffix): + // GroupVerifyWait time groups spent waiting for the verifier + // to finish their last batch before executing + // GroupVerifyWaitAfterFull the part of that after the last shred + // TxLoopAfterFull group/suffix execution after the last shred: + // the execution FullToReplayed actually paid for + // GroupsStraddlingFull groups that started before and finished after + // LargestGroupTransactions / LargestGroupBatches: the biggest group (a + // late open turns the whole backlog into one); + // Batches is 0 when that group was the suffix + // LastGroupEndNanos when the last group (or suffix) finished + GroupVerifyWait Timing + GroupVerifyWaitAfterFull Timing + TxLoopAfterFull Timing + GroupsStraddlingFull uint64 + LargestGroupTransactions uint64 + LargestGroupBatches uint64 + LastGroupEndNanos int64 + // NotOpenedReason is set when the block was executed whole without a // stream having opened for it: why the executor never opened one // ("header_not_seen", "declined:", "waiting_for_parent:…"). diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index 361a20b6a..a117b112c 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -126,8 +126,13 @@ type streamingDeps struct { } type streamingGroup struct { - startedAt, finishedAt time.Time - transactions int + // readyAt is when the group was formed from contiguous decoded batches; + // verifiedAt when the verifier had finished every batch in it (the + // executor waits for that before executing anything); startedAt and + // finishedAt bound the execution itself. + readyAt, verifiedAt, startedAt, finishedAt time.Time + batches, transactions int + suffix bool // the finalize suffix, run on the complete block } // streamingFrontierMark is the replay loop's record of the last executed @@ -679,6 +684,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { var txs []*solana.Transaction var verified []txverify.VerifiedMessageIdentity allVerified := true + readyAt := time.Now() for _, batch := range group { identities, ok, err := batch.WaitVerification(context.Background()) if errors.Is(err, turbine.ErrStreamBatchUnverified) { @@ -693,6 +699,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { txs = append(txs, batch.Transactions...) verified = append(verified, identities...) } + verifiedAt := time.Now() if len(txs) == 0 { return nil } @@ -734,7 +741,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { return fmt.Errorf("group: %w", err) } cur.origin = append(cur.origin, txs...) - cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(txs)}) + cur.groups = append(cur.groups, streamingGroup{readyAt: readyAt, verifiedAt: verifiedAt, startedAt: started, finishedAt: time.Now(), batches: len(group), transactions: len(txs)}) return nil } @@ -887,7 +894,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel if err != nil { return fail("suffix", fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)) } - cur.groups = append(cur.groups, streamingGroup{startedAt: started, finishedAt: time.Now(), transactions: len(suffix)}) + cur.groups = append(cur.groups, streamingGroup{readyAt: started, verifiedAt: started, startedAt: started, finishedAt: time.Now(), transactions: len(suffix), suffix: true}) } if exec.processedSignatures != executionPlan.processedSignatures || exec.processedTxCount != executionPlan.processedTxCount { return fail("counts", fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", @@ -929,6 +936,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel } record.OpenDelay.AddTiming(cur.openedAt.Sub(cur.headerAt)) cur.recordTimeline(record, block, finalizeStart) + cur.recordGroups(record, fullAt) return slotCtx, true, nil } @@ -1038,6 +1046,87 @@ func (cur *streamingSlot) recordTimeline(record *metrics.StreamingExecution, blo } } +// recordGroups writes the per-group view: how long groups waited for the +// verifier before executing, how much of that and of the execution itself +// fell after the last shred (the part FullToReplayed pays for), and the +// largest group (a late open turns the whole backlog into one group). The +// suffix counts as a group that starts after full. +func (cur *streamingSlot) recordGroups(record *metrics.StreamingExecution, fullAt time.Time) { + // Without a full instant nothing is "after full" (as TxLoopBeforeFull + // and FullToReplayed record nothing either); waits and sizes still count. + after := func(from, to time.Time) time.Duration { + if fullAt.IsZero() { + return 0 + } + if fullAt.After(from) { + from = fullAt + } + if to.After(from) { + return to.Sub(from) + } + return 0 + } + for _, group := range cur.groups { + if wait := group.verifiedAt.Sub(group.readyAt); wait > 0 && !group.suffix { + record.GroupVerifyWait.AddTiming(wait) + if late := after(group.readyAt, group.verifiedAt); late > 0 { + record.GroupVerifyWaitAfterFull.AddTiming(late) + } + } + if late := after(group.startedAt, group.finishedAt); late > 0 { + record.TxLoopAfterFull.AddTiming(late) + if !group.suffix && !fullAt.IsZero() && group.startedAt.Before(fullAt) { + record.GroupsStraddlingFull++ + } + } + if uint64(group.transactions) > record.LargestGroupTransactions { + record.LargestGroupTransactions = uint64(group.transactions) + record.LargestGroupBatches = uint64(group.batches) + } + } + if n := len(cur.groups); n > 0 { + record.LastGroupEndNanos = nanosOf(cur.groups[n-1].finishedAt) + } + // The tail cases are worth a per-group line; bounded so a heavy block + // with hundreds of groups does not flood the log. + if record.TxLoopAfterFull.SumNanoseconds > uint64(30*time.Millisecond) || record.GroupVerifyWaitAfterFull.SumNanoseconds > uint64(5*time.Millisecond) { + mlog.Log.FileOnlyf("streaming: slot %d groups (vs last shred): %s", cur.slot, cur.groupTimeline(fullAt, 12)) + } +} + +// groupTimeline renders up to limit groups (the first ones and the last) +// relative to fullAt: "[#0 b=3 tx=1200 ready-150.2 verified-149.8 exec-149.8..-140.1]". +func (cur *streamingSlot) groupTimeline(fullAt time.Time, limit int) string { + rel := func(t time.Time) string { + if t.IsZero() || fullAt.IsZero() { + return "?" + } + return fmt.Sprintf("%+.1f", float64(t.Sub(fullAt).Microseconds())/1e3) + } + var out []byte + render := func(i int) { + g := cur.groups[i] + kind := "" + if g.suffix { + kind = " suffix" + } + out = fmt.Appendf(out, "[#%d%s b=%d tx=%d ready%s verified%s exec%s..%s]", i, kind, g.batches, g.transactions, rel(g.readyAt), rel(g.verifiedAt), rel(g.startedAt), rel(g.finishedAt)) + } + n := len(cur.groups) + if n <= limit { + for i := range cur.groups { + render(i) + } + return string(out) + } + for i := 0; i < limit-1; i++ { + render(i) + } + out = fmt.Appendf(out, "…(%d more)", n-limit) + render(n - 1) + return string(out) +} + // openTimeline renders the open's timeline for the log, relative to the // header's decode instant. func (cur *streamingSlot) openTimeline() string { diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index 6780d713c..a99419e7a 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -331,6 +331,12 @@ func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { require.Equal(rig.t, uint64(1), record.OpenWaitLoop.Count) require.Equal(rig.t, uint64(record.OpenedNanos-record.HeaderSeenNanos), record.OpenWaitLoop.SumNanoseconds, "with nothing to wait for, the whole open delay is the loop's") require.Equal(rig.t, rig.mark.waitEnteredAt.UnixNano(), record.WaitEnteredNanos) + // Every group ran before the completion was even broadcast. + require.Positive(rig.t, record.LastGroupEndNanos) + require.LessOrEqual(rig.t, record.LastGroupEndNanos, record.FullNanos) + require.Zero(rig.t, record.TxLoopAfterFull.Count) + require.Zero(rig.t, record.GroupsStraddlingFull) + require.Contains(rig.t, []uint64{uint64(len(rig.legacyWires)), uint64(len(rig.allWires()))}, record.LargestGroupTransactions, "one or two groups, depending on how the batches were pulled") require.Zero(rig.t, record.OpenWaitPostReplay.Count, "the header arrived after the loop was already waiting") require.Equal(rig.t, record.OpenWaitLoop, record.OpenWaitDispatch, "…so the loop's delay is all dispatch") } diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 24a35932f..fcfa2ee3b 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -1010,3 +1010,46 @@ func TestStreamingNoteWholeBlockReasons(t *testing.T) { h.exec.pruneHeaders(h.frontier) require.Empty(t, h.exec.observed) } + +// The per-group record splits verification waits and execution at the last +// shred, so the execution FullToReplayed paid for is visible separately +// from the work hidden behind reception. +func TestStreamingGroupRecordSplitsAtFull(t *testing.T) { + t0 := time.Unix(1_700_000_000, 0) + at := func(ms int) time.Time { return t0.Add(time.Duration(ms) * time.Millisecond) } + ms := func(timing metrics.Timing) float64 { return float64(timing.SumNanoseconds) / 1e6 } + cur := &streamingSlot{slot: 42, groups: []streamingGroup{ + {readyAt: at(90), verifiedAt: at(92), startedAt: at(92), finishedAt: at(120), batches: 3, transactions: 900}, // entirely before full + {readyAt: at(250), verifiedAt: at(310), startedAt: at(310), finishedAt: at(340), batches: 9, transactions: 5000}, // waited 60 ms for the verifier, 10 of them after full; ran after full + {readyAt: at(280), verifiedAt: at(281), startedAt: at(281), finishedAt: at(320), batches: 1, transactions: 300}, // straddles full + {readyAt: at(340), verifiedAt: at(340), startedAt: at(340), finishedAt: at(350), transactions: 40, suffix: true}, + }} + var r metrics.StreamingExecution + cur.recordGroups(&r, at(300)) + require.Equal(t, 63.0, ms(r.GroupVerifyWait)) + require.Equal(t, uint64(3), r.GroupVerifyWait.Count, "the suffix has no verifier wait") + require.Equal(t, 10.0, ms(r.GroupVerifyWaitAfterFull)) + require.Equal(t, uint64(1), r.GroupVerifyWaitAfterFull.Count) + require.Equal(t, 60.0, ms(r.TxLoopAfterFull), "30 + 20 + 10 ms of execution after the last shred") + require.Equal(t, uint64(3), r.TxLoopAfterFull.Count) + require.Equal(t, uint64(1), r.GroupsStraddlingFull) + require.Equal(t, uint64(5000), r.LargestGroupTransactions) + require.Equal(t, uint64(9), r.LargestGroupBatches) + require.Equal(t, at(350).UnixNano(), r.LastGroupEndNanos) + line := cur.groupTimeline(at(300), 3) + require.Contains(t, line, "[#0 b=3 tx=900 ready-210.0 verified-208.0 exec-208.0..-180.0]") + require.Contains(t, line, "…(1 more)") + require.Contains(t, line, "[#3 suffix b=0 tx=40 ready+40.0 verified+40.0 exec+40.0..+50.0]") + require.NotContains(t, line, "#2 ") + + // No full instant (a block that did not arrive as shreds): nothing is + // "after full", waits and sizes still count. + var whole metrics.StreamingExecution + cur.recordGroups(&whole, time.Time{}) + require.Equal(t, 63.0, ms(whole.GroupVerifyWait)) + require.Zero(t, whole.GroupVerifyWaitAfterFull.Count) + require.Zero(t, whole.TxLoopAfterFull.Count) + require.Zero(t, whole.GroupsStraddlingFull) + require.Equal(t, uint64(5000), whole.LargestGroupTransactions) + require.Equal(t, at(350).UnixNano(), whole.LastGroupEndNanos) +} From b87bd57b5a818e9399a60e798211b6c5de30585a Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 10:45:37 -0500 Subject: [PATCH 055/111] replay: correct streaming timing attribution --- pkg/metrics/metrics.go | 63 +++++++++++++------------- pkg/replay/streaming.go | 54 +++++++++++----------- pkg/replay/streaming_realfeed_test.go | 6 +-- pkg/replay/streaming_test.go | 64 +++++++++++++++++---------- 4 files changed, 102 insertions(+), 85 deletions(-) diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index f27177101..213955d0d 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -201,40 +201,35 @@ type StreamingExecution struct { // timeline cannot support it): // OpenWaitParentArrival header ready → parent's last shred: the child's // header was decoded before its parent was even - // fully received (a leader/arrival gap, not ours) + // fully received (arrival timing; cause not established) // OpenWaitParentReplay parent's last shred (or header ready, whichever // is later) → parent replayed: the parent's own // post-full path held the child; split, when the // parent's admission is known, into - // OpenWaitParentQueue … → the parent's admission: the parent's own - // post-full path in the source (completion, - // verification, and waiting for its ancestors — - // a leader window's earlier slots still replaying) - // OpenWaitParentExec admission → replayed: the parent's execution - // and tail - // OpenWaitLoop parent replayed (or header seen, whichever is - // later) → opened: the replay loop's own latency - // to open once nothing else stood in the way; - // split, when the wait entry is known, into - // OpenWaitPostReplay … → the loop's first wait entry after the - // parent: the parent's post-replay tail - // (promotion, RPC, stats) held the child - // OpenWaitDispatch wait entry (or header seen) → opened: events - // ahead of the header in the feed, the poll - OpenWaitParentArrival Timing - OpenWaitParentReplay Timing - OpenWaitParentQueue Timing - OpenWaitParentExec Timing - OpenWaitLoop Timing - OpenWaitPostReplay Timing - OpenWaitDispatch Timing + // OpenWaitParentPreAdmission … → admission, including any streamed execution + // OpenWaitParentPostAdmission admission → replayed, including finalization + // OpenWaitLoop max(parent replayed, header ready) → opened; + // includes time before the header is handled + // OpenWaitPostReplay loop start → first wait entry after the parent + // OpenWaitDispatch max(wait entry, loop start) → opened + // These are elapsed intervals, not CPU times. Subdivisions must not be + // added to their aggregates. Unknown parent milestones limit attribution. + OpenWaitParentArrival Timing + OpenWaitParentReplay Timing + OpenWaitParentPreAdmission Timing + OpenWaitParentPostAdmission Timing + OpenWaitLoop Timing + OpenWaitPostReplay Timing + OpenWaitDispatch Timing // Groups, from the executor's per-group bookkeeping (each group is the // contiguous set of decoded batches that was ready at one wake-up, or // the finalize suffix): - // GroupVerifyWait time groups spent waiting for the verifier - // to finish their last batch before executing - // GroupVerifyWaitAfterFull the part of that after the last shred + // GroupJoinAssembly verification joins plus batch-slice assembly; + // not pure cryptography or verifier queue time + // GroupJoinAssemblyAfterFull intersection of that interval with post-full + // GroupPreparation identity binding and execution-copy creation + // GroupPreparationAfterFull intersection of preparation with post-full // TxLoopAfterFull group/suffix execution after the last shred: // the execution FullToReplayed actually paid for // GroupsStraddlingFull groups that started before and finished after @@ -242,13 +237,15 @@ type StreamingExecution struct { // late open turns the whole backlog into one); // Batches is 0 when that group was the suffix // LastGroupEndNanos when the last group (or suffix) finished - GroupVerifyWait Timing - GroupVerifyWaitAfterFull Timing - TxLoopAfterFull Timing - GroupsStraddlingFull uint64 - LargestGroupTransactions uint64 - LargestGroupBatches uint64 - LastGroupEndNanos int64 + GroupJoinAssembly Timing + GroupJoinAssemblyAfterFull Timing + GroupPreparation Timing + GroupPreparationAfterFull Timing + TxLoopAfterFull Timing + GroupsStraddlingFull uint64 + LargestGroupTransactions uint64 + LargestGroupBatches uint64 + LastGroupEndNanos int64 // NotOpenedReason is set when the block was executed whole without a // stream having opened for it: why the executor never opened one diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index a117b112c..8eaac27c8 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -127,12 +127,12 @@ type streamingDeps struct { type streamingGroup struct { // readyAt is when the group was formed from contiguous decoded batches; - // verifiedAt when the verifier had finished every batch in it (the - // executor waits for that before executing anything); startedAt and + // joinedAt when the consumer finished joining verification and copying + // batch slices (not the verifier completion instant); startedAt and // finishedAt bound the execution itself. - readyAt, verifiedAt, startedAt, finishedAt time.Time - batches, transactions int - suffix bool // the finalize suffix, run on the complete block + readyAt, joinedAt, startedAt, finishedAt time.Time + batches, transactions int + suffix bool // the finalize suffix, run on the complete block } // streamingFrontierMark is the replay loop's record of the last executed @@ -699,7 +699,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { txs = append(txs, batch.Transactions...) verified = append(verified, identities...) } - verifiedAt := time.Now() + joinedAt := time.Now() if len(txs) == 0 { return nil } @@ -741,7 +741,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { return fmt.Errorf("group: %w", err) } cur.origin = append(cur.origin, txs...) - cur.groups = append(cur.groups, streamingGroup{readyAt: readyAt, verifiedAt: verifiedAt, startedAt: started, finishedAt: time.Now(), batches: len(group), transactions: len(txs)}) + cur.groups = append(cur.groups, streamingGroup{readyAt: readyAt, joinedAt: joinedAt, startedAt: started, finishedAt: time.Now(), batches: len(group), transactions: len(txs)}) return nil } @@ -894,7 +894,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel if err != nil { return fail("suffix", fmt.Errorf("execute block suffix at slot %d: %w", block.Slot, err)) } - cur.groups = append(cur.groups, streamingGroup{readyAt: started, verifiedAt: started, startedAt: started, finishedAt: time.Now(), transactions: len(suffix), suffix: true}) + cur.groups = append(cur.groups, streamingGroup{readyAt: started, joinedAt: started, startedAt: started, finishedAt: time.Now(), transactions: len(suffix), suffix: true}) } if exec.processedSignatures != executionPlan.processedSignatures || exec.processedTxCount != executionPlan.processedTxCount { return fail("counts", fmt.Errorf("streaming execution at slot %d processed %d transactions/%d signatures, block plan has %d/%d", @@ -1026,28 +1026,26 @@ func (cur *streamingSlot) recordTimeline(record *metrics.StreamingExecution, blo if parentReplayed > 0 { replayStart := max(ready, parentFull) add(&record.OpenWaitParentReplay, replayStart, parentReplayed) - // The parent's replay splits at its admission: before it the parent - // was still in the source (completion, verification, waiting for its - // own ancestors); after it, its execution and tail. Queue + Exec == - // ParentReplay whatever the ordering. + // Admission is a milestone, not an execution boundary: streaming + // can execute inside waitForReplayInput before admission. if parentAdmitted > 0 { - add(&record.OpenWaitParentQueue, replayStart, min(parentAdmitted, parentReplayed)) - add(&record.OpenWaitParentExec, max(parentAdmitted, replayStart), parentReplayed) + add(&record.OpenWaitParentPreAdmission, replayStart, min(parentAdmitted, parentReplayed)) + add(&record.OpenWaitParentPostAdmission, max(parentAdmitted, replayStart), parentReplayed) } } - loopStart := max(seen, parentReplayed) + loopStart := max(ready, parentReplayed) add(&record.OpenWaitLoop, loopStart, opened) // The loop's wait splits at its first entry into the replay wait after // the parent: before it is the parent's post-replay tail, after it the // dispatch of the child's header (queued events ahead of it, the poll). if waitEntered > 0 && parentReplayed > 0 && waitEntered > parentReplayed { add(&record.OpenWaitPostReplay, loopStart, min(waitEntered, opened)) - add(&record.OpenWaitDispatch, max(waitEntered, seen), opened) + add(&record.OpenWaitDispatch, max(waitEntered, loopStart), opened) } } -// recordGroups writes the per-group view: how long groups waited for the -// verifier before executing, how much of that and of the execution itself +// recordGroups writes the per-group view: joining/assembly, preparation, +// and execution elapsed intervals, including how much of each interval // fell after the last shred (the part FullToReplayed pays for), and the // largest group (a late open turns the whole backlog into one group). The // suffix counts as a group that starts after full. @@ -1067,10 +1065,16 @@ func (cur *streamingSlot) recordGroups(record *metrics.StreamingExecution, fullA return 0 } for _, group := range cur.groups { - if wait := group.verifiedAt.Sub(group.readyAt); wait > 0 && !group.suffix { - record.GroupVerifyWait.AddTiming(wait) - if late := after(group.readyAt, group.verifiedAt); late > 0 { - record.GroupVerifyWaitAfterFull.AddTiming(late) + if wait := group.joinedAt.Sub(group.readyAt); wait > 0 && !group.suffix { + record.GroupJoinAssembly.AddTiming(wait) + if late := after(group.readyAt, group.joinedAt); late > 0 { + record.GroupJoinAssemblyAfterFull.AddTiming(late) + } + } + if !group.suffix && group.startedAt.After(group.joinedAt) { + record.GroupPreparation.AddTiming(group.startedAt.Sub(group.joinedAt)) + if late := after(group.joinedAt, group.startedAt); late > 0 { + record.GroupPreparationAfterFull.AddTiming(late) } } if late := after(group.startedAt, group.finishedAt); late > 0 { @@ -1089,13 +1093,13 @@ func (cur *streamingSlot) recordGroups(record *metrics.StreamingExecution, fullA } // The tail cases are worth a per-group line; bounded so a heavy block // with hundreds of groups does not flood the log. - if record.TxLoopAfterFull.SumNanoseconds > uint64(30*time.Millisecond) || record.GroupVerifyWaitAfterFull.SumNanoseconds > uint64(5*time.Millisecond) { + if record.TxLoopAfterFull.SumNanoseconds > uint64(30*time.Millisecond) || record.GroupJoinAssemblyAfterFull.SumNanoseconds > uint64(5*time.Millisecond) { mlog.Log.FileOnlyf("streaming: slot %d groups (vs last shred): %s", cur.slot, cur.groupTimeline(fullAt, 12)) } } // groupTimeline renders up to limit groups (the first ones and the last) -// relative to fullAt: "[#0 b=3 tx=1200 ready-150.2 verified-149.8 exec-149.8..-140.1]". +// relative to fullAt: "[#0 b=3 tx=1200 ready-150.2 joined-149.8 exec-149.8..-140.1]". func (cur *streamingSlot) groupTimeline(fullAt time.Time, limit int) string { rel := func(t time.Time) string { if t.IsZero() || fullAt.IsZero() { @@ -1110,7 +1114,7 @@ func (cur *streamingSlot) groupTimeline(fullAt time.Time, limit int) string { if g.suffix { kind = " suffix" } - out = fmt.Appendf(out, "[#%d%s b=%d tx=%d ready%s verified%s exec%s..%s]", i, kind, g.batches, g.transactions, rel(g.readyAt), rel(g.verifiedAt), rel(g.startedAt), rel(g.finishedAt)) + out = fmt.Appendf(out, "[#%d%s b=%d tx=%d ready%s joined%s exec%s..%s]", i, kind, g.batches, g.transactions, rel(g.readyAt), rel(g.joinedAt), rel(g.startedAt), rel(g.finishedAt)) } n := len(cur.groups) if n <= limit { diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index a99419e7a..c9fad4163 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -326,10 +326,10 @@ func (rig *realFeedRig) finalizeAndCompare(reference lifecycleOutcome) { require.LessOrEqual(rig.t, record.FullNanos, record.FinalizeStartNanos) require.Zero(rig.t, record.OpenWaitParentArrival.Count, "the parent was fully received before the header") require.Zero(rig.t, record.OpenWaitParentReplay.Count, "the parent was replayed before the header") - require.Zero(rig.t, record.OpenWaitParentQueue.Count) - require.Zero(rig.t, record.OpenWaitParentExec.Count) + require.Zero(rig.t, record.OpenWaitParentPreAdmission.Count) + require.Zero(rig.t, record.OpenWaitParentPostAdmission.Count) require.Equal(rig.t, uint64(1), record.OpenWaitLoop.Count) - require.Equal(rig.t, uint64(record.OpenedNanos-record.HeaderSeenNanos), record.OpenWaitLoop.SumNanoseconds, "with nothing to wait for, the whole open delay is the loop's") + require.Equal(rig.t, uint64(record.OpenedNanos-record.HeaderReadyNanos), record.OpenWaitLoop.SumNanoseconds, "with nothing to wait for, the whole open delay is the loop's") require.Equal(rig.t, rig.mark.waitEnteredAt.UnixNano(), record.WaitEnteredNanos) // Every group ran before the completion was even broadcast. require.Positive(rig.t, record.LastGroupEndNanos) diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index fcfa2ee3b..b3b4e1847 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -885,8 +885,8 @@ func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { require.Equal(t, at(320).UnixNano(), r.FinalizeStartNanos) require.Equal(t, 50.0, ms(r.OpenWaitParentArrival)) require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) - require.Equal(t, 20.0, ms(r.OpenWaitParentQueue), "the parent sat in the source 20 ms after its last shred") - require.Equal(t, 10.0, ms(r.OpenWaitParentExec)) + require.Equal(t, 20.0, ms(r.OpenWaitParentPreAdmission), "the parent sat in the source 20 ms after its last shred") + require.Equal(t, 10.0, ms(r.OpenWaitParentPostAdmission)) require.Equal(t, 5.0, ms(r.OpenWaitLoop)) require.Equal(t, uint64(1), r.OpenWaitLoop.Count) @@ -894,8 +894,8 @@ func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { // queueing cannot have held the child): the replay wait is all execution. cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(1), openedAt: at(40), parentFullNanos: at(-30).UnixNano(), parentAdmittedAt: at(-5), parentReplayedAt: at(30)} r = record(cur) - require.Zero(t, r.OpenWaitParentQueue.Count) - require.Equal(t, 30.0, ms(r.OpenWaitParentExec)) + require.Zero(t, r.OpenWaitParentPreAdmission.Count) + require.Equal(t, 30.0, ms(r.OpenWaitParentPostAdmission)) require.Equal(t, 30.0, ms(r.OpenWaitParentReplay)) // Parent fully received before the header was even decoded: no arrival @@ -907,12 +907,13 @@ func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { require.Equal(t, 10.0, ms(r.OpenWaitLoop)) // Header handled only after the parent was replayed (the wake-up sat in - // the channel): the loop wait runs from the header being seen. + // the channel): include the queued wake-up before the header was seen. cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(90), openedAt: at(95), parentFullNanos: at(20).UnixNano(), parentReplayedAt: at(60)} r = record(cur) require.Equal(t, 20.0, ms(r.OpenWaitParentArrival)) require.Equal(t, 40.0, ms(r.OpenWaitParentReplay)) - require.Equal(t, 5.0, ms(r.OpenWaitLoop)) + require.Equal(t, 35.0, ms(r.OpenWaitLoop)) + require.Equal(t, 95.0, ms(r.OpenWaitParentArrival)+ms(r.OpenWaitParentReplay)+ms(r.OpenWaitLoop)) // The loop wait splits at the first wait entry after the parent: a header // remembered while the parent executed waited through the parent's @@ -926,12 +927,12 @@ func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { require.Equal(t, 5.0, ms(r.OpenWaitDispatch)) // A header seen only after the wait entry (it arrived while the loop was - // already waiting) was not held by the tail: dispatch only, from seen. + // already waiting) includes both the tail and queued header dispatch. cur = &streamingSlot{headerAt: at(0), headerSeenAt: at(140), openedAt: at(141), parentFullNanos: at(-100).UnixNano(), parentReplayedAt: at(60), waitEnteredAt: at(125)} r = record(cur) - require.Zero(t, r.OpenWaitPostReplay.Count) - require.Equal(t, 1.0, ms(r.OpenWaitDispatch)) - require.Equal(t, 1.0, ms(r.OpenWaitLoop)) + require.Equal(t, 65.0, ms(r.OpenWaitPostReplay)) + require.Equal(t, 16.0, ms(r.OpenWaitDispatch)) + require.Equal(t, 81.0, ms(r.OpenWaitLoop)) require.Contains(t, cur.openTimeline(), "wait entered +125.0ms") // Unknown parent instants (no mark, or a re-based frontier): only the @@ -942,8 +943,8 @@ func TestStreamingTimelineAttributesOpenDelay(t *testing.T) { require.Zero(t, r.ParentReplayedNanos) require.Zero(t, r.OpenWaitParentArrival.Count) require.Zero(t, r.OpenWaitParentReplay.Count) - require.Zero(t, r.OpenWaitParentQueue.Count) - require.Zero(t, r.OpenWaitParentExec.Count) + require.Zero(t, r.OpenWaitParentPreAdmission.Count) + require.Zero(t, r.OpenWaitParentPostAdmission.Count) require.Equal(t, 70.0, ms(r.OpenWaitLoop)) require.Zero(t, r.OpenWaitPostReplay.Count) require.Zero(t, r.OpenWaitDispatch.Count) @@ -1019,17 +1020,17 @@ func TestStreamingGroupRecordSplitsAtFull(t *testing.T) { at := func(ms int) time.Time { return t0.Add(time.Duration(ms) * time.Millisecond) } ms := func(timing metrics.Timing) float64 { return float64(timing.SumNanoseconds) / 1e6 } cur := &streamingSlot{slot: 42, groups: []streamingGroup{ - {readyAt: at(90), verifiedAt: at(92), startedAt: at(92), finishedAt: at(120), batches: 3, transactions: 900}, // entirely before full - {readyAt: at(250), verifiedAt: at(310), startedAt: at(310), finishedAt: at(340), batches: 9, transactions: 5000}, // waited 60 ms for the verifier, 10 of them after full; ran after full - {readyAt: at(280), verifiedAt: at(281), startedAt: at(281), finishedAt: at(320), batches: 1, transactions: 300}, // straddles full - {readyAt: at(340), verifiedAt: at(340), startedAt: at(340), finishedAt: at(350), transactions: 40, suffix: true}, + {readyAt: at(90), joinedAt: at(92), startedAt: at(92), finishedAt: at(120), batches: 3, transactions: 900}, // entirely before full + {readyAt: at(250), joinedAt: at(310), startedAt: at(310), finishedAt: at(340), batches: 9, transactions: 5000}, // waited 60 ms for the verifier, 10 of them after full; ran after full + {readyAt: at(280), joinedAt: at(281), startedAt: at(281), finishedAt: at(320), batches: 1, transactions: 300}, // straddles full + {readyAt: at(340), joinedAt: at(340), startedAt: at(340), finishedAt: at(350), transactions: 40, suffix: true}, }} var r metrics.StreamingExecution cur.recordGroups(&r, at(300)) - require.Equal(t, 63.0, ms(r.GroupVerifyWait)) - require.Equal(t, uint64(3), r.GroupVerifyWait.Count, "the suffix has no verifier wait") - require.Equal(t, 10.0, ms(r.GroupVerifyWaitAfterFull)) - require.Equal(t, uint64(1), r.GroupVerifyWaitAfterFull.Count) + require.Equal(t, 63.0, ms(r.GroupJoinAssembly)) + require.Equal(t, uint64(3), r.GroupJoinAssembly.Count, "the suffix has no verifier wait") + require.Equal(t, 10.0, ms(r.GroupJoinAssemblyAfterFull)) + require.Equal(t, uint64(1), r.GroupJoinAssemblyAfterFull.Count) require.Equal(t, 60.0, ms(r.TxLoopAfterFull), "30 + 20 + 10 ms of execution after the last shred") require.Equal(t, uint64(3), r.TxLoopAfterFull.Count) require.Equal(t, uint64(1), r.GroupsStraddlingFull) @@ -1037,19 +1038,34 @@ func TestStreamingGroupRecordSplitsAtFull(t *testing.T) { require.Equal(t, uint64(9), r.LargestGroupBatches) require.Equal(t, at(350).UnixNano(), r.LastGroupEndNanos) line := cur.groupTimeline(at(300), 3) - require.Contains(t, line, "[#0 b=3 tx=900 ready-210.0 verified-208.0 exec-208.0..-180.0]") + require.Contains(t, line, "[#0 b=3 tx=900 ready-210.0 joined-208.0 exec-208.0..-180.0]") require.Contains(t, line, "…(1 more)") - require.Contains(t, line, "[#3 suffix b=0 tx=40 ready+40.0 verified+40.0 exec+40.0..+50.0]") + require.Contains(t, line, "[#3 suffix b=0 tx=40 ready+40.0 joined+40.0 exec+40.0..+50.0]") require.NotContains(t, line, "#2 ") // No full instant (a block that did not arrive as shreds): nothing is // "after full", waits and sizes still count. var whole metrics.StreamingExecution cur.recordGroups(&whole, time.Time{}) - require.Equal(t, 63.0, ms(whole.GroupVerifyWait)) - require.Zero(t, whole.GroupVerifyWaitAfterFull.Count) + require.Equal(t, 63.0, ms(whole.GroupJoinAssembly)) + require.Zero(t, whole.GroupJoinAssemblyAfterFull.Count) require.Zero(t, whole.TxLoopAfterFull.Count) require.Zero(t, whole.GroupsStraddlingFull) require.Equal(t, uint64(5000), whole.LargestGroupTransactions) require.Equal(t, at(350).UnixNano(), whole.LastGroupEndNanos) } + +// The group stages account for preparation even when it crosses completion. +func TestStreamingGroupPreparationAccounting(t *testing.T) { + base := time.Unix(1700000000, 0) + at := func(ms int) time.Time { return base.Add(time.Duration(ms) * time.Millisecond) } + cur := &streamingSlot{groups: []streamingGroup{{readyAt: at(0), joinedAt: at(10), startedAt: at(30), finishedAt: at(50)}}} + var r metrics.StreamingExecution + cur.recordGroups(&r, at(20)) + require.Equal(t, uint64(10*time.Millisecond), r.GroupJoinAssembly.SumNanoseconds) + require.Equal(t, uint64(20*time.Millisecond), r.GroupPreparation.SumNanoseconds) + require.Zero(t, r.GroupJoinAssemblyAfterFull.SumNanoseconds) + require.Equal(t, uint64(10*time.Millisecond), r.GroupPreparationAfterFull.SumNanoseconds) + require.Equal(t, uint64(20*time.Millisecond), r.TxLoopAfterFull.SumNanoseconds) + require.Equal(t, uint64(30*time.Millisecond), r.GroupJoinAssemblyAfterFull.SumNanoseconds+r.GroupPreparationAfterFull.SumNanoseconds+r.TxLoopAfterFull.SumNanoseconds) +} From cfbdec7f72819c705a9e6cbe5748a5a43b83b51d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 11:11:34 -0500 Subject: [PATCH 056/111] replay: refresh ready batches before streaming group dispatch --- pkg/replay/streaming.go | 14 +++++++++----- pkg/replay/streaming_test.go | 26 ++++++++++++++++++++++++++ pkg/turbine/stream.go | 9 +++++---- pkg/turbine/stream_test.go | 16 +++++++++++++++- 4 files changed, 55 insertions(+), 10 deletions(-) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index 8eaac27c8..cf2eb776e 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -294,7 +294,13 @@ func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { return } if s.current != nil && event.Generation == s.current.generation { + // A previous group can leave many notifications queued. Refresh + // the authoritative ready set before choosing the next group. + if event.Batch.Start < s.current.nextStart { + return // already consumed by a prior refresh + } s.offer(event.Batch) + s.pull() s.consume() return } @@ -315,11 +321,9 @@ func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { case turbine.StreamCompleted: if s.current != nil && event.Generation == s.current.generation { s.current.completed = true - // Wake-ups for the slot's batches precede this event in the - // channel, so everything decoded is already pending; run it now - // (the group minimum no longer applies). Anything a dropped - // wake-up missed runs in the finalize suffix: the prefetch state - // is released at completion, so there is nothing left to pull. + // Released prefetch results remain immutable and owned by this + // generation. Recover ready batches whose wake-ups were dropped. + s.pull() s.consume() return } diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index b3b4e1847..6efaf443b 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -1069,3 +1069,29 @@ func TestStreamingGroupPreparationAccounting(t *testing.T) { require.Equal(t, uint64(20*time.Millisecond), r.TxLoopAfterFull.SumNanoseconds) require.Equal(t, uint64(30*time.Millisecond), r.GroupJoinAssemblyAfterFull.SumNanoseconds+r.GroupPreparationAfterFull.SumNanoseconds+r.TxLoopAfterFull.SumNanoseconds) } + +// One notification must pick up every ready contiguous batch, even when the +// remaining notifications are queued or the generation has just completed. +func TestStreamingEventRefreshesReadyBatches(t *testing.T) { + for _, completed := range []bool{false, true} { + t.Run(map[bool]string{false: "active", true: "completed"}[completed], func(t *testing.T) { + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() + h := newStreamingTestHarness(t) + txs := transferTransactions(t, 9, 1700) + a, b, c := h.batch(t, 1, 3, txs[:3]), h.batch(t, 4, 6, txs[3:6]), h.batch(t, 7, 9, txs[6:]) + h.feed.pending[h.gen] = []*turbine.StreamBatch{c, a, b} + if completed { + h.feed.status[h.gen] = turbine.StreamDone + h.exec.handleEvent(turbine.StreamEvent{Kind: turbine.StreamCompleted, Slot: a.Slot, Generation: h.gen}) + } else { + h.exec.handleEvent(h.event(a)) + } + sameTransactions(t, txs, h.executed()) + require.Len(t, h.exec.current.groups, 1) + h.exec.handleEvent(h.event(b)) + h.exec.handleEvent(h.event(c)) + require.Len(t, h.exec.current.groups, 1, "queued stale notifications do not execute twice") + }) + } +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index 2a6139749..95012e89d 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -227,16 +227,17 @@ func (a *SlotAssembler) streamStatusLocked(g StreamGeneration) StreamStatus { // PendingStreamBatches returns every decoded batch of the generation whose // range starts at or after fromStart, in shred-index order. It reads the // prefetch state directly, so it is the authoritative recovery path after a -// dropped wake-up. The result is empty once the generation is no longer -// active (its batches may still be used by a completed block, but the feed -// has nothing more to offer). +// dropped wake-up. A completed generation still owns its immutable ready +// results, so completion does not hide batches behind queued/lost notifications. +// Cancelled generations return nothing. No new prefetch work is scheduled here. func (a *SlotAssembler) PendingStreamBatches(g StreamGeneration, fromStart uint32) []*StreamBatch { if g.state == nil { return nil } a.mu.Lock() defer a.mu.Unlock() - if a.streamStatusLocked(g) != StreamActive || g.state.prefetch == nil || g.state.prefetch.released { + status := a.streamStatusLocked(g) + if status == StreamGone || g.state.prefetch == nil || (g.state.prefetch.released && status != StreamDone) { return nil } var out []*StreamBatch diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index 79dc0ce25..a9c08edce 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -98,7 +98,21 @@ func TestStreamFeedPublishesBatchesAndCompletion(t *testing.T) { done := nextStreamEvent(t, events, StreamCompleted) require.Equal(t, first.Generation, done.Generation) require.Equal(t, StreamDone, a.StreamStatusOf(first.Generation)) - require.Empty(t, a.PendingStreamBatches(first.Generation, 0), "a completed generation has nothing pending") + retained := a.PendingStreamBatches(first.Generation, first.Batch.Start) + require.Len(t, retained, 2, "completion preserves already-ready entry batches") + for i, batch := range retained { + ids, ok, err := batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, ok) + require.Len(t, ids, len(batch.Transactions)) + offset := 0 + if i == 1 { + offset = 3 + } + for j, tx := range batch.Transactions { + require.Same(t, blk.Transactions[offset+j], tx) + } + } // Pointer identity: the prefix a streaming consumer executed is the block. require.Len(t, blk.Transactions, 7) From d0f8237d7dd94ab08de9831b76f4395be0d6e49e Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 17:59:45 -0500 Subject: [PATCH 057/111] replay: retain account loader metrics across streaming groups --- pkg/metrics/account_loader_test.go | 40 +++++++++ pkg/metrics/metrics.go | 99 +++++++++++++++++++++++ pkg/replay/account_loader_metrics_test.go | 34 ++++++++ pkg/replay/block.go | 44 +++++----- pkg/replay/block_execution.go | 18 +++++ pkg/replay/streaming.go | 3 + pkg/replay/streaming_realfeed_test.go | 8 ++ 7 files changed, 224 insertions(+), 22 deletions(-) create mode 100644 pkg/metrics/account_loader_test.go create mode 100644 pkg/replay/account_loader_metrics_test.go diff --git a/pkg/metrics/account_loader_test.go b/pkg/metrics/account_loader_test.go new file mode 100644 index 000000000..fed0c86ed --- /dev/null +++ b/pkg/metrics/account_loader_test.go @@ -0,0 +1,40 @@ +package metrics + +import ( + "reflect" + "testing" +) + +// Exercise every field so adding a metric without merging it is caught. +func TestAccountLoaderAccumulateAllFields(t *testing.T) { + var src AccountLoader + var fill func(reflect.Value) + fill = func(v reflect.Value) { + for i := 0; i < v.NumField(); i++ { + f := v.Field(i) + if f.Kind() == reflect.Struct { + fill(f) + } else { + f.SetUint(uint64(i + 1)) + } + } + } + fill(reflect.ValueOf(&src).Elem()) + var dst AccountLoader + dst.Accumulate(src) + if dst != src { + t.Fatal("first merge lost fields") + } + dst.Accumulate(src) + var check func(reflect.Value, reflect.Value) + check = func(a, b reflect.Value) { + for i := 0; i < a.NumField(); i++ { + if a.Field(i).Kind() == reflect.Struct { + check(a.Field(i), b.Field(i)) + } else if a.Field(i).Uint() != 2*b.Field(i).Uint() { + t.Errorf("field %s not accumulated", a.Type().Field(i).Name) + } + } + } + check(reflect.ValueOf(dst), reflect.ValueOf(src)) +} diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 213955d0d..8b3903eb6 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -32,6 +32,8 @@ func (t *Timing) AddTimingSince(start time.Time) { // AccountLoader is the per-slot decomposition of LoadBlockAccounts. Counters // describe logical loader work; allocation counters cover objects/data created // directly by the batch loader rather than runtime or Pebble internals. +// Counts sum loader operations across groups, not unique keys/files per block; +// in particular UniqueAppendVecs sums each batch's distinct file count. type AccountLoader struct { AddressTableLookups Timing DedupeBlockAccounts Timing @@ -104,6 +106,103 @@ type AccountLoader struct { SysvarCachePublicationEpochRejects uint64 } +// Accumulate merges completed loader work. Both records must be exclusively +// owned by the replay goroutine; this is not a concurrent snapshot operation. +func (dst *AccountLoader) Accumulate(src AccountLoader) { + dst.AddressTableLookups.Count += src.AddressTableLookups.Count + dst.AddressTableLookups.SumNanoseconds += src.AddressTableLookups.SumNanoseconds + dst.DedupeBlockAccounts.Count += src.DedupeBlockAccounts.Count + dst.DedupeBlockAccounts.SumNanoseconds += src.DedupeBlockAccounts.SumNanoseconds + dst.SourceBatch.Count += src.SourceBatch.Count + dst.SourceBatch.SumNanoseconds += src.SourceBatch.SumNanoseconds + dst.ParentMapBuild.Count += src.ParentMapBuild.Count + dst.ParentMapBuild.SumNanoseconds += src.ParentMapBuild.SumNanoseconds + dst.SysvarUpdates.Count += src.SysvarUpdates.Count + dst.SysvarUpdates.SumNanoseconds += src.SysvarUpdates.SumNanoseconds + dst.SysvarClockRead.Count += src.SysvarClockRead.Count + dst.SysvarClockRead.SumNanoseconds += src.SysvarClockRead.SumNanoseconds + dst.SysvarSlotHashesRead.Count += src.SysvarSlotHashesRead.Count + dst.SysvarSlotHashesRead.SumNanoseconds += src.SysvarSlotHashesRead.SumNanoseconds + dst.SysvarRecentBlockhashesRead.Count += src.SysvarRecentBlockhashesRead.Count + dst.SysvarRecentBlockhashesRead.SumNanoseconds += src.SysvarRecentBlockhashesRead.SumNanoseconds + dst.SysvarSlotHistoryRead.Count += src.SysvarSlotHistoryRead.Count + dst.SysvarSlotHistoryRead.SumNanoseconds += src.SysvarSlotHistoryRead.SumNanoseconds + dst.SysvarStakeHistoryRead.Count += src.SysvarStakeHistoryRead.Count + dst.SysvarStakeHistoryRead.SumNanoseconds += src.SysvarStakeHistoryRead.SumNanoseconds + dst.SysvarLastRestartSlotRead.Count += src.SysvarLastRestartSlotRead.Count + dst.SysvarLastRestartSlotRead.SumNanoseconds += src.SysvarLastRestartSlotRead.SumNanoseconds + dst.WorkingSetLookup.Count += src.WorkingSetLookup.Count + dst.WorkingSetLookup.SumNanoseconds += src.WorkingSetLookup.SumNanoseconds + dst.InProgressLookup.Count += src.InProgressLookup.Count + dst.InProgressLookup.SumNanoseconds += src.InProgressLookup.SumNanoseconds + dst.AppendVecPinWait.Count += src.AppendVecPinWait.Count + dst.AppendVecPinWait.SumNanoseconds += src.AppendVecPinWait.SumNanoseconds + dst.ReadCacheEpochWait.Count += src.ReadCacheEpochWait.Count + dst.ReadCacheEpochWait.SumNanoseconds += src.ReadCacheEpochWait.SumNanoseconds + dst.CacheLookup.Count += src.CacheLookup.Count + dst.CacheLookup.SumNanoseconds += src.CacheLookup.SumNanoseconds + dst.AdmissionFilter.Count += src.AdmissionFilter.Count + dst.AdmissionFilter.SumNanoseconds += src.AdmissionFilter.SumNanoseconds + dst.IndexLookup.Count += src.IndexLookup.Count + dst.IndexLookup.SumNanoseconds += src.IndexLookup.SumNanoseconds + dst.ReadPlanning.Count += src.ReadPlanning.Count + dst.ReadPlanning.SumNanoseconds += src.ReadPlanning.SumNanoseconds + dst.AppendVecRead.Count += src.AppendVecRead.Count + dst.AppendVecRead.SumNanoseconds += src.AppendVecRead.SumNanoseconds + dst.CachePublicationWait.Count += src.CachePublicationWait.Count + dst.CachePublicationWait.SumNanoseconds += src.CachePublicationWait.SumNanoseconds + dst.CachePublication.Count += src.CachePublication.Count + dst.CachePublication.SumNanoseconds += src.CachePublication.SumNanoseconds + dst.SysvarWorkingSetLookup.Count += src.SysvarWorkingSetLookup.Count + dst.SysvarWorkingSetLookup.SumNanoseconds += src.SysvarWorkingSetLookup.SumNanoseconds + dst.SysvarClone.Count += src.SysvarClone.Count + dst.SysvarClone.SumNanoseconds += src.SysvarClone.SumNanoseconds + dst.SysvarAppendVecPinWait.Count += src.SysvarAppendVecPinWait.Count + dst.SysvarAppendVecPinWait.SumNanoseconds += src.SysvarAppendVecPinWait.SumNanoseconds + dst.SysvarInProgressLookup.Count += src.SysvarInProgressLookup.Count + dst.SysvarInProgressLookup.SumNanoseconds += src.SysvarInProgressLookup.SumNanoseconds + dst.SysvarReadCacheEpochWait.Count += src.SysvarReadCacheEpochWait.Count + dst.SysvarReadCacheEpochWait.SumNanoseconds += src.SysvarReadCacheEpochWait.SumNanoseconds + dst.SysvarCacheLookup.Count += src.SysvarCacheLookup.Count + dst.SysvarCacheLookup.SumNanoseconds += src.SysvarCacheLookup.SumNanoseconds + dst.SysvarIndexAndAppendVecRead.Count += src.SysvarIndexAndAppendVecRead.Count + dst.SysvarIndexAndAppendVecRead.SumNanoseconds += src.SysvarIndexAndAppendVecRead.SumNanoseconds + dst.SysvarCachePublicationWait.Count += src.SysvarCachePublicationWait.Count + dst.SysvarCachePublicationWait.SumNanoseconds += src.SysvarCachePublicationWait.SumNanoseconds + dst.SysvarCachePublication.Count += src.SysvarCachePublication.Count + dst.SysvarCachePublication.SumNanoseconds += src.SysvarCachePublication.SumNanoseconds + dst.RequestedKeys += src.RequestedKeys + dst.DurableKeys += src.DurableKeys + dst.ParentAccounts += src.ParentAccounts + dst.WorkingSetHits += src.WorkingSetHits + dst.InProgressHits += src.InProgressHits + dst.PendingFoldHits += src.PendingFoldHits + dst.CacheHits += src.CacheHits + dst.IndexHits += src.IndexHits + dst.IndexMisses += src.IndexMisses + dst.UniqueAppendVecs += src.UniqueAppendVecs + dst.AppendVecChunks += src.AppendVecChunks + dst.AppendVecAccounts += src.AppendVecAccounts + dst.OpenFailures += src.OpenFailures + dst.ReadFailures += src.ReadFailures + dst.RetryAccounts += src.RetryAccounts + dst.CommonCacheAdmissions += src.CommonCacheAdmissions + dst.CommonCacheAdmissionsSkipped += src.CommonCacheAdmissionsSkipped + dst.VoteCacheAdmissions += src.VoteCacheAdmissions + dst.VoteCacheAdmissionsSkipped += src.VoteCacheAdmissionsSkipped + dst.CachePublicationEpochRejects += src.CachePublicationEpochRejects + dst.DecodedAccountObjects += src.DecodedAccountObjects + dst.DecodedAccountBytes += src.DecodedAccountBytes + dst.PlaceholderObjects += src.PlaceholderObjects + dst.SysvarReads += src.SysvarReads + dst.SysvarWorkingSetHits += src.SysvarWorkingSetHits + dst.SysvarInProgressHits += src.SysvarInProgressHits + dst.SysvarPendingFoldHits += src.SysvarPendingFoldHits + dst.SysvarCacheHits += src.SysvarCacheHits + dst.SysvarDurableReads += src.SysvarDurableReads + dst.SysvarCachePublicationEpochRejects += src.SysvarCachePublicationEpochRejects +} + // TurbineIngress records per-slot pre-replay pipeline observations. // It is written to replay_timings.jsonl without high-cardinality metric labels. type TurbineIngress struct { diff --git a/pkg/replay/account_loader_metrics_test.go b/pkg/replay/account_loader_metrics_test.go new file mode 100644 index 000000000..128bd1cef --- /dev/null +++ b/pkg/replay/account_loader_metrics_test.go @@ -0,0 +1,34 @@ +package replay + +import ( + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/metrics" + "github.com/stretchr/testify/require" +) + +func TestAccountLoaderRetainsGroupsAcrossReset(t *testing.T) { + previous := metrics.GlobalBlockReplay.AccountLoader + t.Cleanup(func() { metrics.GlobalBlockReplay.AccountLoader = previous }) + exec := &blockExecution{} + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{RequestedKeys: 99} + func() { + defer exec.captureAccountLoader()() + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 3, IndexHits: 2, AppendVecAccounts: 2, AppendVecReadNanoseconds: 1000}) + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 4, CacheHits: 4}) + }() + require.EqualValues(t, 99, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys, "another slot's record is unchanged") + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} + func() { + defer exec.captureAccountLoader()() + recordAccountLoaderBatchStats(&metrics.GlobalBlockReplay.AccountLoader, accountsdb.BatchReadStats{RequestedKeys: 5, CacheHits: 5}) + }() + require.Zero(t, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys, "speculative work is not published before acceptance") + require.EqualValues(t, 12, exec.accountLoader.RequestedKeys) + require.EqualValues(t, 9, exec.accountLoader.CacheHits) + require.EqualValues(t, 2, exec.accountLoader.AppendVecAccounts) + require.EqualValues(t, time.Microsecond, exec.accountLoader.AppendVecRead.SumNanoseconds) + require.EqualValues(t, 3, exec.accountLoader.AppendVecRead.Count) +} diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 8f2f84820..e6b849b54 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1030,28 +1030,28 @@ func loadBlockAccountsAndUpdateSysvars( } func recordAccountLoaderBatchStats(dst *metrics.AccountLoader, src accountsdb.BatchReadStats) { - dst.RequestedKeys = src.RequestedKeys - dst.DurableKeys = src.DurableKeys - dst.WorkingSetHits = src.WorkingSetHits - dst.InProgressHits = src.InProgressHits - dst.PendingFoldHits = src.PendingFoldHits - dst.CacheHits = src.CacheHits - dst.IndexHits = src.IndexHits - dst.IndexMisses = src.IndexMisses - dst.UniqueAppendVecs = src.UniqueAppendVecs - dst.AppendVecChunks = src.AppendVecChunks - dst.AppendVecAccounts = src.AppendVecAccounts - dst.OpenFailures = src.OpenFailures - dst.ReadFailures = src.ReadFailures - dst.RetryAccounts = src.RetryAccounts - dst.CommonCacheAdmissions = src.CommonCacheAdmissions - dst.CommonCacheAdmissionsSkipped = src.CommonCacheAdmissionsSkipped - dst.VoteCacheAdmissions = src.VoteCacheAdmissions - dst.VoteCacheAdmissionsSkipped = src.VoteCacheAdmissionsSkipped - dst.CachePublicationEpochRejects = src.CachePublicationEpochRejects - dst.DecodedAccountObjects = src.DecodedAccountObjects - dst.DecodedAccountBytes = src.DecodedAccountBytes - dst.PlaceholderObjects = src.PlaceholderObjects + dst.RequestedKeys += src.RequestedKeys + dst.DurableKeys += src.DurableKeys + dst.WorkingSetHits += src.WorkingSetHits + dst.InProgressHits += src.InProgressHits + dst.PendingFoldHits += src.PendingFoldHits + dst.CacheHits += src.CacheHits + dst.IndexHits += src.IndexHits + dst.IndexMisses += src.IndexMisses + dst.UniqueAppendVecs += src.UniqueAppendVecs + dst.AppendVecChunks += src.AppendVecChunks + dst.AppendVecAccounts += src.AppendVecAccounts + dst.OpenFailures += src.OpenFailures + dst.ReadFailures += src.ReadFailures + dst.RetryAccounts += src.RetryAccounts + dst.CommonCacheAdmissions += src.CommonCacheAdmissions + dst.CommonCacheAdmissionsSkipped += src.CommonCacheAdmissionsSkipped + dst.VoteCacheAdmissions += src.VoteCacheAdmissions + dst.VoteCacheAdmissionsSkipped += src.VoteCacheAdmissionsSkipped + dst.CachePublicationEpochRejects += src.CachePublicationEpochRejects + dst.DecodedAccountObjects += src.DecodedAccountObjects + dst.DecodedAccountBytes += src.DecodedAccountBytes + dst.PlaceholderObjects += src.PlaceholderObjects dst.WorkingSetLookup.AddTiming(time.Duration(src.WorkingSetLookupNanoseconds)) dst.InProgressLookup.AddTiming(time.Duration(src.InProgressNanoseconds)) dst.AppendVecPinWait.AddTiming(time.Duration(src.AppendVecPinWaitNanoseconds)) diff --git a/pkg/replay/block_execution.go b/pkg/replay/block_execution.go index 854590479..0a5ebfaa3 100644 --- a/pkg/replay/block_execution.go +++ b/pkg/replay/block_execution.go @@ -75,6 +75,9 @@ type blockExecution struct { txFeeAccumulator fees.TxFeeInfoAccumulator totalCU uint64 + + // Retained across global metric resets while a speculative stream waits. + accountLoader metrics.AccountLoader } // newBlockExecution installs the per-bank trace task, the stage watchdog and @@ -212,6 +215,7 @@ func (exec *blockExecution) installSlotCtx(accts accounts.Accounts, parentAccts // without the whole-block planner, so a streaming caller can start executing // groups before any transaction of the block is known. func (exec *blockExecution) open() error { + defer exec.captureAccountLoader()() block := exec.block exec.setReplayStage("prepare_dependency_planner") if SerializedParameterArena != nil { @@ -366,6 +370,19 @@ func (exec *blockExecution) executeTransactionGroup(txs []*solana.Transaction, i return nil } +// captureAccountLoader isolates this stream's loader work from the global +// record, which may belong to another replayed slot or be reset while waiting. +// Only the replay goroutine may enter this scope; loader workers are joined +// before it exits. Discarded streams never publish their retained totals. +func (exec *blockExecution) captureAccountLoader() func() { + previous := metrics.GlobalBlockReplay.AccountLoader + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} + return func() { + exec.accountLoader.Accumulate(metrics.GlobalBlockReplay.AccountLoader) + metrics.GlobalBlockReplay.AccountLoader = previous + } +} + // loadTransactionAccounts resolves the group's address-table lookups and adds // the pristine parent image of every account the group can touch to the // parent snapshot, exactly as the whole-block loader does for a block, except @@ -374,6 +391,7 @@ func (exec *blockExecution) executeTransactionGroup(txs []*solana.Transaction, i // overlay, so the first image is the right one and later groups must not // replace it. func (exec *blockExecution) loadTransactionAccounts(view *b.Block) error { + defer exec.captureAccountLoader()() phaseStart := time.Now() if err := resolveAddrTableLookups(exec.blockSrc, view); err != nil { return fmt.Errorf("resolve address table lookups at slot %d: %w", view.Slot, err) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index cf2eb776e..ca0850399 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -941,6 +941,9 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel record.OpenDelay.AddTiming(cur.openedAt.Sub(cur.headerAt)) cur.recordTimeline(record, block, finalizeStart) cur.recordGroups(record, fullAt) + // Publish only the accepted stream, including its open, early groups and + // suffix. The finalization record already contains any tail loader work. + metrics.GlobalBlockReplay.AccountLoader.Accumulate(exec.accountLoader) return slotCtx, true, nil } diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index c9fad4163..36087f334 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -369,7 +369,15 @@ func TestStreamingRealFeedExecutesPrefixBeforeCompletionAndMatchesWholeBlock(t * require.Same(t, rig.block.Transactions[i], tx, "the executed prefix is the block, by identity") } + retainedLoader := rig.exec.current.exec.accountLoader + require.Positive(t, retainedLoader.SourceBatch.Count) + // Replay resets the global collector while waiting for the full block. + // Early loader work must survive and be published exactly once. + metrics.GlobalBlockReplay.AccountLoader = metrics.AccountLoader{} rig.finalizeAndCompare(reference) + require.Equal(t, retainedLoader.SourceBatch, metrics.GlobalBlockReplay.AccountLoader.SourceBatch) + require.Equal(t, retainedLoader.RequestedKeys, metrics.GlobalBlockReplay.AccountLoader.RequestedKeys) + require.Equal(t, retainedLoader.ParentAccounts, metrics.GlobalBlockReplay.AccountLoader.ParentAccounts) require.Positive(t, metrics.GlobalBlockReplay.StreamingExecution.TxLoopBeforeFull.Count, "transaction work finished before the slot was full") require.Zero(t, rig.receiver.StreamDroppedEvents()) } From 61aaeaa14a1e49bc2644ff06248ba1d699ee1318 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 18:21:25 -0500 Subject: [PATCH 058/111] turbine: trace completion-critical shred sources and repair sends --- docs/transaction_sigverify_streaming.md | 33 ++++++++ pkg/turbine/assembler.go | 13 +++- pkg/turbine/entry_pipeline_trace.go | 95 ++++++++++++++++++++++-- pkg/turbine/entry_pipeline_trace_test.go | 63 ++++++++++++++++ pkg/turbine/repair.go | 6 ++ 5 files changed, 200 insertions(+), 10 deletions(-) diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index 42e4bf311..cab70c767 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -157,3 +157,36 @@ For four 4,096-transaction components, medians of the three per-run statistics w Completion-finished p99 ranged 46.43–54.52 ms before and 1.477–1.719 ms after. Admission p99 ranged 44.27–53.76 ms before and 0.003206–0.06401 ms after. Every iteration reached its intended request occupancy. For 256-transaction components, completion-finished p99 medians were 3.215→1.165 ms. Shared-host scheduling introduces variation; reserving admission does not remove queued-job or CPU delays, and these 100-sample tails are not a live p99/FAST claim. Total-work throughput was roughly unchanged; no total-work tail improvement is claimed. Original run artifacts are retained in the [evidence archive](streaming-preparation-evidence.md). + +### Completion-critical shred diagnostics + +The optional `MITHRIL_ENTRY_TRACE_MOD`, `MITHRIL_ENTRY_TRACE_SECONDS` (at most +1,800), and `MITHRIL_ENTRY_TRACE_FILE` settings are read at process startup. +Use a new output file for each capture. Tracing is disabled by default. + +Large-block batch records include `critical_shred_index` and its admission +source: `non_repair`, `repair`, or `fec_recovery`. The critical index maximizes +local availability time across the batch **and its preceding DATA_COMPLETE +boundary**. Equal timestamps select the lowest index and report the tie count; +unknown coverage still sets `availability_known=false`. This identifies the +last locally available dependency, not necessarily the replay cursor's current +blocking range. Correlate with execution groups before calling it a replay stall. + +Recovered shreds identify the triggering packet's index, FEC set, coding/data +type and repair status. `non_repair` can include spool hydration. Admission entry +timestamps precede the assembler lock; they are not socket/NIC timestamps. A zero +admission-entry timestamp means it was not sampled. Admission-to-availability +includes local processing and, for recovered data, reconstruction. + +The same JSONL file also contains `event="repair_send"` records. Consumers must +separate these from block reports. Join by `origin_unix_ns`, slot and shred index, +then order by send timestamps; attempt IDs can reset. Start/end bracket the UDP +write, and `success` means only that the local write succeeded. Highest-index +probes are explicitly marked and must not be treated as exact-index requests. +Request records can exist for slots without a large-block report. No response +peer or exact request/response nonce correlation is recorded. + +Both queues are bounded and producers never wait for the writer. The cumulative +`dropped_reports` counter covers queue drops and encoding failures; an absent +repair record is not proof of no request if records were dropped. Tracing does +not change repair scheduling, retry intervals, fanout or verification checks. diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 4ccec5f6e..dc466b03a 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -268,6 +268,10 @@ func (a *SlotAssembler) addShredFrom(shred *Shred, fromRepair bool) (*slotComple // of this immutable shred. Unauthenticated callers and spool hydration use nil // and retain the normal root-computation fallback. func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root *solana.Hash) (*slotCompletionWork, error) { + var admissionEntered int64 + if shred != nil && entryTraceSelected(shred.Slot) { + admissionEntered = entryTraceNow() + } if shred == nil { return nil, nil } @@ -316,7 +320,11 @@ func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root state.noteError(err) return nil, err } - state.traceAcceptedShred(shred) + source := entryShredSource{Path: "non_repair", FEC: shred.FECSetIndex, TriggerIndex: shred.Index, TriggerCoding: shred.Type == ShredTypeCode, TriggerRepair: fromRepair, AdmissionEntered: admissionEntered} + if fromRepair { + source.Path = "repair" + } + state.traceAcceptedShred(shred, source) a.notePrefetchShredLocked(state, shred) if state.firstShredAt.IsZero() { state.firstShredAt = time.Now() @@ -339,7 +347,8 @@ func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root return nil, err } if err == nil { - state.traceAcceptedShred(recoveredShred) + source.Path = "fec_recovery" + state.traceAcceptedShred(recoveredShred, source) a.notePrefetchShredLocked(state, recoveredShred) a.recoveredDataShreds++ } diff --git a/pkg/turbine/entry_pipeline_trace.go b/pkg/turbine/entry_pipeline_trace.go index c2c83b454..b482d0f4f 100644 --- a/pkg/turbine/entry_pipeline_trace.go +++ b/pkg/turbine/entry_pipeline_trace.go @@ -23,6 +23,7 @@ type entryTraceSettings struct { modulo uint64 until time.Time reports chan entryPipelineReport + repairs chan entryRepairTrace } func configureEntryTrace() entryTraceSettings { @@ -44,18 +45,27 @@ func configureEntryTrace() entryTraceSettings { return entryTraceSettings{} } ch := make(chan entryPipelineReport, 8) + repairs := make(chan entryRepairTrace, 256) go func() { defer f.Close() enc := json.NewEncoder(f) - for r := range ch { - r.Dropped = entryTraceDropped.Load() - r.finish() - if err := enc.Encode(r); err != nil { - entryTraceDropped.Add(1) + for { + select { + case r := <-repairs: + r.Dropped = entryTraceDropped.Load() + if err := enc.Encode(r); err != nil { + entryTraceDropped.Add(1) + } + case r := <-ch: + r.Dropped = entryTraceDropped.Load() + r.finish() + if err := enc.Encode(r); err != nil { + entryTraceDropped.Add(1) + } } } }() - return entryTraceSettings{n, time.Now().Add(time.Duration(seconds) * time.Second), ch} + return entryTraceSettings{modulo: n, until: time.Now().Add(time.Duration(seconds) * time.Second), reports: ch, repairs: repairs} } func entryTraceNow() int64 { return time.Since(entryTraceOrigin).Nanoseconds() } @@ -75,13 +85,14 @@ func entryTraceContext(ctx context.Context) bool { type entryPipelineTrace struct { arrivals map[uint32]int64 + sources map[uint32]entryShredSource discovered map[uint32]int64 sealed bool // frozen at first completion claim, including retry/error paths } // Called only after successful admission, under the assembler lock. Includes // FEC-reconstructed data. This is local availability, not a NIC timestamp. -func (s *slotState) traceAcceptedShred(sh *Shred) { +func (s *slotState) traceAcceptedShred(sh *Shred, source ...entryShredSource) { if sh.Type != ShredTypeData { return } @@ -93,7 +104,16 @@ func (s *slotState) traceAcceptedShred(sh *Shred) { s.pipelineTrace = &entryPipelineTrace{arrivals: make(map[uint32]int64), discovered: make(map[uint32]int64)} } if !s.pipelineTrace.sealed { + if _, exists := s.pipelineTrace.arrivals[sh.Index]; exists { + return + } s.pipelineTrace.arrivals[sh.Index] = entryTraceNow() + if len(source) > 0 { + if s.pipelineTrace.sources == nil { + s.pipelineTrace.sources = make(map[uint32]entryShredSource) + } + s.pipelineTrace.sources[sh.Index] = source[0] + } } } @@ -125,6 +145,9 @@ func (t *entryVerificationTrace) observe(j *transactionVerifyJob) { } type entryBatchTraceReport struct { + CriticalIndex *uint32 `json:"critical_shred_index,omitempty"` + CriticalSource *entryShredSource `json:"critical_shred_source,omitempty"` + CriticalTies int `json:"critical_timestamp_ties"` Start uint32 `json:"start"` End uint32 `json:"end"` Transactions int `json:"transactions"` @@ -190,7 +213,18 @@ func (r *entryPipelineReport) finish() { if !ok { row.AvailabilityKnown = false } - row.Available = max(row.Available, at) + if ok && (row.CriticalIndex == nil || at > row.Available) { + index := i + row.CriticalIndex = &index + row.CriticalTies = 1 + row.CriticalSource = nil + if source, exists := r.source.sources[i]; exists { + row.CriticalSource = &source + } + row.Available = at + } else if ok && at == row.Available { + row.CriticalTies++ + } } r.Batches = append(r.Batches, row) } @@ -210,3 +244,48 @@ func queueEntryPipelineReport(s *slotState, b *block.Block, d *entryDecodeTiming entryTraceDropped.Add(1) } } + +// Attribution starts at assembler entry, not socket receipt. A recovered data +// "non_repair" includes direct/spooled admission, not proof of socket origin. +// A recovered shred records the packet that triggered reconstruction; it is not itself a +// received repair response. Missing source fields in older reports mean unknown. +type entryShredSource struct { + Path string `json:"path"` + FEC uint32 `json:"fec_set"` + TriggerIndex uint32 `json:"trigger_index"` + TriggerCoding bool `json:"trigger_coding"` + TriggerRepair bool `json:"trigger_repair"` + AdmissionEntered int64 `json:"admission_entered_ns"` +} + +// Separate JSONL records join by origin/slot/index. The time pair brackets the +// UDP write syscall, NOT delivery. Attempt IDs can reset; order by timestamps. +// Highest-index probes are not exact requests for the returned shred index. +type entryRepairTrace struct { + Event string `json:"event"` + Origin int64 `json:"origin_unix_ns"` + Slot uint64 `json:"slot"` + Index uint32 `json:"index"` + Highest bool `json:"highest_index_probe"` + Attempt uint8 `json:"attempt"` + SendStart int64 `json:"send_start_ns"` + SendEnd int64 `json:"send_end_ns"` + Success bool `json:"success"` + Dropped uint64 `json:"dropped_reports"` +} + +func entryTraceSelected(slot uint64) bool { + c := entryTraceConfig + return c.modulo != 0 && slot%c.modulo == 0 && time.Now().Before(c.until) +} +func traceRepairSend(slot uint64, index uint32, kind repairRequestKind, attempt uint8, start int64, success bool) { + if start == 0 || entryTraceConfig.repairs == nil { + return + } + r := entryRepairTrace{Event: "repair_send", Origin: entryTraceOrigin.UnixNano(), Slot: slot, Index: index, Highest: kind == repairRequestHighestWindowIndex, Attempt: attempt, SendStart: start, SendEnd: entryTraceNow(), Success: success} + select { + case entryTraceConfig.repairs <- r: + default: + entryTraceDropped.Add(1) + } +} diff --git a/pkg/turbine/entry_pipeline_trace_test.go b/pkg/turbine/entry_pipeline_trace_test.go index 6aa493601..07fbdc102 100644 --- a/pkg/turbine/entry_pipeline_trace_test.go +++ b/pkg/turbine/entry_pipeline_trace_test.go @@ -3,6 +3,7 @@ package turbine import ( "context" "testing" + "time" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" @@ -66,3 +67,65 @@ func TestEntryPipelineTraceSealedGeneration(t *testing.T) { s.traceAcceptedShred(&Shred{Type: ShredTypeData, Index: 2}) require.Len(t, s.pipelineTrace.arrivals, 1, "completion/retry cannot mutate a report's frozen arrival map") } + +func TestEntryCriticalShredIncludesBoundaryAndSource(t *testing.T) { + source := &entryPipelineTrace{arrivals: map[uint32]int64{2: 100, 3: 20, 4: 30}, sources: map[uint32]entryShredSource{2: {Path: "fec_recovery", FEC: 0, TriggerIndex: 12, TriggerCoding: true, TriggerRepair: true, AdmissionEntered: 80}}} + r := entryPipelineReport{source: source, all: []*prefetchedShredBatch{{start: 3, end: 4}}} + r.finish() + require.Equal(t, uint32(2), *r.Batches[0].CriticalIndex) + require.Equal(t, "fec_recovery", r.Batches[0].CriticalSource.Path) + require.True(t, r.Batches[0].CriticalSource.TriggerRepair) + require.Equal(t, 1, r.Batches[0].CriticalTies) + source.arrivals[3] = 100 + r.Batches = nil + r.finish() + require.Equal(t, 2, r.Batches[0].CriticalTies) + require.Equal(t, uint32(2), *r.Batches[0].CriticalIndex, "ties select the lowest index deterministically") +} + +func TestEntryTracePreservesFirstAdmission(t *testing.T) { + s := &slotState{pipelineTrace: &entryPipelineTrace{arrivals: make(map[uint32]int64)}} + sh := &Shred{Type: ShredTypeData, Index: 1} + s.traceAcceptedShred(sh, entryShredSource{Path: "repair"}) + first := s.pipelineTrace.arrivals[1] + s.traceAcceptedShred(sh, entryShredSource{Path: "non_repair"}) + require.Equal(t, first, s.pipelineTrace.arrivals[1]) + require.Equal(t, "repair", s.pipelineTrace.sources[1].Path) + s.pipelineTrace.sealed = true + s.traceAcceptedShred(&Shred{Type: ShredTypeData, Index: 2}, entryShredSource{Path: "repair"}) + require.Len(t, s.pipelineTrace.sources, 1) +} + +func TestEntryRepairTraceBoundedAndExplicit(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{repairs: make(chan entryRepairTrace, 1)} + traceRepairSend(15, 10, repairRequestWindowIndex, 2, 100, true) + r := <-entryTraceConfig.repairs + require.Equal(t, "repair_send", r.Event) + require.Equal(t, uint64(15), r.Slot) + require.False(t, r.Highest) + require.True(t, r.Success) + require.Equal(t, uint8(2), r.Attempt) + traceRepairSend(15, 10, repairRequestHighestWindowIndex, 0, 100, false) + dropped := entryTraceDropped.Load() + traceRepairSend(15, 11, repairRequestWindowIndex, 0, 100, true) + require.Equal(t, dropped+1, entryTraceDropped.Load(), "full diagnostic queue never blocks repair") + r = <-entryTraceConfig.repairs + require.True(t, r.Highest) + require.False(t, r.Success) + traceRepairSend(15, 1, repairRequestWindowIndex, 0, 0, true) + require.Empty(t, entryTraceConfig.repairs, "unsampled sends are ignored") +} + +func TestEntryTraceSelectionIsBounded(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{modulo: 5, until: time.Now().Add(time.Minute)} + require.True(t, entryTraceSelected(15)) + require.False(t, entryTraceSelected(16)) + entryTraceConfig.until = time.Now().Add(-time.Second) + require.False(t, entryTraceSelected(15)) + entryTraceConfig = entryTraceSettings{} + require.False(t, entryTraceSelected(15)) +} diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index 980f0b621..3512959c3 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -983,7 +983,12 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair c.byResponse[responseKey] = key c.mu.Unlock() + var traceStart int64 + if entryTraceSelected(slot) { + traceStart = entryTraceNow() + } if _, err := conn.WriteToUDP(packet, peer.Addr); err != nil { + traceRepairSend(slot, index, kind, attempt, traceStart, false) c.mu.Lock() delete(c.outstanding, key) delete(c.byResponse, responseKey) @@ -994,6 +999,7 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair return false } + traceRepairSend(slot, index, kind, attempt, traceStart, true) c.requests.Add(1) return true } From 320ce8da16ead061a377086fbce01381200985a6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 19:39:40 -0500 Subject: [PATCH 059/111] turbine: prioritize the earliest repair gap for streaming heads --- docs/transaction_sigverify_streaming.md | 22 ++++ pkg/turbine/assembler.go | 43 +++++++- pkg/turbine/repair_selection_test.go | 59 +++++++++++ pkg/turbine/repairsim/prefix_repair_test.go | 107 ++++++++++++++++++++ 4 files changed, 229 insertions(+), 2 deletions(-) create mode 100644 pkg/turbine/repairsim/prefix_repair_test.go diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index cab70c767..49846bc7c 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -190,3 +190,25 @@ Both queues are bounded and producers never wait for the writer. The cumulative `dropped_reports` counter covers queue drops and encoding failures; an absent repair record is not proof of no request if records were dropped. Tracing does not change repair scheduling, retry intervals, fanout or verification checks. + +### Repair ordering for streaming + +When a streaming subscriber is installed, the first priority repair slot puts +its earliest missing data span ahead of the usual cheapest-FEC-unlock ordering. +Known FEC sets still request only their recovery deficit; an unknown-layout hole +prioritizes one missing index without guessing its FEC shape. Remaining work, +other priority slots and freshness repair retain their previous ordering. +Request budgets, admission shares, retry intervals and fanout are unchanged. + +This trades completing cheap later sets first for making the contiguous input +prefix available sooner. It helps when request capacity is constrained; it does +not accelerate requests already in flight, guarantee an earlier full block, or +prove improved voting latency. The subscriber and priority head are used as the +scope; the selector does not read the execution cursor. + +`go test ./pkg/turbine/repairsim -run TestStreamingPrefixRepairUnderLimitedBudget -v` +compares both policies using authenticated generated shreds and production FEC +recovery, at fixed request budgets and a 20ms simulated round trip. It checks +identical assembled entries/transactions, request counts and full completion, +and measures availability of the first data span. It does not model production +retry timers, peer loss, execution timing, or reproduce a captured live slot. diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index dc466b03a..6c9649fb3 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -929,6 +929,10 @@ func (s *slotState) addDataShred(shred *Shred) error { } func (s *slotState) repairRequest(maxMissing int) (SlotRepairRequest, bool) { + return s.repairRequestWithPrefix(maxMissing, false) +} + +func (s *slotState) repairRequestWithPrefix(maxMissing int, prefix bool) (SlotRepairRequest, bool) { req := SlotRepairRequest{Slot: s.slot} var maxObserved uint32 @@ -945,7 +949,7 @@ func (s *slotState) repairRequest(maxMissing int) (SlotRepairRequest, bool) { req.HighestDataShredIndex = maxObserved + 1 } - req.MissingDataShreds = s.missingDataForRepair(maxObserved, maxMissing) + req.MissingDataShreds = s.missingDataForRepairWithPrefix(maxObserved, maxMissing, prefix) if len(req.MissingDataShreds) == 0 && !req.NeedHighestDataShred { return SlotRepairRequest{}, false @@ -994,6 +998,10 @@ func (span codedSpan) requestsToUnlock() int { // promises more — without that, the tail waits on a HighestWindowIndex // round trip to be discovered. func (s *slotState) missingDataForRepair(maxObserved uint32, maxMissing int) []uint32 { + return s.missingDataForRepairWithPrefix(maxObserved, maxMissing, false) +} + +func (s *slotState) missingDataForRepairWithPrefix(maxObserved uint32, maxMissing int, prefix bool) []uint32 { spans := make([]codedSpan, 0, len(s.fecSets)) for _, fec := range s.fecSets { if !fec.haveLayout || fec.layout.dataShreds == 0 { @@ -1053,7 +1061,38 @@ func (s *slotState) missingDataForRepair(maxObserved uint32, maxMissing int) []u return spans[order[a]].start < spans[order[b]].start }) + // For a streaming head, one earliest hole gates every later batch. Move + // just that span ahead of cheapest-unlock order; retain deficit capping + // and every existing request/admission limit. Without a known layout, + // prioritize one earliest missing data index rather than guessing a span. + var first []uint32 + if prefix { + earliest := -1 + for _, i := range order { + if earliest < 0 || spans[i].missing[0] < spans[earliest].missing[0] { + earliest = i + } + } + if len(uncovered) > 0 && (earliest < 0 || uncovered[0] < spans[earliest].missing[0]) { + first = uncovered[:1] + uncovered = uncovered[1:] + } else if earliest >= 0 { + first = spans[earliest].missing[:spans[earliest].requestsToUnlock()] + for j, i := range order { + if i == earliest { + order = append(order[:j], order[j+1:]...) + break + } + } + } + } missing := make([]uint32, 0, min(maxMissing, 64)) + for _, index := range first { + if len(missing) >= maxMissing { + return missing + } + missing = append(missing, index) + } for _, i := range order { span := spans[i] for _, index := range span.missing[:span.requestsToUnlock()] { @@ -1342,7 +1381,7 @@ func (a *SlotAssembler) RepairRequestsTiered(maxSlots int, maxMissingPerSlot int HighestDataShredIndex: 0, }) } - if req, ok := state.repairRequest(maxMissing); ok { + if req, ok := state.repairRequestWithPrefix(maxMissing, priorityPin && len(dst) == 0 && a.streamSubscriber != nil); ok { seen[slot] = struct{}{} return append(dst, req) } diff --git a/pkg/turbine/repair_selection_test.go b/pkg/turbine/repair_selection_test.go index 76ebf157f..033ad81a7 100644 --- a/pkg/turbine/repair_selection_test.go +++ b/pkg/turbine/repair_selection_test.go @@ -251,3 +251,62 @@ func TestHeadPolicy(t *testing.T) { t.Fatalf("bulk policy = %+v, want no concurrent duplicate attempts", bulk) } } + +func TestRepairSelectionPrefixBeforeCheapest(t *testing.T) { + s := newRepairSelectionSlot(9) + addCodedSet(s, 0, 32, 32, seq(0, 9), 12) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + got := s.missingDataForRepairWithPrefix(63, 4, true) + if !reflect.DeepEqual(got, seq(10, 13)) { + t.Fatalf("prefix %v", got) + } + all := s.missingDataForRepairWithPrefix(63, 256, true) + if !reflect.DeepEqual(all, append(seq(10, 19), 57)) { + t.Fatalf("remaining order %v", all) + } +} + +func TestRepairSelectionUncodedPrefixBeforeCoded(t *testing.T) { + s := newRepairSelectionSlot(9) + addUncodedData(s, seq(1, 31)...) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + got := s.missingDataForRepairWithPrefix(63, 256, true) + if !reflect.DeepEqual(got, []uint32{0, 57}) { + t.Fatalf("uncoded prefix %v", got) + } + if got := s.missingDataForRepairWithPrefix(63, 0, true); len(got) != 0 { + t.Fatalf("zero budget: %v", got) + } +} + +func TestRepairSelectionPrefixOnlyForStreamingPriorityHead(t *testing.T) { + a := NewSlotAssembler() + for _, slot := range []uint64{9, 10} { + s := newRepairSelectionSlot(slot) + addCodedSet(s, 0, 32, 32, seq(0, 9), 12) + addCodedSet(s, 32, 32, 32, seq(32, 56), 6) + s.haveLast = true + s.lastIndex = 63 + a.slots[slot] = s + } + a.maxObservedSlot = 11 + a.PrioritizeRepairRange(9, 10) + p, _ := a.RepairRequestsTiered(2, 256) + if len(p) != 2 || p[0].MissingDataShreds[0] != 57 { + t.Fatalf("nonstreaming order %v", p) + } + a.SubscribeStream(make(chan StreamEvent, 1)) + p, _ = a.RepairRequestsTiered(2, 256) + if len(p) != 2 || p[0].MissingDataShreds[0] != 10 || p[1].MissingDataShreds[0] != 57 { + t.Fatalf("streaming head scope %v", p) + } + a.SubscribeStream(nil) + p, _ = a.RepairRequestsTiered(2, 256) + if p[0].MissingDataShreds[0] != 57 { + t.Fatal("disabled streaming retained prefix policy") + } +} diff --git a/pkg/turbine/repairsim/prefix_repair_test.go b/pkg/turbine/repairsim/prefix_repair_test.go new file mode 100644 index 000000000..9077b0649 --- /dev/null +++ b/pkg/turbine/repairsim/prefix_repair_test.go @@ -0,0 +1,107 @@ +package repairsim + +import ( + "fmt" + "reflect" + "testing" + "time" + + "github.com/Overclock-Validator/mithril/pkg/block" + "github.com/Overclock-Validator/mithril/pkg/turbine" +) + +// Logical-time experiment using production selection, authenticated packets and +// FEC recovery. A fixed request budget per 20ms round; this measures when the +// first data span is restored, not execution latency or production retry policy. +func TestStreamingPrefixRepairUnderLimitedBudget(t *testing.T) { + ledger := testLedger(t, 1, 8) + for _, budget := range []int{1, 2, 4, 16} { + t.Run(fmt.Sprint(budget), func(t *testing.T) { + type result struct { + prefix, complete time.Duration + requests int + block *block.Block + } + run := func(stream bool) result { + a := turbine.NewSlotAssembler() + if stream { + a.SubscribeStream(make(chan turbine.StreamEvent, 256)) + } + slot := ledger.Slots[0] + feed := func(p Packet) *block.Block { + sh, err := parseAndVerify(p, ledger) + if err != nil { + t.Fatal(err) + } + b, err := a.AddShred(sh) + if err != nil { + t.Fatal(err) + } + return b + } + for j, f := range slot.FECs { + for _, p := range f.Data[:29] { + feed(p) + } + n := 2 + if j == 0 { + n = 1 + } + for _, p := range f.Coding[:n] { + feed(p) + } + } + a.PrioritizeRepairSlot(slot.Number) + r := result{} + for round := 1; round <= 20; round++ { + req := a.RepairRequests(1, 256) + if len(req) == 0 { + t.Fatal("unfinished slot has no repair request") + } + indices := req[0].MissingDataShreds + if len(indices) > budget { + indices = indices[:budget] + } + if len(indices) == 0 { + t.Fatal("no exact repair work") + } + for _, idx := range indices { + r.requests++ + if b := feed(slot.Data[idx]); b != nil { + r.block = b + r.complete = time.Duration(round) * 20 * time.Millisecond + } + } + remaining := a.RepairRequests(1, 256) + hole := false + for _, q := range remaining { + for _, idx := range q.MissingDataShreds { + if idx < 32 { + hole = true + } + } + } + if !hole && r.prefix == 0 { + r.prefix = time.Duration(round) * 20 * time.Millisecond + } + if r.block != nil { + return r + } + } + t.Fatal("did not complete") + return r + } + before, after := run(false), run(true) + t.Logf("first span %v -> %v; full block %v -> %v; requests %d -> %d", before.prefix, after.prefix, before.complete, after.complete, before.requests, after.requests) + if after.prefix > before.prefix || (budget < 9 && after.prefix == before.prefix) { + t.Fatal("prefix did not improve") + } + if after.requests != before.requests || after.complete > before.complete { + t.Fatal("request count or completion regressed") + } + if !reflect.DeepEqual(before.block.Transactions, after.block.Transactions) || !reflect.DeepEqual(before.block.Entries, after.block.Entries) { + t.Fatal("assembled payload mismatch") + } + }) + } +} From 1f755c84e80a98fd0bb0c39dd122943b2d999a3b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:06:26 -0500 Subject: [PATCH 060/111] turbine: trace repair response matching and FEC progress --- docs/transaction_sigverify_streaming.md | 24 ++++++++- pkg/turbine/assembler.go | 17 ++++++ pkg/turbine/entry_pipeline_trace.go | 66 +++++++++++++++++++----- pkg/turbine/entry_pipeline_trace_test.go | 46 +++++++++++++++++ pkg/turbine/repair.go | 9 +++- 5 files changed, 146 insertions(+), 16 deletions(-) diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index 49846bc7c..4ffe0e57a 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -183,8 +183,7 @@ separate these from block reports. Join by `origin_unix_ns`, slot and shred inde then order by send timestamps; attempt IDs can reset. Start/end bracket the UDP write, and `success` means only that the local write succeeded. Highest-index probes are explicitly marked and must not be treated as exact-index requests. -Request records can exist for slots without a large-block report. No response -peer or exact request/response nonce correlation is recorded. +Request records can exist for slots without a large-block report. Send records include the peer endpoint and nonce for response correlation. Both queues are bounded and producers never wait for the writer. The cumulative `dropped_reports` counter covers queue drops and encoding failures; an absent @@ -212,3 +211,24 @@ recovery, at fixed request budgets and a 20ms simulated round trip. It checks identical assembled entries/transactions, request counts and full completion, and measures availability of the first data span. It does not model production retry timers, peer loss, execution timing, or reproduce a captured live slot. + + +Response-effectiveness tracing also emits `repair_response` and +`repair_admission` events. Treat every record with an `event` field as an event, +not a block report. Match sends/responses by origin, peer, nonce and requested +slot/index; use timestamps to disambiguate nonce reuse. Responses record the +returned index, request-registration timestamp and whether the request had +expired. Registration precedes signing/write; use `send_start_ns` for the closer +approximation to network elapsed time. A matched response is not proof that +assembly accepted it: later receive-path checks can still reject it. + +Admission events cover sampled matched-repair shreds reaching an active assembly; +join to responses by slot/returned index and chronology (no nonce is carried into +the assembler). They report accepted/duplicate/rejected, the coding-layout +recovery deficit before/after, and the number of reconstructed data shreds. +Deficit `-1` means unknown layout; zero means enough shards, not necessarily +successful recovery. Admission timestamps are local assembler entry/exit, +including lock wait and processing, not NIC timestamps. Already-completed, +evicted, or completing slots return before this instrumentation. An unmatched +or canceled response is not emitted as a matched response. Missing records, +particularly with drops or capture boundaries, cannot establish packet loss. diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 6c9649fb3..3d9600f9d 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -295,6 +295,16 @@ func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root a.ignoredOldShreds++ return nil, nil } + // Diagnostic only; the deferred observation runs while a.mu is still held. + var repairTrace *entryRepairTrace + if fromRepair && admissionEntered != 0 { + repairTrace = &entryRepairTrace{Event: "repair_admission", Origin: entryTraceOrigin.UnixNano(), Slot: shred.Slot, Index: shred.Index, FEC: shred.FECSetIndex, AdmissionStart: admissionEntered, DeficitBefore: traceFECDeficit(state, shred.FECSetIndex), Outcome: "rejected"} + defer func() { + repairTrace.ResponseAt = entryTraceNow() + repairTrace.DeficitAfter = traceFECDeficit(state, shred.FECSetIndex) + emitRepairTrace(*repairTrace) + }() + } var err error switch shred.Type { case ShredTypeData: @@ -315,6 +325,9 @@ func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root } if err != nil { if errors.Is(err, ErrDuplicateShred) { + if repairTrace != nil { + repairTrace.Outcome = "duplicate" + } return nil, nil } state.noteError(err) @@ -354,6 +367,10 @@ func (a *SlotAssembler) addShredFromWithRoot(shred *Shred, fromRepair bool, root } } + if repairTrace != nil { + repairTrace.Outcome = "accepted" + repairTrace.Recovered = len(recovered) + } a.prefetchEntriesLocked(state) if !state.complete() { return nil, nil diff --git a/pkg/turbine/entry_pipeline_trace.go b/pkg/turbine/entry_pipeline_trace.go index b482d0f4f..38ea1055e 100644 --- a/pkg/turbine/entry_pipeline_trace.go +++ b/pkg/turbine/entry_pipeline_trace.go @@ -245,7 +245,7 @@ func queueEntryPipelineReport(s *slotState, b *block.Block, d *entryDecodeTiming } } -// Attribution starts at assembler entry, not socket receipt. A recovered data +// Attribution starts at assembler entry, not socket receipt. // "non_repair" includes direct/spooled admission, not proof of socket origin. // A recovered shred records the packet that triggered reconstruction; it is not itself a // received repair response. Missing source fields in older reports mean unknown. @@ -262,30 +262,72 @@ type entryShredSource struct { // UDP write syscall, NOT delivery. Attempt IDs can reset; order by timestamps. // Highest-index probes are not exact requests for the returned shred index. type entryRepairTrace struct { - Event string `json:"event"` - Origin int64 `json:"origin_unix_ns"` - Slot uint64 `json:"slot"` - Index uint32 `json:"index"` - Highest bool `json:"highest_index_probe"` - Attempt uint8 `json:"attempt"` - SendStart int64 `json:"send_start_ns"` - SendEnd int64 `json:"send_end_ns"` - Success bool `json:"success"` - Dropped uint64 `json:"dropped_reports"` + AdmissionStart int64 `json:"admission_start_ns,omitempty"` + Peer string `json:"peer,omitempty"` + Nonce uint32 `json:"nonce"` + ResponseAt int64 `json:"response_ns,omitempty"` + RequestedAt int64 `json:"requested_ns,omitempty"` + ReturnedIndex uint32 `json:"returned_index"` + Late bool `json:"late"` + FEC uint32 `json:"fec_set"` + DeficitBefore int `json:"deficit_before"` + DeficitAfter int `json:"deficit_after"` + Recovered int `json:"recovered"` + Outcome string `json:"outcome,omitempty"` + Event string `json:"event"` + Origin int64 `json:"origin_unix_ns"` + Slot uint64 `json:"slot"` + Index uint32 `json:"index"` + Highest bool `json:"highest_index_probe"` + Attempt uint8 `json:"attempt"` + SendStart int64 `json:"send_start_ns"` + SendEnd int64 `json:"send_end_ns"` + Success bool `json:"success"` + Dropped uint64 `json:"dropped_reports"` } func entryTraceSelected(slot uint64) bool { c := entryTraceConfig return c.modulo != 0 && slot%c.modulo == 0 && time.Now().Before(c.until) } -func traceRepairSend(slot uint64, index uint32, kind repairRequestKind, attempt uint8, start int64, success bool) { +func traceRepairSend(slot uint64, index uint32, kind repairRequestKind, attempt uint8, start int64, success bool, binding ...entryRepairTrace) { if start == 0 || entryTraceConfig.repairs == nil { return } r := entryRepairTrace{Event: "repair_send", Origin: entryTraceOrigin.UnixNano(), Slot: slot, Index: index, Highest: kind == repairRequestHighestWindowIndex, Attempt: attempt, SendStart: start, SendEnd: entryTraceNow(), Success: success} + if len(binding) > 0 { + r.Peer = binding[0].Peer + r.Nonce = binding[0].Nonce + } + emitRepairTrace(r) +} + +func emitRepairTrace(r entryRepairTrace) { + if entryTraceConfig.repairs == nil { + return + } select { case entryTraceConfig.repairs <- r: default: entryTraceDropped.Add(1) } } + +// -1 means no authenticated coding layout is known. Zero means sufficient +// shards, not that reconstruction necessarily succeeded (see outcome/recovered). +func traceFECDeficit(s *slotState, index uint32) int { + if s == nil { + return -1 + } + f := s.fecSets[index] + if f == nil || !f.haveLayout { + return -1 + } + return max(0, int(f.layout.dataShreds)-len(f.data)-len(f.coding)) +} +func traceRepairResponse(rec outstandingRepairRequest, sh *Shred, peer string, late bool) { + if !entryTraceSelected(sh.Slot) { + return + } + emitRepairTrace(entryRepairTrace{Event: "repair_response", Origin: entryTraceOrigin.UnixNano(), Slot: sh.Slot, Index: rec.key.index, Highest: rec.key.kind == repairRequestHighestWindowIndex, Attempt: rec.key.attempt, Nonce: rec.nonce, Peer: peer, RequestedAt: entryTraceTime(rec.sentAt), ResponseAt: entryTraceNow(), ReturnedIndex: sh.Index, FEC: sh.FECSetIndex, Late: late}) +} diff --git a/pkg/turbine/entry_pipeline_trace_test.go b/pkg/turbine/entry_pipeline_trace_test.go index 07fbdc102..a39f8f087 100644 --- a/pkg/turbine/entry_pipeline_trace_test.go +++ b/pkg/turbine/entry_pipeline_trace_test.go @@ -2,6 +2,7 @@ package turbine import ( "context" + "net" "testing" "time" @@ -129,3 +130,48 @@ func TestEntryTraceSelectionIsBounded(t *testing.T) { entryTraceConfig = entryTraceSettings{} require.False(t, entryTraceSelected(15)) } + +func TestEntryRepairResponseCorrelation(t *testing.T) { + old := entryTraceConfig + defer func() { entryTraceConfig = old }() + entryTraceConfig = entryTraceSettings{modulo: 1, until: time.Now().Add(time.Minute), repairs: make(chan entryRepairTrace, 8)} + from := &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1), Port: 8000} + addr, _ := repairAddressKeyFromUDP(from) + for _, late := range []bool{false, true} { + c := newPacingTestClient(t) + key := repairRequestKey{kind: repairRequestWindowIndex, slot: 42, index: 3} + rec := outstandingRepairRequest{key: key, nonce: 777, addr: addr, sentAt: time.Now().Add(-time.Second), accountAt: time.Now().Add(time.Second)} + responseKey := repairResponseKey{addr: addr, nonce: 777} + if late { + c.expiredCur[responseKey] = rec + } else { + c.outstanding[key] = rec + c.byResponse[responseKey] = key + } + require.False(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 4, Type: ShredTypeData})) + require.Empty(t, entryTraceConfig.repairs, "wrong index cannot be reported as matched") + require.True(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + r := <-entryTraceConfig.repairs + require.Equal(t, "repair_response", r.Event) + require.Equal(t, uint32(777), r.Nonce) + require.Equal(t, from.String(), r.Peer) + require.Equal(t, late, r.Late) + require.Equal(t, uint32(3), r.ReturnedIndex) + require.Greater(t, r.ResponseAt, r.RequestedAt) + require.False(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + require.Empty(t, entryTraceConfig.repairs, "consumed nonce cannot count twice") + } +} + +func TestEntryFECDeficit(t *testing.T) { + s := newRepairSelectionSlot(42) + require.Equal(t, -1, traceFECDeficit(s, 0)) + addCodedSet(s, 0, 32, 32, seq(0, 19), 8) + require.Equal(t, 4, traceFECDeficit(s, 0)) + s.fecSets[0].data[20] = &Shred{} + require.Equal(t, 3, traceFECDeficit(s, 0)) + for i := uint32(21); i < 32; i++ { + s.fecSets[0].data[i] = &Shred{} + } + require.Zero(t, traceFECDeficit(s, 0), "enough shards is not a negative deficit") +} diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index 3512959c3..3c5a3e69e 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -691,6 +691,9 @@ func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, fr c.observeLatencyLocked(latency) c.mu.Unlock() + if entryTraceSelected(shred.Slot) { + traceRepairResponse(outstanding, shred, from.String(), late) + } if late { c.lateResponses.Add(1) } else { @@ -984,11 +987,13 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair c.mu.Unlock() var traceStart int64 + var traceBinding entryRepairTrace if entryTraceSelected(slot) { traceStart = entryTraceNow() + traceBinding = entryRepairTrace{Peer: peer.Addr.String(), Nonce: nonce} } if _, err := conn.WriteToUDP(packet, peer.Addr); err != nil { - traceRepairSend(slot, index, kind, attempt, traceStart, false) + traceRepairSend(slot, index, kind, attempt, traceStart, false, traceBinding) c.mu.Lock() delete(c.outstanding, key) delete(c.byResponse, responseKey) @@ -999,7 +1004,7 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair return false } - traceRepairSend(slot, index, kind, attempt, traceStart, true) + traceRepairSend(slot, index, kind, attempt, traceStart, true, traceBinding) c.requests.Add(1) return true } From 72757aa3431f22c5c0d9ad65c1175b31102546f6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:37:13 -0500 Subject: [PATCH 061/111] turbine: repair one child prefix while its parent executes --- docs/transaction_sigverify_streaming.md | 24 ++++ pkg/blockstream/turbine_stream.go | 12 +- pkg/replay/streaming.go | 6 +- pkg/replay/streaming_realfeed_test.go | 2 +- pkg/replay/streaming_test.go | 4 +- pkg/turbine/assembler.go | 16 ++- pkg/turbine/child_repair.go | 110 ++++++++++++++++ pkg/turbine/child_repair_test.go | 160 ++++++++++++++++++++++++ pkg/turbine/receiver.go | 15 +++ pkg/turbine/repair.go | 26 +++- pkg/turbine/stream.go | 6 + 11 files changed, 371 insertions(+), 10 deletions(-) create mode 100644 pkg/turbine/child_repair.go create mode 100644 pkg/turbine/child_repair_test.go diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index 4ffe0e57a..bd5dac9f2 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -232,3 +232,27 @@ including lock wait and processing, not NIC timestamps. Already-completed, evicted, or completing slots return before this instrumentation. An unmatched or canceled response is not emitted as a matched response. Missing records, particularly with drops or capture boundaries, cannot establish packet loss. + +### Bounded child repair lookahead + +Replay supplies its exact streaming generation as a repair anchor. The +asynchronous decoder can then recognize a decoded header for the immediate next +slot naming that parent, even while replay executes a parent group. A header +published earlier is recovered from the assembler's ready batches. The hint +permits fetching only; it does not establish fork choice, validate the final +parent block ID, or permit child execution before parent completion. + +After ordinary priority and freshness repair, leftover tokens may request up to +four missing data shreds from the child's earliest incomplete FEC span. Existing +in-flight requests for that child count against the four-request lookahead +allowance. Normal repair can independently exceed that allowance. Global rate, +per-scan and admission limits remain in force; lookahead uses bulk single-attempt +policy, with no new retry/fanout or highest-index probing. If no capacity remains, +the child waits. + +Only one child is tracked. The anchor expires after two seconds and is cleared +on parent finalize/discard, parent reset/update, or stream unsubscribe. Child +completion/reset and changed parent markers invalidate its hint. Generation +checks reject stale headers. Already-sent requests still use normal response and +expiry handling. Notifications and repair wakeups remain nonblocking/coalesced; +no extra workers or polling loop are introduced. diff --git a/pkg/blockstream/turbine_stream.go b/pkg/blockstream/turbine_stream.go index 01585d6f0..5e53b32a9 100644 --- a/pkg/blockstream/turbine_stream.go +++ b/pkg/blockstream/turbine_stream.go @@ -620,6 +620,14 @@ func (bs *BlockSource) PendingStreamBatches(g turbine.StreamGeneration, fromStar // PrioritizeStreamRepair keeps a slot that replay is executing while its // shreds arrive pinned for repair, since the emitter pins the head only when // it observes a gap. -func (bs *BlockSource) PrioritizeStreamRepair(slot uint64) { - bs.prioritizeTurbineRepairRange(slot, slot) +func (bs *BlockSource) PrioritizeStreamRepair(g turbine.StreamGeneration) { + if bs.sourceType != BlockSourceTurbine || !bs.turbineAlpenglowBlockIDHints { + return + } + bs.alpenglowMu.Lock() + receiver := bs.activeTurbineReceiver + bs.alpenglowMu.Unlock() + if receiver != nil { + receiver.PrioritizeStreamRepair(g) + } } diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index ca0850399..a6280af8c 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -92,7 +92,7 @@ type streamingFeed interface { StreamEvents() <-chan turbine.StreamEvent StreamStatusOf(turbine.StreamGeneration) turbine.StreamStatus PendingStreamBatches(turbine.StreamGeneration, uint32) []*turbine.StreamBatch - PrioritizeStreamRepair(uint64) + PrioritizeStreamRepair(turbine.StreamGeneration) } var _ streamingFeed = (*blockstream.BlockSource)(nil) @@ -598,7 +598,7 @@ func (s *streamingExecutor) openStream(header *turbine.StreamBatch) { } s.current = cur metrics.GlobalBlockReplay.StreamingExecution.Opened = 1 - d.feed.PrioritizeStreamRepair(shell.Slot) + d.feed.PrioritizeStreamRepair(header.Generation) mlog.Log.FileOnlyf("streaming: opened slot %d on parent %d | %s", shell.Slot, header.ParentSlot, cur.openTimeline()) s.offer(header) s.pull() @@ -757,6 +757,7 @@ func (s *streamingExecutor) discard(reason string) { } cur := s.current s.current = nil + s.deps.feed.PrioritizeStreamRepair(turbine.StreamGeneration{}) s.stopTicker() s.retire(cur.slot, cur.generation) if obs := s.observed[cur.slot]; obs != nil && obs.generation == cur.generation { @@ -921,6 +922,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel exec.slotCtx.TrackProgramCacheAdds = false exec.slotCtx.TakeProgramCacheAdds() s.current = nil + s.deps.feed.PrioritizeStreamRepair(turbine.StreamGeneration{}) exec.close() // The per-block record is rebuilt from the stream's own bookkeeping: the diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index 36087f334..5210f5da9 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -50,7 +50,7 @@ func (f *receiverFeed) StreamStatusOf(g turbine.StreamGeneration) turbine.Stream func (f *receiverFeed) PendingStreamBatches(g turbine.StreamGeneration, from uint32) []*turbine.StreamBatch { return f.r.PendingStreamBatches(g, from) } -func (f *receiverFeed) PrioritizeStreamRepair(uint64) {} +func (f *receiverFeed) PrioritizeStreamRepair(turbine.StreamGeneration) {} type realFeedRig struct { slot uint64 diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 6efaf443b..29837fc6e 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -55,8 +55,8 @@ func (f *fakeStreamFeed) PendingStreamBatches(g turbine.StreamGeneration, fromSt return out } -func (f *fakeStreamFeed) PrioritizeStreamRepair(slot uint64) { - f.prioritized = append(f.prioritized, slot) +func (f *fakeStreamFeed) PrioritizeStreamRepair(g turbine.StreamGeneration) { + f.prioritized = append(f.prioritized, g.Slot()) } // fakeUnrootedState satisfies the tail interface for eligibility checks; no diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 3d9600f9d..16d910f29 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -82,8 +82,13 @@ type SlotAssembler struct { entryPrefetch *entryPrefetchPool // Streaming feed subscriber (see stream.go); nil when nothing consumes // batches before completion. - streamSubscriber chan<- StreamEvent - streamDroppedEvents uint64 + streamSubscriber chan<- StreamEvent + streamDroppedEvents uint64 + streamRepairParent *slotState + streamRepairChild *slotState + streamRepairInvalidChild *slotState + streamRepairUntil time.Time + streamRepairWake chan<- struct{} } type SlotRepairRequest struct { @@ -645,6 +650,13 @@ func (a *SlotAssembler) RejectAlpenglowBlockID(slot uint64, blockID solana.Hash) func (a *SlotAssembler) ResetSlot(slot uint64) { a.mu.Lock() defer a.mu.Unlock() + if a.streamRepairParent != nil && a.streamRepairParent.slot == slot { + a.streamRepairParent = nil + a.streamRepairChild = nil + } + if a.streamRepairChild != nil && a.streamRepairChild.slot == slot { + a.streamRepairChild = nil + } a.retentionDirty = true a.recordPartialObsLocked(a.slots[slot]) diff --git a/pkg/turbine/child_repair.go b/pkg/turbine/child_repair.go new file mode 100644 index 000000000..786d3ee2d --- /dev/null +++ b/pkg/turbine/child_repair.go @@ -0,0 +1,110 @@ +package turbine + +import "time" + +const childRepairLimit = 4 +const childRepairLifetime = 2 * time.Second + +// SetStreamRepairParent anchors lookahead to the generation currently executing. +// Zero clears the hint on discard/finalize. This grants fetching, never execution +// or fork choice: the child's parent block ID may not be verifiable until full. +func (a *SlotAssembler) SetStreamRepairParent(g StreamGeneration) { + a.mu.Lock() + defer a.mu.Unlock() + a.streamRepairInvalidChild = nil + a.streamRepairParent = nil + a.streamRepairChild = nil + slot := g.Slot() + if g.IsZero() || slot == 0 || a.streamSubscriber == nil { + return + } + p := g.state + if a.streamStatusLocked(g) == StreamGone { + return + } + a.streamRepairParent = p + a.streamRepairUntil = time.Now().Add(childRepairLifetime) + // Reconcile a header published before replay installed the anchor. + if slot == ^uint64(0) { + return + } + c := a.slots[slot+1] + if c == nil || c.prefetch == nil { + return + } + b := c.prefetch.batches[0] + if b == nil || b.ready == nil { + return + } + select { + case <-b.ready: + a.noteChildRepairHeaderLocked(c, b) + default: + } +} + +// Called by the asynchronous decoder, not replay's busy execution goroutine. +func (a *SlotAssembler) noteChildRepairHeaderLocked(s *slotState, b *prefetchedShredBatch) { + p := a.streamRepairParent + if p == nil || s == nil || a.slots[s.slot] != s || b == nil || !b.marker || b.parent == nil { + return + } + if s == p && b.parent.FromUpdateParent { + a.streamRepairParent = nil + a.streamRepairChild = nil + return + } + if p.slot == ^uint64(0) || s.slot != p.slot+1 { + return + } + if b.parent.FromUpdateParent || b.parent.ParentSlot != p.slot { + a.streamRepairInvalidChild = s + a.streamRepairChild = nil + return + } + if a.streamRepairInvalidChild == s || b.start != 0 || b.err != nil || b.parent.ParentSlot != p.slot || a.streamRepairChild == s { + return + } + if time.Now().After(a.streamRepairUntil) || a.streamStatusLocked(StreamGeneration{slot: p.slot, state: p}) == StreamGone { + return + } + a.streamRepairChild = s + // Nonblocking channel send has no callback or lock acquisition. It uses the + // existing coalesced/minimum-spacing repair scheduler, including under a.mu. + select { + case a.streamRepairWake <- struct{}{}: + default: + } +} + +func (a *SlotAssembler) childRepairRequest(now time.Time) (SlotRepairRequest, bool) { + a.mu.Lock() + defer a.mu.Unlock() + p, c := a.streamRepairParent, a.streamRepairChild + if a.streamSubscriber == nil || p == nil || c == nil || now.After(a.streamRepairUntil) { + return SlotRepairRequest{}, false + } + if a.streamStatusLocked(StreamGeneration{slot: p.slot, state: p}) == StreamGone || a.slots[c.slot] != c || c.completing { + return SlotRepairRequest{}, false + } + req, ok := c.repairRequestWithPrefix(childRepairLimit, true) + if !ok || len(req.MissingDataShreds) == 0 { + return SlotRepairRequest{}, false + } + // Only the earliest span, not four unrelated holes or highest-index probes. + first := req.MissingDataShreds[0] + end := first + 1 + for _, f := range c.fecSets { + if f.haveLayout && f.fecSetIndex <= first && first < f.fecSetIndex+uint32(f.layout.dataShreds) { + end = f.fecSetIndex + uint32(f.layout.dataShreds) + break + } + } + n := 1 + for n < len(req.MissingDataShreds) && req.MissingDataShreds[n] < end { + n++ + } + req.MissingDataShreds = req.MissingDataShreds[:n] + req.NeedHighestDataShred = false + return req, true +} diff --git a/pkg/turbine/child_repair_test.go b/pkg/turbine/child_repair_test.go new file mode 100644 index 000000000..46e26c035 --- /dev/null +++ b/pkg/turbine/child_repair_test.go @@ -0,0 +1,160 @@ +package turbine + +import ( + "errors" + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/stretchr/testify/require" + "net" + "testing" + "time" +) + +func childRepairFixture(t *testing.T) (*SlotAssembler, *slotState, *slotState, *prefetchedShredBatch) { + t.Helper() + a := NewSlotAssembler() + a.SubscribeStream(make(chan StreamEvent, 1)) + p := newRepairSelectionSlot(100) + c := newRepairSelectionSlot(101) + addCodedSet(c, 0, 32, 32, seq(0, 19), 8) // deficit4 + addCodedSet(c, 32, 32, 32, seq(32, 56), 6) // cheaper later span + a.slots[100] = p + a.slots[101] = c + a.maxObservedSlot = 101 + done := make(chan struct{}) + close(done) + b := &prefetchedShredBatch{start: 0, marker: true, parent: &AlpenglowParentInfo{ParentSlot: 100}, ready: done} + c.prefetch = &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: b}} + a.SetStreamRepairParent(StreamGeneration{slot: 100, state: p}) + return a, p, c, b +} +func TestChildRepairEarlyHeaderAndDroppedEvents(t *testing.T) { + a, _, c, b := childRepairFixture(t) + req, ok := a.childRepairRequest(time.Now()) + require.True(t, ok) + require.Equal(t, []uint32{20, 21, 22, 23}, req.MissingDataShreds) + require.False(t, req.NeedHighestDataShred) + a.streamRepairChild = nil + a.streamSubscriber <- StreamEvent{} // full notification channel + wake := make(chan struct{}, 1) + a.streamRepairWake = wake + a.mu.Lock() + a.publishStreamBatchReadyLocked(c, b) + a.mu.Unlock() + require.Len(t, wake, 1) + require.Equal(t, uint64(1), a.StreamDroppedEvents()) + _, ok = a.childRepairRequest(time.Now()) + require.True(t, ok) + // Repeated publication is not a new repair wakeup. + <-wake + a.mu.Lock() + a.publishStreamBatchReadyLocked(c, b) + a.mu.Unlock() + require.Empty(t, wake) +} +func TestChildRepairLifecycle(t *testing.T) { + for _, kind := range []string{"expiry", "clear", "child-reset", "parent-reset", "child-complete", "parent-replaced", "unsubscribe", "update-parent", "wrong-parent"} { + t.Run(kind, func(t *testing.T) { + a, p, c, b := childRepairFixture(t) + switch kind { + case "expiry": + a.streamRepairUntil = time.Now().Add(-time.Second) + case "clear": + a.SetStreamRepairParent(StreamGeneration{}) + case "child-reset": + c.prefetch = nil // fixture has no live prefetch workers + a.ResetSlot(c.slot) + case "parent-reset": + a.ResetSlot(p.slot) + case "child-complete": + c.completing = true + case "parent-replaced": + a.slots[p.slot] = newRepairSelectionSlot(p.slot) + case "unsubscribe": + a.SubscribeStream(nil) + case "update-parent", "wrong-parent": + changed := *b + parent := *b.parent + changed.parent = &parent + if kind == "update-parent" { + changed.start = 32 + parent.FromUpdateParent = true + } else { + parent.ParentSlot = 99 + } + a.mu.Lock() + a.noteChildRepairHeaderLocked(c, &changed) + a.noteChildRepairHeaderLocked(c, b) + a.mu.Unlock() + } + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) + }) + } + // Parent assembly can finish while replay is still executing it. + a, p, _, _ := childRepairFixture(t) + delete(a.slots, p.slot) + p.streamCompleted = true + _, ok := a.childRepairRequest(time.Now()) + require.True(t, ok) + a.ResetSlot(p.slot) + _, ok = a.childRepairRequest(time.Now()) + require.False(t, ok) +} +func TestChildRepairRejectsUnrelatedAndInvalidHeader(t *testing.T) { + a, _, c, b := childRepairFixture(t) + a.streamRepairChild = nil + bad := *b + bad.err = errors.New("invalid header") + a.mu.Lock() + a.noteChildRepairHeaderLocked(c, &bad) + a.mu.Unlock() + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) + other := newRepairSelectionSlot(102) + a.mu.Lock() + a.noteChildRepairHeaderLocked(other, b) + a.mu.Unlock() + _, ok = a.childRepairRequest(time.Now()) + require.False(t, ok) +} +func TestChildRepairUsesOnlyRemainingBudget(t *testing.T) { + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + c := newPacingTestClient(t) + peers := []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + req := SlotRepairRequest{Slot: 101, MissingDataShreds: []uint32{20, 21, 22, 23, 24, 25}, NeedHighestDataShred: true} + require.Zero(t, c.sendChildRepair(conn, peers, req, 0, time.Second)) + require.Equal(t, 2, c.sendChildRepair(conn, peers, req, 2, time.Second)) + require.Equal(t, 2, c.sendChildRepair(conn, peers, req, 100, time.Second)) + require.Zero(t, c.sendChildRepair(conn, peers, req, 100, time.Second)) + require.Len(t, c.outstanding, 4) + for k := range c.outstanding { + require.Equal(t, repairRequestWindowIndex, k.kind) + require.Zero(t, k.attempt) + } +} + +func TestChildRepairRejectsStaleParentAnchor(t *testing.T) { + a, p, _, _ := childRepairFixture(t) + a.slots[p.slot] = newRepairSelectionSlot(p.slot) + a.SetStreamRepairParent(StreamGeneration{slot: p.slot, state: p}) + require.Nil(t, a.streamRepairParent) +} + +func TestChildRepairParentUpdateClearsLookahead(t *testing.T) { + a, p, _, b := childRepairFixture(t) + update := *b + info := *b.parent + info.FromUpdateParent = true + update.parent = &info + update.start = 32 + a.mu.Lock() + a.noteChildRepairHeaderLocked(p, &update) + a.mu.Unlock() + _, ok := a.childRepairRequest(time.Now()) + require.False(t, ok) +} diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index 372e44000..b3e67d654 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -208,6 +208,9 @@ func (r *UDPReceiver) SetRepairPeerSource(identity ed25519.PrivateKey, source fu return err } r.repairClient = client + r.assembler.mu.Lock() + r.assembler.streamRepairWake = client.priorityWake + r.assembler.mu.Unlock() return nil } @@ -986,3 +989,15 @@ func (r *UDPReceiver) hydrateLoop(ctx context.Context) { } } } + +// PrioritizeStreamRepair also anchors bounded asynchronous child lookahead. +func (r *UDPReceiver) PrioritizeStreamRepair(g StreamGeneration) { + if r == nil || r.assembler == nil { + return + } + r.assembler.SetStreamRepairParent(g) + slot := g.Slot() + if slot != 0 { + r.PrioritizeRepairSlot(slot) + } +} diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index 3c5a3e69e..36f4cbfb3 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -457,7 +457,8 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { return } priority, edge := assembler.RepairRequestsTiered(repairMaxSlotsPerScan, repairMaxMissingPerSlot) - if len(priority)+len(edge) == 0 { + child, haveChild := assembler.childRepairRequest(time.Now()) + if len(priority)+len(edge) == 0 && !haveChild { return } @@ -483,6 +484,9 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { } edgeDemand := tierSendDemand(edge, 1) want := tierSendDemand(priority, headInitial) + edgeDemand + if haveChild { + want += len(child.MissingDataShreds) + } if want > repairMaxOutstanding { want = repairMaxOutstanding } @@ -506,6 +510,10 @@ func (c *repairClient) repairOnce(conn *net.UDPConn, assembler *SlotAssembler) { // below what the edge can use; leftover head budget flows to the edge. spent := c.sendTier(conn, peers, priority, splitRepairBudget(budget, edgeDemand), headPol, acct) spent += c.sendTier(conn, peers, edge, budget-spent, nil, acct) + // Parent/normal repair and freshness keep first claim on every token. + if haveChild { + spent += c.sendChildRepair(conn, peers, child, budget-spent, acct) + } c.returnRateTokens(budget - spent) } @@ -1567,3 +1575,19 @@ func (c *repairClient) stats() RepairStats { AvgResponseMillis: avgResponseMillis, } } + +// Called after normal priority and freshness work. Existing in-flight child +// requests count against lookahead capacity; no fanout or highest-index probes. +func (c *repairClient) sendChildRepair(conn *net.UDPConn, peers []gossip.RepairPeer, req SlotRepairRequest, budget int, acct time.Duration) int { + if budget <= 0 { + return 0 + } + c.mu.Lock() + room := childRepairLimit - c.outstandingForSlotLocked(req.Slot) + c.mu.Unlock() + if room <= 0 { + return 0 + } + req.NeedHighestDataShred = false + return c.sendTier(conn, peers, []SlotRepairRequest{req}, min(room, budget), nil, acct) +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index 95012e89d..09a222d97 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -181,6 +181,11 @@ func (a *SlotAssembler) SubscribeStream(ch chan<- StreamEvent) { a.mu.Lock() defer a.mu.Unlock() a.streamSubscriber = ch + if ch == nil { + a.streamRepairParent = nil + a.streamRepairChild = nil + a.streamRepairInvalidChild = nil + } } // StreamDroppedEvents reports wake-ups dropped because the subscriber was @@ -297,6 +302,7 @@ func (a *SlotAssembler) publishStreamBatchReadyLocked(s *slotState, batch *prefe if a.streamSubscriber == nil || s == nil || batch == nil { return } + a.noteChildRepairHeaderLocked(s, batch) g := StreamGeneration{slot: s.slot, state: s} a.publishStreamLocked(StreamEvent{Kind: StreamBatchReady, Slot: s.slot, Generation: g, Batch: newStreamBatch(g, batch)}) } From 3e22d9842a95ee998da58c9198ce625d634eda84 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Thu, 17 Sep 2026 00:04:15 -0500 Subject: [PATCH 062/111] turbine: select highest-shred followups after admission --- docs/transaction_sigverify_streaming.md | 15 +++ pkg/turbine/entry_pipeline_trace_test.go | 6 +- pkg/turbine/receiver.go | 7 +- pkg/turbine/repair.go | 78 ++---------- pkg/turbine/repair_followup.go | 69 +++++++++++ pkg/turbine/repair_followup_test.go | 146 +++++++++++++++++++++++ pkg/turbine/repair_pacing_test.go | 16 +-- 7 files changed, 258 insertions(+), 79 deletions(-) create mode 100644 pkg/turbine/repair_followup.go create mode 100644 pkg/turbine/repair_followup_test.go diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index bd5dac9f2..19f868d7f 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -256,3 +256,18 @@ completion/reset and changed parent markers invalidate its hint. Generation checks reject stale headers. Already-sent requests still use normal response and expiry handling. Notifications and repair wakeups remain nonblocking/coalesced; no extra workers or polling loop are introduced. + +### Highest-index repair followups + +A matched highest-index response triggers followup selection only after receiver +admission and FEC recovery. The assembler supplies its current deficit-aware +selection (at most 256 data requests), rather than treating the interval below +the response as missing. Completed, completing, evicted and absent assembler +slots produce no immediate followups. During disk-only catchup, selection waits +for hydration instead of blindly fetching data that may already be spooled. + +Followups retain the shared token bucket, admission limits, bulk retry policy, +and a reserved token for continued highest-index discovery when needed. A +snapshot can still race with subsequent arrivals; this removes known redundant +requests, not every possible duplicate. No assembler lock is held while signing +or sending requests. Response matching and peer credit are unchanged. diff --git a/pkg/turbine/entry_pipeline_trace_test.go b/pkg/turbine/entry_pipeline_trace_test.go index a39f8f087..5816287d4 100644 --- a/pkg/turbine/entry_pipeline_trace_test.go +++ b/pkg/turbine/entry_pipeline_trace_test.go @@ -148,9 +148,9 @@ func TestEntryRepairResponseCorrelation(t *testing.T) { c.outstanding[key] = rec c.byResponse[responseKey] = key } - require.False(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 4, Type: ShredTypeData})) + require.False(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 4, Type: ShredTypeData})) require.Empty(t, entryTraceConfig.repairs, "wrong index cannot be reported as matched") - require.True(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + require.True(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) r := <-entryTraceConfig.repairs require.Equal(t, "repair_response", r.Event) require.Equal(t, uint32(777), r.Nonce) @@ -158,7 +158,7 @@ func TestEntryRepairResponseCorrelation(t *testing.T) { require.Equal(t, late, r.Late) require.Equal(t, uint32(3), r.ReturnedIndex) require.Greater(t, r.ResponseAt, r.RequestedAt) - require.False(t, c.observeShredResponse(nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) + require.False(t, observeRepairForTest(c, nil, nonceTrailer(777), from, &Shred{Slot: 42, Index: 3, Type: ShredTypeData})) require.Empty(t, entryTraceConfig.repairs, "consumed nonce cannot count twice") } } diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index b3e67d654..6fce107f0 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -755,9 +755,9 @@ func (r *UDPReceiver) processPacket(ctx context.Context, conn *net.UDPConn, pack } authenticatedRoot = &root } - matchedRepair := false + matchedRepair, highestRepair := false, false if onRepairSocket && r.repairClient != nil { - matchedRepair = r.repairClient.observeShredResponse(conn, packet, addr, shred) + matchedRepair, highestRepair = r.repairClient.matchShredResponse(packet, addr, shred) } if onRepairSocket && !matchedRepair { r.repairSocketUnmatched.Add(1) @@ -818,6 +818,9 @@ func (r *UDPReceiver) processPacket(ctx context.Context, conn *net.UDPConn, pack } return true } + if highestRepair { + r.repairClient.followupHighestResponse(conn, r.assembler, shred.Slot) + } return r.submitCompletion(ctx, work, false) } diff --git a/pkg/turbine/repair.go b/pkg/turbine/repair.go index 36f4cbfb3..7e19baa1d 100644 --- a/pkg/turbine/repair.go +++ b/pkg/turbine/repair.go @@ -621,21 +621,23 @@ func shredSatisfiesRequest(key repairRequestKey, shred *Shred) bool { } } -// observeShredResponse matches an incoming packet against outstanding repair +// matchShredResponse matches an incoming packet against outstanding repair // requests (responder address + nonce). Returns true when the shred was // delivered BY REPAIR — it answers one of our requests — so the caller can -// attribute it in per-slot repair accounting. -func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, from *net.UDPAddr, shred *Shred) bool { +// attribute it in per-slot repair accounting. highest is a discovery hint for +// followup selection only AFTER the receiver admits the shred; matching and +// peer credit alone do not authorize requests for the claimed range. +func (c *repairClient) matchShredResponse(packet []byte, from *net.UDPAddr, shred *Shred) (matched, highest bool) { if from == nil || shred == nil { - return false + return false, false } nonce, ok := repairproto.ResponseNonce(packet) if !ok { - return false + return false, false } addrKey, ok := repairAddressKeyFromUDP(from) if !ok { - return false + return false, false } responseKey := repairResponseKey{addr: addrKey, nonce: nonce} @@ -669,7 +671,7 @@ func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, fr // so it still expires into a deserved timeout and keeps its in-flight // slot for retry. The peer gets nothing. c.mu.Unlock() - return false + return false, false } if late { // The expired record lives in exactly one generation; deleting from @@ -707,63 +709,7 @@ func (c *repairClient) observeShredResponse(conn *net.UDPConn, packet []byte, fr } else { c.responses.Add(1) } - // shredSatisfiesRequest already guaranteed slot match and a data shred; the - // gap-backfill path below is HWI-only. - if outstanding.key.kind != repairRequestHighestWindowIndex { - return true - } - - peers := c.peerSnapshot(time.Now()) - if len(peers) == 0 { - return true - } - start := outstanding.key.index - gap := 0 - if shred.Index > start { - gap = int(shred.Index - start) - } - ask := gap - if ask > repairMaxFollowupRequests { - ask = repairMaxFollowupRequests - } - chainProbe := !shred.LastInSlot() && shred.Index < maxDataShredsPerSlot-1 - if chainProbe { - ask++ - } - if ask == 0 { - return true - } - // Followups draw from the SAME token bucket as the scan. This path used - // to be unmetered — with hundreds of probed slots it pushed the total - // send rate ~70% past the cap, which is exactly the flood the peer-side - // QoS ban punishes. When the bucket is dry the scan's deficit-aware - // selection covers the slot on its own cadence. - grant := c.takeRateTokens(ask) - if grant <= 0 { - return true - } - windowBudget := grant - if chainProbe && windowBudget > 0 { - windowBudget-- // reserve the chained probe's token - } - // Discovery followups are bulk-paced and go through the same inflight - // dedup as the scan, so a window index already being repaired is not - // re-sent here. - bulk := bulkPolicy() - acct := c.accountingTimeout() - followups := 0 - for index := start; index < shred.Index && followups < windowBudget; index++ { - if c.sendShredAttempt(conn, peers, repairRequestWindowIndex, shred.Slot, index, bulk, acct) { - followups++ - } - } - if chainProbe && followups < grant { - if c.sendShredAttempt(conn, peers, repairRequestHighestWindowIndex, shred.Slot, shred.Index+1, bulk, acct) { - followups++ - } - } - c.returnRateTokens(grant - followups) - return true + return true, outstanding.key.kind == repairRequestHighestWindowIndex } // observeLatencyLocked folds one request->response latency into the EWMA and @@ -821,7 +767,7 @@ func bulkPolicy() retryPolicy { // satisfyDataShred retires WindowIndex requests satisfied by a verified data // shred arriving through any path: a matched repair response, Turbine // broadcast, FEC/spool hydration, or a duplicate response. Request nonce -// matching still happens first in observeShredResponse so the answering peer +// matching still happens first in matchShredResponse so the answering peer // receives its proper timely/late credit. func (c *repairClient) satisfyDataShred(shred *Shred) { if c == nil || shred == nil || shred.Type != ShredTypeData { @@ -950,7 +896,7 @@ func (c *repairClient) sendShredAttempt(conn *net.UDPConn, peers []gossip.Repair // count), then release BEFORE signing. Ed25519 signing is ~tens of // microseconds; at tens of thousands of req/s, holding the lock across it // serialized every send against the response-processing path - // (observeShredResponse needs the same lock) and could stall the UDP + // (matchShredResponse needs the same lock) and could stall the UDP // receive loop into kernel drops. The reserve is enough for coherence: a // response for this attempt cannot arrive until after we WriteToUDP below, // which is strictly after we register outstanding/byResponse. diff --git a/pkg/turbine/repair_followup.go b/pkg/turbine/repair_followup.go new file mode 100644 index 000000000..8b250ca2c --- /dev/null +++ b/pkg/turbine/repair_followup.go @@ -0,0 +1,69 @@ +package turbine + +import ( + "net" + "time" +) + +// highestRepairFollowup snapshots current deficits AFTER response admission and +// FEC recovery. It never treats a highest-index response as proof that earlier +// pieces are missing. The returned selection can race with later arrivals, but +// no assembler lock is held during signing, network sends, or repair locking. +func (a *SlotAssembler) highestRepairFollowup(slot uint64) (SlotRepairRequest, bool) { + a.mu.Lock() + defer a.mu.Unlock() + if a.slotTooOldLocked(slot) { + return SlotRepairRequest{}, false + } + if _, done := a.completedSlots[slot]; done { + return SlotRepairRequest{}, false + } + s := a.slots[slot] + if s == nil || s.completing { + return SlotRepairRequest{}, false + } + return s.repairRequest(repairMaxFollowupRequests) +} + +// Only invoked for a matched highest-index response which passed admission. +// Disk-only catchup slots defer selection until hydration supplies assembler +// state; blindly backfilling those would ignore data already held in the spool. +func (c *repairClient) followupHighestResponse(conn *net.UDPConn, a *SlotAssembler, slot uint64) { + req, ok := a.highestRepairFollowup(slot) + if !ok { + return + } + peers := c.peerSnapshot(time.Now()) + if len(peers) == 0 { + return + } + ask := len(req.MissingDataShreds) + if req.NeedHighestDataShred { + ask++ + } + grant := c.takeRateTokens(ask) + if grant <= 0 { + return + } + // Reserve discovery capacity as before, even when missing data fills the cap. + window := grant + if req.NeedHighestDataShred { + window-- + } + pol, acct := bulkPolicy(), c.accountingTimeout() + sent := 0 + for _, index := range req.MissingDataShreds { + if sent >= window { + break + } + if c.sendShredAttempt(conn, peers, repairRequestWindowIndex, slot, index, pol, acct) { + sent++ + } + } + if req.NeedHighestDataShred && sent < grant { + if c.sendShredAttempt(conn, peers, repairRequestHighestWindowIndex, slot, req.HighestDataShredIndex, pol, acct) { + sent++ + } + } + c.returnRateTokens(grant - sent) +} diff --git a/pkg/turbine/repair_followup_test.go b/pkg/turbine/repair_followup_test.go new file mode 100644 index 000000000..92778b877 --- /dev/null +++ b/pkg/turbine/repair_followup_test.go @@ -0,0 +1,146 @@ +package turbine + +import ( + "context" + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/gossip" + "github.com/stretchr/testify/require" + "net" + "sync" + "testing" + "time" +) + +// Existing protocol/pacing fixtures model cold admission without packet parsing. +// Production invokes followups only after the full receiver admission path. +func observeRepairForTest(c *repairClient, conn *net.UDPConn, packet []byte, from *net.UDPAddr, sh *Shred) bool { + matched, highest := c.matchShredResponse(packet, from, sh) + if highest { + a := NewSlotAssembler() + s := newRepairSelectionSlot(sh.Slot) + s.shreds[sh.Index] = sh + s.haveLast, s.lastIndex = sh.LastInSlot(), sh.Index + a.slots[sh.Slot] = s + c.followupHighestResponse(conn, a, sh.Slot) + } + return matched +} + +func TestHighestRepairFollowupSelection(t *testing.T) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + // A recovered first span must not be requested again; the next span + // lacks twelve data but has eight coding, so only four repairs are needed. + addCodedSet(s, 0, 32, 32, seq(0, 31), 0) + addCodedSet(s, 32, 32, 32, seq(32, 51), 8) + s.haveLast, s.lastIndex = true, 63 + r, ok := a.highestRepairFollowup(50) + require.True(t, ok) + require.Equal(t, []uint32{52, 53, 54, 55}, r.MissingDataShreds) + require.False(t, r.NeedHighestDataShred) + for i := uint32(52); i <= 63; i++ { + s.shreds[i] = &Shred{Index: i} + } + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + delete(s.shreds, 55) + s.completing = true + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + s.completing = false + a.completedSlots[50] = struct{}{} + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) + delete(a.completedSlots, 50) + a.ResetSlot(50) + _, ok = a.highestRepairFollowup(50) + require.False(t, ok) +} + +func TestHighestRepairFollowupColdAndDiscovery(t *testing.T) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + s.shreds[600] = &Shred{Index: 600} + r, ok := a.highestRepairFollowup(50) + require.True(t, ok) + require.Len(t, r.MissingDataShreds, 256) + require.Equal(t, uint32(0), r.MissingDataShreds[0]) + require.True(t, r.NeedHighestDataShred) + require.Equal(t, uint32(601), r.HighestDataShredIndex) + // Possession holes only, even without a coding layout. + for i := uint32(0); i < 600; i++ { + s.shreds[i] = &Shred{Index: i} + } + delete(s.shreds, 299) + r, ok = a.highestRepairFollowup(50) + require.True(t, ok) + require.Equal(t, []uint32{299}, r.MissingDataShreds) +} + +func TestHighestRepairFollowupSendsOnlyDeficit(t *testing.T) { + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + c := newPacingTestClient(t) + c.peerCache = []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + c.peerCacheAt = time.Now() + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + addCodedSet(s, 0, 32, 32, seq(0, 19), 8) + s.haveLast = true + s.lastIndex = 31 + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) // same inflight dedupe + // Race a reset with read-only selection; no old slot is recreated. + var wg sync.WaitGroup + wg.Add(1) + go func() { + defer wg.Done() + for i := 0; i < 100; i++ { + a.highestRepairFollowup(50) + } + }() + a.ResetSlot(50) + wg.Wait() + c.followupHighestResponse(conn, a, 50) + require.Equal(t, uint64(4), c.requests.Load()) +} + +func TestReceiverHighestFollowupUsesAdmittedState(t *testing.T) { + packets := fixtures.DataShreds(t, "mainnet", 102815960) + require.Greater(t, len(packets), 12) + conn, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer conn.Close() + sink, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)}) + require.NoError(t, err) + defer sink.Close() + r := NewUDPReceiver("127.0.0.1:0") + c := newPacingTestClient(t) + r.repairClient = c + c.peerCache = []gossip.RepairPeer{{Addr: sink.LocalAddr().(*net.UDPAddr)}} + c.peerCacheAt = time.Now() + for _, p := range packets[:11] { + require.True(t, r.processPacket(context.Background(), nil, p, nil, false)) + } + from := &net.UDPAddr{IP: net.IPv4(10, 0, 0, 9), Port: 8009} + addr, _ := repairAddressKeyFromUDP(from) + key := repairRequestKey{kind: repairRequestHighestWindowIndex, slot: 102815960, index: 0} + c.outstanding[key] = outstandingRepairRequest{key: key, nonce: 42, addr: addr, sentAt: time.Now()} + c.byResponse[repairResponseKey{addr: addr, nonce: 42}] = key + packet := append(append([]byte(nil), packets[11]...), nonceTrailer(42)...) + require.True(t, r.processPacket(context.Background(), conn, packet, from, true)) + require.Equal(t, uint64(1), c.requests.Load(), "only continued discovery, no already held indices") + for key := range c.outstanding { + require.Equal(t, repairRequestHighestWindowIndex, key.kind) + require.Equal(t, uint32(12), key.index) + } +} diff --git a/pkg/turbine/repair_pacing_test.go b/pkg/turbine/repair_pacing_test.go index 8391915a5..ad747cb1a 100644 --- a/pkg/turbine/repair_pacing_test.go +++ b/pkg/turbine/repair_pacing_test.go @@ -90,7 +90,7 @@ func TestLateResponseMatchedAfterExpiry(t *testing.T) { packet := nonceTrailer(777) shred := &Shred{Slot: 42, Index: 3, Type: ShredTypeData} - if !c.observeShredResponse(nil, packet, from, shred) { + if !observeRepairForTest(c, nil, packet, from, shred) { t.Fatal("late answer must be attributed as a repair delivery") } if c.lateResponses.Load() != 1 { @@ -104,7 +104,7 @@ func TestLateResponseMatchedAfterExpiry(t *testing.T) { } // Second delivery of the same nonce: entry consumed, ordinary broadcast. - if c.observeShredResponse(nil, packet, from, shred) { + if observeRepairForTest(c, nil, packet, from, shred) { t.Fatal("expired entry must be single-use") } } @@ -122,7 +122,7 @@ func TestLateResponseWrongSlotRejected(t *testing.T) { c.byResponse[repairResponseKey{addr: addrKey, nonce: 900}] = reqKey c.expireOutstanding(time.Now()) - if c.observeShredResponse(nil, nonceTrailer(900), from, &Shred{Slot: 43, Index: 3, Type: ShredTypeData}) { + if observeRepairForTest(c, nil, nonceTrailer(900), from, &Shred{Slot: 43, Index: 3, Type: ShredTypeData}) { t.Fatal("wrong-slot late answer must not be attributed as repair") } if c.lateResponses.Load() != 0 { @@ -165,7 +165,7 @@ func TestNonConformingResponseRejected(t *testing.T) { c.addInflightLocked(tc.key.shred(), time.Now()) c.mu.Unlock() - if c.observeShredResponse(nil, nonceTrailer(111), from, tc.shred) { + if observeRepairForTest(c, nil, nonceTrailer(111), from, tc.shred) { t.Fatal("non-conforming answer must not be attributed as a repair delivery") } if c.responses.Load() != 0 || c.lateResponses.Load() != 0 { @@ -211,7 +211,7 @@ func TestLateHighestResponseFiresFollowups(t *testing.T) { c.byResponse[repairResponseKey{addr: addrKey, nonce: 6}] = reqKey c.expireOutstanding(time.Now()) - if !c.observeShredResponse(conn, nonceTrailer(6), from, &Shred{Slot: 50, Index: 200, Type: ShredTypeData}) { + if !observeRepairForTest(c, conn, nonceTrailer(6), from, &Shred{Slot: 50, Index: 200, Type: ShredTypeData}) { t.Fatal("late HWI answer must match") } if c.lateResponses.Load() != 1 || c.responses.Load() != 0 { @@ -448,7 +448,7 @@ func TestRepairAnswerCancelsSiblingAttempts(t *testing.T) { // The ORIGINAL peer answers timely; the sibling is neutral-cancelled. from := &net.UDPAddr{IP: sinkAddr.IP, Port: sinkAddr.Port} - if !c.observeShredResponse(conn, nonceTrailer(o0.nonce), from, &Shred{Slot: 60, Index: 3, Type: ShredTypeData}) { + if !observeRepairForTest(c, conn, nonceTrailer(o0.nonce), from, &Shred{Slot: 60, Index: 3, Type: ShredTypeData}) { t.Fatal("original attempt's answer must match") } c.mu.Lock() @@ -568,7 +568,7 @@ func TestFollowupsAreMeteredByTokenBucket(t *testing.T) { // by the primed outstanding entry, leaving a stray token. c.takeRateTokens(repairMaxRequestsPerSecond) c.takeRateTokens(repairMaxRequestsPerSecond) - if !c.observeShredResponse(conn, packet, from, shred) { + if !observeRepairForTest(c, conn, packet, from, shred) { t.Fatal("response itself must match") } if got := c.requests.Load(); got != 0 { @@ -580,7 +580,7 @@ func TestFollowupsAreMeteredByTokenBucket(t *testing.T) { c.rateRefillAt = time.Now() c.rateTokens = repairMaxRequestsPerSecond c.mu.Unlock() - if !c.observeShredResponse(conn, packet, from, shred) { + if !observeRepairForTest(c, conn, packet, from, shred) { t.Fatal("response itself must match") } // A full bucket sends the whole revealed gap: under the adaptive per-peer From 1ec98614f7ba5f53ffba6e8b745d2bd792a6417b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Thu, 17 Sep 2026 00:30:02 -0500 Subject: [PATCH 063/111] turbine: test spool followups and benchmark fragmented selection --- pkg/turbine/repair_followup_test.go | 76 +++++++++++++++++++++++++++++ 1 file changed, 76 insertions(+) diff --git a/pkg/turbine/repair_followup_test.go b/pkg/turbine/repair_followup_test.go index 92778b877..a8bb6a21f 100644 --- a/pkg/turbine/repair_followup_test.go +++ b/pkg/turbine/repair_followup_test.go @@ -2,6 +2,7 @@ package turbine import ( "context" + "fmt" "github.com/Overclock-Validator/mithril/fixtures" "github.com/Overclock-Validator/mithril/pkg/gossip" "github.com/stretchr/testify/require" @@ -144,3 +145,78 @@ func TestReceiverHighestFollowupUsesAdmittedState(t *testing.T) { require.Equal(t, uint32(12), key.index) } } + +// Disk-only discovery must remain repairable after hydration enters the slot. +func TestHighestRepairFollowupAfterDiskOnlyHydration(t *testing.T) { + const slot = uint64(102815960) + packets := fixtures.DataShreds(t, "mainnet", slot) + r := NewUDPReceiver("127.0.0.1:0") + spool, err := OpenShredSpool(t.TempDir(), 0) + require.NoError(t, err) + defer spool.Close() + r.SetShredSpool(spool) + r.assembler.maxObservedSlot = slot + 1000 + r.SetHydrationWindow(slot-8, slot-1) + require.True(t, r.skipAssemblyForSpool(slot)) + // Seed only a partial range on disk; an authentic highest response extends it. + for _, p := range packets[:10] { + require.True(t, r.processPacket(context.Background(), nil, p, nil, false)) + } + c := newPacingTestClient(t) + r.repairClient = c + from := &net.UDPAddr{IP: net.IPv4(10, 0, 0, 9), Port: 8009} + addr, _ := repairAddressKeyFromUDP(from) + key := repairRequestKey{kind: repairRequestHighestWindowIndex, slot: slot, index: 0} + c.outstanding[key] = outstandingRepairRequest{key: key, nonce: 43, addr: addr, sentAt: time.Now()} + c.byResponse[repairResponseKey{addr: addr, nonce: 43}] = key + packet := append(append([]byte(nil), packets[11]...), nonceTrailer(43)...) + require.True(t, r.processPacket(context.Background(), nil, packet, from, true)) + require.Equal(t, uint64(0), c.requests.Load()) + _, live := r.assembler.HeadShredDetail(slot) + require.False(t, live) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { defer close(done); r.hydrateLoop(ctx) }() + defer func() { cancel(); <-done }() + r.SetRetentionFloor(slot) // replay protects the hydration window during catchup + r.assembler.PrioritizeRepairSlot(slot) + r.SetHydrationWindow(slot, slot) + require.Eventually(t, func() bool { return r.hydratedSlots.Load() > 0 }, time.Second, time.Millisecond) + req, ok := r.assembler.highestRepairFollowup(slot) + require.True(t, ok) + require.Equal(t, []uint32{10}, req.MissingDataShreds) + require.True(t, req.NeedHighestDataShred) + require.Equal(t, uint32(12), req.HighestDataShredIndex) + // Replay priority makes the hole eligible for the ordinary scheduler. + r.assembler.PrioritizeRepairSlot(slot) + priority, _ := r.assembler.RepairRequestsTiered(64, 2048) + require.NotEmpty(t, priority) + require.Equal(t, slot, priority[0].Slot) + require.Equal(t, []uint32{10}, priority[0].MissingDataShreds) +} + +func BenchmarkHighestRepairFollowup(b *testing.B) { + for _, n := range []int{16384, 65536} { + for _, fragmented := range []bool{false, true} { + b.Run(fmt.Sprintf("shreds%d/fragmented%t", n, fragmented), func(b *testing.B) { + a := NewSlotAssembler() + s := newRepairSelectionSlot(50) + a.slots[50] = s + for start := 0; start < n; start += 32 { + last := start + 31 + coding := 0 + if fragmented { + last = start + 19 + coding = 8 + } + addCodedSet(s, uint32(start), 32, 32, seq(uint32(start), uint32(last)), coding) + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + a.highestRepairFollowup(50) + } + }) + } + } +} From eb17bbf8708b28ffcdf09b9e742a5bf3fdb9e7f3 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:46:59 -0500 Subject: [PATCH 064/111] sbpf: restore v2 memory opcodes and differential coverage --- docs/sbpf-interpreter-benchmarks.md | 11 ++- pkg/sbpf/interpreter.go | 80 ++++++++++++++++++++ pkg/sbpf/interpreter_v2_test.go | 113 ++++++++++++++++++++++++++++ pkg/sbpf/perf_differential_test.go | 90 +++++++++++++++++++--- 4 files changed, 281 insertions(+), 13 deletions(-) create mode 100644 pkg/sbpf/interpreter_v2_test.go diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 095eb0b06..c399f4da5 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -32,9 +32,14 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ ## Correctness and comparison boundaries The generated-program harness compares return values, errors, CU usage and memory -for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate -and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and -verifier semantics changes are excluded from this performance work. +for 100,000 generated programs, evenly divided across SBPF v0–v3. The generator +uses v2-specific memory, arithmetic and constant-loading encodings; the dump +reports verifier rejection or execution results, and logs accepted counts per version. +Run the same harness on both trees: set `SBPF_DIFF_OUT` separately on the reference +and candidate and compare the files. `SBPF_CHECK_POOL_ZERO=1` additionally checks +the candidate’s clear-on-return pool invariant; older references may clear on +acquisition instead. ARSH and verifier semantics changes are excluded from this +performance work. For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, identical transaction parallelism, and the same slot interval. Compare normalized diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e7e28fda0..e27392d7c 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1299,6 +1299,86 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { var err error switch ins.Op() { + // In v2 these encodings are memory operations, not MUL/DIV/MOD. + // Run dispatches their non-v2 arithmetic forms on the hot path. + case OpLd1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt1BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(ins.Uimm())) + pc++ + case OpSt1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(r[ins.Src()])) + pc++ + case OpLd2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt2BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(ins.Uimm())) + pc++ + case OpSt2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(r[ins.Src()])) + pc++ + case OpLd8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = v + pc++ + case OpSt8BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, uint64(ins.Imm())) + pc++ + case OpSt8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, r[ins.Src()]) + pc++ case OpDiv32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) pc++ diff --git a/pkg/sbpf/interpreter_v2_test.go b/pkg/sbpf/interpreter_v2_test.go new file mode 100644 index 000000000..39fd1e161 --- /dev/null +++ b/pkg/sbpf/interpreter_v2_test.go @@ -0,0 +1,113 @@ +package sbpf + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" + "github.com/stretchr/testify/require" +) + +// These byte values alias arithmetic instructions outside v2. Exercise all +// relocated memory instructions through Run, not just the cold handler. +func TestInterpreterV2MemoryOpcodes(t *testing.T) { + for _, width := range []int{1, 2, 4, 8} { + loads := map[int]uint8{1: OpLd1BReg, 2: OpLd2BReg, 4: OpLd4BReg, 8: OpLd8BReg} + immediates := map[int]uint8{1: OpSt1BImm, 2: OpSt2BImm, 4: OpSt4BImm, 8: OpSt8BImm} + registers := map[int]uint8{1: OpSt1BReg, 2: OpSt2BReg, 4: OpSt4BReg, 8: OpSt8BReg} + for _, kind := range []string{"load", "store_imm", "store_reg"} { + for _, region := range []string{"heap", "stack", "input", "rodata", "unmapped"} { + t.Run(fmt.Sprintf("%s/%d/%s", kind, width, region), func(t *testing.T) { + const value = uint64(0xfedcba9876543210) + immediate := uint32(0xf123abcd) + var addr uint64 + switch region { + case "heap": + addr = VaddrHeap + 9 + case "stack": + addr = VaddrStack + 9 + case "input": + addr = VaddrInput + 9 + case "rodata": + addr = VaddrProgram + 9 + case "unmapped": + addr = 0x600000009 + } + // Negative offsets and unaligned addresses must behave identically to + // the original interpreter, including the post-instruction exception PC. + text := diffLoadImm64(5, addr+3, sbpfver.SbpfVersionV2) + text = append(text, diffLoadImm64(6, value, sbpfver.SbpfVersionV2)...) + var op Slot + switch kind { + case "load": + op = slot(loads[width], 0, 5, -3, 0) + case "store_imm": + op = slot(immediates[width], 5, 0, -3, immediate) + case "store_reg": + op = slot(registers[width], 5, 6, -3, 0) + } + text = append(text, op, slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(text, sbpfver.SbpfVersionV2) + p.RO = bytes.Repeat([]byte{0xa5}, 32) + require.NoError(t, p.Verify()) + cm := cu.NewComputeMeter(100) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32, Input: bytes.Repeat([]byte{0xa5}, 32), ComputeMeter: &cm, Syscalls: noSyscalls}) + defer ip.Finish() + var memory []byte + switch region { + case "heap": + memory = ip.heap + case "stack": + memory = ip.stack.mem + case "input": + memory = ip.input + case "rodata": + memory = p.RO + } + if memory != nil { + // Use Write for writable VM storage so pooled-memory tracking is kept. + if region != "rodata" { + require.NoError(t, ip.Write(addr-9, bytes.Repeat([]byte{0xa5}, 32))) + } + } + before := append([]byte(nil), memory...) + ret, used, err := ip.Run() + if region == "unmapped" || (region == "rodata" && kind != "load") { + require.Error(t, err) + var exc *Exception + require.ErrorAs(t, err, &exc) + require.Equal(t, int64(5), exc.PC) + var access ExcBadAccess + require.ErrorAs(t, err, &access) + require.Equal(t, addr, access.Addr) + require.Equal(t, uint64(width), access.Size) + require.Equal(t, kind != "load", access.Write) + require.Equal(t, uint64(95), cm.Remaining()) + require.Equal(t, before, memory) + return + } + require.NoError(t, err) + require.Equal(t, uint64(6), used) + require.Equal(t, uint64(94), cm.Remaining()) + var encoded [8]byte + if kind == "load" { + copy(encoded[:], before[9:9+width]) + require.Equal(t, binary.LittleEndian.Uint64(encoded[:]), ret) + } else { + v := value + if kind == "store_imm" { + signed := int32(immediate) + v = uint64(int64(signed)) + } + binary.LittleEndian.PutUint64(encoded[:], v) + copy(before[9:9+width], encoded[:width]) + } + require.Equal(t, before, memory) + }) + } + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go index 1d679d4e4..0d072f564 100644 --- a/pkg/sbpf/perf_differential_test.go +++ b/pkg/sbpf/perf_differential_test.go @@ -2,7 +2,6 @@ package sbpf import ( "bufio" - "encoding/binary" "fmt" "hash/fnv" "math/rand" @@ -97,6 +96,15 @@ func diffRegistry(h uint32) (Syscall, bool) { return nil, false } +// v2 replaces LDDW with MOV32 + HOR64. MOV32 avoids sign-extending the +// low word before ORing in the high word. +func diffLoadImm64(dst uint8, value uint64, ver uint32) []Slot { + if ver == sbpfver.SbpfVersionV2 { + return []Slot{slot(OpMov32Imm, dst, 0, 0, uint32(value)), slot(OpHor64Imm, dst, 0, 0, uint32(value>>32))} + } + return []Slot{slot(OpLddw, dst, 0, 0, uint32(value)), slot(0, 0, 0, 0, uint32(value>>32))} +} + func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 imm := func() uint32 { @@ -117,6 +125,29 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + if ver == sbpfver.SbpfVersionV2 { + // Arithmetic and memory encodings both change in v2. Keep generated ALU + // operations arithmetic rather than generating unintended memory accesses. + replacements := map[uint8]uint8{ + OpMul32Imm: OpLmul32Imm, OpMul32Reg: OpLmul32Reg, + OpMul64Imm: OpLmul64Imm, OpMul64Reg: OpLmul64Reg, + OpDiv32Imm: OpUdiv32Imm, OpDiv32Reg: OpUdiv32Reg, + OpDiv64Imm: OpUdiv64Imm, OpDiv64Reg: OpUdiv64Reg, + OpMod32Imm: OpUrem32Imm, OpMod32Reg: OpUrem32Reg, + OpMod64Imm: OpUrem64Imm, OpMod64Reg: OpUrem64Reg, + } + filtered := alu64[:0] + for _, op := range alu64 { + if op == OpNeg32 || op == OpNeg64 || op == OpLe { + continue + } + if replacement, ok := replacements[op]; ok { + op = replacement + } + filtered = append(filtered, op) + } + alu64 = filtered + } jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} switch rng.Intn(10) { @@ -126,7 +157,7 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { if op == OpLe || op == OpBe { i = []uint32{16, 32, 64}[rng.Intn(3)] } - if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm || op == OpUdiv32Imm || op == OpUdiv64Imm || op == OpUrem32Imm || op == OpUrem64Imm) && i == 0 { i = 3 } switch op { @@ -138,6 +169,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(op, reg(), reg(), 0, i)} case 4: // load ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg} + } // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random var base uint8 var off int16 @@ -154,6 +188,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} case 5: // store ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} + } var base uint8 var off int16 switch rng.Intn(5) { @@ -184,11 +221,11 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { default: // set up pointer registers switch rng.Intn(3) { case 0: // r5 = heap - return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + return diffLoadImm64(5, VaddrHeap, ver) case 1: // r1 = input + small - return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + return diffLoadImm64(1, VaddrInput+uint64(rng.Intn(64)), ver) default: // r9 = random 64-bit - return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + return diffLoadImm64(9, uint64(rng.Uint32())|uint64(rng.Intn(6))<<32, ver) } } } @@ -215,10 +252,14 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { } } // function + store := uint8(OpStxdw) + if ver == sbpfver.SbpfVersionV2 { + store = OpSt8BReg + } body = append(body, slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), slot(OpXor64Reg, 7, 6, 0, 0), - slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(store, 10, 7, int16(-8-rng.Intn(64)), 0), slot(OpExit, 0, 0, 0, 0)) p := mkProgram(body, ver) if ver < sbpfver.SbpfVersionV3 { @@ -231,6 +272,29 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { return p } +// Keep this check in the ordinary suite: merely selecting v2 is insufficient +// if its programs still contain legacy memory opcodes or LDDW and never run. +func TestDifferentialV2Generator(t *testing.T) { + rng := rand.New(rand.NewSource(12345)) + seen := make(map[uint8]bool) + for i := 0; i < 1000; i++ { + p := genProgram(rng, sbpfver.SbpfVersionV2) + if err := p.Verify(); err != nil { + continue + } + for _, ins := range p.Text { + seen[ins.Op()] = true + } + } + for _, op := range []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg, + OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, + OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} { + if !seen[op] { + t.Errorf("no verifier-accepted v2 program contains opcode %#x", op) + } + } +} + func memHash(bs ...[]byte) uint64 { h := fnv.New64a() for _, b := range bs { @@ -255,8 +319,10 @@ func TestDifferentialDump(t *testing.T) { rng := rand.New(rand.NewSource(12345)) const N = 100000 generated, verified := 0, 0 + var verifiedByVersion [4]int for i := 0; i < N; i++ { - ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + // Equal representation of every version, independent of RNG consumption. + ver := uint32(i % 4) p := genProgram(rng, ver) generated++ if err := p.Verify(); err != nil { @@ -264,6 +330,7 @@ func TestDifferentialDump(t *testing.T) { continue } verified++ + verifiedByVersion[ver]++ resolveCallTargetsIfSupported(p) input := make([]byte, 700) for j := range input { @@ -332,7 +399,10 @@ func TestDifferentialDump(t *testing.T) { heapPool.Put(hp) } } - t.Logf("generated=%d verified=%d", generated, verified) + for ver, count := range verifiedByVersion { + if count == 0 { + t.Errorf("no verifier-accepted programs for v%d", ver) + } + } + t.Logf("generated=%d verified=%d verified_by_version=%v", generated, verified, verifiedByVersion) } - -var _ = binary.LittleEndian From 9df40ca09508dc6ab18c2eea7796209155d92750 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:21:56 -0500 Subject: [PATCH 065/111] turbine: release prefetch reservations after failed completion --- pkg/turbine/assembler.go | 4 + pkg/turbine/entry_prefetch.go | 2 +- pkg/turbine/entry_prefetch_test.go | 153 +++++++++++++++++++++++++++++ pkg/turbine/stream.go | 5 + 4 files changed, 163 insertions(+), 1 deletion(-) diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 16d910f29..b324cafaf 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -527,6 +527,10 @@ func (a *SlotAssembler) finalizeCompletion(work *slotCompletionWork, processed p // state so catchup diagnostics report poison instead of a missing slot. state.noteError(processed.err) state.completing = false + // Diagnostics retain the poisoned slot, not a usable stream. Cancel + // readers now; cleanup returns capacity only after they have joined. + state.streamCancelReason = "completion_failed" + a.releasePrefetchLocked(state) a.mu.Unlock() return nil, processed.err } diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go index 1de831111..b055ae594 100644 --- a/pkg/turbine/entry_prefetch.go +++ b/pkg/turbine/entry_prefetch.go @@ -83,7 +83,7 @@ func newEntryPrefetchPool(ctx context.Context, a *SlotAssembler, verifier *trans func (a *SlotAssembler) prefetchEntriesLocked(s *slotState) { p := a.entryPrefetch - if p == nil || p.closed || p.ctx.Err() != nil { + if p == nil || p.closed || p.ctx.Err() != nil || s.streamCancelReason != "" { return } if s.batchIndex == nil && len(s.shreds) != 0 { diff --git a/pkg/turbine/entry_prefetch_test.go b/pkg/turbine/entry_prefetch_test.go index 679ee18b5..2c2aed451 100644 --- a/pkg/turbine/entry_prefetch_test.go +++ b/pkg/turbine/entry_prefetch_test.go @@ -273,6 +273,14 @@ func TestEntryPrefetchInvalidRetainedTransactionFailsClosed(t *testing.T) { } require.ErrorContains(t, finalErr, "transaction 1") require.False(t, a.SlotCompleted(300)) + p.cleanup.Wait() + a.mu.Lock() + g := StreamGeneration{slot: 300, state: a.slots[300]} + slots, bytes := p.slots, p.bytes + a.mu.Unlock() + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Zero(t, slots) + require.Zero(t, bytes) } func TestEntryPrefetchUpdateParentDiscardsInvalidOptimisticPrefix(t *testing.T) { @@ -526,3 +534,148 @@ func TestEntryPrefetchResetRetainsQueuedReservations(t *testing.T) { a.mu.Unlock() require.Zero(t, slots) } + +// Failed full blocks remain available for diagnostics, but must not consume +// the prefetch budget or remain usable streaming generations until retention. +func TestEntryPrefetchFailedCompletionReleasesCapacity(t *testing.T) { + for _, dropEvents := range []bool{false, true} { + t.Run(fmt.Sprintf("drop_events=%v", dropEvents), func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + capacity := 16 + if dropEvents { + capacity = 0 + } + events := make(chan StreamEvent, capacity) + a.SubscribeStream(events) + payload := prefetchTestPayload(t, verifierSignedTransactions(t, 1)) + for i := 0; i <= entryPrefetchSlots; i++ { + slot := uint64(900 + i) + // Zero entry count plus trailing bytes is an invalid component. + batches := prefetchTestShreds(t, slot, payload, make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + batch := waitPrefetchedBatch(t, a, slot, 0) + _, err := batch.verification.wait() + require.NoError(t, err) + a.mu.Lock() + s := a.slots[slot] + g := StreamGeneration{slot: slot, state: s} + a.mu.Unlock() + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + count, last := a.SlotAssemblyErrors(slot) + require.Positive(t, count) + require.Equal(t, failure.Error(), last) + require.False(t, a.SlotCompleted(slot)) + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Empty(t, a.PendingStreamBatches(g, 0)) + if !dropEvents { + event := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, g, event.Generation) + require.Equal(t, "completion_failed", event.Reason) + } + p.cleanup.Wait() + a.mu.Lock() + retained, slots, bytes := a.slots[slot], p.slots, p.bytes + // Repeated release/admission cannot double-refund or resurrect it. + a.releasePrefetchLocked(s) + a.prefetchEntriesLocked(s) + afterSlots := p.slots + a.mu.Unlock() + require.Same(t, s, retained, "preserve poisoned-slot diagnostics") + require.Zero(t, slots) + require.Zero(t, bytes) + require.Zero(t, afterSlots) + } + good := prefetchTestShreds(t, 920, payload, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, good[0])) + waitPrefetchedBatch(t, a, 920, 0) + blk := feedPrefetchShreds(t, a, good[1]) + require.NotNil(t, blk) + require.True(t, blk.TransactionSignaturesVerified()) + }) + } +} + +func TestEntryPrefetchFailedCompletionJoinsReaders(t *testing.T) { + started, release := make(chan struct{}), make(chan struct{}) + var once sync.Once + v := newTransactionVerifier(1, 8, func(*solana.Transaction) error { + close(started) + <-release + return nil + }) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + defer once.Do(func() { close(release) }) + batches := prefetchTestShreds(t, 930, prefetchTestPayload(t, verifierSignedTransactions(t, 1)), make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitSignal(t, started, "prefetch verifier") + waitPrefetchedBatch(t, a, 930, 0) + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + a.mu.Lock() + s := a.slots[930] + released, ctxErr, slots, bytes := s.prefetch.released, s.prefetch.ctx.Err(), p.slots, p.bytes + a.mu.Unlock() + require.True(t, released) + require.ErrorIs(t, ctxErr, context.Canceled) + require.Equal(t, 1, slots, "reader still owns the reservation") + require.Positive(t, bytes) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 930, state: s})) + once.Do(func() { close(release) }) + p.cleanup.Wait() + a.mu.Lock() + slots, bytes = p.slots, p.bytes + a.mu.Unlock() + require.Zero(t, slots) + require.Zero(t, bytes) +} + +// A failed generation that never received a reservation must not acquire one +// later when capacity becomes available (or prefetch is attached). +func TestEntryPrefetchFailedCompletionWithoutReservation(t *testing.T) { + a := NewSlotAssembler() + batches := prefetchTestShreds(t, 940, prefetchTestPayload(t, verifierSignedTransactions(t, 1)), make([]byte, 16)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + var failure error + for _, sh := range batches[1] { + blk, err := a.AddShred(sh) + require.Nil(t, blk) + if err != nil { + failure = err + } + } + require.Error(t, failure) + v := newTransactionVerifier(1, 8, nil) + defer v.closeAndWait() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + a.mu.Lock() + s := a.slots[940] + a.prefetchEntriesLocked(s) + reserved, slots := s.prefetch, p.slots + a.mu.Unlock() + require.Nil(t, reserved) + require.Zero(t, slots) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 940, state: s})) +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index 09a222d97..fe4af4213 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -221,6 +221,11 @@ func (a *SlotAssembler) StreamStatusOf(g StreamGeneration) StreamStatus { func (a *SlotAssembler) streamStatusLocked(g StreamGeneration) StreamStatus { if a.slots[g.slot] == g.state { + // Failed completions retain state for diagnostics. Polling must still + // see cancellation when the bounded event channel dropped its wake-up. + if g.state.streamCancelReason != "" { + return StreamGone + } return StreamActive } if g.state.streamCompleted { From 76a03c29132460645928cb6c837d22bdee3da1a6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:49:34 -0500 Subject: [PATCH 066/111] blockprod: validate transaction serialization before bank execution --- pkg/blockprod/bank.go | 14 ++++++-- pkg/blockprod/bank_test.go | 71 ++++++++++++++++++++++++++++++++++++++ pkg/blockprod/entry.go | 31 ++++++++++++----- 3 files changed, 105 insertions(+), 11 deletions(-) diff --git a/pkg/blockprod/bank.go b/pkg/blockprod/bank.go index b93e07d48..8535e86ec 100644 --- a/pkg/blockprod/bank.go +++ b/pkg/blockprod/bank.go @@ -184,7 +184,8 @@ func (b *WorkingBank) Forge(wire []byte) (ForgeResult, costmodel.ExceedReason) { return b.ForgeTransaction(tx, len(wire)) } -// ForgeTransaction executes and commits a parsed transaction. +// ForgeTransaction executes and commits a parsed transaction. The caller must +// keep it immutable: accepted transactions are retained for entry publication. func (b *WorkingBank) ForgeTransaction(tx *solana.Transaction, wireSize int) (ForgeResult, costmodel.ExceedReason) { return b.forgeTransaction(tx, wireSize, nil) } @@ -203,8 +204,15 @@ func (b *WorkingBank) forgeTransaction(tx *solana.Transaction, wireSize int, pre b.RebateSchedule(wireSize) return ForgeDroppedParse, costmodel.ExceedNone } + // Validate the full wire before execution can charge fees or publish + // account changes. Entry append then reuses this measured size and has no + // fallible serialization step after commit, including on the prepared path. + serializedSize, err := serializedTransactionSize(tx) + if err != nil { + b.RebateSchedule(wireSize) + return ForgeDroppedParse, costmodel.ExceedNone + } var messageHash [32]byte - var err error if prepared != nil { messageHash = prepared.MessageHash() } else { @@ -307,7 +315,7 @@ func (b *WorkingBank) forgeTransaction(tx *solana.Transaction, wireSize int, pre b.costs.Record(cost) execCU, loadedCost := actualExecutionUsage(output) b.costs.Rebate(cost, execCU, loadedCost) - if flushed, batchBytes, didFlush := b.entries.Append(*tx, wireSize); didFlush { + if flushed, batchBytes, didFlush := b.entries.appendSerialized(*tx, wireSize, serializedSize); didFlush { b.entryHash = b.entries.CurrentEntryHash() b.sink.OnEntryBatch(flushed, batchBytes) } diff --git a/pkg/blockprod/bank_test.go b/pkg/blockprod/bank_test.go index 98b9f7d13..13e13029f 100644 --- a/pkg/blockprod/bank_test.go +++ b/pkg/blockprod/bank_test.go @@ -819,3 +819,74 @@ func TestControllerWorkingBank(t *testing.T) { controller.SetWorkingBank(env.Bank) assert.Equal(t, env.Bank, controller.WorkingBank()) } + +func TestWorkingBankSerializationFailureDoesNotCommit(t *testing.T) { + for _, mode := range []string{"ordinary", "prepared"} { + t.Run(mode, func(t *testing.T) { + sink := &captureSink{} + env := NewTestEnv(TestEnvConfig{Sink: sink}) + defer env.Close() + env.SlotCtx.Features.EnableFeature(features.EnableTxV1, 0) + env.Bank.preparer = replay.NewTransactionPreparer(env.SlotCtx.Features) + tx := mustSignedTransfer(t, 1) + _, err := tx.Message.SetVersion(solana.MessageVersionV1) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + prepared := env.Bank.preparer.Prepare(tx) + if mode == "prepared" { + require.NotNil(t, prepared) + } + // Inject a malformed signature list after static preparation to + // ensure that even the prepared path cannot skip full-wire validation. + // The message is unchanged and serializable, but the full V1 wire + // cannot encode more signatures than the message header declares. + tx.Signatures = append(tx.Signatures, solana.Signature{1}) + _, err = replay.TransactionMessageHash(tx) + require.NoError(t, err) + _, err = tx.MarshalBinary() + require.ErrorContains(t, err, "signatures but header requires") + payer, err := env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + dest, err := env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + payerBalance, destBalance := payer.Lamports, dest.Lamports + hash := env.Bank.EntryHash() + require.Equal(t, costmodel.ExceedNone, env.Bank.PrepareSchedule(len(wire))) + require.Positive(t, env.Bank.EntryBuilder().ReservedBytes()) + var result ForgeResult + var reason costmodel.ExceedReason + if mode == "prepared" { + result, reason = env.Bank.ForgePreparedTransaction(tx, len(wire), prepared) + } else { + result, reason = env.Bank.ForgeTransaction(tx, len(wire)) + } + require.Equal(t, ForgeDroppedParse, result) + require.Equal(t, costmodel.ExceedNone, reason) + payer, err = env.SlotCtx.GetAccount(txfixture.PayerPubkey()) + require.NoError(t, err) + dest, err = env.SlotCtx.GetAccount(txfixture.DestPubkey()) + require.NoError(t, err) + require.Equal(t, payerBalance, payer.Lamports) + require.Equal(t, destBalance, dest.Lamports) + require.Zero(t, env.Bank.TxFeeAccumulator().TotalFees) + require.Zero(t, env.Bank.CostTracker().BlockCost()) + require.Zero(t, env.Bank.NumSignatures()) + require.Empty(t, env.Bank.ForgedTransactions()) + require.Empty(t, env.Bank.seenMessages) + require.Empty(t, env.SlotCtx.ModifiedAccts) + require.Zero(t, env.Bank.EntryBuilder().PendingCount()) + require.Zero(t, env.Bank.EntryBuilder().ReservedBytes()) + require.Zero(t, env.Bank.EntryBytes()) + require.Equal(t, hash, env.Bank.EntryHash()) + require.Empty(t, sink.batches) + // The same message remains eligible after correcting the wire. + tx.Signatures = tx.Signatures[:1] + result, reason = env.Bank.ForgeTransaction(tx, len(wire)) + require.Equal(t, ForgeAccepted, result) + require.Equal(t, costmodel.ExceedNone, reason) + require.Zero(t, env.Bank.EntryBuilder().ReservedBytes()) + require.Equal(t, 1, env.Bank.EntryBuilder().PendingCount()) + }) + } +} diff --git a/pkg/blockprod/entry.go b/pkg/blockprod/entry.go index 95e51d600..75ad95a19 100644 --- a/pkg/blockprod/entry.go +++ b/pkg/blockprod/entry.go @@ -105,29 +105,44 @@ func (b *EntryBuilder) dropReservation() { // transaction would overflow the configured batch target. A short leftover is only emitted by // Flush (slot end / Freeze). Appended transactions must remain immutable. func (b *EntryBuilder) Append(tx solana.Transaction, wireSize int) ([]turbine.Entry, int, bool) { - // Canonical component bytes may differ from a transport-size hint. Measure - // once per transaction and reuse the count at flush, preserving slot budgets. + serializedSize, err := serializedTransactionSize(&tx) + if err != nil { + return nil, 0, false + } + return b.appendSerialized(tx, wireSize, serializedSize) +} + +// serializedTransactionSize validates the complete transaction, not just its +// message. In particular, v1 signature-count errors surface only here. +func serializedTransactionSize(tx *solana.Transaction) (int, error) { wire, err := tx.MarshalBinary() if err != nil { _ = statsd.Count(statsd.BlockProductionEntrySerializationErrors, 1, nil) - mlog.Log.Errorf("entry builder: cannot serialize applied transaction: %v", err) - return nil, 0, false + mlog.Log.Errorf("entry builder: cannot serialize transaction: %v", err) + return 0, err } + return len(wire), nil +} + +// appendSerialized cannot fail after bank state is applied: the caller has +// already serialized this exact transaction and must keep it immutable. Keep +// canonical component size separate from the transport reservation-size hint. +func (b *EntryBuilder) appendSerialized(tx solana.Transaction, wireSize, serializedSize int) ([]turbine.Entry, int, bool) { if wireSize <= 0 { - wireSize = len(wire) + wireSize = serializedSize } b.consumeReserved(wireSize) - if b.wouldOverflowBatch(len(wire)) { + if b.wouldOverflowBatch(serializedSize) { flushed, batchBytes := b.flushLocked() b.pendingTxns = append(b.pendingTxns[:0], tx) - b.pendingSerializedBytes = len(wire) + b.pendingSerializedBytes = serializedSize b.pendingWire = wireSize return flushed, batchBytes, true } b.pendingTxns = append(b.pendingTxns, tx) - b.pendingSerializedBytes += len(wire) + b.pendingSerializedBytes += serializedSize b.pendingWire += wireSize return nil, 0, false } From 8082fec48dc668938198dcd40a31e8fe84a1fdb0 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:13:08 -0500 Subject: [PATCH 067/111] turbine: bound stream notification lifetimes and retry prefetch --- pkg/replay/streaming.go | 5 ++ pkg/replay/streaming_realfeed_test.go | 3 + pkg/turbine/assembler.go | 5 +- pkg/turbine/entry_prefetch.go | 36 +++++++++- pkg/turbine/entry_prefetch_bounds_test.go | 85 +++++++++++++++++++++++ pkg/turbine/stream.go | 39 +++++++++-- pkg/turbine/stream_test.go | 56 +++++++++++++++ 7 files changed, 220 insertions(+), 9 deletions(-) diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index a6280af8c..a618e451e 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -288,6 +288,11 @@ func (s *streamingExecutor) handleEvent(event turbine.StreamEvent) { if s == nil { return } + var live bool + event, live = event.Resolve() + if !live { + return + } switch event.Kind { case turbine.StreamBatchReady: if event.Batch == nil { diff --git a/pkg/replay/streaming_realfeed_test.go b/pkg/replay/streaming_realfeed_test.go index 5210f5da9..2c86f7b60 100644 --- a/pkg/replay/streaming_realfeed_test.go +++ b/pkg/replay/streaming_realfeed_test.go @@ -407,6 +407,9 @@ func TestStreamingRealFeedRecoversDroppedWakeups(t *testing.T) { default: t.Fatal("the surviving wake-up is not queued") } + var live bool + survivor, live = survivor.Resolve() + require.True(t, live) require.Zero(t, len(rig.feed.events), "nothing else was published") require.Equal(t, turbine.StreamBatchReady, survivor.Kind) require.Equal(t, uint64(realFeedSlot), survivor.Slot) diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index b324cafaf..ede8842ed 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -146,6 +146,8 @@ func (s *slotState) noteError(err error) { } type slotCompletionWork struct { + // Captured under mu: cancelled prefetch readers are not completion inputs. + ignorePrefetch bool state *slotState queuedAt time.Time observeCollection bool @@ -405,6 +407,7 @@ func (a *SlotAssembler) claimCompletionLocked(state *slotState, reportNonCanonic } return &slotCompletionWork{ state: state, + ignorePrefetch: state.prefetch != nil && state.prefetch.released, queuedAt: now, observeCollection: observeCollection, reportNonCanonical: reportNonCanonical, @@ -450,7 +453,7 @@ func (a *SlotAssembler) processCompletion(ctx context.Context, work *slotComplet decodeStartedAt := time.Now() decodeTimings := entryDecodeTimings{ctx: ctx} - if work.state.prefetch != nil { + if work.state.prefetch != nil && !work.ignorePrefetch { decodeTimings.prefetched = work.state.prefetch.batches } blk, parentInfo, roots, err := work.state.decodeBlock(&decodeTimings) diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go index b055ae594..71ffa1469 100644 --- a/pkg/turbine/entry_prefetch.go +++ b/pkg/turbine/entry_prefetch.go @@ -3,6 +3,7 @@ package turbine import ( "context" "errors" + "sort" "sync" "time" @@ -51,6 +52,7 @@ type slotEntryPrefetch struct { queued, released bool queueDone chan struct{} // closed after the queued/running token retires bytes int + budgetBlocked bool } // All scheduling and accounting use assembler.mu. The packet reader only @@ -64,6 +66,7 @@ type entryPrefetchPool struct { jobs chan *slotState workers, cleanup sync.WaitGroup slots, bytes int + active map[*slotState]struct{} // at most entryPrefetchSlots admitted generations closed bool close sync.Once } @@ -96,6 +99,10 @@ func (a *SlotAssembler) prefetchEntriesLocked(s *slotState) { ctx, cancel := context.WithCancel(withEntryPipelineTrace(p.ctx, s.pipelineTrace)) s.prefetch = &slotEntryPrefetch{pool: p, ctx: ctx, cancel: cancel, batches: make(map[uint32]*prefetchedShredBatch)} p.slots++ + if p.active == nil { + p.active = make(map[*slotState]struct{}) + } + p.active[s] = struct{}{} } p.enqueueLocked(s) } @@ -124,6 +131,7 @@ func (p *entryPrefetchPool) run() { p.a.mu.Unlock() continue } + f.budgetBlocked = false var batch *prefetchedShredBatch var shreds []*Shred var rawSize int @@ -137,10 +145,14 @@ func (p *entryPrefetchPool) run() { } } if size > entryPrefetchBatchBytes { - f.next++ - continue + // Never publish a prefix with an unfillable hole. Completion + // still decodes and verifies the entire valid block normally. + s.streamCancelReason = "prefetch_batch_too_large" + p.a.releasePrefetchLocked(s) + break } if p.bytes+size > entryPrefetchBytes { + f.budgetBlocked = true break } f.next++ @@ -244,10 +256,30 @@ func (a *SlotAssembler) releasePrefetchLocked(s *slotState) { p.a.mu.Lock() p.slots-- p.bytes -= f.bytes + delete(p.active, s) + p.retryBudgetBlockedLocked() p.a.mu.Unlock() }() } +// Retry only admitted generations, oldest slot first, when readers release bytes. +// This avoids both waiting for another shred and scanning all retained slots. +func (p *entryPrefetchPool) retryBudgetBlockedLocked() { + if p.closed || p.ctx.Err() != nil { + return + } + waiting := make([]*slotState, 0, len(p.active)) + for s := range p.active { + if s.prefetch.budgetBlocked && p.a.slots[s.slot] == s { + waiting = append(waiting, s) + } + } + sort.Slice(waiting, func(i, j int) bool { return waiting[i].slot < waiting[j].slot }) + for _, s := range waiting { + p.enqueueLocked(s) + } +} + func (p *entryPrefetchPool) closeAndWait() { p.close.Do(func() { p.cancel() diff --git a/pkg/turbine/entry_prefetch_bounds_test.go b/pkg/turbine/entry_prefetch_bounds_test.go index 189ff25c3..f7c9aef4a 100644 --- a/pkg/turbine/entry_prefetch_bounds_test.go +++ b/pkg/turbine/entry_prefetch_bounds_test.go @@ -2,6 +2,7 @@ package turbine import ( "context" + "fmt" "testing" "time" @@ -38,6 +39,15 @@ func TestEntryPrefetchByteBoundsFallBackToCompleteVerification(t *testing.T) { f := a.slots[slot].prefetch return f != nil && !f.queued && len(f.batches) == 0 }, 3*time.Second, time.Millisecond) + if mode == "oversized_component" { + a.mu.Lock() + state := a.slots[slot] + reason, released := state.streamCancelReason, state.prefetch.released + a.mu.Unlock() + require.Equal(t, "prefetch_batch_too_large", reason) + require.True(t, released) + require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: slot, state: state})) + } blk := feedPrefetchShreds(t, a, batches[1]) require.NotNil(t, blk) require.Len(t, blk.Transactions, 3) @@ -48,3 +58,78 @@ func TestEntryPrefetchByteBoundsFallBackToCompleteVerification(t *testing.T) { }) } } + +// A closed range needs no additional shred to become eligible after another +// generation releases its reservation. Cancellation must not revive stale work. +func TestEntryPrefetchRetriesByteBudgetOnRelease(t *testing.T) { + for _, cancelWaiting := range []bool{false, true} { + t.Run(fmt.Sprint(cancelWaiting), func(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + holderCtx, cancel := context.WithCancel(context.Background()) + holder := &slotState{slot: 399, prefetch: &slotEntryPrefetch{pool: p, ctx: holderCtx, cancel: cancel, bytes: entryPrefetchBytes}} + a.mu.Lock() + a.slots[399] = holder + p.slots = 1 + p.bytes = entryPrefetchBytes + p.active = map[*slotState]struct{}{holder: {}} + a.mu.Unlock() + batches := prefetchTestShreds(t, 400, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + f := a.slots[400].prefetch + return f != nil && f.budgetBlocked && !f.queued + }, 3*time.Second, time.Millisecond) + if cancelWaiting { + a.ResetSlot(400) + } + a.ResetSlot(399) + if cancelWaiting { + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return p.slots == 0 && p.bytes == 0 && len(p.active) == 0 + }, 3*time.Second, time.Millisecond) + } else { + batch := waitPrefetchedBatch(t, a, 400, 0) + _, err := batch.verification.wait() + require.NoError(t, err) + require.Len(t, entryBatchTransactions(batch.entries), 3) + } + }) + } +} + +func TestEntryPrefetchOversizedRangeAfterVerifiedPrefix(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + raw := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + large := prefetchTestPayload(t, verifierSignedTransactions(t, 3)) + large = append(large, make([]byte, entryPrefetchBatchBytes+1-len(large))...) + batches := prefetchTestShreds(t, 401, raw, large, buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + prefix := waitPrefetchedBatch(t, a, 401, 0) + _, err := prefix.verification.wait() + require.NoError(t, err) + require.Nil(t, feedPrefetchShreds(t, a, batches[1])) + require.Eventually(t, func() bool { + a.mu.Lock() + defer a.mu.Unlock() + return a.slots[401].prefetch.released + }, 3*time.Second, time.Millisecond) + blk := feedPrefetchShreds(t, a, batches[2]) + require.NotNil(t, blk) + require.Len(t, blk.Transactions, 6) + require.True(t, blk.TransactionSignaturesVerified()) + timings, ok := blk.TurbineIngressTimings() + require.True(t, ok) + require.Zero(t, timings.EarlyVerifiedTransactions, "cancelled prefix cache is not reused") +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index fe4af4213..5b4f7049e 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -5,6 +5,7 @@ import ( "errors" "sort" "time" + "weak" "github.com/Overclock-Validator/mithril/pkg/txverify" "github.com/gagliardetto/solana-go" @@ -153,13 +154,40 @@ const ( StreamCompleted ) -// StreamEvent is one feed wake-up. +// StreamEvent is an advisory wake-up. Call Resolve before inspecting Generation +// or Batch. Queued notifications hold only weak references, so a stalled +// subscriber cannot retain retired slot buffers outside the prefetch budget. type StreamEvent struct { Kind StreamEventKind Slot uint64 Generation StreamGeneration Batch *StreamBatch Reason string + state weak.Pointer[slotState] + batch weak.Pointer[prefetchedShredBatch] +} + +// Resolve acquires ownership of a still-live notification. A false result means +// the opportunity has expired; whole-block replay remains authoritative. Resolved +// generations retain their state for polling, including after completion. Detached +// events supplied by test feeds are already resolved. +func (e StreamEvent) Resolve() (StreamEvent, bool) { + if !e.Generation.IsZero() { + return e, true + } + state := e.state.Value() + if state == nil { + return StreamEvent{}, false + } + e.Generation = StreamGeneration{slot: e.Slot, state: state} + if e.Kind == StreamBatchReady { + batch := e.batch.Value() + if batch == nil { + return StreamEvent{}, false + } + e.Batch = newStreamBatch(e.Generation, batch) + } + return e, true } // StreamStatus is the assembler's view of a generation. @@ -308,8 +336,7 @@ func (a *SlotAssembler) publishStreamBatchReadyLocked(s *slotState, batch *prefe return } a.noteChildRepairHeaderLocked(s, batch) - g := StreamGeneration{slot: s.slot, state: s} - a.publishStreamLocked(StreamEvent{Kind: StreamBatchReady, Slot: s.slot, Generation: g, Batch: newStreamBatch(g, batch)}) + a.publishStreamLocked(StreamEvent{Kind: StreamBatchReady, Slot: s.slot, state: weak.Make(s), batch: weak.Make(batch)}) } // publishStreamReleaseLocked is called from releasePrefetchLocked, i.e. from @@ -319,10 +346,10 @@ func (a *SlotAssembler) publishStreamReleaseLocked(s *slotState, reason string) if a.streamSubscriber == nil || s == nil { return } - g := StreamGeneration{slot: s.slot, state: s} + state := weak.Make(s) if s.streamCompleted { - a.publishStreamLocked(StreamEvent{Kind: StreamCompleted, Slot: s.slot, Generation: g}) + a.publishStreamLocked(StreamEvent{Kind: StreamCompleted, Slot: s.slot, state: state}) return } - a.publishStreamLocked(StreamEvent{Kind: StreamCancelled, Slot: s.slot, Generation: g, Reason: reason}) + a.publishStreamLocked(StreamEvent{Kind: StreamCancelled, Slot: s.slot, state: state, Reason: reason}) } diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index a9c08edce..0b7acb543 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -2,8 +2,10 @@ package turbine import ( "context" + "runtime" "testing" "time" + "weak" "github.com/Overclock-Validator/mithril/pkg/block" "github.com/gagliardetto/solana-go" @@ -16,6 +18,10 @@ func nextStreamEvent(t *testing.T, ch <-chan StreamEvent, kind StreamEventKind) for { select { case event := <-ch: + event, live := event.Resolve() + if !live { + continue + } if event.Kind == kind { return event } @@ -187,3 +193,53 @@ func TestStreamFeedDropsWakeupsWhenSubscriberIsFull(t *testing.T) { require.Len(t, pending, 1) require.Len(t, pending[0].Transactions, 2) } + +// Notifications must not own retired slots or decoded payloads. Conversely, +// resolving a live event gives the consumer a strong polling handle. +func TestStreamQueuedEventsDoNotRetainRetiredBuffers(t *testing.T) { + events := make(chan StreamEvent, 4) + publish := func() (weak.Pointer[slotState], weak.Pointer[prefetchedShredBatch]) { + a := NewSlotAssembler() + a.SubscribeStream(events) + batch := &prefetchedShredBatch{raw: make([]byte, 1<<20), marker: true} + state := &slotState{slot: 42, prefetch: &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: batch}}} + a.mu.Lock() + a.publishStreamBatchReadyLocked(state, batch) + a.publishStreamReleaseLocked(state, "reset") + a.mu.Unlock() + return weak.Make(state), weak.Make(batch) + } + state, batch := publish() + require.Eventually(t, func() bool { + runtime.GC() + return state.Value() == nil && batch.Value() == nil + }, 3*time.Second, time.Millisecond) + require.Len(t, events, 2) + for len(events) > 0 { + _, live := (<-events).Resolve() + require.False(t, live) + } +} + +func TestStreamResolvedEventRetainsCompletedPollingState(t *testing.T) { + a := NewSlotAssembler() + events := make(chan StreamEvent, 2) + a.SubscribeStream(events) + ready := make(chan struct{}) + close(ready) + batch := &prefetchedShredBatch{ready: ready, marker: true} + state := &slotState{slot: 42, streamCompleted: true, prefetch: &slotEntryPrefetch{batches: map[uint32]*prefetchedShredBatch{0: batch}, released: true}} + a.mu.Lock() + a.publishStreamBatchReadyLocked(state, batch) + a.publishStreamReleaseLocked(state, "") + a.mu.Unlock() + event, live := (<-events).Resolve() + require.True(t, live) + runtime.GC() + require.Equal(t, StreamDone, a.StreamStatusOf(event.Generation)) + require.Len(t, a.PendingStreamBatches(event.Generation, 0), 1) + done, live := (<-events).Resolve() + require.True(t, live) + require.Equal(t, event.Generation, done.Generation) + require.Equal(t, StreamCompleted, done.Kind) +} From 022fb833e7cc8832afa86664d4bb41dfe327327f Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:29:23 -0500 Subject: [PATCH 068/111] sbpf: guard pooled write tracking beyond bitmap capacity --- pkg/sbpf/fastmem.go | 3 +++ pkg/sbpf/interpreter.go | 9 ++++++++ pkg/sbpf/pooling_test.go | 46 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+) diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go index 421f3ed42..1f15dab3b 100644 --- a/pkg/sbpf/fastmem.go +++ b/pkg/sbpf/fastmem.go @@ -25,6 +25,9 @@ type memRegion struct { // emptyRegion never matches any access. var emptyRegion = memRegion{gapShift: 63} +// A uint64 dirty bitmap can describe exactly 64 pages of 4 KiB. +const fastDirtyBytes = 64 * 4096 + const numFastRegions = 6 // index 5 is a permanently empty catch-all // fastRead returns a host pointer for a size-byte read at vma, or nil if the diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e27392d7c..96d3cb25b 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -162,6 +162,15 @@ func (ip *Interpreter) initRegions() { if len(ip.heap) != 0 { ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} } + // Finish relies on complete write tracking before returning pooled storage. + // Larger heaps (or a future larger stack) must use translateInternal's byte + // ranges: shifting the fast-path bitmap beyond page 63 silently loses writes. + // Reads remain fast; current <=256 KiB writable mappings are unchanged. + for _, idx := range []uint64{VaddrStack >> 32, VaddrHeap >> 32} { + if ip.regions[idx].wlen > fastDirtyBytes { + ip.regions[idx].wlen = 0 + } + } if len(ip.inputRegions) == 0 && len(ip.input) != 0 { ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index c4bc5eb0d..4a922a70f 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -2,10 +2,13 @@ package sbpf import ( "bytes" + "encoding/binary" + "fmt" "sync" "testing" "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" "github.com/stretchr/testify/require" ) @@ -87,3 +90,46 @@ func BenchmarkVMCreateAndFinish(b *testing.B) { }) } } + +// Exercise actual stores at and beyond the bitmap boundary. Inspect the returned +// buffer directly: sync.Pool is permitted to discard entries, so a subsequent Get +// alone would not reliably detect a missed clear. +func TestPooledHeapDirtyBitmapBoundary(t *testing.T) { + oldUsePool, oldPool := UsePool, heapPool + UsePool = true + heapPool = &sync.Pool{New: func() any { return newHeap() }} + t.Cleanup(func() { UsePool, heapPool = oldUsePool, oldPool }) + for _, size := range []int{fastDirtyBytes, fastDirtyBytes + 1, 2 * fastDirtyBytes} { + for _, ver := range []uint32{sbpfver.SbpfVersionV0, sbpfver.SbpfVersionV2, sbpfver.SbpfVersionV3} { + t.Run(fmt.Sprintf("%d/v%d", size, ver), func(t *testing.T) { + offsets := []int{0, size - 8} + if size >= fastDirtyBytes+8 { + offsets = append(offsets, fastDirtyBytes-4, fastDirtyBytes) + } + var text []Slot + op := uint8(OpStdw) + if ver == sbpfver.SbpfVersionV2 { + op = OpSt8BImm + } + for _, off := range offsets { + text = append(text, diffLoadImm64(5, VaddrHeap+uint64(off), ver)...) + text = append(text, slot(op, 5, 0, 0, 0x12345678)) + } + text = append(text, slot(OpExit, 0, 0, 0, 0)) + program := mkProgram(text, ver) + require.NoError(t, program.Verify()) + meter := cu.NewComputeMeter(100) + ip := NewInterpreter(program, &VMOpts{HeapMax: size, ComputeMeter: &meter, Syscalls: noSyscalls}) + // Always clear test storage on failure so later cases cannot inherit dirt. + defer func() { clear(ip.heap) }() + require.NotNil(t, ip.fastRead(VaddrHeap+uint64(size-8), 8)) + _, _, err := ip.Run() + require.NoError(t, err) + require.Equal(t, uint64(0x12345678), binary.LittleEndian.Uint64(ip.heap[offsets[len(offsets)-1]:])) + ip.Finish() + require.True(t, bytes.Equal(make([]byte, size), ip.heap), "Finish must clear every written byte before pooling") + require.Equal(t, size <= fastDirtyBytes, ip.regions[VaddrHeap>>32].wlen != 0) + }) + } + } +} From be007c3a6e916ae76c9203451caf42e9849850b6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:43:23 -0500 Subject: [PATCH 069/111] replay: bound speculative signature verification waits --- docs/transaction_sigverify_streaming.md | 20 +++++++ pkg/metrics/metrics.go | 4 ++ pkg/replay/streaming.go | 75 ++++++++++++++++++------- pkg/replay/streaming_lifecycle_test.go | 74 ++++++++++++++---------- pkg/replay/streaming_test.go | 65 +++++++++++++++++++++ pkg/turbine/stream.go | 21 ++++++- pkg/turbine/stream_test.go | 28 +++++++++ 7 files changed, 235 insertions(+), 52 deletions(-) diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index 19f868d7f..cfd39e6cd 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -271,3 +271,23 @@ and a reserved token for continued highest-index discovery when needed. A snapshot can still race with subsequent arrivals; this removes known redundant requests, not every possible duplicate. No assembler lock is held while signing or sending requests. Response matching and peer credit are unchanged. + + +### Bounded speculative verification waits + +Replay joins a streaming group's signature verification with one shared 100 ms +budget, capped by the stream's remaining open lifetime. The watchdog stage is +`streaming_sigverify_wait`. Expiry discards the speculative overlay with reason +`sigverify_timeout`; it is not a signature verdict. Whole-block replay still +requires normal verification before accepting the block. + +The streaming wait only observes immutable verifier results. Timing out does not +cancel the shared request or wait for its workers: turbine continues owning its +transactions and retains reservations until readers finish. The owning completion +and cleanup paths retain their joining waits. This bounds speculative replay's +wait, not the duration of whole-block verification or recovery from a failed worker. + +`StreamingExecution.VerificationWait` records wall time spent joining groups, +including failed joins and discarded streams. It overlaps `GroupJoinAssembly` for +successful groups; do not add them together or interpret it as cryptographic CPU +cost. The 100 ms limit is a conservative fallback budget, not a measured optimum. diff --git a/pkg/metrics/metrics.go b/pkg/metrics/metrics.go index 8b3903eb6..fd807da0b 100644 --- a/pkg/metrics/metrics.go +++ b/pkg/metrics/metrics.go @@ -255,6 +255,10 @@ type VoteRewardDetails struct { // this slot was thrown away and the block was executed whole; DiscardReason // names why. type StreamingExecution struct { + // VerificationWait is wall time joining speculative verification groups, + // including failed joins. It overlaps GroupJoinAssembly for successful + // groups; it is neither crypto CPU time nor additional replay latency. + VerificationWait Timing Opened uint64 Groups uint64 Transactions uint64 diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index a618e451e..addac7bf3 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -60,6 +60,9 @@ const ( defaultStreamingWorkers = 4 defaultStreamingMaxAge = 2 * time.Second streamingPollInterval = 5 * time.Millisecond + // Bound the entire group's verification join, not each batch separately. + // This is a speculative-work budget, not a signature validity deadline. + streamingVerificationWait = 100 * time.Millisecond // Bound speculation across missing leaders; this never advances replay // or establishes that the intervening slots are actually skipped. streamingMaxSlotDistance = uint64(32) @@ -158,14 +161,15 @@ type streamingFrontierMark struct { // that a block executed whole can report why no stream opened for it, and a // stream can report how long its header waited and on what. type streamingObservation struct { - generation turbine.StreamGeneration - parentSlot uint64 - readyAt time.Time // header batch decoded (its wake-up's ReadyAt) - seenAt time.Time // executor first handled the header - frontierAtSeen uint64 - declined string // eligibility reason, when the header was declined - discarded string // discard reason, when a stream opened and was thrown away - openedAt time.Time + verificationWait metrics.Timing + generation turbine.StreamGeneration + parentSlot uint64 + readyAt time.Time // header batch decoded (its wake-up's ReadyAt) + seenAt time.Time // executor first handled the header + frontierAtSeen uint64 + declined string // eligibility reason, when the header was declined + discarded string // discard reason, when a stream opened and was thrown away + openedAt time.Time } // streamingSlot is one in-progress stream. @@ -178,14 +182,15 @@ type streamingSlot struct { // origin holds the block's own transaction objects in executed order; // the bank executes stream-owned copies (block.ExecutionCopies), and the // handshake proves the block by these originals. - origin []*solana.Transaction - nextStart uint32 - pending map[uint32]*turbine.StreamBatch - footerSeen bool - completed bool - openedAt time.Time - headerAt time.Time - groups []streamingGroup + origin []*solana.Transaction + nextStart uint32 + pending map[uint32]*turbine.StreamBatch + footerSeen bool + completed bool + openedAt time.Time + headerAt time.Time + groups []streamingGroup + verificationWait metrics.Timing // timeline: what bounded the open (see metrics.StreamingExecution). Kept // here rather than in the collector because the loop resets the collector // before every wait and the block may arrive several waits after the open. @@ -215,8 +220,9 @@ type streamingExecutor struct { // pruned with headers. observed map[uint64]*streamingObservation ticker *time.Ticker - // executeFn runs one group on the open execution; tests substitute it. - executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error + // Verification observation and group execution hooks; tests substitute them. + waitVerificationFn func(context.Context, *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) + executeFn func(exec *blockExecution, txs []*solana.Transaction, identities *b.PreparedTransactionMessageIdentities, shouldVerifySignatures bool) error } func newStreamingExecutor(deps streamingDeps) *streamingExecutor { @@ -226,6 +232,9 @@ func newStreamingExecutor(deps streamingDeps) *streamingExecutor { retired: make(map[uint64]turbine.StreamGeneration), observed: make(map[uint64]*streamingObservation), executeFn: (*blockExecution).executeTransactionGroup, + waitVerificationFn: func(ctx context.Context, batch *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + return batch.WaitVerification(ctx) + }, } } @@ -694,12 +703,36 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { var verified []txverify.VerifiedMessageIdentity allVerified := true readyAt := time.Now() + deadline := readyAt.Add(streamingVerificationWait) + maxAge := StreamingExecutionCfg.maxOpenAge() + if cur.completed { + maxAge *= streamingHardOpenAgeFactor + } + if ageDeadline := cur.openedAt.Add(maxAge); ageDeadline.Before(deadline) { + deadline = ageDeadline + } + ctx, cancel := context.WithDeadline(context.Background(), deadline) + defer cancel() + cur.exec.setReplayStage("streaming_sigverify_wait") + // Keep failure timings across replay-loop metric resets, just like headers. + joinStarted := time.Now() + recordJoin := func() { + cur.verificationWait.AddTiming(time.Since(joinStarted)) + if obs := s.observed[cur.slot]; obs != nil && obs.generation == cur.generation { + obs.verificationWait = cur.verificationWait + } + cur.exec.setReplayStage("streaming_wait") + } for _, batch := range group { - identities, ok, err := batch.WaitVerification(context.Background()) + identities, ok, err := s.waitVerificationFn(ctx, batch) if errors.Is(err, turbine.ErrStreamBatchUnverified) { ok, err = false, nil } if err != nil { + recordJoin() + if errors.Is(err, context.DeadlineExceeded) { + return errors.New("sigverify_timeout") + } return fmt.Errorf("sigverify: %w", err) } if !ok { @@ -708,6 +741,7 @@ func (s *streamingExecutor) executeGroup(group []*turbine.StreamBatch) error { txs = append(txs, batch.Transactions...) verified = append(verified, identities...) } + recordJoin() joinedAt := time.Now() if len(txs) == 0 { return nil @@ -790,6 +824,7 @@ func (s *streamingExecutor) discard(reason string) { if cur.restoreSysvarCache != nil { cur.restoreSysvarCache() } + metrics.GlobalBlockReplay.StreamingExecution.VerificationWait = cur.verificationWait metrics.GlobalBlockReplay.StreamingExecution.Discarded = 1 metrics.GlobalBlockReplay.StreamingExecution.DiscardReason = reason mlog.Log.FileOnlyf("streaming: discarded slot %d after %d groups (%s)", cur.slot, len(cur.groups), reason) @@ -936,6 +971,7 @@ func (s *streamingExecutor) finalize(block *b.Block, parentBankSysvars *sealevel record := &metrics.GlobalBlockReplay.StreamingExecution discarded, discardReason := record.Discarded, record.DiscardReason *record = metrics.StreamingExecution{Opened: 1, Discarded: discarded, DiscardReason: discardReason} + record.VerificationWait = cur.verificationWait record.Groups = uint64(len(cur.groups)) for _, group := range cur.groups { record.Transactions += uint64(group.transactions) @@ -1187,6 +1223,7 @@ func (s *streamingExecutor) noteWholeBlock(block *b.Block) { default: record.NotOpenedReason = fmt.Sprintf("waiting_for_parent:header_on_parent_%d_seen_at_frontier_%d", obs.parentSlot, obs.frontierAtSeen) } + record.VerificationWait = obs.verificationWait record.HeaderReadyNanos = nanosOf(obs.readyAt) record.HeaderSeenNanos = nanosOf(obs.seenAt) record.OpenedNanos = nanosOf(obs.openedAt) diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go index bbd4b747a..63f466283 100644 --- a/pkg/replay/streaming_lifecycle_test.go +++ b/pkg/replay/streaming_lifecycle_test.go @@ -19,6 +19,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" "github.com/Overclock-Validator/mithril/pkg/turbine" + "github.com/Overclock-Validator/mithril/pkg/txverify" bin "github.com/gagliardetto/binary" "github.com/gagliardetto/solana-go" "github.com/stretchr/testify/require" @@ -355,37 +356,50 @@ func TestStreamingLifecycleMatchesWholeBlock(t *testing.T) { func TestStreamingLifecycleDiscardThenWholeBlock(t *testing.T) { StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} defer func() { StreamingExecutionCfg = StreamingExecutionConfig{} }() - txs := transferTransactions(t, 8, 999_000) - env := newLifecycleEnv(t) - whole := lifecycleWholeBlock(t, env, txs, 2) + for _, reason := range []string{"update_parent", "sigverify_timeout"} { + t.Run(reason, func(t *testing.T) { + txs := transferTransactions(t, 8, 999_000) + env := newLifecycleEnv(t) + whole := lifecycleWholeBlock(t, env, txs, 2) + + tail := &lifecycleTail{durable: env.durable} + statuses := NewTransactionStatusCache() + shell := env.block(nil) + exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, env.parent) + require.NoError(t, exec.open()) + feed := newFakeStreamFeed() + gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) + feed.status[gen] = turbine.StreamActive + s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: env.epochSchedule, tail: tail, transactionStatuses: statuses}) + s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, + exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, + Batch: turbine.NewDetachedStreamBatch(gen, 1, 5, txs[:5], verifiedIdentities(t, txs[:5]))}) + sameTransactions(t, txs[:5], s.current.origin) + sameCopies(t, txs[:5], exec.transactions) + payerNow, err := exec.slotCtx.GetAccountShared(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Less(t, payerNow.Lamports, uint64(3_200_000), "the overlay saw the executed prefix") + + if reason == "sigverify_timeout" { + s.waitVerificationFn = func(context.Context, *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + return nil, false, context.DeadlineExceeded + } + s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, Batch: turbine.NewDetachedStreamBatch(gen, 6, 8, txs[5:], verifiedIdentities(t, txs[5:]))}) + require.Nil(t, s.current) + } else { + s.discard(reason) + } + require.Empty(t, tail.added, "a discarded stream commits nothing") + durablePayer, err := env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) + require.NoError(t, err) + require.Equal(t, uint64(3_200_000), durablePayer.Lamports, "the durable view is untouched") + + again := lifecycleWholeBlock(t, env, txs, 2) + requireSameLifecycleOutcome(t, whole, again) - tail := &lifecycleTail{durable: env.durable} - statuses := NewTransactionStatusCache() - shell := env.block(nil) - exec := newBlockExecution(env.acctsDb, shell, env.epochSchedule, 2, nil, &persistedTracker{}, tail, statuses, false, env.parent) - require.NoError(t, exec.open()) - feed := newFakeStreamFeed() - gen := turbine.NewDetachedStreamGeneration(lifecycleSlot) - feed.status[gen] = turbine.StreamActive - s := newStreamingExecutor(streamingDeps{feed: feed, epochSchedule: env.epochSchedule, tail: tail, transactionStatuses: statuses}) - s.current = &streamingSlot{slot: lifecycleSlot, generation: gen, parentSlot: lifecycleParentSlot, parentID: lifecycleParentBlockID, - exec: exec, pending: make(map[uint32]*turbine.StreamBatch), openedAt: time.Now(), headerAt: time.Now(), nextStart: 1} - s.handleEvent(turbine.StreamEvent{Kind: turbine.StreamBatchReady, Slot: lifecycleSlot, Generation: gen, - Batch: turbine.NewDetachedStreamBatch(gen, 1, 5, txs[:5], verifiedIdentities(t, txs[:5]))}) - sameTransactions(t, txs[:5], s.current.origin) - sameCopies(t, txs[:5], exec.transactions) - payerNow, err := exec.slotCtx.GetAccountShared(txfixture.PayerPubkey()) - require.NoError(t, err) - require.Less(t, payerNow.Lamports, uint64(3_200_000), "the overlay saw the executed prefix") - - s.discard("update_parent") - require.Empty(t, tail.added, "a discarded stream commits nothing") - durablePayer, err := env.durable.GetAccountWithoutLock(txfixture.PayerPubkey()) - require.NoError(t, err) - require.Equal(t, uint64(3_200_000), durablePayer.Lamports, "the durable view is untouched") - - again := lifecycleWholeBlock(t, env, txs, 2) - requireSameLifecycleOutcome(t, whole, again) + }) + } } // V0 / address-lookup-table coverage. Resolving lookups mutates the message diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index 29837fc6e..abb2dfcd3 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -1095,3 +1095,68 @@ func TestStreamingEventRefreshesReadyBatches(t *testing.T) { }) } } + +func TestStreamingVerificationDeadlineAndDiscard(t *testing.T) { + old := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + defer func() { StreamingExecutionCfg = old }() + h := newFinalizeFailureHarness(t) + cur := h.exec.current + h.exec.observed[cur.slot] = &streamingObservation{generation: h.gen} + StreamingExecutionCfg.MinGroupBatches = 2 + var stage string + cur.exec.setReplayStage = func(value string) { stage = value } + var firstContext context.Context + calls := 0 + h.exec.waitVerificationFn = func(ctx context.Context, batch *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + require.Equal(t, "streaming_sigverify_wait", stage) + deadline, ok := ctx.Deadline() + require.True(t, ok) + require.LessOrEqual(t, time.Until(deadline), streamingVerificationWait) + calls++ + if calls == 1 { + firstContext = ctx + return batch.WaitVerification(ctx) + } + require.Equal(t, firstContext, ctx, "one deadline covers all batches") + return nil, false, context.DeadlineExceeded + } + txs := transferTransactions(t, 2, 98123) + h.exec.handleEvent(h.event(h.batch(t, 4, 4, txs[:1]))) + require.Zero(t, calls) + h.exec.handleEvent(h.event(h.batch(t, 5, 5, txs[1:]))) + require.Equal(t, 2, calls) + h.assertUndone(t, "sigverify_timeout") + require.Equal(t, "streaming_wait", stage) + sameTransactions(t, h.txs[:3], cur.origin) + sameCopies(t, h.txs[:3], cur.exec.transactions) + _, ok, err := h.exec.finalize(h.env.exec.block, h.env.exec.parentBankSysvars) + require.NoError(t, err) + require.False(t, ok, "whole-block replay must take over") + metrics.GlobalBlockReplay.StreamingExecution = metrics.StreamingExecution{} + h.exec.noteWholeBlock(h.env.exec.block) + record := metrics.GlobalBlockReplay.StreamingExecution + require.Equal(t, "discarded:sigverify_timeout", record.NotOpenedReason) + require.Equal(t, uint64(2), record.VerificationWait.Count) + require.Positive(t, record.VerificationWait.SumNanoseconds) +} + +func TestStreamingVerificationWaitRespectsRemainingOpenAge(t *testing.T) { + old := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true, MaxOpenAge: time.Second} + defer func() { StreamingExecutionCfg = old }() + h := newStreamingTestHarness(t) + cur := h.exec.current + cur.openedAt = time.Now().Add(-time.Second) + h.exec.waitVerificationFn = func(ctx context.Context, _ *turbine.StreamBatch) ([]txverify.VerifiedMessageIdentity, bool, error) { + deadline, ok := ctx.Deadline() + require.True(t, ok) + require.Equal(t, cur.openedAt.Add(time.Second), deadline) + require.ErrorIs(t, ctx.Err(), context.DeadlineExceeded) + return nil, false, ctx.Err() + } + h.exec.handleEvent(h.event(h.batch(t, 1, 1, transferTransactions(t, 1, 98124)))) + require.Nil(t, h.exec.current) + require.Empty(t, h.executed()) + require.Equal(t, "sigverify_timeout", h.discardReason()) +} diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index 5b4f7049e..a6a51b44a 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -118,12 +118,15 @@ type StreamBatch struct { // consumer must verify signatures itself. var ErrStreamBatchUnverified = errors.New("stream batch has no verification result") -// WaitVerification joins the batch's asynchronous signature verification and +// WaitVerification observes the batch's asynchronous signature verification and // returns the verifier's message identities, one per transaction, bound to // Transactions (see block.PrepareVerifiedTransactionMessageIdentities). A // nil error with verified == false means no result is attached and the // caller must verify itself; any other error means a signature failed (the -// slot is invalid) or ctx ended. +// slot is invalid) or ctx ended. Unlike the owning verifier wait, a context +// timeout returns without cancelling or joining the job: turbine retains the +// immutable transaction storage and joins readers before releasing reservations. +// A caller timing out must not mutate the batch or its transactions. func (sb *StreamBatch) WaitVerification(ctx context.Context) (identities []txverify.VerifiedMessageIdentity, verified bool, err error) { if sb == nil || sb.batch == nil { return nil, false, ErrStreamBatchUnverified @@ -131,9 +134,21 @@ func (sb *StreamBatch) WaitVerification(ctx context.Context) (identities []txver if sb.batch.verification == nil { return nil, false, nil } - if _, err := sb.batch.verification.waitContext(ctx); err != nil { + if ctx == nil { + ctx = context.Background() + } + future := sb.batch.verification + select { + case <-ctx.Done(): + return nil, false, ctx.Err() + case <-future.done: + } + if err := ctx.Err(); err != nil { return nil, false, err } + if future.err != nil { + return nil, false, future.err + } if len(sb.batch.verification.identities) != len(sb.Transactions) { return nil, false, nil } diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index 0b7acb543..50245fce6 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -3,6 +3,8 @@ package turbine import ( "context" "runtime" + "sync" + "sync/atomic" "testing" "time" "weak" @@ -243,3 +245,29 @@ func TestStreamResolvedEventRetainsCompletedPollingState(t *testing.T) { require.Equal(t, event.Generation, done.Generation) require.Equal(t, StreamCompleted, done.Kind) } + +// The streaming observer must return while the owning request is still running. +// Completion/cleanup retain the separate joining wait and own buffer lifetime. +func TestStreamVerificationTimeoutDoesNotCancelOrJoinOwner(t *testing.T) { + done := make(chan struct{}) + var closeOnce sync.Once + defer closeOnce.Do(func() { close(done) }) + var cancelled atomic.Bool + future := &transactionVerification{done: done, cancel: func() { cancelled.Store(true) }} + batch := &StreamBatch{batch: &prefetchedShredBatch{verification: future}} + ctx, cancel := context.WithTimeout(context.Background(), time.Millisecond) + defer cancel() + returned := make(chan error, 1) + go func() { _, _, err := batch.WaitVerification(ctx); returned <- err }() + select { + case err := <-returned: + require.ErrorIs(t, err, context.DeadlineExceeded) + case <-time.After(3 * time.Second): + t.Fatal("observer waited for unfinished owner") + } + require.False(t, cancelled.Load(), "observer must not cancel completion's shared work") + closeOnce.Do(func() { close(done) }) + _, verified, err := batch.WaitVerification(context.Background()) + require.NoError(t, err) + require.True(t, verified, "same request remains usable after the observer leaves") +} From 2678206e1a369c445bc60d2e40d922aa319ceee4 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:59:32 -0500 Subject: [PATCH 070/111] sealevel: match Agave zero-length memory copy behavior --- pkg/sealevel/syscalls_mem.go | 9 +++------ pkg/sealevel/syscalls_mem_test.go | 23 +++++++++++------------ 2 files changed, 14 insertions(+), 18 deletions(-) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 73f96fc67..14a1e39b0 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,13 +24,10 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { - // Translate intentionally bypasses address validation for zero-length slices; - // Read/Write did not. Preserve the old syscall validation and error order. + // Agave's touch_slice_mut / translate_slice return empty slices before + // address lookup for zero length. CU was already charged by the syscall. if n == 0 { - if err := vm.Read(src, nil); err != nil { - return err - } - return vm.Write(dst, nil) + return nil } srcMem, err := vm.Translate(src, n, false) if err != nil { diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 9fe7d22b7..d375a42b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -199,19 +199,18 @@ func TestMemsetBytes(t *testing.T) { } } -func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { +func TestSyscallMemoryZeroLengthSkipsAddressValidation(t *testing.T) { for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { - vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) - want := vm.Read(src, nil) - if want == nil { - want = vm.Write(dst, nil) - } - got := memmoveImplInternal(vm, dst, src, 0) - if want == nil { - require.NoError(t, got) - } else { - require.EqualError(t, got, want.Error()) + for _, fn := range []func(sbpf.VM, uint64, uint64, uint64) (uint64, error){SyscallMemmoveImpl, SyscallMemcpyImpl} { + input := bytes.Repeat([]byte{0x42}, 32) + vm, ctx := newMemSyscallVM(t, input, nil) + before := ctx.ComputeMeter.Remaining() + ret, err := fn(vm, dst, src, 0) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, before-cu.CUMemOpBaseCost, ctx.ComputeMeter.Remaining()) + require.Equal(t, bytes.Repeat([]byte{0x42}, 32), input) } } } @@ -231,7 +230,7 @@ func TestMemoryCopyDifferential(t *testing.T) { dst += sbpf.VaddrInput _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) wantErr := MemOpConsume(refCtx, n) - if wantErr == nil { + if wantErr == nil && n > 0 { buf := make([]byte, n) wantErr = ref.Read(src, buf) if wantErr == nil { From f526c1d4c6150ad730c6ba034cb2cacb774159db Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:59:32 -0500 Subject: [PATCH 071/111] review: close remaining resource, diagnostics and regression gaps --- .github/workflows/go_build.yml | 11 +- cmd/mithril/configcmd/configcmd.go | 6 + cmd/mithril/configcmd/configcmd_test.go | 4 + docs/leader_block_packing.md | 9 +- docs/transaction_sigverify_streaming.md | 9 + pkg/alpenglow/certpool.go | 2 +- pkg/alpenglow/peer_sender_test.go | 12 +- .../readonly_block_prepared_bench_test.go | 16 ++ pkg/consensus/voter.go | 3 +- pkg/costmodel/costmodel_test.go | 18 +- pkg/costmodel/entry_bytes.go | 4 +- pkg/costmodel/limits.go | 1 - pkg/replay/streaming.go | 8 +- pkg/replay/streaming_test.go | 18 ++ pkg/sbpf/interpreter.go | 12 +- pkg/sbpf/perf_differential_test.go | 21 +- pkg/sbpf/vasa_test.go | 13 ++ pkg/sealevel/bpf_loader.go | 4 + pkg/sealevel/sealevel_test.go | 188 +++++++++++------- pkg/turbine/assembler.go | 1 - pkg/turbine/entry_prefetch.go | 6 +- pkg/turbine/entry_prefetch_test.go | 18 +- pkg/turbine/shredspool.go | 4 +- pkg/turbine/stream.go | 2 +- pkg/turbine/stream_test.go | 19 ++ 25 files changed, 301 insertions(+), 108 deletions(-) diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index 4fa7306c9..bdcfcc42c 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -35,13 +35,16 @@ jobs: - name: Voting, checkpoint, streaming and scheduler race regressions # Run the complete affected package suites, including subprocess crash - # recovery and cancellation tests. The independent sealevel suite has - # known base-branch failures documented in the validation report. + # recovery and cancellation tests. The broader sealevel suite still has unrelated legacy + # loader-fixture failures; the interpreter/syscall subset runs below. run: >- go test -race -p 2 -count=1 ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... ./cmd/mithril/node ./cmd/mithril/configcmd - - name: Vote-program deque ownership race regression - run: go test -race -count=1 ./pkg/sealevel -run '^TestProcessNewVoteStateOwnsRetainedDeque$' + - name: Interpreter differential and memory race regressions + run: go test -race -count=1 ./pkg/sbpf/... + + - name: Sealevel interpreter, syscall and vote ownership regressions + run: go test -race -count=1 ./pkg/sealevel -run '^(TestInterpreter_(Noop|Mem|Sha256|Blake3|Keccak256|CreateProgramAddress|TryFindProgramAddress|TestPanic|Secp256k1)|TestSyscall|TestMemoryCopyDifferential$|TestProcessNewVoteStateOwnsRetainedDeque$)' diff --git a/cmd/mithril/configcmd/configcmd.go b/cmd/mithril/configcmd/configcmd.go index 3236b9fcc..54fa1952c 100644 --- a/cmd/mithril/configcmd/configcmd.go +++ b/cmd/mithril/configcmd/configcmd.go @@ -158,6 +158,9 @@ authorized_voter_keypair = "" # BLS derivation signer (empty defaults to id authorized_withdrawer_keypair = "" # Authorized withdrawer keypair path (diagnostics only) tpu_quic_bind_addr = "0.0.0.0:8004" advertised_ip = "" # Required only in validator mode; public IP advertised for TPU QUIC +wait_to_vote_slot = 0 # Minimum slot for new votes; does not bypass recovery checks +tpu_max_buffered_transactions = 0 # 0 = 131,072 buffered transactions +block_completion_reserve_ms = 0 # 0 = 75ms local completion/broadcast reserve tpu_sigverify_workers = 0 # 0 = GOMAXPROCS [consensus] @@ -175,6 +178,9 @@ authorized_voter_keypair = "" # Empty defaults to authorized_withdrawer_keypair = "" tpu_quic_bind_addr = "0.0.0.0:8004" advertised_ip = "" # REQUIRED: public IP advertised for TPU QUIC +wait_to_vote_slot = 0 # Minimum slot for new votes; does not bypass recovery checks +tpu_max_buffered_transactions = 0 # 0 = 131,072 buffered transactions +block_completion_reserve_ms = 0 # 0 = 75ms local completion/broadcast reserve tpu_sigverify_workers = 0 [consensus] diff --git a/cmd/mithril/configcmd/configcmd_test.go b/cmd/mithril/configcmd/configcmd_test.go index 2d62bb8b5..d6905a819 100644 --- a/cmd/mithril/configcmd/configcmd_test.go +++ b/cmd/mithril/configcmd/configcmd_test.go @@ -13,6 +13,10 @@ func TestStarterConfigSignatureVerification(t *testing.T) { v := viper.New() v.SetConfigType("toml") require.NoError(t, v.ReadConfig(strings.NewReader(generateStarterConfig(validator)))) + for _, key := range []string{"validator.wait_to_vote_slot", "validator.tpu_max_buffered_transactions", "validator.block_completion_reserve_ms"} { + require.True(t, v.IsSet(key), key) + require.Zero(t, v.GetInt(key), key) + } require.Equal(t, "auto", v.GetString("sigverify.backend")) require.True(t, v.IsSet("sigverify.workers")) require.Zero(t, v.GetInt("sigverify.workers")) diff --git a/docs/leader_block_packing.md b/docs/leader_block_packing.md index fd76fe466..b30ad57a1 100644 --- a/docs/leader_block_packing.md +++ b/docs/leader_block_packing.md @@ -81,7 +81,14 @@ The corresponding flags are `--tpu-max-buffered-transactions` and `--leader-completion-reserve-ms`. The measured full-prefill trial used 262,144 queue entries and a 60ms reserve. Those are opt-in tuning values; defaults stay unchanged. A larger queue uses additional memory for owned wire, decoded and -prepared objects. A shorter reserve needs measured local finalization/broadcast +prepared objects. `BenchmarkReadonlyPairPreparationMemory` measured 728 allocated +bytes per preparation (9 allocations) on Go 1.26.4 arm64 for the 198-byte fixture. +That is 91 MiB of allocation volume for 131,072 preparations, or 182 MiB for +262,144, in addition to wire/decoded transactions and queue indexes. Allocation +volume includes temporary preparation storage: it is not retained heap or RSS. +More accounts/instructions increase the footprint; these are not worst-case caps. +Reproduce with `go test ./pkg/blockprod -run '^$' -bench '^BenchmarkReadonlyPairPreparationMemory$' -benchmem`. +A shorter reserve needs measured local finalization/broadcast margin and does not change the protocol deadline. Shifting completion also shifts later bank start times, so it does not add the same packing time to all four blocks. diff --git a/docs/transaction_sigverify_streaming.md b/docs/transaction_sigverify_streaming.md index cfd39e6cd..4c40da2bb 100644 --- a/docs/transaction_sigverify_streaming.md +++ b/docs/transaction_sigverify_streaming.md @@ -291,3 +291,12 @@ wait, not the duration of whole-block verification or recovery from a failed wor including failed joins and discarded streams. It overlaps `GroupJoinAssembly` for successful groups; do not add them together or interpret it as cryptographic CPU cost. The 100 ms limit is a conservative fallback budget, not a measured optimum. + + +### Runtime pooling default + +Omitting `tuning.use_pool` now preserves the CLI default (`true`), instead of +silently disabling VM pooling through the config reader's zero value. Explicit +TOML `false` remains supported, and an explicitly supplied CLI flag takes +precedence. This is a runtime behavior change for previously minimal configs; +pooled memory is cleared before reuse. The flag remains the single default source. diff --git a/pkg/alpenglow/certpool.go b/pkg/alpenglow/certpool.go index 33483cc10..4383898d3 100644 --- a/pkg/alpenglow/certpool.go +++ b/pkg/alpenglow/certpool.go @@ -501,6 +501,7 @@ func (p *CertPool) finishSlotAndUnlock(slot uint64, ps *poolSlot, emits []Certif if p.slots[slot] != ps { emits = nil } + p.snap.CertsEmitted += uint64(len(emits)) target := p.publicationTargetLocked() ps.processing = false p.workCond.Broadcast() @@ -988,7 +989,6 @@ func (p *CertPool) maybeAssembleLocked(slot uint64, ps *poolSlot) []Certificate continue } p.emitted[key] = struct{}{} - p.snap.CertsEmitted++ emits = append(emits, cert) } return emits diff --git a/pkg/alpenglow/peer_sender_test.go b/pkg/alpenglow/peer_sender_test.go index dda145c4a..7595643fe 100644 --- a/pkg/alpenglow/peer_sender_test.go +++ b/pkg/alpenglow/peer_sender_test.go @@ -133,7 +133,9 @@ func testVotorBlockedPeer(t *testing.T, action string) { select { case received := <-marker: t.Logf("Healthy marker received in %s while other peer's SendDatagram is blocked", received.Sub(start)) - case <-time.After(250 * time.Millisecond): + case <-badConn.Context().Done(): + t.Fatal("healthy marker did not arrive before stalled peer was released") + case <-time.After(3 * time.Second): t.Fatal("stalled peer delayed healthy delivery") } require.NoError(t, badConn.Context().Err(), "marker must arrive before watchdog releases stalled peer") @@ -144,7 +146,7 @@ func testVotorBlockedPeer(t *testing.T, action string) { go func() { _ = b.Close(); close(closed) }() select { case <-closed: - case <-time.After(500 * time.Millisecond): + case <-time.After(3 * time.Second): t.Fatal("Close waited for stalled SendDatagram") } case "depart": @@ -158,7 +160,7 @@ func testVotorBlockedPeer(t *testing.T, action string) { } select { case <-badSender.done: - case <-time.After(500 * time.Millisecond): + case <-time.After(3 * time.Second): t.Fatal("old sender did not stop") } require.Error(t, badConn.Context().Err()) @@ -204,7 +206,9 @@ func testVotorBlockedPeer(t *testing.T, action string) { require.NoError(t, b.Enqueue(NewVoteMessage(NewSkipVote(markerSlot), testSignatureSeq(0x43), 3))) select { case <-marker: - case <-time.After(250 * time.Millisecond): + case <-badConn.Context().Done(): + t.Fatal("healthy marker did not arrive before stalled peer was released") + case <-time.After(3 * time.Second): t.Fatal("full peer queue delayed healthy delivery") } // Queue age is measured from fanout, so PTO progress no longer restarts diff --git a/pkg/blockprod/readonly_block_prepared_bench_test.go b/pkg/blockprod/readonly_block_prepared_bench_test.go index 68957fcef..ccf2380de 100644 --- a/pkg/blockprod/readonly_block_prepared_bench_test.go +++ b/pkg/blockprod/readonly_block_prepared_bench_test.go @@ -43,3 +43,19 @@ func BenchmarkReadonlyPairPreparedFullBlock(b *testing.B) { } b.ReportMetric(float64(readonlyBlockAccepted), "tx/block") } + +// Allocations per preparation bound the incremental prepared-object footprint; +// excludes the already decoded transaction and wire bytes owned by the queue. +func BenchmarkReadonlyPairPreparationMemory(b *testing.B) { + f := makeReadonlyBlockFixture(b, 1) + setup := f.bank(b, nil) + defer setup.Close() + preparer := replay.NewTransactionPreparer(setup.SlotCtx.Features) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if preparer.Prepare(f.txs[0]) == nil { + b.Fatal("preparation failed") + } + } +} diff --git a/pkg/consensus/voter.go b/pkg/consensus/voter.go index 4a9cbbc15..3f950126b 100644 --- a/pkg/consensus/voter.go +++ b/pkg/consensus/voter.go @@ -384,6 +384,7 @@ func (v *alpenglowVoter) loop() { for _, event := range pending { if err := v.handle(event); err != nil { v.engine.latchSafetyError(fmt.Errorf("reservation retry: %w", err)) + mlog.Log.Errorf("ALPENGLOW VOTING SAFETY: reservation retry: %v", err) return } } @@ -827,7 +828,7 @@ func (v *alpenglowVoter) sign(vote alpenglow.Vote, respectVotingGate bool) (alpe if err := v.engine.safetyError(); err != nil { return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, err } - if v.reservation != nil && !v.reservation.allow(vote.Slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { + if !respectVotingGate && v.reservation != nil && !v.reservation.allow(vote.Slot, v.engine.alpenglowVerifiedFinalityFloor(), false) { return alpenglow.VoteMessage{}, alpenglow.VoteVerifyResult{}, fmt.Errorf("%w: waiting for verified recovery or durable signing reservation", errVoterNotReady) } if respectVotingGate { diff --git a/pkg/costmodel/costmodel_test.go b/pkg/costmodel/costmodel_test.go index f5e41ec45..82160999c 100644 --- a/pkg/costmodel/costmodel_test.go +++ b/pkg/costmodel/costmodel_test.go @@ -81,15 +81,6 @@ func TestEstimateTransactionCostCountsPrecompileSignatures(t *testing.T) { ) } -func TestLimitsForFeaturesRaiseBlockLimitsTo100m(t *testing.T) { - feats := features.NewFeaturesDefault() - assert.Equal(t, uint64(MaxBlockUnitsSIMD0256), LimitsForFeatures(feats).BlockCost) - - feats.EnableFeature(features.RaiseBlockLimitsTo100m, 123) - assert.Equal(t, uint64(MaxBlockUnitsSIMD0286), LimitsForFeatures(feats).BlockCost) - assert.Equal(t, uint64(MaxBlockUnitsSIMD0256), DefaultLimits().BlockCost) -} - func TestWritableAccountsUsesUnsignedWritableRange(t *testing.T) { tx := &solana.Transaction{Message: solana.Message{ Header: solana.MessageHeader{ @@ -284,3 +275,12 @@ func TestCostTrackerAcceptsUnderLimits(t *testing.T) { assert.Equal(t, ExceedNone, tracker.WouldExceed(cost)) _ = wire } + +func TestPackEntryBytesMaxChargesEveryFECSetInBatch(t *testing.T) { + // One initial set, two reserved ending sets, then exactly one full batch. + shreds := uint64((1 + 2 + FECSetsPerBatch) * DataShredsPerFECSet) + want := uint64(DefaultTargetBatchBytes - 8 - MaxMicroblockBytes) + assert.Equal(t, want, PackEntryBytesMax(shreds, MaxMicroblockBytes)) + assert.Equal(t, want, PackEntryBytesMax(shreds+DataShredsPerFECSet-1, MaxMicroblockBytes)) + assert.Zero(t, PackEntryBytesMax(shreds-DataShredsPerFECSet, MaxMicroblockBytes)) +} diff --git a/pkg/costmodel/entry_bytes.go b/pkg/costmodel/entry_bytes.go index 7cfec6481..ebf89fc47 100644 --- a/pkg/costmodel/entry_bytes.go +++ b/pkg/costmodel/entry_bytes.go @@ -1,7 +1,7 @@ package costmodel // PackEntryBytesMax is the shred-safe entry-byte bound: we close a batch when -// the next microblock would not fit in one FEC set, so padding is at most one +// the next microblock would not fit in FECSetsPerBatch FEC sets, so padding is at most one // microblock. The slot cap is still min(this, SIMD-0525), decided at schedule // time like Firedancer pack. // @@ -23,7 +23,7 @@ func PackEntryBytesMax(slotMaxDataShreds, maxMicroblock uint64) uint64 { } middle := fecSets - first - lastFEC minBatch := wmark - maxMicroblock - return middle * minBatch + return (middle / FECSetsPerBatch) * minBatch } // DefaultPackEntryBytes is min(shred-safe, SIMD-0525) minus one ending tick. diff --git a/pkg/costmodel/limits.go b/pkg/costmodel/limits.go index 26091a3d0..0afdf925f 100644 --- a/pkg/costmodel/limits.go +++ b/pkg/costmodel/limits.go @@ -79,7 +79,6 @@ func DefaultLimits() Limits { } } -// LimitsForFeatures returns the legacy 400ms budgets. Live banks must use // LimitsForSlot to apply slot-time reductions at the correct epoch boundary. func LimitsForFeatures(feats *features.Features) Limits { limits := DefaultLimits() diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index addac7bf3..5c4944492 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -381,7 +381,7 @@ func (s *streamingExecutor) handleTick() { func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { frontier := s.deps.frontier() - if header.Slot <= frontier { + if header.Slot <= frontier || header.Slot-frontier > streamingMaxSlotDistance { return } s.headers[header.Slot] = header @@ -441,17 +441,17 @@ func (s *streamingExecutor) retire(slot uint64, g turbine.StreamGeneration) { // frontier can never open. func (s *streamingExecutor) pruneHeaders(frontier uint64) { for slot := range s.headers { - if slot <= frontier { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { delete(s.headers, slot) } } for slot := range s.retired { - if slot <= frontier { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { delete(s.retired, slot) } } for slot := range s.observed { - if slot <= frontier { + if slot <= frontier || slot-frontier > streamingMaxSlotDistance { delete(s.observed, slot) } } diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index abb2dfcd3..d2cba3027 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -1160,3 +1160,21 @@ func TestStreamingVerificationWaitRespectsRemainingOpenAge(t *testing.T) { require.Empty(t, h.executed()) require.Equal(t, "sigverify_timeout", h.discardReason()) } + +func TestStreamingHeaderAdmissionBounds(t *testing.T) { + frontier := uint64(100) + s := newStreamingExecutor(streamingDeps{frontier: func() uint64 { return frontier }}) + for _, slot := range []uint64{100, 101, 132, 133, ^uint64(0)} { + g := turbine.NewDetachedStreamGeneration(slot) + s.rememberHeader(turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, 100, solana.Hash{})) + } + require.Len(t, s.headers, 2) + require.Len(t, s.observed, 2) + require.Contains(t, s.headers, uint64(101)) + require.Contains(t, s.headers, uint64(132)) + frontier = ^uint64(0) - 1 + s.pruneHeaders(frontier) + g := turbine.NewDetachedStreamGeneration(^uint64(0)) + s.rememberHeader(turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, frontier, solana.Hash{})) + require.Len(t, s.headers, 1, "distance comparison must not overflow") +} diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 96d3cb25b..8cd4c681c 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1111,10 +1111,16 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un // same idea as Agave's MappingCache). Only the currently mapped // RegionSize bytes are exposed; anything beyond takes the slow path // again so that OnWrite / growth semantics are preserved. - if region.RegionSize != 0 && (region.Data == nil || uint64(len(region.Data)) >= region.RegionSize) { - cached := memRegion{base: base, start: region.Offset, rlen: region.RegionSize, gapShift: 63} + cacheLen := region.RegionSize + if region.Data == nil { + cacheLen = min(cacheLen, uint64(len(ip.input))-region.HostOffset) + } else { + cacheLen = min(cacheLen, uint64(len(region.Data))) + } + if cacheLen != 0 { + cached := memRegion{base: base, start: region.Offset, rlen: cacheLen, gapShift: 63} if region.Writable { - cached.wlen = region.RegionSize + cached.wlen = cacheLen } ip.regions[VaddrInput>>32] = cached } diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go index 0d072f564..3ad824959 100644 --- a/pkg/sbpf/perf_differential_test.go +++ b/pkg/sbpf/perf_differential_test.go @@ -2,8 +2,10 @@ package sbpf import ( "bufio" + "crypto/sha256" "fmt" "hash/fnv" + "io" "math/rand" "os" "testing" @@ -316,11 +318,14 @@ func TestDifferentialDump(t *testing.T) { w := bufio.NewWriter(f) defer w.Flush() + writeDifferentialDump(t, w, 100000) +} + +func writeDifferentialDump(t *testing.T, w io.Writer, n int) { rng := rand.New(rand.NewSource(12345)) - const N = 100000 generated, verified := 0, 0 var verifiedByVersion [4]int - for i := 0; i < N; i++ { + for i := 0; i < n; i++ { // Equal representation of every version, independent of RNG consumption. ver := uint32(i % 4) p := genProgram(rng, ver) @@ -406,3 +411,15 @@ func TestDifferentialDump(t *testing.T) { } t.Logf("generated=%d verified=%d verified_by_version=%v", generated, verified, verifiedByVersion) } + +// Golden derived from the pre-optimization e1204b32 interpreter with this +// generator (seed 12345). Do not regenerate from candidate output alone. +// Covers 1024 programs per version; dumps compare results, errors, CU and memory. +func TestDifferentialGolden(t *testing.T) { + h := sha256.New() + writeDifferentialDump(t, h, 4096) + const want = "ccc16b4021e784342a7200a3d8911d362d2ab11a5ddefccf1c93880a704426b0" + if got := fmt.Sprintf("%x", h.Sum(nil)); got != want { + t.Fatalf("differential drift: got %s, want %s", got, want) + } +} diff --git a/pkg/sbpf/vasa_test.go b/pkg/sbpf/vasa_test.go index 70296ea43..77912c07a 100644 --- a/pkg/sbpf/vasa_test.go +++ b/pkg/sbpf/vasa_test.go @@ -100,3 +100,16 @@ func TestStackFrameGapsAreLegacyOnly(t *testing.T) { require.Equal(t, VaddrStack+StackFrameSize*2, regs[10]) require.NotNil(t, stack.GetFrame(StackFrameSize)) } + +func TestInputRegionFastCacheClampsToBackingBytes(t *testing.T) { + ip := &Interpreter{input: make([]byte, 8), inputRegions: []InputRegion{{Offset: 0, HostOffset: 4, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}}} + ip.initRegions() + require.NoError(t, ip.Write8(VaddrInput, 7)) + require.NotNil(t, ip.fastRead(VaddrInput+3, 1)) + require.Nil(t, ip.fastRead(VaddrInput+4, 1)) + require.Nil(t, ip.fastWrite(VaddrInput+4, 1)) + _, err := ip.Read8(VaddrInput + 4) + require.Error(t, err) + require.Error(t, ip.Write8(VaddrInput+4, 8)) + require.Equal(t, byte(7), ip.input[4]) +} diff --git a/pkg/sealevel/bpf_loader.go b/pkg/sealevel/bpf_loader.go index 845accd60..e64c04589 100644 --- a/pkg/sealevel/bpf_loader.go +++ b/pkg/sealevel/bpf_loader.go @@ -1361,6 +1361,10 @@ func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey return executeLoadedProgram(execCtx, program, syscallRegistry) } +// All loader cache insertions must pass through this helper. A speculative +// stream records replacements for eviction on discard; the previous program is +// then reloaded from authoritative account data, never retained from that stream. +// Replay must discard its stream before executing a different bank. func addProgramToCache(execCtx *ExecutionCtx, programAddr solana.PublicKey, entry *accountsdb.ProgramCacheEntry) { if execCtx.SlotCtx == nil || execCtx.SlotCtx.AccountsDb == nil { return diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 18309927b..c5b90ebed 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -75,6 +75,7 @@ func TestInterpreter_Noop(t *testing.T) { ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -107,13 +108,16 @@ func TestInterpreter_Memcpy_Strings_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -145,14 +149,17 @@ func TestInterpreter_Memcpy_Do_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -183,14 +190,17 @@ func TestInterpreter_Memmove_Strings_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -222,14 +232,17 @@ func TestInterpreter_Memmove_Do_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.Equal(t, log.Logs, []string{ @@ -259,14 +272,17 @@ func TestInterpreter_Memcpy_Overlapping(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() @@ -297,14 +313,17 @@ func TestInterpreter_Memcmp_Matches(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -338,14 +357,17 @@ func TestInterpreter_Memcmp_Does_Not_Match(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -379,14 +401,17 @@ func TestInterpreter_Memset_Check_Correct(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -429,6 +454,7 @@ func TestInterpreter_Sha256(t *testing.T) { ComputeMeter: &ctx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -462,14 +488,17 @@ func TestInterpreter_Blake3(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -503,14 +532,17 @@ func TestInterpreter_Keccak256(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -545,14 +577,17 @@ func TestInterpreter_CreateProgramAddress(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -590,14 +625,17 @@ func TestInterpreter_TryFindProgramAddress(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -628,18 +666,24 @@ func TestInterpreter_TestPanic(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.Error(t, err) - assert.Equal(t, err.Error(), "exception at 16: SBF program Panicked in some_file_1234.c at 1337:10") + require.Contains(t, err.Error(), "SBF program Panicked in some_file_1234.c at 1337:10") + var exception *sbpf.Exception + require.ErrorAs(t, err, &exception) + require.Equal(t, int64(17), exception.PC) } func TestInterpreter_Secp256k1_Syscall(t *testing.T) { @@ -659,14 +703,17 @@ func TestInterpreter_Secp256k1_Syscall(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: syscalls, - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: syscalls, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() require.NoError(t, err) @@ -2091,6 +2138,7 @@ func (e *executeCase) run(t *testing.T) { interpreter := sbpf.NewInterpreter(program, opts) require.NotNil(t, interpreter) + defer interpreter.Finish() _, _, err = interpreter.Run() assert.NoError(t, err) diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index ede8842ed..3c043af0e 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -1753,7 +1753,6 @@ func (s *slotState) sortedShreds() []*Shred { for _, idx := range indexes { out = append(out, s.shreds[uint32(idx)]) } - sort.Slice(out, func(i, j int) bool { return out[i].Index < out[j].Index }) return out } diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go index 71ffa1469..64f917f9c 100644 --- a/pkg/turbine/entry_prefetch.go +++ b/pkg/turbine/entry_prefetch.go @@ -30,6 +30,7 @@ type prefetchedShredBatch struct { start, end uint32 raw []byte entries []Entry + transactions []*solana.Transaction // immutable pointer view, built once before ready closes parent *AlpenglowParentInfo footer *BlockFooter marker bool @@ -190,8 +191,11 @@ func (p *entryPrefetchPool) run() { if s.pipelineTrace != nil { batch.traceDecodeEnd = entryTraceNow() } + if batch.err == nil && !batch.marker { + batch.transactions = entryBatchTransactions(batch.entries) + } if batch.err == nil && !batch.marker && f.ctx.Err() == nil { - txs := entryBatchTransactions(batch.entries) + txs := batch.transactions if len(txs) > 0 { batch.submittedAt = time.Now() batch.verification, batch.submitErr = p.verifier.submitPrefetchTransactions(f.ctx, txs) diff --git a/pkg/turbine/entry_prefetch_test.go b/pkg/turbine/entry_prefetch_test.go index 2c2aed451..b8fe4bf43 100644 --- a/pkg/turbine/entry_prefetch_test.go +++ b/pkg/turbine/entry_prefetch_test.go @@ -335,8 +335,9 @@ func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { require.NoError(t, err) } require.NotNil(t, work) - ctx, cancel := context.WithCancel(context.Background()) + baseCtx, cancel := context.WithCancel(context.Background()) defer cancel() + ctx := &verificationJoinContext{Context: baseCtx, waiting: make(chan struct{})} canceled := make(chan struct{}) originalCancel := cached.verification.cancel cached.verification.cancel = func() { close(canceled); originalCancel() } @@ -344,7 +345,7 @@ func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { go func() { done <- a.processCompletion(ctx, work) }() // The completion has no expensive decode left and blocks joining this // one already-prepared future; cancel while it owns those transactions. - time.Sleep(20 * time.Millisecond) + waitSignal(t, ctx.waiting, "completion entered verification wait") cancel() waitSignal(t, canceled, "completion canceled its signature request") releaseOnce.Do(func() { close(release) }) @@ -679,3 +680,16 @@ func TestEntryPrefetchFailedCompletionWithoutReservation(t *testing.T) { require.Zero(t, slots) require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 940, state: s})) } + +// Done is consulted by the blocking verification join; this avoids assuming +// completion reaches that join within a fixed amount of wall time. +type verificationJoinContext struct { + context.Context + waiting chan struct{} + once sync.Once +} + +func (c *verificationJoinContext) Done() <-chan struct{} { + c.once.Do(func() { close(c.waiting) }) + return c.Context.Done() +} diff --git a/pkg/turbine/shredspool.go b/pkg/turbine/shredspool.go index 0170b0f76..cebb0c837 100644 --- a/pkg/turbine/shredspool.go +++ b/pkg/turbine/shredspool.go @@ -426,7 +426,9 @@ func (s *ShredSpool) recomputeHighestLocked() { // DiscardSlot removes both persisted packets and the completeness marker for // a poisoned or rejected slot. The journal tombstone prevents an older // completion record from being resurrected if repair immediately recreates a -// partial file with the same slot number. +// partial file with the same slot number. If the journal cannot durably fence +// the old marker, retain the old file and reject replacement writes until a +// retry succeeds; deleting it would permit stale completeness to certify new data. func (s *ShredSpool) DiscardSlot(slot uint64) { if s == nil { return diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index a6a51b44a..a3e0abc40 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -339,7 +339,7 @@ func newStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatc // component boundary: nothing to execute, nothing to select. view.Marker = StreamMarkerFooter default: - view.Transactions = entryBatchTransactions(batch.entries) + view.Transactions = batch.transactions } return view } diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index 50245fce6..132e805e0 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -271,3 +271,22 @@ func TestStreamVerificationTimeoutDoesNotCancelOrJoinOwner(t *testing.T) { require.NoError(t, err) require.True(t, verified, "same request remains usable after the observer leaves") } + +func TestStreamPollingReusesTransactionView(t *testing.T) { + v := newTransactionVerifier(2, 16, nil) + defer v.closeAndWait() + a := NewSlotAssembler() + p := newEntryPrefetchPool(context.Background(), a, v) + defer p.closeAndWait() + batches := prefetchTestShreds(t, 812, prefetchTestPayload(t, verifierSignedTransactions(t, 3)), buildAlpenglowEndingTick(t)) + require.Nil(t, feedPrefetchShreds(t, a, batches[0])) + waitPrefetchedBatch(t, a, 812, 0) + a.mu.Lock() + g := StreamGeneration{slot: 812, state: a.slots[812]} + a.mu.Unlock() + first, second := a.PendingStreamBatches(g, 0), a.PendingStreamBatches(g, 0) + require.Len(t, first, 1) + require.Len(t, second, 1) + require.Len(t, first[0].Transactions, 3) + require.Same(t, &first[0].Transactions[0], &second[0].Transactions[0]) +} From e7d351aa9398ac9f57e487f9f018ff8260e11fcb Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:03:53 -0500 Subject: [PATCH 072/111] test: synchronize completion cancellation on the owning join --- pkg/turbine/entry_prefetch_test.go | 28 ++++++++++++---------------- 1 file changed, 12 insertions(+), 16 deletions(-) diff --git a/pkg/turbine/entry_prefetch_test.go b/pkg/turbine/entry_prefetch_test.go index b8fe4bf43..9dbbd9ada 100644 --- a/pkg/turbine/entry_prefetch_test.go +++ b/pkg/turbine/entry_prefetch_test.go @@ -4,6 +4,8 @@ import ( "context" "errors" "fmt" + "runtime" + "strings" "sync" "sync/atomic" "testing" @@ -335,9 +337,8 @@ func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { require.NoError(t, err) } require.NotNil(t, work) - baseCtx, cancel := context.WithCancel(context.Background()) + ctx, cancel := context.WithCancel(context.Background()) defer cancel() - ctx := &verificationJoinContext{Context: baseCtx, waiting: make(chan struct{})} canceled := make(chan struct{}) originalCancel := cached.verification.cancel cached.verification.cancel = func() { close(canceled); originalCancel() } @@ -345,7 +346,15 @@ func TestEntryPrefetchCanceledCompletionCanRetrySameGeneration(t *testing.T) { go func() { done <- a.processCompletion(ctx, work) }() // The completion has no expensive decode left and blocks joining this // one already-prepared future; cancel while it owns those transactions. - waitSignal(t, ctx.waiting, "completion entered verification wait") + require.Eventually(t, func() bool { + stack := make([]byte, 2<<20) + for _, goroutine := range strings.Split(string(stack[:runtime.Stack(stack, true)]), "\n\n") { + if strings.Contains(goroutine, "(*SlotAssembler).processCompletion") && strings.Contains(goroutine, "(*transactionVerification).waitContext") { + return true + } + } + return false + }, 3*time.Second, time.Millisecond, "completion must enter the owning verification join") cancel() waitSignal(t, canceled, "completion canceled its signature request") releaseOnce.Do(func() { close(release) }) @@ -680,16 +689,3 @@ func TestEntryPrefetchFailedCompletionWithoutReservation(t *testing.T) { require.Zero(t, slots) require.Equal(t, StreamGone, a.StreamStatusOf(StreamGeneration{slot: 940, state: s})) } - -// Done is consulted by the blocking verification join; this avoids assuming -// completion reaches that join within a fixed amount of wall time. -type verificationJoinContext struct { - context.Context - waiting chan struct{} - once sync.Once -} - -func (c *verificationJoinContext) Done() <-chan struct{} { - c.once.Do(func() { close(c.waiting) }) - return c.Context.Done() -} From 7fcdcbb2194632777d1c1ce3a16278f37273f165 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:28:16 -0500 Subject: [PATCH 073/111] sealevel: correct return-data truncation and ABI encodings --- pkg/sealevel/bpf_loader.go | 6 +- pkg/sealevel/loader_wire_test.go | 30 +++++++++ pkg/sealevel/syscalls_call.go | 4 +- pkg/sealevel/syscalls_return_data_test.go | 43 +++++++++++++ pkg/sealevel/syscalls_sysvar.go | 3 +- pkg/sealevel/syscalls_sysvar_layout_test.go | 68 +++++++++++++++++++++ pkg/sealevel/types.go | 4 +- 7 files changed, 151 insertions(+), 7 deletions(-) create mode 100644 pkg/sealevel/loader_wire_test.go create mode 100644 pkg/sealevel/syscalls_return_data_test.go create mode 100644 pkg/sealevel/syscalls_sysvar_layout_test.go diff --git a/pkg/sealevel/bpf_loader.go b/pkg/sealevel/bpf_loader.go index e64c04589..6142b3049 100644 --- a/pkg/sealevel/bpf_loader.go +++ b/pkg/sealevel/bpf_loader.go @@ -123,7 +123,11 @@ func (write *UpgradeableLoaderInstrWrite) MarshalWithEncoder(encoder *bin.Encode return err } - err = encoder.WriteBytes(write.Bytes, true) + // UpgradeableLoaderInstruction uses bincode's fixed-width u64 vector length. + if err = encoder.WriteUint64(uint64(len(write.Bytes)), bin.LE); err != nil { + return err + } + err = encoder.WriteBytes(write.Bytes, false) return err } diff --git a/pkg/sealevel/loader_wire_test.go b/pkg/sealevel/loader_wire_test.go new file mode 100644 index 000000000..e45d375e8 --- /dev/null +++ b/pkg/sealevel/loader_wire_test.go @@ -0,0 +1,30 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + bin "github.com/gagliardetto/binary" + "github.com/stretchr/testify/require" + "testing" +) + +func TestUpgradeableLoaderWriteWireLayout(t *testing.T) { + instruction := UpgradeableLoaderInstrWrite{Offset: 0x12345678, Bytes: []byte{0xAB, 0xCD}} + var buf bytes.Buffer + require.NoError(t, instruction.MarshalWithEncoder(bin.NewBinEncoder(&buf))) + require.Equal(t, []byte{1, 0, 0, 0, 0x78, 0x56, 0x34, 0x12, 2, 0, 0, 0, 0, 0, 0, 0, 0xAB, 0xCD}, buf.Bytes()) + var decoded UpgradeableLoaderInstrWrite + require.NoError(t, decoded.UnmarshalWithDecoder(bin.NewBinDecoder(buf.Bytes()[4:]))) + require.Equal(t, instruction, decoded) +} + +func TestSolAccountMetaCWireLayout(t *testing.T) { + for _, bits := range [][2]byte{{0, 0}, {1, 0}, {0, 1}, {1, 1}} { + meta := SolAccountMetaC{PubkeyAddr: 0x12345678, IsWritable: bits[0], IsSigner: bits[1]} + wire, err := meta.Marshal() + require.NoError(t, err) + want := binary.LittleEndian.AppendUint64(nil, meta.PubkeyAddr) + want = append(want, bits[:]...) + require.Equal(t, want, wire) + } +} diff --git a/pkg/sealevel/syscalls_call.go b/pkg/sealevel/syscalls_call.go index 0ffe7b0c3..b594ad2a5 100644 --- a/pkg/sealevel/syscalls_call.go +++ b/pkg/sealevel/syscalls_call.go @@ -53,11 +53,11 @@ func SyscallGetReturnDataImpl(vm sbpf.VM, returnDataAddr, length, programIdAddr return syscallErr(err) } - if len(returnData) != len(returnDataResult) { + if int(length) != len(returnDataResult) { return syscallErr(SyscallErrInvalidLength) } - copy(returnDataResult, returnData) + copy(returnDataResult, returnData[:length]) var programIdResult []byte programIdResult, err = vm.Translate(programIdAddr, solana.PublicKeyLength, true) diff --git a/pkg/sealevel/syscalls_return_data_test.go b/pkg/sealevel/syscalls_return_data_test.go new file mode 100644 index 000000000..e3252a154 --- /dev/null +++ b/pkg/sealevel/syscalls_return_data_test.go @@ -0,0 +1,43 @@ +package sealevel + +import ( + "bytes" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" + "testing" +) + +func TestSyscallGetReturnDataPrefix(t *testing.T) { + for _, n := range []uint64{0, 1, 7, 32, 64} { + t.Run(fmt.Sprint(n), func(t *testing.T) { + input := bytes.Repeat([]byte{0xCC}, 128) + vm, ctx := newMemSyscallVM(t, input, nil) + ctx.TransactionContext = &TransactionCtx{} + data := bytes.Repeat([]byte{0x5A}, 32) + pk := solana.PublicKey{0x72} + ctx.TransactionContext.SetReturnData(pk, data) + before := ctx.ComputeMeter.Remaining() + got, err := SyscallGetReturnDataImpl(vm, sbpf.VaddrInput, n, sbpf.VaddrInput+64) + require.NoError(t, err) + require.Equal(t, uint64(len(data)), got) + copied := min(n, uint64(len(data))) + actual, err := vm.Translate(sbpf.VaddrInput, 128, false) + require.NoError(t, err) + require.Equal(t, data[:copied], actual[:copied]) + require.Equal(t, bytes.Repeat([]byte{0xCC}, int(64-copied)), actual[copied:64]) + charge := uint64(cu.CUSyscallBaseCost) + if copied != 0 { + require.Equal(t, pk[:], actual[64:96]) + charge += (copied + 32) / cu.CUCpiBytesPerUnit + } else { + require.Equal(t, bytes.Repeat([]byte{0xCC}, 32), actual[64:96]) + } + require.Equal(t, charge, before-ctx.ComputeMeter.Remaining()) + _, retained := ctx.TransactionContext.ReturnData() + require.Equal(t, data, retained) + }) + } +} diff --git a/pkg/sealevel/syscalls_sysvar.go b/pkg/sealevel/syscalls_sysvar.go index acd8929b4..400744653 100644 --- a/pkg/sealevel/syscalls_sysvar.go +++ b/pkg/sealevel/syscalls_sysvar.go @@ -12,7 +12,6 @@ import ( //"github.com/Overclock-Validator/mithril/pkg/mlog" "github.com/Overclock-Validator/mithril/pkg/safemath" "github.com/Overclock-Validator/mithril/pkg/sbpf" - "github.com/Overclock-Validator/mithril/pkg/util" "github.com/gagliardetto/solana-go" ) @@ -172,9 +171,9 @@ func SyscallGetEpochRewardsSysvarImpl(vm sbpf.VM, addr uint64) (uint64, error) { binary.LittleEndian.PutUint64(epochRewardsDst[8:16], epochRewards.NumPartitions) copy(epochRewardsDst[16:48], epochRewards.ParentBlockhash[:]) + // repr(C) u128 uses little-endian low/high limbs on the SBF target. binary.LittleEndian.PutUint64(epochRewardsDst[48:56], epochRewards.TotalPoints.Lo) binary.LittleEndian.PutUint64(epochRewardsDst[56:64], epochRewards.TotalPoints.Hi) - util.ReverseBytesInPlace(epochRewardsDst[48:64]) binary.LittleEndian.PutUint64(epochRewardsDst[64:72], epochRewards.TotalRewards) binary.LittleEndian.PutUint64(epochRewardsDst[72:80], epochRewards.DistributedRewards) diff --git a/pkg/sealevel/syscalls_sysvar_layout_test.go b/pkg/sealevel/syscalls_sysvar_layout_test.go new file mode 100644 index 000000000..1cea7ecb9 --- /dev/null +++ b/pkg/sealevel/syscalls_sysvar_layout_test.go @@ -0,0 +1,68 @@ +package sealevel + +import ( + "encoding/binary" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math" + "testing" +) + +// Replace the old sysvars.so fixture, which expected packed EpochSchedule +// u64 fields at offsets 17/25. The syscall ABI is repr(C): offsets 24/32. +func TestSyscallSysvarLayouts(t *testing.T) { + vm, ctx := newMemSyscallVM(t, make([]byte, 512), nil) + ctx.Accounts = accounts.NewMemAccounts() + clock := SysvarClock{Slot: 1234, EpochStartTimestamp: 2222, Epoch: 1111, LeaderScheduleEpoch: 100000, UnixTimestamp: 3} + rent := SysvarRent{LamportsPerUint8Year: 12, ExemptionThreshold: 34, BurnPercent: 56} + schedule := SysvarEpochSchedule{SlotsPerEpoch: 1111, LeaderScheduleSlotOffset: 2222, Warmup: true, FirstNormalEpoch: 4444, FirstNormalSlot: 5555} + rewards := SysvarEpochRewards{DistributionStartingBlockHeight: 1234, NumPartitions: 4321, TotalRewards: 5656, DistributedRewards: 6767, Active: true} + rewards.TotalPoints.Lo = 0x0123456789abcdef + rewards.TotalPoints.Hi = 0xfedcba9876543210 + rewards.ParentBlockhash[0] = 0x73 + for _, key := range [][32]byte{SysvarClockAddr, SysvarRentAddr, SysvarEpochScheduleAddr, SysvarEpochRewardsAddr, SysvarLastRestartSlotAddr} { + require.NoError(t, ctx.Accounts.SetAccount(&key, &accounts.Account{Key: key, Lamports: 1})) + } + WriteClockSysvar(&ctx.Accounts, clock) + WriteRentSysvar(&ctx.Accounts, rent) + WriteEpochScheduleSysvar(&ctx.Accounts, schedule) + WriteEpochRewardsSysvar(&ctx.Accounts, rewards) + WriteLastRestartSlotSysvar(&ctx.Accounts, SysvarLastRestartSlot{LastRestartSlot: 989898}) + for _, tc := range []struct { + name string + call func(sbpf.VM, uint64) (uint64, error) + want []byte + }{ + {"clock", SyscallGetClockSysvarImpl, appendU64s(1234, 2222, 1111, 100000, 3)}, + {"rent", SyscallGetRentSysvarImpl, append(appendU64s(12, math.Float64bits(34)), 56, 0, 0, 0, 0, 0, 0, 0)}, + {"schedule", SyscallGetEpochScheduleSysvarImpl, appendU64s(1111, 2222, 1, 4444, 5555)}, + {"last restart", SyscallGetLastRestartSlotSysvarImpl, appendU64s(989898)}, + } { + t.Run(tc.name, func(t *testing.T) { + _, err := tc.call(vm, sbpf.VaddrInput) + require.NoError(t, err) + got, err := vm.Translate(sbpf.VaddrInput, uint64(len(tc.want)), false) + require.NoError(t, err) + require.Equal(t, tc.want, got) + clear(got) + }) + } + _, err := SyscallGetEpochRewardsSysvarImpl(vm, sbpf.VaddrInput) + require.NoError(t, err) + got, err := vm.Translate(sbpf.VaddrInput, 96, false) + require.NoError(t, err) + want := make([]byte, 96) + copy(want, appendU64s(1234, 4321)) + copy(want[16:48], rewards.ParentBlockhash[:]) + copy(want[48:], appendU64s(rewards.TotalPoints.Lo, rewards.TotalPoints.Hi, 5656, 6767)) + want[80] = 1 + require.Equal(t, want, got) +} +func appendU64s(values ...uint64) []byte { + var b []byte + for _, v := range values { + b = binary.LittleEndian.AppendUint64(b, v) + } + return b +} diff --git a/pkg/sealevel/types.go b/pkg/sealevel/types.go index 558429374..6d00a25f7 100644 --- a/pkg/sealevel/types.go +++ b/pkg/sealevel/types.go @@ -208,12 +208,12 @@ func (accountMeta *SolAccountMetaC) Marshal() ([]byte, error) { return nil, err } - err = binary.Write(buf, binary.LittleEndian, accountMeta.IsSigner) + err = binary.Write(buf, binary.LittleEndian, accountMeta.IsWritable) if err != nil { return nil, err } - err = binary.Write(buf, binary.LittleEndian, accountMeta.IsWritable) + err = binary.Write(buf, binary.LittleEndian, accountMeta.IsSigner) if err != nil { return nil, err } From 25447e36b163097240c25d40f1b81abab61c4702 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:28:25 -0500 Subject: [PATCH 074/111] test: repair legacy bank fixtures and run full sealevel CI --- .github/workflows/go_build.yml | 5 +- pkg/sealevel/legacy_bank_fixture_test.go | 37 ++++ pkg/sealevel/sealevel_bpf_loader_test.go | 91 +++++++--- pkg/sealevel/sealevel_config_program_test.go | 2 +- pkg/sealevel/sealevel_system_program_test.go | 2 +- pkg/sealevel/sealevel_test.go | 174 ++++--------------- pkg/sealevel/spl_token_demo_test.go | 7 +- pkg/sealevel/sysvar_instructions_test.go | 2 +- 8 files changed, 152 insertions(+), 168 deletions(-) create mode 100644 pkg/sealevel/legacy_bank_fixture_test.go diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index bdcfcc42c..4d9d9bf5d 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -35,8 +35,7 @@ jobs: - name: Voting, checkpoint, streaming and scheduler race regressions # Run the complete affected package suites, including subprocess crash - # recovery and cancellation tests. The broader sealevel suite still has unrelated legacy - # loader-fixture failures; the interpreter/syscall subset runs below. + # recovery and cancellation tests. run: >- go test -race -p 2 -count=1 ./pkg/alpenglow ./pkg/consensus ./pkg/replay @@ -47,4 +46,4 @@ jobs: run: go test -race -count=1 ./pkg/sbpf/... - name: Sealevel interpreter, syscall and vote ownership regressions - run: go test -race -count=1 ./pkg/sealevel -run '^(TestInterpreter_(Noop|Mem|Sha256|Blake3|Keccak256|CreateProgramAddress|TryFindProgramAddress|TestPanic|Secp256k1)|TestSyscall|TestMemoryCopyDifferential$|TestProcessNewVoteStateOwnsRetainedDeque$)' + run: go test -race -count=1 ./pkg/sealevel diff --git a/pkg/sealevel/legacy_bank_fixture_test.go b/pkg/sealevel/legacy_bank_fixture_test.go new file mode 100644 index 000000000..cbf7cf81a --- /dev/null +++ b/pkg/sealevel/legacy_bank_fixture_test.go @@ -0,0 +1,37 @@ +package sealevel + +import ( + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" + "testing" +) + +// Legacy instruction fixtures predate the bank and program-cache dependencies +// used by loaded programs. Give each fixture isolated state, as replay does. +func initializeLegacyBankFixture(t *testing.T, ctx *ExecutionCtx) { + t.Helper() + ctx.RecordInnerInstructions = true + if ctx.Log == nil { + ctx.Log = &LogRecorder{} + } + t.Cleanup(func() { + if t.Failed() { + t.Logf("program logs: %#v", ctx.Log) + } + }) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + if ctx.Accounts == nil { + ctx.Accounts = accounts.NewMemAccounts() + } + if ctx.SlotCtx == nil { + ctx.SlotCtx = &SlotCtx{} + } + ctx.SlotCtx.Accounts = ctx.Accounts + ctx.SlotCtx.AccountsDb = &accountsdb.AccountsDb{ProgramCache: cache} + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} +} diff --git a/pkg/sealevel/sealevel_bpf_loader_test.go b/pkg/sealevel/sealevel_bpf_loader_test.go index f20f13890..ab1b89d47 100644 --- a/pkg/sealevel/sealevel_bpf_loader_test.go +++ b/pkg/sealevel/sealevel_bpf_loader_test.go @@ -48,6 +48,7 @@ func TestExecute_Tx_BpfLoader_InitializeBuffer_Success(t *testing.T) { instrData := make([]byte, 4) binary.LittleEndian.AppendUint32(instrData, UpgradeableLoaderInstrTypeInitializeBuffer) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -91,6 +92,7 @@ func TestExecute_Tx_BpfLoader_InitializeBuffer_Buffer_Acct_Already_Initialize_Fa instrData := make([]byte, 4) binary.LittleEndian.AppendUint32(instrData, UpgradeableLoaderInstrTypeInitializeBuffer) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrAccountAlreadyInitialized, err) @@ -143,6 +145,7 @@ func TestExecute_Tx_BpfLoader_Write_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -203,6 +206,7 @@ func TestExecute_Tx_BpfLoader_Write_Offset_Too_Large_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrAccountDataTooSmall, err) } @@ -254,6 +258,7 @@ func TestExecute_Tx_BpfLoader_Write_Buffer_Authority_Didnt_Sign_Failure(t *testi txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -310,6 +315,7 @@ func TestExecute_Tx_BpfLoader_Write_Incorrect_Authority_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -346,8 +352,9 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Not_Enough_Instr_Accts_Failure(t *tes txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Success(t *testing.T) { @@ -391,6 +398,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -445,6 +453,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -499,6 +508,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Immutable_Failure(t *testing.T txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -549,6 +559,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Wrong_Upgrade_Authority_Failur txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -594,6 +605,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Authority_Didnt_Sign_Failure(t txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -633,6 +645,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_No_New_Authority_Failure(t *te txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -678,6 +691,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_Buffer_Uninitialized_Account_Failure( txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -723,6 +737,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Immutable_Failure(t *test txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -768,6 +783,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Authority_Didnt_Sign_Fail txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -818,6 +834,7 @@ func TestExecute_Tx_BpfLoader_SetAuthority_ProgramData_Wrong_Authority_Failure(t txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -855,9 +872,10 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Not_Enough_Instr_Accts_Failure txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success(t *testing.T) { @@ -902,7 +920,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -958,7 +977,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Success(t *testing txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1014,7 +1034,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Immutable_Failure(t *te txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1066,7 +1087,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Wrong_Upgrade_Authority txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1113,7 +1135,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Authority_Didnt_Sign_Fa txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1160,7 +1183,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_New_Authority_Didnt_Sig txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1207,7 +1231,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_Buffer_Uninitialized_Account_F txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1254,7 +1279,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Immutable_Failure( txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1301,7 +1327,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Authority_Didnt_Si txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1348,7 +1375,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_New_Authority_Didn txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1400,7 +1428,8 @@ func TestExecute_Tx_BpfLoader_SetAuthorityChecked_ProgramData_Wrong_Authority_Fa txCtx := NewTransactionCtx(*transactionAccts, 5, 64) f := features.NewFeaturesDefault() f.EnableFeature(features.EnableBpfLoaderSetAuthorityCheckedIx, 0) - execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + execCtx := ExecutionCtx{Features: *f, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1446,6 +1475,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1502,6 +1532,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Immutable_Failure(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrImmutable, err) } @@ -1547,6 +1578,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Authority_Didnt_Sign_Failure(t *testi txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrMissingRequiredSignature, err) } @@ -1598,6 +1630,7 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Wrong_Authority_Failure(t *testing.T) txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -1636,6 +1669,7 @@ func TestExecute_Tx_BpfLoader_Close_Uninitialized_Success(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1680,6 +1714,7 @@ func TestExecute_Tx_BpfLoader_Close_Recipient_Same_As_Account_Being_Closed_Failu txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1723,8 +1758,9 @@ func TestExecute_Tx_BpfLoader_Close_Buffer_Not_Enough_Accounts(t *testing.T) { txCtx := NewTransactionCtx(*transactionAccts, 5, 64) execCtx := ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeterDefault()} + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_Close_ProgramData_Success(t *testing.T) { @@ -1787,6 +1823,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Success(t *testing.T) { clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -1861,8 +1898,9 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Not_Enough_Accounts_Failure(t *t clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Not_Writable_Failure(t *testing.T) { @@ -1925,6 +1963,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Not_Writable_Failur clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -1990,6 +2029,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Program_Acct_Wrong_Owner_Failure clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectProgramId, err) } @@ -2055,6 +2095,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Already_Deployed_In_This_Block_F clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2120,6 +2161,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_ProgramData_Not_A_Program_Acct_F clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2185,6 +2227,7 @@ func TestExecute_Tx_BpfLoader_Close_ProgramData_Nonclosable_Account_Failure(t *t clockAcct.Lamports = 1 execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) WriteClockSysvar(&execCtx.Accounts, clock) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidArgument, err) } @@ -2265,6 +2308,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_Success(t *testing.T) { execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2367,6 +2411,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_Extend_By_Zero_Bytes_Failure(t *test execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrInvalidInstructionData, err) } @@ -2447,6 +2492,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Not_Enou execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrNotEnoughAccountKeys, err) } @@ -2487,8 +2533,8 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Success( payerPrivateKey, err := solana.NewRandomPrivateKey() assert.NoError(t, err) payerPubkey := payerPrivateKey.PublicKey() - payerAcct := accounts.Account{Key: payerPubkey, Lamports: 10, Data: make([]byte, 0), Owner: a.SystemProgramAddr, Executable: false, RentEpoch: 100} - origPayerBalance := uint64(10) + payerAcct := accounts.Account{Key: payerPubkey, Lamports: 1_000_000, Data: make([]byte, 0), Owner: a.SystemProgramAddr, Executable: false, RentEpoch: 100} + origPayerBalance := uint64(1_000_000) // program account programPrivKey, err := solana.NewRandomPrivateKey() @@ -2540,6 +2586,7 @@ func TestExecute_Tx_BpfLoader_ExtendProgram_With_Rent_Exemption_Payment_Success( execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2666,10 +2713,11 @@ func TestExecute_Tx_BpfLoader_Upgrade_Success(t *testing.T) { rent.ExemptionThreshold = 1 rent.BurnPercent = 0 - rentAcct := accounts.Account{} + rentAcct := accounts.Account{Lamports: 1} execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2777,10 +2825,11 @@ func TestExecute_Tx_BpfLoader_Upgrade_Buffer_Wrong_Authority_Failure(t *testing. rent.LamportsPerUint8Year = 1 rent.ExemptionThreshold = 1 rent.BurnPercent = 0 - rentAcct := accounts.Account{} + rentAcct := accounts.Account{Lamports: 1} execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, InstrErrIncorrectAuthority, err) } @@ -2885,6 +2934,7 @@ func TestExecute_Tx_BpfLoader_DeployWithMaxDataLen_Success(t *testing.T) { execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) WriteRentSysvar(&execCtx.Accounts, rent) + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) @@ -2970,6 +3020,7 @@ func TestExecute_Tx_BpfLoader_Invoke_Bpf_Program_Success(t *testing.T) { execCtx.SlotCtx = new(SlotCtx) execCtx.SlotCtx.Slot = 1337 + initializeLegacyBankFixture(t, &execCtx) err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) } diff --git a/pkg/sealevel/sealevel_config_program_test.go b/pkg/sealevel/sealevel_config_program_test.go index 0941cd82c..3f33a5216 100644 --- a/pkg/sealevel/sealevel_config_program_test.go +++ b/pkg/sealevel/sealevel_config_program_test.go @@ -53,7 +53,7 @@ func TestExecute_Tx_Config_Program_Success(t *testing.T) { acct, err := txCtx.Accounts.GetAccount(1) require.NoError(t, err) - hasNewData := bytes.HasSuffix(acct.Data, []byte("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb")) + hasNewData := bytes.Equal(acct.Data[:len(instrData)], instrData) assert.Equal(t, true, hasNewData) } diff --git a/pkg/sealevel/sealevel_system_program_test.go b/pkg/sealevel/sealevel_system_program_test.go index 952461fa7..73ce90e13 100644 --- a/pkg/sealevel/sealevel_system_program_test.go +++ b/pkg/sealevel/sealevel_system_program_test.go @@ -195,7 +195,7 @@ func TestExecute_Tx_System_Program_CreateAccount_Not_Enough_Accts_Failure(t *tes WriteRentSysvar(&execCtx.Accounts, rent) err = execCtx.ProcessInstruction(instrBytes, instructionAccts, []uint64{0}) - assert.Equal(t, InstrErrNotEnoughAccountKeys, err) + assert.Equal(t, InstrErrMissingAccount, err) } func TestExecute_Tx_System_Program_CreateAccount_New_Acct_Has_Lamports_Failure(t *testing.T) { diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index c5b90ebed..41ac410be 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" _ "embed" + "encoding/binary" "encoding/json" "fmt" "io/fs" @@ -784,7 +785,7 @@ func TestInterpreter_Get_Stack_Height_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -856,7 +857,7 @@ func TestInterpreter_ReturnData_Syscalls(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -941,132 +942,13 @@ func TestInterpreter_Poseidon_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) assert.Equal(t, nil, err) } -func TestInterpreter_Get_Sysvar_Syscalls(t *testing.T) { - // program data account - programDataPrivKey, err := solana.NewRandomPrivateKey() - assert.NoError(t, err) - programDataPubkey := programDataPrivKey.PublicKey() - programDataAcctState := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgramData, ProgramData: UpgradeableLoaderStateProgramData{Slot: 0, UpgradeAuthorityAddress: nil}} - validProgramBytes := fixtures.Load(t, "sbpf", "sysvars.so") - programDataStateWriter := new(bytes.Buffer) - programDataStateEncoder := bin.NewBinEncoder(programDataStateWriter) - err = programDataAcctState.MarshalWithEncoder(programDataStateEncoder) - assert.NoError(t, err) - programDataStateWriter.Write(validProgramBytes) - programDataStateBytes := make([]byte, len(validProgramBytes)+upgradeableLoaderSizeOfProgramDataMetaData) - copy(programDataStateBytes, programDataStateWriter.Bytes()) - copy(programDataStateBytes[upgradeableLoaderSizeOfProgramDataMetaData:], validProgramBytes) - - programDataAcct := accounts.Account{Key: programDataPubkey, Lamports: 0, Data: programDataStateBytes, Owner: a.BpfLoaderUpgradeableAddr, Executable: false, RentEpoch: 100} - - // program account - programAcctState := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgram, Program: UpgradeableLoaderStateProgram{ProgramDataAddress: programDataAcct.Key}} - programWriter := new(bytes.Buffer) - programEncoder := bin.NewBinEncoder(programWriter) - err = programAcctState.MarshalWithEncoder(programEncoder) - assert.NoError(t, err) - programBytes := programWriter.Bytes() - programPrivKey, err := solana.NewRandomPrivateKey() - assert.NoError(t, err) - programPubkey := programPrivKey.PublicKey() - programData := make([]byte, 5000) - copy(programData, programBytes) - programAcct := accounts.Account{Key: programPubkey, Lamports: 10000, Data: programData, Owner: a.BpfLoaderUpgradeableAddr, Executable: true, RentEpoch: 100} - - instrData := make([]byte, 0) - - transactionAccts := NewTransactionAccounts([]accounts.Account{programAcct}) - - acctMetas := []AccountMeta{{Pubkey: programAcct.Key, IsSigner: false, IsWritable: false}} - - instructionAccts := InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) - - txCtx := NewTransactionCtx(*transactionAccts, 5, 64) - var log LogRecorder - execCtx := ExecutionCtx{Log: &log, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(10000000000)} - - execCtx.Accounts = accounts.NewMemAccounts() - var clock SysvarClock - clock.Slot = 1234 - clock.Epoch = 1111 - clock.EpochStartTimestamp = 2222 - clock.UnixTimestamp = 3 - clock.LeaderScheduleEpoch = 100000 - clockAcct := accounts.Account{} - clockAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct) - WriteClockSysvar(&execCtx.Accounts, clock) - - var rent SysvarRent - rent.LamportsPerUint8Year = 12 - rent.ExemptionThreshold = 34 - rent.BurnPercent = 56 - - rentAcct := accounts.Account{} - rentAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct) - WriteRentSysvar(&execCtx.Accounts, rent) - - var epochSchedule SysvarEpochSchedule - epochSchedule.SlotsPerEpoch = 1111 - epochSchedule.LeaderScheduleSlotOffset = 2222 - epochSchedule.Warmup = true - epochSchedule.FirstNormalEpoch = 4444 - epochSchedule.FirstNormalSlot = 5555 - - epochScheduleAcct := accounts.Account{} - epochScheduleAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarEpochScheduleAddr, &epochScheduleAcct) - WriteEpochScheduleSysvar(&execCtx.Accounts, epochSchedule) - - var lastRestartSlot SysvarLastRestartSlot - lastRestartSlot.LastRestartSlot = 989898 - lastRestartSlotAcct := accounts.Account{} - lastRestartSlotAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarLastRestartSlotAddr, &lastRestartSlotAcct) - WriteLastRestartSlotSysvar(&execCtx.Accounts, lastRestartSlot) - - var epochRewards SysvarEpochRewards - epochRewards.DistributionStartingBlockHeight = 1234 - epochRewards.NumPartitions = 4321 - copy(epochRewards.ParentBlockhash[:], "abaaaaaaaaaaaaaaaaaaaaaaaaaaaada") - epochRewards.TotalPoints.Lo = 0xffffffffffffffff - epochRewards.TotalPoints.Hi = 0xeeeeeeeeeeeeeeee - epochRewards.TotalRewards = 5656 - epochRewards.DistributedRewards = 6767 - epochRewards.Active = false - epochRewardsAcct := accounts.Account{} - epochRewardsAcct.Lamports = 1 - execCtx.Accounts.SetAccount(&SysvarEpochRewardsAddr, &epochRewardsAcct) - WriteEpochRewardsSysvar(&execCtx.Accounts, epochRewards) - - f := features.NewFeaturesDefault() - f.EnableFeature(features.LastRestartSlotSysvar, 0) - f.EnableFeature(features.EnablePartitionedEpochReward, 0) - execCtx.Features = *f - - pk := [32]byte(programDataAcct.Key) - err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) - assert.NoError(t, err) - - execCtx.SlotCtx = new(SlotCtx) - execCtx.SlotCtx.Slot = 1337 - - err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, nil, err) - - for _, l := range log.Logs { - fmt.Printf("log: %s\n", l) - } -} - func TestInterpreter_AltBn128_Ops_Syscall(t *testing.T) { // program data account programDataPrivKey, err := solana.NewRandomPrivateKey() @@ -1111,7 +993,7 @@ func TestInterpreter_AltBn128_Ops_Syscall(t *testing.T) { var log LogRecorder execCtx := ExecutionCtx{Log: &log, TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(10000000000)} - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 execCtx.SlotCtx.Accounts = accounts.NewMemAccounts() @@ -1237,7 +1119,7 @@ func TestInterpreter_Alloc_Free_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1301,7 +1183,7 @@ func TestInterpreter_Alt_Bn128_Compression_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1365,7 +1247,7 @@ func TestInterpreter_Validate_Point_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1429,7 +1311,7 @@ func TestInterpreter_Curve_Group_Ops_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1493,7 +1375,7 @@ func TestInterpreter_Curve_Multiscalar_Mul_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1557,7 +1439,7 @@ func TestInterpreter_Log_Data_Syscall(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1632,7 +1514,7 @@ func TestInterpreter_Cpi_C_System_Program_Allocate(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1712,7 +1594,7 @@ func TestInterpreter_Cpi_Rust_System_Program_Allocate(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1794,7 +1676,7 @@ func TestInterpreter_Cpi_C_Bpf_Program_Call(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1886,7 +1768,7 @@ func executeFirstBpfProgramAndReturnExecCtx(t *testing.T, log *LogRecorder, acct err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -1962,14 +1844,21 @@ func TestInterpreter_Get_Processed_Sibling_Instruction_Test(t *testing.T) { fmt.Printf("******** second program call is %s\n", programAcct.Key) + // The introspection ELF only reads siblings; it does not invoke Allocate. + // Execute that preceding sibling explicitly before asking for indices 0/1. + allocateData := make([]byte, 12) + binary.LittleEndian.PutUint32(allocateData, SystemProgramInstrTypeAllocate) + binary.LittleEndian.PutUint64(allocateData[4:], 16) + allocateAccounts := InstructionAcctsFromAccountMetas([]AccountMeta{{Pubkey: acctToAlloc.Key, IsSigner: true, IsWritable: true}}, execCtx.TransactionContext.Accounts) + require.NoError(t, execCtx.ProcessInstruction(allocateData, allocateAccounts, []uint64{2})) + err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{1}) - assert.NoError(t, err) + require.NoError(t, err) - // test that the program logs from the CPI'd program (which calls get_processed_sibling_instruction) - // are as expected - expected := fmt.Sprintf("Program log: ******** sibling instruction 0 program id: %s", programAcct.Key) + // Check the introspection program reports both completed top-level siblings. + expected := fmt.Sprintf("Program log: ******** sibling instruction 0 program id: %s", solana.PublicKey(a.SystemProgramAddr)) assert.Equal(t, expected, log.Logs[1]) - expected = fmt.Sprintf("Program log: ******** sibling instruction 0 instruction data: %s", reformatHexBytes(instrData)) + expected = fmt.Sprintf("Program log: ******** sibling instruction 0 instruction data: %s", reformatHexBytes(allocateData)) assert.Equal(t, expected, log.Logs[2]) expected = fmt.Sprintf("Program log: ******** sibling instruction 1 program id: %s", firstProgramAcct.Key) @@ -2015,7 +1904,7 @@ func TestInterpreter_Test_Memo_Program_With_LoaderV2(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -2032,7 +1921,7 @@ func TestInterpreter_Test_Memo_Program_With_LoaderV2(t *testing.T) { instrData[1] = 0xff err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) - assert.Equal(t, nil, err) + assert.Equal(t, InstrErrInvalidInstructionData, err) expected = fmt.Sprintf("Program log: Signed by %s", signerPubkey) containsExpected = strings.HasPrefix(log.Logs[2], expected) @@ -2090,7 +1979,7 @@ func TestInterpreter_Test_Deprecated_Loader(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) @@ -2135,6 +2024,9 @@ func (e *executeCase) run(t *testing.T) { tx.PushInstructionCtx(InstructionCtx{}) opts := tx.newVMOpts(&e.Params) opts.Tracer = testLogger{t} + ctx := opts.Context.(*ExecutionCtx) + ctx.ComputeMeter = cu.NewComputeMeter(uint64(opts.MaxCU)) + opts.ComputeMeter = &ctx.ComputeMeter interpreter := sbpf.NewInterpreter(program, opts) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/spl_token_demo_test.go b/pkg/sealevel/spl_token_demo_test.go index c0ede8c86..0a6cc6441 100644 --- a/pkg/sealevel/spl_token_demo_test.go +++ b/pkg/sealevel/spl_token_demo_test.go @@ -24,7 +24,7 @@ var splTokenProgramAddr = base58.MustDecodeFromString("TokenkegQfeZyiNwAJbNbGKPF // spl token program later. func setupSplTokenProgramAccount(t *testing.T, accts *accounts.Accounts) accounts.Account { programBytes := fixtures.Load(t, "sbpf", "spl-token.so") - splTokenAcct := accounts.Account{Key: splTokenProgramAddr, Lamports: 0, Data: programBytes, Owner: a.BpfLoader2Addr, Executable: true, RentEpoch: 100} + splTokenAcct := accounts.Account{Key: splTokenProgramAddr, Lamports: 1, Data: programBytes, Owner: a.BpfLoader2Addr, Executable: true, RentEpoch: 100} pk := [32]byte(splTokenProgramAddr) err := (*accts).SetAccount(&pk, &splTokenAcct) @@ -201,6 +201,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { {Pubkey: SysvarRentAddr, IsSigner: false, IsWritable: false}} instructionAccts := InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeMint: execute SPL token InitializeMint instruction err := execCtx.ProcessInstruction(initMintInstrData, instructionAccts, []uint64{0}) @@ -231,6 +232,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeAccount: execute SPL token InitializeMint instruction err = execCtx.ProcessInstruction(initAccountInstrData, instructionAccts, []uint64{0}) @@ -260,6 +262,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // InitializeAccount: execute SPL token InitializeMint instruction err = execCtx.ProcessInstruction(initAccountInstrData, instructionAccts, []uint64{0}) @@ -281,6 +284,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // MintTo: serialize up a MintTo instruction numTokensToMint := uint64(61616161) @@ -306,6 +310,7 @@ func Test_Spl_Token_Program_Demo(t *testing.T) { instructionAccts = InstructionAcctsFromAccountMetas(acctMetas, *transactionAccts) execCtx.TransactionContext = NewTransactionCtx(*transactionAccts, 5, 64) + initializeLegacyBankFixture(t, execCtx) // Transfer: serialize up a Transfer instruction numTokensToTransfer := uint64(1337) diff --git a/pkg/sealevel/sysvar_instructions_test.go b/pkg/sealevel/sysvar_instructions_test.go index 568fa8d98..67fb040cd 100644 --- a/pkg/sealevel/sysvar_instructions_test.go +++ b/pkg/sealevel/sysvar_instructions_test.go @@ -113,7 +113,7 @@ func TestExecute_Tx_Sysvar_Instructions_Bpf_Test(t *testing.T) { err = execCtx.Accounts.SetAccount(&pk, &programDataAcct) assert.NoError(t, err) - execCtx.SlotCtx = new(SlotCtx) + initializeLegacyBankFixture(t, &execCtx) execCtx.SlotCtx.Slot = 1337 err = execCtx.ProcessInstruction(instrData, instructionAccts, []uint64{0}) From 02da5330f7f2502752119a6c8916b7a24e687609 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:28:40 -0500 Subject: [PATCH 075/111] test: compare streaming replay across program and account mutations --- pkg/replay/streaming_lifecycle_test.go | 1 + pkg/replay/streaming_program_mix_test.go | 193 +++++++++++++++++++++++ 2 files changed, 194 insertions(+) create mode 100644 pkg/replay/streaming_program_mix_test.go diff --git a/pkg/replay/streaming_lifecycle_test.go b/pkg/replay/streaming_lifecycle_test.go index 63f466283..6ee885ef8 100644 --- a/pkg/replay/streaming_lifecycle_test.go +++ b/pkg/replay/streaming_lifecycle_test.go @@ -166,6 +166,7 @@ func newLifecycleEnv(t *testing.T) *lifecycleEnv { func (env *lifecycleEnv) block(txs []*solana.Transaction) *b.Block { return &b.Block{ Slot: lifecycleSlot, + VoteTimestamps: make(map[solana.PublicKey]sealevel.BlockTimestamp), Epoch: 0, ParentSlot: lifecycleParentSlot, ParentBankhash: [32]byte{0x88}, diff --git a/pkg/replay/streaming_program_mix_test.go b/pkg/replay/streaming_program_mix_test.go new file mode 100644 index 000000000..3391eb15f --- /dev/null +++ b/pkg/replay/streaming_program_mix_test.go @@ -0,0 +1,193 @@ +package replay + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/tpu/txfixture" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/gagliardetto/solana-go/programs/system" + "github.com/stretchr/testify/require" +) + +func signLifecycleInstructions(t *testing.T, instructions ...solana.Instruction) *solana.Transaction { + t.Helper() + tx, err := solana.NewTransaction(instructions, txfixture.TestBlockhash(), solana.TransactionPayer(txfixture.PayerPubkey())) + require.NoError(t, err) + key := txfixture.PayerPrivateKey() + _, err = tx.Sign(func(pk solana.PublicKey) *solana.PrivateKey { + if pk == key.PublicKey() { + return &key + } + return nil + }) + require.NoError(t, err) + wire, err := tx.MarshalBinary() + require.NoError(t, err) + decoded, err := solana.TransactionFromBytes(wire) + require.NoError(t, err) + return decoded +} + +// Each case has a real write consumed or replaced in a subsequent group. Fresh +// wire decoding avoids accidentally sharing resolved transactions across paths. +func TestStreamingLifecycleProgramMutations(t *testing.T) { + previous := StreamingExecutionCfg + StreamingExecutionCfg = StreamingExecutionConfig{Enabled: true} + t.Cleanup(func() { StreamingExecutionCfg = previous }) + payer := txfixture.PayerPubkey() + key := solana.PublicKey{0xC1} + previousVote := global.VoteCacheItem(key) + t.Cleanup(func() { + if previousVote == nil { + global.DeleteVoteCacheItem(key) + } else { + global.PutVoteCacheItem(key, previousVote) + } + }) + for _, kind := range []string{"nonce", "lookup extension", "vote commission", "program upgrade", "program deployment"} { + t.Run(kind, func(t *testing.T) { + setup := func() (*lifecycleEnv, []*solana.Transaction) { + env := newLifecycleEnv(t) + env.acctsDb.InitCaches() + t.Cleanup(env.acctsDb.ProgramCache.Close) + t.Cleanup(env.acctsDb.VoteAcctCache.Close) + t.Cleanup(env.acctsDb.CommonAcctsCache.Close) + put := func(pk solana.PublicKey, owner solana.PublicKey, data []byte, executable bool) { + require.NoError(t, env.durable.SetAccountWithoutLock(pk, &accounts.Account{Key: pk, Owner: owner, Lamports: 100_000_000, Data: data, Executable: executable, RentEpoch: ^uint64(0)})) + } + put(payer, addresses.SystemProgramAddr, nil, false) + native := func(pk solana.PublicKey) { put(pk, addresses.NativeLoaderAddr, nil, true) } + var txs []*solana.Transaction + switch kind { + case "nonce": + state := sealevel.NonceStateVersions{Type: sealevel.NonceVersionCurrent, Current: sealevel.NonceData{IsInitialized: true, Authority: payer, DurableNonce: [32]byte{0xAA}, FeeCalculator: sealevel.FeeCalculator{LamportsPerSignature: 5000}}} + data, err := state.Marshal() + require.NoError(t, err) + put(key, addresses.SystemProgramAddr, data, false) + // Repeated advances have different messages but the same nonce account; + // only the first may advance in this bank. Later instructions must see it. + for i := uint64(1); i <= 3; i++ { + txs = append(txs, signLifecycleInstructions(t, system.NewAdvanceNonceAccountInstruction(key, solana.SysVarRecentBlockHashesPubkey, payer).Build(), system.NewTransferInstruction(i, payer, txfixture.DestPubkey()).Build())) + } + case "lookup extension": + native(addresses.AddressLookupTableAddr) + table := lookupTableAccount(t, txfixture.DestPubkey()) + table.Lamports = 100_000_000 + require.NoError(t, env.durable.SetAccountWithoutLock(lifecycleTableKey, table)) + for i := byte(1); i <= 3; i++ { + instruction := sealevel.AddrLookupTableInstrExtendLookupTable{NewAddresses: []solana.PublicKey{{0xD2, i}}} + var b bytes.Buffer + require.NoError(t, instruction.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.AddressLookupTableAddr, solana.AccountMetaSlice{solana.Meta(lifecycleTableKey).WRITE(), solana.Meta(payer).SIGNER()}, b.Bytes()))) + } + case "vote commission": + global.DeleteVoteCacheItem(key) + native(addresses.VoteProgramAddr) + state := sealevel.VoteStateVersions{Type: sealevel.VoteStateVersionCurrent, Current: sealevel.VoteState{AuthorizedWithdrawer: payer, Commission: 30}} + var b bytes.Buffer + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&b))) + data := make([]byte, sealevel.VoteStateV3Size) + copy(data, b.Bytes()) + put(key, addresses.VoteProgramAddr, data, false) + for _, commission := range []byte{20, 10, 5} { + data := binary.LittleEndian.AppendUint32(nil, sealevel.VoteProgramInstrTypeUpdateCommission) + data = append(data, commission) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.VoteProgramAddr, solana.AccountMetaSlice{solana.Meta(key).WRITE(), solana.Meta(payer).SIGNER()}, data))) + } + case "program upgrade", "program deployment": + native(addresses.BpfLoaderUpgradeableAddr) + programDataKey := solana.PublicKey{0xC2} + bufferKey := solana.PublicKey{0xC3} + elf := fixtures.Load(t, "sbpf", "noop_aligned.so") + encode := func(state sealevel.UpgradeableLoaderState, size int) []byte { + var b bytes.Buffer + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&b))) + out := make([]byte, size) + copy(out, b.Bytes()) + return out + } + put(key, addresses.BpfLoaderUpgradeableAddr, encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeProgram, Program: sealevel.UpgradeableLoaderStateProgram{ProgramDataAddress: programDataKey}}, 36), true) + pd := encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeProgramData, ProgramData: sealevel.UpgradeableLoaderStateProgramData{Slot: 1, UpgradeAuthorityAddress: &payer}}, 45+len(elf)) + copy(pd[45:], elf) + put(programDataKey, addresses.BpfLoaderUpgradeableAddr, pd, false) + buf := encode(sealevel.UpgradeableLoaderState{Type: sealevel.UpgradeableLoaderStateTypeBuffer, Buffer: sealevel.UpgradeableLoaderStateBuffer{AuthorityAddress: &payer}}, 37+len(elf)) + copy(buf[37:], elf) + put(bufferKey, addresses.BpfLoaderUpgradeableAddr, buf, false) + if kind == "program deployment" { + programDataKey, _, err := solana.FindProgramAddress([][]byte{key[:]}, addresses.BpfLoaderUpgradeableAddr) + require.NoError(t, err) + put(key, addresses.BpfLoaderUpgradeableAddr, make([]byte, 36), false) + // No pre-funded PDA: Deploy creates it with a signed CPI. + require.NoError(t, env.durable.SetAccountWithoutLock(programDataKey, &accounts.Account{Key: programDataKey, Owner: addresses.SystemProgramAddr})) + write := sealevel.UpgradeableLoaderInstrWrite{Offset: 0, Bytes: elf[:8]} + var b bytes.Buffer + require.NoError(t, write.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(bufferKey).WRITE(), solana.Meta(payer).SIGNER()}, b.Bytes()))) + deploy := sealevel.UpgradeableLoaderInstrDeployWithMaxDataLen{MaxDataLen: uint64(len(elf))} + b.Reset() + require.NoError(t, deploy.MarshalWithEncoder(bin.NewBinEncoder(&b))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(payer).WRITE().SIGNER(), solana.Meta(programDataKey).WRITE(), solana.Meta(key).WRITE(), solana.Meta(bufferKey).WRITE(), solana.Meta(sealevel.SysvarRentAddr), solana.Meta(sealevel.SysvarClockAddr), solana.Meta(addresses.SystemProgramAddr), solana.Meta(payer).SIGNER()}, b.Bytes()))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{2}))) + break + } + + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{1}))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(addresses.BpfLoaderUpgradeableAddr, solana.AccountMetaSlice{solana.Meta(programDataKey).WRITE(), solana.Meta(key).WRITE(), solana.Meta(bufferKey).WRITE(), solana.Meta(payer).WRITE(), solana.Meta(sealevel.SysvarRentAddr), solana.Meta(sealevel.SysvarClockAddr), solana.Meta(payer).SIGNER()}, binary.LittleEndian.AppendUint32(nil, sealevel.UpgradeableLoaderInstrTypeUpgrade)))) + txs = append(txs, signLifecycleInstructions(t, solana.NewInstruction(key, nil, []byte{2}))) + } + return env, txs + } + env, txs := setup() + whole := lifecycleWholeBlock(t, env, txs, 2) + switch kind { + case "nonce": + require.Contains(t, whole.delta, key) + state, err := sealevel.UnmarshalNonceStateVersions(whole.delta[key].Data) + require.NoError(t, err) + require.NotEqual(t, [32]byte{0xAA}, state.State().DurableNonce) + case "lookup extension": + require.Contains(t, whole.delta, lifecycleTableKey) + require.Len(t, whole.delta[lifecycleTableKey].Data, sealevel.AddressLookupTableMetaSize+4*32) + case "vote commission": + require.Contains(t, whole.delta, key) + state, err := sealevel.UnmarshalVersionedVoteState(whole.delta[key].Data) + require.NoError(t, err) + require.Equal(t, byte(5), state.ConvertToCurrent().Commission) + case "program upgrade", "program deployment": + pk := solana.PublicKey{0xC2} + if kind == "program deployment" { + var err error + pk, _, err = solana.FindProgramAddress([][]byte{key[:]}, addresses.BpfLoaderUpgradeableAddr) + require.NoError(t, err) + require.True(t, whole.delta[key].Executable) + } + require.Contains(t, whole.delta, pk) + state, err := sealevel.UnmarshalUpgradeableLoaderState(whole.delta[pk].Data) + require.NoError(t, err) + require.Equal(t, lifecycleSlot, state.ProgramData.Slot) + } + for _, workers := range []int{1, 4} { + for suffix := 0; suffix <= 3; suffix++ { + t.Run(fmt.Sprintf("workers%d/suffix%d", workers, suffix), func(t *testing.T) { + env, txs := setup() + var splits []int + for i := 1; i < 3-suffix; i++ { + splits = append(splits, i) + } + got, _ := lifecycleStream(t, env, txs, splits, suffix, workers) + requireSameLifecycleOutcome(t, whole, got) + }) + } + } + }) + } +} From 26a6a812dda9bc205e5754169738ed9d5e22e16b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:00:21 -0500 Subject: [PATCH 076/111] sealevel: track sibling header writes in pooled VM memory --- pkg/sealevel/syscalls_call.go | 2 +- pkg/sealevel/syscalls_sibling_pool_test.go | 42 ++++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) create mode 100644 pkg/sealevel/syscalls_sibling_pool_test.go diff --git a/pkg/sealevel/syscalls_call.go b/pkg/sealevel/syscalls_call.go index b594ad2a5..193f44baa 100644 --- a/pkg/sealevel/syscalls_call.go +++ b/pkg/sealevel/syscalls_call.go @@ -170,7 +170,7 @@ func SyscallGetProcessedSiblingInstructionImpl(vm sbpf.VM, index, metaAddr, prog } if instrCtxFound != nil { - resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, false) + resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, true) if err != nil { return syscallErr(err) } diff --git a/pkg/sealevel/syscalls_sibling_pool_test.go b/pkg/sealevel/syscalls_sibling_pool_test.go new file mode 100644 index 000000000..a3b3b9e22 --- /dev/null +++ b/pkg/sealevel/syscalls_sibling_pool_test.go @@ -0,0 +1,42 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func TestSiblingHeaderWriteIsClearedOnFinish(t *testing.T) { + old := sbpf.UsePool + sbpf.UsePool = true + t.Cleanup(func() { sbpf.UsePool = old }) + for _, addr := range []uint64{sbpf.VaddrStack + 128, sbpf.VaddrHeap + 128} { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(100000), TransactionContext: &TransactionCtx{ + InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}, + }} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram}, &sbpf.VMOpts{HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + // Keep a read-only view so the test itself does not mark the header dirty. + header, err := vm.Translate(addr, ProcessedSiblingInstructionSize, false) + require.NoError(t, err) + require.Equal(t, make([]byte, 16), header) + found, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, addr, 0, 0, 0) + require.NoError(t, err) + require.Equal(t, uint64(1), found) + require.Equal(t, byte(3), header[0]) + vm.Finish() + // Inspect before allocating another VM: this is deterministic and does not + // depend on sync.Pool choosing a particular backing buffer on the next Get. + require.Equal(t, make([]byte, 16), header) + } +} + +func TestSiblingHeaderRejectsReadOnlyDestination(t *testing.T) { + data := make([]byte, 16) + vm, ctx := newMemSyscallVM(t, nil, []sbpf.InputRegion{{RegionSize: 16, AddressSpaceReserved: 16, Data: data}}) + ctx.TransactionContext = &TransactionCtx{InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}} + _, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, sbpf.VaddrInput, 0, 0, 0) + require.Error(t, err) + require.Equal(t, make([]byte, 16), data) +} From 8238acd9e7c3721e3df8a0d4ed7f3c7bfb47f737 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:00:21 -0500 Subject: [PATCH 077/111] replay: preserve independent banks and pending fork switches --- pkg/global/global_ctx.go | 14 ++++++++++-- pkg/replay/alpenglow_switch.go | 10 +++++++++ pkg/replay/alpenglow_switch_test.go | 13 +++++++++++ pkg/replay/alpenglow_unwind_test.go | 24 ++++++++++++++++++++ pkg/replay/block.go | 4 +++- pkg/replay/block_execution.go | 35 +++++++++++++++++++++++------ pkg/replay/block_execution_test.go | 22 ++++++++++++++++++ pkg/replay/promotion.go | 8 +++---- pkg/replay/streaming.go | 6 ++--- pkg/replay/streaming_test.go | 17 ++++++++++++++ 10 files changed, 135 insertions(+), 18 deletions(-) diff --git a/pkg/global/global_ctx.go b/pkg/global/global_ctx.go index f8241f771..d99d9c5c6 100644 --- a/pkg/global/global_ctx.go +++ b/pkg/global/global_ctx.go @@ -104,9 +104,19 @@ func PendingStakeEntriesSnapshot() []accountsdb.StakeIndexEntry { return out } +// DropPendingStakePubkeys drops only the discarded bank's slot. A local leader +// can have pending entries at later slots even while replay is behind it. +func DropPendingStakePubkeys(slot uint64) int { + instance.pendingStakeMutex.Lock() + defer instance.pendingStakeMutex.Unlock() + dropped := len(instance.pendingStakeBySlot[slot]) + delete(instance.pendingStakeBySlot, slot) + return dropped +} + // DropPendingStakePubkeysFrom discards pending entries for slots >= fromSlot. -// Called by the fork-switch unwind so wrong-fork stake entries never reach the -// durable index. Returns the number of entries dropped. +// This is a global reset operation. Per-bank discard and replay unwind must +// use DropPendingStakePubkeys so they preserve independent leader banks. func DropPendingStakePubkeysFrom(fromSlot uint64) int { instance.pendingStakeMutex.Lock() defer instance.pendingStakeMutex.Unlock() diff --git a/pkg/replay/alpenglow_switch.go b/pkg/replay/alpenglow_switch.go index f025ae410..2d384a096 100644 --- a/pkg/replay/alpenglow_switch.go +++ b/pkg/replay/alpenglow_switch.go @@ -117,6 +117,16 @@ func newAlpenglowSwitchSweeper(engine consensusengine.Engine) *alpenglowSwitchSw return s } +// peek tests for a switch without consuming the sweep's version/frontier gate. +// Streaming admission must leave a detected switch for the replay loop to apply. +func (s *alpenglowSwitchSweeper) peek(executed map[uint64]solana.Hash, lastRooted, tip uint64) *CertifiedSwitch { + if s == nil { + return nil + } + snapshot := *s + return snapshot.sweep(executed, lastRooted, tip) +} + // sweep walks consumed block/skip outcomes in (lastRooted, tip] and returns // the first contradiction with a decisive chain decision. tip includes trailing // skips even when the executed bank remains at an earlier slot. diff --git a/pkg/replay/alpenglow_switch_test.go b/pkg/replay/alpenglow_switch_test.go index cd8084682..35cd7ca17 100644 --- a/pkg/replay/alpenglow_switch_test.go +++ b/pkg/replay/alpenglow_switch_test.go @@ -455,3 +455,16 @@ func TestWaitForAlpenglowReplayInputHonorsCancellation(t *testing.T) { require.Nil(t, parentSwitch) require.Nil(t, certifiedSwitch) } + +func TestSwitchPeekDoesNotConsumeDecision(t *testing.T) { + q := &fakeChainQuery{certified: map[uint64]alpenglow.BlockID{101: {Slot: 101, Hash: swHash(9)}}, skipped: map[uint64]bool{}, version: 1} + s := newTestSweeper(q) + executed := map[uint64]solana.Hash{101: swHash(1)} + before := *s + first := s.peek(executed, 100, 101) + require.NotNil(t, first) + require.Equal(t, before, *s) + require.Equal(t, first, s.peek(executed, 100, 101)) + require.Equal(t, first, s.sweep(executed, 100, 101), "replay still receives the switch") + require.Nil(t, s.sweep(executed, 100, 101), "normal sweep still consumes its gate") +} diff --git a/pkg/replay/alpenglow_unwind_test.go b/pkg/replay/alpenglow_unwind_test.go index dd478e0fc..52e949c2b 100644 --- a/pkg/replay/alpenglow_unwind_test.go +++ b/pkg/replay/alpenglow_unwind_test.go @@ -7,10 +7,12 @@ import ( "testing" "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/global" "github.com/Overclock-Validator/mithril/pkg/rewards" "github.com/Overclock-Validator/mithril/pkg/sealevel" "github.com/Overclock-Validator/mithril/pkg/state" bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" "github.com/mr-tron/base58" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -373,3 +375,25 @@ func assertUnwindFallbackReason(t *testing.T, want string, sw *CertifiedSwitch, assert.Nil(t, bankSysvars) assert.Equal(t, want, reason) } + +func TestUnwindPreservesPendingLeaderStakeEntries(t *testing.T) { + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + for _, slot := range []uint64{8, 9, 12} { + global.EnqueuePendingStakePubkey(slot, solana.PublicKey{byte(slot), 0xFA}) + t.Cleanup(func() { global.DropPendingStakePubkeys(slot) }) + } + for _, slot := range []uint64{8, 9} { + tail.Add(slot, []*accounts.Account{testAccount(1, slot)}, testHashBytes(byte(slot))) + } + tail.unwind(8) + entries := global.PendingStakeEntriesSnapshot() + for _, slot := range []uint64{8, 9, 12} { + found := false + for _, entry := range entries { + if entry.Pubkey == (solana.PublicKey{byte(slot), 0xFA}) { + found = true + } + } + require.Equal(t, slot == 12, found) + } +} diff --git a/pkg/replay/block.go b/pkg/replay/block.go index ec305a9d9..f71df8c98 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -2553,7 +2553,9 @@ func ReplayBlocks( rewardsInFlight: func() bool { return partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 }, - switchPending: func() bool { return sweepWhileWaiting != nil && sweepWhileWaiting() != nil }, + switchPending: func() bool { + return switchSweeper.peek(alpenglowExecutedBlockIDs, mithrilState.LastRootedSlot, replayFrontier) != nil + }, executedBlockID: func(slot uint64) (solana.Hash, bool) { if id, ok := alpenglowExecutedBlockIDs[slot]; ok { return id, true diff --git a/pkg/replay/block_execution.go b/pkg/replay/block_execution.go index 0a5ebfaa3..6e72c1388 100644 --- a/pkg/replay/block_execution.go +++ b/pkg/replay/block_execution.go @@ -444,7 +444,7 @@ func (exec *blockExecution) loadTransactionAccounts(view *b.Block) error { // runTransactionGroup is parallelTxLoop over a transaction slice with a plan // built for the group alone (indices are group-local). Without the planner -// (txParallelism == 0, or an unresolvable lookup) it runs sequentially, which +// (txParallelism <= 1) it runs sequentially, which // is always correct because groups are consumed in block order. func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execute []bool, shouldVerifySignatures bool) ([]*fees.TxFeeInfo, []uint64, error) { slotCtx := exec.slotCtx @@ -456,13 +456,15 @@ func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execu if workers > len(txs) { workers = len(txs) } + view := &b.Block{Transactions: txs} + if !canUseDependencyPlanner(view) { + return nil, nil, errors.New("streaming group has unresolved address tables") + } var plan *dependencyPlan if workers > 1 { plannerBuildStart := time.Now() - plannerAccounts, available := plannerAccountsForBlock(&b.Block{Transactions: txs}) - if available { - plan = buildDependencyPlan(plannerAccounts) - } + plannerAccounts, _ := plannerAccountsForBlock(view) + plan = buildDependencyPlan(plannerAccounts) metrics.GlobalBlockReplay.DependencyPlannerBuild.AddTimingSince(plannerBuildStart) } if plan == nil { @@ -470,12 +472,17 @@ func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execu if !execute[idx] { continue } - feeInfos[idx], computeUnits[idx], _ = ProcessTransaction(slotCtx, &exec.sigverifyWg, tx, nil, dbgOpts, nil, shouldVerifySignatures) + var txErr error + feeInfos[idx], computeUnits[idx], txErr = ProcessTransaction(slotCtx, &exec.sigverifyWg, tx, nil, dbgOpts, nil, shouldVerifySignatures) + if feeInfos[idx] == nil { + return nil, nil, streamingTransactionError(idx, txErr) + } } return feeInfos, computeUnits, nil } metrics.GlobalBlockReplay.DependencyPlannerPrepared = 1 + txErrors := make([]error, len(txs)) do := make(chan int, len(txs)) done := make(chan int, len(txs)) plannerDone := make(chan struct{}) @@ -500,7 +507,7 @@ func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execu done <- idx continue } - feeInfos[idx], computeUnits[idx], _ = ProcessTransaction(slotCtx, &exec.sigverifyWg, txs[idx], nil, dbgOpts, workerArena, shouldVerifySignatures) + feeInfos[idx], computeUnits[idx], txErrors[idx] = ProcessTransaction(slotCtx, &exec.sigverifyWg, txs[idx], nil, dbgOpts, workerArena, shouldVerifySignatures) done <- idx } }(i) @@ -508,9 +515,23 @@ func (exec *blockExecution) runTransactionGroup(txs []*solana.Transaction, execu wg.Wait() close(done) <-plannerDone + for idx := range txs { + if execute[idx] && feeInfos[idx] == nil { + return nil, nil, streamingTransactionError(idx, txErrors[idx]) + } + } return feeInfos, computeUnits, nil } +// Instruction failures still carry charged fees and remain valid block entries. +// A missing fee result means transaction admission failed: discard the bank. +func streamingTransactionError(index int, err error) error { + if err == nil { + err = errors.New("missing fee result") + } + return fmt.Errorf("unprocessable streaming transaction %d: %w", index, err) +} + // reportNilFeeInfo reproduces ProcessBlock's diagnostic for a transaction whose // fee information is missing, which only happens when blockhash validation // failed for a transaction the block claims to have processed. diff --git a/pkg/replay/block_execution_test.go b/pkg/replay/block_execution_test.go index 5bcff38ba..31556fdf2 100644 --- a/pkg/replay/block_execution_test.go +++ b/pkg/replay/block_execution_test.go @@ -309,3 +309,25 @@ func TestExecuteTransactionGroupRefusesClosedExecution(t *testing.T) { err := env.exec.executeTransactionGroup(transferTransactions(t, 1, 300), nil, false) require.ErrorIs(t, err, errBlockExecutionClosed) } + +func TestStreamingGroupRejectsUnprocessableTransactions(t *testing.T) { + for _, workers := range []int{0, 4} { + env := newGroupExecutionEnv(t, workers, 10_000_000) + defer env.cleanup() + txs := transferTransactions(t, 2, 1) + txs[0].Message.RecentBlockhash = solana.Hash{0xFA} + err := env.exec.executeTransactionGroup(txs, nil, false) + require.ErrorIs(t, err, TxErrInvalidBlockhash) + } +} + +func TestStreamingGroupRejectsUnresolvedLookupsBeforeExecution(t *testing.T) { + for _, workers := range []int{0, 4} { + env := newGroupExecutionEnv(t, workers, 10_000_000) + defer env.cleanup() + txs := decodeTransactions(t, [][]byte{signedV0TransferViaTableWire(t, 1)}) + err := env.exec.executeTransactionGroup(txs, nil, false) + require.ErrorContains(t, err, "unresolved address tables") + require.Zero(t, env.exec.slotCtx.TotalComputeUnitsConsumed) + } +} diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index 5472ec4ed..f6cedc25a 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -610,13 +610,11 @@ func (p *asyncPromoter) stop() { // the caller validates the pair and falls back to rooted-checkpoint re-replay. func (t *unrootedTail) unwind(fromSlot uint64) (*state.ResumeContext, *sealevel.BankSysvars) { t.overlay.EvictFrom(fromSlot) - // Branch-scoped side effect: stake pubkeys enqueued by the evicted slots - // must never reach the durable index — drop them with the state. - if dropped := global.DropPendingStakePubkeysFrom(fromSlot); dropped > 0 { - mlog.Log.Infof("fork unwind: dropped %d pending stake-index entries from slots >= %d", dropped, fromSlot) - } + // Only replay-owned held slots are unwound. A future local leader bank + // is not part of this tail and must retain its pending stake entries. for s := range t.bankhashes { if s >= fromSlot { + global.DropPendingStakePubkeys(s) delete(t.bankhashes, s) } } diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index 5c4944492..98e800c4e 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -813,14 +813,14 @@ func (s *streamingExecutor) discard(reason string) { } } // Deferred vote-cache changes die with the SlotCtx; the stake index - // entries are keyed by slot and the stream is the only bank above - // the frontier. + // entries belong to this slot; a concurrent leader bank may own + // entries at later slots. exec.slotCtx.PendingVoteCache = nil exec.slotCtx.PendingVoteCacheDeletes = nil exec.slotCtx.VoteStakeDirty = false } } - global.DropPendingStakePubkeysFrom(cur.slot) + global.DropPendingStakePubkeys(cur.slot) if cur.restoreSysvarCache != nil { cur.restoreSysvarCache() } diff --git a/pkg/replay/streaming_test.go b/pkg/replay/streaming_test.go index d2cba3027..d716e84a3 100644 --- a/pkg/replay/streaming_test.go +++ b/pkg/replay/streaming_test.go @@ -1178,3 +1178,20 @@ func TestStreamingHeaderAdmissionBounds(t *testing.T) { s.rememberHeader(turbine.NewDetachedStreamMarker(g, 0, 0, turbine.StreamMarkerHeader, frontier, solana.Hash{})) require.Len(t, s.headers, 1, "distance comparison must not overflow") } + +func TestStreamingDiscardPreservesLaterLeaderStakeEntries(t *testing.T) { + h := newFinalizeFailureHarness(t) + leaderSlot := h.exec.current.slot + 4 + leaderStake := solana.PublicKey{0xF7, 0xD1} + global.EnqueuePendingStakePubkey(leaderSlot, leaderStake) + t.Cleanup(func() { global.DropPendingStakePubkeys(leaderSlot) }) + h.exec.discard("test") + entries := global.PendingStakeEntriesSnapshot() + found := false + for _, entry := range entries { + if entry.Pubkey == leaderStake { + found = true + } + } + require.True(t, found, "discard must not remove the independent leader bank's stake entries") +} From 868524b9efbfe2e457c0d6393e68eea64cfb45d8 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:00:21 -0500 Subject: [PATCH 078/111] turbine: cancel completed streams when delivery is abandoned --- pkg/turbine/assembler.go | 9 +++++---- pkg/turbine/completion_pool.go | 9 +++++---- pkg/turbine/receiver.go | 20 ++++++++++++++------ pkg/turbine/repair_followup.go | 4 ++++ pkg/turbine/stream.go | 17 +++++++++++++++++ pkg/turbine/stream_test.go | 31 +++++++++++++++++++++++++++++++ 6 files changed, 76 insertions(+), 14 deletions(-) diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 3c043af0e..9a1516f29 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -166,10 +166,11 @@ type processedSlotCompletion struct { } type slotCompletionResult struct { - block *block.Block - err error - hydrated bool - pending bool + generation StreamGeneration + block *block.Block + err error + hydrated bool + pending bool } type fecLayout struct { diff --git a/pkg/turbine/completion_pool.go b/pkg/turbine/completion_pool.go index 451b0095c..e493b681d 100644 --- a/pkg/turbine/completion_pool.go +++ b/pkg/turbine/completion_pool.go @@ -83,10 +83,11 @@ func newSlotCompletionPool(assembler *SlotAssembler, resetGate *sync.RWMutex, on p.resetGate.RUnlock() } p.results <- slotCompletionResult{ - block: blk, - err: err, - hydrated: queued.hydrated, - pending: pending, + generation: StreamGeneration{slot: queued.work.state.slot, state: queued.work.state}, + block: blk, + err: err, + hydrated: queued.hydrated, + pending: pending, } } }() diff --git a/pkg/turbine/receiver.go b/pkg/turbine/receiver.go index 6fce107f0..2e197980a 100644 --- a/pkg/turbine/receiver.go +++ b/pkg/turbine/receiver.go @@ -847,16 +847,18 @@ func (r *UDPReceiver) submitCompletion(ctx context.Context, work *slotCompletion } r.slotResetMu.RUnlock() return r.handleCompletionResult(ctx, slotCompletionResult{ - block: blk, - err: err, - hydrated: hydrated, - pending: pending, + generation: StreamGeneration{slot: work.state.slot, state: work.state}, + block: blk, + err: err, + hydrated: hydrated, + pending: pending, }) } func (r *UDPReceiver) consumeCompletionResults(ctx context.Context, results <-chan slotCompletionResult) { for result := range results { if ctx.Err() != nil { + r.assembler.cancelUndeliveredStream(result.generation) if result.pending && result.block != nil { r.finishPendingBlock(result.block.Slot) } @@ -884,10 +886,16 @@ func (r *UDPReceiver) handleCompletionResult(ctx context.Context, result slotCom if result.hydrated { r.hydratedFromDisk.Add(1) } + var emitted bool if result.pending { - return r.emitPendingAssembled(ctx, result.block) + emitted = r.emitPendingAssembled(ctx, result.block) + } else { + emitted = r.emitAssembled(ctx, result.block) + } + if !emitted { + r.assembler.cancelUndeliveredStream(result.generation) } - return r.emitAssembled(ctx, result.block) + return emitted } // skipAssemblyForSpool implements the catchup RAM policy: with a hydration diff --git a/pkg/turbine/repair_followup.go b/pkg/turbine/repair_followup.go index 8b250ca2c..659ad728d 100644 --- a/pkg/turbine/repair_followup.go +++ b/pkg/turbine/repair_followup.go @@ -37,6 +37,10 @@ func (c *repairClient) followupHighestResponse(conn *net.UDPConn, a *SlotAssembl if len(peers) == 0 { return } + // A matched discovery response can request at most 256 missing pieces plus + // one highest-index probe. This response-driven burst shares the global + // token bucket (not the periodic head-share quota); inflight dedup suppresses + // repeat sends and unused tokens are returned below. ask := len(req.MissingDataShreds) if req.NeedHighestDataShred { ask++ diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index a3e0abc40..617ba086f 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -277,6 +277,23 @@ func (a *SlotAssembler) streamStatusLocked(g StreamGeneration) StreamStatus { return StreamGone } +// cancelUndeliveredStream closes a completed generation whose result was +// abandoned during receiver shutdown. Identity binding avoids cancelling a +// replacement generation assembled for the same slot. +func (a *SlotAssembler) cancelUndeliveredStream(g StreamGeneration) { + if g.state == nil { + return + } + a.mu.Lock() + defer a.mu.Unlock() + if !g.state.streamCompleted { + return + } + g.state.streamCompleted = false + g.state.streamCancelReason = "delivery_cancelled" + a.publishStreamReleaseLocked(g.state, g.state.streamCancelReason) +} + // PendingStreamBatches returns every decoded batch of the generation whose // range starts at or after fromStart, in shred-index order. It reads the // prefetch state directly, so it is the authoritative recovery path after a diff --git a/pkg/turbine/stream_test.go b/pkg/turbine/stream_test.go index 132e805e0..bb0277916 100644 --- a/pkg/turbine/stream_test.go +++ b/pkg/turbine/stream_test.go @@ -290,3 +290,34 @@ func TestStreamPollingReusesTransactionView(t *testing.T) { require.Len(t, first[0].Transactions, 3) require.Same(t, &first[0].Transactions[0], &second[0].Transactions[0]) } + +func TestCompletedStreamCancelledWhenDeliveryAbandoned(t *testing.T) { + for _, queued := range []bool{false, true} { + a := &SlotAssembler{slots: make(map[uint64]*slotState)} + events := make(chan StreamEvent, 2) + a.SubscribeStream(events) + g := NewDetachedStreamGeneration(101) + g.state.streamCompleted = true + replacement := NewDetachedStreamGeneration(101) + a.slots[101] = replacement.state + r := &UDPReceiver{assembler: a, blocks: make(chan *block.Block), pendingBlocks: make(map[uint64]int)} + ctx, cancel := context.WithCancel(context.Background()) + cancel() + result := slotCompletionResult{block: &block.Block{Slot: 101}, generation: g, pending: true} + r.startPendingBlock(101) + if queued { + results := make(chan slotCompletionResult, 1) + results <- result + close(results) + r.consumeCompletionResults(ctx, results) + } else { + require.False(t, r.handleCompletionResult(ctx, result)) + } + require.Equal(t, StreamGone, a.StreamStatusOf(g)) + require.Equal(t, StreamActive, a.StreamStatusOf(replacement)) + require.Empty(t, r.pendingBlocks) + event := nextStreamEvent(t, events, StreamCancelled) + require.Equal(t, g, event.Generation) + require.Equal(t, "delivery_cancelled", event.Reason) + } +} From bcfc01242d89c7a216e5fa8c935a56c77a30b63f Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:43:02 -0500 Subject: [PATCH 079/111] replay: remove obsolete streaming missing-fee panic --- pkg/replay/block_execution.go | 23 ----------------------- 1 file changed, 23 deletions(-) diff --git a/pkg/replay/block_execution.go b/pkg/replay/block_execution.go index 6e72c1388..5e5e5518e 100644 --- a/pkg/replay/block_execution.go +++ b/pkg/replay/block_execution.go @@ -358,9 +358,6 @@ func (exec *blockExecution) executeTransactionGroup(txs []*solana.Transaction, i continue } exec.totalCU += computeUnits[idx] - if txFeeInfo == nil { - reportNilFeeInfo(exec.slotCtx, txs[idx], slot) - } exec.txFeeAccumulator.Add(txFeeInfo) } exec.slotCtx.TotalComputeUnitsConsumed = exec.totalCU @@ -532,26 +529,6 @@ func streamingTransactionError(index int, err error) error { return fmt.Errorf("unprocessable streaming transaction %d: %w", index, err) } -// reportNilFeeInfo reproduces ProcessBlock's diagnostic for a transaction whose -// fee information is missing, which only happens when blockhash validation -// failed for a transaction the block claims to have processed. -func reportNilFeeInfo(slotCtx *sealevel.SlotCtx, tx *solana.Transaction, slot uint64) { - var recentBlockhashes sealevel.SysvarRecentBlockhashes - if bankSysvars := slotCtx.BankSysvars(); bankSysvars != nil { - recentBlockhashes, _ = bankSysvars.RecentBlockhashes() - } - mlog.Log.Errorf("txFeeInfo is nil for tx %s in slot %d", tx.Signatures[0], slot) - mlog.Log.Errorf(" tx blockhash: %s", tx.Message.RecentBlockhash) - mlog.Log.Errorf(" LatestEvictedBlockhash: %x", slotCtx.LatestEvictedBlockhash[:8]) - if len(recentBlockhashes) > 0 { - mlog.Log.Errorf(" RecentBlockhashes: %d entries, newest=%x, oldest=%x", - len(recentBlockhashes), recentBlockhashes[0].Blockhash[:8], recentBlockhashes[len(recentBlockhashes)-1].Blockhash[:8]) - } else { - mlog.Log.Errorf(" RecentBlockhashes: nil or empty!") - } - panic(fmt.Sprintf("txFeeInfo is nil - blockhash validation failed for tx %s", tx.Signatures[0])) -} - // finalize runs the end-of-block phases over the open SlotCtx: fees to the // leader, rent, incinerator, the Alpenglow footer clock and vote rewards, bank // sysvar finalization, bank hash, footer verification, state publication and From 1b7da5699e5293e3c983e4dcde71e9438075a209 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:07:11 -0500 Subject: [PATCH 080/111] turbine: prioritize older pinned repair dependencies --- pkg/turbine/assembler.go | 6 +++++- pkg/turbine/repair_selection_test.go | 15 +++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/pkg/turbine/assembler.go b/pkg/turbine/assembler.go index 9a1516f29..6ef1d89d8 100644 --- a/pkg/turbine/assembler.go +++ b/pkg/turbine/assembler.go @@ -1426,7 +1426,11 @@ func (a *SlotAssembler) RepairRequestsTiered(maxSlots int, maxMissingPerSlot int } a.prunePriorityRepairSlotsLocked() - for _, slot := range a.priorityRepairOrder { + // Pin insertion order is retention policy, not dependency order. An older + // parent can be discovered after its child; give that parent the head share. + ordered := append([]uint64(nil), a.priorityRepairOrder...) + sort.Slice(ordered, func(i, j int) bool { return ordered[i] < ordered[j] }) + for _, slot := range ordered { // HEAD FIRST: the first priority slot — the one gating emission — // may list up to repairHeadMaxMissing, several times the per-slot // cap, so its admission share stays full at any response latency. diff --git a/pkg/turbine/repair_selection_test.go b/pkg/turbine/repair_selection_test.go index 033ad81a7..39cf4985a 100644 --- a/pkg/turbine/repair_selection_test.go +++ b/pkg/turbine/repair_selection_test.go @@ -310,3 +310,18 @@ func TestRepairSelectionPrefixOnlyForStreamingPriorityHead(t *testing.T) { t.Fatal("disabled streaming retained prefix policy") } } + +func TestRepairPriorityParentPinnedAfterChild(t *testing.T) { + a := NewSlotAssembler() + a.maxObservedSlot = 101 + a.retentionFloor = 100 + a.PrioritizeRepairRange(101, 101) + a.PrioritizeRepairRange(100, 100) + priority, _ := a.RepairRequestsTiered(2, 16) + if len(priority) != 2 || priority[0].Slot != 100 || priority[1].Slot != 101 { + t.Fatalf("parent must precede earlier-pinned child: %+v", priority) + } + if a.priorityRepairOrder[0] != 101 { + t.Fatal("selection changed pin retention order") + } +} From 16e4c8e05373f712b0eef1f663f92087b1be7342 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:07:11 -0500 Subject: [PATCH 081/111] turbine: reuse immutable ready stream views --- pkg/replay/streaming.go | 2 +- pkg/turbine/child_repair_test.go | 6 +++--- pkg/turbine/entry_prefetch.go | 4 ++++ pkg/turbine/stream.go | 17 ++++++++++++++-- pkg/turbine/stream_view_test.go | 33 ++++++++++++++++++++++++++++++++ 5 files changed, 56 insertions(+), 6 deletions(-) create mode 100644 pkg/turbine/stream_view_test.go diff --git a/pkg/replay/streaming.go b/pkg/replay/streaming.go index 98e800c4e..2a02820bb 100644 --- a/pkg/replay/streaming.go +++ b/pkg/replay/streaming.go @@ -403,7 +403,7 @@ func (s *streamingExecutor) rememberHeader(header *turbine.StreamBatch) { // Only slots within the bounded lookahead are worth the lookup, and a generation this // executor already retired (discarded, or declined as ineligible) is never // brought back: whole-block execution owns it from then on. A recovered -// header's ReadyAt is the lookup time, so OpenDelay reads as ~0 for it. +// header retains its original readiness time, including time before this lookup. func (s *streamingExecutor) recoverHeader(slot uint64, g turbine.StreamGeneration) { if g.IsZero() { return diff --git a/pkg/turbine/child_repair_test.go b/pkg/turbine/child_repair_test.go index 46e26c035..b6718bd97 100644 --- a/pkg/turbine/child_repair_test.go +++ b/pkg/turbine/child_repair_test.go @@ -72,7 +72,7 @@ func TestChildRepairLifecycle(t *testing.T) { case "unsubscribe": a.SubscribeStream(nil) case "update-parent", "wrong-parent": - changed := *b + changed := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} parent := *b.parent changed.parent = &parent if kind == "update-parent" { @@ -103,7 +103,7 @@ func TestChildRepairLifecycle(t *testing.T) { func TestChildRepairRejectsUnrelatedAndInvalidHeader(t *testing.T) { a, _, c, b := childRepairFixture(t) a.streamRepairChild = nil - bad := *b + bad := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} bad.err = errors.New("invalid header") a.mu.Lock() a.noteChildRepairHeaderLocked(c, &bad) @@ -147,7 +147,7 @@ func TestChildRepairRejectsStaleParentAnchor(t *testing.T) { func TestChildRepairParentUpdateClearsLookahead(t *testing.T) { a, p, _, b := childRepairFixture(t) - update := *b + update := prefetchedShredBatch{start: b.start, end: b.end, parent: b.parent, marker: b.marker, ready: b.ready} info := *b.parent info.FromUpdateParent = true update.parent = &info diff --git a/pkg/turbine/entry_prefetch.go b/pkg/turbine/entry_prefetch.go index 64f917f9c..28076cfb3 100644 --- a/pkg/turbine/entry_prefetch.go +++ b/pkg/turbine/entry_prefetch.go @@ -27,6 +27,9 @@ type shredBatchRange struct{ start, end uint32 } // Fields are immutable after ready closes; signature readers own its decoded // transactions until verification.done closes. type prefetchedShredBatch struct { + viewOnce sync.Once // protects the immutable stream view, including concurrent Resolve + view *StreamBatch + readyAt time.Time // set before ready closes start, end uint32 raw []byte entries []Entry @@ -201,6 +204,7 @@ func (p *entryPrefetchPool) run() { batch.verification, batch.submitErr = p.verifier.submitPrefetchTransactions(f.ctx, txs) } } + batch.readyAt = time.Now() close(ready) p.a.mu.Lock() f.queued = false diff --git a/pkg/turbine/stream.go b/pkg/turbine/stream.go index 617ba086f..9e8def5a6 100644 --- a/pkg/turbine/stream.go +++ b/pkg/turbine/stream.go @@ -107,7 +107,9 @@ type StreamBatch struct { // Transactions is empty for markers and for batches that failed to decode. Transactions []*solana.Transaction // Err is the decode error; a batch with Err makes the whole slot invalid. - Err error + Err error + // ReadyAt is the original publication time, preserved across polling and + // notification recovery. It is not the time a consumer looked up the batch. ReadyAt time.Time batch *prefetchedShredBatch @@ -329,13 +331,24 @@ func (a *SlotAssembler) PendingStreamBatches(g StreamGeneration, fromStart uint3 // newStreamBatch builds the immutable view; it must only be called after the // batch's ready channel closed (its fields are immutable from then on). func newStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatch { + batch.viewOnce.Do(func() { + batch.view = buildStreamBatch(g, batch) + }) + return batch.view +} + +func buildStreamBatch(g StreamGeneration, batch *prefetchedShredBatch) *StreamBatch { + readyAt := batch.readyAt + if readyAt.IsZero() { // detached test batches have no prefetch publication + readyAt = time.Now() + } view := &StreamBatch{ Slot: g.slot, Generation: g, Start: batch.start, End: batch.end, Err: batch.err, - ReadyAt: time.Now(), + ReadyAt: readyAt, batch: batch, } switch { diff --git a/pkg/turbine/stream_view_test.go b/pkg/turbine/stream_view_test.go new file mode 100644 index 000000000..8f861fb3c --- /dev/null +++ b/pkg/turbine/stream_view_test.go @@ -0,0 +1,33 @@ +package turbine + +import ( + "sync" + "testing" + "time" +) + +func TestReadyStreamViewReused(t *testing.T) { + g := NewDetachedStreamGeneration(42) + ready := make(chan struct{}) + at := time.Now().Add(-time.Second) + b := &prefetchedShredBatch{ready: ready, readyAt: at, start: 1, end: 2} + close(ready) + expected := newStreamBatch(g, b) + var wg sync.WaitGroup + for i := 0; i < 16; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for j := 0; j < 100; j++ { + got := newStreamBatch(g, b) + if got != expected || !got.ReadyAt.Equal(at) { + t.Error("view or readiness time changed") + } + } + }() + } + wg.Wait() + if n := testing.AllocsPerRun(100, func() { _ = newStreamBatch(g, b) }); n != 0 { + t.Fatalf("cached view allocates: %v", n) + } +} From be29319f2802cf1a56a3edfdbc1c397d2fb7b197 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:07:11 -0500 Subject: [PATCH 082/111] sealevel: bind cached executables to bank-visible source and features --- pkg/accountsdb/accountsdb.go | 22 +++- pkg/accountsdb/program_cache_version_test.go | 31 ++++++ pkg/replay/remaining_compute_units_test.go | 4 +- pkg/sealevel/bpf_loader.go | 106 ++++++++----------- pkg/sealevel/loader_v4.go | 49 +++------ pkg/sealevel/program_cache_version_test.go | 72 +++++++++++++ pkg/sealevel/program_workloads_bench_test.go | 4 +- 7 files changed, 190 insertions(+), 98 deletions(-) create mode 100644 pkg/accountsdb/program_cache_version_test.go create mode 100644 pkg/sealevel/program_cache_version_test.go diff --git a/pkg/accountsdb/accountsdb.go b/pkg/accountsdb/accountsdb.go index 49714f731..acc1666e5 100644 --- a/pkg/accountsdb/accountsdb.go +++ b/pkg/accountsdb/accountsdb.go @@ -9,6 +9,7 @@ import ( "errors" "fmt" "log" + "maps" "os" "path/filepath" "runtime/trace" @@ -18,6 +19,7 @@ import ( "github.com/Overclock-Validator/mithril/pkg/accounts" "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/features" "github.com/Overclock-Validator/mithril/pkg/mlog" "github.com/Overclock-Validator/mithril/pkg/sbpf" "github.com/cockroachdb/pebble" @@ -251,6 +253,24 @@ func (accountsDb *AccountsDb) InitCaches() { type ProgramCacheEntry struct { Program *sbpf.Program DeploymentSlot uint64 + // Executables are reusable across banks only for identical source and loader + // features. Slot alone is not a version: competing forks can deploy at the + // same slot. Fields are immutable after cache publication. + sourceBytes []byte + sourceBound bool + sourceFeatures features.Features +} + +// BindSource must be called before publishing the entry; published bindings +// must never be mutated, including when another bank replaces the cache key. +func (entry *ProgramCacheEntry) BindSource(source []byte, f *features.Features) { + entry.sourceBytes = bytes.Clone(source) + entry.sourceBound = true + entry.sourceFeatures = *f.Clone() +} + +func (entry *ProgramCacheEntry) MatchesSource(source []byte, f *features.Features) bool { + return entry != nil && entry.sourceBound && bytes.Equal(entry.sourceBytes, source) && maps.Equal(entry.sourceFeatures, *f) } func programCacheCapacityUnits() int { @@ -287,7 +307,7 @@ func (entry *ProgramCacheEntry) CostUnits() uint32 { if entry == nil || entry.Program == nil { return 1 } - bytes := entry.Program.MemoryBytes() + bytes := entry.Program.MemoryBytes() + uint64(len(entry.sourceBytes)) units := (bytes + programCacheCostUnitBytes - 1) / programCacheCostUnitBytes if units == 0 { return 1 diff --git a/pkg/accountsdb/program_cache_version_test.go b/pkg/accountsdb/program_cache_version_test.go new file mode 100644 index 000000000..1d9d962a0 --- /dev/null +++ b/pkg/accountsdb/program_cache_version_test.go @@ -0,0 +1,31 @@ +package accountsdb + +import ( + "github.com/Overclock-Validator/mithril/pkg/features" + "testing" +) + +func TestProgramCacheSourceBinding(t *testing.T) { + f := features.NewFeaturesDefault() + source := []byte{1, 2, 3} + entry := &ProgramCacheEntry{DeploymentSlot: 100} + if entry.MatchesSource(source, f) { + t.Fatal("unbound entry matched") + } + entry.BindSource(source, f) + if !entry.MatchesSource(source, f) { + t.Fatal("identical source missed") + } + source[0]++ + if entry.MatchesSource(source, f) { + t.Fatal("same-slot fork source matched") + } + source[0]-- + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + if entry.MatchesSource(source, f) { + t.Fatal("different feature environment matched") + } + if !entry.MatchesSource(source, features.NewFeaturesDefault()) { + t.Fatal("binding retained mutable feature map") + } +} diff --git a/pkg/replay/remaining_compute_units_test.go b/pkg/replay/remaining_compute_units_test.go index ddec0e686..2cf0ace1c 100644 --- a/pkg/replay/remaining_compute_units_test.go +++ b/pkg/replay/remaining_compute_units_test.go @@ -77,7 +77,9 @@ func TestRemainingComputeUnitsPreservesSuccessfulNonceAdvance(t *testing.T) { } program := &sbpf.Program{Text: text, TextBytes: textBytes, TextVA: sbpf.VaddrProgram} require.NoError(t, program.Verify()) - slotCtx.AccountsDb.AddProgramToCache(programKey, &accountsdb.ProgramCacheEntry{Program: program}) + entry := &accountsdb.ProgramCacheEntry{Program: program} + entry.BindSource(nil, slotCtx.Features) + slotCtx.AccountsDb.AddProgramToCache(programKey, entry) tx, err := solana.NewTransaction([]solana.Instruction{ system.NewAdvanceNonceAccountInstruction(nonceKey, solana.SysVarRecentBlockHashesPubkey, payer).Build(), diff --git a/pkg/sealevel/bpf_loader.go b/pkg/sealevel/bpf_loader.go index 6142b3049..16ba486cc 100644 --- a/pkg/sealevel/bpf_loader.go +++ b/pkg/sealevel/bpf_loader.go @@ -1342,6 +1342,11 @@ func executeLoadedProgram(execCtx *ExecutionCtx, program *sbpf.Program, syscallR func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey, programData []byte, syscallRegistry sbpf.SyscallRegistry) error { start := time.Now() + // The caller has already validated the bank-visible loader metadata. Never + // let a global cache hit bypass that validation or select a different fork. + if entry, ok := execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAddr); ok && entry.MatchesSource(programData, &execCtx.Features) { + return executeLoadedProgram(execCtx, entry.Program, syscallRegistry) + } loader, err := loader.NewLoaderWithSyscalls(programData, syscallRegistry, false, &execCtx.Features) if err != nil { return InstrErrUnsupportedProgramId @@ -1356,6 +1361,7 @@ func executeProgramFromBytes(execCtx *ExecutionCtx, programAddr solana.PublicKey } entry := &accountsdb.ProgramCacheEntry{Program: program} + entry.BindSource(programData, &execCtx.Features) if !execCtx.IsSimulation { addProgramToCache(execCtx, programAddr, entry) } @@ -1495,36 +1501,27 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { } var programBytes []byte - var loadedProgram *sbpf.Program - var hasLoadedProgram bool var programAcctKey solana.PublicKey programOwner := programAcct.Owner() if programOwner == a.BpfLoader2Addr || programOwner == a.BpfLoaderDeprecatedAddr { - var programCacheEntry *accountsdb.ProgramCacheEntry - programCacheEntry, hasLoadedProgram = execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAcct.Key()) - if hasLoadedProgram { - programAcctKey = programAcct.Key() - loadedProgram = programCacheEntry.Program - } else { // program is not cached - if len(programAcct.Data()) == 0 { - var paTmp *accounts.Account - paTmp, err = execCtx.SlotCtx.GetAccount(programAcct.Key()) + if len(programAcct.Data()) == 0 { + var paTmp *accounts.Account + paTmp, err = execCtx.SlotCtx.GetAccount(programAcct.Key()) + if err != nil { + paTmp, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcct.Key()) if err != nil { - paTmp, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcct.Key()) - if err != nil { - //mlog.Log.Debugf("unable to get account %s from accountsdb", programAcct.Key()) - return InstrErrUnsupportedProgramId - } + //mlog.Log.Debugf("unable to get account %s from accountsdb", programAcct.Key()) + return InstrErrUnsupportedProgramId } - programBytes = paTmp.Data - } else { - programBytes = programAcct.Data() } - programAcctKey = programAcct.Key() + programBytes = paTmp.Data + } else { + programBytes = programAcct.Data() } + programAcctKey = programAcct.Key() } else if programOwner == a.BpfLoaderUpgradeableAddr { var programAcctState *UpgradeableLoaderState @@ -1552,49 +1549,38 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { } start := time.Now() - var programCacheEntry *accountsdb.ProgramCacheEntry - programCacheEntry, hasLoadedProgram = execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(programAcctState.Program.ProgramDataAddress) - if hasLoadedProgram { - if programCacheEntry.DeploymentSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } - programAcctKey = programAcctState.Program.ProgramDataAddress - loadedProgram = programCacheEntry.Program - metrics.GlobalBlockReplay.GetProgramDataCached.AddTimingSince(start) - } else { // program is not cached - programDataAcct, err := execCtx.SlotCtx.GetAccount(programAcctState.Program.ProgramDataAddress) + programDataAcct, err := execCtx.SlotCtx.GetAccount(programAcctState.Program.ProgramDataAddress) + if err != nil { + programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcctState.Program.ProgramDataAddress) if err != nil { - programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(programAcctState.Program.ProgramDataAddress) - if err != nil { - return InstrErrUnsupportedProgramId - } - metrics.GlobalBlockReplay.GetProgramDataUncachedAccountsDb.AddTimingSince(start) - } else { - metrics.GlobalBlockReplay.GetProgramDataUncachedAccounts.AddTimingSince(start) + return InstrErrUnsupportedProgramId } + metrics.GlobalBlockReplay.GetProgramDataUncachedAccountsDb.AddTimingSince(start) + } else { + metrics.GlobalBlockReplay.GetProgramDataUncachedAccounts.AddTimingSince(start) + } - start = time.Now() - programDataAcctState, err := UnmarshalUpgradeableLoaderState(programDataAcct.Data) - if err != nil { - return err - } + start = time.Now() + programDataAcctState, err := UnmarshalUpgradeableLoaderState(programDataAcct.Data) + if err != nil { + return err + } - if programDataAcctState.Type != UpgradeableLoaderStateTypeProgramData { - return InstrErrUnsupportedProgramId - } + if programDataAcctState.Type != UpgradeableLoaderStateTypeProgramData { + return InstrErrUnsupportedProgramId + } - programDataSlot := programDataAcctState.ProgramData.Slot - if programDataSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } + programDataSlot := programDataAcctState.ProgramData.Slot + if programDataSlot >= execCtx.SlotCtx.Slot { + return InstrErrInvalidAccountData + } - if len(programDataAcct.Data) < upgradeableLoaderSizeOfProgramDataMetaData { - return InstrErrUnsupportedProgramId - } - programAcctKey = programAcctState.Program.ProgramDataAddress - programBytes = programDataAcct.Data[upgradeableLoaderSizeOfProgramDataMetaData:] - metrics.GlobalBlockReplay.GetProgramDataUncachedMarshal.AddTimingSince(start) + if len(programDataAcct.Data) < upgradeableLoaderSizeOfProgramDataMetaData { + return InstrErrUnsupportedProgramId } + programAcctKey = programAcctState.Program.ProgramDataAddress + programBytes = programDataAcct.Data[upgradeableLoaderSizeOfProgramDataMetaData:] + metrics.GlobalBlockReplay.GetProgramDataUncachedMarshal.AddTimingSince(start) } else { return InstrErrUnsupportedProgramId } @@ -1605,13 +1591,7 @@ func BpfLoaderProgramExecute(execCtx *ExecutionCtx) error { return Syscalls(&execCtx.Features, false, u) }) - // two cases here: we're either executing from the program cache, so from a pre-parsed/loaded program, or from bytes if - // the the program was not found in the cache. - if hasLoadedProgram { - err = executeLoadedProgram(execCtx, loadedProgram, syscallRegistry) - } else { - err = executeProgramFromBytes(execCtx, programAcctKey, programBytes, syscallRegistry) - } + err = executeProgramFromBytes(execCtx, programAcctKey, programBytes, syscallRegistry) return err } diff --git a/pkg/sealevel/loader_v4.go b/pkg/sealevel/loader_v4.go index 942aaf892..8ab33effb 100644 --- a/pkg/sealevel/loader_v4.go +++ b/pkg/sealevel/loader_v4.go @@ -274,38 +274,28 @@ func LoaderV4Execute(execCtx *ExecutionCtx) error { return err } - var loadedProgram *sbpf.Program var programBytes []byte - - programCacheEntry, hasLoadedProgram := execCtx.SlotCtx.AccountsDb.MaybeGetProgramFromCache(program.Key()) - if hasLoadedProgram { - if programCacheEntry.DeploymentSlot >= execCtx.SlotCtx.Slot { - return InstrErrInvalidAccountData - } - loadedProgram = programCacheEntry.Program - } else { - programDataAcct, err := execCtx.SlotCtx.GetAccount(program.Key()) - if err != nil { - programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(program.Key()) - if err != nil { - return InstrErrUnsupportedProgramId - } - } - - state, err := decodeLoaderV4State(programDataAcct.Data) + programDataAcct, err := execCtx.SlotCtx.GetAccount(program.Key()) + if err != nil { + programDataAcct, err = execCtx.SlotCtx.GetAccountFromAccountsDb(program.Key()) if err != nil { return InstrErrUnsupportedProgramId } + } - if state.Status == LoaderV4StatusRetracted { - return InstrErrUnsupportedProgramId - } - if state.Slot >= execCtx.SlotCtx.Slot { - return InstrErrUnsupportedProgramId - } + state, err := decodeLoaderV4State(programDataAcct.Data) + if err != nil { + return InstrErrUnsupportedProgramId + } - programBytes = programDataAcct.Data[loaderV4ProgramDataOffset:] + if state.Status == LoaderV4StatusRetracted { + return InstrErrUnsupportedProgramId } + if state.Slot >= execCtx.SlotCtx.Slot { + return InstrErrUnsupportedProgramId + } + + programBytes = programDataAcct.Data[loaderV4ProgramDataOffset:] syscallRegistry := sbpf.SyscallRegistry(func(u uint32) (sbpf.Syscall, bool) { return Syscalls(&execCtx.Features, false, u) @@ -313,13 +303,8 @@ func LoaderV4Execute(execCtx *ExecutionCtx) error { program.Drop() - // two cases here: we're either executing from the program cache, so from a pre-parsed/loaded program, or from bytes if - // the the program was not found in the cache. - if hasLoadedProgram { - err = executeLoadedProgram(execCtx, loadedProgram, syscallRegistry) - } else { - err = executeProgramFromBytes(execCtx, program.Key(), programBytes, syscallRegistry) - } + err = executeProgramFromBytes(execCtx, program.Key(), programBytes, syscallRegistry) + } return err diff --git a/pkg/sealevel/program_cache_version_test.go b/pkg/sealevel/program_cache_version_test.go new file mode 100644 index 000000000..83089af25 --- /dev/null +++ b/pkg/sealevel/program_cache_version_test.go @@ -0,0 +1,72 @@ +package sealevel + +import ( + "bytes" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/features" + bin "github.com/gagliardetto/binary" + "github.com/stretchr/testify/require" + "testing" +) + +func TestExecutionRejectsOtherBankProgramCache(t *testing.T) { + for _, kind := range []string{"unbound", "same-slot-fork", "different-features"} { + t.Run(kind, func(t *testing.T) { + w := expandedProgramWorkloads(t)[0] + run := workloadRunner(t, w, false) + ctx, err := run() + require.NoError(t, err) + poison := &accountsdb.ProgramCacheEntry{DeploymentSlot: 1336} // nil executable: using it would panic + f := features.NewFeaturesDefault() + switch kind { + case "same-slot-fork": + poison.BindSource([]byte("another fork's program"), f) + case "different-features": + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + poison.BindSource(w.elf, f) + } + ctx.SlotCtx.AccountsDb.AddProgramToCache(w.program, poison) + next, err := run() + require.NoError(t, err) + w.check(t, next) + require.Equal(t, ctx.ComputeMeter.Used(), next.ComputeMeter.Used()) + }) + } +} + +func TestWarmCacheCannotBypassDeploymentSlot(t *testing.T) { + key, dataKey := benchPubkey(80), benchPubkey(81) + var encoded bytes.Buffer + state := UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgram, Program: UpgradeableLoaderStateProgram{ProgramDataAddress: dataKey}} + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&encoded))) + program := accounts.Account{Key: key, Owner: a.BpfLoaderUpgradeableAddr, Executable: true, Lamports: 10000000, Data: encoded.Bytes()} + tx := NewTransactionAccounts([]accounts.Account{program}) + ctx := newBenchExecCtx(tx, 100) + initializeLegacyBankFixture(t, ctx) + ctx.SlotCtx.Slot = 100 + var data bytes.Buffer + state = UpgradeableLoaderState{Type: UpgradeableLoaderStateTypeProgramData, ProgramData: UpgradeableLoaderStateProgramData{Slot: 100}} + require.NoError(t, state.MarshalWithEncoder(bin.NewBinEncoder(&data))) + require.NoError(t, ctx.Accounts.SetAccount((*[32]byte)(&dataKey), &accounts.Account{Key: dataKey, Owner: a.BpfLoaderUpgradeableAddr, Lamports: 10000000, Data: data.Bytes()})) + // The cache describes a previous bank. Its older deployment slot must not + // hide this bank's deployment, which cannot be invoked in the same slot. + ctx.SlotCtx.AccountsDb.AddProgramToCache(dataKey, &accountsdb.ProgramCacheEntry{DeploymentSlot: 99}) + err := ctx.ProcessInstruction(nil, nil, []uint64{0}) + require.ErrorIs(t, err, InstrErrInvalidAccountData) +} + +func TestWarmCacheCannotBypassLoaderV4Retraction(t *testing.T) { + key := benchPubkey(82) + state := LoaderV4State{Slot: 99, Status: LoaderV4StatusRetracted} + program := accounts.Account{Key: key, Owner: a.LoaderV4Addr, Executable: true, Lamports: 10000000, Data: state.Marshal()} + tx := NewTransactionAccounts([]accounts.Account{program}) + ctx := newBenchExecCtx(tx, 100) + initializeLegacyBankFixture(t, ctx) + ctx.SlotCtx.Slot = 100 + require.NoError(t, ctx.Accounts.SetAccount((*[32]byte)(&key), &program)) + ctx.SlotCtx.AccountsDb.AddProgramToCache(key, &accountsdb.ProgramCacheEntry{DeploymentSlot: 99}) + err := ctx.ProcessInstruction(nil, nil, []uint64{0}) + require.ErrorIs(t, err, InstrErrUnsupportedProgramId) +} diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go index 649d35ddd..fd58e60b3 100644 --- a/pkg/sealevel/program_workloads_bench_test.go +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -126,7 +126,9 @@ func workloadRunner(t testing.TB, w programWorkload, vasa bool) func() (*Executi require.NoError(t, err) t.Cleanup(cache.Close) db := &accountsdb.AccountsDb{ProgramCache: cache} - db.AddProgramToCache(w.program, &accountsdb.ProgramCacheEntry{Program: prog}) + entry := &accountsdb.ProgramCacheEntry{Program: prog} + entry.BindSource(w.elf, f) + db.AddProgramToCache(w.program, entry) _, cached := db.MaybeGetProgramFromCache(w.program) require.True(t, cached, "warm program must be cached") return func() (*ExecutionCtx, error) { From fda34f664d76b09d5716a83c1ec33ccdcee7fb9e Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:24:25 +0100 Subject: [PATCH 083/111] rewards: preserve credits on inactive Alpenglow stakes Retain activation status from the existing points calculation and exclude inactive stakes from the Alpenglow skipped-reward credit advance. Preserve explicit forced-update cases. Add an epoch-116 fixture that reproduces the original slot 6264001 bank hash before the fix and the expected footer hash afterward. Include reward regressions in the race CI job. --- .github/workflows/go_build.yml | 2 +- pkg/rewards/alpenglow_rewards_test.go | 53 + pkg/rewards/inactive_stakes_test.go | 118 ++ pkg/rewards/rewards.go | 22 +- pkg/rewards/testdata/epoch116/README.md | 49 + .../testdata/epoch116/inactive-stakes.json | 1693 +++++++++++++++++ 6 files changed, 1930 insertions(+), 7 deletions(-) create mode 100644 pkg/rewards/inactive_stakes_test.go create mode 100644 pkg/rewards/testdata/epoch116/README.md create mode 100644 pkg/rewards/testdata/epoch116/inactive-stakes.json diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index 4d9d9bf5d..7b7d01159 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -38,7 +38,7 @@ jobs: # recovery and cancellation tests. run: >- go test -race -p 2 -count=1 - ./pkg/alpenglow ./pkg/consensus ./pkg/replay + ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... ./cmd/mithril/node ./cmd/mithril/configcmd diff --git a/pkg/rewards/alpenglow_rewards_test.go b/pkg/rewards/alpenglow_rewards_test.go index 76136886d..ae9d451bb 100644 --- a/pkg/rewards/alpenglow_rewards_test.go +++ b/pkg/rewards/alpenglow_rewards_test.go @@ -116,6 +116,59 @@ func TestAlpenglowEarnedPointsAreNotCreditsOnly(t *testing.T) { )) } +func TestAlpenglowSkippedRewardCreditsRespectStakeActivation(t *testing.T) { + votePubkey := solana.PublicKey{1} + voteState := &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{{ + Epoch: 115, Credits: 2_000, PrevCredits: 1_000, + }}}, + } + mode := RewardCalculationMode{ + FullAlpenglow: true, + RewardEpochDelegatedStakes: map[solana.PublicKey]uint64{votePubkey: 1_000_000}, + } + for _, tc := range []struct { + name string + activation, deactivation uint64 + advance bool + }{ + {"fully cooled in rewarded epoch", 103, 114, false}, + {"not yet activating", 116, math.MaxUint64, false}, + {"activating in rewarded epoch", 115, math.MaxUint64, true}, + {"effective fractional reward", 103, math.MaxUint64, true}, + {"still cooling in rewarded epoch", 103, 115, true}, + } { + t.Run(tc.name, func(t *testing.T) { + delegation := &sealevel.Delegation{ + VoterPubkey: votePubkey, StakeLamports: 1, + ActivationEpoch: tc.activation, DeactivationEpoch: tc.deactivation, + CreditsObserved: 1_000, + } + pcs := calculateStakePointsAndCredits(solana.PublicKey{}, &sealevel.SysvarStakeHistory{}, + delegation, voteState, nil, 115, mode) + require.True(t, pcs.Points.Eq(wide.Uint128{})) + require.Equal(t, uint64(2_000), pcs.NewCreditsObserved) + require.Equal(t, tc.advance, shouldForceCreditsOnly(pcs, 1, tc.activation, 115, 1_000, mode)) + }) + } +} + +func TestInactiveStakePreservesExplicitCreditUpdates(t *testing.T) { + pcs := CalculatedStakePoints{NewCreditsObserved: 2_000, Inactive: true} + mode := RewardCalculationMode{FullAlpenglow: true} + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, mode)) + require.True(t, shouldForceCreditsOnly(pcs, 0, 103, 115, 1_000, mode), "disabled inflation") + require.True(t, shouldForceCreditsOnly(pcs, 1, 115, 115, 1_000, mode), "activation epoch") + pcs.ForceCreditsUpdateWithSkippedReward = true + require.True(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 3_000, mode), "vote credit rewind") + + // Tower does not inherit Alpenglow's automatic skipped-reward advance. + pcs.ForceCreditsUpdateWithSkippedReward = false + pcs.Inactive = false + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, RewardCalculationMode{})) +} + func TestInflationRewardsUseHistoricalSlotTimeTransitions(t *testing.T) { schedule := &sealevel.SysvarEpochSchedule{ SlotsPerEpoch: 54_000, diff --git a/pkg/rewards/inactive_stakes_test.go b/pkg/rewards/inactive_stakes_test.go new file mode 100644 index 000000000..c1edc7c95 --- /dev/null +++ b/pkg/rewards/inactive_stakes_test.go @@ -0,0 +1,118 @@ +package rewards + +import ( + "crypto/sha256" + "encoding/json" + "fmt" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/bankhash" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/lthash" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// This reduced incident fixture holds all unrelated slot effects constant. +// The production reward calculator, spool distributor and bank hasher must +// leave the 33 fully cooled stakes unchanged. See testdata/epoch116/README.md. +func TestEpoch116InactiveStakeBankHash(t *testing.T) { + var fixture struct { + ParentBankhash string `json:"parent_bankhash"` + Blockhash string `json:"blockhash"` + ExpectedBankhash string `json:"expected_bankhash"` + OriginalBankhash string `json:"original_bankhash"` + BaseLtHash []byte `json:"base_accounts_lt_hash"` + StakeHistory []byte `json:"stake_history"` + Stakes []struct { + Account *accounts.Account `json:"account"` + VoteEpochCredits sealevel.EpochCredits `json:"vote_epoch_credits"` + OriginalUpdatedDataSHA256 string `json:"original_updated_data_sha256"` + } `json:"stakes"` + } + raw, err := os.ReadFile("testdata/epoch116/inactive-stakes.json") + require.NoError(t, err) + require.NoError(t, json.Unmarshal(raw, &fixture)) + require.Len(t, fixture.Stakes, 33) + + dir := t.TempDir() + require.NoError(t, os.MkdirAll(filepath.Join(dir, "accounts"), 0o755)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "largest_file_id"), make([]byte, 8), 0o644)) + db, err := accountsdb.OpenDb(dir) + require.NoError(t, err) + db.InitCaches() + t.Cleanup(db.CloseDb) + global.ClearPendingStakePubkeys() + t.Cleanup(global.ClearPendingStakePubkeys) + + parents := accounts.NewMemAccounts() + var storedAccounts, erroneousUpdates []*accounts.Account + votes := make(map[solana.PublicKey]*sealevel.VoteStateVersions) + for _, row := range fixture.Stakes { + acct := row.Account + stake, err := sealevel.UnmarshalStakeState(acct.Data) + require.NoError(t, err) + require.Equal(t, uint64(114), stake.Stake.Stake.Delegation.DeactivationEpoch) + require.Equal(t, uint64(115), row.VoteEpochCredits.Epoch) + voteKey := stake.Stake.Stake.Delegation.VoterPubkey + votes[voteKey] = &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{row.VoteEpochCredits}}, + } + require.NoError(t, parents.SetAccountWithoutLock(acct.Key, acct)) + storedAccounts = append(storedAccounts, acct) + global.EnqueuePendingStakePubkey(6264000, acct.Key) + + // Independently reconstruct the logged erroneous write, and verify its + // byte hash before using it to establish the original bad bank hash. + bad := acct.Clone() + stake.Stake.Stake.CreditsObserved = row.VoteEpochCredits.Credits + require.NoError(t, sealevel.MarshalStakeStakeInto(stake, bad.Data)) + require.Equal(t, row.OriginalUpdatedDataSHA256, fmt.Sprintf("%x", sha256.Sum256(bad.Data))) + erroneousUpdates = append(erroneousUpdates, bad) + } + stored := make(chan struct{}) + require.NoError(t, db.StoreAccounts(storedAccounts, 6264000, func() { close(stored) })) + <-stored + count, err := global.FlushPendingStakePubkeysThrough(dir, 6264000) + require.NoError(t, err) + require.Equal(t, len(fixture.Stakes), count) + db.RootedDurable = true + + f := &features.Features{} + f.EnableFeature(features.AccountsLtHash, 0) + f.EnableFeature(features.RemoveAccountsDeltaHash, 0) + calculateHash := func(updates []*accounts.Account) string { + ctx := &sealevel.SlotCtx{ + Features: f, ParentAccts: parents, + AcctsLtHash: new(lthash.LtHash).InitWithHash(fixture.BaseLtHash), + } + return solana.HashFromBytes(bankhash.CalculateBankHash(ctx, nil, updates, + solana.MustHashFromBase58(fixture.ParentBankhash), 0, + solana.MustHashFromBase58(fixture.Blockhash))).String() + } + require.Equal(t, fixture.OriginalBankhash, calculateHash(erroneousUpdates)) + + var history sealevel.SysvarStakeHistory + require.NoError(t, history.UnmarshalWithDecoder(bin.NewBinDecoder(fixture.StakeHistory))) + newRateEpoch := uint64(0) + result, err := CalculateRewardsStreaming(db, 6264000, &history, &newRateEpoch, + votes, PointValue{Rewards: 12922370184029}, 115, [32]byte{}, + &sealevel.SlotCtx{Features: f}, f, RewardCalculationMode{FullAlpenglow: true}) + require.NoError(t, err) + updated, _, distributed, burned := DistributeStakingRewardsFromSpool( + db, result.SpoolDir, result.SpoolSlot, 0, 6264001, nil) + require.Zero(t, distributed) + require.Zero(t, burned) + require.Equal(t, fixture.ExpectedBankhash, calculateHash(updated)) + require.Zero(t, result.NumStakeRewards) + require.Equal(t, uint64(1), result.NumPartitions) + require.Empty(t, updated) +} diff --git a/pkg/rewards/rewards.go b/pkg/rewards/rewards.go index ce0e824a0..bfff0ebfc 100644 --- a/pkg/rewards/rewards.go +++ b/pkg/rewards/rewards.go @@ -43,6 +43,9 @@ type CalculatedStakePoints struct { Points wide.Uint128 NewCreditsObserved uint64 ForceCreditsUpdateWithSkippedReward bool + // Inactive is set by Alpenglow points calculation when the delegation + // has neither effective nor activating stake in the rewarded epoch. + Inactive bool } const legacyInflationSlotsPerYear = 78_892_314.984 @@ -697,11 +700,14 @@ func calculateStakePointsAndCredits( } newObserved = max(newObserved, latest.Credits) - effectiveStake := delegation.StakeActivatingAndDeactivating( + status := delegation.StakeActivatingAndDeactivating( rewardedEpoch, stakeHistory, newRateActivationEpoch, - ).Effective - if earnedCredits == 0 || effectiveStake == 0 { - return CalculatedStakePoints{NewCreditsObserved: newObserved} + ) + if earnedCredits == 0 || status.Effective == 0 { + return CalculatedStakePoints{ + NewCreditsObserved: newObserved, + Inactive: status.Effective == 0 && status.Activating == 0, + } } totalStake := mode.RewardEpochDelegatedStakes[delegation.VoterPubkey] if totalStake == 0 { @@ -711,7 +717,7 @@ func calculateStakePointsAndCredits( } } points := wide.Uint128FromUint64(earnedCredits). - Mul(wide.Uint128FromUint64(effectiveStake)). + Mul(wide.Uint128FromUint64(status.Effective)). Div(wide.Uint128FromUint64(totalStake)) return CalculatedStakePoints{Points: points, NewCreditsObserved: newObserved} } @@ -789,10 +795,14 @@ func shouldForceCreditsOnly( pointValueRewards, activationEpoch, rewardedEpoch, creditsObserved uint64, mode RewardCalculationMode, ) bool { + // Agave's skipped-reward credit advance applies only to effective or + // activating Alpenglow stakes. Fully cooled stakes must retain their + // account bytes, even though their vote account has earned new credits. + // The explicit forced-update cases still take precedence. return pcs.ForceCreditsUpdateWithSkippedReward || pointValueRewards == 0 || activationEpoch == rewardedEpoch || - (mode.FullAlpenglow && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) + (mode.FullAlpenglow && !pcs.Inactive && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) } // CalculateRewardsStreaming performs a streaming calculation of stake rewards. diff --git a/pkg/rewards/testdata/epoch116/README.md b/pkg/rewards/testdata/epoch116/README.md new file mode 100644 index 000000000..41cd2cb6a --- /dev/null +++ b/pkg/rewards/testdata/epoch116/README.md @@ -0,0 +1,49 @@ +# Epoch 115 → 116 inactive-stake regression + +`inactive-stakes.json` is a reduced fixture from the Alpenglow failure at slot +6,264,001 on 2026-09-21, running Mithril `320ce8da`. It contains the 33 fully +cooled stake accounts that Mithril incorrectly rewrote, their vote accounts' +epoch-115 credits, and the saved StakeHistory sysvar. All 33 delegated stakes +had deactivation epoch 114 and zero effective/activating stake in epoch 115. + +The source was the supplied `/mnt/mithril-accounts` checkpoint at slot 6,264,000 +and `footer-bankhash-mismatch-slot-6264001.json` in the supplied logs. AccountsDB +was opened read-only. Account bytes are public chain data; no keys or validator +identity files are included. + +To isolate the defect, `base_accounts_lt_hash` includes all correct slot effects +and the unchanged 33 accounts. The other 792 modified accounts were reconstructed +from the parent checkpoint and deterministic slot updates; all 825 reconstructed +accounts matched the diagnostic's individual SHA-256 data hashes. Combining +their deltas with the saved parent LtHash reproduced both the diagnostic LtHash +checksum and the original bad bank hash. Undoing only the 33 credit-only writes +then reproduced the exact expected footer hash: + +| State | Bank hash | +| --- | --- | +| Original 33 erroneous writes | `CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG` | +| Preserve the 33 inactive accounts | `BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm` | + +`TestEpoch116InactiveStakeBankHash` checks each reconstructed erroneous account +against its recorded data hash, checks the bad bank hash, and then runs the +production streaming calculator, spool distributor and bank hasher. Before the +fix, that path emitted 33 zero-lamport writes and produced the bad hash. After +the fix it emits no writes for these stakes and produces the expected hash. +Other slot effects are held constant; this is not a full signed-shred replay or +a replay of later slots. + +Reference behavior: + +- [Agave ab655329, inflation_rewards/mod.rs:254–270](https://github.com/anza-xyz/agave/blob/ab6553293094e59dee7d3e7c928c7fa1023d0684/runtime/src/inflation_rewards/mod.rs#L254-L270) + restricts skipped-reward credit advancement to effective or activating stake, + preserving the explicit forced-update cases. +- [Firedancer 57d39904, fd_rewards.c:649–678](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L649-L678) + rejects an inactive credit-only update. Its + [points calculation at lines 908–917](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L908-L917) + retains the inactive flag from the existing activation-status calculation. + +Run with: + +```sh +go test ./pkg/rewards -run TestEpoch116InactiveStakeBankHash -count=1 -v +``` diff --git a/pkg/rewards/testdata/epoch116/inactive-stakes.json b/pkg/rewards/testdata/epoch116/inactive-stakes.json new file mode 100644 index 000000000..e6dec5b48 --- /dev/null +++ b/pkg/rewards/testdata/epoch116/inactive-stakes.json @@ -0,0 +1,1693 @@ +{ + "parent_bankhash": "B2Yi5ecGCiZRpvjdrLeQj2SdSncKgi6zEmVJxpMfzfyB", + "blockhash": "GBQcgk9JL3YFaK1GvpKLGpCbpmJ3My4oYWztnAJ3wirY", + "expected_bankhash": "BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm", + "original_bankhash": "CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG", + "base_accounts_lt_hash": "Mnn6NyRVWRvlXK1fJ68LNsMwgwvv5fH/brKehzoh9D+CaKo7hT/Wa1o3UkLLI7xs14mAj7EgyHCauhEVsMQGISCDkpWwejJy5tH/1Ov8yb+FM+1ZkSIO1vL8vsxDGipgAGfNXO+nGnh/ldg6Jbv66RgGdXJtR4b2IMUD970fj9z7229YpWP7PIwy70TEJ5VQuS16bXRwJlA8cRSSn7RMx9/yW2G4WhS+XAGi13N1uPsIHUkm6DePr4WPRsUi4UjMP3F8fpYUCAyVi8NMeFBPPGHZa/ZkR2C3NTk1Fs/jpI3UTw7MI61RnDFZs2vd4YYx6gr2g2jRKBX2/3NgSEK0112FCnB6T9V+LM6b+flCUlL5OddAd5MtN+4gDN2DlHAX20jb8VTWqwvb602v877ha/L+YK09+D+Gr9+iNxx8CSCAycb9moun0LGVSg6TgenXfhwKQbYlLtbIB7bD6YCrLBeP4A9rifOBIobmidjeYBjF7kicq8vMrH3j+flLSsn4fHXsssP9edqitP69ycRm7nLrYEFTR5+tl985wzrWi9DtOnImREgGo1oHBqGxXe6BksLdnCSei1i9SOX3oEaxKWf8bcf28ocL3pf5JKw3cBNALWTdjda/w/CMkg4N6MB6pUurImpm1M+Sg9O2Qyvv6v24RgG/TBdqMTmt3b46einSa+KZO72BQ+c4KkihabsRBpJNYpLQAiSH1Xm6FWoSuu63pRnGRyYokF3JC7K7M5wYpqE2EYJUUVM/SH7uS/JhOmWZXtGSpBcq0kQjHfIIjsJhrD2om9NJgVxEf4DhwK3ZTHUdTE5SY5aOwB/HVNlaTQd2Ab1ZsJncKAGIz1LcJpLB1uI3CtjtTwKUYrYTukHpWl2OQySqy0gmEzNPqmDUbhMUuEHcvJBUw30wLusR/BmlR9RErlxtigecyID+K9O9sTssRnAfuCY6Hm9SQP0rd6VRbIDqcfvSp4nOdYIZhx84IcN3SkN2Q9ipCc/OBCf0b2vIUu/7YS7VPv0gUQCKgg+kHJzZzUD5vHsOTw2623uxug2Y3tkiaVgrh861s2lJTjsSz351rMZQc0nzjis+GAoOuzo34zpGX3duLHYzxxdiRcX0jSvM4C9CVX47dQsYnltzVgOfXbeLYxA19SBjrhErooboDLe5OfeFeHGMIUO01sQ7dXXTzLZt03Opb8UjTbAtV3zCN4l08vObalYPFm4zEuTGYVnWIrJp4d2sEg/p7ejmWQNif/36zSLANp6ILgJqbVp4uATjuPPQD/nEcqlTspMy3f9yDK2Du3DjOcRNR6w17AtQvKsyH5dNo3w9JWG0a9HeBciz5CwdqYZ8cTi9jouD9Yj8roxAH9l63gNNJ+RgO62/17EBw9fS3pbNDfyJbZfvbnFq1KpIHe0LqK0K7hL7zTpcu/KgDKJLLZZe1YeDUm8jlVlhpbuDCBr3oj5N2p37581r5thU0F4f8//vqHm8G7i2UWt7sXtxJJ5JhUDV7RLYLfmFRc7ZXBJ7Z9qAxTo4WIWWfb3QsxSgZK2VfBFerxHzE8QlDCukgAHnjRxx/sVAHU5kqTI+jHpXsDogLmCTvaTuIW7x2WAmdEazWRi//5J9COsBt3wOwQwwBq7nRfj8w7Zo/M5xwbdpTv/JsvLQKmc51uaYVBSHZy8q54g7jo+j75jiA6TCh4KzMejWbH+c9Ew44ATW5mLiPFCJxmkZLI9Dl5XXxTtspSkHlTinziKS8+FCFUD8iCdnXKkym5gD1oO0imsGvBRs1qqnwbIaOv7esnnh4n8V3ML5La3du54etTnaGlAI35lGv2AEc2otPbHDvffI3PsMuEi9YkktqO7rJmFieUIU/uNiMevudlQqqrWM499kCOI7C6drxi8KfdcCd9b88nGSL6MlKu7F0cwttJWCb2uDTWEA2vYbxpUtDxZ8xYNeCdKDHJbNGk9viaLneQ0BzE3GQLnZPEYAuVkAbh+mWN0o/kmL8DQF4wpa1xBFUBBD6zhhM5zL6yAF5zT5k+bGkrIGS6+D0OteyVZnB+2y/ttjRknqhc+5CQVr90dEBsk+ivEsz7Ta/vZ4ibYJOgHq6z4OfagQWaWCiYBaQoTfKhLpekY3HrIMhQSsXuH8UjiRULhj0mqLnX7UarLLCTAvqEsvCgQxH1vzNbeomzI3HlJtWTrQz/vv8DOPXzJE/aYuxbyTAhq6uZH6TkH4qmv9lOh5dLbtQVZq1uf0upbJgiamwRyf+PGKqBe+rcYz3Mp/2aJk+jAYx4WGVdd66COzTvV///vH+M+/E7Yj4ueEnqsQVuLMtjAW+xTu9UiA/eb/mMVBpzeRaOUELDeYHwiKR8HiwLH73UO2AGuFsCAhqQPsYP/SSOaKvJinVKe7uclMgEgSAknr2UTxVjoPay18vynGYJOYDBrxFaxaDQbCglcjxfJW016E9SGRbxxlZZOqpr9qZtO20MgGMsgotBKniJ87ssxlJDJ+Kx7w2MWcIN5dv7kFj9KVIiBJ206fg6fQUHB+/4g0x7rBedhfmDtWizaNyqNcpWHfRMhoIPI+nOij3u8ASv/Z5HgqvFDeqn3MBix5Z/4+vRpS24pN83kvdbaGnyLSZ6jNuq2wfdN2PYYlnUczLvrwD7QjUkf+mf3PVhZxM+56+lnYG52tToqXIqAtrq4fkJq73lCOLwnd+8cCkdbQtx3qQF+NiR1eMiewNwz6cVTKoMdNqeYsGmXRNMY=", + "stake_history": "dAAAAAAAAABzAAAAAAAAAJ5AeDrqWjwAaanDLwG0AQBXHmbgZhcAAHIAAAAAAAAAJXm6zkOLNwDYUEVSW28GAPX9Y+xHMAAAcQAAAAAAAACbOD5ij1s7AH8vYWThAAAANhpRt13RAwBwAAAAAAAAAEKuN0SOQEAAGg55mi/jAABjcyFX7dUGAG8AAAAAAAAA243TwqZJPgCgvqaXc7gCAGxJJAa+wQAAbgAAAAAAAADlcqREpWI+AF6z7ht6CQAA2yghg6ciAABtAAAAAAAAAOSauglaSj4A7jLIrkcZAABAJ1t/KwEAAGwAAAAAAAAA1bfp3PxJPgBswjk12BAAAMxhZcmqEAAAawAAAAAAAACgpYOCjzs+ADdbCt+GKwAAyzEg90YdAABqAAAAAAAAAHdjrA3dQj4A26mxDlISAAB8hECCzhkAAGkAAAAAAAAAAEdfVIHpPQBW2hsNHo8AAC+XfvrrNQAAaAAAAAAAAABo9vZqfuU9AA58Eg7PBAAABIpq+vsAAABnAAAAAAAAAJFweNQULT4AqV72RLRIAAD2LXPQeJAAAGYAAAAAAAAA9qEC3Yo/PgCF7iNpyQ4AAAZZwTZvIQAAZQAAAAAAAAAuQbhfCTk+ANbBFd9uDAIAstBbUhgGAgBkAAAAAAAAADB84TH1QT4Agz3YD9wMAAAUQT/28xUAAGMAAAAAAAAAgc99eYkCPgDGnMraNEECAIGF5S/2AQIAYgAAAAAAAABNb2WWlro9AGWphlSdhQAAyYxLbNc9AABhAAAAAAAAAFmW5xtaxj0AF1Ks46MDAACV5g+0lg8AAGAAAAAAAAAAQgPK8m0kPgD7oWlICjIAALtPvWhHkAAAXwAAAAAAAADvAGYj8wo+AHtuvjg5PwIA4Gqgee8lAgBeAAAAAAAAAPRPsb1kNj0A6J0nTzorAwC0i2yT2VYCAF0AAAAAAAAAIiDIxHosPgBZmiIiYSYAACkVK22lHAEAXAAAAAAAAAAJvmQRggs+AAIGnLbqIAAAWMutxRwAAABbAAAAAAAAAL/QM8VAMD4Aw7etwKcAAADauqqckSUAAFoAAAAAAAAA65RRlJUVPgDMSCTC02cAAIfN5q1YTQAAWQAAAAAAAACP/BZcKsU9ACAmYvigdAAAiDfz8mMkAABYAAAAAAAAAEK6x8WZyT0ArEiebikKAACSNlCKxQ4AAFcAAAAAAAAACTBqaG7SPQCoR3XOSAsAAF71j7lJFAAAVgAAAAAAAAAjWmA61589AMtdU9w4VwAABJfvbs4kAABVAAAAAAAAAG/JhOUTaT0AUMvSkd44AAB1Iv0rRAIAAFQAAAAAAAAAfK7kidvtPQBNKEsZw0gAAC4+dp63zQAAUwAAAAAAAAB8tlqi8ts9APvZJnnSHwAA/WCY9BIOAABSAAAAAAAAAMAYELB0AD4A+dGs4HEaAADOy8rLIT8AAFEAAAAAAAAA0TQODTb8PQD+ZIrMvQ0AAKLsbq2qCQAAUAAAAAAAAACb6i9EQwM+AEicn0T6PAAAyAM4YTVEAABPAAAAAAAAAJeDquB9Qj4A29/hUTE6AADzZwr3lnkAAE4AAAAAAAAA3YhSTtglPgA1M80MFSYAAGDVIn6dCQAATQAAAAAAAADavbp48hw+ADUoxH3YCAAAmmSrqRsAAABMAAAAAAAAAM9KZ53SHT4AFPzaJG8EAABBh/RQegUAAEsAAAAAAAAA+HspIqN4PQAJ3QG6ZLgAAIXaa4phEwAASgAAAAAAAAC0Hzu/esY6AHt5mUByuQIAmiDhzXUHAABJAAAAAAAAAEMqtGP2cD0A84gQpRYFAABAZYJ8w68CAEgAAAAAAAAADk3OwuxvPQDAS3n7GBkAAE+VB8A4GAAARwAAAAAAAADRsZ0Vx1U9AHLLb4SiHgAAZfdTUKYEAABGAAAAAAAAAKNb7H87Qj0AZZCx2/oiAACT9t0tmQ8AAEUAAAAAAAAAQvmdRxdOPQDVa544vgcAAJYPouDEEwAARAAAAAAAAADx6iXLmDU8APdU+ynaJwEA6/70KIYPAABDAAAAAAAAANZyds2GODwAFLJEEZYGAACa0y6VrgkAAEIAAAAAAAAAe6+cLig2PABhXBlazAsAAC0teOaYCQAAQQAAAAAAAACxff6Ze1E8ACrBuEaqiwAAJkWtgCmnAABAAAAAAAAAAPLJlmbtbzwATFPM6gYKAABz5YOHoygAAD8AAAAAAAAAlXqLG0GDPADGPdE0LRAAAJvWmUOwIwAAPgAAAAAAAAAGKrZkAoU8AFZorVhRMQAALFQFTUEzAAA9AAAAAAAAAAoUqYH8sTsA4IDLEML4AAC9/5aE6iUAADwAAAAAAAAAuzL+y6ZrPAAB8TW7cSgAACoQ4FtL4gAAOwAAAAAAAAAn1bWd5RM8ANwhM7NEiQAA9kqDwa4xAAA6AAAAAAAAAFxafIPkIDwAK2eWKPQ2AADVJYEPHkQAADkAAAAAAAAAViHBYFraOwD4j7g0FpAAADQ7TdK4SQAAOAAAAAAAAAC0B1p+6RA8ALCVqnGzOQAAclnWd2pwAAA3AAAAAAAAABJqVZSu8TsAxZ4RSV55AADnTpfITVoAADYAAAAAAAAATOvQCGvlOwAInOaCLw4AAGlNI3AVAgAANQAAAAAAAADiOkL7a3I7AP3nWviafwIAPurnf8cMAgA0AAAAAAAAAISnBw5hVDsASNjGSPMhAABXGeKPGwQAADMAAAAAAAAA75Bzrs90OwAB3HKPxhQDADDloUJeNQMAMgAAAAAAAADdNGOwmVQ6AKrWSFPxbgEAhqz8EelOAAAxAAAAAAAAAKVlV2M5bTsAt7J6MWEZAABWgGutNTIBADAAAAAAAAAATdF/eGFmOwCMYa1ndx8CAFnG6qTIGAIALwAAAAAAAADaHqDgbWs6APgZHPi53gEAfGTrf/rjAAAuAAAAAAAAAHVDwpv5ozoAnUFqwhSoAAAwz2LP1eAAAC0AAAAAAAAAaE1WTy5QOgC7Lb6VWeQAANrlRLzBkAAALAAAAAAAAABMwuR/ciM6AEHpLe8wUwAAcI0k8acmAAArAAAAAAAAAGUV3bRCGDoAkDAWSOEkAACeekS34RkAACoAAAAAAAAArx9e+fYeOgB8ySqUlQgAAAs3i/h6DwAAKQAAAAAAAACNHcp9cl06AJaDFRAWbgAATpEP8MasAAAoAAAAAAAAAJSotzIIpzoAFwSV826JAAAy92cXMtMAACcAAAAAAAAAPFUSdQ65OgDoJJ6NhCcAADCwfvHBOQAAJgAAAAAAAAAA6p5t5546AMm2R+E9JAAAo0xN31AKAAAlAAAAAAAAAO6h+UfxQToAnx24iv5qAAAddIZ3PQ4AACQAAAAAAAAAKa2tRNmFOgACcRzbgS0AANYPtIOhcQAAIwAAAAAAAAAGOyVz3zI6ALqiF0x0+QAAi9QI3qumAAAiAAAAAAAAAOcmAEoM+jgA4sBkWs5BAQCzLciCLAkAACEAAAAAAAAAdn8sBEvPNABKZYiSUOEFAKq488AelgAAIAAAAAAAAAD0lsvPNDw1AL9v6SJ28wAAlbOWU4hgAQAfAAAAAAAAAGk2uzPpvTUADkS44IFUBAD3aJUNxGQFAB4AAAAAAAAAG4YELojOMwBT2ECjQoQEAAUoip3hlAIAHQAAAAAAAADYscqSXXQzAFufDzqhIgEAGMvVnnbIAAAcAAAAAAAAALtrLOOo4TIAHUaer7SSAAAAAAAAAAAAABsAAAAAAAAAPlfpgiWuLgD+x4A4R4oEAAAAAAAAAAAAGgAAAAAAAABhNJyrrrQrAPe6b3aatQcA4T05ZoT1AAAZAAAAAAAAAOgqRqBFESoAOA1GOSyqCgD7mGj70iUCABgAAAAAAAAAQUvKQ90zKQBbee1BlPgEAIO7vG7m1wIAFwAAAAAAAAB0/UZWrDMpAFfZoR9TcQgAbNhDyqKIBgAWAAAAAAAAAGTa4bqZwSoArSc7BflKAgATQNWp51cJABUAAAAAAAAAuJa1PExVKgAIutFU5XQAADZwSSvTCAAAFAAAAAAAAABdiCSRT04qAEvh4+XMBwAA6Ma1tAYBAAATAAAAAAAAALBNfKM56ykAxKmk5NtrAAAgpzgl+ggAABIAAAAAAAAAcrdEoMDgKQCuPiLWZmYAANwwSk8mXAAAEQAAAAAAAAAUXTuSI9EnAILkep/EswIA4YaSbF2kAAAQAAAAAAAAAEhOd7PGYiUAJd7/3yVuAgAAAAAAAAAAAA8AAAAAAAAAVP+zxFlMIgDlgfGO3iIFAAAAAAAAAAAADgAAAAAAAABT2ixTNXcfALK/SZfhyAcAAAAAAAAAAAANAAAAAAAAAESpbbv03RwARWz5UURhCgAAAAAAAAAAAAwAAAAAAAAAQFrTd6Z7GgDMXyiGppQMAAAAAAAAAAAACwAAAAAAAABcnB0IxEsYAH3pX8hj/w0AAAAAAAAAAAAKAAAAAAAAAJCaKkMgShYA38jJYAvPDwAAAAAAAAAAAAkAAAAAAAAAF58Ll+hyFACFLkwOVlARAAAAAAAAAAAACAAAAAAAAAAxtAZmmsISAEKE8AX0/hIAAAAAAAAAAAAHAAAAAAAAAHXDQI4BNhEAACv2ahmEFAAAAAAAAAAAAAYAAAAAAAAACt0HxSrKDwDjq1Ie9r8VAAAAAAAAAAAABQAAAAAAAACxrhYtYXwOAN0of1y9rhYAAAAAAAAAAAAEAAAAAAAAAC2YPXwpSg0Ap97CpRE4FwAAAAAAAAAAAAMAAAAAAAAANf2HdjwxDAD7OmBvB/AXAAAAAAAAAAAAAgAAAAAAAADgL1gZgy8LACVZ0BHMARgAAAAAAAAAAAABAAAAAAAAADurkkgTQwoASjP+BeaxFQAAAAAAAAAAAAAAAAAAAAAAgGxxLSlqCQDgMpi3KeAVAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + "stakes": [ + { + "account": { + "Slot": 6210151, + "Key": "3mZwezEBuKcCs42NfAEZfdCbCMyPaV6wnus263mgrhjP", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAI9JSTu3nHzfydAzLjeA87Aht6Ost2Mol90Y42gdbSqB9HisvA0AAABnAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACZzNzwgAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7e91c9547ff107e827f22f226df525ee83be7e9eb22bc8afd812c92c235c69ba", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9430889728015, + "PrevCredits": 9349889838233 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "453ETE3U7Usc4ss87sszs3EvTR5ZYQQWcZHgs9XiAZdY", + "Lamports": 3306766663971, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADSIHvAB+hqCeyUMqrtA7RF7gf0pF5+WOdx24l28NcLZoyuE6gEDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAALJPnDAREAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "88983f1b8c3153b155fc54b8ffcdbe468b3356ce007cdab77d95cab1d3c50dc3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18772882129308, + "PrevCredits": 18699280524299 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "462wmoxLUzHfxcURxwkMj5f7cmeVXeSeaFU5uXd9jrQE", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAB7glTPNCA5Xp0adNqBsSaCIXSwoB3DhCXGyxm9mwIQut+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAABJX296lgQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c0c400478bf441c374e56141b0122846da672b9b1a5d01157374dadfa83649", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5082302209818, + "PrevCredits": 5044345724745 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4aNZYt3giNatv4J58sYHwRNs8L6ZQpLqG9GZiyqrXUqS", + "Lamports": 2329209663118, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAK8Fa5ghODdNV981iyfO/aADxuCTmU56Dtg15Wf9Xk+LDhmUTx4CAAABAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAeHbzX0wkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "78674eaaabfbe62cf2e558bd33e01dd53a33a7a4024e52eba673bb1acf73ff83", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10909280173169, + "PrevCredits": 10805462179102 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4u4XDhXRtXA9QqP8uC9vpoyw4KLaeWHm7g2gP4aLDGK1", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABtRAg7KdM61d5wq/bIMit41UurMHM2/gwzsJrDkSJWr9HisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC8dSEI5AgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f49eb225b3f858e7609ffecf5e9d2734cb28fe754ca697800e0a44a144ab0f3d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9855957906862, + "PrevCredits": 9775481976252 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4zrGtmZj2s7akr4FyiJ7E3fQY4n59cZoVEmL192GttRq", + "Lamports": 3956852941834, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAApBQ/86W982eATp1soYW/TvCCjcBXX1dK637et+U7Qaio6tRpkDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABz+BFZ1gkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6a0c148dbf978bfec9e6e5e3227ad57e6089c056b2d23730b85be7bdcc2a3384", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10900522634265, + "PrevCredits": 10816222001267 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "5ZEUi72iZM7ozwisZV2jezjgR3BLBQToz8ZxpdnqYaue", + "Lamports": 10521999279304, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEvFm+VinHP4nTd2zkbv4cVplRt3TlVeTEidtXhPv7vISK/k15EJAABiAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABDEW5ATQcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c42ad17b70bd37283facbe1655a9401660d0ba96e7c57fad1ba452c46c77c8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8111781721836, + "PrevCredits": 8028374831427 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "6dNDcZRYeX7zoGirAbYxjJVCoHg14KLcy1ndb5a6hLtp", + "Lamports": 92402274844, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALJZyCh5aoPz7tWdy6VqoilN4lxeoCa10yfJpJBoYSa8nPx3gxUAAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACEXlHvNAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "4914fbc352aae3350f8dd39a94733fc3bd927e1ccd2ed471ac81318d759594f9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9114167602897, + "PrevCredits": 9023446408836 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7Xwjq1Pu1ibBufp4mdu51yrRzUoadak98sdj6BQhGmBH", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMb+8swy4nz1EbP63bJ/XUX49DQYo/CiTC3dFaYhgfYs9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACE/LmcrgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "01df0af720551818974311706b7a5ebb9c5b333eb90e877e0f227f08fd4cf9e3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8523194289275, + "PrevCredits": 8446535138436 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7kGbq8H94roSeiCvasCE8kWq2Kx1dCTZQTUnkiPqibTg", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAaFSKZ/FpeRshn8N9+v+AkQycTuIPwqrs5Yf2mgYnGv9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAF9xJssgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f04a2844299ed1e7c980b160a89a508f61a135ee5d333ddbb948aabb00b80a6d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8544891256299, + "PrevCredits": 8462898755333 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7uyNRdfhmk6SwJPfM172LF4ajEUKUkP9RC2okeB3RZsC", + "Lamports": 48931548636, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABLFnpY8I58W61QVYzmx6wfM3seg+Q0PxwXVhMb9YvpaXFhpZAsAAAAzAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABZwrzIpQYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d2a561a19b6f7cd09f21985876978ae5ab1e11c7cc9262c5a17caf9af6e0c92b", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7371534095782, + "PrevCredits": 7309107184217 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8QYDcU8pvmMKvovzUiqWq4H7ZFiWpbet4HwUnDy5XTYK", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAFgYhGqxkhQFABeF03ECpK9RUUF1hdavhflLAu/MvP/rAMqaOwAAAABtAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAs5ZJ1sQUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ff84dc2a8426d4deb78786afc87899cbb4c052b6c20346fac76f7fcd865026c5", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6318982736717, + "PrevCredits": 6259739911468 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8n3LHx61MeuZTF28hkHvsiEtKSEXLUQDYCMHS8sCQ264", + "Lamports": 3058382969924, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAM9GQ73y9miIEoMD1/GtvVm6Z6RC6u1sNUKM6BM6jWJ0xMaxFcgCAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABQGTGKHhEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "20b17fb68d98d290dfa4742237ac719219e23147a8f7cdfb27ac1be790f9e8d8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18900384617754, + "PrevCredits": 18822865164624 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "98joChCUmuTBYZx8xHDJzJzbF37HPuP4EHqm3BwD8KSo", + "Lamports": 132002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAite2/mh1w1kW4RMXwAjDtR4kt3TcBCr/oy5ngX7hOQACjQux4AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACf+Oklb0QAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c11ae1d9490b3f0876c515a564337a0170c21fae8c3cb3f43c66e5dee2cd34f", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 76179624781196, + "PrevCredits": 75244168149151 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "9PAPBcFDq536ZzMWhNywZAYPZt5mF92VadrJc5ctoQqY", + "Lamports": 1894659468576, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALo09fj+TmHMhldcHlC9iwfK4NZlNl+rkERD8XdoYYH1oFdeIrkBAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC9udrpXwwAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ba19d0de0f1d7ccded2e93dd4b9d57b8077397c396005d1918b4d6dd719196b7", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 13714985011803, + "PrevCredits": 13606084852157 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ALrrFs8DAkNHjyce6LhwkotcBQp8U9dGKb4prUHZp5GF", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACZSY+h/WqtJAAWy9FwQ5zvCPyGbQUYwp5yepcxIn0qbAMqaOwAAAABwAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACXOQj8AgEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a96af37ed2c6d36cfc77971b4432cb7c62f4e045b3883a5b7e2e6434fa21dec9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 1147511243813, + "PrevCredits": 1112329959831 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ARPzersuLPyg9ZJaxYuKc2eDKKg8t4qWXpTy9TzKjLJU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAF6PbC1s5mob1cKneVa1heDI7d/UjwFUwoZKG6DN1okV9QSscAoAAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAZRdBpGAcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c3b359b3e2c72b291d1703977f464395b1772f9197466913c9f87661d3ebac0", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7864995243760, + "PrevCredits": 7801435866393 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "AXC6s8QGstr1nnTJ5ranGEeSZ7rcHKVeQav4r4J8icJL", + "Lamports": 2951443346384, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAD2DXLDpqE+RtusMr3795aLUMJhXsVQnXUHgtyhqt1j0UJ6YL68CAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACQUqzjgQ8AAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0e44daa0385ca084667f7e44f0fb96571b6b963f6bd046bf66c2c7fb31e232ce", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 17293215270489, + "PrevCredits": 17050544919184 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "BWiQJLo1TGhbNp3ionXVFPnSgfGimGU9FVHm1hdwLKbJ", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADUkqa5BeE51KvSv12H//GyC8rI6ScrXABI9UawzUXiot+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAADlkymHrAQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8cf243491cd54f9d1efc732287caf92bd392dfc1ad410c74c94e2d8f4e4deb18", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5181125525357, + "PrevCredits": 5139048535013 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "CbFTDpV9HjcnVQReCUwuZmLMp6PjMoWZBA7VTvLzszQ", + "Lamports": 5114239502629, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANp1F9fR0omLpsfk9YL0X7w0EuA6D8E39bIn27KGqRgppfNKwKYEAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADfLChErwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6bf6057cfa3c4d2a7d786bb5f325032604cca8eb30391468f2caf1b71406ce57", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9639820890790, + "PrevCredits": 9548855782623 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "D5Zx5bsdVDg2kBZkPvaF59ULdKd8LACtScaWE7NEAbtw", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANbDQeBxoTnMt62Ygze0/L0wq3Tv5ynlZXdjhuu6lY/4AMqaOwAAAABuAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABO6xz/EAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2480dc54ffb62581eec0ca3bcb7201b2ba4260a97bdc186e2f2345241abade3c", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6731352374437, + "PrevCredits": 6670069328718 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "DNCCPW9Bsv8SdcDMFDA7b2HT5jKxgUnBrA2gkKXwe5Bg", + "Lamports": 44252278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEZlguB/hNNN/i6p/xPZEHXCTr6qIxawtvTnx9A4WE77t1qBTQoAAAAWAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAJpOYcEAUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a02301fdf180b2f542124caf7da178cae4397acc833030b9dca28116d6e2b747", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5630002073869, + "PrevCredits": 5566762492937 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "E2TDLwJNmzvg6K7MhrkQ3HV4n4mp2zL49FqGwV4tFUfC", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAIZbuXtVwiaoNZBvLxeR/DEy46f1oQ/1DEqV1R+wxiWyAMqaOwAAAABvAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA+JHwoWwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2ecdc0e446f6e0c112c947c79b6976cbdeea776ecb4fe62ece89b1f725e3ae48", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9263388828449, + "PrevCredits": 9187614270526 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "EA3MVKFbieGtDZDG4bmCiwnE2HWndMuWcjMjRNSrbMBC", + "Lamports": 3787600669028, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACGc29KCGn33qCWHYLj1Li6Q1LaT9ffcV8Q3PJ6AssCM5NN03nEDAAAGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAChfXpwQQkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "c0bf9bd502e0b675a186b4b855812ae1915d04110a951499d911e6174ed2b2b2", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10252916692527, + "PrevCredits": 10176664599969 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Eijmmf2xFvjStkcu6WQRj93x5iYP1uv5zuZP8PARdqd", + "Lamports": 3449661803946, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAC3bR7fFvVz26LQA9LQq1JGFKWXl7hg6BaC5sp0SpTflKvi6LyMDAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABF19K7FxEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a4810c657cb2161f3c47f0452d78dc74d9d4597763d82c3dfd48b3d1dc43dfe9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18871239559140, + "PrevCredits": 18793633077061 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "FxhNhXDkoqMTcXqyHxYXW8BJAdpWH5qdyz3qgpiySrKn", + "Lamports": 71250450995, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIums8S4lhAAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ab580bd8bb513651006208d529d58dbe8e40d0306141e4b9bbe32c0559040af8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "GAyfc6a5E5uzXQgQ4KXHuJyunyGNtLGZgUUdWKEL4Df9", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA0QDCxOaKoUheu2WKN7BFVixKbK4ogx2BYksnFTbTzL9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACYANgjswcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d635c01a4fc8bc710b4ca6d9b0b0ba960b9ab3ad8ac0359ee8adbdd62ad30f3a", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8554183692312, + "PrevCredits": 8465981898904 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "H2dM1xYSeuZaujtByMa4jvPwUVYjfxzUL3K9iVLaCW9G", + "Lamports": 64469922880, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAiQo+w+o08xym3k1JBF+HYHIJZGTDaZaufK/simzs1bwB6SAg8AAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACe3jqUQyIAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b39033f338419cea17108c3989a6331d8272749095fcf3ad856031cbab264110", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 38085420659301, + "PrevCredits": 37673645039262 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Hh8u21hJnAvE9PMNBmNTsDUTseY6bYfwvbZQgj8R2vcU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJOHGqQMr9AgJ5hIj044ilcd/38t2ijdTubnJyZ6iT+d9QSscAoAAABGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABymDuk3wUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0569f4f3dbc16ab5c284f6efaefd1e2b6b5b8db18ffa1b1fbd0da98468c54d78", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6525335322632, + "PrevCredits": 6458091214962 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "HicPaN5ogXhE5jQ99WotjGUQLPJ11bdCqXjJhMKUq5N", + "Lamports": 1690908095830, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJ1KF6URsOxn3mdoAgUdjof1ae904ReOis3chjrZx0h+1h/XsYkBAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADt4ogGQgUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "47ad9ddb61ff1edee564852ac0e48ff45b47a377b0ca1a5f1e0aef74a27627b4", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5826444097536, + "PrevCredits": 5781135614701 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "c6Ev8GfADHdrGtvH1VqwH7JmN28PisZgGfUy64RE7np", + "Lamports": 3782196946839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMDb0jorArBVkC7lcchWRZHVxqD+PrkqtMdnxFvaxkbfF5JenHADAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA1TQUrswgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b361cd2d770b1b3432d58f8098c654f300754fab58787d6033dae1e724618467", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9647673972566, + "PrevCredits": 9565613935925 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "eHboMWq5bmx1DxdKw79zpse75XLk2GEgeRnJG9nysJn", + "Lamports": 5125666350406, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAPznsTwPeDl/u5+jig9MvOBeS/RKUvjM0z9ITapNTvPoxs9iaakEAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAEpkAsaggAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8e052b16e36ecb6e80919ada5456816e25e3b804fe9119319cff9861e7242e25", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9334067720448, + "PrevCredits": 9252101989892 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "v5pSSEZAuvG9ewGhdMNVcTVHeJPzFnGse3mX4FBEyLT", + "Lamports": 1082110856099, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIumI3ay8vsAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "053f427dd888bbe45c6e0f772faf6eac1f9ede5822cc74477c757eaac75cb8b1", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + } + ] +} \ No newline at end of file From 2d0ea6f96867e1b91eba8a2c1a83d55ade961b50 Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:24:36 +0100 Subject: [PATCH 084/111] replay: checkpoint reward windows only after verified completion Keep the epoch boundary replayable when distribution consumes its last spool before a failed bank verification. Apply the hold again after finality gating, and commit the full reward window atomically so ordinary fold batches cannot leave an unrecoverable active EpochRewards checkpoint. Cover failed distribution, finality lag, completion generations, small fold batches and failed-commit retry. Document the epoch-116 diagnosis, bank-hash evidence and verified recovery checkpoint. --- docs/epoch116-reward-divergence.md | 82 +++++++++++++++++++++++++++ pkg/replay/block.go | 36 ++++++++---- pkg/replay/promotion.go | 17 ++++++ pkg/replay/rewards_retirement.go | 13 +++++ pkg/replay/rewards_retirement_test.go | 63 ++++++++++++++++++++ 5 files changed, 201 insertions(+), 10 deletions(-) create mode 100644 docs/epoch116-reward-divergence.md diff --git a/docs/epoch116-reward-divergence.md b/docs/epoch116-reward-divergence.md new file mode 100644 index 000000000..9d93ab52c --- /dev/null +++ b/docs/epoch116-reward-divergence.md @@ -0,0 +1,82 @@ +# Epoch 115 → 116 reward divergence + +The supplied run (`320ce8da`, started 2026-09-18) passed the epoch-boundary bank +at slot **6,264,000**, then stopped at **6,264,001** with a footer bank-hash +mismatch. Both blocks had zero transactions. The mismatch was a deterministic +reward-account state divergence, not a transaction-execution panic. + +At the boundary, rewards for epoch 115 were calculated and vote commissions +were paid. The following bank distributed one staking-reward partition with +724 stake records. Of those records, 33 paid zero lamports but advanced +`credits_observed` on stakes that had deactivated in epoch 114 and were fully +inactive in epoch 115. Agave leaves those accounts unchanged. The missing +effective-or-activating condition in `shouldForceCreditsOnly` caused the extra +writes and changed the LtHash. + +Reconstructing all 825 modified accounts from the saved parent accounts and +the diagnostic's per-account hashes reproduced the original computed bank +hash. Removing only those 33 writes reproduced the expected footer hash +exactly. The [reduced fixture and regression](../pkg/rewards/testdata/epoch116/README.md) +preserve this evidence and link the Agave/Firedancer rules. The code retains +activation status from the existing calculation, so no extra stake scan is +needed. Credit rewinds, disabled inflation, activation-epoch updates and +fractional rewards on effective stakes retain their prior behavior. + +## Checkpoint failure and recovery + +The distribution decremented its remaining-partition counter to zero before +the failing bank's footer was verified. The promotion guard used only that +counter, so graceful shutdown released the epoch hold and persisted slot +6,264,000 with an active EpochRewards sysvar. The spool had already been +consumed. That checkpoint cannot resume distribution directly. + +Promotion now waits for the successfully verified bank's immutable inactive +EpochRewards state. It also waits for that bank to pass the normal finality +and verification gates. The boundary through completion is committed in one +fold, including when the configured batch size is smaller than the rewards +window. A failed bank, a finality cutoff inside the window, or a failed fold +keeps the prior checkpoint. Memory remains bounded by the existing tail cap; +the completion fold can be larger than the usual batch, once per epoch. + +The supplied data includes a retained fold and transaction-status checkpoint +at **6,263,999**, the last slot of epoch 115. Read-only validation confirmed its +57,677,519-byte transaction-status checkpoint, complete retained-root coverage, +selected block identity, inactive EpochRewards state, and all 122 account undo +targets for reverting the boundary bank. With the fixed binary, the existing +`run --rewind-to-slot 6263999` option can restore that boundary and recalculate +rewards, provided historical blocks/shreds remain available. Keep the normal +node configuration and signing-history recovery checks. A direct restart from +6,264,000 cannot recover the missing distribution bookkeeping. + +The supplied local shred spool does not contain slots 6,264,000–6,264,001. On +2026-09-22 the configured primary RPC also reported 6,264,001 as pruned (first +available block 6,764,691). Replaying from the retained boundary therefore needs +another historical source; otherwise use the fixed binary with a fresh +snapshot. Original accounts, ledger, logs and signing history were preserved; +the fix was not deployed to a live validator during diagnosis. + +## Validation + +The branch was fetched and fast-forwarded from `320ce8da` to `be29319f` before +implementation. The reduced production-path regression failed on `be29319f` +with the incident's exact bad hash and passed after the fix. Additional tests +cover inactive, activating, effective and cooling stakes, forced credit-update +exceptions, failed distribution, finality lag, distribution generations, +atomic completion folds and failed-commit retry. + +Passed locally with Go 1.27.1 on linux/amd64: + +```sh +go test -race ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -count=1 +GOMAXPROCS=2 go test -race -p 2 -count=1 \ + ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards \ + ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... \ + ./cmd/mithril/node ./cmd/mithril/configcmd +go vet ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash +go build ./cmd/mithril +git diff --check +``` + +The CI regression job now includes `pkg/rewards`, so the incident fixture runs +on pull requests. These checks do not establish live post-restart catch-up or +later-slot parity; the original node was not restarted. diff --git a/pkg/replay/block.go b/pkg/replay/block.go index f71df8c98..e571c98d0 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1941,8 +1941,8 @@ func ReplayBlocks( var highestExecutedSlot uint64 // highest slot ProcessBlock has executed; bounds the promotion-gate walk // While partitioned rewards distribute, promotion holds below the boundary // block so a crash-resume always re-runs it (the distribution bookkeeping is - // RAM-only and not reconstructible mid-window). Self-clears when the window - // completes (NumRewardPartitionsRemaining reaches 0). + // RAM-only and not reconstructible mid-window). Release requires a verified + // completion bank, committed atomically with the whole rewards window. var rewardsHoldBelowSlot uint64 // Alpenglow finality identities captured at observe/ingest time for the promotion // gate (the tracker's own state may be pruned by promotion time). Pruned as slots @@ -2187,13 +2187,9 @@ func ReplayBlocks( } promoteThrough := safePromoteTarget(lastRootedWatermark, verifierRequired, verifiedWM, replayDivergenceFloor) // Partitioned-rewards window: hold promotion below the boundary block - // until every partition distributes, so a crash-resume re-runs the - // boundary and rebuilds the RAM-only distribution bookkeeping. - if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 { - if promoteThrough >= rewardsHoldBelowSlot { - promoteThrough = rewardsHoldBelowSlot - 1 - } - } + // until the completion bank verifies and is eligible to fold, so a + // failed distribution re-runs the boundary and rebuilds its bookkeeping. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { // Operator signal: promotion is fully stalled (verifier lag, // divergence floor, or rewards hold) while finality has run at @@ -2224,6 +2220,8 @@ func ReplayBlocks( mlog.Log.FileOnlyf("alpenglow gate: checked=%d matched=%d no_finality=%d no_local_id=%d", gateStats.checked, gateStats.matched, gateStats.noFinality, gateStats.noLocalID) } + // The finality gate can stop before the verified completion bank. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { return false } @@ -2237,6 +2235,18 @@ func ReplayBlocks( if res := promoter.drain(); res != nil { applyFoldOutcome(res) } + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr := unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + if jerr != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", jerr) + return false + } + if err := runFoldJob(unrootedTailState.committer, job); err != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", err) + return false + } + applyFoldOutcome(&foldResult{job: job}) + } promotedThrough, rootedCtx, perr := unrootedTailState.flush(promoteThrough) if perr != nil { mlog.Log.Errorf("rooted-durable: forced fold stopped at slot %d: %v", promotedThrough, perr) @@ -2250,7 +2260,13 @@ func ReplayBlocks( // when idle; completions are applied at the top of this function on a // later iteration. if !promoter.inFlight { - job, jerr := unrootedTailState.buildFoldJob(promoteThrough, false) + var job *foldJob + var jerr error + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr = unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + } else { + job, jerr = unrootedTailState.buildFoldJob(promoteThrough, false) + } if jerr != nil { mlog.Log.Errorf("rooted-durable: %v; watermark held back", jerr) return false diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index f6cedc25a..5debf2ffa 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -390,6 +390,23 @@ type foldResult struct { err error } +// buildRewardsCompletionFoldJob puts the entire rewards window in one commit. +// A normal batch cutoff inside that window would leave an active EpochRewards +// checkpoint without the RAM-only spool bookkeeping needed to resume it. The +// retained tail already bounds the size of this once-per-epoch fold. +func (t *unrootedTail) buildRewardsCompletionFoldJob(through uint64) (*foldJob, error) { + whole := *t + whole.batchSlots = t.overlay.HeldSlots() + job, err := whole.buildFoldJob(through, true) + if err != nil { + return nil, err + } + if job == nil || job.through != through { + return nil, fmt.Errorf("rewards completion bank %d is absent from retained fold prefix", through) + } + return job, nil +} + // buildFoldJob snapshots the FIRST fold chunk of the rooted prefix <= through // (loop thread). force also takes a trailing partial chunk. Returns nil when // no chunk is ready. A missing chunk-top context is an error — a context-less diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go index a03a00cbf..18783e382 100644 --- a/pkg/replay/rewards_retirement.go +++ b/pkg/replay/rewards_retirement.go @@ -15,6 +15,19 @@ type partitionedRewardsCompletion struct { slot uint64 } +// limitPromotion keeps the boundary replayable until a successfully verified +// completion bank is eligible for promotion. Consuming the last spool changes +// the RAM counter before footer verification and is not completion evidence. +func (c *partitionedRewardsCompletion) limitPromotion(info *rewards.PartitionedRewardDistributionInfo, boundary, through uint64) uint64 { + if boundary == 0 || info == nil { + return through + } + if info.NumRewardPartitionsRemaining != 0 || c.info != info || c.slot == 0 || through < c.slot { + return min(through, boundary-1) + } + return through +} + // observeBank must run only after successful block execution/publication, using // that bank's immutable sysvars (never the speculative global sysvar cache). // If the first completed bank lacks evidence, recording a later descendant is diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go index 460584eee..aaa169ae1 100644 --- a/pkg/replay/rewards_retirement_test.go +++ b/pkg/replay/rewards_retirement_test.go @@ -3,6 +3,7 @@ package replay import ( "bytes" "encoding/base64" + "fmt" "testing" "github.com/Overclock-Validator/mithril/pkg/accounts" @@ -53,6 +54,68 @@ func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { require.False(t, completed.retire(&info, 100), "retirement is one-shot") } +func TestRewardsPromotionRetainsBoundaryAfterFailedDistribution(t *testing.T) { + const boundary = uint64(6264000) + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: 1} + var completed partitionedRewardsCompletion + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + // Distribution consumes the final spool before ProcessBlock checks the + // footer. The epoch-116 failure took this path; no successful bank was + // observed, so forced shutdown must not persist the boundary bank. + info.NumRewardPartitionsRemaining = 0 + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + completed.observeBank(info, nil) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + completed.observeBank(info, testUnwindBankSysvars(t, boundary+1, 50)) + require.Equal(t, boundary-2, completed.limitPromotion(info, boundary, boundary-2)) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary), "finality stopped inside rewards window") + require.Equal(t, boundary+1, completed.limitPromotion(info, boundary, boundary+1)) + + next := &rewards.PartitionedRewardDistributionInfo{} + require.Equal(t, boundary-1, completed.limitPromotion(next, boundary, boundary+2), "completion belongs to another distribution") + require.Equal(t, boundary+2, completed.limitPromotion(nil, boundary, boundary+2)) + require.Equal(t, boundary+2, completed.limitPromotion(info, 0, boundary+2)) +} + +func TestRewardsCompletionFoldCannotCheckpointInsideWindow(t *testing.T) { + for _, batchSize := range []int{1, 2, 128} { + t.Run(fmt.Sprintf("batch_%d", batchSize), func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 7} + tail := asyncTestTail(fc, 5, 6, 7, 8) + tail.batchSlots = batchSize + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + through := completed.limitPromotion(info, 5, 8) + require.Equal(t, uint64(8), through) + + job, err := tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NotNil(t, job) + require.Equal(t, uint64(7), job.through) + require.Len(t, job.chunk, 3, "the entire distribution must share one commit") + require.Equal(t, batchSize, tail.batchSlots, "normal batching is unchanged") + require.Error(t, runFoldJob(fc, job)) + require.Empty(t, fc.throughs) + require.False(t, completed.retire(&info, 4)) + + fc.failOn = 0 + job, err = tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NoError(t, runFoldJob(fc, job)) + require.Equal(t, []uint64{7}, fc.throughs) + tail.applyFoldJob(job) + require.True(t, completed.retire(&info, 7)) + require.Equal(t, 1, tail.overlay.HeldSlots(), "later banks remain buffered") + }) + } + tail := asyncTestTail(&fakeCommitter{durable: accounts.NewMemAccounts()}, 5, 6) + _, err := tail.buildRewardsCompletionFoldJob(7) + require.ErrorContains(t, err, "absent from retained fold prefix") +} + func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} From c1f713461b4f95d2559d6387db6018f40da85dd5 Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:27:41 +0100 Subject: [PATCH 085/111] docs: remove epoch 116 incident report --- docs/epoch116-reward-divergence.md | 82 ------------------------------ 1 file changed, 82 deletions(-) delete mode 100644 docs/epoch116-reward-divergence.md diff --git a/docs/epoch116-reward-divergence.md b/docs/epoch116-reward-divergence.md deleted file mode 100644 index 9d93ab52c..000000000 --- a/docs/epoch116-reward-divergence.md +++ /dev/null @@ -1,82 +0,0 @@ -# Epoch 115 → 116 reward divergence - -The supplied run (`320ce8da`, started 2026-09-18) passed the epoch-boundary bank -at slot **6,264,000**, then stopped at **6,264,001** with a footer bank-hash -mismatch. Both blocks had zero transactions. The mismatch was a deterministic -reward-account state divergence, not a transaction-execution panic. - -At the boundary, rewards for epoch 115 were calculated and vote commissions -were paid. The following bank distributed one staking-reward partition with -724 stake records. Of those records, 33 paid zero lamports but advanced -`credits_observed` on stakes that had deactivated in epoch 114 and were fully -inactive in epoch 115. Agave leaves those accounts unchanged. The missing -effective-or-activating condition in `shouldForceCreditsOnly` caused the extra -writes and changed the LtHash. - -Reconstructing all 825 modified accounts from the saved parent accounts and -the diagnostic's per-account hashes reproduced the original computed bank -hash. Removing only those 33 writes reproduced the expected footer hash -exactly. The [reduced fixture and regression](../pkg/rewards/testdata/epoch116/README.md) -preserve this evidence and link the Agave/Firedancer rules. The code retains -activation status from the existing calculation, so no extra stake scan is -needed. Credit rewinds, disabled inflation, activation-epoch updates and -fractional rewards on effective stakes retain their prior behavior. - -## Checkpoint failure and recovery - -The distribution decremented its remaining-partition counter to zero before -the failing bank's footer was verified. The promotion guard used only that -counter, so graceful shutdown released the epoch hold and persisted slot -6,264,000 with an active EpochRewards sysvar. The spool had already been -consumed. That checkpoint cannot resume distribution directly. - -Promotion now waits for the successfully verified bank's immutable inactive -EpochRewards state. It also waits for that bank to pass the normal finality -and verification gates. The boundary through completion is committed in one -fold, including when the configured batch size is smaller than the rewards -window. A failed bank, a finality cutoff inside the window, or a failed fold -keeps the prior checkpoint. Memory remains bounded by the existing tail cap; -the completion fold can be larger than the usual batch, once per epoch. - -The supplied data includes a retained fold and transaction-status checkpoint -at **6,263,999**, the last slot of epoch 115. Read-only validation confirmed its -57,677,519-byte transaction-status checkpoint, complete retained-root coverage, -selected block identity, inactive EpochRewards state, and all 122 account undo -targets for reverting the boundary bank. With the fixed binary, the existing -`run --rewind-to-slot 6263999` option can restore that boundary and recalculate -rewards, provided historical blocks/shreds remain available. Keep the normal -node configuration and signing-history recovery checks. A direct restart from -6,264,000 cannot recover the missing distribution bookkeeping. - -The supplied local shred spool does not contain slots 6,264,000–6,264,001. On -2026-09-22 the configured primary RPC also reported 6,264,001 as pruned (first -available block 6,764,691). Replaying from the retained boundary therefore needs -another historical source; otherwise use the fixed binary with a fresh -snapshot. Original accounts, ledger, logs and signing history were preserved; -the fix was not deployed to a live validator during diagnosis. - -## Validation - -The branch was fetched and fast-forwarded from `320ce8da` to `be29319f` before -implementation. The reduced production-path regression failed on `be29319f` -with the incident's exact bad hash and passed after the fix. Additional tests -cover inactive, activating, effective and cooling stakes, forced credit-update -exceptions, failed distribution, finality lag, distribution generations, -atomic completion folds and failed-commit retry. - -Passed locally with Go 1.27.1 on linux/amd64: - -```sh -go test -race ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -count=1 -GOMAXPROCS=2 go test -race -p 2 -count=1 \ - ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards \ - ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... \ - ./cmd/mithril/node ./cmd/mithril/configcmd -go vet ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -go build ./cmd/mithril -git diff --check -``` - -The CI regression job now includes `pkg/rewards`, so the incident fixture runs -on pull requests. These checks do not establish live post-restart catch-up or -later-slot parity; the original node was not restarted. From 008258f58c13c0b6721d052f6bbf6b24d6d8cdff Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:24:25 +0100 Subject: [PATCH 086/111] rewards: preserve credits on inactive Alpenglow stakes Retain activation status from the existing points calculation and exclude inactive stakes from the Alpenglow skipped-reward credit advance. Preserve explicit forced-update cases. Add an epoch-116 fixture that reproduces the original slot 6264001 bank hash before the fix and the expected footer hash afterward. Include reward regressions in the race CI job. --- .github/workflows/go_build.yml | 3 + pkg/rewards/alpenglow_rewards_test.go | 53 + pkg/rewards/inactive_stakes_test.go | 118 ++ pkg/rewards/rewards.go | 22 +- pkg/rewards/testdata/epoch116/README.md | 49 + .../testdata/epoch116/inactive-stakes.json | 1693 +++++++++++++++++ 6 files changed, 1932 insertions(+), 6 deletions(-) create mode 100644 pkg/rewards/inactive_stakes_test.go create mode 100644 pkg/rewards/testdata/epoch116/README.md create mode 100644 pkg/rewards/testdata/epoch116/inactive-stakes.json diff --git a/.github/workflows/go_build.yml b/.github/workflows/go_build.yml index 50383467f..bb29441f7 100644 --- a/.github/workflows/go_build.yml +++ b/.github/workflows/go_build.yml @@ -17,3 +17,6 @@ jobs: - name: Build run: go build -v ./cmd/mithril + + - name: Rewards and replay regressions + run: GOMAXPROCS=2 go test -race -p 2 -count=1 ./pkg/rewards ./pkg/replay diff --git a/pkg/rewards/alpenglow_rewards_test.go b/pkg/rewards/alpenglow_rewards_test.go index 76136886d..ae9d451bb 100644 --- a/pkg/rewards/alpenglow_rewards_test.go +++ b/pkg/rewards/alpenglow_rewards_test.go @@ -116,6 +116,59 @@ func TestAlpenglowEarnedPointsAreNotCreditsOnly(t *testing.T) { )) } +func TestAlpenglowSkippedRewardCreditsRespectStakeActivation(t *testing.T) { + votePubkey := solana.PublicKey{1} + voteState := &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{{ + Epoch: 115, Credits: 2_000, PrevCredits: 1_000, + }}}, + } + mode := RewardCalculationMode{ + FullAlpenglow: true, + RewardEpochDelegatedStakes: map[solana.PublicKey]uint64{votePubkey: 1_000_000}, + } + for _, tc := range []struct { + name string + activation, deactivation uint64 + advance bool + }{ + {"fully cooled in rewarded epoch", 103, 114, false}, + {"not yet activating", 116, math.MaxUint64, false}, + {"activating in rewarded epoch", 115, math.MaxUint64, true}, + {"effective fractional reward", 103, math.MaxUint64, true}, + {"still cooling in rewarded epoch", 103, 115, true}, + } { + t.Run(tc.name, func(t *testing.T) { + delegation := &sealevel.Delegation{ + VoterPubkey: votePubkey, StakeLamports: 1, + ActivationEpoch: tc.activation, DeactivationEpoch: tc.deactivation, + CreditsObserved: 1_000, + } + pcs := calculateStakePointsAndCredits(solana.PublicKey{}, &sealevel.SysvarStakeHistory{}, + delegation, voteState, nil, 115, mode) + require.True(t, pcs.Points.Eq(wide.Uint128{})) + require.Equal(t, uint64(2_000), pcs.NewCreditsObserved) + require.Equal(t, tc.advance, shouldForceCreditsOnly(pcs, 1, tc.activation, 115, 1_000, mode)) + }) + } +} + +func TestInactiveStakePreservesExplicitCreditUpdates(t *testing.T) { + pcs := CalculatedStakePoints{NewCreditsObserved: 2_000, Inactive: true} + mode := RewardCalculationMode{FullAlpenglow: true} + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, mode)) + require.True(t, shouldForceCreditsOnly(pcs, 0, 103, 115, 1_000, mode), "disabled inflation") + require.True(t, shouldForceCreditsOnly(pcs, 1, 115, 115, 1_000, mode), "activation epoch") + pcs.ForceCreditsUpdateWithSkippedReward = true + require.True(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 3_000, mode), "vote credit rewind") + + // Tower does not inherit Alpenglow's automatic skipped-reward advance. + pcs.ForceCreditsUpdateWithSkippedReward = false + pcs.Inactive = false + require.False(t, shouldForceCreditsOnly(pcs, 1, 103, 115, 1_000, RewardCalculationMode{})) +} + func TestInflationRewardsUseHistoricalSlotTimeTransitions(t *testing.T) { schedule := &sealevel.SysvarEpochSchedule{ SlotsPerEpoch: 54_000, diff --git a/pkg/rewards/inactive_stakes_test.go b/pkg/rewards/inactive_stakes_test.go new file mode 100644 index 000000000..c1edc7c95 --- /dev/null +++ b/pkg/rewards/inactive_stakes_test.go @@ -0,0 +1,118 @@ +package rewards + +import ( + "crypto/sha256" + "encoding/json" + "fmt" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + "github.com/Overclock-Validator/mithril/pkg/bankhash" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/global" + "github.com/Overclock-Validator/mithril/pkg/lthash" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + bin "github.com/gagliardetto/binary" + "github.com/gagliardetto/solana-go" + "github.com/stretchr/testify/require" +) + +// This reduced incident fixture holds all unrelated slot effects constant. +// The production reward calculator, spool distributor and bank hasher must +// leave the 33 fully cooled stakes unchanged. See testdata/epoch116/README.md. +func TestEpoch116InactiveStakeBankHash(t *testing.T) { + var fixture struct { + ParentBankhash string `json:"parent_bankhash"` + Blockhash string `json:"blockhash"` + ExpectedBankhash string `json:"expected_bankhash"` + OriginalBankhash string `json:"original_bankhash"` + BaseLtHash []byte `json:"base_accounts_lt_hash"` + StakeHistory []byte `json:"stake_history"` + Stakes []struct { + Account *accounts.Account `json:"account"` + VoteEpochCredits sealevel.EpochCredits `json:"vote_epoch_credits"` + OriginalUpdatedDataSHA256 string `json:"original_updated_data_sha256"` + } `json:"stakes"` + } + raw, err := os.ReadFile("testdata/epoch116/inactive-stakes.json") + require.NoError(t, err) + require.NoError(t, json.Unmarshal(raw, &fixture)) + require.Len(t, fixture.Stakes, 33) + + dir := t.TempDir() + require.NoError(t, os.MkdirAll(filepath.Join(dir, "accounts"), 0o755)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "largest_file_id"), make([]byte, 8), 0o644)) + db, err := accountsdb.OpenDb(dir) + require.NoError(t, err) + db.InitCaches() + t.Cleanup(db.CloseDb) + global.ClearPendingStakePubkeys() + t.Cleanup(global.ClearPendingStakePubkeys) + + parents := accounts.NewMemAccounts() + var storedAccounts, erroneousUpdates []*accounts.Account + votes := make(map[solana.PublicKey]*sealevel.VoteStateVersions) + for _, row := range fixture.Stakes { + acct := row.Account + stake, err := sealevel.UnmarshalStakeState(acct.Data) + require.NoError(t, err) + require.Equal(t, uint64(114), stake.Stake.Stake.Delegation.DeactivationEpoch) + require.Equal(t, uint64(115), row.VoteEpochCredits.Epoch) + voteKey := stake.Stake.Stake.Delegation.VoterPubkey + votes[voteKey] = &sealevel.VoteStateVersions{ + Type: sealevel.VoteStateVersionV4, + V4: sealevel.VoteState4{EpochCredits: []sealevel.EpochCredits{row.VoteEpochCredits}}, + } + require.NoError(t, parents.SetAccountWithoutLock(acct.Key, acct)) + storedAccounts = append(storedAccounts, acct) + global.EnqueuePendingStakePubkey(6264000, acct.Key) + + // Independently reconstruct the logged erroneous write, and verify its + // byte hash before using it to establish the original bad bank hash. + bad := acct.Clone() + stake.Stake.Stake.CreditsObserved = row.VoteEpochCredits.Credits + require.NoError(t, sealevel.MarshalStakeStakeInto(stake, bad.Data)) + require.Equal(t, row.OriginalUpdatedDataSHA256, fmt.Sprintf("%x", sha256.Sum256(bad.Data))) + erroneousUpdates = append(erroneousUpdates, bad) + } + stored := make(chan struct{}) + require.NoError(t, db.StoreAccounts(storedAccounts, 6264000, func() { close(stored) })) + <-stored + count, err := global.FlushPendingStakePubkeysThrough(dir, 6264000) + require.NoError(t, err) + require.Equal(t, len(fixture.Stakes), count) + db.RootedDurable = true + + f := &features.Features{} + f.EnableFeature(features.AccountsLtHash, 0) + f.EnableFeature(features.RemoveAccountsDeltaHash, 0) + calculateHash := func(updates []*accounts.Account) string { + ctx := &sealevel.SlotCtx{ + Features: f, ParentAccts: parents, + AcctsLtHash: new(lthash.LtHash).InitWithHash(fixture.BaseLtHash), + } + return solana.HashFromBytes(bankhash.CalculateBankHash(ctx, nil, updates, + solana.MustHashFromBase58(fixture.ParentBankhash), 0, + solana.MustHashFromBase58(fixture.Blockhash))).String() + } + require.Equal(t, fixture.OriginalBankhash, calculateHash(erroneousUpdates)) + + var history sealevel.SysvarStakeHistory + require.NoError(t, history.UnmarshalWithDecoder(bin.NewBinDecoder(fixture.StakeHistory))) + newRateEpoch := uint64(0) + result, err := CalculateRewardsStreaming(db, 6264000, &history, &newRateEpoch, + votes, PointValue{Rewards: 12922370184029}, 115, [32]byte{}, + &sealevel.SlotCtx{Features: f}, f, RewardCalculationMode{FullAlpenglow: true}) + require.NoError(t, err) + updated, _, distributed, burned := DistributeStakingRewardsFromSpool( + db, result.SpoolDir, result.SpoolSlot, 0, 6264001, nil) + require.Zero(t, distributed) + require.Zero(t, burned) + require.Equal(t, fixture.ExpectedBankhash, calculateHash(updated)) + require.Zero(t, result.NumStakeRewards) + require.Equal(t, uint64(1), result.NumPartitions) + require.Empty(t, updated) +} diff --git a/pkg/rewards/rewards.go b/pkg/rewards/rewards.go index ce0e824a0..bfff0ebfc 100644 --- a/pkg/rewards/rewards.go +++ b/pkg/rewards/rewards.go @@ -43,6 +43,9 @@ type CalculatedStakePoints struct { Points wide.Uint128 NewCreditsObserved uint64 ForceCreditsUpdateWithSkippedReward bool + // Inactive is set by Alpenglow points calculation when the delegation + // has neither effective nor activating stake in the rewarded epoch. + Inactive bool } const legacyInflationSlotsPerYear = 78_892_314.984 @@ -697,11 +700,14 @@ func calculateStakePointsAndCredits( } newObserved = max(newObserved, latest.Credits) - effectiveStake := delegation.StakeActivatingAndDeactivating( + status := delegation.StakeActivatingAndDeactivating( rewardedEpoch, stakeHistory, newRateActivationEpoch, - ).Effective - if earnedCredits == 0 || effectiveStake == 0 { - return CalculatedStakePoints{NewCreditsObserved: newObserved} + ) + if earnedCredits == 0 || status.Effective == 0 { + return CalculatedStakePoints{ + NewCreditsObserved: newObserved, + Inactive: status.Effective == 0 && status.Activating == 0, + } } totalStake := mode.RewardEpochDelegatedStakes[delegation.VoterPubkey] if totalStake == 0 { @@ -711,7 +717,7 @@ func calculateStakePointsAndCredits( } } points := wide.Uint128FromUint64(earnedCredits). - Mul(wide.Uint128FromUint64(effectiveStake)). + Mul(wide.Uint128FromUint64(status.Effective)). Div(wide.Uint128FromUint64(totalStake)) return CalculatedStakePoints{Points: points, NewCreditsObserved: newObserved} } @@ -789,10 +795,14 @@ func shouldForceCreditsOnly( pointValueRewards, activationEpoch, rewardedEpoch, creditsObserved uint64, mode RewardCalculationMode, ) bool { + // Agave's skipped-reward credit advance applies only to effective or + // activating Alpenglow stakes. Fully cooled stakes must retain their + // account bytes, even though their vote account has earned new credits. + // The explicit forced-update cases still take precedence. return pcs.ForceCreditsUpdateWithSkippedReward || pointValueRewards == 0 || activationEpoch == rewardedEpoch || - (mode.FullAlpenglow && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) + (mode.FullAlpenglow && !pcs.Inactive && pcs.Points.Eq(wide.Uint128{}) && pcs.NewCreditsObserved != creditsObserved) } // CalculateRewardsStreaming performs a streaming calculation of stake rewards. diff --git a/pkg/rewards/testdata/epoch116/README.md b/pkg/rewards/testdata/epoch116/README.md new file mode 100644 index 000000000..41cd2cb6a --- /dev/null +++ b/pkg/rewards/testdata/epoch116/README.md @@ -0,0 +1,49 @@ +# Epoch 115 → 116 inactive-stake regression + +`inactive-stakes.json` is a reduced fixture from the Alpenglow failure at slot +6,264,001 on 2026-09-21, running Mithril `320ce8da`. It contains the 33 fully +cooled stake accounts that Mithril incorrectly rewrote, their vote accounts' +epoch-115 credits, and the saved StakeHistory sysvar. All 33 delegated stakes +had deactivation epoch 114 and zero effective/activating stake in epoch 115. + +The source was the supplied `/mnt/mithril-accounts` checkpoint at slot 6,264,000 +and `footer-bankhash-mismatch-slot-6264001.json` in the supplied logs. AccountsDB +was opened read-only. Account bytes are public chain data; no keys or validator +identity files are included. + +To isolate the defect, `base_accounts_lt_hash` includes all correct slot effects +and the unchanged 33 accounts. The other 792 modified accounts were reconstructed +from the parent checkpoint and deterministic slot updates; all 825 reconstructed +accounts matched the diagnostic's individual SHA-256 data hashes. Combining +their deltas with the saved parent LtHash reproduced both the diagnostic LtHash +checksum and the original bad bank hash. Undoing only the 33 credit-only writes +then reproduced the exact expected footer hash: + +| State | Bank hash | +| --- | --- | +| Original 33 erroneous writes | `CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG` | +| Preserve the 33 inactive accounts | `BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm` | + +`TestEpoch116InactiveStakeBankHash` checks each reconstructed erroneous account +against its recorded data hash, checks the bad bank hash, and then runs the +production streaming calculator, spool distributor and bank hasher. Before the +fix, that path emitted 33 zero-lamport writes and produced the bad hash. After +the fix it emits no writes for these stakes and produces the expected hash. +Other slot effects are held constant; this is not a full signed-shred replay or +a replay of later slots. + +Reference behavior: + +- [Agave ab655329, inflation_rewards/mod.rs:254–270](https://github.com/anza-xyz/agave/blob/ab6553293094e59dee7d3e7c928c7fa1023d0684/runtime/src/inflation_rewards/mod.rs#L254-L270) + restricts skipped-reward credit advancement to effective or activating stake, + preserving the explicit forced-update cases. +- [Firedancer 57d39904, fd_rewards.c:649–678](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L649-L678) + rejects an inactive credit-only update. Its + [points calculation at lines 908–917](https://github.com/firedancer-io/firedancer/blob/57d39904e3886731b96b9174ff8763ee7c36e3ad/src/flamenco/rewards/fd_rewards.c#L908-L917) + retains the inactive flag from the existing activation-status calculation. + +Run with: + +```sh +go test ./pkg/rewards -run TestEpoch116InactiveStakeBankHash -count=1 -v +``` diff --git a/pkg/rewards/testdata/epoch116/inactive-stakes.json b/pkg/rewards/testdata/epoch116/inactive-stakes.json new file mode 100644 index 000000000..e6dec5b48 --- /dev/null +++ b/pkg/rewards/testdata/epoch116/inactive-stakes.json @@ -0,0 +1,1693 @@ +{ + "parent_bankhash": "B2Yi5ecGCiZRpvjdrLeQj2SdSncKgi6zEmVJxpMfzfyB", + "blockhash": "GBQcgk9JL3YFaK1GvpKLGpCbpmJ3My4oYWztnAJ3wirY", + "expected_bankhash": "BBXWdsTHc3EdN8zXbcZZ8vwqtYjzggD3gp8CqkZuBvBm", + "original_bankhash": "CCY2QcFoGGAodDaB9RCAW2RUbbZm2xMo9dJw8tKHdDcG", + "base_accounts_lt_hash": "Mnn6NyRVWRvlXK1fJ68LNsMwgwvv5fH/brKehzoh9D+CaKo7hT/Wa1o3UkLLI7xs14mAj7EgyHCauhEVsMQGISCDkpWwejJy5tH/1Ov8yb+FM+1ZkSIO1vL8vsxDGipgAGfNXO+nGnh/ldg6Jbv66RgGdXJtR4b2IMUD970fj9z7229YpWP7PIwy70TEJ5VQuS16bXRwJlA8cRSSn7RMx9/yW2G4WhS+XAGi13N1uPsIHUkm6DePr4WPRsUi4UjMP3F8fpYUCAyVi8NMeFBPPGHZa/ZkR2C3NTk1Fs/jpI3UTw7MI61RnDFZs2vd4YYx6gr2g2jRKBX2/3NgSEK0112FCnB6T9V+LM6b+flCUlL5OddAd5MtN+4gDN2DlHAX20jb8VTWqwvb602v877ha/L+YK09+D+Gr9+iNxx8CSCAycb9moun0LGVSg6TgenXfhwKQbYlLtbIB7bD6YCrLBeP4A9rifOBIobmidjeYBjF7kicq8vMrH3j+flLSsn4fHXsssP9edqitP69ycRm7nLrYEFTR5+tl985wzrWi9DtOnImREgGo1oHBqGxXe6BksLdnCSei1i9SOX3oEaxKWf8bcf28ocL3pf5JKw3cBNALWTdjda/w/CMkg4N6MB6pUurImpm1M+Sg9O2Qyvv6v24RgG/TBdqMTmt3b46einSa+KZO72BQ+c4KkihabsRBpJNYpLQAiSH1Xm6FWoSuu63pRnGRyYokF3JC7K7M5wYpqE2EYJUUVM/SH7uS/JhOmWZXtGSpBcq0kQjHfIIjsJhrD2om9NJgVxEf4DhwK3ZTHUdTE5SY5aOwB/HVNlaTQd2Ab1ZsJncKAGIz1LcJpLB1uI3CtjtTwKUYrYTukHpWl2OQySqy0gmEzNPqmDUbhMUuEHcvJBUw30wLusR/BmlR9RErlxtigecyID+K9O9sTssRnAfuCY6Hm9SQP0rd6VRbIDqcfvSp4nOdYIZhx84IcN3SkN2Q9ipCc/OBCf0b2vIUu/7YS7VPv0gUQCKgg+kHJzZzUD5vHsOTw2623uxug2Y3tkiaVgrh861s2lJTjsSz351rMZQc0nzjis+GAoOuzo34zpGX3duLHYzxxdiRcX0jSvM4C9CVX47dQsYnltzVgOfXbeLYxA19SBjrhErooboDLe5OfeFeHGMIUO01sQ7dXXTzLZt03Opb8UjTbAtV3zCN4l08vObalYPFm4zEuTGYVnWIrJp4d2sEg/p7ejmWQNif/36zSLANp6ILgJqbVp4uATjuPPQD/nEcqlTspMy3f9yDK2Du3DjOcRNR6w17AtQvKsyH5dNo3w9JWG0a9HeBciz5CwdqYZ8cTi9jouD9Yj8roxAH9l63gNNJ+RgO62/17EBw9fS3pbNDfyJbZfvbnFq1KpIHe0LqK0K7hL7zTpcu/KgDKJLLZZe1YeDUm8jlVlhpbuDCBr3oj5N2p37581r5thU0F4f8//vqHm8G7i2UWt7sXtxJJ5JhUDV7RLYLfmFRc7ZXBJ7Z9qAxTo4WIWWfb3QsxSgZK2VfBFerxHzE8QlDCukgAHnjRxx/sVAHU5kqTI+jHpXsDogLmCTvaTuIW7x2WAmdEazWRi//5J9COsBt3wOwQwwBq7nRfj8w7Zo/M5xwbdpTv/JsvLQKmc51uaYVBSHZy8q54g7jo+j75jiA6TCh4KzMejWbH+c9Ew44ATW5mLiPFCJxmkZLI9Dl5XXxTtspSkHlTinziKS8+FCFUD8iCdnXKkym5gD1oO0imsGvBRs1qqnwbIaOv7esnnh4n8V3ML5La3du54etTnaGlAI35lGv2AEc2otPbHDvffI3PsMuEi9YkktqO7rJmFieUIU/uNiMevudlQqqrWM499kCOI7C6drxi8KfdcCd9b88nGSL6MlKu7F0cwttJWCb2uDTWEA2vYbxpUtDxZ8xYNeCdKDHJbNGk9viaLneQ0BzE3GQLnZPEYAuVkAbh+mWN0o/kmL8DQF4wpa1xBFUBBD6zhhM5zL6yAF5zT5k+bGkrIGS6+D0OteyVZnB+2y/ttjRknqhc+5CQVr90dEBsk+ivEsz7Ta/vZ4ibYJOgHq6z4OfagQWaWCiYBaQoTfKhLpekY3HrIMhQSsXuH8UjiRULhj0mqLnX7UarLLCTAvqEsvCgQxH1vzNbeomzI3HlJtWTrQz/vv8DOPXzJE/aYuxbyTAhq6uZH6TkH4qmv9lOh5dLbtQVZq1uf0upbJgiamwRyf+PGKqBe+rcYz3Mp/2aJk+jAYx4WGVdd66COzTvV///vH+M+/E7Yj4ueEnqsQVuLMtjAW+xTu9UiA/eb/mMVBpzeRaOUELDeYHwiKR8HiwLH73UO2AGuFsCAhqQPsYP/SSOaKvJinVKe7uclMgEgSAknr2UTxVjoPay18vynGYJOYDBrxFaxaDQbCglcjxfJW016E9SGRbxxlZZOqpr9qZtO20MgGMsgotBKniJ87ssxlJDJ+Kx7w2MWcIN5dv7kFj9KVIiBJ206fg6fQUHB+/4g0x7rBedhfmDtWizaNyqNcpWHfRMhoIPI+nOij3u8ASv/Z5HgqvFDeqn3MBix5Z/4+vRpS24pN83kvdbaGnyLSZ6jNuq2wfdN2PYYlnUczLvrwD7QjUkf+mf3PVhZxM+56+lnYG52tToqXIqAtrq4fkJq73lCOLwnd+8cCkdbQtx3qQF+NiR1eMiewNwz6cVTKoMdNqeYsGmXRNMY=", + "stake_history": "dAAAAAAAAABzAAAAAAAAAJ5AeDrqWjwAaanDLwG0AQBXHmbgZhcAAHIAAAAAAAAAJXm6zkOLNwDYUEVSW28GAPX9Y+xHMAAAcQAAAAAAAACbOD5ij1s7AH8vYWThAAAANhpRt13RAwBwAAAAAAAAAEKuN0SOQEAAGg55mi/jAABjcyFX7dUGAG8AAAAAAAAA243TwqZJPgCgvqaXc7gCAGxJJAa+wQAAbgAAAAAAAADlcqREpWI+AF6z7ht6CQAA2yghg6ciAABtAAAAAAAAAOSauglaSj4A7jLIrkcZAABAJ1t/KwEAAGwAAAAAAAAA1bfp3PxJPgBswjk12BAAAMxhZcmqEAAAawAAAAAAAACgpYOCjzs+ADdbCt+GKwAAyzEg90YdAABqAAAAAAAAAHdjrA3dQj4A26mxDlISAAB8hECCzhkAAGkAAAAAAAAAAEdfVIHpPQBW2hsNHo8AAC+XfvrrNQAAaAAAAAAAAABo9vZqfuU9AA58Eg7PBAAABIpq+vsAAABnAAAAAAAAAJFweNQULT4AqV72RLRIAAD2LXPQeJAAAGYAAAAAAAAA9qEC3Yo/PgCF7iNpyQ4AAAZZwTZvIQAAZQAAAAAAAAAuQbhfCTk+ANbBFd9uDAIAstBbUhgGAgBkAAAAAAAAADB84TH1QT4Agz3YD9wMAAAUQT/28xUAAGMAAAAAAAAAgc99eYkCPgDGnMraNEECAIGF5S/2AQIAYgAAAAAAAABNb2WWlro9AGWphlSdhQAAyYxLbNc9AABhAAAAAAAAAFmW5xtaxj0AF1Ks46MDAACV5g+0lg8AAGAAAAAAAAAAQgPK8m0kPgD7oWlICjIAALtPvWhHkAAAXwAAAAAAAADvAGYj8wo+AHtuvjg5PwIA4Gqgee8lAgBeAAAAAAAAAPRPsb1kNj0A6J0nTzorAwC0i2yT2VYCAF0AAAAAAAAAIiDIxHosPgBZmiIiYSYAACkVK22lHAEAXAAAAAAAAAAJvmQRggs+AAIGnLbqIAAAWMutxRwAAABbAAAAAAAAAL/QM8VAMD4Aw7etwKcAAADauqqckSUAAFoAAAAAAAAA65RRlJUVPgDMSCTC02cAAIfN5q1YTQAAWQAAAAAAAACP/BZcKsU9ACAmYvigdAAAiDfz8mMkAABYAAAAAAAAAEK6x8WZyT0ArEiebikKAACSNlCKxQ4AAFcAAAAAAAAACTBqaG7SPQCoR3XOSAsAAF71j7lJFAAAVgAAAAAAAAAjWmA61589AMtdU9w4VwAABJfvbs4kAABVAAAAAAAAAG/JhOUTaT0AUMvSkd44AAB1Iv0rRAIAAFQAAAAAAAAAfK7kidvtPQBNKEsZw0gAAC4+dp63zQAAUwAAAAAAAAB8tlqi8ts9APvZJnnSHwAA/WCY9BIOAABSAAAAAAAAAMAYELB0AD4A+dGs4HEaAADOy8rLIT8AAFEAAAAAAAAA0TQODTb8PQD+ZIrMvQ0AAKLsbq2qCQAAUAAAAAAAAACb6i9EQwM+AEicn0T6PAAAyAM4YTVEAABPAAAAAAAAAJeDquB9Qj4A29/hUTE6AADzZwr3lnkAAE4AAAAAAAAA3YhSTtglPgA1M80MFSYAAGDVIn6dCQAATQAAAAAAAADavbp48hw+ADUoxH3YCAAAmmSrqRsAAABMAAAAAAAAAM9KZ53SHT4AFPzaJG8EAABBh/RQegUAAEsAAAAAAAAA+HspIqN4PQAJ3QG6ZLgAAIXaa4phEwAASgAAAAAAAAC0Hzu/esY6AHt5mUByuQIAmiDhzXUHAABJAAAAAAAAAEMqtGP2cD0A84gQpRYFAABAZYJ8w68CAEgAAAAAAAAADk3OwuxvPQDAS3n7GBkAAE+VB8A4GAAARwAAAAAAAADRsZ0Vx1U9AHLLb4SiHgAAZfdTUKYEAABGAAAAAAAAAKNb7H87Qj0AZZCx2/oiAACT9t0tmQ8AAEUAAAAAAAAAQvmdRxdOPQDVa544vgcAAJYPouDEEwAARAAAAAAAAADx6iXLmDU8APdU+ynaJwEA6/70KIYPAABDAAAAAAAAANZyds2GODwAFLJEEZYGAACa0y6VrgkAAEIAAAAAAAAAe6+cLig2PABhXBlazAsAAC0teOaYCQAAQQAAAAAAAACxff6Ze1E8ACrBuEaqiwAAJkWtgCmnAABAAAAAAAAAAPLJlmbtbzwATFPM6gYKAABz5YOHoygAAD8AAAAAAAAAlXqLG0GDPADGPdE0LRAAAJvWmUOwIwAAPgAAAAAAAAAGKrZkAoU8AFZorVhRMQAALFQFTUEzAAA9AAAAAAAAAAoUqYH8sTsA4IDLEML4AAC9/5aE6iUAADwAAAAAAAAAuzL+y6ZrPAAB8TW7cSgAACoQ4FtL4gAAOwAAAAAAAAAn1bWd5RM8ANwhM7NEiQAA9kqDwa4xAAA6AAAAAAAAAFxafIPkIDwAK2eWKPQ2AADVJYEPHkQAADkAAAAAAAAAViHBYFraOwD4j7g0FpAAADQ7TdK4SQAAOAAAAAAAAAC0B1p+6RA8ALCVqnGzOQAAclnWd2pwAAA3AAAAAAAAABJqVZSu8TsAxZ4RSV55AADnTpfITVoAADYAAAAAAAAATOvQCGvlOwAInOaCLw4AAGlNI3AVAgAANQAAAAAAAADiOkL7a3I7AP3nWviafwIAPurnf8cMAgA0AAAAAAAAAISnBw5hVDsASNjGSPMhAABXGeKPGwQAADMAAAAAAAAA75Bzrs90OwAB3HKPxhQDADDloUJeNQMAMgAAAAAAAADdNGOwmVQ6AKrWSFPxbgEAhqz8EelOAAAxAAAAAAAAAKVlV2M5bTsAt7J6MWEZAABWgGutNTIBADAAAAAAAAAATdF/eGFmOwCMYa1ndx8CAFnG6qTIGAIALwAAAAAAAADaHqDgbWs6APgZHPi53gEAfGTrf/rjAAAuAAAAAAAAAHVDwpv5ozoAnUFqwhSoAAAwz2LP1eAAAC0AAAAAAAAAaE1WTy5QOgC7Lb6VWeQAANrlRLzBkAAALAAAAAAAAABMwuR/ciM6AEHpLe8wUwAAcI0k8acmAAArAAAAAAAAAGUV3bRCGDoAkDAWSOEkAACeekS34RkAACoAAAAAAAAArx9e+fYeOgB8ySqUlQgAAAs3i/h6DwAAKQAAAAAAAACNHcp9cl06AJaDFRAWbgAATpEP8MasAAAoAAAAAAAAAJSotzIIpzoAFwSV826JAAAy92cXMtMAACcAAAAAAAAAPFUSdQ65OgDoJJ6NhCcAADCwfvHBOQAAJgAAAAAAAAAA6p5t5546AMm2R+E9JAAAo0xN31AKAAAlAAAAAAAAAO6h+UfxQToAnx24iv5qAAAddIZ3PQ4AACQAAAAAAAAAKa2tRNmFOgACcRzbgS0AANYPtIOhcQAAIwAAAAAAAAAGOyVz3zI6ALqiF0x0+QAAi9QI3qumAAAiAAAAAAAAAOcmAEoM+jgA4sBkWs5BAQCzLciCLAkAACEAAAAAAAAAdn8sBEvPNABKZYiSUOEFAKq488AelgAAIAAAAAAAAAD0lsvPNDw1AL9v6SJ28wAAlbOWU4hgAQAfAAAAAAAAAGk2uzPpvTUADkS44IFUBAD3aJUNxGQFAB4AAAAAAAAAG4YELojOMwBT2ECjQoQEAAUoip3hlAIAHQAAAAAAAADYscqSXXQzAFufDzqhIgEAGMvVnnbIAAAcAAAAAAAAALtrLOOo4TIAHUaer7SSAAAAAAAAAAAAABsAAAAAAAAAPlfpgiWuLgD+x4A4R4oEAAAAAAAAAAAAGgAAAAAAAABhNJyrrrQrAPe6b3aatQcA4T05ZoT1AAAZAAAAAAAAAOgqRqBFESoAOA1GOSyqCgD7mGj70iUCABgAAAAAAAAAQUvKQ90zKQBbee1BlPgEAIO7vG7m1wIAFwAAAAAAAAB0/UZWrDMpAFfZoR9TcQgAbNhDyqKIBgAWAAAAAAAAAGTa4bqZwSoArSc7BflKAgATQNWp51cJABUAAAAAAAAAuJa1PExVKgAIutFU5XQAADZwSSvTCAAAFAAAAAAAAABdiCSRT04qAEvh4+XMBwAA6Ma1tAYBAAATAAAAAAAAALBNfKM56ykAxKmk5NtrAAAgpzgl+ggAABIAAAAAAAAAcrdEoMDgKQCuPiLWZmYAANwwSk8mXAAAEQAAAAAAAAAUXTuSI9EnAILkep/EswIA4YaSbF2kAAAQAAAAAAAAAEhOd7PGYiUAJd7/3yVuAgAAAAAAAAAAAA8AAAAAAAAAVP+zxFlMIgDlgfGO3iIFAAAAAAAAAAAADgAAAAAAAABT2ixTNXcfALK/SZfhyAcAAAAAAAAAAAANAAAAAAAAAESpbbv03RwARWz5UURhCgAAAAAAAAAAAAwAAAAAAAAAQFrTd6Z7GgDMXyiGppQMAAAAAAAAAAAACwAAAAAAAABcnB0IxEsYAH3pX8hj/w0AAAAAAAAAAAAKAAAAAAAAAJCaKkMgShYA38jJYAvPDwAAAAAAAAAAAAkAAAAAAAAAF58Ll+hyFACFLkwOVlARAAAAAAAAAAAACAAAAAAAAAAxtAZmmsISAEKE8AX0/hIAAAAAAAAAAAAHAAAAAAAAAHXDQI4BNhEAACv2ahmEFAAAAAAAAAAAAAYAAAAAAAAACt0HxSrKDwDjq1Ie9r8VAAAAAAAAAAAABQAAAAAAAACxrhYtYXwOAN0of1y9rhYAAAAAAAAAAAAEAAAAAAAAAC2YPXwpSg0Ap97CpRE4FwAAAAAAAAAAAAMAAAAAAAAANf2HdjwxDAD7OmBvB/AXAAAAAAAAAAAAAgAAAAAAAADgL1gZgy8LACVZ0BHMARgAAAAAAAAAAAABAAAAAAAAADurkkgTQwoASjP+BeaxFQAAAAAAAAAAAAAAAAAAAAAAgGxxLSlqCQDgMpi3KeAVAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA", + "stakes": [ + { + "account": { + "Slot": 6210151, + "Key": "3mZwezEBuKcCs42NfAEZfdCbCMyPaV6wnus263mgrhjP", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAI9JSTu3nHzfydAzLjeA87Aht6Ost2Mol90Y42gdbSqB9HisvA0AAABnAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACZzNzwgAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7e91c9547ff107e827f22f226df525ee83be7e9eb22bc8afd812c92c235c69ba", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9430889728015, + "PrevCredits": 9349889838233 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "453ETE3U7Usc4ss87sszs3EvTR5ZYQQWcZHgs9XiAZdY", + "Lamports": 3306766663971, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADSIHvAB+hqCeyUMqrtA7RF7gf0pF5+WOdx24l28NcLZoyuE6gEDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAALJPnDAREAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "88983f1b8c3153b155fc54b8ffcdbe468b3356ce007cdab77d95cab1d3c50dc3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18772882129308, + "PrevCredits": 18699280524299 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "462wmoxLUzHfxcURxwkMj5f7cmeVXeSeaFU5uXd9jrQE", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAB7glTPNCA5Xp0adNqBsSaCIXSwoB3DhCXGyxm9mwIQut+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAABJX296lgQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c0c400478bf441c374e56141b0122846da672b9b1a5d01157374dadfa83649", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5082302209818, + "PrevCredits": 5044345724745 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4aNZYt3giNatv4J58sYHwRNs8L6ZQpLqG9GZiyqrXUqS", + "Lamports": 2329209663118, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAK8Fa5ghODdNV981iyfO/aADxuCTmU56Dtg15Wf9Xk+LDhmUTx4CAAABAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAeHbzX0wkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "78674eaaabfbe62cf2e558bd33e01dd53a33a7a4024e52eba673bb1acf73ff83", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10909280173169, + "PrevCredits": 10805462179102 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4u4XDhXRtXA9QqP8uC9vpoyw4KLaeWHm7g2gP4aLDGK1", + "Lamports": 59002277492, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABtRAg7KdM61d5wq/bIMit41UurMHM2/gwzsJrDkSJWr9HisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC8dSEI5AgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f49eb225b3f858e7609ffecf5e9d2734cb28fe754ca697800e0a44a144ab0f3d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9855957906862, + "PrevCredits": 9775481976252 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "4zrGtmZj2s7akr4FyiJ7E3fQY4n59cZoVEmL192GttRq", + "Lamports": 3956852941834, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAApBQ/86W982eATp1soYW/TvCCjcBXX1dK637et+U7Qaio6tRpkDAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABz+BFZ1gkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6a0c148dbf978bfec9e6e5e3227ad57e6089c056b2d23730b85be7bdcc2a3384", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10900522634265, + "PrevCredits": 10816222001267 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "5ZEUi72iZM7ozwisZV2jezjgR3BLBQToz8ZxpdnqYaue", + "Lamports": 10521999279304, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEvFm+VinHP4nTd2zkbv4cVplRt3TlVeTEidtXhPv7vISK/k15EJAABiAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABDEW5ATQcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a0c42ad17b70bd37283facbe1655a9401660d0ba96e7c57fad1ba452c46c77c8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8111781721836, + "PrevCredits": 8028374831427 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "6dNDcZRYeX7zoGirAbYxjJVCoHg14KLcy1ndb5a6hLtp", + "Lamports": 92402274844, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALJZyCh5aoPz7tWdy6VqoilN4lxeoCa10yfJpJBoYSa8nPx3gxUAAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACEXlHvNAgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "4914fbc352aae3350f8dd39a94733fc3bd927e1ccd2ed471ac81318d759594f9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9114167602897, + "PrevCredits": 9023446408836 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7Xwjq1Pu1ibBufp4mdu51yrRzUoadak98sdj6BQhGmBH", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMb+8swy4nz1EbP63bJ/XUX49DQYo/CiTC3dFaYhgfYs9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACE/LmcrgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "01df0af720551818974311706b7a5ebb9c5b333eb90e877e0f227f08fd4cf9e3", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8523194289275, + "PrevCredits": 8446535138436 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7kGbq8H94roSeiCvasCE8kWq2Kx1dCTZQTUnkiPqibTg", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAaFSKZ/FpeRshn8N9+v+AkQycTuIPwqrs5Yf2mgYnGv9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAF9xJssgcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "f04a2844299ed1e7c980b160a89a508f61a135ee5d333ddbb948aabb00b80a6d", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8544891256299, + "PrevCredits": 8462898755333 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "7uyNRdfhmk6SwJPfM172LF4ajEUKUkP9RC2okeB3RZsC", + "Lamports": 48931548636, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAABLFnpY8I58W61QVYzmx6wfM3seg+Q0PxwXVhMb9YvpaXFhpZAsAAAAzAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABZwrzIpQYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d2a561a19b6f7cd09f21985876978ae5ab1e11c7cc9262c5a17caf9af6e0c92b", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7371534095782, + "PrevCredits": 7309107184217 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8QYDcU8pvmMKvovzUiqWq4H7ZFiWpbet4HwUnDy5XTYK", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAFgYhGqxkhQFABeF03ECpK9RUUF1hdavhflLAu/MvP/rAMqaOwAAAABtAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAs5ZJ1sQUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ff84dc2a8426d4deb78786afc87899cbb4c052b6c20346fac76f7fcd865026c5", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6318982736717, + "PrevCredits": 6259739911468 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "8n3LHx61MeuZTF28hkHvsiEtKSEXLUQDYCMHS8sCQ264", + "Lamports": 3058382969924, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAM9GQ73y9miIEoMD1/GtvVm6Z6RC6u1sNUKM6BM6jWJ0xMaxFcgCAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABQGTGKHhEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "20b17fb68d98d290dfa4742237ac719219e23147a8f7cdfb27ac1be790f9e8d8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18900384617754, + "PrevCredits": 18822865164624 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "98joChCUmuTBYZx8xHDJzJzbF37HPuP4EHqm3BwD8KSo", + "Lamports": 132002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAite2/mh1w1kW4RMXwAjDtR4kt3TcBCr/oy5ngX7hOQACjQux4AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACf+Oklb0QAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c11ae1d9490b3f0876c515a564337a0170c21fae8c3cb3f43c66e5dee2cd34f", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 76179624781196, + "PrevCredits": 75244168149151 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "9PAPBcFDq536ZzMWhNywZAYPZt5mF92VadrJc5ctoQqY", + "Lamports": 1894659468576, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAALo09fj+TmHMhldcHlC9iwfK4NZlNl+rkERD8XdoYYH1oFdeIrkBAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAC9udrpXwwAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ba19d0de0f1d7ccded2e93dd4b9d57b8077397c396005d1918b4d6dd719196b7", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 13714985011803, + "PrevCredits": 13606084852157 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ALrrFs8DAkNHjyce6LhwkotcBQp8U9dGKb4prUHZp5GF", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACZSY+h/WqtJAAWy9FwQ5zvCPyGbQUYwp5yepcxIn0qbAMqaOwAAAABwAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACXOQj8AgEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a96af37ed2c6d36cfc77971b4432cb7c62f4e045b3883a5b7e2e6434fa21dec9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 1147511243813, + "PrevCredits": 1112329959831 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "ARPzersuLPyg9ZJaxYuKc2eDKKg8t4qWXpTy9TzKjLJU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAF6PbC1s5mob1cKneVa1heDI7d/UjwFUwoZKG6DN1okV9QSscAoAAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAZRdBpGAcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "7c3b359b3e2c72b291d1703977f464395b1772f9197466913c9f87661d3ebac0", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7864995243760, + "PrevCredits": 7801435866393 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "AXC6s8QGstr1nnTJ5ranGEeSZ7rcHKVeQav4r4J8icJL", + "Lamports": 2951443346384, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAD2DXLDpqE+RtusMr3795aLUMJhXsVQnXUHgtyhqt1j0UJ6YL68CAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACQUqzjgQ8AAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0e44daa0385ca084667f7e44f0fb96571b6b963f6bd046bf66c2c7fb31e232ce", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 17293215270489, + "PrevCredits": 17050544919184 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "BWiQJLo1TGhbNp3ionXVFPnSgfGimGU9FVHm1hdwLKbJ", + "Lamports": 30092278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAADUkqa5BeE51KvSv12H//GyC8rI6ScrXABI9UawzUXiot+aAAQcAAAA0AAAAAAAAAHIAAAAAAAAAAAAAAAAAAADlkymHrAQAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8cf243491cd54f9d1efc732287caf92bd392dfc1ad410c74c94e2d8f4e4deb18", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5181125525357, + "PrevCredits": 5139048535013 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "CbFTDpV9HjcnVQReCUwuZmLMp6PjMoWZBA7VTvLzszQ", + "Lamports": 5114239502629, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANp1F9fR0omLpsfk9YL0X7w0EuA6D8E39bIn27KGqRgppfNKwKYEAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADfLChErwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "6bf6057cfa3c4d2a7d786bb5f325032604cca8eb30391468f2caf1b71406ce57", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9639820890790, + "PrevCredits": 9548855782623 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "D5Zx5bsdVDg2kBZkPvaF59ULdKd8LACtScaWE7NEAbtw", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAANbDQeBxoTnMt62Ygze0/L0wq3Tv5ynlZXdjhuu6lY/4AMqaOwAAAABuAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABO6xz/EAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2480dc54ffb62581eec0ca3bcb7201b2ba4260a97bdc186e2f2345241abade3c", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6731352374437, + "PrevCredits": 6670069328718 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "DNCCPW9Bsv8SdcDMFDA7b2HT5jKxgUnBrA2gkKXwe5Bg", + "Lamports": 44252278839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAEZlguB/hNNN/i6p/xPZEHXCTr6qIxawtvTnx9A4WE77t1qBTQoAAAAWAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAJpOYcEAUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a02301fdf180b2f542124caf7da178cae4397acc833030b9dca28116d6e2b747", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5630002073869, + "PrevCredits": 5566762492937 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "E2TDLwJNmzvg6K7MhrkQ3HV4n4mp2zL49FqGwV4tFUfC", + "Lamports": 1002282880, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAIZbuXtVwiaoNZBvLxeR/DEy46f1oQ/1DEqV1R+wxiWyAMqaOwAAAABvAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA+JHwoWwgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "2ecdc0e446f6e0c112c947c79b6976cbdeea776ecb4fe62ece89b1f725e3ae48", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9263388828449, + "PrevCredits": 9187614270526 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "EA3MVKFbieGtDZDG4bmCiwnE2HWndMuWcjMjRNSrbMBC", + "Lamports": 3787600669028, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACGc29KCGn33qCWHYLj1Li6Q1LaT9ffcV8Q3PJ6AssCM5NN03nEDAAAGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAChfXpwQQkAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "c0bf9bd502e0b675a186b4b855812ae1915d04110a951499d911e6174ed2b2b2", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 10252916692527, + "PrevCredits": 10176664599969 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Eijmmf2xFvjStkcu6WQRj93x5iYP1uv5zuZP8PARdqd", + "Lamports": 3449661803946, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAC3bR7fFvVz26LQA9LQq1JGFKWXl7hg6BaC5sp0SpTflKvi6LyMDAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABF19K7FxEAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "a4810c657cb2161f3c47f0452d78dc74d9d4597763d82c3dfd48b3d1dc43dfe9", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 18871239559140, + "PrevCredits": 18793633077061 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "FxhNhXDkoqMTcXqyHxYXW8BJAdpWH5qdyz3qgpiySrKn", + "Lamports": 71250450995, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIums8S4lhAAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "ab580bd8bb513651006208d529d58dbe8e40d0306141e4b9bbe32c0559040af8", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "GAyfc6a5E5uzXQgQ4KXHuJyunyGNtLGZgUUdWKEL4Df9", + "Lamports": 59002277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA0QDCxOaKoUheu2WKN7BFVixKbK4ogx2BYksnFTbTzL9XisvA0AAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACYANgjswcAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "d635c01a4fc8bc710b4ca6d9b0b0ba960b9ab3ad8ac0359ee8adbdd62ad30f3a", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 8554183692312, + "PrevCredits": 8465981898904 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "H2dM1xYSeuZaujtByMa4jvPwUVYjfxzUL3K9iVLaCW9G", + "Lamports": 64469922880, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAiQo+w+o08xym3k1JBF+HYHIJZGTDaZaufK/simzs1bwB6SAg8AAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAACe3jqUQyIAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b39033f338419cea17108c3989a6331d8272749095fcf3ad856031cbab264110", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 38085420659301, + "PrevCredits": 37673645039262 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "Hh8u21hJnAvE9PMNBmNTsDUTseY6bYfwvbZQgj8R2vcU", + "Lamports": 44842277493, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJOHGqQMr9AgJ5hIj044ilcd/38t2ijdTubnJyZ6iT+d9QSscAoAAABGAAAAAAAAAHIAAAAAAAAAAAAAAAAAAABymDuk3wUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "0569f4f3dbc16ab5c284f6efaefd1e2b6b5b8db18ffa1b1fbd0da98468c54d78", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 6525335322632, + "PrevCredits": 6458091214962 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "HicPaN5ogXhE5jQ99WotjGUQLPJ11bdCqXjJhMKUq5N", + "Lamports": 1690908095830, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAJ1KF6URsOxn3mdoAgUdjof1ae904ReOis3chjrZx0h+1h/XsYkBAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAADt4ogGQgUAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "47ad9ddb61ff1edee564852ac0e48ff45b47a377b0ca1a5f1e0aef74a27627b4", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 5826444097536, + "PrevCredits": 5781135614701 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "c6Ev8GfADHdrGtvH1VqwH7JmN28PisZgGfUy64RE7np", + "Lamports": 3782196946839, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAMDb0jorArBVkC7lcchWRZHVxqD+PrkqtMdnxFvaxkbfF5JenHADAAACAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAA1TQUrswgAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "b361cd2d770b1b3432d58f8098c654f300754fab58787d6033dae1e724618467", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9647673972566, + "PrevCredits": 9565613935925 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "eHboMWq5bmx1DxdKw79zpse75XLk2GEgeRnJG9nysJn", + "Lamports": 5125666350406, + "Data": "AgAAAIDVIgAAAAAA/LTRFge4YpO0iugL7CcyV2OPIzVpqNRFyNcRlWoBALL8tNEWB7hik7SK6AvsJzJXY48jNWmo1EXI1xGVagEAsgAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAPznsTwPeDl/u5+jig9MvOBeS/RKUvjM0z9ITapNTvPoxs9iaakEAAAAAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAEpkAsaggAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "8e052b16e36ecb6e80919ada5456816e25e3b804fe9119319cff9861e7242e25", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 9334067720448, + "PrevCredits": 9252101989892 + } + }, + { + "account": { + "Slot": 6210151, + "Key": "v5pSSEZAuvG9ewGhdMNVcTVHeJPzFnGse3mX4FBEyLT", + "Lamports": 1082110856099, + "Data": "AgAAAIDVIgAAAAAAzi6d3+ZZh+CVMQop8DVJNLt57mKRukRhiZHz0dnIWEDOLp3f5lmH4JUxCinwNUk0u3nuYpG6RGGJkfPR2chYQAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAHeMPT3PZmkZZcARjam+dHkyZP326BpHWqy/ktZchIumI3ay8vsAAAAgAAAAAAAAAHIAAAAAAAAAAAAAAAAAAAAXAHi5oAYAAAAAAAA=", + "Owner": [ + 6, + 161, + 216, + 23, + 145, + 55, + 84, + 42, + 152, + 52, + 55, + 189, + 254, + 42, + 122, + 178, + 85, + 127, + 83, + 92, + 138, + 120, + 114, + 43, + 104, + 164, + 157, + 192, + 0, + 0, + 0, + 0 + ], + "Executable": false, + "RentEpoch": 18446744073709551615, + "IsDummy": false + }, + "original_updated_data_sha256": "053f427dd888bbe45c6e0f772faf6eac1f9ede5822cc74477c757eaac75cb8b1", + "vote_epoch_credits": { + "Epoch": 115, + "Credits": 7382306449058, + "PrevCredits": 7287376183319 + } + } + ] +} \ No newline at end of file From 5bfcdd6900c93a36c67671246a5f8d1e0da6f185 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:21:18 -0500 Subject: [PATCH 087/111] replay: retire completed rewards bookkeeping after durable promotion --- docs/rewards-unwind-retirement.md | 58 +++++++++++++ pkg/replay/block.go | 7 ++ pkg/replay/rewards_retirement.go | 49 +++++++++++ pkg/replay/rewards_retirement_test.go | 119 ++++++++++++++++++++++++++ 4 files changed, 233 insertions(+) create mode 100644 docs/rewards-unwind-retirement.md create mode 100644 pkg/replay/rewards_retirement.go create mode 100644 pkg/replay/rewards_retirement_test.go diff --git a/docs/rewards-unwind-retirement.md b/docs/rewards-unwind-retirement.md new file mode 100644 index 000000000..edf207be8 --- /dev/null +++ b/docs/rewards-unwind-retirement.md @@ -0,0 +1,58 @@ +# Retiring durable rewards bookkeeping + +A completed partitioned-rewards distribution used to leave its in-memory +descriptor alive for the rest of the replay attempt. The fork-switch guard +rejects any such descriptor because account-overlay unwind cannot restore the +consumed spool or its distribution counters. This is necessary while completion +is speculative, but unnecessarily forces checkpoint replay after completion +has become durable. + +Replay now observes the inactive EpochRewards sysvar in a successfully executed +bank's immutable snapshot, with zero partitions remaining. It remembers that +bank's slot and the exact distribution descriptor. Only applying a successful +durable fold through that slot retires the descriptor. Later bank observations +do not move the completion slot forward. A new descriptor/epoch invalidates the +old evidence; missing sysvars or unknown completion retain the old fallback. + +## Safety and recovery contract + +- Completion in memory, certificate finality, and submitting a fold do not + authorize retirement. Failed folds leave the durable watermark unchanged. +- Active distribution and completed-but-not-durable distribution retain the + existing rewards guard. No spool reconstruction or rewards rollback is added. +- After retirement, in-memory switches still require the existing epoch, + vote/stake-cache, parent-context, sysvar and transaction-status checks. + Switches at/below the durable watermark still require durable recovery. +- Completion evidence is replay-thread-owned and process-local. It does not + change checkpoint formats, signing reservations, persisted vote history, + clean-shutdown rules or restart authorization. Restart retains the existing + persisted EpochRewards validation. No extra file or disk sync is introduced. + +## Incident motivating the change + +On Zen 5, distribution completed at slot 3,942,001. At a later parent-linked +switch, the durable checkpoint was already 3,944,067; child 3,944,076 selected +parent 3,944,073, abandoning the suffix from 3,944,074. The remaining descriptor +forced the rewards-window fallback even though completion was below the root. +Checkpoint recovery re-fetched previously received blocks, with logged waits +of 2.739 seconds and 0.967 seconds. A buffered 665-transaction block waited +3,613.510 ms for replay admission and then executed in 7.520 ms. + +These are incident observations, not a before/after benchmark or a measurement +of checkpoint encoding/fsync time. Thirteen observed FAST aggregates omitted +our vote during the recovery interval; that does not prove absence from every +FAST aggregate or a single cause for all thirteen omissions. No live latency +improvement is established until a comparable switch exercises the new path. + +## Validation + +`rewards_retirement_test.go` covers active/missing bank state, unknown completion, +the exact durable boundary, later-bank observations, generation changes, failed +and successful folds, and an exact-parent unwind after retirement (including +account values, resume state and immutable rewards sysvars). Existing unwind +tests still require fallback for zero-remaining bookkeeping without retirement, +cross-epoch switches, dirty vote/stake caches and invalid parent snapshots. + +Full replay/rewards race suites passed locally and in the combined native +build; native node recovery/checkpoint race tests, vet and validator build also +passed. These are software tests, not mainnet power-loss qualification. diff --git a/pkg/replay/block.go b/pkg/replay/block.go index 495ed53a8..c31153b1a 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1747,6 +1747,7 @@ func ReplayBlocks( var unwoundParentBankSysvars *sealevel.BankSysvars var partitionedEpochRewardsEnabled bool var partitionedRewardsInfo *rewards.PartitionedRewardDistributionInfo + var rewardsCompletion partitionedRewardsCompletion var featuresActivatedInFirstSlot []*accounts.Account var parentFeaturesActivatedInFirstSlot []*accounts.Account @@ -2063,6 +2064,10 @@ func ReplayBlocks( mithrilState.LastRootedSlot = promotedThrough mithrilState.LastRootedBankhash = rootedCtx.Bankhash mithrilState.LastRootedContext = rootedCtx + if rewardsCompletion.retire(&partitionedRewardsInfo, promotedThrough) { + rewardsHoldBelowSlot = 0 + mlog.Log.Infof("epoch rewards bookkeeping retired through durable slot %d; later fork switches may unwind in memory", promotedThrough) + } if transactionStatuses.Root(promotedThrough) { mlog.Log.Infof("transaction status cache reconstructed complete %d-root coverage through durable slot %d", maxTransactionStatusRoots, promotedThrough) @@ -2914,6 +2919,7 @@ func ReplayBlocks( boundaryParentCtx = epochBoundaryParentCtx(acctsDb, block, currentEpoch, replayCtx.CurrentFeatures) } partitionedRewardsInfo = handleEpochTransition(acctsDb, partitionedEpochRewardsEnabled, boundaryParentCtx, replayCtx, epochSchedule, replayCtx.CurrentFeatures, block, currentEpoch, rpcc, dbgOpts) + rewardsCompletion = partitionedRewardsCompletion{} currentEpoch = block.Epoch justCrossedEpochBoundary = true // While partitioned rewards are distributing, hold durable promotion @@ -3026,6 +3032,7 @@ func ReplayBlocks( } // The successful child now owns its derived snapshot. Any later bank uses // lastSlotCtx; the one-shot retained unwind bridge is no longer needed. + rewardsCompletion.observeBank(partitionedRewardsInfo, lastSlotCtx.BankSysvars()) unwoundParentBankSysvars = nil postProcessBlockStart := processBlockEnd statusViewStart := time.Now() diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go new file mode 100644 index 000000000..a03a00cbf --- /dev/null +++ b/pkg/replay/rewards_retirement.go @@ -0,0 +1,49 @@ +package replay + +import ( + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" +) + +// partitionedRewardsCompletion is replay-thread-owned, process-local evidence +// that a successfully executed bank contains all effects of this distribution. +// It is not a checkpoint or signing authority. Until that bank is durable, +// tryInLoopUnwind must still reject even a zero-remaining distribution: its +// spool has been consumed and cannot be rolled back with the account overlay. +type partitionedRewardsCompletion struct { + info *rewards.PartitionedRewardDistributionInfo + slot uint64 +} + +// observeBank must run only after successful block execution/publication, using +// that bank's immutable sysvars (never the speculative global sysvar cache). +// If the first completed bank lacks evidence, recording a later descendant is +// conservative: retirement then waits for that later bank to become durable. +func (c *partitionedRewardsCompletion) observeBank(info *rewards.PartitionedRewardDistributionInfo, bank *sealevel.BankSysvars) { + if c.info != info { + *c = partitionedRewardsCompletion{info: info} + } + if info == nil || c.slot != 0 || info.NumRewardPartitionsRemaining != 0 || bank == nil || bank.Slot() == 0 { + return + } + epochRewards, ok := bank.EpochRewards() + if ok && !epochRewards.Active { + c.slot = bank.Slot() + } +} + +// retire is called only when replay applies a successfully committed fold and +// advances LastRootedSlot. Finality, an enqueued/in-flight fold, and a failed +// commit do not acknowledge durability. At this boundary every rewards effect +// is in AccountsDB; in-memory switches above it cannot undo distribution. +// Switches at/below it still take durable recovery, whose persisted +// EpochRewards validation remains unchanged. Restart loses this optional +// evidence and reconstructs state through the existing recovery path. +func (c *partitionedRewardsCompletion) retire(info **rewards.PartitionedRewardDistributionInfo, durableSlot uint64) bool { + if *info == nil || *info != c.info || c.slot == 0 || durableSlot < c.slot || (*info).NumRewardPartitionsRemaining != 0 { + return false + } + *info = nil + *c = partitionedRewardsCompletion{} + return true +} diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go new file mode 100644 index 000000000..460584eee --- /dev/null +++ b/pkg/replay/rewards_retirement_test.go @@ -0,0 +1,119 @@ +package replay + +import ( + "bytes" + "encoding/base64" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/rewards" + "github.com/Overclock-Validator/mithril/pkg/sealevel" + "github.com/Overclock-Validator/mithril/pkg/state" + bin "github.com/gagliardetto/binary" + "github.com/mr-tron/base58" + "github.com/stretchr/testify/require" +) + +func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { + active := &sealevel.SysvarEpochRewards{Active: true} + var raw bytes.Buffer + require.NoError(t, active.MarshalWithEncoder(bin.NewBinEncoder(&raw))) + activeBank, err := sealevel.NewBankSysvars(5, &accounts.Account{Key: sealevel.SysvarEpochRewardsAddr, Data: raw.Bytes()}) + require.NoError(t, err) + missingBank, err := sealevel.NewBankSysvars(5) + require.NoError(t, err) + for _, tc := range []struct { + name string + remaining uint64 + bank *sealevel.BankSysvars + }{ + {"active distribution", 1, testUnwindBankSysvars(t, 5, 50)}, + {"active bank", 0, activeBank}, + {"missing bank", 0, nil}, + {"missing rewards", 0, missingBank}, + {"unknown slot", 0, testUnwindBankSysvars(t, 0, 50)}, + } { + t.Run(tc.name, func(t *testing.T) { + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: tc.remaining} + var completed partitionedRewardsCompletion + completed.observeBank(info, tc.bank) + require.False(t, completed.retire(&info, 100)) + require.NotNil(t, info) + }) + } + + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + require.False(t, completed.retire(&info, 100), "zero remaining without observed completion is insufficient") + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + require.False(t, completed.retire(&info, 4), "uncommitted completion must retain the guard") + require.True(t, completed.retire(&info, 5), "later observations must not postpone recorded completion") + require.Nil(t, info) + require.False(t, completed.retire(&info, 100), "retirement is one-shot") +} + +func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { + old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} + next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} + var completed partitionedRewardsCompletion + completed.observeBank(old, testUnwindBankSysvars(t, 5, 50)) + require.False(t, completed.retire(&next, 100), "old completion cannot retire new bookkeeping") + completed.observeBank(next, testUnwindBankSysvars(t, 11, 60)) + require.False(t, completed.retire(&next, 10)) + require.True(t, completed.retire(&next, 11)) +} + +func TestRewardsRetirementWaitsForSuccessfulFold(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 5} + tail := asyncTestTail(fc, 5, 6) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + job, err := tail.buildFoldJob(6, true) + require.NoError(t, err) + require.NotNil(t, job) + root := uint64(4) + require.False(t, completed.retire(&info, root), "capturing a job does not make its bank durable") + require.Error(t, runFoldJob(fc, job)) + require.False(t, completed.retire(&info, root), "a failed fold leaves the old durable root") + fc.failOn = 0 + require.NoError(t, runFoldJob(fc, job)) + ctx := tail.applyFoldJob(job) + require.NotNil(t, ctx) + root = job.through + require.True(t, completed.retire(&info, root)) +} + +func TestRewardsRetirementAllowsExactParentUnwind(t *testing.T) { + resetVoteStakeDirty() + t.Cleanup(resetVoteStakeDirty) + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 5, 50)) + tail := newUnrootedTail(&fakeDurable{}, &fakeCommitter{durable: accounts.NewMemAccounts()}, 512, 1, "") + parent := &state.ResumeContext{Slot: 7, Bankhash: base58.Encode(make([]byte, 32)), AcctsLtHash: base64.StdEncoding.EncodeToString(make([]byte, 2048)), Capitalization: 700} + bank := testUnwindBankSysvars(t, 7, 50) + tail.Add(7, []*accounts.Account{testAccount(1, 71)}, testHashBytes(7)) + tail.SetContext(7, parent, bank) + tail.Add(8, []*accounts.Account{testAccount(1, 81)}, testHashBytes(8)) + tail.SetContext(8, &state.ResumeContext{Slot: 8}, testUnwindBankSysvars(t, 8, 999)) + sw := &CertifiedSwitch{Slot: 8} + ms := &state.MithrilState{LastRootedSlot: 4} + sched := &sealevel.SysvarEpochSchedule{SlotsPerEpoch: 432000} + rs, _, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Nil(t, rs) + require.Equal(t, unwindFallbackRewardsWindow, reason) + ms.LastRootedSlot = 5 + markVoteStakeDirty(5) // completed reward writes are also below the durable root + require.True(t, completed.retire(&info, ms.LastRootedSlot)) + rs, restored, reason := tryInLoopUnwind(sw, tail, ms, sched, 0, info) + require.Empty(t, reason) + require.Same(t, bank, restored, "use the surviving bank, never abandoned reward sysvars") + want, err := ResumeStateFromRootedContext(parent, nil) + require.NoError(t, err) + require.Equal(t, want, rs, "resume state must match rebuilding the exact retained parent") + acct, err := tail.GetAccount(8, testAccount(1, 0).Key) + require.NoError(t, err) + require.Equal(t, uint64(71), acct.Lamports, "abandoned account writes must be removed") +} From 35d73a7e49072b363d67abf0c7e2284071f339e8 Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:24:36 +0100 Subject: [PATCH 088/111] replay: checkpoint reward windows only after verified completion Keep the epoch boundary replayable when distribution consumes its last spool before a failed bank verification. Apply the hold again after finality gating, and commit the full reward window atomically so ordinary fold batches cannot leave an unrecoverable active EpochRewards checkpoint. Cover failed distribution, finality lag, completion generations, small fold batches and failed-commit retry. Document the epoch-116 diagnosis, bank-hash evidence and verified recovery checkpoint. --- docs/epoch116-reward-divergence.md | 82 +++++++++++++++++++++++++++ pkg/replay/block.go | 36 ++++++++---- pkg/replay/promotion.go | 17 ++++++ pkg/replay/rewards_retirement.go | 13 +++++ pkg/replay/rewards_retirement_test.go | 63 ++++++++++++++++++++ 5 files changed, 201 insertions(+), 10 deletions(-) create mode 100644 docs/epoch116-reward-divergence.md diff --git a/docs/epoch116-reward-divergence.md b/docs/epoch116-reward-divergence.md new file mode 100644 index 000000000..9d93ab52c --- /dev/null +++ b/docs/epoch116-reward-divergence.md @@ -0,0 +1,82 @@ +# Epoch 115 → 116 reward divergence + +The supplied run (`320ce8da`, started 2026-09-18) passed the epoch-boundary bank +at slot **6,264,000**, then stopped at **6,264,001** with a footer bank-hash +mismatch. Both blocks had zero transactions. The mismatch was a deterministic +reward-account state divergence, not a transaction-execution panic. + +At the boundary, rewards for epoch 115 were calculated and vote commissions +were paid. The following bank distributed one staking-reward partition with +724 stake records. Of those records, 33 paid zero lamports but advanced +`credits_observed` on stakes that had deactivated in epoch 114 and were fully +inactive in epoch 115. Agave leaves those accounts unchanged. The missing +effective-or-activating condition in `shouldForceCreditsOnly` caused the extra +writes and changed the LtHash. + +Reconstructing all 825 modified accounts from the saved parent accounts and +the diagnostic's per-account hashes reproduced the original computed bank +hash. Removing only those 33 writes reproduced the expected footer hash +exactly. The [reduced fixture and regression](../pkg/rewards/testdata/epoch116/README.md) +preserve this evidence and link the Agave/Firedancer rules. The code retains +activation status from the existing calculation, so no extra stake scan is +needed. Credit rewinds, disabled inflation, activation-epoch updates and +fractional rewards on effective stakes retain their prior behavior. + +## Checkpoint failure and recovery + +The distribution decremented its remaining-partition counter to zero before +the failing bank's footer was verified. The promotion guard used only that +counter, so graceful shutdown released the epoch hold and persisted slot +6,264,000 with an active EpochRewards sysvar. The spool had already been +consumed. That checkpoint cannot resume distribution directly. + +Promotion now waits for the successfully verified bank's immutable inactive +EpochRewards state. It also waits for that bank to pass the normal finality +and verification gates. The boundary through completion is committed in one +fold, including when the configured batch size is smaller than the rewards +window. A failed bank, a finality cutoff inside the window, or a failed fold +keeps the prior checkpoint. Memory remains bounded by the existing tail cap; +the completion fold can be larger than the usual batch, once per epoch. + +The supplied data includes a retained fold and transaction-status checkpoint +at **6,263,999**, the last slot of epoch 115. Read-only validation confirmed its +57,677,519-byte transaction-status checkpoint, complete retained-root coverage, +selected block identity, inactive EpochRewards state, and all 122 account undo +targets for reverting the boundary bank. With the fixed binary, the existing +`run --rewind-to-slot 6263999` option can restore that boundary and recalculate +rewards, provided historical blocks/shreds remain available. Keep the normal +node configuration and signing-history recovery checks. A direct restart from +6,264,000 cannot recover the missing distribution bookkeeping. + +The supplied local shred spool does not contain slots 6,264,000–6,264,001. On +2026-09-22 the configured primary RPC also reported 6,264,001 as pruned (first +available block 6,764,691). Replaying from the retained boundary therefore needs +another historical source; otherwise use the fixed binary with a fresh +snapshot. Original accounts, ledger, logs and signing history were preserved; +the fix was not deployed to a live validator during diagnosis. + +## Validation + +The branch was fetched and fast-forwarded from `320ce8da` to `be29319f` before +implementation. The reduced production-path regression failed on `be29319f` +with the incident's exact bad hash and passed after the fix. Additional tests +cover inactive, activating, effective and cooling stakes, forced credit-update +exceptions, failed distribution, finality lag, distribution generations, +atomic completion folds and failed-commit retry. + +Passed locally with Go 1.27.1 on linux/amd64: + +```sh +go test -race ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -count=1 +GOMAXPROCS=2 go test -race -p 2 -count=1 \ + ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards \ + ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... \ + ./cmd/mithril/node ./cmd/mithril/configcmd +go vet ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash +go build ./cmd/mithril +git diff --check +``` + +The CI regression job now includes `pkg/rewards`, so the incident fixture runs +on pull requests. These checks do not establish live post-restart catch-up or +later-slot parity; the original node was not restarted. diff --git a/pkg/replay/block.go b/pkg/replay/block.go index c31153b1a..ecb4e92df 100644 --- a/pkg/replay/block.go +++ b/pkg/replay/block.go @@ -1929,8 +1929,8 @@ func ReplayBlocks( var highestExecutedSlot uint64 // highest slot ProcessBlock has executed; bounds the promotion-gate walk // While partitioned rewards distribute, promotion holds below the boundary // block so a crash-resume always re-runs it (the distribution bookkeeping is - // RAM-only and not reconstructible mid-window). Self-clears when the window - // completes (NumRewardPartitionsRemaining reaches 0). + // RAM-only and not reconstructible mid-window). Release requires a verified + // completion bank, committed atomically with the whole rewards window. var rewardsHoldBelowSlot uint64 // Alpenglow finality identities captured at observe/ingest time for the promotion // gate (the tracker's own state may be pruned by promotion time). Pruned as slots @@ -2175,13 +2175,9 @@ func ReplayBlocks( } promoteThrough := safePromoteTarget(lastRootedWatermark, verifierRequired, verifiedWM, replayDivergenceFloor) // Partitioned-rewards window: hold promotion below the boundary block - // until every partition distributes, so a crash-resume re-runs the - // boundary and rebuilds the RAM-only distribution bookkeeping. - if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && partitionedRewardsInfo.NumRewardPartitionsRemaining > 0 { - if promoteThrough >= rewardsHoldBelowSlot { - promoteThrough = rewardsHoldBelowSlot - 1 - } - } + // until the completion bank verifies and is eligible to fold, so a + // failed distribution re-runs the boundary and rebuilds its bookkeeping. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { // Operator signal: promotion is fully stalled (verifier lag, // divergence floor, or rewards hold) while finality has run at @@ -2212,6 +2208,8 @@ func ReplayBlocks( mlog.Log.FileOnlyf("alpenglow gate: checked=%d matched=%d no_finality=%d no_local_id=%d", gateStats.checked, gateStats.matched, gateStats.noFinality, gateStats.noLocalID) } + // The finality gate can stop before the verified completion bank. + promoteThrough = rewardsCompletion.limitPromotion(partitionedRewardsInfo, rewardsHoldBelowSlot, promoteThrough) if promoteThrough <= mithrilState.LastRootedSlot { return false } @@ -2225,6 +2223,18 @@ func ReplayBlocks( if res := promoter.drain(); res != nil { applyFoldOutcome(res) } + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr := unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + if jerr != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", jerr) + return false + } + if err := runFoldJob(unrootedTailState.committer, job); err != nil { + mlog.Log.Errorf("rooted-durable: rewards completion fold: %v", err) + return false + } + applyFoldOutcome(&foldResult{job: job}) + } promotedThrough, rootedCtx, perr := unrootedTailState.flush(promoteThrough) if perr != nil { mlog.Log.Errorf("rooted-durable: forced fold stopped at slot %d: %v", promotedThrough, perr) @@ -2238,7 +2248,13 @@ func ReplayBlocks( // when idle; completions are applied at the top of this function on a // later iteration. if !promoter.inFlight { - job, jerr := unrootedTailState.buildFoldJob(promoteThrough, false) + var job *foldJob + var jerr error + if rewardsHoldBelowSlot > 0 && partitionedRewardsInfo != nil && promoteThrough >= rewardsHoldBelowSlot { + job, jerr = unrootedTailState.buildRewardsCompletionFoldJob(rewardsCompletion.slot) + } else { + job, jerr = unrootedTailState.buildFoldJob(promoteThrough, false) + } if jerr != nil { mlog.Log.Errorf("rooted-durable: %v; watermark held back", jerr) return false diff --git a/pkg/replay/promotion.go b/pkg/replay/promotion.go index c750009ea..51d966a44 100644 --- a/pkg/replay/promotion.go +++ b/pkg/replay/promotion.go @@ -388,6 +388,23 @@ type foldResult struct { err error } +// buildRewardsCompletionFoldJob puts the entire rewards window in one commit. +// A normal batch cutoff inside that window would leave an active EpochRewards +// checkpoint without the RAM-only spool bookkeeping needed to resume it. The +// retained tail already bounds the size of this once-per-epoch fold. +func (t *unrootedTail) buildRewardsCompletionFoldJob(through uint64) (*foldJob, error) { + whole := *t + whole.batchSlots = t.overlay.HeldSlots() + job, err := whole.buildFoldJob(through, true) + if err != nil { + return nil, err + } + if job == nil || job.through != through { + return nil, fmt.Errorf("rewards completion bank %d is absent from retained fold prefix", through) + } + return job, nil +} + // buildFoldJob snapshots the FIRST fold chunk of the rooted prefix <= through // (loop thread). force also takes a trailing partial chunk. Returns nil when // no chunk is ready. A missing chunk-top context is an error — a context-less diff --git a/pkg/replay/rewards_retirement.go b/pkg/replay/rewards_retirement.go index a03a00cbf..18783e382 100644 --- a/pkg/replay/rewards_retirement.go +++ b/pkg/replay/rewards_retirement.go @@ -15,6 +15,19 @@ type partitionedRewardsCompletion struct { slot uint64 } +// limitPromotion keeps the boundary replayable until a successfully verified +// completion bank is eligible for promotion. Consuming the last spool changes +// the RAM counter before footer verification and is not completion evidence. +func (c *partitionedRewardsCompletion) limitPromotion(info *rewards.PartitionedRewardDistributionInfo, boundary, through uint64) uint64 { + if boundary == 0 || info == nil { + return through + } + if info.NumRewardPartitionsRemaining != 0 || c.info != info || c.slot == 0 || through < c.slot { + return min(through, boundary-1) + } + return through +} + // observeBank must run only after successful block execution/publication, using // that bank's immutable sysvars (never the speculative global sysvar cache). // If the first completed bank lacks evidence, recording a later descendant is diff --git a/pkg/replay/rewards_retirement_test.go b/pkg/replay/rewards_retirement_test.go index 460584eee..aaa169ae1 100644 --- a/pkg/replay/rewards_retirement_test.go +++ b/pkg/replay/rewards_retirement_test.go @@ -3,6 +3,7 @@ package replay import ( "bytes" "encoding/base64" + "fmt" "testing" "github.com/Overclock-Validator/mithril/pkg/accounts" @@ -53,6 +54,68 @@ func TestRewardsRetirementRequiresCompletedBankAndDurability(t *testing.T) { require.False(t, completed.retire(&info, 100), "retirement is one-shot") } +func TestRewardsPromotionRetainsBoundaryAfterFailedDistribution(t *testing.T) { + const boundary = uint64(6264000) + info := &rewards.PartitionedRewardDistributionInfo{NumRewardPartitionsRemaining: 1} + var completed partitionedRewardsCompletion + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + // Distribution consumes the final spool before ProcessBlock checks the + // footer. The epoch-116 failure took this path; no successful bank was + // observed, so forced shutdown must not persist the boundary bank. + info.NumRewardPartitionsRemaining = 0 + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + completed.observeBank(info, nil) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary+1)) + + completed.observeBank(info, testUnwindBankSysvars(t, boundary+1, 50)) + require.Equal(t, boundary-2, completed.limitPromotion(info, boundary, boundary-2)) + require.Equal(t, boundary-1, completed.limitPromotion(info, boundary, boundary), "finality stopped inside rewards window") + require.Equal(t, boundary+1, completed.limitPromotion(info, boundary, boundary+1)) + + next := &rewards.PartitionedRewardDistributionInfo{} + require.Equal(t, boundary-1, completed.limitPromotion(next, boundary, boundary+2), "completion belongs to another distribution") + require.Equal(t, boundary+2, completed.limitPromotion(nil, boundary, boundary+2)) + require.Equal(t, boundary+2, completed.limitPromotion(info, 0, boundary+2)) +} + +func TestRewardsCompletionFoldCannotCheckpointInsideWindow(t *testing.T) { + for _, batchSize := range []int{1, 2, 128} { + t.Run(fmt.Sprintf("batch_%d", batchSize), func(t *testing.T) { + fc := &fakeCommitter{durable: accounts.NewMemAccounts(), failOn: 7} + tail := asyncTestTail(fc, 5, 6, 7, 8) + tail.batchSlots = batchSize + info := &rewards.PartitionedRewardDistributionInfo{} + var completed partitionedRewardsCompletion + completed.observeBank(info, testUnwindBankSysvars(t, 7, 50)) + through := completed.limitPromotion(info, 5, 8) + require.Equal(t, uint64(8), through) + + job, err := tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NotNil(t, job) + require.Equal(t, uint64(7), job.through) + require.Len(t, job.chunk, 3, "the entire distribution must share one commit") + require.Equal(t, batchSize, tail.batchSlots, "normal batching is unchanged") + require.Error(t, runFoldJob(fc, job)) + require.Empty(t, fc.throughs) + require.False(t, completed.retire(&info, 4)) + + fc.failOn = 0 + job, err = tail.buildRewardsCompletionFoldJob(completed.slot) + require.NoError(t, err) + require.NoError(t, runFoldJob(fc, job)) + require.Equal(t, []uint64{7}, fc.throughs) + tail.applyFoldJob(job) + require.True(t, completed.retire(&info, 7)) + require.Equal(t, 1, tail.overlay.HeldSlots(), "later banks remain buffered") + }) + } + tail := asyncTestTail(&fakeCommitter{durable: accounts.NewMemAccounts()}, 5, 6) + _, err := tail.buildRewardsCompletionFoldJob(7) + require.ErrorContains(t, err, "absent from retained fold prefix") +} + func TestRewardsRetirementDoesNotCrossGenerations(t *testing.T) { old := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 1} next := &rewards.PartitionedRewardDistributionInfo{SpoolSlot: 10} From 0fe2fc17345af7a8594e871fc7a3fad1091f036a Mon Sep 17 00:00:00 2001 From: Shaun Colley <79477620+smcio@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:27:41 +0100 Subject: [PATCH 089/111] docs: remove epoch 116 incident report --- docs/epoch116-reward-divergence.md | 82 ------------------------------ 1 file changed, 82 deletions(-) delete mode 100644 docs/epoch116-reward-divergence.md diff --git a/docs/epoch116-reward-divergence.md b/docs/epoch116-reward-divergence.md deleted file mode 100644 index 9d93ab52c..000000000 --- a/docs/epoch116-reward-divergence.md +++ /dev/null @@ -1,82 +0,0 @@ -# Epoch 115 → 116 reward divergence - -The supplied run (`320ce8da`, started 2026-09-18) passed the epoch-boundary bank -at slot **6,264,000**, then stopped at **6,264,001** with a footer bank-hash -mismatch. Both blocks had zero transactions. The mismatch was a deterministic -reward-account state divergence, not a transaction-execution panic. - -At the boundary, rewards for epoch 115 were calculated and vote commissions -were paid. The following bank distributed one staking-reward partition with -724 stake records. Of those records, 33 paid zero lamports but advanced -`credits_observed` on stakes that had deactivated in epoch 114 and were fully -inactive in epoch 115. Agave leaves those accounts unchanged. The missing -effective-or-activating condition in `shouldForceCreditsOnly` caused the extra -writes and changed the LtHash. - -Reconstructing all 825 modified accounts from the saved parent accounts and -the diagnostic's per-account hashes reproduced the original computed bank -hash. Removing only those 33 writes reproduced the expected footer hash -exactly. The [reduced fixture and regression](../pkg/rewards/testdata/epoch116/README.md) -preserve this evidence and link the Agave/Firedancer rules. The code retains -activation status from the existing calculation, so no extra stake scan is -needed. Credit rewinds, disabled inflation, activation-epoch updates and -fractional rewards on effective stakes retain their prior behavior. - -## Checkpoint failure and recovery - -The distribution decremented its remaining-partition counter to zero before -the failing bank's footer was verified. The promotion guard used only that -counter, so graceful shutdown released the epoch hold and persisted slot -6,264,000 with an active EpochRewards sysvar. The spool had already been -consumed. That checkpoint cannot resume distribution directly. - -Promotion now waits for the successfully verified bank's immutable inactive -EpochRewards state. It also waits for that bank to pass the normal finality -and verification gates. The boundary through completion is committed in one -fold, including when the configured batch size is smaller than the rewards -window. A failed bank, a finality cutoff inside the window, or a failed fold -keeps the prior checkpoint. Memory remains bounded by the existing tail cap; -the completion fold can be larger than the usual batch, once per epoch. - -The supplied data includes a retained fold and transaction-status checkpoint -at **6,263,999**, the last slot of epoch 115. Read-only validation confirmed its -57,677,519-byte transaction-status checkpoint, complete retained-root coverage, -selected block identity, inactive EpochRewards state, and all 122 account undo -targets for reverting the boundary bank. With the fixed binary, the existing -`run --rewind-to-slot 6263999` option can restore that boundary and recalculate -rewards, provided historical blocks/shreds remain available. Keep the normal -node configuration and signing-history recovery checks. A direct restart from -6,264,000 cannot recover the missing distribution bookkeeping. - -The supplied local shred spool does not contain slots 6,264,000–6,264,001. On -2026-09-22 the configured primary RPC also reported 6,264,001 as pruned (first -available block 6,764,691). Replaying from the retained boundary therefore needs -another historical source; otherwise use the fixed binary with a fresh -snapshot. Original accounts, ledger, logs and signing history were preserved; -the fix was not deployed to a live validator during diagnosis. - -## Validation - -The branch was fetched and fast-forwarded from `320ce8da` to `be29319f` before -implementation. The reduced production-path regression failed on `be29319f` -with the incident's exact bad hash and passed after the fix. Additional tests -cover inactive, activating, effective and cooling stakes, forced credit-update -exceptions, failed distribution, finality lag, distribution generations, -atomic completion folds and failed-commit retry. - -Passed locally with Go 1.27.1 on linux/amd64: - -```sh -go test -race ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -count=1 -GOMAXPROCS=2 go test -race -p 2 -count=1 \ - ./pkg/alpenglow ./pkg/consensus ./pkg/replay ./pkg/rewards \ - ./pkg/turbine ./pkg/sigverify ./pkg/blockprod/... \ - ./cmd/mithril/node ./cmd/mithril/configcmd -go vet ./pkg/rewards ./pkg/replay ./pkg/accountsdb ./pkg/bankhash -go build ./cmd/mithril -git diff --check -``` - -The CI regression job now includes `pkg/rewards`, so the incident fixture runs -on pull requests. These checks do not establish live post-restart catch-up or -later-slot parity; the original node was not restarted. From fd4ae2f8c7cc334bad9f1339bc531ace30279ab5 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:59:15 -0500 Subject: [PATCH 090/111] fix: preserve boolean flag defaults and explicit config overrides --- cmd/mithril/node/config_bool.go | 15 ++++++ cmd/mithril/node/config_bool_test.go | 38 ++++++++++++++ cmd/mithril/node/node.go | 9 +--- pkg/sbpf/pooling_test.go | 75 ++++++++++++++++++++++++++++ 4 files changed, 130 insertions(+), 7 deletions(-) create mode 100644 cmd/mithril/node/config_bool.go create mode 100644 cmd/mithril/node/config_bool_test.go create mode 100644 pkg/sbpf/pooling_test.go diff --git a/cmd/mithril/node/config_bool.go b/cmd/mithril/node/config_bool.go new file mode 100644 index 000000000..e779efb61 --- /dev/null +++ b/cmd/mithril/node/config_bool.go @@ -0,0 +1,15 @@ +package node + +import "github.com/spf13/pflag" + +// resolveBoolOption preserves explicit false at either precedence level. +// Defaults use DefValue, not a flag value potentially left by a previous run. +func resolveBoolOption(flag *pflag.Flag, configured bool, configuredValue bool) bool { + if flag != nil && flag.Changed { + return flag.Value.String() == "true" + } + if configured { + return configuredValue + } + return flag != nil && flag.DefValue == "true" +} diff --git a/cmd/mithril/node/config_bool_test.go b/cmd/mithril/node/config_bool_test.go new file mode 100644 index 000000000..ab23313d0 --- /dev/null +++ b/cmd/mithril/node/config_bool_test.go @@ -0,0 +1,38 @@ +package node + +import ( + "github.com/spf13/pflag" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" + "strings" + "testing" +) + +func TestResolveBoolOptionPrecedence(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + defaultValue, want bool + }{ + {"omitted true default", "", "", true, true}, + {"omitted false default", "", "", false, false}, + {"TOML false", "enabled=false", "", true, false}, + {"TOML true", "enabled=true", "", false, true}, + {"CLI false beats TOML true", "enabled=true", "false", true, false}, + {"CLI true beats TOML false", "enabled=false", "true", false, true}, + {"CLI false without TOML", "", "false", true, false}, + } { + t.Run(tc.name, func(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(tc.toml))) + flags := pflag.NewFlagSet("test", pflag.ContinueOnError) + flags.Bool("enabled", tc.defaultValue, "") + if tc.cli != "" { + require.NoError(t, flags.Set("enabled", tc.cli)) + } + require.Equal(t, tc.want, resolveBoolOption(flags.Lookup("enabled"), v.IsSet("enabled"), v.GetBool("enabled"))) + }) + } + require.False(t, resolveBoolOption(nil, false, false)) + require.True(t, resolveBoolOption(nil, true, true)) +} diff --git a/cmd/mithril/node/node.go b/cmd/mithril/node/node.go index 66a81e3e1..57c8f76a0 100644 --- a/cmd/mithril/node/node.go +++ b/cmd/mithril/node/node.go @@ -742,14 +742,9 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { return 0 } - // Helper to get bool: CLI flag if explicitly set, otherwise TOML config + // Match numeric options: explicit CLI, configured value, then flag default. getBool := func(cliKey, tomlKey string) bool { - if flagChanged(cliKey) { - if f := cmd.Flags().Lookup(cliKey); f != nil { - return f.Value.String() == "true" - } - } - return config.GetBool(tomlKey) + return resolveBoolOption(cmd.Flags().Lookup(cliKey), config.IsSet(tomlKey), config.GetBool(tomlKey)) } // Helper to get string slice: CLI flag if explicitly set, otherwise TOML config diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go new file mode 100644 index 000000000..38d1028f8 --- /dev/null +++ b/pkg/sbpf/pooling_test.go @@ -0,0 +1,75 @@ +package sbpf + +import ( + "bytes" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/stretchr/testify/require" +) + +func poolingInterpreter(heap int) *Interpreter { + meter := cu.NewComputeMeter(100) + return NewInterpreter(testV3Program([]Slot{testSlot(OpExit, 0, 0, 0, 0)}, nil), + &VMOpts{HeapMax: heap, ComputeMeter: &meter}) +} + +func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { + old := UsePool + UsePool = true + t.Cleanup(func() { UsePool = old }) + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for i := 0; i < 30; i++ { + parent := poolingInterpreter(32 * 1024) + if !bytes.Equal(parent.heap, make([]byte, len(parent.heap))) || + !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { + t.Error("pooled VM exposed data from an earlier execution") + } + parent.heap[0] = byte(worker + 1) + parent.stack.mem[0] = byte(worker + 1) + child := poolingInterpreter(256 * 1024) + for j := range child.heap { + child.heap[j] = 0xab + } + for j := range child.stack.mem { + child.stack.mem[j] = 0xcd + } + child.Finish() + if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { + t.Error("nested VM storage aliased its active parent") + } + parent.Finish() + } + }(worker) + } + wg.Wait() + ip := poolingInterpreter(32 * 1024) + defer ip.Finish() + require.Len(t, ip.heap, 32*1024) + _, err := ip.Translate(VaddrHeap+32*1024, 1, false) + require.Error(t, err, "pool capacity must not widen the requested heap mapping") +} + +func BenchmarkVMCreateAndFinish(b *testing.B) { + old := UsePool + b.Cleanup(func() { UsePool = old }) + for _, pooled := range []bool{false, true} { + name := "fresh" + if pooled { + name = "pooled" + } + b.Run(name, func(b *testing.B) { + UsePool = pooled + b.ReportAllocs() + for i := 0; i < b.N; i++ { + ip := poolingInterpreter(32 * 1024) + ip.Finish() + } + }) + } +} From c7d9f6cca2985fa61ff2a700565303cf065487ef Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:59:15 -0500 Subject: [PATCH 091/111] fix: own retained vote deque storage before pooled reuse --- pkg/sealevel/vote_deque_ownership_test.go | 29 +++++++++++++++++++++++ pkg/sealevel/vote_program.go | 9 ++++++- 2 files changed, 37 insertions(+), 1 deletion(-) create mode 100644 pkg/sealevel/vote_deque_ownership_test.go diff --git a/pkg/sealevel/vote_deque_ownership_test.go b/pkg/sealevel/vote_deque_ownership_test.go new file mode 100644 index 000000000..b7d50f43b --- /dev/null +++ b/pkg/sealevel/vote_deque_ownership_test.go @@ -0,0 +1,29 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" + "github.com/gammazero/deque" + "github.com/stretchr/testify/require" +) + +func TestProcessNewVoteStateOwnsRetainedDeque(t *testing.T) { + // Model the TowerSync scratch deque being returned to its pool and reused. + scratch := new(deque.Deque[LandedVote]) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 100, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 101, ConfirmationCount: 1}}) + state := new(VoteState) + require.NoError(t, processNewVoteState(state, scratch, nil, nil, 0, 101, features.Features{})) + cached := newVoteState4FromCurrent(state, solana.PublicKey{}) + want := []LandedVote{state.Votes.At(0), state.Votes.At(1)} + + scratch.Clear() + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 200, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 201, ConfirmationCount: 1}}) + for i, vote := range want { + require.Equal(t, vote, state.Votes.At(i)) + require.Equal(t, vote, cached.Votes.At(i)) + } +} diff --git a/pkg/sealevel/vote_program.go b/pkg/sealevel/vote_program.go index 43a8500c2..679d1787d 100644 --- a/pkg/sealevel/vote_program.go +++ b/pkg/sealevel/vote_program.go @@ -1951,7 +1951,14 @@ func processNewVoteState(voteState *VoteState, newState *deque.Deque[LandedVote] } voteState.RootSlot = newRoot - voteState.Votes = *newState + // newState may be a pooled deque. Own the backing storage before its + // caller returns it to the pool: the resulting state can escape into the + // shared vote cache after this instruction completes. + var owned deque.Deque[LandedVote] + for i := 0; i < newState.Len(); i++ { + owned.PushBack(newState.At(i)) + } + voteState.Votes = owned return nil } From 8a424ba45678218619a8f865bcb5b2c64419dc39 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:12:58 -0500 Subject: [PATCH 092/111] sbpf: reject overflow in contiguous virtual memory ranges --- pkg/sbpf/interpreter.go | 6 +++--- pkg/sbpf/translate_overflow_test.go | 25 +++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 3 deletions(-) create mode 100644 pkg/sbpf/translate_overflow_test.go diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 4fa0f784a..5f351a5d6 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1186,7 +1186,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.ro)) { + if lo+size < lo || lo+size > uint64(len(ip.ro)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds program read") } return unsafe.Pointer(&ip.ro[lo]), nil @@ -1203,7 +1203,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.heap)) { + if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } return unsafe.Pointer(&ip.heap[lo]), nil @@ -1214,7 +1214,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if len(ip.inputRegions) != 0 { return ip.translateInputRegion(lo, size, write) } - if lo+size > uint64(len(ip.input)) { + if lo+size < lo || lo+size > uint64(len(ip.input)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds input access") } return unsafe.Pointer(&ip.input[lo]), nil diff --git a/pkg/sbpf/translate_overflow_test.go b/pkg/sbpf/translate_overflow_test.go new file mode 100644 index 000000000..98d8ee2d0 --- /dev/null +++ b/pkg/sbpf/translate_overflow_test.go @@ -0,0 +1,25 @@ +package sbpf + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestTranslateRejectsOverflowingRange(t *testing.T) { + ip := &Interpreter{ro: make([]byte, 32), heap: make([]byte, 32), input: make([]byte, 32)} + for _, base := range []uint64{VaddrProgram, VaddrHeap, VaddrInput} { + for _, write := range []bool{false, true} { + for _, offset := range []uint64{1, 31, 33, 0xffffffff} { + for _, size := range []uint64{^uint64(0), ^uint64(0) - 15} { + _, err := ip.Translate(base+offset, size, write) + require.Error(t, err, "base=%x offset=%d size=%d write=%v", base, offset, size, write) + } + } + } + got, err := ip.Translate(base+31, 1, false) + require.NoError(t, err) + require.Len(t, got, 1) + _, err = ip.Translate(base+31, 2, false) + require.Error(t, err) + } +} From c12362ba620058d5fd6539523331ac1af8e7bf92 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:14:24 -0500 Subject: [PATCH 093/111] sealevel: remove SHA-256 syscall descriptor and digest allocations --- docs/sha256-syscall.md | 42 ++++ pkg/sealevel/sealevel_test.go | 12 +- pkg/sealevel/syscalls_hash.go | 15 +- pkg/sealevel/syscalls_sha256_bench_test.go | 268 +++++++++++++++++++++ 4 files changed, 325 insertions(+), 12 deletions(-) create mode 100644 docs/sha256-syscall.md create mode 100644 pkg/sealevel/syscalls_sha256_bench_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md new file mode 100644 index 000000000..58a83abca --- /dev/null +++ b/docs/sha256-syscall.md @@ -0,0 +1,42 @@ +# SHA-256 syscall overhead + +The syscall decodes the already-translated slice descriptor array directly and +writes the final digest into the translated output buffer. It retains streaming +SHA-256, slice order, memory translations, CU charges and validation order. Output +is written only after all inputs have been read, preserving overlapping-buffer +behavior. No special case for a particular on-chain program is introduced. + +A bounded 55-byte input-buffer prototype was slower than this simpler path and +is retained only as a benchmark comparison. The baseline reference is copied +from combined review commit `bd17683a`. + +Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; +medians below. Each benchmark runs serially through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. This does +not include VM instruction dispatch, a complete program, or block replay. + +| Input | Original | Direct decoding/output | Buffered prototype | +|---|---:|---:|---:| +| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | +| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | +| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | +| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | + +The two-slice case removes four allocations (112 bytes) per call. This is an +ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native +Zen 5 and captured heavy-block replay before deployment decisions. + +Reproduce with: + +```sh +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +``` + +Differential tests compare hashes, return/error values, remaining CU and all +input/output memory over valid inputs, invalid descriptors/addresses, depleted +budgets and output aliasing. The existing SHA program fixture also executes. +Testing exposed pre-existing overflow in contiguous VM region bounds checks; +that correction and its regression test are a separate preceding commit. Both +benchmark variants use the corrected VM. The old SHA fixture also needed its +compute-meter pointer initialized for the current interpreter API. diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ddfe230a..6ad64f3a9 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -416,13 +416,15 @@ func TestInterpreter_Sha256(t *testing.T) { syscalls.Register("my_memcmp", SyscallMemcmp) var log LogRecorder + ctx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: ctx, + ComputeMeter: &ctx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_hash.go b/pkg/sealevel/syscalls_hash.go index b73a1cd0c..33edd31d0 100644 --- a/pkg/sealevel/syscalls_hash.go +++ b/pkg/sealevel/syscalls_hash.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" "crypto/sha256" + "encoding/binary" "fmt" "math/big" @@ -49,15 +50,13 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 } var data []byte - reader := bytes.NewReader(vals) + // Translate validated the complete descriptor array above. Decode directly + // to avoid allocating a reader and a temporary buffer for each slice. for count := uint64(0); count < valsLen; count++ { - var vec VectorDescrC - err = vec.Unmarshal(reader) - if err != nil { - return syscallErr(err) - } + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} data, err = vm.Translate(vec.Addr, vec.Len, false) if err != nil { @@ -73,7 +72,9 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 hasher.Write(data) } } - copy(hashResult[:], hasher.Sum(nil)) + // All inputs have been read before writing, including when output aliases + // input memory. Append into the translated destination without allocating. + hasher.Sum(hashResult[:0]) return syscallSuccess(0) } diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go new file mode 100644 index 000000000..b4240f54e --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -0,0 +1,268 @@ +package sealevel + +import ( + "bytes" + "crypto/sha256" + "encoding/binary" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +// Frozen syscall implementation from bd17683a; keep independent for differential tests. +func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256BaselineReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + reader := bytes.NewReader(vals) + + for count := uint64(0); count < valsLen; count++ { + + var vec VectorDescrC + err = vec.Unmarshal(reader) + if err != nil { + return syscallErr(err) + } + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + hasher.Write(data) + } + } + copy(hashResult[:], hasher.Sum(nil)) + return syscallSuccess(0) +} + +// Experimental bounded-buffer variant retained only for benchmark comparison. +func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256SmallInputReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only + // this bounded case; larger inputs retain streaming hashing. + var small [55]byte + buffered := 0 + streaming := false + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + + for count := uint64(0); count < valsLen; count++ { + + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + if !streaming && len(data) <= len(small)-buffered { + buffered += copy(small[buffered:], data) + } else { + if !streaming { + hasher.Write(small[:buffered]) + streaming = true + } + hasher.Write(data) + } + } + } + if streaming { + hasher.Sum(hashResult[:0]) + } else { + digest := sha256.Sum256(small[:buffered]) + copy(hashResult, digest[:]) + } + return syscallSuccess(0) +} + +type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) + +func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { + mem := make([]byte, 32768) + pos := 8192 + for i, n := range sizes { + binary.LittleEndian.PutUint64(mem[i*16:], sbpf.VaddrInput+uint64(pos)) + binary.LittleEndian.PutUint64(mem[i*16+8:], uint64(n)) + for j := 0; j < n; j++ { + mem[pos+j] = byte(i + j) + } + pos += n + } + return mem, sbpf.VaddrInput, uint64(len(sizes)), sbpf.VaddrInput + 4096 +} + +func sha256VM(mem []byte, budget uint64) (*sbpf.Interpreter, *ExecutionCtx) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(budget)} + vm := sbpf.NewInterpreter(&sbpf.Program{}, &sbpf.VMOpts{Input: mem, HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + return vm, ctx +} + +func TestSha256SyscallDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(1234)) + for i := 0; i < 700; i++ { + sizes := make([]int, rng.Intn(8)) + for j := range sizes { + sizes[j] = rng.Intn(80) + } + if i < 8 { + sizes = [][]int{nil, {0}, {32, 4}, {55}, {56}, {32, 24}, {4096}, {0, 32, 0, 4}}[i] + } + mem, a, n, out := sha256Fixture(sizes) + budget := uint64(100000) + switch i % 11 { + case 1: + budget = uint64(rng.Intn(250)) + case 2: + out = 1 + case 3: + a = 1 + case 4: + if n > 0 { + binary.LittleEndian.PutUint64(mem, 1) + } + case 5: + if n > 0 { + binary.LittleEndian.PutUint64(mem[8:], ^uint64(0)) + } + case 6: + out = sbpf.VaddrInput + 8192 // output overlaps the input + case 7: + out = a // output overlaps descriptors + case 8: + n = cu.CUSha256MaxSlices + 1 + case 9: + n = cu.CUSha256MaxSlices + case 10: + a = sbpf.VaddrInput + uint64(len(mem)-1) + } + var wantMem []byte + var wantRet, wantCU uint64 + var wantErr string + for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + buf := append([]byte(nil), mem...) + vm, ctx := sha256VM(buf, budget) + ret, err := fn(vm, a, n, out) + remaining := ctx.ComputeMeter.Remaining() + vm.Finish() + if k == 0 { + wantMem = buf + wantRet = ret + wantCU = remaining + wantErr = fmt.Sprint(err) + continue + } + require.Equal(t, wantRet, ret, "case %d variant %d", i, k) + require.Equal(t, wantErr, fmt.Sprint(err), "case %d variant %d", i, k) + require.Equal(t, wantCU, remaining, "case %d variant %d", i, k) + require.Equal(t, wantMem, buf, "case %d variant %d", i, k) + } + } +} + +var sha256BenchDigest [32]byte + +func BenchmarkSha256Syscall(b *testing.B) { + for _, tc := range []struct { + name string + sizes []int + }{{"empty", nil}, {"36_contiguous", []int{36}}, {"32_plus_4", []int{32, 4}}, {"55", []int{55}}, {"56", []int{56}}, {"1232", []int{1232}}, {"4096", []int{4096}}} { + for _, variant := range []struct { + name string + fn sha256Call + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + b.Run(tc.name+"/"+variant.name, func(b *testing.B) { + mem, a, n, out := sha256Fixture(tc.sizes) + vm, ctx := sha256VM(mem, ^uint64(0)) + defer vm.Finish() + b.ReportAllocs() + b.ResetTimer() + for j := 0; j < b.N; j++ { + ctx.ComputeMeter = cu.NewComputeMeter(100000) + if _, err := variant.fn(vm, a, n, out); err != nil { + b.Fatal(err) + } + } + }) + } + } + b.Run("raw36", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + sha256BenchDigest = sha256.Sum256(data[:]) + } + }) +} From bb23409b8a7f2bd32e6191221ace9c6383fe52af Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:20:08 -0500 Subject: [PATCH 094/111] test: measure captured SHA loop and document Zen 5 acceleration --- docs/sha256-syscall.md | 47 ++++++++++ pkg/sealevel/syscalls_sha256_loop_test.go | 102 ++++++++++++++++++++++ 2 files changed, 149 insertions(+) create mode 100644 pkg/sealevel/syscalls_sha256_loop_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 58a83abca..10c1b59b2 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -40,3 +40,50 @@ Testing exposed pre-existing overflow in contiguous VM region bounds checks; that correction and its regression test are a separate preceding commit. Both benchmark variants use the corrected VM. The old SHA fixture also needed its compute-meter pointer initialized for the current interpreter API. + + +## Zen 5 acceleration and captured loop + +The live Go 1.26.4 validator on Ryzen 7 9700X had its actual +`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is +already active. No validator runtime setting was changed. + +The isolated loop harness copies text slots 518–545 from the captured SBF v0 +program, resolves the SHA syscall relocation, and supplies 1,000 iterations and +zero initial state. It retains descriptor setup, stack accesses, digest copying, +counter update and loop branching. A test checks its result against a Go hash +chain and checks equal CU consumption for both syscall implementations. This +excludes transaction loading, account dependencies, CPI and the remaining program. + +A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, +affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. +Three 150 ms samples (medians, per hash iteration): + +| Isolated loop | Time | +|---|---:| +| Original syscall | 234.5 ns | +| Optimized syscall | 165.9 ns | +| Dispatch-only diagnostic control | 109.2 ns | +| Go hash chain without VM | 54.69 ns | + +The optimized loop takes about 29% less time. The dispatch-only control replaces +the syscall with a no-op: it omits hashing, translations and syscall CU charging, +and is only an overhead diagnostic, never a valid execution implementation. +The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; +the buffered-input prototype remained slower at 71–73 ns. These short tests +share a host with other processes; they are not isolated-core latency guarantees. + +A separate short CPU profile of the optimized loop attributed 36.2% cumulative +sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI +compression routine. Most remaining sampled work was VM execution: instruction +dispatch/decoding, stack address translation, loads/stores and compute metering. +Cumulative and flat percentages overlap and must not be added. This profile is +of the harness, not of full-block replay or the live validator. + +The next execution experiment should target measured VM overhead and then replay +captured blocks; these results do not justify a claimed 29% block-time improvement. + +```sh +go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +``` diff --git a/pkg/sealevel/syscalls_sha256_loop_test.go b/pkg/sealevel/syscalls_sha256_loop_test.go new file mode 100644 index 000000000..32c22d9f0 --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_loop_test.go @@ -0,0 +1,102 @@ +package sealevel + +import ( + "crypto/sha256" + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +// This is text slots 518..545 from the captured AogGeA81 program's hash loop. +// Captured ELF SHA-256: b3286f96f5611ee7db62dc11ea9aab1afe80908c08a0fbd3fdd783879d70ed88. +// The captured ELF uses SBF v0. Only the unresolved syscall relocation is replaced. The harness supplies a +// loop bound and zero initial state; transaction loading, CPI and the rest of +// the program are deliberately excluded. +func sha256LoopProgram(iterations uint32) *sbpf.Program { + text := []sbpf.Slot{ + sbpf.Slot(sbpf.OpMov64Imm) | 7<<8 | sbpf.Slot(iterations)<<32, + 0xa1bf, 0xffffffb000000107, 0xfe701a7b, 0xa1bf, 0xfffffe1000000107, + 0xfe601a7b, 0xffb08a63, 0x4fe780a7a, 0x20fe680a7a, 0xa1bf, + 0xfffffe6000000107, 0xa3bf, 0xffffffe000000307, 0x2000002b7, + sbpf.Slot(sbpf.OpCall) | sbpf.Slot(hash_sol_sha256)<<32, + 0xffe0a179, 0xfe101a7b, 0xffe8a179, 0xfe181a7b, 0xfff0a179, + 0xfe201a7b, 0xfff8a179, 0xfe281a7b, 0x100000807, 0x81bf, + 0x2000000167, 0x2000000177, 0xffe471ad, + // Return the first digest word so the harness can check the computation. + 0xfe10a079, sbpf.Slot(sbpf.OpExit), + } + return &sbpf.Program{Text: text} +} + +func runSha256Loop(p *sbpf.Program, fn sha256Call) (uint64, uint64, error) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(10000000)} + vm := sbpf.NewInterpreter(p, &sbpf.VMOpts{Context: ctx, ComputeMeter: &ctx.ComputeMeter, Syscalls: func(hash uint32) (sbpf.Syscall, bool) { return sbpf.SyscallFunc3(fn), hash == hash_sol_sha256 }}) + ret, _, err := vm.Run() + used := ctx.ComputeMeter.Used() + vm.Finish() + return ret, used, err +} + +func TestSha256CapturedLoop(t *testing.T) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + require.NoError(t, p.Verify()) + var data [36]byte + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + want := binary.LittleEndian.Uint64(data[:8]) + var wantCU uint64 + for _, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { + got, used, err := runSha256Loop(p, fn) + require.NoError(t, err) + require.Equal(t, want, got) + if wantCU == 0 { + wantCU = used + } + require.Equal(t, wantCU, used) + } +} + +func BenchmarkSha256CapturedLoop(b *testing.B) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + for _, v := range []struct { + name string + fn sha256Call + }{ + {"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, + // Diagnostic lower bound: no hashing, translations or syscall CU charging. + // It is not a valid implementation and must never be used in replay. + {"dispatch_only", func(sbpf.VM, uint64, uint64, uint64) (uint64, error) { return 0, nil }}, + } { + b.Run(v.name, func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + if _, _, err := runSha256Loop(p, v.fn); err != nil { + b.Fatal(err) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) + } + b.Run("raw_chain", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + clear(data[:]) + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + } + sha256BenchDigest = sha256.Sum256(data[:]) + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) +} From de8ea09db034334bc54a710fc00beb2c9cc66451 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 095/111] sbpf: cut per-instruction overhead in the interpreter (~2x on SPL Token transfer) Consensus-neutral performance changes to pkg/sbpf, validated against the unmodified interpreter with a 100k-program differential corpus (identical return values, errors/PCs, CU consumed, meter remaining, memory contents, input-region state) plus the package's unit tests: - meter instructions with a local due/budget pair synced around syscalls and on exit (Agave's due_insn_count scheme) instead of calling ComputeMeter.Consume per instruction - move cold opcodes to executeCold so Run drops below the compiler's "big function" threshold and Consume/Read*/Push/Pop/fast paths inline - zero only the dirty range of the pooled stack/heap in Finish (page bitmap on the fast path, byte range on the translate path) instead of 256 KiB + HeapMax per execution - per-window fast-path address translation table (Agave aligned mapping layout, branch-free v0 frame gaps, one-entry cache for VASA input regions) - 16-wide register file (no bounds checks on r[dst]/r[src]), in-place call-frame Push/Pop, precomputed internal call targets per Program pooling_test writes through the VM's translation layer now, since the pool only re-zeroes memory the VM saw written (all production writes go through translation). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/cu/cu.go | 5 + pkg/sbpf/fastmem.go | 59 ++ pkg/sbpf/interpreter.go | 1236 +++++++++++++++++++++---------------- pkg/sbpf/loader/loader.go | 4 +- pkg/sbpf/pooling_test.go | 26 +- pkg/sbpf/program.go | 23 + pkg/sbpf/stack.go | 74 ++- 7 files changed, 873 insertions(+), 554 deletions(-) create mode 100644 pkg/sbpf/fastmem.go diff --git a/pkg/cu/cu.go b/pkg/cu/cu.go index 79d337199..f90aa1e2c 100644 --- a/pkg/cu/cu.go +++ b/pkg/cu/cu.go @@ -33,6 +33,11 @@ func (cm *ComputeMeter) Consume(cost uint64) error { return nil } +// Disabled reports whether metering is currently switched off (Consume is a no-op). +func (cm *ComputeMeter) Disabled() bool { + return cm.disable +} + func (cm *ComputeMeter) Used() uint64 { return cm.startingBalance - cm.computeMeter } diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go new file mode 100644 index 000000000..421f3ed42 --- /dev/null +++ b/pkg/sbpf/fastmem.go @@ -0,0 +1,59 @@ +package sbpf + +import "unsafe" + +// memRegion describes one of the fixed 4 GiB virtual address windows +// (rodata, stack, heap, input) as a contiguous host buffer, so that the +// interpreter can translate the common case without a function call. +// +// rlen / wlen are the readable / writable byte lengths (wlen == 0 for +// read-only windows). gapShift/gapMask implement SBPF v0 stack frame gaps +// exactly like Agave's MemoryRegion::vm_gap_shift (gapShift = 63 and +// gapMask = 0 for windows without gaps, which makes the gap logic a no-op). +// A window that needs special handling (VASA input regions, ...) has +// rlen = wlen = 0 and falls back to translateInternal. +type memRegion struct { + base unsafe.Pointer // host address of window offset `start` + start uint64 // offset of the region within its 4 GiB window + rlen uint64 + wlen uint64 + gapShift uint64 + gapMask uint64 + dirty uint64 // 4 KiB page bitmap of fast-path writes (stack/heap only matter) +} + +// emptyRegion never matches any access. +var emptyRegion = memRegion{gapShift: 63} + +const numFastRegions = 6 // index 5 is a permanently empty catch-all + +// fastRead returns a host pointer for a size-byte read at vma, or nil if the +// access is not covered by the fast path (caller falls back to Read*). +func (ip *Interpreter) fastRead(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + // Truncating to 32 bits makes lo < start wrap to a value >= 2^32-start, + // which is always > rlen (start+rlen < 2^32), so one compare suffices. + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.rlen || inGap != 0 { + return nil + } + return unsafe.Add(reg.base, off) +} + +// fastWrite is the write counterpart of fastRead; it also records the dirty +// range so Finish only needs to zero what was touched. +func (ip *Interpreter) fastWrite(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.wlen || inGap != 0 { + return nil + } + // Mark the 4 KiB page (and, conservatively, the next one, since an access + // is at most 8 bytes and may straddle a page boundary) as dirty. + reg.dirty |= 3 << (off >> 12) + return unsafe.Add(reg.base, off) +} diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 5f351a5d6..e7e28fda0 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1,6 +1,7 @@ package sbpf import ( + "errors" "fmt" "math" "math/bits" @@ -46,6 +47,31 @@ type Interpreter struct { sbpfVersion sbpfver.SbpfVersion programId solana.PublicKey txSignature solana.Signature + + // callTargets[pc] is the resolved internal-function target of the `call imm` + // at pc (or -1). Computed once per Program at load time. + callTargets []int64 + + // Fast-path translation table indexed by (vaddr >> 32); see fastmem.go. + regions [numFastRegions]memRegion + // dirtyLo/dirtyHi: byte range written through translateInternal; + // memRegion.dirty: 4 KiB page bitmap of writes through the fast path. + dirtyLo [numFastRegions]uint64 + dirtyHi [numFastRegions]uint64 +} + +// dirtyRange returns the union of the byte ranges that may have been written +// in window idx (stack or heap), as [lo, hi). +func (ip *Interpreter) dirtyRange(idx uint64, size uint64) (lo, hi uint64) { + lo, hi = ip.dirtyLo[idx], ip.dirtyHi[idx] + if pages := ip.regions[idx].dirty; pages != 0 { + plo := uint64(bits.TrailingZeros64(pages)) << 12 + phi := uint64(bits.Len64(pages)) << 12 + lo = min(lo, plo) + hi = max(hi, phi) + } + hi = min(hi, size) + return lo, hi } type TraceSink interface { @@ -76,12 +102,13 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { heap = slices.Grow(heap, opts.HeapMax-len(heap)) } heap = heap[:opts.HeapMax] - clear(heap) + // Buffers in the pool are zeroed (for their dirty range) in Finish, so + // no clear is needed here. } else { heap = newHeap() } - return &Interpreter{ + ip := &Interpreter{ textVA: p.TextVA, textBytes: p.TextBytes, text: p.Text, @@ -103,17 +130,56 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { sbpfVersion: p.SbpfVersion, programId: opts.ProgramId, txSignature: opts.TxSignature, + callTargets: p.CallTargets, + } + ip.initRegions() + return ip +} + +// initRegions fills the fast-path translation table. Windows that need the +// full logic in translateInternal are left empty (rlen = wlen = 0). +func (ip *Interpreter) initRegions() { + for i := range ip.dirtyLo { + ip.dirtyLo[i] = math.MaxUint64 + ip.dirtyHi[i] = 0 + ip.regions[i].gapShift = 63 + } + if len(ip.ro) != 0 { + idx := VaddrProgram >> 32 + if ip.sbpfVersion.EnableLowerRodataVaddr() { + idx = 0 + } + ip.regions[idx] = memRegion{base: unsafe.Pointer(&ip.ro[0]), rlen: uint64(len(ip.ro)), gapShift: 63} + } + if len(ip.stack.mem) != 0 { + r := memRegion{base: unsafe.Pointer(&ip.stack.mem[0]), rlen: StackMax, wlen: StackMax, gapShift: 63} + if ip.stack.stackFrameGaps { + r.gapShift = 12 // log2(StackFrameSize) + r.gapMask = GapMask + } + ip.regions[VaddrStack>>32] = r + } + if len(ip.heap) != 0 { + ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} + } + if len(ip.inputRegions) == 0 && len(ip.input) != 0 { + ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } } func (ip *Interpreter) Finish() { if UsePool { + lo, hi := ip.dirtyRange(VaddrHeap>>32, uint64(len(ip.heap))) + if hi > lo { + clear(ip.heap[lo:hi]) + } heapPool.Put(ip.heap) } + ip.stack.MarkDirty(ip.dirtyRange(VaddrStack>>32, StackMax)) ip.stack.Finish() } -func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, error) { +func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[16]uint64) (int64, error) { var taken bool dst := uint32(r[ins.Dst()]) src := uint32(r[ins.Src()]) @@ -200,7 +266,7 @@ func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, e // // This function may panic given code that doesn't pass the static verifier. func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { - var r [11]uint64 + var r [16]uint64 // 16 (not 11) so that r[ins.Dst()] (4-bit field) needs no bounds check r[1] = VaddrInput r[2] = ip.inputDataVaddr @@ -216,137 +282,228 @@ func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { // initialize pc to program entry point pc := int64(ip.entry) + // Loop-invariant state hoisted into locals so the compiler can keep them in + // registers (fields of ip may alias with the unsafe stores in the loop and + // would otherwise be reloaded on every instruction). + text := ip.text + tracing := ip.enableTracing + jmp32 := ip.sbpfVersion.EnableJmp32() + moveMem := ip.sbpfVersion.MoveMemoryInstructionClasses() + pqr := ip.sbpfVersion.EnablePqr() + staticSyscalls := ip.sbpfVersion.EnableStaticSyscalls() + callTargets := ip.callTargets + + // Instruction metering (mirrors Agave's due_insn_count / previous_instruction_meter): + // count executed instructions locally and only sync with the shared compute + // meter around syscalls and on exit. `budget` is the number of instructions + // we may still execute before the meter would be exhausted. + meter := ip.computeMeter + var budget, due uint64 + reloadBudget := func() { + if meter.Disabled() { + budget = math.MaxUint64 + } else { + budget = meter.Remaining() + } + due = 0 + // A syscall (CPI in particular) may have changed the input regions; + // drop the cached input-region fast path entry, it is re-populated on + // the next slow-path translation. (With a plain, region-less input the + // entry is static and stays.) + if len(ip.inputRegions) != 0 { + ip.regions[VaddrInput>>32] = emptyRegion + } + } + flushDue := func() { + if due != 0 { + _ = meter.Consume(due) + due = 0 + } + } + reloadBudget() + mainLoop: for i := 0; true; i++ { // Fetch - if pc < 0 || pc >= int64(len(ip.text)) { + if pc < 0 || pc >= int64(len(text)) { + flushDue() return 0, 0, &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, ExcExecutionOverrun), } } - ins := ip.getSlot(pc) - if ip.enableTracing { + ins := text[pc] + if tracing { regsDump := fmt.Sprintf("%016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x", r[0], r[1], r[2], r[3], r[4], r[5], r[6], r[7], r[8], r[9], r[10]) fmt.Printf("% 5d [%s]: %s\n", i, strings.ToUpper(regsDump), ip.disassemble(ins, 0)) } - err = ip.computeMeter.Consume(1) - if err != nil { + // Meter: identical semantics to Consume(1) before each instruction. + if due == budget { + err = cu.ErrComputeExceeded break mainLoop } + due++ // Execute - if ip.sbpfVersion.EnableJmp32() && ins.Op()&0x07 == ClassPqr { + if jmp32 && ins.Op()&0x07 == ClassPqr { pc, err = ip.executeJmp32(ins, pc, &r) goto postExecute } switch ins.Op() { case OpLdxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 1); p != nil { + r[ins.Dst()] = uint64(*(*uint8)(p)) + } else { + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 2); p != nil { + r[ins.Dst()] = uint64(*(*uint16)(p)) + } else { + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 4); p != nil { + r[ins.Dst()] = uint64(*(*uint32)(p)) + } else { + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v + if p := ip.fastRead(vma, 8); p != nil { + r[ins.Dst()] = uint64(*(*uint64)(p)) + } else { + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpStb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(ins.Uimm()) + } else { + err = ip.Write8(vma, uint8(ins.Uimm())) + } pc++ case OpSth: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(ins.Uimm()) + } else { + err = ip.Write16(vma, uint16(ins.Uimm())) + } pc++ case OpStw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(ins.Uimm()) + } else { + err = ip.Write32(vma, uint32(ins.Uimm())) + } pc++ case OpStdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(ins.Imm()) + } else { + err = ip.Write64(vma, uint64(ins.Imm())) + } pc++ case OpStxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(r[ins.Src()]) + } else { + err = ip.Write8(vma, uint8(r[ins.Src()])) + } pc++ case OpStxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(r[ins.Src()]) + } else { + err = ip.Write16(vma, uint16(r[ins.Src()])) + } pc++ case OpStxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(r[ins.Src()]) + } else { + err = ip.Write32(vma, uint32(r[ins.Src()])) + } pc++ case OpStxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(r[ins.Src()]) + } else { + err = ip.Write64(vma, uint64(r[ins.Src()])) + } pc++ case OpAdd32Imm: r[ins.Dst()] = ip.signExtension(int32(r[ins.Dst()]) + ins.Imm()) @@ -383,343 +540,6 @@ mainLoop: case OpMul32Imm: r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * ins.Imm()) pc++ - case OpMul32Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd1BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpMul64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) - pc++ - } - case OpMul64Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= r[ins.Src()] - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) - pc++ - } - case OpDiv32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpDiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd2BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpLd4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) - pc++ - case OpDiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] /= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) - pc++ - } - case OpDiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) - pc++ - } - case OpSt4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) - pc++ - case OpLmul32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) - pc++ - case OpLmul32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) - pc++ - case OpLmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= uint64(int64(ins.Imm())) - pc++ - case OpLmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= r[ins.Src()] - pc++ - case OpUhmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) - r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() - pc++ - case OpUhmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) - r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() - pc++ - case OpShmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - imm128 := wide.Int128FromInt64(int64(ins.Imm())) - r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() - pc++ - case OpShmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - src128 := wide.Int128FromInt64(int64(r[ins.Src()])) - r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() - pc++ - case OpUdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpUdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] /= uint64(ins.Uimm()) - pc++ - case OpUdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpUrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] = uint64(r[ins.Dst()] % src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] %= uint64(ins.Uimm()) - pc++ - case OpUrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= r[ins.Src()] - } else { - err = ExcDivideByZero - } - pc++ - case OpSdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) - pc++ - case OpSdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) - pc++ - case OpSdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) - pc++ - case OpSrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) - pc++ - case OpSrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) - } else { - err = ExcDivideByZero - break - } - pc++ case OpOr32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) | ins.Uimm()) pc++ @@ -768,66 +588,6 @@ mainLoop: case OpRsh64Reg: r[ins.Dst()] >>= r[ins.Src()] & 0x3f pc++ - case OpNeg32: - if ip.sbpfVersion.DisableNeg() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) - pc++ - case OpNeg64: - if !ip.sbpfVersion.DisableNeg() { - r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt4BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) - pc++ - } - case OpMod32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpMod32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd8BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v - pc++ - } - case OpMod64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] %= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) - pc++ - } - case OpMod64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) - pc++ - } case OpXor32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) ^ ins.Uimm()) pc++ @@ -856,53 +616,6 @@ mainLoop: case OpMov64Reg: r[ins.Dst()] = r[ins.Src()] pc++ - case OpArsh32Imm: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) - pc++ - case OpArsh32Reg: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) - pc++ - case OpArsh64Imm: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) - pc++ - case OpArsh64Reg: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) - pc++ - case OpHor64Imm: - if !ip.sbpfVersion.DisableLddw() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] |= uint64(ins.Uimm()) << 32 - pc++ - case OpLe: - if ip.sbpfVersion.DisableLe() { - err = ExcInvalidInstr - break - } - switch ins.Uimm() { - case 16: - r[ins.Dst()] &= math.MaxUint16 - case 32: - r[ins.Dst()] &= math.MaxUint32 - case 64: - r[ins.Dst()] &= math.MaxUint64 - default: - err = ExcUnsupportedInstruction - } - pc++ - case OpBe: - switch ins.Uimm() { - case 16: - r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) - case 32: - r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) - case 64: - r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) - default: - err = ExcUnsupportedInstruction - } - pc++ case OpLddw: if ip.sbpfVersion.DisableLddw() { err = ExcInvalidInstr @@ -1024,14 +737,16 @@ mainLoop: } pc++ case OpCall: - if ip.sbpfVersion.EnableStaticSyscalls() { + if staticSyscalls { if ins.Src() == 0 { sc, ok := ip.syscalls(ins.Uimm()) if !ok { err = ExcCallDest{ins.Uimm()} break } + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } @@ -1042,7 +757,7 @@ mainLoop: err = ExcCallDest{uint32(targetPC)} break } - if ok := ip.stack.Push(r[:], pc+1); !ok { + if ok := ip.stack.Push(&r, pc+1); !ok { err = ExcCallDepth } pc = targetPC @@ -1051,60 +766,147 @@ mainLoop: } } else { if sc, ok := ip.syscalls(ins.Uimm()); ok { + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } pc++ - } else if target, ok := ip.funcs[ins.Uimm()]; ok { - ok = ip.stack.Push(r[:], pc+1) + } else { + var target int64 + var ok bool + if callTargets != nil { + target = callTargets[pc] + ok = target >= 0 + } else { + target, ok = ip.funcs[ins.Uimm()] + } if !ok { + err = ExcCallDest{ins.Uimm()} + break + } + if !ip.stack.Push(&r, pc+1) { err = ExcCallDepth } pc = target - } else { - err = ExcCallDest{ins.Uimm()} } } - case OpCallx: - var target uint64 - if ip.sbpfVersion.CallXUsesSrcReg() { - target = r[ins.Src()] - } else if ip.sbpfVersion.CallXUsesDstReg() { - target = r[ins.Dst()] + case OpExit: + var ok bool + pc, ok = ip.stack.Pop(&r) + if !ok { + ret = r[0] + break mainLoop + } + case OpMul32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) + pc++ + case OpMul64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= uint64(ins.Imm()) + pc++ + case OpMul64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpDiv32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) } else { - target = r[ins.Uimm()] + err = ExcDivideByZero } - - if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { - err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + pc++ + case OpDiv64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - targetPC := int64((target - ip.textVA) / 8) - if ok := ip.stack.Push(r[:], pc+1); !ok { - err = ExcCallDepth + r[ins.Dst()] /= uint64(ins.Imm()) + pc++ + case OpDiv64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - pc = targetPC - case OpExit: - var ok bool - pc, ok = ip.stack.Pop(r[:]) - if !ok { - ret = r[0] - break mainLoop + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero } + pc++ + case OpMod32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpMod64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] %= uint64(ins.Imm()) + pc++ + case OpMod64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= src + } else { + err = ExcDivideByZero + } + pc++ + case OpArsh32Imm: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) + pc++ + case OpArsh32Reg: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) + pc++ + case OpArsh64Imm: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) + pc++ + case OpArsh64Reg: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) + pc++ default: - err = ExcUnsupportedInstruction - return + pc, err = ip.executeCold(ins, pc, &r) + if err == errUnknownOpcode { + // Preserve the original behaviour for an unknown opcode: + // a bare (unwrapped) ExcUnsupportedInstruction. + flushDue() + return 0, 0, ExcUnsupportedInstruction + } } // Post execute postExecute: - if err == cu.ErrComputeExceeded { - err = ExcOutOfCU - } - if err != nil { + flushDue() + if err == cu.ErrComputeExceeded { + err = ExcOutOfCU + } exc := &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, err), @@ -1117,6 +919,9 @@ mainLoop: } } + flushDue() + // NB: when the loop exits because the meter is exhausted, err is the bare + // cu.ErrComputeExceeded (not wrapped in an Exception), as before. cuConsumed = ip.initialInstrMeter - ip.computeMeter.Remaining() return @@ -1198,6 +1003,11 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } + if write { + off := StackMax - uint64(len(mem)) + ip.dirtyLo[VaddrStack>>32] = min(ip.dirtyLo[VaddrStack>>32], off) + ip.dirtyHi[VaddrStack>>32] = max(ip.dirtyHi[VaddrStack>>32], off+size) + } return unsafe.Pointer(&mem[0]), nil case VaddrHeap >> 32: if size == 0 { @@ -1206,6 +1016,10 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } + if write { + ip.dirtyLo[VaddrHeap>>32] = min(ip.dirtyLo[VaddrHeap>>32], lo) + ip.dirtyHi[VaddrHeap>>32] = max(ip.dirtyHi[VaddrHeap>>32], lo+size) + } return unsafe.Pointer(&ip.heap[lo]), nil case VaddrInput >> 32: if size == 0 { @@ -1255,6 +1069,8 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } if write && (!region.Writable || requestedLen > region.RegionSize) && region.OnWrite != nil { + // The callback may replace region.Data / grow the region: drop the cache. + ip.regions[VaddrInput>>32] = emptyRegion if err := region.OnWrite(region, requestedLen); err != nil { return nil, err } @@ -1263,23 +1079,37 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un if !write || !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } + ip.regions[VaddrInput>>32] = emptyRegion region.RegionSize = region.AddressSpaceReserved } if write && !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "write to readonly input region") } + var base unsafe.Pointer if region.Data != nil { if requestedLen > uint64(len(region.Data)) { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } - return unsafe.Pointer(®ion.Data[regionOffset]), nil + base = unsafe.Pointer(unsafe.SliceData(region.Data)) + } else { + hostOffset := region.HostOffset + regionOffset + if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { + return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + } + base = unsafe.Pointer(&ip.input[region.HostOffset]) } - - hostOffset := region.HostOffset + regionOffset - if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { - return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + // Cache this region for the interpreter's fast path (one-entry cache, + // same idea as Agave's MappingCache). Only the currently mapped + // RegionSize bytes are exposed; anything beyond takes the slow path + // again so that OnWrite / growth semantics are preserved. + if region.RegionSize != 0 && (region.Data == nil || uint64(len(region.Data)) >= region.RegionSize) { + cached := memRegion{base: base, start: region.Offset, rlen: region.RegionSize, gapShift: 63} + if region.Writable { + cached.wlen = region.RegionSize + } + ip.regions[VaddrInput>>32] = cached } - return unsafe.Pointer(&ip.input[hostOffset]), nil + return unsafe.Add(base, regionOffset), nil } func (ip *Interpreter) TranslateInput(addr uint64, size uint64) ([]byte, error) { @@ -1335,6 +1165,7 @@ func (ip *Interpreter) SetInputRegionData(addr uint64, data []byte, length uint6 } region.RegionSize = length region.Writable = writable + ip.regions[VaddrInput>>32] = emptyRegion return true } @@ -1461,3 +1292,360 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { *(*uint64)(ptr) = x return nil } + +// executeCold handles the less frequently executed opcodes. Keeping them out of +// Run keeps that function below the compiler's "big function" threshold so the +// hot helpers (metering, fast memory translation, stack push/pop) stay inlinable. +func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { + var err error + switch ins.Op() { + case OpDiv32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpLd4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, uint32(r[ins.Src()])) + pc++ + case OpLmul32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) + pc++ + case OpLmul32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) + pc++ + case OpLmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= uint64(int64(ins.Imm())) + pc++ + case OpLmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpUhmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) + r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() + pc++ + case OpUhmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) + r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() + pc++ + case OpShmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + imm128 := wide.Int128FromInt64(int64(ins.Imm())) + r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() + pc++ + case OpShmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + src128 := wide.Int128FromInt64(int64(r[ins.Src()])) + r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() + pc++ + case OpUdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpUdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] /= uint64(ins.Uimm()) + pc++ + case OpUdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpUrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] = uint64(r[ins.Dst()] % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] %= uint64(ins.Uimm()) + pc++ + case OpUrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= r[ins.Src()] + } else { + err = ExcDivideByZero + } + pc++ + case OpSdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) + pc++ + case OpSdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) + pc++ + case OpSdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) + pc++ + case OpSrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) + pc++ + case OpSrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpNeg32: + if ip.sbpfVersion.DisableNeg() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) + pc++ + case OpNeg64: + if !ip.sbpfVersion.DisableNeg() { + r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) + pc++ + } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { + // OpSt4BImm + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, ins.Uimm()) + pc++ + } + case OpMod32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpHor64Imm: + if !ip.sbpfVersion.DisableLddw() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] |= uint64(ins.Uimm()) << 32 + pc++ + case OpLe: + if ip.sbpfVersion.DisableLe() { + err = ExcInvalidInstr + break + } + switch ins.Uimm() { + case 16: + r[ins.Dst()] &= math.MaxUint16 + case 32: + r[ins.Dst()] &= math.MaxUint32 + case 64: + r[ins.Dst()] &= math.MaxUint64 + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpBe: + switch ins.Uimm() { + case 16: + r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) + case 32: + r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) + case 64: + r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpCallx: + var target uint64 + if ip.sbpfVersion.CallXUsesSrcReg() { + target = r[ins.Src()] + } else if ip.sbpfVersion.CallXUsesDstReg() { + target = r[ins.Dst()] + } else { + target = r[ins.Uimm()] + } + + if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { + err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + break + } + targetPC := int64((target - ip.textVA) / 8) + if ok := ip.stack.Push(r, pc+1); !ok { + err = ExcCallDepth + break + } + pc = targetPC + default: + err = errUnknownOpcode + } + return pc, err +} + +// errUnknownOpcode is an internal sentinel returned by executeCold for an +// opcode that is not handled by either switch; Run turns it into the bare +// ExcUnsupportedInstruction return of the original implementation. +var errUnknownOpcode = errors.New("unknown opcode") diff --git a/pkg/sbpf/loader/loader.go b/pkg/sbpf/loader/loader.go index 7b9dc52ca..23a26ec76 100644 --- a/pkg/sbpf/loader/loader.go +++ b/pkg/sbpf/loader/loader.go @@ -162,7 +162,7 @@ func parseSlots(bs []byte) []sbpf.Slot { } func (l *Loader) getProgram() *sbpf.Program { - return &sbpf.Program{ + p := &sbpf.Program{ RO: l.program, TextBytes: l.text, Text: parseSlots(l.text), @@ -171,4 +171,6 @@ func (l *Loader) getProgram() *sbpf.Program { Funcs: l.funcs, SbpfVersion: l.sbpfVersion(), } + p.ResolveCallTargets() + return p } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index 38d1028f8..c4bc5eb0d 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -30,14 +30,28 @@ func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { t.Error("pooled VM exposed data from an earlier execution") } - parent.heap[0] = byte(worker + 1) - parent.stack.mem[0] = byte(worker + 1) + // Write through the VM's translation layer, as programs and + // syscalls do: the pool only re-zeroes memory the VM saw written. + if err := parent.Write8(VaddrHeap, byte(worker+1)); err != nil { + t.Error(err) + } + if err := parent.Write8(VaddrStack, byte(worker+1)); err != nil { + t.Error(err) + } child := poolingInterpreter(256 * 1024) - for j := range child.heap { - child.heap[j] = 0xab + childHeap, err := child.Translate(VaddrHeap, uint64(len(child.heap)), true) + if err != nil { + t.Error(err) + } + for j := range childHeap { + childHeap[j] = 0xab + } + childStack, err := child.Translate(VaddrStack, StackMax, true) + if err != nil { + t.Error(err) } - for j := range child.stack.mem { - child.stack.mem[j] = 0xcd + for j := range childStack { + childStack[j] = 0xcd } child.Finish() if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index c112be14d..969a15469 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -13,6 +13,29 @@ type Program struct { Entrypoint uint64 // PC Funcs map[uint32]int64 SbpfVersion sbpfver.SbpfVersion + + // CallTargets[pc] holds the resolved internal function target for a + // `call imm` slot at pc (non-static-syscall versions), or -1. + CallTargets []int64 +} + +// ResolveCallTargets precomputes CallTargets from Funcs so the interpreter +// does not need a map lookup per call instruction. +func (p *Program) ResolveCallTargets() { + if p.SbpfVersion.EnableStaticSyscalls() { + p.CallTargets = nil + return + } + targets := make([]int64, len(p.Text)) + for pc, slot := range p.Text { + targets[pc] = -1 + if slot.Op() == OpCall { + if t, ok := p.Funcs[slot.Uimm()]; ok { + targets[pc] = t + } + } + } + p.CallTargets = targets } func (p *Program) MemoryBytes() uint64 { diff --git a/pkg/sbpf/stack.go b/pkg/sbpf/stack.go index 3f3331b6e..edc662c18 100644 --- a/pkg/sbpf/stack.go +++ b/pkg/sbpf/stack.go @@ -36,6 +36,12 @@ type Stack struct { shadow []Frame dynamicStackFrames bool stackFrameGaps bool + // dirtyLo/dirtyHi bound the physical byte range of mem that may have been + // written during this execution. Finish only has to zero this range before + // returning the buffer to the pool. + dirtyLo uint64 + dirtyHi uint64 + maxDepth int } // Frame is an entry on the shadow stack. @@ -92,9 +98,11 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { } s := Stack{ - mem: m, - sp: VaddrStack, - shadow: sh, + mem: m, + sp: VaddrStack, + shadow: sh, + dirtyLo: StackMax, + dirtyHi: 0, } var sz uint64 @@ -115,10 +123,12 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { func (s *Stack) Finish() { if UsePool { s.mem = s.mem[:StackMax] - clear(s.mem) + if s.dirtyHi > s.dirtyLo { + clear(s.mem[s.dirtyLo:s.dirtyHi]) + } stackMemPool.Put(s.mem) s.shadow = s.shadow[:StackDepth] - clear(s.shadow) + clear(s.shadow[:max(s.maxDepth, 1)]) s.shadow = s.shadow[:1] stackShadowPool.Put(s.shadow) } @@ -153,21 +163,38 @@ func (s *Stack) GetFrame(addr uint32) []byte { } } +// MarkDirty records that physical stack bytes [off, off+size) may be written. +func (s *Stack) MarkDirty(lo, hi uint64) { + if hi <= lo { + return + } + s.dirtyLo = min(s.dirtyLo, lo) + s.dirtyHi = max(s.dirtyHi, hi) +} + // Push allocates a new call frame. // // Saves the given nonvolatile regs, return address, // and current frame pointer. // Returns the new frame pointer. -func (s *Stack) Push(regs []uint64, ret int64) bool { - if ok := len(s.shadow) < cap(s.shadow); !ok { +func (s *Stack) Push(regs *[16]uint64, ret int64) bool { + n := len(s.shadow) + if n >= cap(s.shadow) { return false } - - frame := Frame{RetAddr: ret} - copy(frame.NVRegs[:], regs[6:10]) - frame.FramePtr = regs[10] - - s.shadow = append(s.shadow, frame) + // Write the frame in place (no temporary Frame value / copy) to avoid + // store-forwarding stalls in this very hot path. + s.shadow = s.shadow[:n+1] + f := &s.shadow[n] + f.RetAddr = ret + f.NVRegs[0] = regs[6] + f.NVRegs[1] = regs[7] + f.NVRegs[2] = regs[8] + f.NVRegs[3] = regs[9] + f.FramePtr = regs[10] + if n+1 > s.maxDepth { + s.maxDepth = n + 1 + } if !s.dynamicStackFrames { if s.stackFrameGaps { @@ -185,16 +212,17 @@ func (s *Stack) Push(regs []uint64, ret int64) bool { // Restores saved nonvolatile regs into provided slice. // Returns saved return address and returns true upon success, // and returns false if no call frames are left. -func (s *Stack) Pop(regs []uint64) (int64, bool) { - if len(s.shadow) <= 1 { +func (s *Stack) Pop(regs *[16]uint64) (int64, bool) { + n := len(s.shadow) + if n <= 1 { return 0, false } - - var frame Frame - frame, s.shadow = s.shadow[len(s.shadow)-1], s.shadow[:len(s.shadow)-1] - - copy(regs[6:10], frame.NVRegs[:]) - regs[10] = frame.FramePtr - - return frame.RetAddr, true + f := &s.shadow[n-1] + regs[6] = f.NVRegs[0] + regs[7] = f.NVRegs[1] + regs[8] = f.NVRegs[2] + regs[9] = f.NVRegs[3] + regs[10] = f.FramePtr + s.shadow = s.shadow[:n-1] + return f.RetAddr, true } From faa0cb0285b4b98f38eba08bb14de4069a39596d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 096/111] sbpf: add interpreter benchmarks and a differential test corpus - perf_bench_test.go: synthetic ALU / load-store / call loops and interpreter setup+teardown - loader/token_perf_bench_test.go: real SPL Token Transfer through the loader/verifier/interpreter with sealevel-equivalent syscalls, in the aligned and VASA input layouts - perf_differential_test.go: deterministic random program corpus; run on two builds with SBPF_DIFF_OUT= and diff the outputs; SBPF_CHECK_POOL_ZERO=1 asserts pooled buffers come back zeroed Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sbpf/loader/token_perf_bench_test.go | 425 +++++++++++++++++++++++ pkg/sbpf/perf_bench_test.go | 193 ++++++++++ pkg/sbpf/perf_differential_test.go | 338 ++++++++++++++++++ 3 files changed, 956 insertions(+) create mode 100644 pkg/sbpf/loader/token_perf_bench_test.go create mode 100644 pkg/sbpf/perf_bench_test.go create mode 100644 pkg/sbpf/perf_differential_test.go diff --git a/pkg/sbpf/loader/token_perf_bench_test.go b/pkg/sbpf/loader/token_perf_bench_test.go new file mode 100644 index 000000000..c2534e586 --- /dev/null +++ b/pkg/sbpf/loader/token_perf_bench_test.go @@ -0,0 +1,425 @@ +package loader_test + +import ( + "encoding/binary" + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" +) + +// ---- minimal syscall set mirroring pkg/sealevel semantics (CU costs + memory behaviour) ---- + +const ( + cuSyscallBase = 100 + cuMemOpBase = 10 + cuCpiBytesPerCU = 250 +) + +type stats struct { + logs int + memcpy int + memset int + memcmp int + memmove int + bytes uint64 +} + +var st stats + +func memOpConsume(vm sbpf.VM, n uint64) error { + cost := max(uint64(cuMemOpBase), n/cuCpiBytesPerCU) + return vm.ComputeMeter().Consume(cost) +} + +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { + srcBuf := make([]byte, n) // same allocation pattern as sealevel/syscalls_mem.go + err = vm.Read(src, srcBuf) + if err != nil { + return + } + err = vm.Write(dst, srcBuf) + return +} + +func isNonOverlapping(src, srcLen, dst, dstLen uint64) bool { + if src > dst { + return src-dst >= dstLen + } + return dst-src >= srcLen +} + +var syscallMemcpy = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memcpy++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + if !isNonOverlapping(src, n, dst, n) { + return 0, errors.New("overlapping") + } + if n == 0 { + return 0, nil + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemmove = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memmove++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemcmp = sbpf.SyscallFunc4(func(vm sbpf.VM, a1, a2, n, res uint64) (uint64, error) { + st.memcmp++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + s1, err := vm.Translate(a1, n, false) + if err != nil { + return 0, err + } + s2, err := vm.Translate(a2, n, false) + if err != nil { + return 0, err + } + r := int32(0) + for i := uint64(0); i < n; i++ { + if s1[i] != s2[i] { + r = int32(s1[i]) - int32(s2[i]) + break + } + } + out, err := vm.Translate(res, 4, true) + if err != nil { + return 0, err + } + binary.LittleEndian.PutUint32(out, uint32(r)) + return 0, nil +}) + +var syscallMemset = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, c, n uint64) (uint64, error) { + st.memset++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + mem, err := vm.Translate(dst, n, true) + if err != nil { + return 0, err + } + for i := uint64(0); i < n; i++ { + mem[i] = byte(c) + } + return 0, nil +}) + +var syscallLog = sbpf.SyscallFunc2(func(vm sbpf.VM, ptr, strlen uint64) (uint64, error) { + st.logs++ + if err := vm.ComputeMeter().Consume(max(uint64(cuSyscallBase), strlen)); err != nil { + return 0, err + } + buf := make([]byte, strlen) + if err := vm.Read(ptr, buf); err != nil { + return 0, err + } + _ = string(buf) + return 0, nil +}) + +var syscallLog64 = sbpf.SyscallFunc5(func(vm sbpf.VM, a, b, c, d, e uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogPubkey = sbpf.SyscallFunc1(func(vm sbpf.VM, a uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogCUs = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallAbort = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { return 0, errors.New("abort") }) +var syscallPanic = sbpf.SyscallFunc4(func(vm sbpf.VM, f, l, line, col uint64) (uint64, error) { + return 0, errors.New("panic") +}) +var syscallAllocFree = sbpf.SyscallFunc2(func(vm sbpf.VM, size, free uint64) (uint64, error) { + if free != 0 { + return 0, nil + } + hs := (vm.HeapSize() + 7) &^ 7 + addr := sbpf.VaddrHeap + hs + hs += size + if hs > vm.HeapMax() { + return 0, nil + } + vm.UpdateHeapSize(hs) + return addr, nil +}) + +var registry = map[uint32]sbpf.Syscall{ + sbpf.SymbolHash("abort"): syscallAbort, + sbpf.SymbolHash("sol_panic_"): syscallPanic, + sbpf.SymbolHash("sol_log_"): syscallLog, + sbpf.SymbolHash("sol_log_64_"): syscallLog64, + sbpf.SymbolHash("sol_log_pubkey"): syscallLogPubkey, + sbpf.SymbolHash("sol_log_compute_units_"): syscallLogCUs, + sbpf.SymbolHash("sol_memcpy_"): syscallMemcpy, + sbpf.SymbolHash("sol_memmove_"): syscallMemmove, + sbpf.SymbolHash("sol_memcmp_"): syscallMemcmp, + sbpf.SymbolHash("sol_memset_"): syscallMemset, + sbpf.SymbolHash("sol_alloc_free_"): syscallAllocFree, +} + +var syscalls = sbpf.SyscallRegistry(func(h uint32) (sbpf.Syscall, bool) { + s, ok := registry[h] + return s, ok +}) + +// ---- aligned input serialization (BPF loader v2/v3 format, no direct mapping) ---- + +const maxPermittedDataIncrease = 10 * 1024 + +type acct struct { + key, owner [32]byte + lamports uint64 + data []byte + signer, writable bool +} + +func serializeAligned(accts []acct, instrData []byte, programId [32]byte) []byte { + out := binary.LittleEndian.AppendUint64(nil, uint64(len(accts))) + for _, a := range accts { + out = append(out, 0xff) + out = append(out, b2u8(a.signer), b2u8(a.writable), 0) + out = append(out, 0, 0, 0, 0) // original_data_len + out = append(out, a.key[:]...) + out = append(out, a.owner[:]...) + out = binary.LittleEndian.AppendUint64(out, a.lamports) + out = binary.LittleEndian.AppendUint64(out, uint64(len(a.data))) + out = append(out, a.data...) + pad := maxPermittedDataIncrease + ((8 - len(a.data)%8) % 8) + out = append(out, make([]byte, pad)...) + out = binary.LittleEndian.AppendUint64(out, ^uint64(0)) // rent epoch + } + out = binary.LittleEndian.AppendUint64(out, uint64(len(instrData))) + out = append(out, instrData...) + out = append(out, programId[:]...) + return out +} + +func b2u8(b bool) byte { + if b { + return 1 + } + return 0 +} + +// SPL token account layout (165 bytes) +func tokenAccount(mint, owner [32]byte, amount uint64) []byte { + d := make([]byte, 165) + copy(d[0:32], mint[:]) + copy(d[32:64], owner[:]) + binary.LittleEndian.PutUint64(d[64:72], amount) + // delegate: COption none (4 bytes 0) + 32 + d[108] = 1 // state = Initialized + // is_native COption none, delegated_amount 0, close_authority none + return d +} + +func key(b byte) [32]byte { + var k [32]byte + for i := range k { + k[i] = b + } + return k +} + +func loadTokenProgram(tb testing.TB) *sbpf.Program { + elfBytes := fixtures.Load(tb, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + tb.Fatal(err) + } + p, err := l.Load() + if err != nil { + tb.Fatal(err) + } + if err := p.Verify(); err != nil { + tb.Fatal(err) + } + return p +} + +func transferInput(programId [32]byte) ([]byte, []acct) { + mint := key(0x11) + authority := key(0x22) + src := key(0x33) + dst := key(0x44) + accts := []acct{ + {key: src, owner: programId, lamports: 2039280, data: tokenAccount(mint, authority, 1_000_000), writable: true}, + {key: dst, owner: programId, lamports: 2039280, data: tokenAccount(mint, key(0x55), 5), writable: true}, + {key: authority, owner: key(0), lamports: 1_000_000_000, data: nil, signer: true}, + } + instr := append([]byte{3}, binary.LittleEndian.AppendUint64(nil, 1000)...) + return serializeAligned(accts, instr, programId), accts +} + +func runTransfer(tb testing.TB, p *sbpf.Program, input []byte) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransfer(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, _ := transferInput(programId) + st = stats{} + ret, used := runTransfer(t, p, input) + t.Logf("ret=%d cuUsed=%d stats=%+v", ret, used, st) + // verify balances changed in the serialized input + // account 0 data starts at 8 + 8 + 32+32+8+8 = 96 + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + off1 := 8 + (8 + 32 + 32 + 8 + 8 + 165 + maxPermittedDataIncrease + 3 + 8) + dstAmt := binary.LittleEndian.Uint64(input[off1+88+64:]) + t.Logf("src=%d dst=%d", srcAmt, dstAmt) + if ret != 0 || srcAmt != 999_000 || dstAmt != 1005 { + t.Fatalf("unexpected result ret=%d src=%d dst=%d", ret, srcAmt, dstAmt) + } +} + +func BenchmarkTokenTransfer(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, _ := transferInput(programId) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + var used uint64 + for i := 0; i < b.N; i++ { + copy(input, orig) + _, used = runTransfer(b, p, input) + } + b.StopTimer() + b.ReportMetric(float64(used), "cu/op") + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/float64(used), "ns/cu") +} + +func BenchmarkTokenLoadVerify(b *testing.B) { + elfBytes := fixtures.Load(b, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + b.ReportAllocs() + for i := 0; i < b.N; i++ { + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + b.Fatal(err) + } + p, err := l.Load() + if err != nil { + b.Fatal(err) + } + if err := p.Verify(); err != nil { + b.Fatal(err) + } + } +} + +// ---- VASA layout (VirtualAddressSpaceAdjustments active, direct mapping off) ---- +// Mirrors serializeParametersAligned with vasa=true, directMapping=false: +// same bytes as the aligned layout, but the input window is split into +// metadata regions and per-account data regions. + +func vasaRegions(accts []acct, instrLen int) []sbpf.InputRegion { + var regions []sbpf.InputRegion + var regionStart, hostRegionStart uint64 + vmOff := uint64(8) + for i, a := range accts { + l := vmOff // host offset == vm offset in this layout + dataLen := uint64(len(a.data)) + align := (8 - dataLen%8) % 8 + reserved := dataLen + maxPermittedDataIncrease + dataStart := vmOff + 88 + if dataStart > regionStart { + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: dataStart - regionStart, AddressSpaceReserved: dataStart - regionStart, Writable: true, AccountIndex: -1}) + } + regions = append(regions, sbpf.InputRegion{Offset: dataStart, HostOffset: l + 88, RegionSize: dataLen, + AddressSpaceReserved: reserved, Writable: a.writable, AccountIndex: i}) + hostRegionStart = l + 88 + reserved + regionStart = dataStart + reserved + vmOff += 88 + reserved + align + 8 + } + end := vmOff + 8 + uint64(instrLen) + 32 + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: end - regionStart, AddressSpaceReserved: end - regionStart, Writable: true, AccountIndex: -1}) + return regions +} + +func runTransferVasa(tb testing.TB, p *sbpf.Program, input []byte, regions []sbpf.InputRegion) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + // regions are mutated by the VM (RegionSize on growth), so copy per run + rc := append([]sbpf.InputRegion(nil), regions...) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + InputRegions: rc, + DisableStackFrameGaps: true, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransferVasa(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + if regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize != uint64(len(input)) { + t.Fatalf("region layout mismatch: %d vs %d", regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize, len(input)) + } + ret, used := runTransferVasa(t, p, input, regions) + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + if ret != 0 || srcAmt != 999_000 { + t.Fatalf("unexpected ret=%d src=%d", ret, srcAmt) + } + t.Logf("ret=%d cu=%d regions=%d", ret, used, len(regions)) +} + +func BenchmarkTokenTransferVasa(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + copy(input, orig) + runTransferVasa(b, p, input, regions) + } +} diff --git a/pkg/sbpf/perf_bench_test.go b/pkg/sbpf/perf_bench_test.go new file mode 100644 index 000000000..1d34d4d7a --- /dev/null +++ b/pkg/sbpf/perf_bench_test.go @@ -0,0 +1,193 @@ +package sbpf + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +func slot(op uint8, dst uint8, src uint8, off int16, imm uint32) Slot { + return Slot(op) | Slot(dst)<<8 | Slot(src)<<12 | Slot(uint16(off))<<16 | Slot(imm)<<32 +} + +func slotsToBytes(slots []Slot) []byte { + out := make([]byte, len(slots)*SlotSize) + for i, s := range slots { + binary.LittleEndian.PutUint64(out[i*SlotSize:], uint64(s)) + } + return out +} + +func mkProgram(text []Slot, ver uint32) *Program { + return &Program{ + TextBytes: slotsToBytes(text), + Text: text, + TextVA: VaddrProgram, + Entrypoint: 0, + Funcs: map[uint32]int64{}, + SbpfVersion: sbpfver.SbpfVersion{Version: ver}, + } +} + +var noSyscalls = SyscallRegistry(func(uint32) (Syscall, bool) { return nil, false }) + +// resolveCallTargetsIfSupported precomputes internal call targets on trees +// that have Program.ResolveCallTargets (the loader does this at load time); +// it is a no-op on the baseline tree so the same benchmark code runs on both. +func resolveCallTargetsIfSupported(p *Program) { + if r, ok := any(p).(interface{ ResolveCallTargets() }); ok { + r.ResolveCallTargets() + } +} + +// aluLoop: r1 = N; loop: r2 += r1; r2 ^= r3; r3 = r2; r3 *= 7; r3 >>= 3; r1 -= 1; jne r1,0 loop; exit +// 7 instructions per iteration. +func aluLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Imm, 3, 0, 0, 3), + // loop @3 + slot(OpAdd64Reg, 2, 1, 0, 0), + slot(OpXor64Reg, 2, 3, 0, 0), + slot(OpMov64Reg, 3, 2, 0, 0), + slot(OpMul64Imm, 3, 0, 0, 7), + slot(OpRsh64Imm, 3, 0, 0, 3), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -7, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// memLoop: writes and reads 8 byte values on the stack frame and heap. +// r1 = N; r4 = r10 - 4096 (frame base); r5 = heap base +// loop: stxdw [r4+0], r1; ldxdw r6, [r4+0]; add r2, r6; stxdw [r5+8], r2; ldxdw r7,[r5+8]; xor r2,r7 ; r1 -= 1; jne +func memLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Reg, 4, 10, 0, 0), + slot(OpAdd64Imm, 4, 0, 0, uint32(0xfffff000)), // r4 = r10 - 4096 + slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), + slot(0, 0, 0, 0, uint32(VaddrHeap>>32)), + // loop @6 + slot(OpStxdw, 4, 1, 0, 0), + slot(OpLdxdw, 6, 4, 0, 0), + slot(OpAdd64Reg, 2, 6, 0, 0), + slot(OpStxdw, 5, 2, 8, 0), + slot(OpLdxdw, 7, 5, 8, 0), + slot(OpXor64Reg, 2, 7, 0, 0), + slot(OpStxw, 5, 2, 16, 0), + slot(OpLdxb, 8, 5, 16, 0), + slot(OpAdd64Reg, 2, 8, 0, 0), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -11, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// callLoop: calls a tiny function N times (tests Push/Pop + call resolution) +func callLoopProgram(n uint32, ver uint32) *Program { + fnPC := int64(7) + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + // loop @2 + slot(OpCall, 0, 0, 0, 0), // patched below + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -3, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + // fn @7 + slot(OpAdd64Imm, 2, 0, 0, 3), + slot(OpXor64Reg, 2, 1, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + p := mkProgram(text, ver) + if ver >= sbpfver.SbpfVersionV3 { + // relative call: target = pc + imm + 1 ; pc=2 -> imm = 7-2-1 = 4 + text[2] = slot(OpCall, 0, 1, 0, uint32(fnPC-2-1)) + } else { + h := PCHash(uint64(fnPC)) + p.Funcs[h] = fnPC + text[2] = slot(OpCall, 0, 0, 0, h) + } + p.TextBytes = slotsToBytes(text) + return p +} + +func runProgram(b *testing.B, p *Program, input []byte, syscalls SyscallRegistry, budget uint64) uint64 { + cm := cu.NewComputeMeter(budget) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, _, err := ip.Run() + ip.Finish() + if err != nil { + b.Fatal(err) + } + return ret +} + +func benchLoop(b *testing.B, p *Program, insnsPerRun uint64) { + if err := p.Verify(); err != nil { + b.Fatal(err) + } + resolveCallTargetsIfSupported(p) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1<<40) + } + b.StopTimer() + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(uint64(b.N)*insnsPerRun), "ns/insn") +} + +const loopN = 200_000 + +func BenchmarkAluLoopV0(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 0), 3+7*loopN+2) } +func BenchmarkAluLoopV3(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 3), 3+7*loopN+2) } +func BenchmarkMemLoopV0(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 0), 5+11*loopN+2) } +func BenchmarkMemLoopV3(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 3), 5+11*loopN+2) } +func BenchmarkCallLoopV0(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 0), 2+6*loopN+2) +} +func BenchmarkCallLoopV3(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 3), 2+6*loopN+2) +} + +// Interpreter setup/teardown cost only (tiny program). +func BenchmarkNewInterpreterAndExit(b *testing.B) { + p := mkProgram([]Slot{slot(OpExit, 0, 0, 0, 0)}, 0) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1000) + } +} + +func TestSyntheticProgramsRun(t *testing.T) { + for _, ver := range []uint32{0, 3} { + for _, p := range []*Program{aluLoopProgram(1000, ver), memLoopProgram(1000, ver), callLoopProgram(1000, ver)} { + if err := p.Verify(); err != nil { + t.Fatal(err) + } + cm := cu.NewComputeMeter(1 << 30) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32 * 1024, Syscalls: noSyscalls, ComputeMeter: &cm}) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + t.Fatal(err) + } + t.Logf("ver=%d ret=%d cu=%d", ver, ret, used) + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go new file mode 100644 index 000000000..1d679d4e4 --- /dev/null +++ b/pkg/sbpf/perf_differential_test.go @@ -0,0 +1,338 @@ +package sbpf + +import ( + "bufio" + "encoding/binary" + "fmt" + "hash/fnv" + "math/rand" + "os" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +// Differential test: generates deterministic pseudo-random programs and dumps +// (return value, error string, CU consumed, memory hash) per program to the +// file named by SBPF_DIFF_OUT. Running it against the baseline and the +// optimized interpreter and diffing the two files checks that observable +// behaviour is identical. + +type diffSyscall struct { + fn func(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) +} + +func (s diffSyscall) Invoke(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) { + return s.fn(vm, r1, r2, r3, r4, r5) +} + +var ( + hashPoke = SymbolHash("poke") // write r3 bytes of value r2 at r1 via vm.Write + hashPeek = SymbolHash("peek") // read 8 bytes at r1 -> r0 + hashCopy = SymbolHash("copy") // copy r3 bytes from r2 to r1 (Translate based) + hashBurn = SymbolHash("burn") // consume r1 CU + hashSetLen = SymbolHash("setlen") // SetInputRegionLength(r1, r2, r3!=0) +) + +func diffRegistry(h uint32) (Syscall, bool) { + switch h { + case hashPoke: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + buf := make([]byte, r3) + for i := range buf { + buf[i] = byte(r2 + uint64(i)) + } + return 0, vm.Write(r1, buf) + }}, true + case hashPeek: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + return vm.Read64(r1) + }}, true + case hashCopy: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + src, err := vm.Translate(r2, r3, false) + if err != nil { + return 0, err + } + dst, err := vm.Translate(r1, r3, true) + if err != nil { + return 0, err + } + copy(dst, src) + return 0, nil + }}, true + case hashBurn: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + return 0, vm.ComputeMeter().Consume(r1 & 0xff) + }}, true + case hashSetLen: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + ip := vm.(*Interpreter) + ok := ip.SetInputRegionLength(r1, r2, r3 != 0) + if ok { + return 1, nil + } + return 0, nil + }}, true + } + return nil, false +} + +func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { + reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 + imm := func() uint32 { + switch rng.Intn(4) { + case 0: + return uint32(rng.Intn(16)) + case 1: + return uint32(int32(-rng.Intn(16))) + case 2: + return rng.Uint32() + default: + return uint32(rng.Intn(4096)) + } + } + alu64 := []uint8{OpAdd64Imm, OpAdd64Reg, OpSub64Imm, OpSub64Reg, OpMul64Imm, OpMul64Reg, OpDiv64Imm, OpDiv64Reg, + OpOr64Imm, OpOr64Reg, OpAnd64Imm, OpAnd64Reg, OpLsh64Imm, OpLsh64Reg, OpRsh64Imm, OpRsh64Reg, OpMod64Imm, OpMod64Reg, + OpXor64Imm, OpXor64Reg, OpMov64Imm, OpMov64Reg, OpArsh64Imm, OpArsh64Reg, OpNeg64, + OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, + OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, + OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, + OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} + switch rng.Intn(10) { + case 0, 1, 2, 3: // alu + op := alu64[rng.Intn(len(alu64))] + i := imm() + if op == OpLe || op == OpBe { + i = []uint32{16, 32, 64}[rng.Intn(3)] + } + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + i = 3 + } + switch op { + case OpLsh32Imm, OpRsh32Imm, OpArsh32Imm: + i = uint32(rng.Intn(32)) + case OpLsh64Imm, OpRsh64Imm, OpArsh64Imm: + i = uint32(rng.Intn(64)) + } + return []Slot{slot(op, reg(), reg(), 0, i)} + case 4: // load + ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random + var base uint8 + var off int16 + switch rng.Intn(4) { + case 0: + base, off = 10, int16(-rng.Intn(4096)) + case 1: + base, off = 5, int16(rng.Intn(1024)) + case 2: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} + case 5: // store + ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + var base uint8 + var off int16 + switch rng.Intn(5) { + case 0, 1: + base, off = 10, int16(-rng.Intn(4096)) + case 2: + base, off = 5, int16(rng.Intn(1024)) + case 3: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(8)], base, reg(), off, imm())} + case 6: // forward conditional jump (never backwards: guarantees termination) + maxOff := n - pc - 2 + if maxOff <= 0 { + return []Slot{slot(OpMov64Imm, reg(), 0, 0, imm())} + } + return []Slot{slot(jmp[rng.Intn(len(jmp))], reg(), reg(), int16(rng.Intn(min(maxOff, 8))), imm())} + case 7: // syscall + hs := []uint32{hashPoke, hashPeek, hashCopy, hashBurn, hashSetLen} + return []Slot{slot(OpCall, 0, 0, 0, hs[rng.Intn(len(hs))])} + case 8: // internal call + if ver >= sbpfver.SbpfVersionV3 { + return []Slot{slot(OpCall, 0, 1, 0, uint32(fnPC-int64(pc)-1))} + } + return []Slot{slot(OpCall, 0, 0, 0, PCHash(uint64(fnPC)))} + default: // set up pointer registers + switch rng.Intn(3) { + case 0: // r5 = heap + return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + case 1: // r1 = input + small + return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + default: // r9 = random 64-bit + return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + } + } +} + +func genProgram(rng *rand.Rand, ver uint32) *Program { + n := 8 + rng.Intn(120) + // layout: [0, n) main body then exit; fn at fnPC: a few ALU ops + exit + body := make([]Slot, 0, n+16) + fnPC := int64(n + 1) + for len(body) < n { + body = append(body, randSlot(rng, len(body), n, ver, fnPC)...) + } + if len(body) > n { + body = body[:n-1] // drop a cut lddw pair + } + body = append(body, slot(OpExit, 0, 0, 0, 0)) + // fix up forward jumps that would land on the second slot of an lddw + for pc := range body { + if body[pc].Op()&0x07 == ClassJmp && body[pc].Op() != OpCall && body[pc].Op() != OpExit { + dst := pc + int(body[pc].Off()) + 1 + if dst < len(body) && body[dst].Op() == 0 { + body[pc] = body[pc]&^(Slot(0xffff)<<16) | Slot(uint16(body[pc].Off()+1))<<16 + } + } + } + // function + body = append(body, + slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), + slot(OpXor64Reg, 7, 6, 0, 0), + slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(body, ver) + if ver < sbpfver.SbpfVersionV3 { + p.Funcs[PCHash(uint64(fnPC))] = fnPC + } + p.RO = make([]byte, 256) + for i := range p.RO { + p.RO[i] = byte(i * 7) + } + return p +} + +func memHash(bs ...[]byte) uint64 { + h := fnv.New64a() + for _, b := range bs { + h.Write(b) + } + return h.Sum64() +} + +func TestDifferentialDump(t *testing.T) { + out := os.Getenv("SBPF_DIFF_OUT") + if out == "" { + t.Skip("SBPF_DIFF_OUT not set") + } + f, err := os.Create(out) + if err != nil { + t.Fatal(err) + } + defer f.Close() + w := bufio.NewWriter(f) + defer w.Flush() + + rng := rand.New(rand.NewSource(12345)) + const N = 100000 + generated, verified := 0, 0 + for i := 0; i < N; i++ { + ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + p := genProgram(rng, ver) + generated++ + if err := p.Verify(); err != nil { + fmt.Fprintf(w, "%d ver=%d VERIFY_FAIL %v\n", i, ver, err) + continue + } + verified++ + resolveCallTargetsIfSupported(p) + input := make([]byte, 700) + for j := range input { + input[j] = byte(j) + } + var regions []InputRegion + useRegions := rng.Intn(2) == 0 + if useRegions { + regions = []InputRegion{ + {Offset: 0, HostOffset: 0, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}, + {Offset: 100, HostOffset: 100, RegionSize: 150, AddressSpaceReserved: 300, Writable: rng.Intn(2) == 0, AccountIndex: 0}, + {Offset: 400, HostOffset: 400, RegionSize: 300, AddressSpaceReserved: 300, Writable: true, AccountIndex: -1}, + } + } + budget := uint64(1 + rng.Intn(400)) + if rng.Intn(4) == 0 { + budget = 100000 + } + cm := cu.NewComputeMeter(budget) + heapMax := 4096 * (1 + rng.Intn(4)) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: heapMax, + Syscalls: diffRegistry, + ComputeMeter: &cm, + Input: input, + InputRegions: regions, + DisableStackFrameGaps: rng.Intn(3) == 0, + }) + var ret, cuUsed uint64 + var runErr error + func() { + defer func() { + if r := recover(); r != nil { + runErr = fmt.Errorf("PANIC: %v", r) + } + }() + ret, cuUsed, runErr = ip.Run() + }() + errStr := "" + if runErr != nil { + errStr = runErr.Error() + } + h := memHash(ip.stack.mem, ip.heap, input) + regionSizes := "" + for _, r := range ip.inputRegions { + regionSizes += fmt.Sprintf("%d/%v,", r.RegionSize, r.Writable) + } + fmt.Fprintf(w, "%d ver=%d budget=%d ret=%d cu=%d remaining=%d err=%q mem=%x regions=%s\n", + i, ver, budget, ret, cuUsed, cm.Remaining(), errStr, h, regionSizes) + ip.Finish() + if os.Getenv("SBPF_CHECK_POOL_ZERO") != "" { + // The buffers just returned to the pool must be all-zero. + st := stackMemPool.Get().([]byte) + hp := heapPool.Get().([]byte) + for j, b := range st[:StackMax] { + if b != 0 { + t.Fatalf("program %d: pooled stack not zeroed at %d", i, j) + } + } + for j, b := range hp[:cap(hp)] { + if b != 0 { + t.Fatalf("program %d: pooled heap not zeroed at %d", i, j) + } + } + stackMemPool.Put(st) + heapPool.Put(hp) + } + } + t.Logf("generated=%d verified=%d", generated, verified) +} + +var _ = binary.LittleEndian From daf2acfbd37343c7354c794f5d0cf5bf76e4f21d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 097/111] sealevel: add native program micro-benchmarks (System transfer, Vote TowerSync) Measured through ExecutionCtx.ProcessInstruction so instruction-context push/pop, lamport-sum checks and timing metrics are included; each has a NoTiming variant (SkipTimingMetrics) to quantify instrumentation cost, plus a vote-state (de)serialization round trip. NOTE: written without a local build of pkg/sealevel (sandbox cannot fetch its dependencies); expect to fix compile errors on first run. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/native_perf_bench_test.go | 287 +++++++++++++++++++++++++ 1 file changed, 287 insertions(+) create mode 100644 pkg/sealevel/native_perf_bench_test.go diff --git a/pkg/sealevel/native_perf_bench_test.go b/pkg/sealevel/native_perf_bench_test.go new file mode 100644 index 000000000..447931695 --- /dev/null +++ b/pkg/sealevel/native_perf_bench_test.go @@ -0,0 +1,287 @@ +package sealevel + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" +) + +// Micro-benchmarks for the natively implemented programs that are still on the +// mainnet hot path (System transfer, Vote TowerSync), measured through the same +// ExecutionCtx.ProcessInstruction entry point replay uses, so the per-instruction +// plumbing (instruction context push/pop, lamport-sum checks, timing metrics) +// is included. Run with: +// +// go test ./pkg/sealevel/ -run XXX -bench 'Native|VoteState|Timing' -benchmem -cpu 1 -count 5 + +func benchPubkey(b byte) solana.PublicKey { + var pk solana.PublicKey + for i := range pk { + pk[i] = b + } + return pk +} + +// newBenchExecCtx mirrors newSystemProgramTestExecCtx without testing.T. +func newBenchExecCtx(txAccts *TransactionAccounts, clockSlot uint64, enabled ...features.FeatureGate) *ExecutionCtx { + txCtx := NewTransactionCtx(*txAccts, 5, 64) + execCtx := &ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(1 << 62)} + execCtx.Accounts = accounts.NewMemAccounts() + + clockAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct); err != nil { + panic(err) + } + WriteClockSysvar(&execCtx.Accounts, SysvarClock{Slot: clockSlot, Epoch: 0}) + + rentAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct); err != nil { + panic(err) + } + WriteRentSysvar(&execCtx.Accounts, SysvarRent{LamportsPerUint8Year: 3480, ExemptionThreshold: 2, BurnPercent: 50}) + + f := features.NewFeaturesDefault() + for _, gate := range enabled { + f.EnableFeature(gate, 0) + } + execCtx.Features = *f + return execCtx +} + +// resetTxCtx gives the execution context a fresh instruction trace/stack for +// the next instruction (each ProcessInstruction consumes one trace slot). +func resetTxCtx(execCtx *ExecutionCtx, txAccts *TransactionAccounts) { + execCtx.TransactionContext = NewTransactionCtx(*txAccts, 5, 64) +} + +// ---------------------------------------------------------------- System + +func encodeSystemTransfer(lamports uint64) []byte { + out := binary.LittleEndian.AppendUint32(nil, uint32(SystemProgramInstrTypeTransfer)) + return binary.LittleEndian.AppendUint64(out, lamports) +} + +func benchmarkSystemTransfer(b *testing.B, skipTiming bool) { + systemProgramAcct := accounts.Account{Key: a.SystemProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + from := accounts.Account{Key: benchPubkey(0x11), Lamports: 1 << 60, Data: []byte{}, Owner: a.SystemProgramAddr} + to := accounts.Account{Key: benchPubkey(0x22), Lamports: 1_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{systemProgramAcct, from, to}) + metas := []AccountMeta{ + {Pubkey: from.Key, IsSigner: true, IsWritable: true}, + {Pubkey: to.Key, IsSigner: false, IsWritable: true}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + instr := encodeSystemTransfer(1) + + execCtx := newBenchExecCtx(txAccts, 1234) + execCtx.SkipTimingMetrics = skipTiming + + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + resetTxCtx(execCtx, txAccts) + if err := execCtx.ProcessInstruction(instr, instrAccts, []uint64{0}); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + if got := txAccts.Accounts[2].Lamports; got != 1_000_000+uint64(b.N) { + b.Fatalf("destination lamports %d, want %d", got, 1_000_000+uint64(b.N)) + } +} + +func BenchmarkNativeSystemTransfer(b *testing.B) { benchmarkSystemTransfer(b, false) } +func BenchmarkNativeSystemTransferNoTiming(b *testing.B) { benchmarkSystemTransfer(b, true) } + +// ---------------------------------------------------------------- Vote + +const ( + benchVoteRoot = uint64(1000) + benchVoteLastSlot = benchVoteRoot + MaxLockoutHistory // 1031 + benchClockSlot = benchVoteLastSlot + 2 +) + +func benchSlotHash(slot uint64) [32]byte { + var h [32]byte + binary.LittleEndian.PutUint64(h[:], slot*0x9e3779b97f4a7c15) + binary.LittleEndian.PutUint64(h[8:], ^slot) + h[31] = 0xa5 + return h +} + +// benchSlotHashes builds a 512-entry SlotHashes sysvar (newest first) that +// covers every slot the benchmark tower refers to. +func benchSlotHashes() SysvarSlotHashes { + const n = 512 + newest := benchVoteLastSlot + 1 + sh := make(SysvarSlotHashes, 0, n) + for i := uint64(0); i < n; i++ { + s := newest - i + sh = append(sh, SlotHash{Slot: s, Hash: benchSlotHash(s)}) + } + return sh +} + +// benchInitialVoteState is a fully populated current-version vote state with a +// full 31-entry tower ending at benchVoteLastSlot. +func benchInitialVoteState(voter solana.PublicKey) *VoteState { + vs := &VoteState{ + NodePubkey: voter, + AuthorizedWithdrawer: voter, + Commission: 10, + PriorVoters: PriorVoters{Index: 31, IsEmpty: true}, + EpochCredits: []EpochCredits{{Epoch: 0, Credits: 1000, PrevCredits: 0}}, + LastTimestamp: BlockTimestamp{Slot: benchVoteLastSlot, Timestamp: 1_700_000_000}, + } + vs.AuthorizedVoters.AuthorizedVoters.Set(0, voter) + root := benchVoteRoot + vs.RootSlot = &root + for i := uint64(0); i < MaxLockoutHistory; i++ { + vs.Votes.PushBack(LandedVote{ + Latency: 1, + Lockout: VoteLockout{Slot: benchVoteRoot + 1 + i, ConfirmationCount: uint32(MaxLockoutHistory - i)}, + }) + } + return vs +} + +func benchSerializedVoteState(vs *VoteState) []byte { + versioned := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + data := make([]byte, VoteStateV3Size) + if err := WriteVersionedVoteStateInPlace(data, versioned); err != nil { + panic(err) + } + return data +} + +// encodeTowerSync encodes a TowerSync that advances the tower by one slot: +// root = old root + 1, lockouts = old lockouts shifted by one slot plus the +// new slot, i.e. exactly what a validator sends every slot. +func encodeTowerSync() []byte { + root := benchVoteRoot + 1 + out := binary.LittleEndian.AppendUint32(nil, uint32(VoteProgramInstrTypeTowerSync)) + out = binary.LittleEndian.AppendUint64(out, root) + out = append(out, byte(MaxLockoutHistory)) // compact-u16, < 0x80 + prev := root + for i := uint64(0); i < MaxLockoutHistory; i++ { + slot := root + 1 + i + out = binary.AppendUvarint(out, slot-prev) + out = append(out, byte(MaxLockoutHistory-i)) + prev = slot + } + last := root + MaxLockoutHistory // benchVoteLastSlot + 1 + h := benchSlotHash(last) + out = append(out, h[:]...) + out = append(out, 1) // Some(timestamp) + out = binary.LittleEndian.AppendUint64(out, uint64(1_700_000_001)) + var blockID [32]byte + out = append(out, blockID[:]...) + return out +} + +type voteBench struct { + execCtx *ExecutionCtx + txAccts *TransactionAccounts + instr []byte + instrAccts []InstructionAccount + initial []byte +} + +func newVoteBench(skipTiming bool) *voteBench { + voter := benchPubkey(0x33) + votePk := benchPubkey(0x44) + initial := benchSerializedVoteState(benchInitialVoteState(voter)) + + voteProgramAcct := accounts.Account{Key: a.VoteProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + voteAcct := accounts.Account{Key: votePk, Lamports: 1_000_000_000, Data: append([]byte(nil), initial...), Owner: a.VoteProgramAddr} + voterAcct := accounts.Account{Key: voter, Lamports: 1_000_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{voteProgramAcct, voteAcct, voterAcct}) + metas := []AccountMeta{ + {Pubkey: votePk, IsSigner: false, IsWritable: true}, + {Pubkey: voter, IsSigner: true, IsWritable: false}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + + execCtx := newBenchExecCtx(txAccts, benchClockSlot, + features.EnableTowerSyncIx, + features.VoteStateAddVoteLatency, + features.TimelyVoteCredits, + features.DeprecateUnusedLegacyVotePlumbing, + ) + execCtx.SkipTimingMetrics = skipTiming + shAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarSlotHashesAddr, &shAcct); err != nil { + panic(err) + } + WriteSlotHashesSysvar(&execCtx.Accounts, benchSlotHashes()) + + return &voteBench{execCtx: execCtx, txAccts: txAccts, instr: encodeTowerSync(), instrAccts: instrAccts, initial: initial} +} + +// step runs one TowerSync against the initial vote state (the account data is +// rewound to the initial state first so every iteration performs identical work). +func (vb *voteBench) step() error { + copy(vb.txAccts.Accounts[1].Data, vb.initial) + resetTxCtx(vb.execCtx, vb.txAccts) + return vb.execCtx.ProcessInstruction(vb.instr, vb.instrAccts, []uint64{0}) +} + +func TestNativeVoteTowerSyncBenchSetup(t *testing.T) { + vb := newVoteBench(false) + if err := vb.step(); err != nil { + t.Fatalf("TowerSync failed: %v", err) + } + versioned, err := UnmarshalVersionedVoteState(vb.txAccts.Accounts[1].Data) + if err != nil { + t.Fatal(err) + } + vs := versioned.ConvertToCurrent() + if vs.Votes.Len() != MaxLockoutHistory { + t.Fatalf("tower length %d, want %d", vs.Votes.Len(), MaxLockoutHistory) + } + if last := vs.Votes.Back().Lockout.Slot; last != benchVoteLastSlot+1 { + t.Fatalf("last voted slot %d, want %d", last, benchVoteLastSlot+1) + } + if vs.RootSlot == nil || *vs.RootSlot != benchVoteRoot+1 { + t.Fatalf("root %v, want %d", vs.RootSlot, benchVoteRoot+1) + } +} + +func benchmarkVoteTowerSync(b *testing.B, skipTiming bool) { + vb := newVoteBench(skipTiming) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := vb.step(); err != nil { + b.Fatal(err) + } + } +} + +func BenchmarkNativeVoteTowerSync(b *testing.B) { benchmarkVoteTowerSync(b, false) } +func BenchmarkNativeVoteTowerSyncNoTiming(b *testing.B) { benchmarkVoteTowerSync(b, true) } + +// BenchmarkVoteStateRoundTrip isolates vote-state (de)serialization: decode +// the account, convert to current, re-encode — the fixed cost of every vote. +func BenchmarkVoteStateRoundTrip(b *testing.B) { + data := benchSerializedVoteState(benchInitialVoteState(benchPubkey(0x33))) + out := make([]byte, VoteStateV3Size) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + versioned, err := UnmarshalVersionedVoteState(data) + if err != nil { + b.Fatal(err) + } + vs := versioned.ConvertToCurrent() + cur := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + if err := WriteVersionedVoteStateInPlace(out, cur); err != nil { + b.Fatal(err) + } + } +} From 929471b59b295b26363a09ae530aaf2d167a9441 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 098/111] test: update VASA stack registers for fixed-size interpreter state --- pkg/sbpf/vasa_test.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/pkg/sbpf/vasa_test.go b/pkg/sbpf/vasa_test.go index e7d64034c..70296ea43 100644 --- a/pkg/sbpf/vasa_test.go +++ b/pkg/sbpf/vasa_test.go @@ -73,7 +73,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { gapped := NewStack(version, false) defer gapped.Finish() - gappedRegs := make([]uint64, 11) + gappedRegs := new([16]uint64) gappedRegs[10] = VaddrStack + StackFrameSize require.True(t, gapped.Push(gappedRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*3, gappedRegs[10]) @@ -81,7 +81,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { contiguous := NewStack(version, true) defer contiguous.Finish() - contiguousRegs := make([]uint64, 11) + contiguousRegs := new([16]uint64) contiguousRegs[10] = VaddrStack + StackFrameSize require.True(t, contiguous.Push(contiguousRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, contiguousRegs[10]) @@ -94,7 +94,7 @@ func TestStackFrameGapsAreLegacyOnly(t *testing.T) { stack := NewStack(version, false) defer stack.Finish() - regs := make([]uint64, 11) + regs := new([16]uint64) regs[10] = VaddrStack + StackFrameSize require.True(t, stack.Push(regs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, regs[10]) From 3d31a10aee935cbbd6b736b2465d19db344980f0 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 099/111] test: benchmark verified token arithmetic and CPI workloads with warm programs --- pkg/sealevel/program_workloads_bench_test.go | 195 +++++++++++++++++++ 1 file changed, 195 insertions(+) create mode 100644 pkg/sealevel/program_workloads_bench_test.go diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go new file mode 100644 index 000000000..cc6b83e78 --- /dev/null +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -0,0 +1,195 @@ +package sealevel + +import ( + "encoding/binary" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" +) + +// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// No live account writes or network calls occur in this harness. Each invocation +// gets fresh account data; the program cache is warm and shared between runs. +type programWorkload struct { + name string + elf []byte + program solana.PublicKey + accts []accounts.Account + metas []AccountMeta + instruction []byte + check func(testing.TB, *ExecutionCtx) +} + +func expandedProgramWorkloads(t testing.TB) []programWorkload { + t.Helper() + program := benchPubkey(0x91) + from := accounts.Account{Key: benchPubkey(0x31), Owner: program, Lamports: 10000} + to := accounts.Account{Key: benchPubkey(0x32), Owner: program, Lamports: 10000} + cases := []programWorkload{{name: "BPF_LamportTransfer", elf: fixtures.Load(t, "sbpf", "cpi_c_to_bpf.so"), program: program, accts: []accounts.Account{from, to}, metas: []AccountMeta{{Pubkey: program}, {Pubkey: from.Key, IsSigner: true, IsWritable: true}, {Pubkey: to.Key, IsSigner: true, IsWritable: true}}, instruction: []byte{0}, check: func(t testing.TB, ctx *ExecutionCtx) { + src, err := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, err) + dst, err := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, err) + require.Equal(t, uint64(9000), src.Lamports) + require.Equal(t, uint64(11000), dst.Lamports) + }}} + + pda, bump, err := solana.FindProgramAddress([][]byte{[]byte("You pass butter")}, program) + require.NoError(t, err) + cases = append(cases, programWorkload{name: "CPI_Rust_SystemAllocate", elf: fixtures.Load(t, "sbpf", "cpi_rust_to_system_program_allocate.so"), program: program, + accts: []accounts.Account{{Key: a.SystemProgramAddr, Owner: a.NativeLoaderAddr, Executable: true, Lamports: 10000}, {Key: pda, Owner: a.SystemProgramAddr, Lamports: 10000}}, + metas: []AccountMeta{{Pubkey: a.SystemProgramAddr}, {Pubkey: pda, IsSigner: true, IsWritable: true}}, instruction: []byte{bump}, check: func(t testing.TB, ctx *ExecutionCtx) { + acct, e := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, e) + require.Len(t, acct.Data, 1337) + require.NotEmpty(t, ctx.InnerInstrs) + }}) + dir := os.Getenv("MITHRIL_PROGRAM_BENCH_DIR") + if dir == "" { + return cases + } + arithmetic, err := os.ReadFile(filepath.Join(dir, "rotation_compute.so")) + require.NoError(t, err) + for _, iterations := range []uint32{500, 5000} { + n := iterations + data := append([]byte("RC01"), 0, 0, 0, 0) + binary.LittleEndian.PutUint32(data[4:], n) + name := "Arithmetic_500" + if n == 5000 { + name = "Arithmetic_5000" + } + cases = append(cases, programWorkload{name: name, elf: arithmetic, program: program, instruction: data, check: func(t testing.TB, ctx *ExecutionCtx) { + x := uint64(0x9e3779b97f4a7c15) + for i := uint32(0); i < n; i++ { + x = ((x << 7) | (x >> 57)) ^ (uint64(i) + 0x517cc1b727220a95) + } + _, got := ctx.TransactionContext.ReturnData() + require.Len(t, got, 8) + require.Equal(t, x, binary.LittleEndian.Uint64(got)) + }}) + } + token, err := os.ReadFile(filepath.Join(dir, "token2022.so")) + require.NoError(t, err) + mint, auth := benchPubkey(0x51), benchPubkey(0x52) + tokenData := func(amount uint64) []byte { + d := make([]byte, 165) + copy(d, mint[:]) + copy(d[32:], auth[:]) + binary.LittleEndian.PutUint64(d[64:], amount) + d[108] = 1 + return d + } + mintData := make([]byte, 82) + mintData[44] = 6 + mintData[45] = 1 + src := accounts.Account{Key: benchPubkey(0x53), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(1000000)} + dst := accounts.Account{Key: benchPubkey(0x54), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(5)} + instr := append([]byte{12}, binary.LittleEndian.AppendUint64(nil, 1000)...) + instr = append(instr, 6) + cases = append(cases, programWorkload{name: "Token2022_TransferChecked", elf: token, program: solana.Token2022ProgramID, + accts: []accounts.Account{src, {Key: mint, Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: mintData}, dst, {Key: auth, Owner: a.SystemProgramAddr, Lamports: 10000000}}, + metas: []AccountMeta{{Pubkey: src.Key, IsWritable: true}, {Pubkey: mint}, {Pubkey: dst.Key, IsWritable: true}, {Pubkey: auth, IsSigner: true}}, instruction: instr, + check: func(t testing.TB, ctx *ExecutionCtx) { + s, e := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, e) + d, e := ctx.TransactionContext.Accounts.GetAccount(3) + require.NoError(t, e) + require.Equal(t, uint64(999000), binary.LittleEndian.Uint64(s.Data[64:])) + require.Equal(t, uint64(1005), binary.LittleEndian.Uint64(d.Data[64:])) + }}) + return cases +} +func workloadRunner(t testing.TB, w programWorkload, vasa bool) func() (*ExecutionCtx, error) { + t.Helper() + f := features.NewFeaturesDefault() + if vasa { + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + } + l, err := loader.NewLoaderWithSyscalls(w.elf, func(h uint32) (sbpf.Syscall, bool) { return Syscalls(f, false, h) }, false, f) + require.NoError(t, err) + prog, err := l.Load() + require.NoError(t, err) + require.NoError(t, prog.Verify()) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + db := &accountsdb.AccountsDb{ProgramCache: cache} + db.AddProgramToCache(w.program, &accountsdb.ProgramCacheEntry{Program: prog}) + _, cached := db.MaybeGetProgramFromCache(w.program) + require.True(t, cached, "warm program must be cached") + return func() (*ExecutionCtx, error) { + list := make([]accounts.Account, len(w.accts)+1) + list[0] = accounts.Account{Key: w.program, Owner: a.BpfLoader2Addr, Lamports: 10000000, Executable: true, Data: w.elf} + for i, acct := range w.accts { + list[i+1] = acct + list[i+1].Data = append([]byte(nil), acct.Data...) + } + tx := NewTransactionAccounts(list) + ctx := newBenchExecCtx(tx, 1337) + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} + ctx.Features = *f + ctx.ComputeMeter = cu.NewComputeMeter(1400000) + ctx.SlotCtx = &SlotCtx{Slot: 1337, AccountsDb: db} + ctx.Log = &LogRecorder{} + ctx.RecordInnerInstructions = true + err := ctx.ProcessInstruction(w.instruction, InstructionAcctsFromAccountMetas(w.metas, *tx), []uint64{0}) + return ctx, err + } +} +func TestProgramWorkloadResults(t *testing.T) { + for _, w := range expandedProgramWorkloads(t) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + t.Run(name, func(t *testing.T) { + run := workloadRunner(t, w, vasa) + ctx, err := run() + require.NoError(t, err) + w.check(t, ctx) + t.Logf("cu=%d inner=%d", ctx.ComputeMeter.Used(), len(ctx.InnerInstrs)) + }) + } + } +} +func BenchmarkProgramWorkloads(b *testing.B) { + for _, w := range expandedProgramWorkloads(b) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + b.Run(name, func(b *testing.B) { + run := workloadRunner(b, w, vasa) + ctx, err := run() + require.NoError(b, err) + w.check(b, ctx) + used := ctx.ComputeMeter.Used() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + ctx, err = run() + if err != nil { + b.Fatal(err) + } + } + b.StopTimer() + w.check(b, ctx) + b.ReportMetric(float64(used), "cu/op") + }) + } + } +} From 8401d577324142e458996e92acc0e8357651fb45 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:21:44 -0500 Subject: [PATCH 100/111] docs: record individual-program and matched Alpenglow replay measurements --- docs/sbpf-interpreter-benchmarks.md | 103 ++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 docs/sbpf-interpreter-benchmarks.md diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md new file mode 100644 index 000000000..73eb970e5 --- /dev/null +++ b/docs/sbpf-interpreter-benchmarks.md @@ -0,0 +1,103 @@ +# Interpreter performance validation + +This experiment compares `9db7dcb` plus identical benchmark code with the same +base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift +semantics patch is excluded. Three VASA tests were updated to pass the new +`*[16]uint64` register type; no further production optimization was added during +this measurement pass. + +## Individual programs + +Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five +one-second samples per variant with alternating run order. The existing validator +continued running. CPU 15 shares a physical core with CPU 7; these are shared-host +measurements rather than an isolated-machine throughput ceiling. + +The harness has a preloaded program cache and asserts a cache hit before timing. +Every invocation gets fresh account data and execution context. It measures +instruction setup, serialization, execution and publication together; it is not +just the interpreter loop. Timing instrumentation and instruction recording are +enabled identically on both versions. Tests check arithmetic return values, +post-transfer balances, allocation results, and recorded CPI. Requested budgets +are equal and the measured CU charges match between variants. + +| Workload | CU | Median before → after | Speedup | +|---|---:|---:|---:| +| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | +| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | +| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | +| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | +| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | +| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | +| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | + +The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. +The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from +the earlier SPL Token loader-only benchmark: they use different program binaries, +instructions, and include execution-context setup. + +Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` +and `token2022.so` to enable those external fixtures. Without it, the in-repository +BPF/CPI fixtures still run. Pinned input SHA-256 values: + +- Arithmetic ELF: `db7c55d6563c879e35dfe2b24edb0fe0515d5a5ae627fe00e3e247001c786441`. + Source: `ag-transaction-bench` at `7e5a263fa5a1c72088f191daf5b7c5d2484c997c`, + `transaction-bench/program/src/rotation_compute.c`. +- Token-2022 ELF: `a794161408080f690dac00832f45b3c3e2b71f1339586667ad1f979cf91d5b68`. + Public Alpenglow program `TokenzQdBNbLqP5VEhdkAS6EPFLC1PHnBqCXEpPxuEb`, + fetched at RPC context slot 4,231,444, program-data account + `DoU57AYuPFu2QU514RktNPG22QhApEjnKxnBcu4BHDTY`. Strip its 45-byte upgradeable + loader metadata before saving the ELF. Verify the hash; do not silently replace + it with a later deployment. + +``` +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^TestProgramWorkloadResults$' -v +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^$' -bench '^BenchmarkProgramWorkloads$' \ + -benchtime=1s -count=5 -benchmem +``` + +## Recorded Alpenglow blocks + +Each run bootstrapped a fresh isolated AccountsDB from the same public full +snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay +covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and +142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, +nice 19. Two paired runs used baseline/candidate then candidate/baseline order. +The live validator's AccountsDB and configuration were not used or changed. + +All 233 per-slot bank hashes matched between baseline and candidate in both +pairs, excluding the run-specific comment header in `bankhash.log`. + +| Work measured | Pair 1 before → after | Pair 2 before → after | +|---|---:|---:| +| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | +| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | +| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | +| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | +| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | + +The sample includes blocks around 40–46 million CU. Three inspected non-empty +blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and +Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries +attribute whole-transaction CU to every participating program and must not be +summed as if they were exclusive per-program execution costs. + +Whole-block p95 did not improve, and the p99 changes are small/variable. The +slowest candidate blocks in the first pair contained no sBPF execution; their +large timers were dispatch and signature verification. These single-CPU replay +results do not establish a live voting/FAST improvement or production parallel +replay latency. They exclude network wait from ProcessBlock and are not elapsed +end-to-end catch-up times. + +## Correctness and limits + +- Native Zen 5 baseline and candidate differential outputs match for 100,000 + deterministic generated programs; candidate pool-zero checks pass. +- Baseline/candidate workload effects and CU charges match. Targeted race tests + for the interpreter, loader, and workload harness pass; vet passes. +- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; + therefore no complete sealevel test-suite pass is claimed. Broader conformance + testing remains separate from this performance experiment. +- No candidate was deployed and no validator restart was needed. From d8fd0646950dfb27a5725be1f668e16f7bcf9863 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:39:11 +0000 Subject: [PATCH 101/111] sealevel: copy, compare and fill VM memory without temporaries or byte loops sol_memcpy_/sol_memmove_ read the source into a fresh heap buffer and wrote it back; the copy now goes directly between the two translated slices with Go's memmove-semantics copy (overlap handled, source translated first so error precedence is unchanged, and a copy-on-write/growth of the destination region still reads the pre-write bytes because the source slice keeps the previous backing buffer alive). sol_memcmp_ uses bytes.Equal for the common equal case and word-skips to the first differing byte otherwise; sol_memset_ uses clear for zero and a doubling copy for other values. An SPL Token transfer issues two memcpy and four memcmp calls, so this is a small, allocation-free win rather than a large one. Tests: memcmpResult against the previous byte loop on 100k random inputs, memsetBytes over sizes and values, and VM-level memmove/memcpy overlap, error-ordering, copy-on-write-region and memcmp/memset checks. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/syscalls_mem.go | 77 +++++++++--- pkg/sealevel/syscalls_mem_test.go | 199 ++++++++++++++++++++++++++++++ 2 files changed, 256 insertions(+), 20 deletions(-) create mode 100644 pkg/sealevel/syscalls_mem_test.go diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 7bea5331e..29834a9ce 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -1,6 +1,7 @@ package sealevel import ( + "bytes" "encoding/binary" //"github.com/Overclock-Validator/mithril/pkg/mlog" @@ -15,14 +16,24 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { return execCtx.ComputeMeter.Consume(cost) } -func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { - srcBuf := make([]byte, n) - err = vm.Read(src, srcBuf) +// memmoveImplInternal copies n bytes from src to dst inside the VM without a +// temporary buffer. The source is translated first so a bad source address is +// reported before a bad destination, as before. Go's copy has memmove +// semantics, so overlapping ranges within one region are handled; and when +// the destination translation grows or copy-on-writes an account region, the +// source slice still refers to the previous backing buffer, whose bytes are +// exactly what the old read-then-write sequence would have copied. +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + srcMem, err := vm.Translate(src, n, false) if err != nil { - return + return err + } + dstMem, err := vm.Translate(dst, n, true) + if err != nil { + return err } - err = vm.Write(dst, srcBuf) - return + copy(dstMem, srcMem) + return nil } // SyscallMemcpyImpl is the implementation of the memcpy (sol_memcpy_) syscall. @@ -76,6 +87,26 @@ func SyscallMemmoveImpl(vm sbpf.VM, dst, src, n uint64) (uint64, error) { var SyscallMemmove = sbpf.SyscallFunc3(SyscallMemmoveImpl) +// memcmpResult returns the C memcmp result of two equal-length slices: zero +// when they are equal, otherwise the difference of the first differing bytes +// as unsigned values, matching Agave's `(b1 as i32) - (b2 as i32)`. +func memcmpResult(a, b []byte) int32 { + if bytes.Equal(a, b) { + return 0 + } + // The slices differ: skip equal 8-byte words, then locate the byte. + i := 0 + for i+8 <= len(a) && binary.LittleEndian.Uint64(a[i:]) == binary.LittleEndian.Uint64(b[i:]) { + i += 8 + } + for ; i < len(a); i++ { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + // SyscallMemcmpImpl is the implementation for the memcmp (sol_memcmp_) syscall. func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemcmp") @@ -96,15 +127,7 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, return syscallErr(err) } - cmpResult := int32(0) - for count := uint64(0); count < n; count++ { - b1 := slice1[count] - b2 := slice2[count] - if b1 != b2 { - cmpResult = int32(b1) - int32(b2) - break - } - } + cmpResult := memcmpResult(slice1, slice2) resultSlice, err := vm.Translate(resultAddr, 4, true) if err != nil { @@ -118,7 +141,23 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, var SyscallMemcmp = sbpf.SyscallFunc4(SyscallMemcmpImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// memsetBytes fills mem with c using the runtime's block clear for zero and a +// doubling copy otherwise, instead of a byte-at-a-time loop. +func memsetBytes(mem []byte, c byte) { + if len(mem) == 0 { + return + } + if c == 0 { + clear(mem) + return + } + mem[0] = c + for filled := 1; filled < len(mem); filled *= 2 { + copy(mem[filled:], mem[:filled]) + } +} + +// SyscallMemsetImpl is the implementation for the memset (sol_memset_) syscall. func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemset") @@ -133,16 +172,14 @@ func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { return syscallErr(err) } - for i := uint64(0); i < n; i++ { - mem[i] = byte(c) - } + memsetBytes(mem, byte(c)) return syscallSuccess(0) } var SyscallMemset = sbpf.SyscallFunc3(SyscallMemsetImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// SyscallAllocFreeImpl is the implementation for the alloc/free (sol_alloc_free_) syscall. func SyscallAllocFreeImpl(vm sbpf.VM, size, freeAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallAllocFreeImpl") diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go new file mode 100644 index 000000000..3b02daea0 --- /dev/null +++ b/pkg/sealevel/syscalls_mem_test.go @@ -0,0 +1,199 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + "math/rand" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + feat "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func newMemSyscallVM(t *testing.T, input []byte, regions []sbpf.InputRegion) (*sbpf.Interpreter, *ExecutionCtx) { + t.Helper() + features := feat.NewFeaturesDefault() + execCtx := &ExecutionCtx{Features: *features, ComputeMeter: cu.NewComputeMeter(1_000_000)} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram, Funcs: map[uint32]int64{}}, &sbpf.VMOpts{ + Input: input, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, + InputRegions: regions, + }) + t.Cleanup(vm.Finish) + return vm, execCtx +} + +func TestSyscallMemmoveOverlapping(t *testing.T) { + for _, test := range []struct { + name string + dst, src, n uint64 + expectMemcpyE bool + }{ + {name: "forward-overlap", dst: 4, src: 0, n: 16, expectMemcpyE: true}, + {name: "backward-overlap", dst: 0, src: 4, n: 16, expectMemcpyE: true}, + {name: "disjoint", dst: 40, src: 0, n: 16}, + {name: "adjacent", dst: 16, src: 0, n: 16}, + {name: "empty", dst: 0, src: 0, n: 0}, + } { + t.Run(test.name, func(t *testing.T) { + input := make([]byte, 64) + for i := range input { + input[i] = byte(i + 1) + } + want := append([]byte(nil), input...) + copy(want[test.dst:test.dst+test.n], want[test.src:test.src+test.n]) + + vm, _ := newMemSyscallVM(t, input, nil) + ret, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, want, input, "memmove must have Go copy (memmove) semantics") + + // memcpy: the same result for disjoint ranges, an error for overlap. + input2 := make([]byte, 64) + for i := range input2 { + input2[i] = byte(i + 1) + } + vm2, _ := newMemSyscallVM(t, input2, nil) + _, err = SyscallMemcpyImpl(vm2, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + if test.expectMemcpyE { + require.ErrorIs(t, err, SyscallErrCopyOverlapping) + } else { + require.NoError(t, err) + require.Equal(t, want, input2) + } + }) + } +} + +func TestSyscallMemmoveBadAddressOrder(t *testing.T) { + input := make([]byte, 32) + vm, _ := newMemSyscallVM(t, input, nil) + // Unreadable source is reported before an unwritable destination. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput+100, 8) + require.Error(t, err) + var badAccess sbpf.ExcBadAccess + require.ErrorAs(t, err, &badAccess) + require.False(t, badAccess.Write, "the source translation must fail first") + // Readable source, write to a read-only region. + _, err = SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput, 8) + require.Error(t, err) + require.ErrorAs(t, err, &badAccess) + require.True(t, badAccess.Write) +} + +func TestSyscallMemmoveIntoGrowingInputRegion(t *testing.T) { + // The destination region copy-on-writes and grows on first write; the + // source slice taken before that must still yield the original bytes. + original := []byte{1, 2, 3, 4, 5, 6, 7, 8} + var replaced []byte + region := sbpf.InputRegion{ + Offset: 0, + RegionSize: uint64(len(original)), + AddressSpaceReserved: 32, + Writable: false, + AccountIndex: 0, + Data: original, + OnWrite: func(region *sbpf.InputRegion, requestedLen uint64) error { + replaced = make([]byte, 32) + copy(replaced, region.Data) + region.Data = replaced + region.RegionSize = 32 + region.Writable = true + return nil + }, + } + vm, _ := newMemSyscallVM(t, nil, []sbpf.InputRegion{region}) + // Copy the first 4 bytes over bytes 4..8 within the same region: the + // source translation sees the original buffer, the destination the clone. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+4, sbpf.VaddrInput, 4) + require.NoError(t, err) + require.NotNil(t, replaced, "the first write must trigger the copy-on-write hook") + require.Equal(t, []byte{1, 2, 3, 4, 1, 2, 3, 4}, replaced[:8]) + require.Equal(t, []byte{1, 2, 3, 4, 5, 6, 7, 8}, original, "the shared buffer must stay untouched") +} + +func TestSyscallMemcmpAndMemset(t *testing.T) { + input := make([]byte, 128) + for i := range input { + input[i] = byte(i) + } + vm, _ := newMemSyscallVM(t, input, nil) + + // memcmp of equal and differing 32-byte keys, result written at 96. + copy(input[32:64], input[0:32]) + _, err := SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0), int32(binary.LittleEndian.Uint32(input[96:]))) + input[63] = 0xff + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(31)-int32(0xff), int32(binary.LittleEndian.Uint32(input[96:]))) + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput+32, sbpf.VaddrInput, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) + + // memset 0xab over 33 bytes, then zero over 9 bytes. + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) + require.Equal(t, byte(97), input[97]) + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) + require.Equal(t, make([]byte, 9), input[70:79]) + require.Equal(t, bytes.Repeat([]byte{0xab}, 18), input[79:97]) +} + +// The byte loop the syscalls used before memcmpResult; kept as the reference. +func referenceMemcmp(a, b []byte) int32 { + for i := range a { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + +func TestMemcmpResultMatchesByteLoop(t *testing.T) { + rng := rand.New(rand.NewSource(7)) + for iter := 0; iter < 100000; iter++ { + n := rng.Intn(70) + a := make([]byte, n) + rng.Read(a) + b := append([]byte(nil), a...) + if n > 0 && rng.Intn(4) != 0 { + b[rng.Intn(n)] = byte(rng.Intn(256)) + if rng.Intn(2) == 0 { + b[rng.Intn(n)] ^= byte(1 + rng.Intn(255)) + } + } + if got, want := memcmpResult(a, b), referenceMemcmp(a, b); got != want { + t.Fatalf("n=%d a=%x b=%x: got %d want %d", n, a, b, got, want) + } + } + if memcmpResult([]byte{0xff}, []byte{0x00}) != 255 || memcmpResult([]byte{0x00}, []byte{0xff}) != -255 { + t.Fatal("memcmp must return the unsigned byte difference") + } + if memcmpResult(nil, nil) != 0 { + t.Fatal("empty compare must be 0") + } +} + +func TestMemsetBytes(t *testing.T) { + for _, n := range []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 100, 1023, 4096, 10001} { + for _, c := range []byte{0, 1, 0x7f, 0xff} { + mem := make([]byte, n) + for i := range mem { + mem[i] = byte(i) + } + memsetBytes(mem, c) + if !bytes.Equal(mem, bytes.Repeat([]byte{c}, n)) { + t.Fatalf("n=%d c=%d: %x", n, c, mem) + } + } + } +} From 7c8cfa3c1d8f610d4e99b4fa3dc083a05e8b5784 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:41:52 +0000 Subject: [PATCH 102/111] lthash: vectorize MixIn/MixOut with AVX2 on amd64 LtHash.MixIn/MixOut are 1024-lane uint16 add/subtract loops that run twice per modified account in the accounts delta hash (once for the old value, once for the new). The scalar loop costs ~560 ns per call in the sandbox and roughly 350 ns on Zen 5, so a block with ~10k modified accounts spends several milliseconds of worker CPU on lane arithmetic alone. On amd64 with AVX2 the lanes are now mixed with VPADDW/VPSUBW, 16 lanes per instruction, four vectors per iteration, unaligned loads and stores (28 ns per call here, 20x). Dispatch is a package variable set from cpu.X86.HasAVX2 (golang.org/x/sys is already a direct dependency); other architectures, CPUs without AVX2 and the purego build tag keep the portable loops, which remain the reference. Equals now compares the two arrays directly (runtime memequal) instead of a lane loop. Tests compare the assembly and the dispatched functions against the portable loops on random lanes including wrap-around values, check that MixOut inverts MixIn, that aliased operands behave, and that the generic fallback is selectable; go vet's asmdecl check passes and the package builds under -tags purego and GOARCH=arm64. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/lthash/lthash.go | 18 ++---- pkg/lthash/mix.go | 16 +++++ pkg/lthash/mix_amd64.go | 32 ++++++++++ pkg/lthash/mix_amd64.s | 56 +++++++++++++++++ pkg/lthash/mix_amd64_test.go | 51 +++++++++++++++ pkg/lthash/mix_generic.go | 6 ++ pkg/lthash/mix_test.go | 116 +++++++++++++++++++++++++++++++++++ 7 files changed, 283 insertions(+), 12 deletions(-) create mode 100644 pkg/lthash/mix.go create mode 100644 pkg/lthash/mix_amd64.go create mode 100644 pkg/lthash/mix_amd64.s create mode 100644 pkg/lthash/mix_amd64_test.go create mode 100644 pkg/lthash/mix_generic.go create mode 100644 pkg/lthash/mix_test.go diff --git a/pkg/lthash/lthash.go b/pkg/lthash/lthash.go index 9c4aaee1b..faa09a8a5 100644 --- a/pkg/lthash/lthash.go +++ b/pkg/lthash/lthash.go @@ -120,16 +120,15 @@ func (ltHash *LtHash) Clone() *LtHash { return new } +// MixIn adds other's 1024 lanes to ltHash, lane-wise modulo 2^16. The +// lane arithmetic is vectorized where the platform supports it (see mix.go). func (ltHash *LtHash) MixIn(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] + other.value[i] - } + mixIn(<Hash.value, &other.value) } +// MixOut subtracts other's lanes from ltHash, the inverse of MixIn. func (ltHash *LtHash) MixOut(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] - other.value[i] - } + mixOut(<Hash.value, &other.value) } func (ltHash *LtHash) Add(other *LtHash) *LtHash { @@ -143,12 +142,7 @@ func (ltHash *LtHash) Sub(other *LtHash) *LtHash { } func (ltHash *LtHash) Equals(other *LtHash) bool { - for i, element := range ltHash.value { - if element != other.value[i] { - return false - } - } - return true + return ltHash.value == other.value } func (ltHash *LtHash) Checksum() []byte { diff --git a/pkg/lthash/mix.go b/pkg/lthash/mix.go new file mode 100644 index 000000000..6cce9a4d9 --- /dev/null +++ b/pkg/lthash/mix.go @@ -0,0 +1,16 @@ +package lthash + +// mixInGeneric and mixOutGeneric are the portable lane loops. Every +// architecture-specific implementation must produce identical results: the +// lanes are independent uint16 additions and subtractions modulo 2^16. +func mixInGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] += src[i] + } +} + +func mixOutGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] -= src[i] + } +} diff --git a/pkg/lthash/mix_amd64.go b/pkg/lthash/mix_amd64.go new file mode 100644 index 000000000..f4818245c --- /dev/null +++ b/pkg/lthash/mix_amd64.go @@ -0,0 +1,32 @@ +//go:build amd64 && !purego + +package lthash + +import "golang.org/x/sys/cpu" + +// useAVX2 selects the vector lane loops. cpu.X86.HasAVX2 already includes +// the operating-system XSAVE/YMM-state check. Tests flip it to compare the +// two implementations on the same machine. +var useAVX2 = cpu.X86.HasAVX2 + +func mixIn(dst, src *[numElements]uint16) { + if useAVX2 { + mixInAVX2(dst, src) + return + } + mixInGeneric(dst, src) +} + +func mixOut(dst, src *[numElements]uint16) { + if useAVX2 { + mixOutAVX2(dst, src) + return + } + mixOutGeneric(dst, src) +} + +//go:noescape +func mixInAVX2(dst, src *[numElements]uint16) + +//go:noescape +func mixOutAVX2(dst, src *[numElements]uint16) diff --git a/pkg/lthash/mix_amd64.s b/pkg/lthash/mix_amd64.s new file mode 100644 index 000000000..2233de57e --- /dev/null +++ b/pkg/lthash/mix_amd64.s @@ -0,0 +1,56 @@ +//go:build amd64 && !purego + +#include "textflag.h" + +// The LtHash value is 1024 uint16 lanes = 2048 bytes = 16 iterations of +// four 32-byte YMM vectors. VPADDW/VPSUBW operate on 16-bit lanes modulo +// 2^16, exactly like the generic Go loop. Loads and stores are unaligned +// (VMOVDQU): LtHash values live inside Go structs with 2-byte alignment. + +// func mixInAVX2(dst, src *[1024]uint16) +TEXT ·mixInAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixin_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPADDW (SI)(AX*1), Y0, Y0 + VPADDW 32(SI)(AX*1), Y1, Y1 + VPADDW 64(SI)(AX*1), Y2, Y2 + VPADDW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixin_loop + VZEROUPPER + RET + +// func mixOutAVX2(dst, src *[1024]uint16) +TEXT ·mixOutAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixout_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPSUBW (SI)(AX*1), Y0, Y0 + VPSUBW 32(SI)(AX*1), Y1, Y1 + VPSUBW 64(SI)(AX*1), Y2, Y2 + VPSUBW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixout_loop + VZEROUPPER + RET diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go new file mode 100644 index 000000000..0cd378ea2 --- /dev/null +++ b/pkg/lthash/mix_amd64_test.go @@ -0,0 +1,51 @@ +//go:build amd64 && !purego + +package lthash + +import ( + "math/rand" + "testing" +) + +// TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has +// AVX2) against the portable loops so the comparison does not depend on the +// dispatch variable. +func TestMixAVX2AgainstGeneric(t *testing.T) { + if !useAVX2 { + t.Skip("no AVX2 on this machine") + } + rng := rand.New(rand.NewSource(6)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + want, got := *dst, *dst + mixInGeneric(&want, src) + mixInAVX2(&got, src) + if got != want { + t.Fatalf("mixInAVX2 diverges (iteration %d)", iter) + } + want, got = *dst, *dst + mixOutGeneric(&want, src) + mixOutAVX2(&got, src) + if got != want { + t.Fatalf("mixOutAVX2 diverges (iteration %d)", iter) + } + } +} + +// TestMixGenericFallbackSelectable makes sure the dispatch honours the flag, +// so a machine without AVX2 takes the portable path. +func TestMixGenericFallbackSelectable(t *testing.T) { + saved := useAVX2 + defer func() { useAVX2 = saved }() + useAVX2 = false + rng := rand.New(rand.NewSource(8)) + dst := randomLanes(rng) + src := randomLanes(rng) + want := *dst + mixInGeneric(&want, src) + mixIn(dst, src) + if *dst != want { + t.Fatal("generic fallback must be used when AVX2 is disabled") + } +} diff --git a/pkg/lthash/mix_generic.go b/pkg/lthash/mix_generic.go new file mode 100644 index 000000000..79e7b0bb4 --- /dev/null +++ b/pkg/lthash/mix_generic.go @@ -0,0 +1,6 @@ +//go:build !amd64 || purego + +package lthash + +func mixIn(dst, src *[numElements]uint16) { mixInGeneric(dst, src) } +func mixOut(dst, src *[numElements]uint16) { mixOutGeneric(dst, src) } diff --git a/pkg/lthash/mix_test.go b/pkg/lthash/mix_test.go new file mode 100644 index 000000000..272051689 --- /dev/null +++ b/pkg/lthash/mix_test.go @@ -0,0 +1,116 @@ +package lthash + +import ( + "math/rand" + "testing" +) + +func randomLanes(rng *rand.Rand) *[numElements]uint16 { + var lanes [numElements]uint16 + for i := range lanes { + switch rng.Intn(8) { + case 0: + lanes[i] = 0 + case 1: + lanes[i] = 0xffff + case 2: + lanes[i] = 0x8000 + default: + lanes[i] = uint16(rng.Uint32()) + } + } + return &lanes +} + +// TestMixMatchesGeneric checks the platform mixIn/mixOut against the +// portable loops, including wrap-around lanes, and that MixOut inverts MixIn. +func TestMixMatchesGeneric(t *testing.T) { + rng := rand.New(rand.NewSource(3)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + wantIn := *dst + mixInGeneric(&wantIn, src) + gotIn := *dst + mixIn(&gotIn, src) + if gotIn != wantIn { + t.Fatalf("mixIn diverges from the generic loop (iteration %d)", iter) + } + wantOut := *dst + mixOutGeneric(&wantOut, src) + gotOut := *dst + mixOut(&gotOut, src) + if gotOut != wantOut { + t.Fatalf("mixOut diverges from the generic loop (iteration %d)", iter) + } + roundTrip := gotIn + mixOut(&roundTrip, src) + if roundTrip != *dst { + t.Fatalf("mixOut does not invert mixIn (iteration %d)", iter) + } + } + // In-place: mixing a value into itself doubles every lane. + dst := randomLanes(rng) + want := *dst + for i := range want { + want[i] *= 2 + } + mixIn(dst, dst) + if *dst != want { + t.Fatal("mixIn with aliased operands must double every lane") + } + mixOut(dst, dst) + if *dst != [numElements]uint16{} { + t.Fatal("mixOut with aliased operands must clear every lane") + } +} + +func TestLtHashMixInMixOutAndEquals(t *testing.T) { + rng := rand.New(rand.NewSource(4)) + var a, b, c LtHash + a.value = *randomLanes(rng) + b.value = *randomLanes(rng) + c = *a.Clone() + c.MixIn(&b) + if c.Equals(&a) { + t.Fatal("mixing in a random value must change the hash") + } + c.MixOut(&b) + if !c.Equals(&a) { + t.Fatal("MixOut must undo MixIn") + } + c.value[numElements-1]++ + if c.Equals(&a) { + t.Fatal("Equals must see a last-lane difference") + } +} + +func BenchmarkMixIn(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixIn(dst, src) + } +} + +func BenchmarkMixInGeneric(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixInGeneric(dst, src) + } +} + +func BenchmarkMixOut(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixOut(dst, src) + } +} From 928f9207f82f7d7bccf936b612f60d33faf6afdf Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 103/111] test: preserve zero-length memory validation and repair VM fixtures --- pkg/sealevel/sealevel_test.go | 12 +++++++----- pkg/sealevel/syscalls_mem.go | 8 ++++++++ pkg/sealevel/syscalls_mem_test.go | 21 ++++++++++++++++++++- 3 files changed, 35 insertions(+), 6 deletions(-) diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ad64f3a9..18309927b 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -65,12 +65,14 @@ func TestInterpreter_Noop(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 29834a9ce..73f96fc67 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,6 +24,14 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + // Translate intentionally bypasses address validation for zero-length slices; + // Read/Write did not. Preserve the old syscall validation and error order. + if n == 0 { + if err := vm.Read(src, nil); err != nil { + return err + } + return vm.Write(dst, nil) + } srcMem, err := vm.Translate(src, n, false) if err != nil { return err diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 3b02daea0..6cf5b5ba9 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -137,10 +137,11 @@ func TestSyscallMemcmpAndMemset(t *testing.T) { require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) // memset 0xab over 33 bytes, then zero over 9 bytes. + sentinel := input[97] // The preceding memcmp result overwrote bytes 96..99. _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) - require.Equal(t, byte(97), input[97]) + require.Equal(t, sentinel, input[97]) _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) @@ -197,3 +198,21 @@ func TestMemsetBytes(t *testing.T) { } } } + +func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { + for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { + for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { + vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) + want := vm.Read(src, nil) + if want == nil { + want = vm.Write(dst, nil) + } + got := memmoveImplInternal(vm, dst, src, 0) + if want == nil { + require.NoError(t, got) + } else { + require.EqualError(t, got, want.Error()) + } + } + } +} From fdbc0198e73b4ec239c326f49ef1feef60e83085 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 104/111] test: cover unaligned and aliased AVX2 hash lanes --- pkg/lthash/mix_amd64_test.go | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go index 0cd378ea2..110c1b39a 100644 --- a/pkg/lthash/mix_amd64_test.go +++ b/pkg/lthash/mix_amd64_test.go @@ -5,6 +5,7 @@ package lthash import ( "math/rand" "testing" + "unsafe" ) // TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has @@ -49,3 +50,25 @@ func TestMixGenericFallbackSelectable(t *testing.T) { t.Fatal("generic fallback must be used when AVX2 is disabled") } } + +func TestMixAVX2UnalignedAndAliased(t *testing.T) { + if !useAVX2 { + t.Skip("AVX2 unavailable") + } + rng := rand.New(rand.NewSource(19)) + for off := 0; off < 32; off += 2 { + storage := make([]byte, numElements*2+32) + dst := (*[numElements]uint16)(unsafe.Pointer(&storage[off])) + *dst = *randomLanes(rng) + want := *dst + mixInGeneric(&want, &want) + mixInAVX2(dst, dst) + if *dst != want { + t.Fatalf("aliased addition offset %d", off) + } + mixOutAVX2(dst, dst) + if *dst != ([numElements]uint16{}) { + t.Fatalf("aliased subtraction offset %d", off) + } + } +} From 5c7a83aadfa94a78d91423e475d35dcd643e4586 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:15:39 -0500 Subject: [PATCH 105/111] sbpf: account for resolved call targets in program cache cost --- pkg/sbpf/program.go | 2 +- pkg/sbpf/program_memory_test.go | 12 ++++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) create mode 100644 pkg/sbpf/program_memory_test.go diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index 969a15469..a353a4594 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -42,7 +42,7 @@ func (p *Program) MemoryBytes() uint64 { if p == nil { return 0 } - total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + uint64(len(p.CallTargets))*8 if len(p.RO) == 0 { total += uint64(len(p.TextBytes)) } diff --git a/pkg/sbpf/program_memory_test.go b/pkg/sbpf/program_memory_test.go new file mode 100644 index 000000000..3360cf45d --- /dev/null +++ b/pkg/sbpf/program_memory_test.go @@ -0,0 +1,12 @@ +package sbpf + +import "testing" + +func TestProgramMemoryIncludesResolvedCalls(t *testing.T) { + p := &Program{Text: make([]Slot, 20)} + before := p.MemoryBytes() + p.ResolveCallTargets() + if got := p.MemoryBytes() - before; got != 20*8 { + t.Fatalf("resolved-call cache bytes = %d, want 160", got) + } +} From ad31b3d2211aa9779128834f368d303e2478d52b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:25:52 -0500 Subject: [PATCH 106/111] test: retain syscall differential coverage and execution reproduction guide --- docs/sbpf-interpreter-benchmarks.md | 80 +++++++++++++---------------- pkg/sealevel/syscalls_mem_test.go | 31 +++++++++++ 2 files changed, 68 insertions(+), 43 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 73eb970e5..d78653991 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -58,46 +58,40 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ -benchtime=1s -count=5 -benchmem ``` -## Recorded Alpenglow blocks - -Each run bootstrapped a fresh isolated AccountsDB from the same public full -snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay -covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and -142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, -nice 19. Two paired runs used baseline/candidate then candidate/baseline order. -The live validator's AccountsDB and configuration were not used or changed. - -All 233 per-slot bank hashes matched between baseline and candidate in both -pairs, excluding the run-specific comment header in `bankhash.log`. - -| Work measured | Pair 1 before → after | Pair 2 before → after | -|---|---:|---:| -| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | -| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | -| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | -| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | -| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | - -The sample includes blocks around 40–46 million CU. Three inspected non-empty -blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and -Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries -attribute whole-transaction CU to every participating program and must not be -summed as if they were exclusive per-program execution costs. - -Whole-block p95 did not improve, and the p99 changes are small/variable. The -slowest candidate blocks in the first pair contained no sBPF execution; their -large timers were dispatch and signature verification. These single-CPU replay -results do not establish a live voting/FAST improvement or production parallel -replay latency. They exclude network wait from ProcessBlock and are not elapsed -end-to-end catch-up times. - -## Correctness and limits - -- Native Zen 5 baseline and candidate differential outputs match for 100,000 - deterministic generated programs; candidate pool-zero checks pass. -- Baseline/candidate workload effects and CU charges match. Targeted race tests - for the interpreter, loader, and workload harness pass; vet passes. -- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; - therefore no complete sealevel test-suite pass is claimed. Broader conformance - testing remains separate from this performance experiment. -- No candidate was deployed and no validator restart was needed. +## Correctness and comparison boundaries + +The generated-program harness compares return values, errors, CU usage and memory +for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate +and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and +verifier semantics changes are excluded from this performance work. + +For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, +identical transaction parallelism, and the same slot interval. Compare normalized +per-slot bank hashes and slot sets before interpreting timings. Compare exact +`ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate +run order and retain raw outputs plus commit IDs outside the merge diff. + +The PR description links the recorded Alpenglow replay results and raw evidence. +Single-core shared-host results do not establish multicore contention or live FAST +inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe +a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now +supplies its execution context's compute meter. + +## Memory syscalls and LtHash + +Memory syscalls retain CU charges, source-before-destination error order, +zero-length behavior, memcpy overlap rejection and memmove overlap support. +Tests cover copy-on-write/growing regions and differential memory/CU results. + +LtHash uses AVX2 only when supported by both CPU and OS; other architectures and +`-tags purego` use portable loops. The vector path preserves 16-bit wraparound and +in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cover +both paths. Component speedups are not block-latency speedups. + +```sh +go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test -tags purego ./pkg/lthash +go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' +SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 +go test ./pkg/lthash -run '^$' -bench BenchmarkMix -benchmem -count=5 +``` diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 6cf5b5ba9..9fe7d22b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -216,3 +216,34 @@ func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { } } } + +// Compare the syscall with the old read-then-write path, including charged CU. +func TestMemoryCopyDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(127)) + for i := 0; i < 300; i++ { + data := make([]byte, 128) + rng.Read(data) + actual, want := append([]byte(nil), data...), append([]byte(nil), data...) + vm, ctx := newMemSyscallVM(t, actual, nil) + ref, refCtx := newMemSyscallVM(t, want, nil) + src, dst, n := uint64(rng.Intn(145)), uint64(rng.Intn(145)), uint64(rng.Intn(80)) + src += sbpf.VaddrInput + dst += sbpf.VaddrInput + _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) + wantErr := MemOpConsume(refCtx, n) + if wantErr == nil { + buf := make([]byte, n) + wantErr = ref.Read(src, buf) + if wantErr == nil { + wantErr = ref.Write(dst, buf) + } + } + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + require.Equal(t, want, actual) + require.Equal(t, refCtx.ComputeMeter.Remaining(), ctx.ComputeMeter.Remaining()) + } +} From 1ac88785b702657f5a66e8a50e88adda6ff1892c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:00:42 -0500 Subject: [PATCH 107/111] test: trim execution benchmark experiments and documentation --- docs/sbpf-interpreter-benchmarks.md | 49 ++------- docs/sha256-syscall.md | 107 ++++++------------- pkg/sealevel/program_workloads_bench_test.go | 2 +- pkg/sealevel/syscalls_sha256_bench_test.go | 81 +------------- 4 files changed, 43 insertions(+), 196 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index d78653991..095eb0b06 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -1,40 +1,11 @@ -# Interpreter performance validation +# Execution performance validation -This experiment compares `9db7dcb` plus identical benchmark code with the same -base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift -semantics patch is excluded. Three VASA tests were updated to pass the new -`*[16]uint64` register type; no further production optimization was added during -this measurement pass. +## Program workloads -## Individual programs - -Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five -one-second samples per variant with alternating run order. The existing validator -continued running. CPU 15 shares a physical core with CPU 7; these are shared-host -measurements rather than an isolated-machine throughput ceiling. - -The harness has a preloaded program cache and asserts a cache hit before timing. -Every invocation gets fresh account data and execution context. It measures -instruction setup, serialization, execution and publication together; it is not -just the interpreter loop. Timing instrumentation and instruction recording are -enabled identically on both versions. Tests check arithmetic return values, -post-transfer balances, allocation results, and recorded CPI. Requested budgets -are equal and the measured CU charges match between variants. - -| Workload | CU | Median before → after | Speedup | -|---|---:|---:|---:| -| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | -| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | -| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | -| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | -| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | -| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | -| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | - -The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. -The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from -the earlier SPL Token loader-only benchmark: they use different program binaries, -instructions, and include execution-context setup. +The program harness measures instruction setup, serialization, execution and +publication with a warm program cache and fresh account data per invocation. +It checks return values, account updates, CPI and CU consumption. Loader-only +benchmarks separately measure VM execution and program loading. Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` and `token2022.so` to enable those external fixtures. Without it, the in-repository @@ -71,11 +42,9 @@ per-slot bank hashes and slot sets before interpreting timings. Compare exact `ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate run order and retain raw outputs plus commit IDs outside the merge diff. -The PR description links the recorded Alpenglow replay results and raw evidence. +Record tested commit IDs, hardware, Go version, affinity and parallelism with results. Single-core shared-host results do not establish multicore contention or live FAST -inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe -a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now -supplies its execution context's compute meter. +inclusion gains. ## Memory syscalls and LtHash @@ -89,7 +58,7 @@ in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cov both paths. Component speedups are not block-latency speedups. ```sh -go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader go test -tags purego ./pkg/lthash go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 10c1b59b2..4045b2bac 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -1,89 +1,44 @@ -# SHA-256 syscall overhead +# SHA-256 syscall validation -The syscall decodes the already-translated slice descriptor array directly and -writes the final digest into the translated output buffer. It retains streaming -SHA-256, slice order, memory translations, CU charges and validation order. Output -is written only after all inputs have been read, preserving overlapping-buffer -behavior. No special case for a particular on-chain program is introduced. +The syscall decodes the translated slice descriptors directly and writes the +final digest into the translated output buffer. It retains streaming SHA-256, +slice order, memory translations, CU charges and validation order. Output is +written only after all inputs have been read, preserving overlapping-buffer +behavior. -A bounded 55-byte input-buffer prototype was slower than this simpler path and -is retained only as a benchmark comparison. The baseline reference is copied -from combined review commit `bd17683a`. - -Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; -medians below. Each benchmark runs serially through a real interpreter's memory -translation and CU meter, with VM creation outside the timed region. This does -not include VM instruction dispatch, a complete program, or block replay. - -| Input | Original | Direct decoding/output | Buffered prototype | -|---|---:|---:|---:| -| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | -| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | -| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | -| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | - -The two-slice case removes four allocations (112 bytes) per call. This is an -ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native -Zen 5 and captured heavy-block replay before deployment decisions. - -Reproduce with: - -```sh -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 -go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' -``` +The frozen reference implementation in the test harness supports differential +checks and before/after benchmarks. Both variants use the same VM, including +its contiguous-region bounds-overflow fix. Differential tests compare hashes, return/error values, remaining CU and all input/output memory over valid inputs, invalid descriptors/addresses, depleted budgets and output aliasing. The existing SHA program fixture also executes. -Testing exposed pre-existing overflow in contiguous VM region bounds checks; -that correction and its regression test are a separate preceding commit. Both -benchmark variants use the corrected VM. The old SHA fixture also needed its -compute-meter pointer initialized for the current interpreter API. +## Benchmarks -## Zen 5 acceleration and captured loop +`BenchmarkSha256Syscall` measures the syscall through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. It covers +empty, single-slice, multiple-slice and larger inputs; it excludes instruction +dispatch and transaction execution. -The live Go 1.26.4 validator on Ryzen 7 9700X had its actual -`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is -already active. No validator runtime setting was changed. +`BenchmarkSha256CapturedLoop` uses a captured SBF v0 hash loop with 1,000 +iterations and zero initial state. It retains descriptor setup, stack accesses, +digest copying, counter updates and branching. Its test checks the result against +a Go hash chain and checks equal CU consumption for both syscall implementations. +The captured ELF hash and extracted instruction range are recorded in the test. +Transaction loading, CPI and the rest of the original program are excluded. -The isolated loop harness copies text slots 518–545 from the captured SBF v0 -program, resolves the SHA syscall relocation, and supplies 1,000 iterations and -zero initial state. It retains descriptor setup, stack accesses, digest copying, -counter update and loop branching. A test checks its result against a Go hash -chain and checks equal CU consumption for both syscall implementations. This -excludes transaction loading, account dependencies, CPI and the remaining program. - -A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, -affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. -Three 150 ms samples (medians, per hash iteration): - -| Isolated loop | Time | -|---|---:| -| Original syscall | 234.5 ns | -| Optimized syscall | 165.9 ns | -| Dispatch-only diagnostic control | 109.2 ns | -| Go hash chain without VM | 54.69 ns | - -The optimized loop takes about 29% less time. The dispatch-only control replaces -the syscall with a no-op: it omits hashing, translations and syscall CU charging, -and is only an overhead diagnostic, never a valid execution implementation. -The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; -the buffered-input prototype remained slower at 71–73 ns. These short tests -share a host with other processes; they are not isolated-core latency guarantees. - -A separate short CPU profile of the optimized loop attributed 36.2% cumulative -sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI -compression routine. Most remaining sampled work was VM execution: instruction -dispatch/decoding, stack address translation, loads/stores and compute metering. -Cumulative and flat percentages overlap and must not be added. This profile is -of the harness, not of full-block replay or the live validator. - -The next execution experiment should target measured VM overhead and then replay -captured blocks; these results do not justify a claimed 29% block-time improvement. +The dispatch-only control omits hashing, translations and syscall CU charging; +it is an overhead diagnostic, not a valid execution implementation. The raw Go +hash chain provides another comparison outside the VM. Neither control can +establish a full-block speedup. ```sh -go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +go test ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256|TestSha256CapturedLoop' +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256(Syscall|CapturedLoop)$' -benchtime=1s -count=5 -benchmem ``` + +Alternate baseline/candidate run order and record commit IDs, Go version, +hardware and CPU affinity. Report component timings separately from full-program +and replay timings; hardware SHA acceleration also affects these results. diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go index cc6b83e78..649d35ddd 100644 --- a/pkg/sealevel/program_workloads_bench_test.go +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -19,7 +19,7 @@ import ( "github.com/stretchr/testify/require" ) -// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// External ELF inputs are pinned by SHA-256 in docs/sbpf-interpreter-benchmarks.md. // No live account writes or network calls occur in this harness. Each invocation // gets fresh account data; the program cache is warm and shared between runs. type programWorkload struct { diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go index b4240f54e..6c972f37f 100644 --- a/pkg/sealevel/syscalls_sha256_bench_test.go +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -14,8 +14,6 @@ import ( // Frozen syscall implementation from bd17683a; keep independent for differential tests. func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256BaselineReference") - if valsLen > cu.CUSha256MaxSlices { return syscallErr(SyscallErrTooManySlices) } @@ -73,81 +71,6 @@ func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) return syscallSuccess(0) } -// Experimental bounded-buffer variant retained only for benchmark comparison. -func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256SmallInputReference") - - if valsLen > cu.CUSha256MaxSlices { - return syscallErr(SyscallErrTooManySlices) - } - - execCtx := executionCtx(vm) - err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) - if err != nil { - return syscallCuErr() - } - - hashResult, err := vm.Translate(resultsAddr, 32, true) - if err != nil { - return syscallErr(err) - } - - hasher := sha256.New() - // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only - // this bounded case; larger inputs retain streaming hashing. - var small [55]byte - buffered := 0 - streaming := false - if valsLen > 0 { - var vals []byte - - // The data at 'valsAddr' consists of an array of 'slice references', which consists - // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that - // refers to an input value to hash. - // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above - vals, err = vm.Translate(valsAddr, valsLen*16, false) - if err != nil { - return syscallErr(err) - } - - var data []byte - - for count := uint64(0); count < valsLen; count++ { - - offset := count * 16 - vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} - - data, err = vm.Translate(vec.Addr, vec.Len, false) - if err != nil { - return syscallErr(err) - } - - cost := max(vec.Len/2, cu.CUMemOpBaseCost) - err = execCtx.ComputeMeter.Consume(cost) - if err != nil { - return syscallCuErr() - } - - if !streaming && len(data) <= len(small)-buffered { - buffered += copy(small[buffered:], data) - } else { - if !streaming { - hasher.Write(small[:buffered]) - streaming = true - } - hasher.Write(data) - } - } - } - if streaming { - hasher.Sum(hashResult[:0]) - } else { - digest := sha256.Sum256(small[:buffered]) - copy(hashResult, digest[:]) - } - return syscallSuccess(0) -} - type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { @@ -211,7 +134,7 @@ func TestSha256SyscallDifferential(t *testing.T) { var wantMem []byte var wantRet, wantCU uint64 var wantErr string - for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + for k, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { buf := append([]byte(nil), mem...) vm, ctx := sha256VM(buf, budget) ret, err := fn(vm, a, n, out) @@ -242,7 +165,7 @@ func BenchmarkSha256Syscall(b *testing.B) { for _, variant := range []struct { name string fn sha256Call - }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}} { b.Run(tc.name+"/"+variant.name, func(b *testing.B) { mem, a, n, out := sha256Fixture(tc.sizes) vm, ctx := sha256VM(mem, ^uint64(0)) From 23a67db4c2bfec5b08b3d7aec471f7b9c83379dd Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:46:59 -0500 Subject: [PATCH 108/111] sbpf: restore v2 memory opcodes and differential coverage --- docs/sbpf-interpreter-benchmarks.md | 11 ++- pkg/sbpf/interpreter.go | 80 ++++++++++++++++++++ pkg/sbpf/interpreter_v2_test.go | 113 ++++++++++++++++++++++++++++ pkg/sbpf/perf_differential_test.go | 90 +++++++++++++++++++--- 4 files changed, 281 insertions(+), 13 deletions(-) create mode 100644 pkg/sbpf/interpreter_v2_test.go diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 095eb0b06..c399f4da5 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -32,9 +32,14 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ ## Correctness and comparison boundaries The generated-program harness compares return values, errors, CU usage and memory -for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate -and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and -verifier semantics changes are excluded from this performance work. +for 100,000 generated programs, evenly divided across SBPF v0–v3. The generator +uses v2-specific memory, arithmetic and constant-loading encodings; the dump +reports verifier rejection or execution results, and logs accepted counts per version. +Run the same harness on both trees: set `SBPF_DIFF_OUT` separately on the reference +and candidate and compare the files. `SBPF_CHECK_POOL_ZERO=1` additionally checks +the candidate’s clear-on-return pool invariant; older references may clear on +acquisition instead. ARSH and verifier semantics changes are excluded from this +performance work. For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, identical transaction parallelism, and the same slot interval. Compare normalized diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e7e28fda0..e27392d7c 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1299,6 +1299,86 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { var err error switch ins.Op() { + // In v2 these encodings are memory operations, not MUL/DIV/MOD. + // Run dispatches their non-v2 arithmetic forms on the hot path. + case OpLd1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt1BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(ins.Uimm())) + pc++ + case OpSt1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(r[ins.Src()])) + pc++ + case OpLd2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt2BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(ins.Uimm())) + pc++ + case OpSt2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(r[ins.Src()])) + pc++ + case OpLd8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = v + pc++ + case OpSt8BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, uint64(ins.Imm())) + pc++ + case OpSt8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, r[ins.Src()]) + pc++ case OpDiv32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) pc++ diff --git a/pkg/sbpf/interpreter_v2_test.go b/pkg/sbpf/interpreter_v2_test.go new file mode 100644 index 000000000..39fd1e161 --- /dev/null +++ b/pkg/sbpf/interpreter_v2_test.go @@ -0,0 +1,113 @@ +package sbpf + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" + "github.com/stretchr/testify/require" +) + +// These byte values alias arithmetic instructions outside v2. Exercise all +// relocated memory instructions through Run, not just the cold handler. +func TestInterpreterV2MemoryOpcodes(t *testing.T) { + for _, width := range []int{1, 2, 4, 8} { + loads := map[int]uint8{1: OpLd1BReg, 2: OpLd2BReg, 4: OpLd4BReg, 8: OpLd8BReg} + immediates := map[int]uint8{1: OpSt1BImm, 2: OpSt2BImm, 4: OpSt4BImm, 8: OpSt8BImm} + registers := map[int]uint8{1: OpSt1BReg, 2: OpSt2BReg, 4: OpSt4BReg, 8: OpSt8BReg} + for _, kind := range []string{"load", "store_imm", "store_reg"} { + for _, region := range []string{"heap", "stack", "input", "rodata", "unmapped"} { + t.Run(fmt.Sprintf("%s/%d/%s", kind, width, region), func(t *testing.T) { + const value = uint64(0xfedcba9876543210) + immediate := uint32(0xf123abcd) + var addr uint64 + switch region { + case "heap": + addr = VaddrHeap + 9 + case "stack": + addr = VaddrStack + 9 + case "input": + addr = VaddrInput + 9 + case "rodata": + addr = VaddrProgram + 9 + case "unmapped": + addr = 0x600000009 + } + // Negative offsets and unaligned addresses must behave identically to + // the original interpreter, including the post-instruction exception PC. + text := diffLoadImm64(5, addr+3, sbpfver.SbpfVersionV2) + text = append(text, diffLoadImm64(6, value, sbpfver.SbpfVersionV2)...) + var op Slot + switch kind { + case "load": + op = slot(loads[width], 0, 5, -3, 0) + case "store_imm": + op = slot(immediates[width], 5, 0, -3, immediate) + case "store_reg": + op = slot(registers[width], 5, 6, -3, 0) + } + text = append(text, op, slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(text, sbpfver.SbpfVersionV2) + p.RO = bytes.Repeat([]byte{0xa5}, 32) + require.NoError(t, p.Verify()) + cm := cu.NewComputeMeter(100) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32, Input: bytes.Repeat([]byte{0xa5}, 32), ComputeMeter: &cm, Syscalls: noSyscalls}) + defer ip.Finish() + var memory []byte + switch region { + case "heap": + memory = ip.heap + case "stack": + memory = ip.stack.mem + case "input": + memory = ip.input + case "rodata": + memory = p.RO + } + if memory != nil { + // Use Write for writable VM storage so pooled-memory tracking is kept. + if region != "rodata" { + require.NoError(t, ip.Write(addr-9, bytes.Repeat([]byte{0xa5}, 32))) + } + } + before := append([]byte(nil), memory...) + ret, used, err := ip.Run() + if region == "unmapped" || (region == "rodata" && kind != "load") { + require.Error(t, err) + var exc *Exception + require.ErrorAs(t, err, &exc) + require.Equal(t, int64(5), exc.PC) + var access ExcBadAccess + require.ErrorAs(t, err, &access) + require.Equal(t, addr, access.Addr) + require.Equal(t, uint64(width), access.Size) + require.Equal(t, kind != "load", access.Write) + require.Equal(t, uint64(95), cm.Remaining()) + require.Equal(t, before, memory) + return + } + require.NoError(t, err) + require.Equal(t, uint64(6), used) + require.Equal(t, uint64(94), cm.Remaining()) + var encoded [8]byte + if kind == "load" { + copy(encoded[:], before[9:9+width]) + require.Equal(t, binary.LittleEndian.Uint64(encoded[:]), ret) + } else { + v := value + if kind == "store_imm" { + signed := int32(immediate) + v = uint64(int64(signed)) + } + binary.LittleEndian.PutUint64(encoded[:], v) + copy(before[9:9+width], encoded[:width]) + } + require.Equal(t, before, memory) + }) + } + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go index 1d679d4e4..0d072f564 100644 --- a/pkg/sbpf/perf_differential_test.go +++ b/pkg/sbpf/perf_differential_test.go @@ -2,7 +2,6 @@ package sbpf import ( "bufio" - "encoding/binary" "fmt" "hash/fnv" "math/rand" @@ -97,6 +96,15 @@ func diffRegistry(h uint32) (Syscall, bool) { return nil, false } +// v2 replaces LDDW with MOV32 + HOR64. MOV32 avoids sign-extending the +// low word before ORing in the high word. +func diffLoadImm64(dst uint8, value uint64, ver uint32) []Slot { + if ver == sbpfver.SbpfVersionV2 { + return []Slot{slot(OpMov32Imm, dst, 0, 0, uint32(value)), slot(OpHor64Imm, dst, 0, 0, uint32(value>>32))} + } + return []Slot{slot(OpLddw, dst, 0, 0, uint32(value)), slot(0, 0, 0, 0, uint32(value>>32))} +} + func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 imm := func() uint32 { @@ -117,6 +125,29 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + if ver == sbpfver.SbpfVersionV2 { + // Arithmetic and memory encodings both change in v2. Keep generated ALU + // operations arithmetic rather than generating unintended memory accesses. + replacements := map[uint8]uint8{ + OpMul32Imm: OpLmul32Imm, OpMul32Reg: OpLmul32Reg, + OpMul64Imm: OpLmul64Imm, OpMul64Reg: OpLmul64Reg, + OpDiv32Imm: OpUdiv32Imm, OpDiv32Reg: OpUdiv32Reg, + OpDiv64Imm: OpUdiv64Imm, OpDiv64Reg: OpUdiv64Reg, + OpMod32Imm: OpUrem32Imm, OpMod32Reg: OpUrem32Reg, + OpMod64Imm: OpUrem64Imm, OpMod64Reg: OpUrem64Reg, + } + filtered := alu64[:0] + for _, op := range alu64 { + if op == OpNeg32 || op == OpNeg64 || op == OpLe { + continue + } + if replacement, ok := replacements[op]; ok { + op = replacement + } + filtered = append(filtered, op) + } + alu64 = filtered + } jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} switch rng.Intn(10) { @@ -126,7 +157,7 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { if op == OpLe || op == OpBe { i = []uint32{16, 32, 64}[rng.Intn(3)] } - if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm || op == OpUdiv32Imm || op == OpUdiv64Imm || op == OpUrem32Imm || op == OpUrem64Imm) && i == 0 { i = 3 } switch op { @@ -138,6 +169,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(op, reg(), reg(), 0, i)} case 4: // load ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg} + } // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random var base uint8 var off int16 @@ -154,6 +188,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} case 5: // store ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} + } var base uint8 var off int16 switch rng.Intn(5) { @@ -184,11 +221,11 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { default: // set up pointer registers switch rng.Intn(3) { case 0: // r5 = heap - return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + return diffLoadImm64(5, VaddrHeap, ver) case 1: // r1 = input + small - return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + return diffLoadImm64(1, VaddrInput+uint64(rng.Intn(64)), ver) default: // r9 = random 64-bit - return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + return diffLoadImm64(9, uint64(rng.Uint32())|uint64(rng.Intn(6))<<32, ver) } } } @@ -215,10 +252,14 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { } } // function + store := uint8(OpStxdw) + if ver == sbpfver.SbpfVersionV2 { + store = OpSt8BReg + } body = append(body, slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), slot(OpXor64Reg, 7, 6, 0, 0), - slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(store, 10, 7, int16(-8-rng.Intn(64)), 0), slot(OpExit, 0, 0, 0, 0)) p := mkProgram(body, ver) if ver < sbpfver.SbpfVersionV3 { @@ -231,6 +272,29 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { return p } +// Keep this check in the ordinary suite: merely selecting v2 is insufficient +// if its programs still contain legacy memory opcodes or LDDW and never run. +func TestDifferentialV2Generator(t *testing.T) { + rng := rand.New(rand.NewSource(12345)) + seen := make(map[uint8]bool) + for i := 0; i < 1000; i++ { + p := genProgram(rng, sbpfver.SbpfVersionV2) + if err := p.Verify(); err != nil { + continue + } + for _, ins := range p.Text { + seen[ins.Op()] = true + } + } + for _, op := range []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg, + OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, + OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} { + if !seen[op] { + t.Errorf("no verifier-accepted v2 program contains opcode %#x", op) + } + } +} + func memHash(bs ...[]byte) uint64 { h := fnv.New64a() for _, b := range bs { @@ -255,8 +319,10 @@ func TestDifferentialDump(t *testing.T) { rng := rand.New(rand.NewSource(12345)) const N = 100000 generated, verified := 0, 0 + var verifiedByVersion [4]int for i := 0; i < N; i++ { - ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + // Equal representation of every version, independent of RNG consumption. + ver := uint32(i % 4) p := genProgram(rng, ver) generated++ if err := p.Verify(); err != nil { @@ -264,6 +330,7 @@ func TestDifferentialDump(t *testing.T) { continue } verified++ + verifiedByVersion[ver]++ resolveCallTargetsIfSupported(p) input := make([]byte, 700) for j := range input { @@ -332,7 +399,10 @@ func TestDifferentialDump(t *testing.T) { heapPool.Put(hp) } } - t.Logf("generated=%d verified=%d", generated, verified) + for ver, count := range verifiedByVersion { + if count == 0 { + t.Errorf("no verifier-accepted programs for v%d", ver) + } + } + t.Logf("generated=%d verified=%d verified_by_version=%v", generated, verified, verifiedByVersion) } - -var _ = binary.LittleEndian From b182b33b3049f4f7de73a974346074bb3f503399 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:29:23 -0500 Subject: [PATCH 109/111] sbpf: guard pooled write tracking beyond bitmap capacity --- pkg/sbpf/fastmem.go | 3 +++ pkg/sbpf/interpreter.go | 9 ++++++++ pkg/sbpf/pooling_test.go | 46 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+) diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go index 421f3ed42..1f15dab3b 100644 --- a/pkg/sbpf/fastmem.go +++ b/pkg/sbpf/fastmem.go @@ -25,6 +25,9 @@ type memRegion struct { // emptyRegion never matches any access. var emptyRegion = memRegion{gapShift: 63} +// A uint64 dirty bitmap can describe exactly 64 pages of 4 KiB. +const fastDirtyBytes = 64 * 4096 + const numFastRegions = 6 // index 5 is a permanently empty catch-all // fastRead returns a host pointer for a size-byte read at vma, or nil if the diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e27392d7c..96d3cb25b 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -162,6 +162,15 @@ func (ip *Interpreter) initRegions() { if len(ip.heap) != 0 { ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} } + // Finish relies on complete write tracking before returning pooled storage. + // Larger heaps (or a future larger stack) must use translateInternal's byte + // ranges: shifting the fast-path bitmap beyond page 63 silently loses writes. + // Reads remain fast; current <=256 KiB writable mappings are unchanged. + for _, idx := range []uint64{VaddrStack >> 32, VaddrHeap >> 32} { + if ip.regions[idx].wlen > fastDirtyBytes { + ip.regions[idx].wlen = 0 + } + } if len(ip.inputRegions) == 0 && len(ip.input) != 0 { ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index c4bc5eb0d..4a922a70f 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -2,10 +2,13 @@ package sbpf import ( "bytes" + "encoding/binary" + "fmt" "sync" "testing" "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" "github.com/stretchr/testify/require" ) @@ -87,3 +90,46 @@ func BenchmarkVMCreateAndFinish(b *testing.B) { }) } } + +// Exercise actual stores at and beyond the bitmap boundary. Inspect the returned +// buffer directly: sync.Pool is permitted to discard entries, so a subsequent Get +// alone would not reliably detect a missed clear. +func TestPooledHeapDirtyBitmapBoundary(t *testing.T) { + oldUsePool, oldPool := UsePool, heapPool + UsePool = true + heapPool = &sync.Pool{New: func() any { return newHeap() }} + t.Cleanup(func() { UsePool, heapPool = oldUsePool, oldPool }) + for _, size := range []int{fastDirtyBytes, fastDirtyBytes + 1, 2 * fastDirtyBytes} { + for _, ver := range []uint32{sbpfver.SbpfVersionV0, sbpfver.SbpfVersionV2, sbpfver.SbpfVersionV3} { + t.Run(fmt.Sprintf("%d/v%d", size, ver), func(t *testing.T) { + offsets := []int{0, size - 8} + if size >= fastDirtyBytes+8 { + offsets = append(offsets, fastDirtyBytes-4, fastDirtyBytes) + } + var text []Slot + op := uint8(OpStdw) + if ver == sbpfver.SbpfVersionV2 { + op = OpSt8BImm + } + for _, off := range offsets { + text = append(text, diffLoadImm64(5, VaddrHeap+uint64(off), ver)...) + text = append(text, slot(op, 5, 0, 0, 0x12345678)) + } + text = append(text, slot(OpExit, 0, 0, 0, 0)) + program := mkProgram(text, ver) + require.NoError(t, program.Verify()) + meter := cu.NewComputeMeter(100) + ip := NewInterpreter(program, &VMOpts{HeapMax: size, ComputeMeter: &meter, Syscalls: noSyscalls}) + // Always clear test storage on failure so later cases cannot inherit dirt. + defer func() { clear(ip.heap) }() + require.NotNil(t, ip.fastRead(VaddrHeap+uint64(size-8), 8)) + _, _, err := ip.Run() + require.NoError(t, err) + require.Equal(t, uint64(0x12345678), binary.LittleEndian.Uint64(ip.heap[offsets[len(offsets)-1]:])) + ip.Finish() + require.True(t, bytes.Equal(make([]byte, size), ip.heap), "Finish must clear every written byte before pooling") + require.Equal(t, size <= fastDirtyBytes, ip.regions[VaddrHeap>>32].wlen != 0) + }) + } + } +} From 9585fa44b236b7b8bed840791557a187fac821f6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:59:32 -0500 Subject: [PATCH 110/111] sealevel: match Agave zero-length memory copy behavior --- pkg/sealevel/syscalls_mem.go | 9 +++------ pkg/sealevel/syscalls_mem_test.go | 23 +++++++++++------------ 2 files changed, 14 insertions(+), 18 deletions(-) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 73f96fc67..14a1e39b0 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,13 +24,10 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { - // Translate intentionally bypasses address validation for zero-length slices; - // Read/Write did not. Preserve the old syscall validation and error order. + // Agave's touch_slice_mut / translate_slice return empty slices before + // address lookup for zero length. CU was already charged by the syscall. if n == 0 { - if err := vm.Read(src, nil); err != nil { - return err - } - return vm.Write(dst, nil) + return nil } srcMem, err := vm.Translate(src, n, false) if err != nil { diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 9fe7d22b7..d375a42b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -199,19 +199,18 @@ func TestMemsetBytes(t *testing.T) { } } -func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { +func TestSyscallMemoryZeroLengthSkipsAddressValidation(t *testing.T) { for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { - vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) - want := vm.Read(src, nil) - if want == nil { - want = vm.Write(dst, nil) - } - got := memmoveImplInternal(vm, dst, src, 0) - if want == nil { - require.NoError(t, got) - } else { - require.EqualError(t, got, want.Error()) + for _, fn := range []func(sbpf.VM, uint64, uint64, uint64) (uint64, error){SyscallMemmoveImpl, SyscallMemcpyImpl} { + input := bytes.Repeat([]byte{0x42}, 32) + vm, ctx := newMemSyscallVM(t, input, nil) + before := ctx.ComputeMeter.Remaining() + ret, err := fn(vm, dst, src, 0) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, before-cu.CUMemOpBaseCost, ctx.ComputeMeter.Remaining()) + require.Equal(t, bytes.Repeat([]byte{0x42}, 32), input) } } } @@ -231,7 +230,7 @@ func TestMemoryCopyDifferential(t *testing.T) { dst += sbpf.VaddrInput _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) wantErr := MemOpConsume(refCtx, n) - if wantErr == nil { + if wantErr == nil && n > 0 { buf := make([]byte, n) wantErr = ref.Read(src, buf) if wantErr == nil { From 52a1aad8c0791877a92c05ced08421e80496ca01 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:00:21 -0500 Subject: [PATCH 111/111] sealevel: track sibling header writes in pooled VM memory --- pkg/sealevel/syscalls_call.go | 2 +- pkg/sealevel/syscalls_sibling_pool_test.go | 42 ++++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) create mode 100644 pkg/sealevel/syscalls_sibling_pool_test.go diff --git a/pkg/sealevel/syscalls_call.go b/pkg/sealevel/syscalls_call.go index 0ffe7b0c3..761069485 100644 --- a/pkg/sealevel/syscalls_call.go +++ b/pkg/sealevel/syscalls_call.go @@ -170,7 +170,7 @@ func SyscallGetProcessedSiblingInstructionImpl(vm sbpf.VM, index, metaAddr, prog } if instrCtxFound != nil { - resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, false) + resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, true) if err != nil { return syscallErr(err) } diff --git a/pkg/sealevel/syscalls_sibling_pool_test.go b/pkg/sealevel/syscalls_sibling_pool_test.go new file mode 100644 index 000000000..a3b3b9e22 --- /dev/null +++ b/pkg/sealevel/syscalls_sibling_pool_test.go @@ -0,0 +1,42 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func TestSiblingHeaderWriteIsClearedOnFinish(t *testing.T) { + old := sbpf.UsePool + sbpf.UsePool = true + t.Cleanup(func() { sbpf.UsePool = old }) + for _, addr := range []uint64{sbpf.VaddrStack + 128, sbpf.VaddrHeap + 128} { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(100000), TransactionContext: &TransactionCtx{ + InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}, + }} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram}, &sbpf.VMOpts{HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + // Keep a read-only view so the test itself does not mark the header dirty. + header, err := vm.Translate(addr, ProcessedSiblingInstructionSize, false) + require.NoError(t, err) + require.Equal(t, make([]byte, 16), header) + found, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, addr, 0, 0, 0) + require.NoError(t, err) + require.Equal(t, uint64(1), found) + require.Equal(t, byte(3), header[0]) + vm.Finish() + // Inspect before allocating another VM: this is deterministic and does not + // depend on sync.Pool choosing a particular backing buffer on the next Get. + require.Equal(t, make([]byte, 16), header) + } +} + +func TestSiblingHeaderRejectsReadOnlyDestination(t *testing.T) { + data := make([]byte, 16) + vm, ctx := newMemSyscallVM(t, nil, []sbpf.InputRegion{{RegionSize: 16, AddressSpaceReserved: 16, Data: data}}) + ctx.TransactionContext = &TransactionCtx{InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}} + _, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, sbpf.VaddrInput, 0, 0, 0) + require.Error(t, err) + require.Equal(t, make([]byte, 16), data) +}