From 915f61946a55a7ae78981510876daabd73419c0d Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:59:15 -0500 Subject: [PATCH 01/22] fix: preserve boolean flag defaults and explicit config overrides --- cmd/mithril/node/config_bool.go | 15 ++++++ cmd/mithril/node/config_bool_test.go | 38 ++++++++++++++ cmd/mithril/node/node.go | 9 +--- pkg/sbpf/pooling_test.go | 75 ++++++++++++++++++++++++++++ 4 files changed, 130 insertions(+), 7 deletions(-) create mode 100644 cmd/mithril/node/config_bool.go create mode 100644 cmd/mithril/node/config_bool_test.go create mode 100644 pkg/sbpf/pooling_test.go diff --git a/cmd/mithril/node/config_bool.go b/cmd/mithril/node/config_bool.go new file mode 100644 index 000000000..e779efb61 --- /dev/null +++ b/cmd/mithril/node/config_bool.go @@ -0,0 +1,15 @@ +package node + +import "github.com/spf13/pflag" + +// resolveBoolOption preserves explicit false at either precedence level. +// Defaults use DefValue, not a flag value potentially left by a previous run. +func resolveBoolOption(flag *pflag.Flag, configured bool, configuredValue bool) bool { + if flag != nil && flag.Changed { + return flag.Value.String() == "true" + } + if configured { + return configuredValue + } + return flag != nil && flag.DefValue == "true" +} diff --git a/cmd/mithril/node/config_bool_test.go b/cmd/mithril/node/config_bool_test.go new file mode 100644 index 000000000..ab23313d0 --- /dev/null +++ b/cmd/mithril/node/config_bool_test.go @@ -0,0 +1,38 @@ +package node + +import ( + "github.com/spf13/pflag" + "github.com/spf13/viper" + "github.com/stretchr/testify/require" + "strings" + "testing" +) + +func TestResolveBoolOptionPrecedence(t *testing.T) { + for _, tc := range []struct { + name, toml, cli string + defaultValue, want bool + }{ + {"omitted true default", "", "", true, true}, + {"omitted false default", "", "", false, false}, + {"TOML false", "enabled=false", "", true, false}, + {"TOML true", "enabled=true", "", false, true}, + {"CLI false beats TOML true", "enabled=true", "false", true, false}, + {"CLI true beats TOML false", "enabled=false", "true", false, true}, + {"CLI false without TOML", "", "false", true, false}, + } { + t.Run(tc.name, func(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(tc.toml))) + flags := pflag.NewFlagSet("test", pflag.ContinueOnError) + flags.Bool("enabled", tc.defaultValue, "") + if tc.cli != "" { + require.NoError(t, flags.Set("enabled", tc.cli)) + } + require.Equal(t, tc.want, resolveBoolOption(flags.Lookup("enabled"), v.IsSet("enabled"), v.GetBool("enabled"))) + }) + } + require.False(t, resolveBoolOption(nil, false, false)) + require.True(t, resolveBoolOption(nil, true, true)) +} diff --git a/cmd/mithril/node/node.go b/cmd/mithril/node/node.go index 66a81e3e1..57c8f76a0 100644 --- a/cmd/mithril/node/node.go +++ b/cmd/mithril/node/node.go @@ -742,14 +742,9 @@ func initConfigAndBindFlags(cmd *cobra.Command) error { return 0 } - // Helper to get bool: CLI flag if explicitly set, otherwise TOML config + // Match numeric options: explicit CLI, configured value, then flag default. getBool := func(cliKey, tomlKey string) bool { - if flagChanged(cliKey) { - if f := cmd.Flags().Lookup(cliKey); f != nil { - return f.Value.String() == "true" - } - } - return config.GetBool(tomlKey) + return resolveBoolOption(cmd.Flags().Lookup(cliKey), config.IsSet(tomlKey), config.GetBool(tomlKey)) } // Helper to get string slice: CLI flag if explicitly set, otherwise TOML config diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go new file mode 100644 index 000000000..38d1028f8 --- /dev/null +++ b/pkg/sbpf/pooling_test.go @@ -0,0 +1,75 @@ +package sbpf + +import ( + "bytes" + "sync" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/stretchr/testify/require" +) + +func poolingInterpreter(heap int) *Interpreter { + meter := cu.NewComputeMeter(100) + return NewInterpreter(testV3Program([]Slot{testSlot(OpExit, 0, 0, 0, 0)}, nil), + &VMOpts{HeapMax: heap, ComputeMeter: &meter}) +} + +func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { + old := UsePool + UsePool = true + t.Cleanup(func() { UsePool = old }) + var wg sync.WaitGroup + for worker := 0; worker < 8; worker++ { + wg.Add(1) + go func(worker int) { + defer wg.Done() + for i := 0; i < 30; i++ { + parent := poolingInterpreter(32 * 1024) + if !bytes.Equal(parent.heap, make([]byte, len(parent.heap))) || + !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { + t.Error("pooled VM exposed data from an earlier execution") + } + parent.heap[0] = byte(worker + 1) + parent.stack.mem[0] = byte(worker + 1) + child := poolingInterpreter(256 * 1024) + for j := range child.heap { + child.heap[j] = 0xab + } + for j := range child.stack.mem { + child.stack.mem[j] = 0xcd + } + child.Finish() + if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { + t.Error("nested VM storage aliased its active parent") + } + parent.Finish() + } + }(worker) + } + wg.Wait() + ip := poolingInterpreter(32 * 1024) + defer ip.Finish() + require.Len(t, ip.heap, 32*1024) + _, err := ip.Translate(VaddrHeap+32*1024, 1, false) + require.Error(t, err, "pool capacity must not widen the requested heap mapping") +} + +func BenchmarkVMCreateAndFinish(b *testing.B) { + old := UsePool + b.Cleanup(func() { UsePool = old }) + for _, pooled := range []bool{false, true} { + name := "fresh" + if pooled { + name = "pooled" + } + b.Run(name, func(b *testing.B) { + UsePool = pooled + b.ReportAllocs() + for i := 0; i < b.N; i++ { + ip := poolingInterpreter(32 * 1024) + ip.Finish() + } + }) + } +} From cb360c4eb141e37ff3b61c0b6cb23e1f2e7cb35c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:59:15 -0500 Subject: [PATCH 02/22] fix: own retained vote deque storage before pooled reuse --- pkg/sealevel/vote_deque_ownership_test.go | 29 +++++++++++++++++++++++ pkg/sealevel/vote_program.go | 9 ++++++- 2 files changed, 37 insertions(+), 1 deletion(-) create mode 100644 pkg/sealevel/vote_deque_ownership_test.go diff --git a/pkg/sealevel/vote_deque_ownership_test.go b/pkg/sealevel/vote_deque_ownership_test.go new file mode 100644 index 000000000..b7d50f43b --- /dev/null +++ b/pkg/sealevel/vote_deque_ownership_test.go @@ -0,0 +1,29 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" + "github.com/gammazero/deque" + "github.com/stretchr/testify/require" +) + +func TestProcessNewVoteStateOwnsRetainedDeque(t *testing.T) { + // Model the TowerSync scratch deque being returned to its pool and reused. + scratch := new(deque.Deque[LandedVote]) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 100, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 101, ConfirmationCount: 1}}) + state := new(VoteState) + require.NoError(t, processNewVoteState(state, scratch, nil, nil, 0, 101, features.Features{})) + cached := newVoteState4FromCurrent(state, solana.PublicKey{}) + want := []LandedVote{state.Votes.At(0), state.Votes.At(1)} + + scratch.Clear() + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 200, ConfirmationCount: 2}}) + scratch.PushBack(LandedVote{Lockout: VoteLockout{Slot: 201, ConfirmationCount: 1}}) + for i, vote := range want { + require.Equal(t, vote, state.Votes.At(i)) + require.Equal(t, vote, cached.Votes.At(i)) + } +} diff --git a/pkg/sealevel/vote_program.go b/pkg/sealevel/vote_program.go index 43a8500c2..679d1787d 100644 --- a/pkg/sealevel/vote_program.go +++ b/pkg/sealevel/vote_program.go @@ -1951,7 +1951,14 @@ func processNewVoteState(voteState *VoteState, newState *deque.Deque[LandedVote] } voteState.RootSlot = newRoot - voteState.Votes = *newState + // newState may be a pooled deque. Own the backing storage before its + // caller returns it to the pool: the resulting state can escape into the + // shared vote cache after this instruction completes. + var owned deque.Deque[LandedVote] + for i := 0; i < newState.Len(); i++ { + owned.PushBack(newState.At(i)) + } + voteState.Votes = owned return nil } From be7a914a7465a334257429f22e8510ddd3e5b697 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:12:58 -0500 Subject: [PATCH 03/22] sbpf: reject overflow in contiguous virtual memory ranges --- pkg/sbpf/interpreter.go | 6 +++--- pkg/sbpf/translate_overflow_test.go | 25 +++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 3 deletions(-) create mode 100644 pkg/sbpf/translate_overflow_test.go diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 4fa0f784a..5f351a5d6 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1186,7 +1186,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.ro)) { + if lo+size < lo || lo+size > uint64(len(ip.ro)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds program read") } return unsafe.Pointer(&ip.ro[lo]), nil @@ -1203,7 +1203,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } - if lo+size > uint64(len(ip.heap)) { + if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } return unsafe.Pointer(&ip.heap[lo]), nil @@ -1214,7 +1214,7 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if len(ip.inputRegions) != 0 { return ip.translateInputRegion(lo, size, write) } - if lo+size > uint64(len(ip.input)) { + if lo+size < lo || lo+size > uint64(len(ip.input)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds input access") } return unsafe.Pointer(&ip.input[lo]), nil diff --git a/pkg/sbpf/translate_overflow_test.go b/pkg/sbpf/translate_overflow_test.go new file mode 100644 index 000000000..98d8ee2d0 --- /dev/null +++ b/pkg/sbpf/translate_overflow_test.go @@ -0,0 +1,25 @@ +package sbpf + +import ( + "github.com/stretchr/testify/require" + "testing" +) + +func TestTranslateRejectsOverflowingRange(t *testing.T) { + ip := &Interpreter{ro: make([]byte, 32), heap: make([]byte, 32), input: make([]byte, 32)} + for _, base := range []uint64{VaddrProgram, VaddrHeap, VaddrInput} { + for _, write := range []bool{false, true} { + for _, offset := range []uint64{1, 31, 33, 0xffffffff} { + for _, size := range []uint64{^uint64(0), ^uint64(0) - 15} { + _, err := ip.Translate(base+offset, size, write) + require.Error(t, err, "base=%x offset=%d size=%d write=%v", base, offset, size, write) + } + } + } + got, err := ip.Translate(base+31, 1, false) + require.NoError(t, err) + require.Len(t, got, 1) + _, err = ip.Translate(base+31, 2, false) + require.Error(t, err) + } +} From 584791f00861e24dcb877abeaecb5edc5761933c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:14:24 -0500 Subject: [PATCH 04/22] sealevel: remove SHA-256 syscall descriptor and digest allocations --- docs/sha256-syscall.md | 42 ++++ pkg/sealevel/sealevel_test.go | 12 +- pkg/sealevel/syscalls_hash.go | 15 +- pkg/sealevel/syscalls_sha256_bench_test.go | 268 +++++++++++++++++++++ 4 files changed, 325 insertions(+), 12 deletions(-) create mode 100644 docs/sha256-syscall.md create mode 100644 pkg/sealevel/syscalls_sha256_bench_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md new file mode 100644 index 000000000..58a83abca --- /dev/null +++ b/docs/sha256-syscall.md @@ -0,0 +1,42 @@ +# SHA-256 syscall overhead + +The syscall decodes the already-translated slice descriptor array directly and +writes the final digest into the translated output buffer. It retains streaming +SHA-256, slice order, memory translations, CU charges and validation order. Output +is written only after all inputs have been read, preserving overlapping-buffer +behavior. No special case for a particular on-chain program is introduced. + +A bounded 55-byte input-buffer prototype was slower than this simpler path and +is retained only as a benchmark comparison. The baseline reference is copied +from combined review commit `bd17683a`. + +Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; +medians below. Each benchmark runs serially through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. This does +not include VM instruction dispatch, a complete program, or block replay. + +| Input | Original | Direct decoding/output | Buffered prototype | +|---|---:|---:|---:| +| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | +| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | +| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | +| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | + +The two-slice case removes four allocations (112 bytes) per call. This is an +ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native +Zen 5 and captured heavy-block replay before deployment decisions. + +Reproduce with: + +```sh +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +``` + +Differential tests compare hashes, return/error values, remaining CU and all +input/output memory over valid inputs, invalid descriptors/addresses, depleted +budgets and output aliasing. The existing SHA program fixture also executes. +Testing exposed pre-existing overflow in contiguous VM region bounds checks; +that correction and its regression test are a separate preceding commit. Both +benchmark variants use the corrected VM. The old SHA fixture also needed its +compute-meter pointer initialized for the current interpreter API. diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ddfe230a..6ad64f3a9 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -416,13 +416,15 @@ func TestInterpreter_Sha256(t *testing.T) { syscalls.Register("my_memcmp", SyscallMemcmp) var log LogRecorder + ctx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: ctx, + ComputeMeter: &ctx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_hash.go b/pkg/sealevel/syscalls_hash.go index b73a1cd0c..33edd31d0 100644 --- a/pkg/sealevel/syscalls_hash.go +++ b/pkg/sealevel/syscalls_hash.go @@ -3,6 +3,7 @@ package sealevel import ( "bytes" "crypto/sha256" + "encoding/binary" "fmt" "math/big" @@ -49,15 +50,13 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 } var data []byte - reader := bytes.NewReader(vals) + // Translate validated the complete descriptor array above. Decode directly + // to avoid allocating a reader and a temporary buffer for each slice. for count := uint64(0); count < valsLen; count++ { - var vec VectorDescrC - err = vec.Unmarshal(reader) - if err != nil { - return syscallErr(err) - } + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} data, err = vm.Translate(vec.Addr, vec.Len, false) if err != nil { @@ -73,7 +72,9 @@ func SyscallSha256Impl(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint6 hasher.Write(data) } } - copy(hashResult[:], hasher.Sum(nil)) + // All inputs have been read before writing, including when output aliases + // input memory. Append into the translated destination without allocating. + hasher.Sum(hashResult[:0]) return syscallSuccess(0) } diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go new file mode 100644 index 000000000..b4240f54e --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -0,0 +1,268 @@ +package sealevel + +import ( + "bytes" + "crypto/sha256" + "encoding/binary" + "fmt" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" + "math/rand" + "testing" +) + +// Frozen syscall implementation from bd17683a; keep independent for differential tests. +func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256BaselineReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + reader := bytes.NewReader(vals) + + for count := uint64(0); count < valsLen; count++ { + + var vec VectorDescrC + err = vec.Unmarshal(reader) + if err != nil { + return syscallErr(err) + } + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + hasher.Write(data) + } + } + copy(hashResult[:], hasher.Sum(nil)) + return syscallSuccess(0) +} + +// Experimental bounded-buffer variant retained only for benchmark comparison. +func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { + //mlog.Log.Debugf("sha256SmallInputReference") + + if valsLen > cu.CUSha256MaxSlices { + return syscallErr(SyscallErrTooManySlices) + } + + execCtx := executionCtx(vm) + err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) + if err != nil { + return syscallCuErr() + } + + hashResult, err := vm.Translate(resultsAddr, 32, true) + if err != nil { + return syscallErr(err) + } + + hasher := sha256.New() + // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only + // this bounded case; larger inputs retain streaming hashing. + var small [55]byte + buffered := 0 + streaming := false + if valsLen > 0 { + var vals []byte + + // The data at 'valsAddr' consists of an array of 'slice references', which consists + // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that + // refers to an input value to hash. + // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above + vals, err = vm.Translate(valsAddr, valsLen*16, false) + if err != nil { + return syscallErr(err) + } + + var data []byte + + for count := uint64(0); count < valsLen; count++ { + + offset := count * 16 + vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} + + data, err = vm.Translate(vec.Addr, vec.Len, false) + if err != nil { + return syscallErr(err) + } + + cost := max(vec.Len/2, cu.CUMemOpBaseCost) + err = execCtx.ComputeMeter.Consume(cost) + if err != nil { + return syscallCuErr() + } + + if !streaming && len(data) <= len(small)-buffered { + buffered += copy(small[buffered:], data) + } else { + if !streaming { + hasher.Write(small[:buffered]) + streaming = true + } + hasher.Write(data) + } + } + } + if streaming { + hasher.Sum(hashResult[:0]) + } else { + digest := sha256.Sum256(small[:buffered]) + copy(hashResult, digest[:]) + } + return syscallSuccess(0) +} + +type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) + +func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { + mem := make([]byte, 32768) + pos := 8192 + for i, n := range sizes { + binary.LittleEndian.PutUint64(mem[i*16:], sbpf.VaddrInput+uint64(pos)) + binary.LittleEndian.PutUint64(mem[i*16+8:], uint64(n)) + for j := 0; j < n; j++ { + mem[pos+j] = byte(i + j) + } + pos += n + } + return mem, sbpf.VaddrInput, uint64(len(sizes)), sbpf.VaddrInput + 4096 +} + +func sha256VM(mem []byte, budget uint64) (*sbpf.Interpreter, *ExecutionCtx) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(budget)} + vm := sbpf.NewInterpreter(&sbpf.Program{}, &sbpf.VMOpts{Input: mem, HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + return vm, ctx +} + +func TestSha256SyscallDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(1234)) + for i := 0; i < 700; i++ { + sizes := make([]int, rng.Intn(8)) + for j := range sizes { + sizes[j] = rng.Intn(80) + } + if i < 8 { + sizes = [][]int{nil, {0}, {32, 4}, {55}, {56}, {32, 24}, {4096}, {0, 32, 0, 4}}[i] + } + mem, a, n, out := sha256Fixture(sizes) + budget := uint64(100000) + switch i % 11 { + case 1: + budget = uint64(rng.Intn(250)) + case 2: + out = 1 + case 3: + a = 1 + case 4: + if n > 0 { + binary.LittleEndian.PutUint64(mem, 1) + } + case 5: + if n > 0 { + binary.LittleEndian.PutUint64(mem[8:], ^uint64(0)) + } + case 6: + out = sbpf.VaddrInput + 8192 // output overlaps the input + case 7: + out = a // output overlaps descriptors + case 8: + n = cu.CUSha256MaxSlices + 1 + case 9: + n = cu.CUSha256MaxSlices + case 10: + a = sbpf.VaddrInput + uint64(len(mem)-1) + } + var wantMem []byte + var wantRet, wantCU uint64 + var wantErr string + for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + buf := append([]byte(nil), mem...) + vm, ctx := sha256VM(buf, budget) + ret, err := fn(vm, a, n, out) + remaining := ctx.ComputeMeter.Remaining() + vm.Finish() + if k == 0 { + wantMem = buf + wantRet = ret + wantCU = remaining + wantErr = fmt.Sprint(err) + continue + } + require.Equal(t, wantRet, ret, "case %d variant %d", i, k) + require.Equal(t, wantErr, fmt.Sprint(err), "case %d variant %d", i, k) + require.Equal(t, wantCU, remaining, "case %d variant %d", i, k) + require.Equal(t, wantMem, buf, "case %d variant %d", i, k) + } + } +} + +var sha256BenchDigest [32]byte + +func BenchmarkSha256Syscall(b *testing.B) { + for _, tc := range []struct { + name string + sizes []int + }{{"empty", nil}, {"36_contiguous", []int{36}}, {"32_plus_4", []int{32, 4}}, {"55", []int{55}}, {"56", []int{56}}, {"1232", []int{1232}}, {"4096", []int{4096}}} { + for _, variant := range []struct { + name string + fn sha256Call + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + b.Run(tc.name+"/"+variant.name, func(b *testing.B) { + mem, a, n, out := sha256Fixture(tc.sizes) + vm, ctx := sha256VM(mem, ^uint64(0)) + defer vm.Finish() + b.ReportAllocs() + b.ResetTimer() + for j := 0; j < b.N; j++ { + ctx.ComputeMeter = cu.NewComputeMeter(100000) + if _, err := variant.fn(vm, a, n, out); err != nil { + b.Fatal(err) + } + } + }) + } + } + b.Run("raw36", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + sha256BenchDigest = sha256.Sum256(data[:]) + } + }) +} From f944a0e26652f6963cde855c14202607ba7fa2c7 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:20:08 -0500 Subject: [PATCH 05/22] test: measure captured SHA loop and document Zen 5 acceleration --- docs/sha256-syscall.md | 47 ++++++++++ pkg/sealevel/syscalls_sha256_loop_test.go | 102 ++++++++++++++++++++++ 2 files changed, 149 insertions(+) create mode 100644 pkg/sealevel/syscalls_sha256_loop_test.go diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 58a83abca..10c1b59b2 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -40,3 +40,50 @@ Testing exposed pre-existing overflow in contiguous VM region bounds checks; that correction and its regression test are a separate preceding commit. Both benchmark variants use the corrected VM. The old SHA fixture also needed its compute-meter pointer initialized for the current interpreter API. + + +## Zen 5 acceleration and captured loop + +The live Go 1.26.4 validator on Ryzen 7 9700X had its actual +`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is +already active. No validator runtime setting was changed. + +The isolated loop harness copies text slots 518–545 from the captured SBF v0 +program, resolves the SHA syscall relocation, and supplies 1,000 iterations and +zero initial state. It retains descriptor setup, stack accesses, digest copying, +counter update and loop branching. A test checks its result against a Go hash +chain and checks equal CU consumption for both syscall implementations. This +excludes transaction loading, account dependencies, CPI and the remaining program. + +A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, +affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. +Three 150 ms samples (medians, per hash iteration): + +| Isolated loop | Time | +|---|---:| +| Original syscall | 234.5 ns | +| Optimized syscall | 165.9 ns | +| Dispatch-only diagnostic control | 109.2 ns | +| Go hash chain without VM | 54.69 ns | + +The optimized loop takes about 29% less time. The dispatch-only control replaces +the syscall with a no-op: it omits hashing, translations and syscall CU charging, +and is only an overhead diagnostic, never a valid execution implementation. +The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; +the buffered-input prototype remained slower at 71–73 ns. These short tests +share a host with other processes; they are not isolated-core latency guarantees. + +A separate short CPU profile of the optimized loop attributed 36.2% cumulative +sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI +compression routine. Most remaining sampled work was VM execution: instruction +dispatch/decoding, stack address translation, loads/stores and compute metering. +Cumulative and flat percentages overlap and must not be added. This profile is +of the harness, not of full-block replay or the live validator. + +The next execution experiment should target measured VM overhead and then replay +captured blocks; these results do not justify a claimed 29% block-time improvement. + +```sh +go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +``` diff --git a/pkg/sealevel/syscalls_sha256_loop_test.go b/pkg/sealevel/syscalls_sha256_loop_test.go new file mode 100644 index 000000000..32c22d9f0 --- /dev/null +++ b/pkg/sealevel/syscalls_sha256_loop_test.go @@ -0,0 +1,102 @@ +package sealevel + +import ( + "crypto/sha256" + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +// This is text slots 518..545 from the captured AogGeA81 program's hash loop. +// Captured ELF SHA-256: b3286f96f5611ee7db62dc11ea9aab1afe80908c08a0fbd3fdd783879d70ed88. +// The captured ELF uses SBF v0. Only the unresolved syscall relocation is replaced. The harness supplies a +// loop bound and zero initial state; transaction loading, CPI and the rest of +// the program are deliberately excluded. +func sha256LoopProgram(iterations uint32) *sbpf.Program { + text := []sbpf.Slot{ + sbpf.Slot(sbpf.OpMov64Imm) | 7<<8 | sbpf.Slot(iterations)<<32, + 0xa1bf, 0xffffffb000000107, 0xfe701a7b, 0xa1bf, 0xfffffe1000000107, + 0xfe601a7b, 0xffb08a63, 0x4fe780a7a, 0x20fe680a7a, 0xa1bf, + 0xfffffe6000000107, 0xa3bf, 0xffffffe000000307, 0x2000002b7, + sbpf.Slot(sbpf.OpCall) | sbpf.Slot(hash_sol_sha256)<<32, + 0xffe0a179, 0xfe101a7b, 0xffe8a179, 0xfe181a7b, 0xfff0a179, + 0xfe201a7b, 0xfff8a179, 0xfe281a7b, 0x100000807, 0x81bf, + 0x2000000167, 0x2000000177, 0xffe471ad, + // Return the first digest word so the harness can check the computation. + 0xfe10a079, sbpf.Slot(sbpf.OpExit), + } + return &sbpf.Program{Text: text} +} + +func runSha256Loop(p *sbpf.Program, fn sha256Call) (uint64, uint64, error) { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(10000000)} + vm := sbpf.NewInterpreter(p, &sbpf.VMOpts{Context: ctx, ComputeMeter: &ctx.ComputeMeter, Syscalls: func(hash uint32) (sbpf.Syscall, bool) { return sbpf.SyscallFunc3(fn), hash == hash_sol_sha256 }}) + ret, _, err := vm.Run() + used := ctx.ComputeMeter.Used() + vm.Finish() + return ret, used, err +} + +func TestSha256CapturedLoop(t *testing.T) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + require.NoError(t, p.Verify()) + var data [36]byte + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + want := binary.LittleEndian.Uint64(data[:8]) + var wantCU uint64 + for _, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { + got, used, err := runSha256Loop(p, fn) + require.NoError(t, err) + require.Equal(t, want, got) + if wantCU == 0 { + wantCU = used + } + require.Equal(t, wantCU, used) + } +} + +func BenchmarkSha256CapturedLoop(b *testing.B) { + const iterations = 1000 + p := sha256LoopProgram(iterations) + for _, v := range []struct { + name string + fn sha256Call + }{ + {"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, + // Diagnostic lower bound: no hashing, translations or syscall CU charging. + // It is not a valid implementation and must never be used in replay. + {"dispatch_only", func(sbpf.VM, uint64, uint64, uint64) (uint64, error) { return 0, nil }}, + } { + b.Run(v.name, func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + if _, _, err := runSha256Loop(p, v.fn); err != nil { + b.Fatal(err) + } + } + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) + } + b.Run("raw_chain", func(b *testing.B) { + var data [36]byte + b.ReportAllocs() + for j := 0; j < b.N; j++ { + clear(data[:]) + for i := uint32(0); i < iterations; i++ { + binary.LittleEndian.PutUint32(data[32:], i) + d := sha256.Sum256(data[:]) + copy(data[:32], d[:]) + } + } + sha256BenchDigest = sha256.Sum256(data[:]) + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N*iterations), "ns/hash") + }) +} From 75db2e0149234204216d345ba82151d9a74c44ad Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 06/22] sbpf: cut per-instruction overhead in the interpreter (~2x on SPL Token transfer) Consensus-neutral performance changes to pkg/sbpf, validated against the unmodified interpreter with a 100k-program differential corpus (identical return values, errors/PCs, CU consumed, meter remaining, memory contents, input-region state) plus the package's unit tests: - meter instructions with a local due/budget pair synced around syscalls and on exit (Agave's due_insn_count scheme) instead of calling ComputeMeter.Consume per instruction - move cold opcodes to executeCold so Run drops below the compiler's "big function" threshold and Consume/Read*/Push/Pop/fast paths inline - zero only the dirty range of the pooled stack/heap in Finish (page bitmap on the fast path, byte range on the translate path) instead of 256 KiB + HeapMax per execution - per-window fast-path address translation table (Agave aligned mapping layout, branch-free v0 frame gaps, one-entry cache for VASA input regions) - 16-wide register file (no bounds checks on r[dst]/r[src]), in-place call-frame Push/Pop, precomputed internal call targets per Program pooling_test writes through the VM's translation layer now, since the pool only re-zeroes memory the VM saw written (all production writes go through translation). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/cu/cu.go | 5 + pkg/sbpf/fastmem.go | 59 ++ pkg/sbpf/interpreter.go | 1236 +++++++++++++++++++++---------------- pkg/sbpf/loader/loader.go | 4 +- pkg/sbpf/pooling_test.go | 26 +- pkg/sbpf/program.go | 23 + pkg/sbpf/stack.go | 74 ++- 7 files changed, 873 insertions(+), 554 deletions(-) create mode 100644 pkg/sbpf/fastmem.go diff --git a/pkg/cu/cu.go b/pkg/cu/cu.go index 79d337199..f90aa1e2c 100644 --- a/pkg/cu/cu.go +++ b/pkg/cu/cu.go @@ -33,6 +33,11 @@ func (cm *ComputeMeter) Consume(cost uint64) error { return nil } +// Disabled reports whether metering is currently switched off (Consume is a no-op). +func (cm *ComputeMeter) Disabled() bool { + return cm.disable +} + func (cm *ComputeMeter) Used() uint64 { return cm.startingBalance - cm.computeMeter } diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go new file mode 100644 index 000000000..421f3ed42 --- /dev/null +++ b/pkg/sbpf/fastmem.go @@ -0,0 +1,59 @@ +package sbpf + +import "unsafe" + +// memRegion describes one of the fixed 4 GiB virtual address windows +// (rodata, stack, heap, input) as a contiguous host buffer, so that the +// interpreter can translate the common case without a function call. +// +// rlen / wlen are the readable / writable byte lengths (wlen == 0 for +// read-only windows). gapShift/gapMask implement SBPF v0 stack frame gaps +// exactly like Agave's MemoryRegion::vm_gap_shift (gapShift = 63 and +// gapMask = 0 for windows without gaps, which makes the gap logic a no-op). +// A window that needs special handling (VASA input regions, ...) has +// rlen = wlen = 0 and falls back to translateInternal. +type memRegion struct { + base unsafe.Pointer // host address of window offset `start` + start uint64 // offset of the region within its 4 GiB window + rlen uint64 + wlen uint64 + gapShift uint64 + gapMask uint64 + dirty uint64 // 4 KiB page bitmap of fast-path writes (stack/heap only matter) +} + +// emptyRegion never matches any access. +var emptyRegion = memRegion{gapShift: 63} + +const numFastRegions = 6 // index 5 is a permanently empty catch-all + +// fastRead returns a host pointer for a size-byte read at vma, or nil if the +// access is not covered by the fast path (caller falls back to Read*). +func (ip *Interpreter) fastRead(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + // Truncating to 32 bits makes lo < start wrap to a value >= 2^32-start, + // which is always > rlen (start+rlen < 2^32), so one compare suffices. + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.rlen || inGap != 0 { + return nil + } + return unsafe.Add(reg.base, off) +} + +// fastWrite is the write counterpart of fastRead; it also records the dirty +// range so Finish only needs to zero what was touched. +func (ip *Interpreter) fastWrite(vma uint64, size uint64) unsafe.Pointer { + reg := &ip.regions[min(vma>>32, numFastRegions-1)] + lo := vma & 0xffffffff + inGap := (lo >> reg.gapShift) & 1 + off := uint64(uint32((((lo & reg.gapMask) >> 1) | (lo &^ reg.gapMask)) - reg.start)) + if off+size > reg.wlen || inGap != 0 { + return nil + } + // Mark the 4 KiB page (and, conservatively, the next one, since an access + // is at most 8 bytes and may straddle a page boundary) as dirty. + reg.dirty |= 3 << (off >> 12) + return unsafe.Add(reg.base, off) +} diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index 5f351a5d6..e7e28fda0 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1,6 +1,7 @@ package sbpf import ( + "errors" "fmt" "math" "math/bits" @@ -46,6 +47,31 @@ type Interpreter struct { sbpfVersion sbpfver.SbpfVersion programId solana.PublicKey txSignature solana.Signature + + // callTargets[pc] is the resolved internal-function target of the `call imm` + // at pc (or -1). Computed once per Program at load time. + callTargets []int64 + + // Fast-path translation table indexed by (vaddr >> 32); see fastmem.go. + regions [numFastRegions]memRegion + // dirtyLo/dirtyHi: byte range written through translateInternal; + // memRegion.dirty: 4 KiB page bitmap of writes through the fast path. + dirtyLo [numFastRegions]uint64 + dirtyHi [numFastRegions]uint64 +} + +// dirtyRange returns the union of the byte ranges that may have been written +// in window idx (stack or heap), as [lo, hi). +func (ip *Interpreter) dirtyRange(idx uint64, size uint64) (lo, hi uint64) { + lo, hi = ip.dirtyLo[idx], ip.dirtyHi[idx] + if pages := ip.regions[idx].dirty; pages != 0 { + plo := uint64(bits.TrailingZeros64(pages)) << 12 + phi := uint64(bits.Len64(pages)) << 12 + lo = min(lo, plo) + hi = max(hi, phi) + } + hi = min(hi, size) + return lo, hi } type TraceSink interface { @@ -76,12 +102,13 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { heap = slices.Grow(heap, opts.HeapMax-len(heap)) } heap = heap[:opts.HeapMax] - clear(heap) + // Buffers in the pool are zeroed (for their dirty range) in Finish, so + // no clear is needed here. } else { heap = newHeap() } - return &Interpreter{ + ip := &Interpreter{ textVA: p.TextVA, textBytes: p.TextBytes, text: p.Text, @@ -103,17 +130,56 @@ func NewInterpreter(p *Program, opts *VMOpts) *Interpreter { sbpfVersion: p.SbpfVersion, programId: opts.ProgramId, txSignature: opts.TxSignature, + callTargets: p.CallTargets, + } + ip.initRegions() + return ip +} + +// initRegions fills the fast-path translation table. Windows that need the +// full logic in translateInternal are left empty (rlen = wlen = 0). +func (ip *Interpreter) initRegions() { + for i := range ip.dirtyLo { + ip.dirtyLo[i] = math.MaxUint64 + ip.dirtyHi[i] = 0 + ip.regions[i].gapShift = 63 + } + if len(ip.ro) != 0 { + idx := VaddrProgram >> 32 + if ip.sbpfVersion.EnableLowerRodataVaddr() { + idx = 0 + } + ip.regions[idx] = memRegion{base: unsafe.Pointer(&ip.ro[0]), rlen: uint64(len(ip.ro)), gapShift: 63} + } + if len(ip.stack.mem) != 0 { + r := memRegion{base: unsafe.Pointer(&ip.stack.mem[0]), rlen: StackMax, wlen: StackMax, gapShift: 63} + if ip.stack.stackFrameGaps { + r.gapShift = 12 // log2(StackFrameSize) + r.gapMask = GapMask + } + ip.regions[VaddrStack>>32] = r + } + if len(ip.heap) != 0 { + ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} + } + if len(ip.inputRegions) == 0 && len(ip.input) != 0 { + ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } } func (ip *Interpreter) Finish() { if UsePool { + lo, hi := ip.dirtyRange(VaddrHeap>>32, uint64(len(ip.heap))) + if hi > lo { + clear(ip.heap[lo:hi]) + } heapPool.Put(ip.heap) } + ip.stack.MarkDirty(ip.dirtyRange(VaddrStack>>32, StackMax)) ip.stack.Finish() } -func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, error) { +func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[16]uint64) (int64, error) { var taken bool dst := uint32(r[ins.Dst()]) src := uint32(r[ins.Src()]) @@ -200,7 +266,7 @@ func (ip *Interpreter) executeJmp32(ins Slot, pc int64, r *[11]uint64) (int64, e // // This function may panic given code that doesn't pass the static verifier. func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { - var r [11]uint64 + var r [16]uint64 // 16 (not 11) so that r[ins.Dst()] (4-bit field) needs no bounds check r[1] = VaddrInput r[2] = ip.inputDataVaddr @@ -216,137 +282,228 @@ func (ip *Interpreter) Run() (ret uint64, cuConsumed uint64, err error) { // initialize pc to program entry point pc := int64(ip.entry) + // Loop-invariant state hoisted into locals so the compiler can keep them in + // registers (fields of ip may alias with the unsafe stores in the loop and + // would otherwise be reloaded on every instruction). + text := ip.text + tracing := ip.enableTracing + jmp32 := ip.sbpfVersion.EnableJmp32() + moveMem := ip.sbpfVersion.MoveMemoryInstructionClasses() + pqr := ip.sbpfVersion.EnablePqr() + staticSyscalls := ip.sbpfVersion.EnableStaticSyscalls() + callTargets := ip.callTargets + + // Instruction metering (mirrors Agave's due_insn_count / previous_instruction_meter): + // count executed instructions locally and only sync with the shared compute + // meter around syscalls and on exit. `budget` is the number of instructions + // we may still execute before the meter would be exhausted. + meter := ip.computeMeter + var budget, due uint64 + reloadBudget := func() { + if meter.Disabled() { + budget = math.MaxUint64 + } else { + budget = meter.Remaining() + } + due = 0 + // A syscall (CPI in particular) may have changed the input regions; + // drop the cached input-region fast path entry, it is re-populated on + // the next slow-path translation. (With a plain, region-less input the + // entry is static and stays.) + if len(ip.inputRegions) != 0 { + ip.regions[VaddrInput>>32] = emptyRegion + } + } + flushDue := func() { + if due != 0 { + _ = meter.Consume(due) + due = 0 + } + } + reloadBudget() + mainLoop: for i := 0; true; i++ { // Fetch - if pc < 0 || pc >= int64(len(ip.text)) { + if pc < 0 || pc >= int64(len(text)) { + flushDue() return 0, 0, &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, ExcExecutionOverrun), } } - ins := ip.getSlot(pc) - if ip.enableTracing { + ins := text[pc] + if tracing { regsDump := fmt.Sprintf("%016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x, %016x", r[0], r[1], r[2], r[3], r[4], r[5], r[6], r[7], r[8], r[9], r[10]) fmt.Printf("% 5d [%s]: %s\n", i, strings.ToUpper(regsDump), ip.disassemble(ins, 0)) } - err = ip.computeMeter.Consume(1) - if err != nil { + // Meter: identical semantics to Consume(1) before each instruction. + if due == budget { + err = cu.ErrComputeExceeded break mainLoop } + due++ // Execute - if ip.sbpfVersion.EnableJmp32() && ins.Op()&0x07 == ClassPqr { + if jmp32 && ins.Op()&0x07 == ClassPqr { pc, err = ip.executeJmp32(ins, pc, &r) goto postExecute } switch ins.Op() { case OpLdxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 1); p != nil { + r[ins.Dst()] = uint64(*(*uint8)(p)) + } else { + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 2); p != nil { + r[ins.Dst()] = uint64(*(*uint16)(p)) + } else { + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) + if p := ip.fastRead(vma, 4); p != nil { + r[ins.Dst()] = uint64(*(*uint32)(p)) + } else { + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpLdxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v + if p := ip.fastRead(vma, 8); p != nil { + r[ins.Dst()] = uint64(*(*uint64)(p)) + } else { + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = uint64(v) + } pc++ case OpStb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(ins.Uimm()) + } else { + err = ip.Write8(vma, uint8(ins.Uimm())) + } pc++ case OpSth: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(ins.Uimm()) + } else { + err = ip.Write16(vma, uint16(ins.Uimm())) + } pc++ case OpStw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(ins.Uimm()) + } else { + err = ip.Write32(vma, uint32(ins.Uimm())) + } pc++ case OpStdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(ins.Imm()) + } else { + err = ip.Write64(vma, uint64(ins.Imm())) + } pc++ case OpStxb: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) + if p := ip.fastWrite(vma, 1); p != nil { + *(*uint8)(p) = uint8(r[ins.Src()]) + } else { + err = ip.Write8(vma, uint8(r[ins.Src()])) + } pc++ case OpStxh: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) + if p := ip.fastWrite(vma, 2); p != nil { + *(*uint16)(p) = uint16(r[ins.Src()]) + } else { + err = ip.Write16(vma, uint16(r[ins.Src()])) + } pc++ case OpStxw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) + if p := ip.fastWrite(vma, 4); p != nil { + *(*uint32)(p) = uint32(r[ins.Src()]) + } else { + err = ip.Write32(vma, uint32(r[ins.Src()])) + } pc++ case OpStxdw: - if ip.sbpfVersion.MoveMemoryInstructionClasses() { + if moveMem { err = ExcInvalidInstr break } vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) + if p := ip.fastWrite(vma, 8); p != nil { + *(*uint64)(p) = uint64(r[ins.Src()]) + } else { + err = ip.Write64(vma, uint64(r[ins.Src()])) + } pc++ case OpAdd32Imm: r[ins.Dst()] = ip.signExtension(int32(r[ins.Dst()]) + ins.Imm()) @@ -383,343 +540,6 @@ mainLoop: case OpMul32Imm: r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * ins.Imm()) pc++ - case OpMul32Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd1BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint8 - v, err = ip.Read8(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpMul64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(ins.Uimm())) - pc++ - } - case OpMul64Reg: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] *= r[ins.Src()] - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt1BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write8(vma, uint8(r[ins.Src()])) - pc++ - } - case OpDiv32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpDiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd2BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint16 - v, err = ip.Read16(vma) - r[ins.Dst()] = uint64(v) - pc++ - } - case OpLd4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint32 - v, err = ip.Read32(vma) - r[ins.Dst()] = uint64(v) - pc++ - case OpDiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] /= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(ins.Uimm())) - pc++ - } - case OpDiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt2BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write16(vma, uint16(r[ins.Src()])) - pc++ - } - case OpSt4BReg: - if !ip.sbpfVersion.MoveMemoryInstructionClasses() { - err = ExcInvalidInstr - break - } - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, uint32(r[ins.Src()])) - pc++ - case OpLmul32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) - pc++ - case OpLmul32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) - pc++ - case OpLmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= uint64(int64(ins.Imm())) - pc++ - case OpLmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] *= r[ins.Src()] - pc++ - case OpUhmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) - r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() - pc++ - case OpUhmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Uint128FromUint64(r[ins.Dst()]) - regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) - r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() - pc++ - case OpShmul64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - imm128 := wide.Int128FromInt64(int64(ins.Imm())) - r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() - pc++ - case OpShmul64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) - src128 := wide.Int128FromInt64(int64(r[ins.Src()])) - r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() - pc++ - case OpUdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) - pc++ - case OpUdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] /= uint64(ins.Uimm()) - pc++ - case OpUdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] /= src - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpUrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] = uint64(r[ins.Dst()] % src) - } else { - err = ExcDivideByZero - } - pc++ - case OpUrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] %= uint64(ins.Uimm()) - pc++ - case OpUrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= r[ins.Src()] - } else { - err = ExcDivideByZero - } - pc++ - case OpSdiv32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) - pc++ - case OpSdiv32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSdiv64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) - pc++ - case OpSdiv64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem32Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) - pc++ - case OpSrem32Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int32(r[ins.Src()]); src != 0 { - if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) - } else { - err = ExcDivideByZero - break - } - pc++ - case OpSrem64Imm: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) - pc++ - case OpSrem64Reg: - if !ip.sbpfVersion.EnablePqr() { - err = ExcInvalidInstr - break - } - if src := int64(r[ins.Src()]); src != 0 { - if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { - err = ExcDivideOverflow - break - } - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) - } else { - err = ExcDivideByZero - break - } - pc++ case OpOr32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) | ins.Uimm()) pc++ @@ -768,66 +588,6 @@ mainLoop: case OpRsh64Reg: r[ins.Dst()] >>= r[ins.Src()] & 0x3f pc++ - case OpNeg32: - if ip.sbpfVersion.DisableNeg() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) - pc++ - case OpNeg64: - if !ip.sbpfVersion.DisableNeg() { - r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt4BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write32(vma, ins.Uimm()) - pc++ - } - case OpMod32Imm: - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) - pc++ - case OpMod32Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := uint32(r[ins.Src()]); src != 0 { - r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpLd8BReg - vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) - var v uint64 - v, err = ip.Read64(vma) - r[ins.Dst()] = v - pc++ - } - case OpMod64Imm: - if !ip.sbpfVersion.EnablePqr() { - r[ins.Dst()] %= uint64(ins.Imm()) - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BImm - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, uint64(ins.Imm())) - pc++ - } - case OpMod64Reg: - if !ip.sbpfVersion.EnablePqr() { - if src := r[ins.Src()]; src != 0 { - r[ins.Dst()] %= src - } else { - err = ExcDivideByZero - } - pc++ - } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { - // OpSt8BReg - vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) - err = ip.Write64(vma, r[ins.Src()]) - pc++ - } case OpXor32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) ^ ins.Uimm()) pc++ @@ -856,53 +616,6 @@ mainLoop: case OpMov64Reg: r[ins.Dst()] = r[ins.Src()] pc++ - case OpArsh32Imm: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) - pc++ - case OpArsh32Reg: - r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) - pc++ - case OpArsh64Imm: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) - pc++ - case OpArsh64Reg: - r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) - pc++ - case OpHor64Imm: - if !ip.sbpfVersion.DisableLddw() { - err = ExcInvalidInstr - break - } - r[ins.Dst()] |= uint64(ins.Uimm()) << 32 - pc++ - case OpLe: - if ip.sbpfVersion.DisableLe() { - err = ExcInvalidInstr - break - } - switch ins.Uimm() { - case 16: - r[ins.Dst()] &= math.MaxUint16 - case 32: - r[ins.Dst()] &= math.MaxUint32 - case 64: - r[ins.Dst()] &= math.MaxUint64 - default: - err = ExcUnsupportedInstruction - } - pc++ - case OpBe: - switch ins.Uimm() { - case 16: - r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) - case 32: - r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) - case 64: - r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) - default: - err = ExcUnsupportedInstruction - } - pc++ case OpLddw: if ip.sbpfVersion.DisableLddw() { err = ExcInvalidInstr @@ -1024,14 +737,16 @@ mainLoop: } pc++ case OpCall: - if ip.sbpfVersion.EnableStaticSyscalls() { + if staticSyscalls { if ins.Src() == 0 { sc, ok := ip.syscalls(ins.Uimm()) if !ok { err = ExcCallDest{ins.Uimm()} break } + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } @@ -1042,7 +757,7 @@ mainLoop: err = ExcCallDest{uint32(targetPC)} break } - if ok := ip.stack.Push(r[:], pc+1); !ok { + if ok := ip.stack.Push(&r, pc+1); !ok { err = ExcCallDepth } pc = targetPC @@ -1051,60 +766,147 @@ mainLoop: } } else { if sc, ok := ip.syscalls(ins.Uimm()); ok { + flushDue() r[0], err = sc.Invoke(ip, r[1], r[2], r[3], r[4], r[5]) + reloadBudget() if err != nil { err = ExcSyscallError{Err: err} } pc++ - } else if target, ok := ip.funcs[ins.Uimm()]; ok { - ok = ip.stack.Push(r[:], pc+1) + } else { + var target int64 + var ok bool + if callTargets != nil { + target = callTargets[pc] + ok = target >= 0 + } else { + target, ok = ip.funcs[ins.Uimm()] + } if !ok { + err = ExcCallDest{ins.Uimm()} + break + } + if !ip.stack.Push(&r, pc+1) { err = ExcCallDepth } pc = target - } else { - err = ExcCallDest{ins.Uimm()} } } - case OpCallx: - var target uint64 - if ip.sbpfVersion.CallXUsesSrcReg() { - target = r[ins.Src()] - } else if ip.sbpfVersion.CallXUsesDstReg() { - target = r[ins.Dst()] + case OpExit: + var ok bool + pc, ok = ip.stack.Pop(&r) + if !ok { + ret = r[0] + break mainLoop + } + case OpMul32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] = uint64(int32(r[ins.Dst()]) * int32(r[ins.Src()])) + pc++ + case OpMul64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= uint64(ins.Imm()) + pc++ + case OpMul64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpDiv32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) } else { - target = r[ins.Uimm()] + err = ExcDivideByZero } - - if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { - err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + pc++ + case OpDiv64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - targetPC := int64((target - ip.textVA) / 8) - if ok := ip.stack.Push(r[:], pc+1); !ok { - err = ExcCallDepth + r[ins.Dst()] /= uint64(ins.Imm()) + pc++ + case OpDiv64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) break } - pc = targetPC - case OpExit: - var ok bool - pc, ok = ip.stack.Pop(r[:]) - if !ok { - ret = r[0] - break mainLoop + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero } + pc++ + case OpMod32Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpMod64Imm: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + r[ins.Dst()] %= uint64(ins.Imm()) + pc++ + case OpMod64Reg: + if pqr { + pc, err = ip.executeCold(ins, pc, &r) + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= src + } else { + err = ExcDivideByZero + } + pc++ + case OpArsh32Imm: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> ins.Uimm())) + pc++ + case OpArsh32Reg: + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) >> uint32(r[ins.Src()]))) + pc++ + case OpArsh64Imm: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> ins.Imm()) + pc++ + case OpArsh64Reg: + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) >> (r[ins.Src()])) + pc++ default: - err = ExcUnsupportedInstruction - return + pc, err = ip.executeCold(ins, pc, &r) + if err == errUnknownOpcode { + // Preserve the original behaviour for an unknown opcode: + // a bare (unwrapped) ExcUnsupportedInstruction. + flushDue() + return 0, 0, ExcUnsupportedInstruction + } } // Post execute postExecute: - if err == cu.ErrComputeExceeded { - err = ExcOutOfCU - } - if err != nil { + flushDue() + if err == cu.ErrComputeExceeded { + err = ExcOutOfCU + } exc := &Exception{ PC: pc, Detail: fmt.Errorf("tx: %s, programId: %s - %w:", ip.txSignature, ip.programId, err), @@ -1117,6 +919,9 @@ mainLoop: } } + flushDue() + // NB: when the loop exits because the meter is exhausted, err is the bare + // cu.ErrComputeExceeded (not wrapped in an Exception), as before. cuConsumed = ip.initialInstrMeter - ip.computeMeter.Remaining() return @@ -1198,6 +1003,11 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if size == 0 { return emptySlice, nil } + if write { + off := StackMax - uint64(len(mem)) + ip.dirtyLo[VaddrStack>>32] = min(ip.dirtyLo[VaddrStack>>32], off) + ip.dirtyHi[VaddrStack>>32] = max(ip.dirtyHi[VaddrStack>>32], off+size) + } return unsafe.Pointer(&mem[0]), nil case VaddrHeap >> 32: if size == 0 { @@ -1206,6 +1016,10 @@ func (ip *Interpreter) translateInternal(addr uint64, size uint64, write bool) ( if lo+size < lo || lo+size > uint64(len(ip.heap)) { return nil, NewExcBadAccess(addr, size, write, "out-of-bounds heap access") } + if write { + ip.dirtyLo[VaddrHeap>>32] = min(ip.dirtyLo[VaddrHeap>>32], lo) + ip.dirtyHi[VaddrHeap>>32] = max(ip.dirtyHi[VaddrHeap>>32], lo+size) + } return unsafe.Pointer(&ip.heap[lo]), nil case VaddrInput >> 32: if size == 0 { @@ -1255,6 +1069,8 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } if write && (!region.Writable || requestedLen > region.RegionSize) && region.OnWrite != nil { + // The callback may replace region.Data / grow the region: drop the cache. + ip.regions[VaddrInput>>32] = emptyRegion if err := region.OnWrite(region, requestedLen); err != nil { return nil, err } @@ -1263,23 +1079,37 @@ func (ip *Interpreter) translateInputRegion(offset, size uint64, write bool) (un if !write || !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } + ip.regions[VaddrInput>>32] = emptyRegion region.RegionSize = region.AddressSpaceReserved } if write && !region.Writable { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "write to readonly input region") } + var base unsafe.Pointer if region.Data != nil { if requestedLen > uint64(len(region.Data)) { return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") } - return unsafe.Pointer(®ion.Data[regionOffset]), nil + base = unsafe.Pointer(unsafe.SliceData(region.Data)) + } else { + hostOffset := region.HostOffset + regionOffset + if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { + return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + } + base = unsafe.Pointer(&ip.input[region.HostOffset]) } - - hostOffset := region.HostOffset + regionOffset - if hostOffset < region.HostOffset || hostOffset+size < hostOffset || hostOffset+size > uint64(len(ip.input)) { - return nil, NewExcBadAccess(VaddrInput+offset, size, write, "out-of-bounds input access") + // Cache this region for the interpreter's fast path (one-entry cache, + // same idea as Agave's MappingCache). Only the currently mapped + // RegionSize bytes are exposed; anything beyond takes the slow path + // again so that OnWrite / growth semantics are preserved. + if region.RegionSize != 0 && (region.Data == nil || uint64(len(region.Data)) >= region.RegionSize) { + cached := memRegion{base: base, start: region.Offset, rlen: region.RegionSize, gapShift: 63} + if region.Writable { + cached.wlen = region.RegionSize + } + ip.regions[VaddrInput>>32] = cached } - return unsafe.Pointer(&ip.input[hostOffset]), nil + return unsafe.Add(base, regionOffset), nil } func (ip *Interpreter) TranslateInput(addr uint64, size uint64) ([]byte, error) { @@ -1335,6 +1165,7 @@ func (ip *Interpreter) SetInputRegionData(addr uint64, data []byte, length uint6 } region.RegionSize = length region.Writable = writable + ip.regions[VaddrInput>>32] = emptyRegion return true } @@ -1461,3 +1292,360 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { *(*uint64)(ptr) = x return nil } + +// executeCold handles the less frequently executed opcodes. Keeping them out of +// Run keeps that function below the compiler's "big function" threshold so the +// hot helpers (metering, fast memory translation, stack push/pop) stay inlinable. +func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { + var err error + switch ins.Op() { + case OpDiv32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpLd4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint32 + v, err = ip.Read32(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt4BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, uint32(r[ins.Src()])) + pc++ + case OpLmul32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * ins.Uimm()) + pc++ + case OpLmul32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) * uint32(r[ins.Src()])) + pc++ + case OpLmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= uint64(int64(ins.Imm())) + pc++ + case OpLmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] *= r[ins.Src()] + pc++ + case OpUhmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + imm128 := wide.Uint128FromUint64(uint64(ins.Uimm())) + r[ins.Dst()] = dst128.Mul(imm128).RShiftN(64).Uint64() + pc++ + case OpUhmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Uint128FromUint64(r[ins.Dst()]) + regSrc128 := wide.Uint128FromUint64(r[ins.Src()]) + r[ins.Dst()] = dst128.Mul(regSrc128).RShiftN(64).Uint64() + pc++ + case OpShmul64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + imm128 := wide.Int128FromInt64(int64(ins.Imm())) + r[ins.Dst()] = dst128.Mul(imm128).Uint128().RShiftN(64).Uint64() + pc++ + case OpShmul64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + dst128 := wide.Int128FromInt64(int64(r[ins.Dst()])) + src128 := wide.Int128FromInt64(int64(r[ins.Src()])) + r[ins.Dst()] = dst128.Mul(src128).Uint128().RShiftN(64).Uint64() + pc++ + case OpUdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) + pc++ + case OpUdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := uint32(r[ins.Src()]); src != 0 { + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] /= uint64(ins.Uimm()) + pc++ + case OpUdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] /= src + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpUrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] = uint64(r[ins.Dst()] % src) + } else { + err = ExcDivideByZero + } + pc++ + case OpUrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] %= uint64(ins.Uimm()) + pc++ + case OpUrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := r[ins.Src()]; src != 0 { + r[ins.Dst()] %= r[ins.Src()] + } else { + err = ExcDivideByZero + } + pc++ + case OpSdiv32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / ins.Imm())) + pc++ + case OpSdiv32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) / src)) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSdiv64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / int64(ins.Imm())) + pc++ + case OpSdiv64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) / src) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem32Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int32(r[ins.Dst()]) == math.MinInt32 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % ins.Imm())) + pc++ + case OpSrem32Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int32(r[ins.Src()]); src != 0 { + if int32(r[ins.Dst()]) == math.MinInt32 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(uint32(int32(r[ins.Dst()]) % int32(r[ins.Src()]))) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpSrem64Imm: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if int64(r[ins.Dst()]) == math.MinInt64 && ins.Imm() == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(ins.Imm())) + pc++ + case OpSrem64Reg: + if !ip.sbpfVersion.EnablePqr() { + err = ExcInvalidInstr + break + } + if src := int64(r[ins.Src()]); src != 0 { + if int64(r[ins.Dst()]) == math.MinInt64 && src == -1 { + err = ExcDivideOverflow + break + } + r[ins.Dst()] = uint64(int64(r[ins.Dst()]) % int64(r[ins.Src()])) + } else { + err = ExcDivideByZero + break + } + pc++ + case OpNeg32: + if ip.sbpfVersion.DisableNeg() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] = uint64(-int32(r[ins.Dst()])) + pc++ + case OpNeg64: + if !ip.sbpfVersion.DisableNeg() { + r[ins.Dst()] = uint64(-int64(r[ins.Dst()])) + pc++ + } else if ip.sbpfVersion.MoveMemoryInstructionClasses() { + // OpSt4BImm + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write32(vma, ins.Uimm()) + pc++ + } + case OpMod32Imm: + r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) % ins.Uimm()) + pc++ + case OpHor64Imm: + if !ip.sbpfVersion.DisableLddw() { + err = ExcInvalidInstr + break + } + r[ins.Dst()] |= uint64(ins.Uimm()) << 32 + pc++ + case OpLe: + if ip.sbpfVersion.DisableLe() { + err = ExcInvalidInstr + break + } + switch ins.Uimm() { + case 16: + r[ins.Dst()] &= math.MaxUint16 + case 32: + r[ins.Dst()] &= math.MaxUint32 + case 64: + r[ins.Dst()] &= math.MaxUint64 + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpBe: + switch ins.Uimm() { + case 16: + r[ins.Dst()] = uint64(bits.ReverseBytes16(uint16(r[ins.Dst()]))) + case 32: + r[ins.Dst()] = uint64(bits.ReverseBytes32(uint32(r[ins.Dst()]))) + case 64: + r[ins.Dst()] = bits.ReverseBytes64(r[ins.Dst()]) + default: + err = ExcUnsupportedInstruction + } + pc++ + case OpCallx: + var target uint64 + if ip.sbpfVersion.CallXUsesSrcReg() { + target = r[ins.Src()] + } else if ip.sbpfVersion.CallXUsesDstReg() { + target = r[ins.Dst()] + } else { + target = r[ins.Uimm()] + } + + if target < ip.textVA || target >= VaddrStack || target >= ip.textVA+uint64(len(ip.text)*8) { + err = NewExcBadAccess(target, 8, false, "jump out-of-bounds") + break + } + targetPC := int64((target - ip.textVA) / 8) + if ok := ip.stack.Push(r, pc+1); !ok { + err = ExcCallDepth + break + } + pc = targetPC + default: + err = errUnknownOpcode + } + return pc, err +} + +// errUnknownOpcode is an internal sentinel returned by executeCold for an +// opcode that is not handled by either switch; Run turns it into the bare +// ExcUnsupportedInstruction return of the original implementation. +var errUnknownOpcode = errors.New("unknown opcode") diff --git a/pkg/sbpf/loader/loader.go b/pkg/sbpf/loader/loader.go index 7b9dc52ca..23a26ec76 100644 --- a/pkg/sbpf/loader/loader.go +++ b/pkg/sbpf/loader/loader.go @@ -162,7 +162,7 @@ func parseSlots(bs []byte) []sbpf.Slot { } func (l *Loader) getProgram() *sbpf.Program { - return &sbpf.Program{ + p := &sbpf.Program{ RO: l.program, TextBytes: l.text, Text: parseSlots(l.text), @@ -171,4 +171,6 @@ func (l *Loader) getProgram() *sbpf.Program { Funcs: l.funcs, SbpfVersion: l.sbpfVersion(), } + p.ResolveCallTargets() + return p } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index 38d1028f8..c4bc5eb0d 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -30,14 +30,28 @@ func TestPooledVMIsolationAcrossNestedAndConcurrentExecutions(t *testing.T) { !bytes.Equal(parent.stack.mem, make([]byte, len(parent.stack.mem))) { t.Error("pooled VM exposed data from an earlier execution") } - parent.heap[0] = byte(worker + 1) - parent.stack.mem[0] = byte(worker + 1) + // Write through the VM's translation layer, as programs and + // syscalls do: the pool only re-zeroes memory the VM saw written. + if err := parent.Write8(VaddrHeap, byte(worker+1)); err != nil { + t.Error(err) + } + if err := parent.Write8(VaddrStack, byte(worker+1)); err != nil { + t.Error(err) + } child := poolingInterpreter(256 * 1024) - for j := range child.heap { - child.heap[j] = 0xab + childHeap, err := child.Translate(VaddrHeap, uint64(len(child.heap)), true) + if err != nil { + t.Error(err) + } + for j := range childHeap { + childHeap[j] = 0xab + } + childStack, err := child.Translate(VaddrStack, StackMax, true) + if err != nil { + t.Error(err) } - for j := range child.stack.mem { - child.stack.mem[j] = 0xcd + for j := range childStack { + childStack[j] = 0xcd } child.Finish() if parent.heap[0] != byte(worker+1) || parent.stack.mem[0] != byte(worker+1) { diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index c112be14d..969a15469 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -13,6 +13,29 @@ type Program struct { Entrypoint uint64 // PC Funcs map[uint32]int64 SbpfVersion sbpfver.SbpfVersion + + // CallTargets[pc] holds the resolved internal function target for a + // `call imm` slot at pc (non-static-syscall versions), or -1. + CallTargets []int64 +} + +// ResolveCallTargets precomputes CallTargets from Funcs so the interpreter +// does not need a map lookup per call instruction. +func (p *Program) ResolveCallTargets() { + if p.SbpfVersion.EnableStaticSyscalls() { + p.CallTargets = nil + return + } + targets := make([]int64, len(p.Text)) + for pc, slot := range p.Text { + targets[pc] = -1 + if slot.Op() == OpCall { + if t, ok := p.Funcs[slot.Uimm()]; ok { + targets[pc] = t + } + } + } + p.CallTargets = targets } func (p *Program) MemoryBytes() uint64 { diff --git a/pkg/sbpf/stack.go b/pkg/sbpf/stack.go index 3f3331b6e..edc662c18 100644 --- a/pkg/sbpf/stack.go +++ b/pkg/sbpf/stack.go @@ -36,6 +36,12 @@ type Stack struct { shadow []Frame dynamicStackFrames bool stackFrameGaps bool + // dirtyLo/dirtyHi bound the physical byte range of mem that may have been + // written during this execution. Finish only has to zero this range before + // returning the buffer to the pool. + dirtyLo uint64 + dirtyHi uint64 + maxDepth int } // Frame is an entry on the shadow stack. @@ -92,9 +98,11 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { } s := Stack{ - mem: m, - sp: VaddrStack, - shadow: sh, + mem: m, + sp: VaddrStack, + shadow: sh, + dirtyLo: StackMax, + dirtyHi: 0, } var sz uint64 @@ -115,10 +123,12 @@ func NewStack(sbpfVer sbpfver.SbpfVersion, disableStackFrameGaps bool) Stack { func (s *Stack) Finish() { if UsePool { s.mem = s.mem[:StackMax] - clear(s.mem) + if s.dirtyHi > s.dirtyLo { + clear(s.mem[s.dirtyLo:s.dirtyHi]) + } stackMemPool.Put(s.mem) s.shadow = s.shadow[:StackDepth] - clear(s.shadow) + clear(s.shadow[:max(s.maxDepth, 1)]) s.shadow = s.shadow[:1] stackShadowPool.Put(s.shadow) } @@ -153,21 +163,38 @@ func (s *Stack) GetFrame(addr uint32) []byte { } } +// MarkDirty records that physical stack bytes [off, off+size) may be written. +func (s *Stack) MarkDirty(lo, hi uint64) { + if hi <= lo { + return + } + s.dirtyLo = min(s.dirtyLo, lo) + s.dirtyHi = max(s.dirtyHi, hi) +} + // Push allocates a new call frame. // // Saves the given nonvolatile regs, return address, // and current frame pointer. // Returns the new frame pointer. -func (s *Stack) Push(regs []uint64, ret int64) bool { - if ok := len(s.shadow) < cap(s.shadow); !ok { +func (s *Stack) Push(regs *[16]uint64, ret int64) bool { + n := len(s.shadow) + if n >= cap(s.shadow) { return false } - - frame := Frame{RetAddr: ret} - copy(frame.NVRegs[:], regs[6:10]) - frame.FramePtr = regs[10] - - s.shadow = append(s.shadow, frame) + // Write the frame in place (no temporary Frame value / copy) to avoid + // store-forwarding stalls in this very hot path. + s.shadow = s.shadow[:n+1] + f := &s.shadow[n] + f.RetAddr = ret + f.NVRegs[0] = regs[6] + f.NVRegs[1] = regs[7] + f.NVRegs[2] = regs[8] + f.NVRegs[3] = regs[9] + f.FramePtr = regs[10] + if n+1 > s.maxDepth { + s.maxDepth = n + 1 + } if !s.dynamicStackFrames { if s.stackFrameGaps { @@ -185,16 +212,17 @@ func (s *Stack) Push(regs []uint64, ret int64) bool { // Restores saved nonvolatile regs into provided slice. // Returns saved return address and returns true upon success, // and returns false if no call frames are left. -func (s *Stack) Pop(regs []uint64) (int64, bool) { - if len(s.shadow) <= 1 { +func (s *Stack) Pop(regs *[16]uint64) (int64, bool) { + n := len(s.shadow) + if n <= 1 { return 0, false } - - var frame Frame - frame, s.shadow = s.shadow[len(s.shadow)-1], s.shadow[:len(s.shadow)-1] - - copy(regs[6:10], frame.NVRegs[:]) - regs[10] = frame.FramePtr - - return frame.RetAddr, true + f := &s.shadow[n-1] + regs[6] = f.NVRegs[0] + regs[7] = f.NVRegs[1] + regs[8] = f.NVRegs[2] + regs[9] = f.NVRegs[3] + regs[10] = f.FramePtr + s.shadow = s.shadow[:n-1] + return f.RetAddr, true } From 5b8f33f6c28243e40b683a6690fae9795cd77ed4 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 07/22] sbpf: add interpreter benchmarks and a differential test corpus - perf_bench_test.go: synthetic ALU / load-store / call loops and interpreter setup+teardown - loader/token_perf_bench_test.go: real SPL Token Transfer through the loader/verifier/interpreter with sealevel-equivalent syscalls, in the aligned and VASA input layouts - perf_differential_test.go: deterministic random program corpus; run on two builds with SBPF_DIFF_OUT= and diff the outputs; SBPF_CHECK_POOL_ZERO=1 asserts pooled buffers come back zeroed Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sbpf/loader/token_perf_bench_test.go | 425 +++++++++++++++++++++++ pkg/sbpf/perf_bench_test.go | 193 ++++++++++ pkg/sbpf/perf_differential_test.go | 338 ++++++++++++++++++ 3 files changed, 956 insertions(+) create mode 100644 pkg/sbpf/loader/token_perf_bench_test.go create mode 100644 pkg/sbpf/perf_bench_test.go create mode 100644 pkg/sbpf/perf_differential_test.go diff --git a/pkg/sbpf/loader/token_perf_bench_test.go b/pkg/sbpf/loader/token_perf_bench_test.go new file mode 100644 index 000000000..c2534e586 --- /dev/null +++ b/pkg/sbpf/loader/token_perf_bench_test.go @@ -0,0 +1,425 @@ +package loader_test + +import ( + "encoding/binary" + "errors" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" +) + +// ---- minimal syscall set mirroring pkg/sealevel semantics (CU costs + memory behaviour) ---- + +const ( + cuSyscallBase = 100 + cuMemOpBase = 10 + cuCpiBytesPerCU = 250 +) + +type stats struct { + logs int + memcpy int + memset int + memcmp int + memmove int + bytes uint64 +} + +var st stats + +func memOpConsume(vm sbpf.VM, n uint64) error { + cost := max(uint64(cuMemOpBase), n/cuCpiBytesPerCU) + return vm.ComputeMeter().Consume(cost) +} + +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { + srcBuf := make([]byte, n) // same allocation pattern as sealevel/syscalls_mem.go + err = vm.Read(src, srcBuf) + if err != nil { + return + } + err = vm.Write(dst, srcBuf) + return +} + +func isNonOverlapping(src, srcLen, dst, dstLen uint64) bool { + if src > dst { + return src-dst >= dstLen + } + return dst-src >= srcLen +} + +var syscallMemcpy = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memcpy++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + if !isNonOverlapping(src, n, dst, n) { + return 0, errors.New("overlapping") + } + if n == 0 { + return 0, nil + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemmove = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, src, n uint64) (uint64, error) { + st.memmove++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + return 0, memmoveImplInternal(vm, dst, src, n) +}) + +var syscallMemcmp = sbpf.SyscallFunc4(func(vm sbpf.VM, a1, a2, n, res uint64) (uint64, error) { + st.memcmp++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + s1, err := vm.Translate(a1, n, false) + if err != nil { + return 0, err + } + s2, err := vm.Translate(a2, n, false) + if err != nil { + return 0, err + } + r := int32(0) + for i := uint64(0); i < n; i++ { + if s1[i] != s2[i] { + r = int32(s1[i]) - int32(s2[i]) + break + } + } + out, err := vm.Translate(res, 4, true) + if err != nil { + return 0, err + } + binary.LittleEndian.PutUint32(out, uint32(r)) + return 0, nil +}) + +var syscallMemset = sbpf.SyscallFunc3(func(vm sbpf.VM, dst, c, n uint64) (uint64, error) { + st.memset++ + st.bytes += n + if err := memOpConsume(vm, n); err != nil { + return 0, err + } + mem, err := vm.Translate(dst, n, true) + if err != nil { + return 0, err + } + for i := uint64(0); i < n; i++ { + mem[i] = byte(c) + } + return 0, nil +}) + +var syscallLog = sbpf.SyscallFunc2(func(vm sbpf.VM, ptr, strlen uint64) (uint64, error) { + st.logs++ + if err := vm.ComputeMeter().Consume(max(uint64(cuSyscallBase), strlen)); err != nil { + return 0, err + } + buf := make([]byte, strlen) + if err := vm.Read(ptr, buf); err != nil { + return 0, err + } + _ = string(buf) + return 0, nil +}) + +var syscallLog64 = sbpf.SyscallFunc5(func(vm sbpf.VM, a, b, c, d, e uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogPubkey = sbpf.SyscallFunc1(func(vm sbpf.VM, a uint64) (uint64, error) { + st.logs++ + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallLogCUs = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { + return 0, vm.ComputeMeter().Consume(100) +}) +var syscallAbort = sbpf.SyscallFunc0(func(vm sbpf.VM) (uint64, error) { return 0, errors.New("abort") }) +var syscallPanic = sbpf.SyscallFunc4(func(vm sbpf.VM, f, l, line, col uint64) (uint64, error) { + return 0, errors.New("panic") +}) +var syscallAllocFree = sbpf.SyscallFunc2(func(vm sbpf.VM, size, free uint64) (uint64, error) { + if free != 0 { + return 0, nil + } + hs := (vm.HeapSize() + 7) &^ 7 + addr := sbpf.VaddrHeap + hs + hs += size + if hs > vm.HeapMax() { + return 0, nil + } + vm.UpdateHeapSize(hs) + return addr, nil +}) + +var registry = map[uint32]sbpf.Syscall{ + sbpf.SymbolHash("abort"): syscallAbort, + sbpf.SymbolHash("sol_panic_"): syscallPanic, + sbpf.SymbolHash("sol_log_"): syscallLog, + sbpf.SymbolHash("sol_log_64_"): syscallLog64, + sbpf.SymbolHash("sol_log_pubkey"): syscallLogPubkey, + sbpf.SymbolHash("sol_log_compute_units_"): syscallLogCUs, + sbpf.SymbolHash("sol_memcpy_"): syscallMemcpy, + sbpf.SymbolHash("sol_memmove_"): syscallMemmove, + sbpf.SymbolHash("sol_memcmp_"): syscallMemcmp, + sbpf.SymbolHash("sol_memset_"): syscallMemset, + sbpf.SymbolHash("sol_alloc_free_"): syscallAllocFree, +} + +var syscalls = sbpf.SyscallRegistry(func(h uint32) (sbpf.Syscall, bool) { + s, ok := registry[h] + return s, ok +}) + +// ---- aligned input serialization (BPF loader v2/v3 format, no direct mapping) ---- + +const maxPermittedDataIncrease = 10 * 1024 + +type acct struct { + key, owner [32]byte + lamports uint64 + data []byte + signer, writable bool +} + +func serializeAligned(accts []acct, instrData []byte, programId [32]byte) []byte { + out := binary.LittleEndian.AppendUint64(nil, uint64(len(accts))) + for _, a := range accts { + out = append(out, 0xff) + out = append(out, b2u8(a.signer), b2u8(a.writable), 0) + out = append(out, 0, 0, 0, 0) // original_data_len + out = append(out, a.key[:]...) + out = append(out, a.owner[:]...) + out = binary.LittleEndian.AppendUint64(out, a.lamports) + out = binary.LittleEndian.AppendUint64(out, uint64(len(a.data))) + out = append(out, a.data...) + pad := maxPermittedDataIncrease + ((8 - len(a.data)%8) % 8) + out = append(out, make([]byte, pad)...) + out = binary.LittleEndian.AppendUint64(out, ^uint64(0)) // rent epoch + } + out = binary.LittleEndian.AppendUint64(out, uint64(len(instrData))) + out = append(out, instrData...) + out = append(out, programId[:]...) + return out +} + +func b2u8(b bool) byte { + if b { + return 1 + } + return 0 +} + +// SPL token account layout (165 bytes) +func tokenAccount(mint, owner [32]byte, amount uint64) []byte { + d := make([]byte, 165) + copy(d[0:32], mint[:]) + copy(d[32:64], owner[:]) + binary.LittleEndian.PutUint64(d[64:72], amount) + // delegate: COption none (4 bytes 0) + 32 + d[108] = 1 // state = Initialized + // is_native COption none, delegated_amount 0, close_authority none + return d +} + +func key(b byte) [32]byte { + var k [32]byte + for i := range k { + k[i] = b + } + return k +} + +func loadTokenProgram(tb testing.TB) *sbpf.Program { + elfBytes := fixtures.Load(tb, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + tb.Fatal(err) + } + p, err := l.Load() + if err != nil { + tb.Fatal(err) + } + if err := p.Verify(); err != nil { + tb.Fatal(err) + } + return p +} + +func transferInput(programId [32]byte) ([]byte, []acct) { + mint := key(0x11) + authority := key(0x22) + src := key(0x33) + dst := key(0x44) + accts := []acct{ + {key: src, owner: programId, lamports: 2039280, data: tokenAccount(mint, authority, 1_000_000), writable: true}, + {key: dst, owner: programId, lamports: 2039280, data: tokenAccount(mint, key(0x55), 5), writable: true}, + {key: authority, owner: key(0), lamports: 1_000_000_000, data: nil, signer: true}, + } + instr := append([]byte{3}, binary.LittleEndian.AppendUint64(nil, 1000)...) + return serializeAligned(accts, instr, programId), accts +} + +func runTransfer(tb testing.TB, p *sbpf.Program, input []byte) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransfer(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, _ := transferInput(programId) + st = stats{} + ret, used := runTransfer(t, p, input) + t.Logf("ret=%d cuUsed=%d stats=%+v", ret, used, st) + // verify balances changed in the serialized input + // account 0 data starts at 8 + 8 + 32+32+8+8 = 96 + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + off1 := 8 + (8 + 32 + 32 + 8 + 8 + 165 + maxPermittedDataIncrease + 3 + 8) + dstAmt := binary.LittleEndian.Uint64(input[off1+88+64:]) + t.Logf("src=%d dst=%d", srcAmt, dstAmt) + if ret != 0 || srcAmt != 999_000 || dstAmt != 1005 { + t.Fatalf("unexpected result ret=%d src=%d dst=%d", ret, srcAmt, dstAmt) + } +} + +func BenchmarkTokenTransfer(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, _ := transferInput(programId) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + var used uint64 + for i := 0; i < b.N; i++ { + copy(input, orig) + _, used = runTransfer(b, p, input) + } + b.StopTimer() + b.ReportMetric(float64(used), "cu/op") + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(b.N)/float64(used), "ns/cu") +} + +func BenchmarkTokenLoadVerify(b *testing.B) { + elfBytes := fixtures.Load(b, "sbpf", "spl-token.so") + f := features.NewFeaturesDefault() + b.ReportAllocs() + for i := 0; i < b.N; i++ { + l, err := loader.NewLoaderWithSyscalls(elfBytes, syscalls, false, f) + if err != nil { + b.Fatal(err) + } + p, err := l.Load() + if err != nil { + b.Fatal(err) + } + if err := p.Verify(); err != nil { + b.Fatal(err) + } + } +} + +// ---- VASA layout (VirtualAddressSpaceAdjustments active, direct mapping off) ---- +// Mirrors serializeParametersAligned with vasa=true, directMapping=false: +// same bytes as the aligned layout, but the input window is split into +// metadata regions and per-account data regions. + +func vasaRegions(accts []acct, instrLen int) []sbpf.InputRegion { + var regions []sbpf.InputRegion + var regionStart, hostRegionStart uint64 + vmOff := uint64(8) + for i, a := range accts { + l := vmOff // host offset == vm offset in this layout + dataLen := uint64(len(a.data)) + align := (8 - dataLen%8) % 8 + reserved := dataLen + maxPermittedDataIncrease + dataStart := vmOff + 88 + if dataStart > regionStart { + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: dataStart - regionStart, AddressSpaceReserved: dataStart - regionStart, Writable: true, AccountIndex: -1}) + } + regions = append(regions, sbpf.InputRegion{Offset: dataStart, HostOffset: l + 88, RegionSize: dataLen, + AddressSpaceReserved: reserved, Writable: a.writable, AccountIndex: i}) + hostRegionStart = l + 88 + reserved + regionStart = dataStart + reserved + vmOff += 88 + reserved + align + 8 + } + end := vmOff + 8 + uint64(instrLen) + 32 + regions = append(regions, sbpf.InputRegion{Offset: regionStart, HostOffset: hostRegionStart, + RegionSize: end - regionStart, AddressSpaceReserved: end - regionStart, Writable: true, AccountIndex: -1}) + return regions +} + +func runTransferVasa(tb testing.TB, p *sbpf.Program, input []byte, regions []sbpf.InputRegion) (uint64, uint64) { + cm := cu.NewComputeMeter(200_000) + // regions are mutated by the VM (RegionSize on growth), so copy per run + rc := append([]sbpf.InputRegion(nil), regions...) + ip := sbpf.NewInterpreter(p, &sbpf.VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + InputRegions: rc, + DisableStackFrameGaps: true, + }) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + tb.Fatal(err) + } + return ret, used +} + +func TestTokenTransferVasa(t *testing.T) { + p := loadTokenProgram(t) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + if regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize != uint64(len(input)) { + t.Fatalf("region layout mismatch: %d vs %d", regions[len(regions)-1].Offset+regions[len(regions)-1].RegionSize, len(input)) + } + ret, used := runTransferVasa(t, p, input, regions) + srcAmt := binary.LittleEndian.Uint64(input[96+64:]) + if ret != 0 || srcAmt != 999_000 { + t.Fatalf("unexpected ret=%d src=%d", ret, srcAmt) + } + t.Logf("ret=%d cu=%d regions=%d", ret, used, len(regions)) +} + +func BenchmarkTokenTransferVasa(b *testing.B) { + p := loadTokenProgram(b) + programId := key(0x99) + input, accts := transferInput(programId) + regions := vasaRegions(accts, 9) + orig := append([]byte(nil), input...) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + copy(input, orig) + runTransferVasa(b, p, input, regions) + } +} diff --git a/pkg/sbpf/perf_bench_test.go b/pkg/sbpf/perf_bench_test.go new file mode 100644 index 000000000..1d34d4d7a --- /dev/null +++ b/pkg/sbpf/perf_bench_test.go @@ -0,0 +1,193 @@ +package sbpf + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +func slot(op uint8, dst uint8, src uint8, off int16, imm uint32) Slot { + return Slot(op) | Slot(dst)<<8 | Slot(src)<<12 | Slot(uint16(off))<<16 | Slot(imm)<<32 +} + +func slotsToBytes(slots []Slot) []byte { + out := make([]byte, len(slots)*SlotSize) + for i, s := range slots { + binary.LittleEndian.PutUint64(out[i*SlotSize:], uint64(s)) + } + return out +} + +func mkProgram(text []Slot, ver uint32) *Program { + return &Program{ + TextBytes: slotsToBytes(text), + Text: text, + TextVA: VaddrProgram, + Entrypoint: 0, + Funcs: map[uint32]int64{}, + SbpfVersion: sbpfver.SbpfVersion{Version: ver}, + } +} + +var noSyscalls = SyscallRegistry(func(uint32) (Syscall, bool) { return nil, false }) + +// resolveCallTargetsIfSupported precomputes internal call targets on trees +// that have Program.ResolveCallTargets (the loader does this at load time); +// it is a no-op on the baseline tree so the same benchmark code runs on both. +func resolveCallTargetsIfSupported(p *Program) { + if r, ok := any(p).(interface{ ResolveCallTargets() }); ok { + r.ResolveCallTargets() + } +} + +// aluLoop: r1 = N; loop: r2 += r1; r2 ^= r3; r3 = r2; r3 *= 7; r3 >>= 3; r1 -= 1; jne r1,0 loop; exit +// 7 instructions per iteration. +func aluLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Imm, 3, 0, 0, 3), + // loop @3 + slot(OpAdd64Reg, 2, 1, 0, 0), + slot(OpXor64Reg, 2, 3, 0, 0), + slot(OpMov64Reg, 3, 2, 0, 0), + slot(OpMul64Imm, 3, 0, 0, 7), + slot(OpRsh64Imm, 3, 0, 0, 3), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -7, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// memLoop: writes and reads 8 byte values on the stack frame and heap. +// r1 = N; r4 = r10 - 4096 (frame base); r5 = heap base +// loop: stxdw [r4+0], r1; ldxdw r6, [r4+0]; add r2, r6; stxdw [r5+8], r2; ldxdw r7,[r5+8]; xor r2,r7 ; r1 -= 1; jne +func memLoopProgram(n uint32, ver uint32) *Program { + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + slot(OpMov64Reg, 4, 10, 0, 0), + slot(OpAdd64Imm, 4, 0, 0, uint32(0xfffff000)), // r4 = r10 - 4096 + slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), + slot(0, 0, 0, 0, uint32(VaddrHeap>>32)), + // loop @6 + slot(OpStxdw, 4, 1, 0, 0), + slot(OpLdxdw, 6, 4, 0, 0), + slot(OpAdd64Reg, 2, 6, 0, 0), + slot(OpStxdw, 5, 2, 8, 0), + slot(OpLdxdw, 7, 5, 8, 0), + slot(OpXor64Reg, 2, 7, 0, 0), + slot(OpStxw, 5, 2, 16, 0), + slot(OpLdxb, 8, 5, 16, 0), + slot(OpAdd64Reg, 2, 8, 0, 0), + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -11, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + return mkProgram(text, ver) +} + +// callLoop: calls a tiny function N times (tests Push/Pop + call resolution) +func callLoopProgram(n uint32, ver uint32) *Program { + fnPC := int64(7) + text := []Slot{ + slot(OpMov64Imm, 1, 0, 0, n), + slot(OpMov64Imm, 2, 0, 0, 1), + // loop @2 + slot(OpCall, 0, 0, 0, 0), // patched below + slot(OpSub64Imm, 1, 0, 0, 1), + slot(OpJneImm, 1, 0, -3, 0), + slot(OpMov64Reg, 0, 2, 0, 0), + slot(OpExit, 0, 0, 0, 0), + // fn @7 + slot(OpAdd64Imm, 2, 0, 0, 3), + slot(OpXor64Reg, 2, 1, 0, 0), + slot(OpExit, 0, 0, 0, 0), + } + p := mkProgram(text, ver) + if ver >= sbpfver.SbpfVersionV3 { + // relative call: target = pc + imm + 1 ; pc=2 -> imm = 7-2-1 = 4 + text[2] = slot(OpCall, 0, 1, 0, uint32(fnPC-2-1)) + } else { + h := PCHash(uint64(fnPC)) + p.Funcs[h] = fnPC + text[2] = slot(OpCall, 0, 0, 0, h) + } + p.TextBytes = slotsToBytes(text) + return p +} + +func runProgram(b *testing.B, p *Program, input []byte, syscalls SyscallRegistry, budget uint64) uint64 { + cm := cu.NewComputeMeter(budget) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: 32 * 1024, + Syscalls: syscalls, + ComputeMeter: &cm, + Input: input, + }) + ret, _, err := ip.Run() + ip.Finish() + if err != nil { + b.Fatal(err) + } + return ret +} + +func benchLoop(b *testing.B, p *Program, insnsPerRun uint64) { + if err := p.Verify(); err != nil { + b.Fatal(err) + } + resolveCallTargetsIfSupported(p) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1<<40) + } + b.StopTimer() + b.ReportMetric(float64(b.Elapsed().Nanoseconds())/float64(uint64(b.N)*insnsPerRun), "ns/insn") +} + +const loopN = 200_000 + +func BenchmarkAluLoopV0(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 0), 3+7*loopN+2) } +func BenchmarkAluLoopV3(b *testing.B) { benchLoop(b, aluLoopProgram(loopN, 3), 3+7*loopN+2) } +func BenchmarkMemLoopV0(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 0), 5+11*loopN+2) } +func BenchmarkMemLoopV3(b *testing.B) { benchLoop(b, memLoopProgram(loopN, 3), 5+11*loopN+2) } +func BenchmarkCallLoopV0(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 0), 2+6*loopN+2) +} +func BenchmarkCallLoopV3(b *testing.B) { + benchLoop(b, callLoopProgram(loopN, 3), 2+6*loopN+2) +} + +// Interpreter setup/teardown cost only (tiny program). +func BenchmarkNewInterpreterAndExit(b *testing.B) { + p := mkProgram([]Slot{slot(OpExit, 0, 0, 0, 0)}, 0) + b.ReportAllocs() + for i := 0; i < b.N; i++ { + runProgram(b, p, nil, noSyscalls, 1000) + } +} + +func TestSyntheticProgramsRun(t *testing.T) { + for _, ver := range []uint32{0, 3} { + for _, p := range []*Program{aluLoopProgram(1000, ver), memLoopProgram(1000, ver), callLoopProgram(1000, ver)} { + if err := p.Verify(); err != nil { + t.Fatal(err) + } + cm := cu.NewComputeMeter(1 << 30) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32 * 1024, Syscalls: noSyscalls, ComputeMeter: &cm}) + ret, used, err := ip.Run() + ip.Finish() + if err != nil { + t.Fatal(err) + } + t.Logf("ver=%d ret=%d cu=%d", ver, ret, used) + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go new file mode 100644 index 000000000..1d679d4e4 --- /dev/null +++ b/pkg/sbpf/perf_differential_test.go @@ -0,0 +1,338 @@ +package sbpf + +import ( + "bufio" + "encoding/binary" + "fmt" + "hash/fnv" + "math/rand" + "os" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" +) + +// Differential test: generates deterministic pseudo-random programs and dumps +// (return value, error string, CU consumed, memory hash) per program to the +// file named by SBPF_DIFF_OUT. Running it against the baseline and the +// optimized interpreter and diffing the two files checks that observable +// behaviour is identical. + +type diffSyscall struct { + fn func(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) +} + +func (s diffSyscall) Invoke(vm VM, r1, r2, r3, r4, r5 uint64) (uint64, error) { + return s.fn(vm, r1, r2, r3, r4, r5) +} + +var ( + hashPoke = SymbolHash("poke") // write r3 bytes of value r2 at r1 via vm.Write + hashPeek = SymbolHash("peek") // read 8 bytes at r1 -> r0 + hashCopy = SymbolHash("copy") // copy r3 bytes from r2 to r1 (Translate based) + hashBurn = SymbolHash("burn") // consume r1 CU + hashSetLen = SymbolHash("setlen") // SetInputRegionLength(r1, r2, r3!=0) +) + +func diffRegistry(h uint32) (Syscall, bool) { + switch h { + case hashPoke: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + buf := make([]byte, r3) + for i := range buf { + buf[i] = byte(r2 + uint64(i)) + } + return 0, vm.Write(r1, buf) + }}, true + case hashPeek: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + return vm.Read64(r1) + }}, true + case hashCopy: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + if r3 > 4096 { + r3 = 4096 + } + src, err := vm.Translate(r2, r3, false) + if err != nil { + return 0, err + } + dst, err := vm.Translate(r1, r3, true) + if err != nil { + return 0, err + } + copy(dst, src) + return 0, nil + }}, true + case hashBurn: + return diffSyscall{func(vm VM, r1, _, _, _, _ uint64) (uint64, error) { + return 0, vm.ComputeMeter().Consume(r1 & 0xff) + }}, true + case hashSetLen: + return diffSyscall{func(vm VM, r1, r2, r3, _, _ uint64) (uint64, error) { + if err := vm.ComputeMeter().Consume(10); err != nil { + return 0, err + } + ip := vm.(*Interpreter) + ok := ip.SetInputRegionLength(r1, r2, r3 != 0) + if ok { + return 1, nil + } + return 0, nil + }}, true + } + return nil, false +} + +func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { + reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 + imm := func() uint32 { + switch rng.Intn(4) { + case 0: + return uint32(rng.Intn(16)) + case 1: + return uint32(int32(-rng.Intn(16))) + case 2: + return rng.Uint32() + default: + return uint32(rng.Intn(4096)) + } + } + alu64 := []uint8{OpAdd64Imm, OpAdd64Reg, OpSub64Imm, OpSub64Reg, OpMul64Imm, OpMul64Reg, OpDiv64Imm, OpDiv64Reg, + OpOr64Imm, OpOr64Reg, OpAnd64Imm, OpAnd64Reg, OpLsh64Imm, OpLsh64Reg, OpRsh64Imm, OpRsh64Reg, OpMod64Imm, OpMod64Reg, + OpXor64Imm, OpXor64Reg, OpMov64Imm, OpMov64Reg, OpArsh64Imm, OpArsh64Reg, OpNeg64, + OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, + OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, + OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, + OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} + switch rng.Intn(10) { + case 0, 1, 2, 3: // alu + op := alu64[rng.Intn(len(alu64))] + i := imm() + if op == OpLe || op == OpBe { + i = []uint32{16, 32, 64}[rng.Intn(3)] + } + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + i = 3 + } + switch op { + case OpLsh32Imm, OpRsh32Imm, OpArsh32Imm: + i = uint32(rng.Intn(32)) + case OpLsh64Imm, OpRsh64Imm, OpArsh64Imm: + i = uint32(rng.Intn(64)) + } + return []Slot{slot(op, reg(), reg(), 0, i)} + case 4: // load + ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random + var base uint8 + var off int16 + switch rng.Intn(4) { + case 0: + base, off = 10, int16(-rng.Intn(4096)) + case 1: + base, off = 5, int16(rng.Intn(1024)) + case 2: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} + case 5: // store + ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + var base uint8 + var off int16 + switch rng.Intn(5) { + case 0, 1: + base, off = 10, int16(-rng.Intn(4096)) + case 2: + base, off = 5, int16(rng.Intn(1024)) + case 3: + base, off = 1, int16(rng.Intn(600)) + default: + base, off = reg(), int16(rng.Intn(65536)-32768) + } + return []Slot{slot(ops[rng.Intn(8)], base, reg(), off, imm())} + case 6: // forward conditional jump (never backwards: guarantees termination) + maxOff := n - pc - 2 + if maxOff <= 0 { + return []Slot{slot(OpMov64Imm, reg(), 0, 0, imm())} + } + return []Slot{slot(jmp[rng.Intn(len(jmp))], reg(), reg(), int16(rng.Intn(min(maxOff, 8))), imm())} + case 7: // syscall + hs := []uint32{hashPoke, hashPeek, hashCopy, hashBurn, hashSetLen} + return []Slot{slot(OpCall, 0, 0, 0, hs[rng.Intn(len(hs))])} + case 8: // internal call + if ver >= sbpfver.SbpfVersionV3 { + return []Slot{slot(OpCall, 0, 1, 0, uint32(fnPC-int64(pc)-1))} + } + return []Slot{slot(OpCall, 0, 0, 0, PCHash(uint64(fnPC)))} + default: // set up pointer registers + switch rng.Intn(3) { + case 0: // r5 = heap + return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + case 1: // r1 = input + small + return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + default: // r9 = random 64-bit + return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + } + } +} + +func genProgram(rng *rand.Rand, ver uint32) *Program { + n := 8 + rng.Intn(120) + // layout: [0, n) main body then exit; fn at fnPC: a few ALU ops + exit + body := make([]Slot, 0, n+16) + fnPC := int64(n + 1) + for len(body) < n { + body = append(body, randSlot(rng, len(body), n, ver, fnPC)...) + } + if len(body) > n { + body = body[:n-1] // drop a cut lddw pair + } + body = append(body, slot(OpExit, 0, 0, 0, 0)) + // fix up forward jumps that would land on the second slot of an lddw + for pc := range body { + if body[pc].Op()&0x07 == ClassJmp && body[pc].Op() != OpCall && body[pc].Op() != OpExit { + dst := pc + int(body[pc].Off()) + 1 + if dst < len(body) && body[dst].Op() == 0 { + body[pc] = body[pc]&^(Slot(0xffff)<<16) | Slot(uint16(body[pc].Off()+1))<<16 + } + } + } + // function + body = append(body, + slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), + slot(OpXor64Reg, 7, 6, 0, 0), + slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(body, ver) + if ver < sbpfver.SbpfVersionV3 { + p.Funcs[PCHash(uint64(fnPC))] = fnPC + } + p.RO = make([]byte, 256) + for i := range p.RO { + p.RO[i] = byte(i * 7) + } + return p +} + +func memHash(bs ...[]byte) uint64 { + h := fnv.New64a() + for _, b := range bs { + h.Write(b) + } + return h.Sum64() +} + +func TestDifferentialDump(t *testing.T) { + out := os.Getenv("SBPF_DIFF_OUT") + if out == "" { + t.Skip("SBPF_DIFF_OUT not set") + } + f, err := os.Create(out) + if err != nil { + t.Fatal(err) + } + defer f.Close() + w := bufio.NewWriter(f) + defer w.Flush() + + rng := rand.New(rand.NewSource(12345)) + const N = 100000 + generated, verified := 0, 0 + for i := 0; i < N; i++ { + ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + p := genProgram(rng, ver) + generated++ + if err := p.Verify(); err != nil { + fmt.Fprintf(w, "%d ver=%d VERIFY_FAIL %v\n", i, ver, err) + continue + } + verified++ + resolveCallTargetsIfSupported(p) + input := make([]byte, 700) + for j := range input { + input[j] = byte(j) + } + var regions []InputRegion + useRegions := rng.Intn(2) == 0 + if useRegions { + regions = []InputRegion{ + {Offset: 0, HostOffset: 0, RegionSize: 100, AddressSpaceReserved: 100, Writable: true, AccountIndex: -1}, + {Offset: 100, HostOffset: 100, RegionSize: 150, AddressSpaceReserved: 300, Writable: rng.Intn(2) == 0, AccountIndex: 0}, + {Offset: 400, HostOffset: 400, RegionSize: 300, AddressSpaceReserved: 300, Writable: true, AccountIndex: -1}, + } + } + budget := uint64(1 + rng.Intn(400)) + if rng.Intn(4) == 0 { + budget = 100000 + } + cm := cu.NewComputeMeter(budget) + heapMax := 4096 * (1 + rng.Intn(4)) + ip := NewInterpreter(p, &VMOpts{ + HeapMax: heapMax, + Syscalls: diffRegistry, + ComputeMeter: &cm, + Input: input, + InputRegions: regions, + DisableStackFrameGaps: rng.Intn(3) == 0, + }) + var ret, cuUsed uint64 + var runErr error + func() { + defer func() { + if r := recover(); r != nil { + runErr = fmt.Errorf("PANIC: %v", r) + } + }() + ret, cuUsed, runErr = ip.Run() + }() + errStr := "" + if runErr != nil { + errStr = runErr.Error() + } + h := memHash(ip.stack.mem, ip.heap, input) + regionSizes := "" + for _, r := range ip.inputRegions { + regionSizes += fmt.Sprintf("%d/%v,", r.RegionSize, r.Writable) + } + fmt.Fprintf(w, "%d ver=%d budget=%d ret=%d cu=%d remaining=%d err=%q mem=%x regions=%s\n", + i, ver, budget, ret, cuUsed, cm.Remaining(), errStr, h, regionSizes) + ip.Finish() + if os.Getenv("SBPF_CHECK_POOL_ZERO") != "" { + // The buffers just returned to the pool must be all-zero. + st := stackMemPool.Get().([]byte) + hp := heapPool.Get().([]byte) + for j, b := range st[:StackMax] { + if b != 0 { + t.Fatalf("program %d: pooled stack not zeroed at %d", i, j) + } + } + for j, b := range hp[:cap(hp)] { + if b != 0 { + t.Fatalf("program %d: pooled heap not zeroed at %d", i, j) + } + } + stackMemPool.Put(st) + heapPool.Put(hp) + } + } + t.Logf("generated=%d verified=%d", generated, verified) +} + +var _ = binary.LittleEndian From 3f0ec6cdcf59816a394cea57cad9c86763c3396f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 02:14:34 +0000 Subject: [PATCH 08/22] sealevel: add native program micro-benchmarks (System transfer, Vote TowerSync) Measured through ExecutionCtx.ProcessInstruction so instruction-context push/pop, lamport-sum checks and timing metrics are included; each has a NoTiming variant (SkipTimingMetrics) to quantify instrumentation cost, plus a vote-state (de)serialization round trip. NOTE: written without a local build of pkg/sealevel (sandbox cannot fetch its dependencies); expect to fix compile errors on first run. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/native_perf_bench_test.go | 287 +++++++++++++++++++++++++ 1 file changed, 287 insertions(+) create mode 100644 pkg/sealevel/native_perf_bench_test.go diff --git a/pkg/sealevel/native_perf_bench_test.go b/pkg/sealevel/native_perf_bench_test.go new file mode 100644 index 000000000..447931695 --- /dev/null +++ b/pkg/sealevel/native_perf_bench_test.go @@ -0,0 +1,287 @@ +package sealevel + +import ( + "encoding/binary" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/accounts" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/gagliardetto/solana-go" +) + +// Micro-benchmarks for the natively implemented programs that are still on the +// mainnet hot path (System transfer, Vote TowerSync), measured through the same +// ExecutionCtx.ProcessInstruction entry point replay uses, so the per-instruction +// plumbing (instruction context push/pop, lamport-sum checks, timing metrics) +// is included. Run with: +// +// go test ./pkg/sealevel/ -run XXX -bench 'Native|VoteState|Timing' -benchmem -cpu 1 -count 5 + +func benchPubkey(b byte) solana.PublicKey { + var pk solana.PublicKey + for i := range pk { + pk[i] = b + } + return pk +} + +// newBenchExecCtx mirrors newSystemProgramTestExecCtx without testing.T. +func newBenchExecCtx(txAccts *TransactionAccounts, clockSlot uint64, enabled ...features.FeatureGate) *ExecutionCtx { + txCtx := NewTransactionCtx(*txAccts, 5, 64) + execCtx := &ExecutionCtx{TransactionContext: txCtx, ComputeMeter: cu.NewComputeMeter(1 << 62)} + execCtx.Accounts = accounts.NewMemAccounts() + + clockAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarClockAddr, &clockAcct); err != nil { + panic(err) + } + WriteClockSysvar(&execCtx.Accounts, SysvarClock{Slot: clockSlot, Epoch: 0}) + + rentAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarRentAddr, &rentAcct); err != nil { + panic(err) + } + WriteRentSysvar(&execCtx.Accounts, SysvarRent{LamportsPerUint8Year: 3480, ExemptionThreshold: 2, BurnPercent: 50}) + + f := features.NewFeaturesDefault() + for _, gate := range enabled { + f.EnableFeature(gate, 0) + } + execCtx.Features = *f + return execCtx +} + +// resetTxCtx gives the execution context a fresh instruction trace/stack for +// the next instruction (each ProcessInstruction consumes one trace slot). +func resetTxCtx(execCtx *ExecutionCtx, txAccts *TransactionAccounts) { + execCtx.TransactionContext = NewTransactionCtx(*txAccts, 5, 64) +} + +// ---------------------------------------------------------------- System + +func encodeSystemTransfer(lamports uint64) []byte { + out := binary.LittleEndian.AppendUint32(nil, uint32(SystemProgramInstrTypeTransfer)) + return binary.LittleEndian.AppendUint64(out, lamports) +} + +func benchmarkSystemTransfer(b *testing.B, skipTiming bool) { + systemProgramAcct := accounts.Account{Key: a.SystemProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + from := accounts.Account{Key: benchPubkey(0x11), Lamports: 1 << 60, Data: []byte{}, Owner: a.SystemProgramAddr} + to := accounts.Account{Key: benchPubkey(0x22), Lamports: 1_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{systemProgramAcct, from, to}) + metas := []AccountMeta{ + {Pubkey: from.Key, IsSigner: true, IsWritable: true}, + {Pubkey: to.Key, IsSigner: false, IsWritable: true}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + instr := encodeSystemTransfer(1) + + execCtx := newBenchExecCtx(txAccts, 1234) + execCtx.SkipTimingMetrics = skipTiming + + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + resetTxCtx(execCtx, txAccts) + if err := execCtx.ProcessInstruction(instr, instrAccts, []uint64{0}); err != nil { + b.Fatal(err) + } + } + b.StopTimer() + if got := txAccts.Accounts[2].Lamports; got != 1_000_000+uint64(b.N) { + b.Fatalf("destination lamports %d, want %d", got, 1_000_000+uint64(b.N)) + } +} + +func BenchmarkNativeSystemTransfer(b *testing.B) { benchmarkSystemTransfer(b, false) } +func BenchmarkNativeSystemTransferNoTiming(b *testing.B) { benchmarkSystemTransfer(b, true) } + +// ---------------------------------------------------------------- Vote + +const ( + benchVoteRoot = uint64(1000) + benchVoteLastSlot = benchVoteRoot + MaxLockoutHistory // 1031 + benchClockSlot = benchVoteLastSlot + 2 +) + +func benchSlotHash(slot uint64) [32]byte { + var h [32]byte + binary.LittleEndian.PutUint64(h[:], slot*0x9e3779b97f4a7c15) + binary.LittleEndian.PutUint64(h[8:], ^slot) + h[31] = 0xa5 + return h +} + +// benchSlotHashes builds a 512-entry SlotHashes sysvar (newest first) that +// covers every slot the benchmark tower refers to. +func benchSlotHashes() SysvarSlotHashes { + const n = 512 + newest := benchVoteLastSlot + 1 + sh := make(SysvarSlotHashes, 0, n) + for i := uint64(0); i < n; i++ { + s := newest - i + sh = append(sh, SlotHash{Slot: s, Hash: benchSlotHash(s)}) + } + return sh +} + +// benchInitialVoteState is a fully populated current-version vote state with a +// full 31-entry tower ending at benchVoteLastSlot. +func benchInitialVoteState(voter solana.PublicKey) *VoteState { + vs := &VoteState{ + NodePubkey: voter, + AuthorizedWithdrawer: voter, + Commission: 10, + PriorVoters: PriorVoters{Index: 31, IsEmpty: true}, + EpochCredits: []EpochCredits{{Epoch: 0, Credits: 1000, PrevCredits: 0}}, + LastTimestamp: BlockTimestamp{Slot: benchVoteLastSlot, Timestamp: 1_700_000_000}, + } + vs.AuthorizedVoters.AuthorizedVoters.Set(0, voter) + root := benchVoteRoot + vs.RootSlot = &root + for i := uint64(0); i < MaxLockoutHistory; i++ { + vs.Votes.PushBack(LandedVote{ + Latency: 1, + Lockout: VoteLockout{Slot: benchVoteRoot + 1 + i, ConfirmationCount: uint32(MaxLockoutHistory - i)}, + }) + } + return vs +} + +func benchSerializedVoteState(vs *VoteState) []byte { + versioned := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + data := make([]byte, VoteStateV3Size) + if err := WriteVersionedVoteStateInPlace(data, versioned); err != nil { + panic(err) + } + return data +} + +// encodeTowerSync encodes a TowerSync that advances the tower by one slot: +// root = old root + 1, lockouts = old lockouts shifted by one slot plus the +// new slot, i.e. exactly what a validator sends every slot. +func encodeTowerSync() []byte { + root := benchVoteRoot + 1 + out := binary.LittleEndian.AppendUint32(nil, uint32(VoteProgramInstrTypeTowerSync)) + out = binary.LittleEndian.AppendUint64(out, root) + out = append(out, byte(MaxLockoutHistory)) // compact-u16, < 0x80 + prev := root + for i := uint64(0); i < MaxLockoutHistory; i++ { + slot := root + 1 + i + out = binary.AppendUvarint(out, slot-prev) + out = append(out, byte(MaxLockoutHistory-i)) + prev = slot + } + last := root + MaxLockoutHistory // benchVoteLastSlot + 1 + h := benchSlotHash(last) + out = append(out, h[:]...) + out = append(out, 1) // Some(timestamp) + out = binary.LittleEndian.AppendUint64(out, uint64(1_700_000_001)) + var blockID [32]byte + out = append(out, blockID[:]...) + return out +} + +type voteBench struct { + execCtx *ExecutionCtx + txAccts *TransactionAccounts + instr []byte + instrAccts []InstructionAccount + initial []byte +} + +func newVoteBench(skipTiming bool) *voteBench { + voter := benchPubkey(0x33) + votePk := benchPubkey(0x44) + initial := benchSerializedVoteState(benchInitialVoteState(voter)) + + voteProgramAcct := accounts.Account{Key: a.VoteProgramAddr, Lamports: 1, Data: []byte{}, Owner: a.NativeLoaderAddr, Executable: true} + voteAcct := accounts.Account{Key: votePk, Lamports: 1_000_000_000, Data: append([]byte(nil), initial...), Owner: a.VoteProgramAddr} + voterAcct := accounts.Account{Key: voter, Lamports: 1_000_000_000, Data: []byte{}, Owner: a.SystemProgramAddr} + txAccts := NewTransactionAccounts([]accounts.Account{voteProgramAcct, voteAcct, voterAcct}) + metas := []AccountMeta{ + {Pubkey: votePk, IsSigner: false, IsWritable: true}, + {Pubkey: voter, IsSigner: true, IsWritable: false}, + } + instrAccts := InstructionAcctsFromAccountMetas(metas, *txAccts) + + execCtx := newBenchExecCtx(txAccts, benchClockSlot, + features.EnableTowerSyncIx, + features.VoteStateAddVoteLatency, + features.TimelyVoteCredits, + features.DeprecateUnusedLegacyVotePlumbing, + ) + execCtx.SkipTimingMetrics = skipTiming + shAcct := accounts.Account{Lamports: 1} + if err := execCtx.Accounts.SetAccount(&SysvarSlotHashesAddr, &shAcct); err != nil { + panic(err) + } + WriteSlotHashesSysvar(&execCtx.Accounts, benchSlotHashes()) + + return &voteBench{execCtx: execCtx, txAccts: txAccts, instr: encodeTowerSync(), instrAccts: instrAccts, initial: initial} +} + +// step runs one TowerSync against the initial vote state (the account data is +// rewound to the initial state first so every iteration performs identical work). +func (vb *voteBench) step() error { + copy(vb.txAccts.Accounts[1].Data, vb.initial) + resetTxCtx(vb.execCtx, vb.txAccts) + return vb.execCtx.ProcessInstruction(vb.instr, vb.instrAccts, []uint64{0}) +} + +func TestNativeVoteTowerSyncBenchSetup(t *testing.T) { + vb := newVoteBench(false) + if err := vb.step(); err != nil { + t.Fatalf("TowerSync failed: %v", err) + } + versioned, err := UnmarshalVersionedVoteState(vb.txAccts.Accounts[1].Data) + if err != nil { + t.Fatal(err) + } + vs := versioned.ConvertToCurrent() + if vs.Votes.Len() != MaxLockoutHistory { + t.Fatalf("tower length %d, want %d", vs.Votes.Len(), MaxLockoutHistory) + } + if last := vs.Votes.Back().Lockout.Slot; last != benchVoteLastSlot+1 { + t.Fatalf("last voted slot %d, want %d", last, benchVoteLastSlot+1) + } + if vs.RootSlot == nil || *vs.RootSlot != benchVoteRoot+1 { + t.Fatalf("root %v, want %d", vs.RootSlot, benchVoteRoot+1) + } +} + +func benchmarkVoteTowerSync(b *testing.B, skipTiming bool) { + vb := newVoteBench(skipTiming) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + if err := vb.step(); err != nil { + b.Fatal(err) + } + } +} + +func BenchmarkNativeVoteTowerSync(b *testing.B) { benchmarkVoteTowerSync(b, false) } +func BenchmarkNativeVoteTowerSyncNoTiming(b *testing.B) { benchmarkVoteTowerSync(b, true) } + +// BenchmarkVoteStateRoundTrip isolates vote-state (de)serialization: decode +// the account, convert to current, re-encode — the fixed cost of every vote. +func BenchmarkVoteStateRoundTrip(b *testing.B) { + data := benchSerializedVoteState(benchInitialVoteState(benchPubkey(0x33))) + out := make([]byte, VoteStateV3Size) + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + versioned, err := UnmarshalVersionedVoteState(data) + if err != nil { + b.Fatal(err) + } + vs := versioned.ConvertToCurrent() + cur := &VoteStateVersions{Type: VoteStateVersionCurrent, Current: *vs} + if err := WriteVersionedVoteStateInPlace(out, cur); err != nil { + b.Fatal(err) + } + } +} From e8dcadc057f05320e01703c5142e8ec82b14f6ef Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 09/22] test: update VASA stack registers for fixed-size interpreter state --- pkg/sbpf/vasa_test.go | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/pkg/sbpf/vasa_test.go b/pkg/sbpf/vasa_test.go index e7d64034c..70296ea43 100644 --- a/pkg/sbpf/vasa_test.go +++ b/pkg/sbpf/vasa_test.go @@ -73,7 +73,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { gapped := NewStack(version, false) defer gapped.Finish() - gappedRegs := make([]uint64, 11) + gappedRegs := new([16]uint64) gappedRegs[10] = VaddrStack + StackFrameSize require.True(t, gapped.Push(gappedRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*3, gappedRegs[10]) @@ -81,7 +81,7 @@ func TestStackFrameGapsCanBeDisabled(t *testing.T) { contiguous := NewStack(version, true) defer contiguous.Finish() - contiguousRegs := make([]uint64, 11) + contiguousRegs := new([16]uint64) contiguousRegs[10] = VaddrStack + StackFrameSize require.True(t, contiguous.Push(contiguousRegs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, contiguousRegs[10]) @@ -94,7 +94,7 @@ func TestStackFrameGapsAreLegacyOnly(t *testing.T) { stack := NewStack(version, false) defer stack.Finish() - regs := make([]uint64, 11) + regs := new([16]uint64) regs[10] = VaddrStack + StackFrameSize require.True(t, stack.Push(regs, 0)) require.Equal(t, VaddrStack+StackFrameSize*2, regs[10]) From 203e105e305743ecf40bb5ebc2f5222afa56d6c2 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:19:02 -0500 Subject: [PATCH 10/22] test: benchmark verified token arithmetic and CPI workloads with warm programs --- pkg/sealevel/program_workloads_bench_test.go | 195 +++++++++++++++++++ 1 file changed, 195 insertions(+) create mode 100644 pkg/sealevel/program_workloads_bench_test.go diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go new file mode 100644 index 000000000..cc6b83e78 --- /dev/null +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -0,0 +1,195 @@ +package sealevel + +import ( + "encoding/binary" + "os" + "path/filepath" + "testing" + + "github.com/Overclock-Validator/mithril/fixtures" + "github.com/Overclock-Validator/mithril/pkg/accounts" + "github.com/Overclock-Validator/mithril/pkg/accountsdb" + a "github.com/Overclock-Validator/mithril/pkg/addresses" + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/Overclock-Validator/mithril/pkg/sbpf/loader" + "github.com/gagliardetto/solana-go" + "github.com/maypok86/otter" + "github.com/stretchr/testify/require" +) + +// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// No live account writes or network calls occur in this harness. Each invocation +// gets fresh account data; the program cache is warm and shared between runs. +type programWorkload struct { + name string + elf []byte + program solana.PublicKey + accts []accounts.Account + metas []AccountMeta + instruction []byte + check func(testing.TB, *ExecutionCtx) +} + +func expandedProgramWorkloads(t testing.TB) []programWorkload { + t.Helper() + program := benchPubkey(0x91) + from := accounts.Account{Key: benchPubkey(0x31), Owner: program, Lamports: 10000} + to := accounts.Account{Key: benchPubkey(0x32), Owner: program, Lamports: 10000} + cases := []programWorkload{{name: "BPF_LamportTransfer", elf: fixtures.Load(t, "sbpf", "cpi_c_to_bpf.so"), program: program, accts: []accounts.Account{from, to}, metas: []AccountMeta{{Pubkey: program}, {Pubkey: from.Key, IsSigner: true, IsWritable: true}, {Pubkey: to.Key, IsSigner: true, IsWritable: true}}, instruction: []byte{0}, check: func(t testing.TB, ctx *ExecutionCtx) { + src, err := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, err) + dst, err := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, err) + require.Equal(t, uint64(9000), src.Lamports) + require.Equal(t, uint64(11000), dst.Lamports) + }}} + + pda, bump, err := solana.FindProgramAddress([][]byte{[]byte("You pass butter")}, program) + require.NoError(t, err) + cases = append(cases, programWorkload{name: "CPI_Rust_SystemAllocate", elf: fixtures.Load(t, "sbpf", "cpi_rust_to_system_program_allocate.so"), program: program, + accts: []accounts.Account{{Key: a.SystemProgramAddr, Owner: a.NativeLoaderAddr, Executable: true, Lamports: 10000}, {Key: pda, Owner: a.SystemProgramAddr, Lamports: 10000}}, + metas: []AccountMeta{{Pubkey: a.SystemProgramAddr}, {Pubkey: pda, IsSigner: true, IsWritable: true}}, instruction: []byte{bump}, check: func(t testing.TB, ctx *ExecutionCtx) { + acct, e := ctx.TransactionContext.Accounts.GetAccount(2) + require.NoError(t, e) + require.Len(t, acct.Data, 1337) + require.NotEmpty(t, ctx.InnerInstrs) + }}) + dir := os.Getenv("MITHRIL_PROGRAM_BENCH_DIR") + if dir == "" { + return cases + } + arithmetic, err := os.ReadFile(filepath.Join(dir, "rotation_compute.so")) + require.NoError(t, err) + for _, iterations := range []uint32{500, 5000} { + n := iterations + data := append([]byte("RC01"), 0, 0, 0, 0) + binary.LittleEndian.PutUint32(data[4:], n) + name := "Arithmetic_500" + if n == 5000 { + name = "Arithmetic_5000" + } + cases = append(cases, programWorkload{name: name, elf: arithmetic, program: program, instruction: data, check: func(t testing.TB, ctx *ExecutionCtx) { + x := uint64(0x9e3779b97f4a7c15) + for i := uint32(0); i < n; i++ { + x = ((x << 7) | (x >> 57)) ^ (uint64(i) + 0x517cc1b727220a95) + } + _, got := ctx.TransactionContext.ReturnData() + require.Len(t, got, 8) + require.Equal(t, x, binary.LittleEndian.Uint64(got)) + }}) + } + token, err := os.ReadFile(filepath.Join(dir, "token2022.so")) + require.NoError(t, err) + mint, auth := benchPubkey(0x51), benchPubkey(0x52) + tokenData := func(amount uint64) []byte { + d := make([]byte, 165) + copy(d, mint[:]) + copy(d[32:], auth[:]) + binary.LittleEndian.PutUint64(d[64:], amount) + d[108] = 1 + return d + } + mintData := make([]byte, 82) + mintData[44] = 6 + mintData[45] = 1 + src := accounts.Account{Key: benchPubkey(0x53), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(1000000)} + dst := accounts.Account{Key: benchPubkey(0x54), Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: tokenData(5)} + instr := append([]byte{12}, binary.LittleEndian.AppendUint64(nil, 1000)...) + instr = append(instr, 6) + cases = append(cases, programWorkload{name: "Token2022_TransferChecked", elf: token, program: solana.Token2022ProgramID, + accts: []accounts.Account{src, {Key: mint, Owner: solana.Token2022ProgramID, Lamports: 10000000, Data: mintData}, dst, {Key: auth, Owner: a.SystemProgramAddr, Lamports: 10000000}}, + metas: []AccountMeta{{Pubkey: src.Key, IsWritable: true}, {Pubkey: mint}, {Pubkey: dst.Key, IsWritable: true}, {Pubkey: auth, IsSigner: true}}, instruction: instr, + check: func(t testing.TB, ctx *ExecutionCtx) { + s, e := ctx.TransactionContext.Accounts.GetAccount(1) + require.NoError(t, e) + d, e := ctx.TransactionContext.Accounts.GetAccount(3) + require.NoError(t, e) + require.Equal(t, uint64(999000), binary.LittleEndian.Uint64(s.Data[64:])) + require.Equal(t, uint64(1005), binary.LittleEndian.Uint64(d.Data[64:])) + }}) + return cases +} +func workloadRunner(t testing.TB, w programWorkload, vasa bool) func() (*ExecutionCtx, error) { + t.Helper() + f := features.NewFeaturesDefault() + if vasa { + f.EnableFeature(features.VirtualAddressSpaceAdjustments, 0) + } + l, err := loader.NewLoaderWithSyscalls(w.elf, func(h uint32) (sbpf.Syscall, bool) { return Syscalls(f, false, h) }, false, f) + require.NoError(t, err) + prog, err := l.Load() + require.NoError(t, err) + require.NoError(t, prog.Verify()) + cache, err := otter.MustBuilder[solana.PublicKey, *accountsdb.ProgramCacheEntry](1024).Cost(func(solana.PublicKey, *accountsdb.ProgramCacheEntry) uint32 { return 1 }).Build() + require.NoError(t, err) + t.Cleanup(cache.Close) + db := &accountsdb.AccountsDb{ProgramCache: cache} + db.AddProgramToCache(w.program, &accountsdb.ProgramCacheEntry{Program: prog}) + _, cached := db.MaybeGetProgramFromCache(w.program) + require.True(t, cached, "warm program must be cached") + return func() (*ExecutionCtx, error) { + list := make([]accounts.Account, len(w.accts)+1) + list[0] = accounts.Account{Key: w.program, Owner: a.BpfLoader2Addr, Lamports: 10000000, Executable: true, Data: w.elf} + for i, acct := range w.accts { + list[i+1] = acct + list[i+1].Data = append([]byte(nil), acct.Data...) + } + tx := NewTransactionAccounts(list) + ctx := newBenchExecCtx(tx, 1337) + ctx.TransactionContext.ComputeBudgetLimits = &ComputeBudgetLimits{UpdatedHeapBytes: 32768} + ctx.Features = *f + ctx.ComputeMeter = cu.NewComputeMeter(1400000) + ctx.SlotCtx = &SlotCtx{Slot: 1337, AccountsDb: db} + ctx.Log = &LogRecorder{} + ctx.RecordInnerInstructions = true + err := ctx.ProcessInstruction(w.instruction, InstructionAcctsFromAccountMetas(w.metas, *tx), []uint64{0}) + return ctx, err + } +} +func TestProgramWorkloadResults(t *testing.T) { + for _, w := range expandedProgramWorkloads(t) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + t.Run(name, func(t *testing.T) { + run := workloadRunner(t, w, vasa) + ctx, err := run() + require.NoError(t, err) + w.check(t, ctx) + t.Logf("cu=%d inner=%d", ctx.ComputeMeter.Used(), len(ctx.InnerInstrs)) + }) + } + } +} +func BenchmarkProgramWorkloads(b *testing.B) { + for _, w := range expandedProgramWorkloads(b) { + for _, vasa := range []bool{false, true} { + name := w.name + if vasa { + name += "_VASA" + } + b.Run(name, func(b *testing.B) { + run := workloadRunner(b, w, vasa) + ctx, err := run() + require.NoError(b, err) + w.check(b, ctx) + used := ctx.ComputeMeter.Used() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + ctx, err = run() + if err != nil { + b.Fatal(err) + } + } + b.StopTimer() + w.check(b, ctx) + b.ReportMetric(float64(used), "cu/op") + }) + } + } +} From 1f4a69bee9d4122b838b37ce3100a7fff800e8cb Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:21:44 -0500 Subject: [PATCH 11/22] docs: record individual-program and matched Alpenglow replay measurements --- docs/sbpf-interpreter-benchmarks.md | 103 ++++++++++++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 docs/sbpf-interpreter-benchmarks.md diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md new file mode 100644 index 000000000..73eb970e5 --- /dev/null +++ b/docs/sbpf-interpreter-benchmarks.md @@ -0,0 +1,103 @@ +# Interpreter performance validation + +This experiment compares `9db7dcb` plus identical benchmark code with the same +base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift +semantics patch is excluded. Three VASA tests were updated to pass the new +`*[16]uint64` register type; no further production optimization was added during +this measurement pass. + +## Individual programs + +Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five +one-second samples per variant with alternating run order. The existing validator +continued running. CPU 15 shares a physical core with CPU 7; these are shared-host +measurements rather than an isolated-machine throughput ceiling. + +The harness has a preloaded program cache and asserts a cache hit before timing. +Every invocation gets fresh account data and execution context. It measures +instruction setup, serialization, execution and publication together; it is not +just the interpreter loop. Timing instrumentation and instruction recording are +enabled identically on both versions. Tests check arithmetic return values, +post-transfer balances, allocation results, and recorded CPI. Requested budgets +are equal and the measured CU charges match between variants. + +| Workload | CU | Median before → after | Speedup | +|---|---:|---:|---:| +| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | +| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | +| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | +| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | +| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | +| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | +| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | + +The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. +The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from +the earlier SPL Token loader-only benchmark: they use different program binaries, +instructions, and include execution-context setup. + +Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` +and `token2022.so` to enable those external fixtures. Without it, the in-repository +BPF/CPI fixtures still run. Pinned input SHA-256 values: + +- Arithmetic ELF: `db7c55d6563c879e35dfe2b24edb0fe0515d5a5ae627fe00e3e247001c786441`. + Source: `ag-transaction-bench` at `7e5a263fa5a1c72088f191daf5b7c5d2484c997c`, + `transaction-bench/program/src/rotation_compute.c`. +- Token-2022 ELF: `a794161408080f690dac00832f45b3c3e2b71f1339586667ad1f979cf91d5b68`. + Public Alpenglow program `TokenzQdBNbLqP5VEhdkAS6EPFLC1PHnBqCXEpPxuEb`, + fetched at RPC context slot 4,231,444, program-data account + `DoU57AYuPFu2QU514RktNPG22QhApEjnKxnBcu4BHDTY`. Strip its 45-byte upgradeable + loader metadata before saving the ELF. Verify the hash; do not silently replace + it with a later deployment. + +``` +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^TestProgramWorkloadResults$' -v +MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ + go test ./pkg/sealevel -run '^$' -bench '^BenchmarkProgramWorkloads$' \ + -benchtime=1s -count=5 -benchmem +``` + +## Recorded Alpenglow blocks + +Each run bootstrapped a fresh isolated AccountsDB from the same public full +snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay +covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and +142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, +nice 19. Two paired runs used baseline/candidate then candidate/baseline order. +The live validator's AccountsDB and configuration were not used or changed. + +All 233 per-slot bank hashes matched between baseline and candidate in both +pairs, excluding the run-specific comment header in `bankhash.log`. + +| Work measured | Pair 1 before → after | Pair 2 before → after | +|---|---:|---:| +| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | +| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | +| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | +| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | +| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | + +The sample includes blocks around 40–46 million CU. Three inspected non-empty +blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and +Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries +attribute whole-transaction CU to every participating program and must not be +summed as if they were exclusive per-program execution costs. + +Whole-block p95 did not improve, and the p99 changes are small/variable. The +slowest candidate blocks in the first pair contained no sBPF execution; their +large timers were dispatch and signature verification. These single-CPU replay +results do not establish a live voting/FAST improvement or production parallel +replay latency. They exclude network wait from ProcessBlock and are not elapsed +end-to-end catch-up times. + +## Correctness and limits + +- Native Zen 5 baseline and candidate differential outputs match for 100,000 + deterministic generated programs; candidate pool-zero checks pass. +- Baseline/candidate workload effects and CU charges match. Targeted race tests + for the interpreter, loader, and workload harness pass; vet passes. +- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; + therefore no complete sealevel test-suite pass is claimed. Broader conformance + testing remains separate from this performance experiment. +- No candidate was deployed and no validator restart was needed. From 2361bc9c62954d1ff3a310893517cb8eebf17da6 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:39:11 +0000 Subject: [PATCH 12/22] sealevel: copy, compare and fill VM memory without temporaries or byte loops sol_memcpy_/sol_memmove_ read the source into a fresh heap buffer and wrote it back; the copy now goes directly between the two translated slices with Go's memmove-semantics copy (overlap handled, source translated first so error precedence is unchanged, and a copy-on-write/growth of the destination region still reads the pre-write bytes because the source slice keeps the previous backing buffer alive). sol_memcmp_ uses bytes.Equal for the common equal case and word-skips to the first differing byte otherwise; sol_memset_ uses clear for zero and a doubling copy for other values. An SPL Token transfer issues two memcpy and four memcmp calls, so this is a small, allocation-free win rather than a large one. Tests: memcmpResult against the previous byte loop on 100k random inputs, memsetBytes over sizes and values, and VM-level memmove/memcpy overlap, error-ordering, copy-on-write-region and memcmp/memset checks. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/sealevel/syscalls_mem.go | 77 +++++++++--- pkg/sealevel/syscalls_mem_test.go | 199 ++++++++++++++++++++++++++++++ 2 files changed, 256 insertions(+), 20 deletions(-) create mode 100644 pkg/sealevel/syscalls_mem_test.go diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 7bea5331e..29834a9ce 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -1,6 +1,7 @@ package sealevel import ( + "bytes" "encoding/binary" //"github.com/Overclock-Validator/mithril/pkg/mlog" @@ -15,14 +16,24 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { return execCtx.ComputeMeter.Consume(cost) } -func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) (err error) { - srcBuf := make([]byte, n) - err = vm.Read(src, srcBuf) +// memmoveImplInternal copies n bytes from src to dst inside the VM without a +// temporary buffer. The source is translated first so a bad source address is +// reported before a bad destination, as before. Go's copy has memmove +// semantics, so overlapping ranges within one region are handled; and when +// the destination translation grows or copy-on-writes an account region, the +// source slice still refers to the previous backing buffer, whose bytes are +// exactly what the old read-then-write sequence would have copied. +func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + srcMem, err := vm.Translate(src, n, false) if err != nil { - return + return err + } + dstMem, err := vm.Translate(dst, n, true) + if err != nil { + return err } - err = vm.Write(dst, srcBuf) - return + copy(dstMem, srcMem) + return nil } // SyscallMemcpyImpl is the implementation of the memcpy (sol_memcpy_) syscall. @@ -76,6 +87,26 @@ func SyscallMemmoveImpl(vm sbpf.VM, dst, src, n uint64) (uint64, error) { var SyscallMemmove = sbpf.SyscallFunc3(SyscallMemmoveImpl) +// memcmpResult returns the C memcmp result of two equal-length slices: zero +// when they are equal, otherwise the difference of the first differing bytes +// as unsigned values, matching Agave's `(b1 as i32) - (b2 as i32)`. +func memcmpResult(a, b []byte) int32 { + if bytes.Equal(a, b) { + return 0 + } + // The slices differ: skip equal 8-byte words, then locate the byte. + i := 0 + for i+8 <= len(a) && binary.LittleEndian.Uint64(a[i:]) == binary.LittleEndian.Uint64(b[i:]) { + i += 8 + } + for ; i < len(a); i++ { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + // SyscallMemcmpImpl is the implementation for the memcmp (sol_memcmp_) syscall. func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemcmp") @@ -96,15 +127,7 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, return syscallErr(err) } - cmpResult := int32(0) - for count := uint64(0); count < n; count++ { - b1 := slice1[count] - b2 := slice2[count] - if b1 != b2 { - cmpResult = int32(b1) - int32(b2) - break - } - } + cmpResult := memcmpResult(slice1, slice2) resultSlice, err := vm.Translate(resultAddr, 4, true) if err != nil { @@ -118,7 +141,23 @@ func SyscallMemcmpImpl(vm sbpf.VM, addr1, addr2, n, resultAddr uint64) (uint64, var SyscallMemcmp = sbpf.SyscallFunc4(SyscallMemcmpImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// memsetBytes fills mem with c using the runtime's block clear for zero and a +// doubling copy otherwise, instead of a byte-at-a-time loop. +func memsetBytes(mem []byte, c byte) { + if len(mem) == 0 { + return + } + if c == 0 { + clear(mem) + return + } + mem[0] = c + for filled := 1; filled < len(mem); filled *= 2 { + copy(mem[filled:], mem[:filled]) + } +} + +// SyscallMemsetImpl is the implementation for the memset (sol_memset_) syscall. func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { //mlog.Log.Debugf("SyscallMemset") @@ -133,16 +172,14 @@ func SyscallMemsetImpl(vm sbpf.VM, dst, c, n uint64) (uint64, error) { return syscallErr(err) } - for i := uint64(0); i < n; i++ { - mem[i] = byte(c) - } + memsetBytes(mem, byte(c)) return syscallSuccess(0) } var SyscallMemset = sbpf.SyscallFunc3(SyscallMemsetImpl) -// SyscallMemcmpImpl is the implementation for the memset (sol_memset_) syscall. +// SyscallAllocFreeImpl is the implementation for the alloc/free (sol_alloc_free_) syscall. func SyscallAllocFreeImpl(vm sbpf.VM, size, freeAddr uint64) (uint64, error) { //mlog.Log.Debugf("SyscallAllocFreeImpl") diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go new file mode 100644 index 000000000..3b02daea0 --- /dev/null +++ b/pkg/sealevel/syscalls_mem_test.go @@ -0,0 +1,199 @@ +package sealevel + +import ( + "bytes" + "encoding/binary" + "math/rand" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + feat "github.com/Overclock-Validator/mithril/pkg/features" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func newMemSyscallVM(t *testing.T, input []byte, regions []sbpf.InputRegion) (*sbpf.Interpreter, *ExecutionCtx) { + t.Helper() + features := feat.NewFeaturesDefault() + execCtx := &ExecutionCtx{Features: *features, ComputeMeter: cu.NewComputeMeter(1_000_000)} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram, Funcs: map[uint32]int64{}}, &sbpf.VMOpts{ + Input: input, + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, + InputRegions: regions, + }) + t.Cleanup(vm.Finish) + return vm, execCtx +} + +func TestSyscallMemmoveOverlapping(t *testing.T) { + for _, test := range []struct { + name string + dst, src, n uint64 + expectMemcpyE bool + }{ + {name: "forward-overlap", dst: 4, src: 0, n: 16, expectMemcpyE: true}, + {name: "backward-overlap", dst: 0, src: 4, n: 16, expectMemcpyE: true}, + {name: "disjoint", dst: 40, src: 0, n: 16}, + {name: "adjacent", dst: 16, src: 0, n: 16}, + {name: "empty", dst: 0, src: 0, n: 0}, + } { + t.Run(test.name, func(t *testing.T) { + input := make([]byte, 64) + for i := range input { + input[i] = byte(i + 1) + } + want := append([]byte(nil), input...) + copy(want[test.dst:test.dst+test.n], want[test.src:test.src+test.n]) + + vm, _ := newMemSyscallVM(t, input, nil) + ret, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, want, input, "memmove must have Go copy (memmove) semantics") + + // memcpy: the same result for disjoint ranges, an error for overlap. + input2 := make([]byte, 64) + for i := range input2 { + input2[i] = byte(i + 1) + } + vm2, _ := newMemSyscallVM(t, input2, nil) + _, err = SyscallMemcpyImpl(vm2, sbpf.VaddrInput+test.dst, sbpf.VaddrInput+test.src, test.n) + if test.expectMemcpyE { + require.ErrorIs(t, err, SyscallErrCopyOverlapping) + } else { + require.NoError(t, err) + require.Equal(t, want, input2) + } + }) + } +} + +func TestSyscallMemmoveBadAddressOrder(t *testing.T) { + input := make([]byte, 32) + vm, _ := newMemSyscallVM(t, input, nil) + // Unreadable source is reported before an unwritable destination. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput+100, 8) + require.Error(t, err) + var badAccess sbpf.ExcBadAccess + require.ErrorAs(t, err, &badAccess) + require.False(t, badAccess.Write, "the source translation must fail first") + // Readable source, write to a read-only region. + _, err = SyscallMemmoveImpl(vm, sbpf.VaddrProgram, sbpf.VaddrInput, 8) + require.Error(t, err) + require.ErrorAs(t, err, &badAccess) + require.True(t, badAccess.Write) +} + +func TestSyscallMemmoveIntoGrowingInputRegion(t *testing.T) { + // The destination region copy-on-writes and grows on first write; the + // source slice taken before that must still yield the original bytes. + original := []byte{1, 2, 3, 4, 5, 6, 7, 8} + var replaced []byte + region := sbpf.InputRegion{ + Offset: 0, + RegionSize: uint64(len(original)), + AddressSpaceReserved: 32, + Writable: false, + AccountIndex: 0, + Data: original, + OnWrite: func(region *sbpf.InputRegion, requestedLen uint64) error { + replaced = make([]byte, 32) + copy(replaced, region.Data) + region.Data = replaced + region.RegionSize = 32 + region.Writable = true + return nil + }, + } + vm, _ := newMemSyscallVM(t, nil, []sbpf.InputRegion{region}) + // Copy the first 4 bytes over bytes 4..8 within the same region: the + // source translation sees the original buffer, the destination the clone. + _, err := SyscallMemmoveImpl(vm, sbpf.VaddrInput+4, sbpf.VaddrInput, 4) + require.NoError(t, err) + require.NotNil(t, replaced, "the first write must trigger the copy-on-write hook") + require.Equal(t, []byte{1, 2, 3, 4, 1, 2, 3, 4}, replaced[:8]) + require.Equal(t, []byte{1, 2, 3, 4, 5, 6, 7, 8}, original, "the shared buffer must stay untouched") +} + +func TestSyscallMemcmpAndMemset(t *testing.T) { + input := make([]byte, 128) + for i := range input { + input[i] = byte(i) + } + vm, _ := newMemSyscallVM(t, input, nil) + + // memcmp of equal and differing 32-byte keys, result written at 96. + copy(input[32:64], input[0:32]) + _, err := SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0), int32(binary.LittleEndian.Uint32(input[96:]))) + input[63] = 0xff + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput, sbpf.VaddrInput+32, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(31)-int32(0xff), int32(binary.LittleEndian.Uint32(input[96:]))) + _, err = SyscallMemcmpImpl(vm, sbpf.VaddrInput+32, sbpf.VaddrInput, 32, sbpf.VaddrInput+96) + require.NoError(t, err) + require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) + + // memset 0xab over 33 bytes, then zero over 9 bytes. + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) + require.Equal(t, byte(97), input[97]) + _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) + require.NoError(t, err) + require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) + require.Equal(t, make([]byte, 9), input[70:79]) + require.Equal(t, bytes.Repeat([]byte{0xab}, 18), input[79:97]) +} + +// The byte loop the syscalls used before memcmpResult; kept as the reference. +func referenceMemcmp(a, b []byte) int32 { + for i := range a { + if a[i] != b[i] { + return int32(a[i]) - int32(b[i]) + } + } + return 0 +} + +func TestMemcmpResultMatchesByteLoop(t *testing.T) { + rng := rand.New(rand.NewSource(7)) + for iter := 0; iter < 100000; iter++ { + n := rng.Intn(70) + a := make([]byte, n) + rng.Read(a) + b := append([]byte(nil), a...) + if n > 0 && rng.Intn(4) != 0 { + b[rng.Intn(n)] = byte(rng.Intn(256)) + if rng.Intn(2) == 0 { + b[rng.Intn(n)] ^= byte(1 + rng.Intn(255)) + } + } + if got, want := memcmpResult(a, b), referenceMemcmp(a, b); got != want { + t.Fatalf("n=%d a=%x b=%x: got %d want %d", n, a, b, got, want) + } + } + if memcmpResult([]byte{0xff}, []byte{0x00}) != 255 || memcmpResult([]byte{0x00}, []byte{0xff}) != -255 { + t.Fatal("memcmp must return the unsigned byte difference") + } + if memcmpResult(nil, nil) != 0 { + t.Fatal("empty compare must be 0") + } +} + +func TestMemsetBytes(t *testing.T) { + for _, n := range []int{0, 1, 2, 3, 7, 8, 9, 31, 32, 33, 100, 1023, 4096, 10001} { + for _, c := range []byte{0, 1, 0x7f, 0xff} { + mem := make([]byte, n) + for i := range mem { + mem[i] = byte(i) + } + memsetBytes(mem, c) + if !bytes.Equal(mem, bytes.Repeat([]byte{c}, n)) { + t.Fatalf("n=%d c=%d: %x", n, c, mem) + } + } + } +} From 32e149e67d0ccecb6d284a4cdeccece8841c6177 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 16 Sep 2026 03:41:52 +0000 Subject: [PATCH 13/22] lthash: vectorize MixIn/MixOut with AVX2 on amd64 LtHash.MixIn/MixOut are 1024-lane uint16 add/subtract loops that run twice per modified account in the accounts delta hash (once for the old value, once for the new). The scalar loop costs ~560 ns per call in the sandbox and roughly 350 ns on Zen 5, so a block with ~10k modified accounts spends several milliseconds of worker CPU on lane arithmetic alone. On amd64 with AVX2 the lanes are now mixed with VPADDW/VPSUBW, 16 lanes per instruction, four vectors per iteration, unaligned loads and stores (28 ns per call here, 20x). Dispatch is a package variable set from cpu.X86.HasAVX2 (golang.org/x/sys is already a direct dependency); other architectures, CPUs without AVX2 and the purego build tag keep the portable loops, which remain the reference. Equals now compares the two arrays directly (runtime memequal) instead of a lane loop. Tests compare the assembly and the dispatched functions against the portable loops on random lanes including wrap-around values, check that MixOut inverts MixIn, that aliased operands behave, and that the generic fallback is selectable; go vet's asmdecl check passes and the package builds under -tags purego and GOARCH=arm64. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_013ctTQDHudYoF3FhmgmvY2y --- pkg/lthash/lthash.go | 18 ++---- pkg/lthash/mix.go | 16 +++++ pkg/lthash/mix_amd64.go | 32 ++++++++++ pkg/lthash/mix_amd64.s | 56 +++++++++++++++++ pkg/lthash/mix_amd64_test.go | 51 +++++++++++++++ pkg/lthash/mix_generic.go | 6 ++ pkg/lthash/mix_test.go | 116 +++++++++++++++++++++++++++++++++++ 7 files changed, 283 insertions(+), 12 deletions(-) create mode 100644 pkg/lthash/mix.go create mode 100644 pkg/lthash/mix_amd64.go create mode 100644 pkg/lthash/mix_amd64.s create mode 100644 pkg/lthash/mix_amd64_test.go create mode 100644 pkg/lthash/mix_generic.go create mode 100644 pkg/lthash/mix_test.go diff --git a/pkg/lthash/lthash.go b/pkg/lthash/lthash.go index 9c4aaee1b..faa09a8a5 100644 --- a/pkg/lthash/lthash.go +++ b/pkg/lthash/lthash.go @@ -120,16 +120,15 @@ func (ltHash *LtHash) Clone() *LtHash { return new } +// MixIn adds other's 1024 lanes to ltHash, lane-wise modulo 2^16. The +// lane arithmetic is vectorized where the platform supports it (see mix.go). func (ltHash *LtHash) MixIn(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] + other.value[i] - } + mixIn(<Hash.value, &other.value) } +// MixOut subtracts other's lanes from ltHash, the inverse of MixIn. func (ltHash *LtHash) MixOut(other *LtHash) { - for i := range numElements { - ltHash.value[i] = ltHash.value[i] - other.value[i] - } + mixOut(<Hash.value, &other.value) } func (ltHash *LtHash) Add(other *LtHash) *LtHash { @@ -143,12 +142,7 @@ func (ltHash *LtHash) Sub(other *LtHash) *LtHash { } func (ltHash *LtHash) Equals(other *LtHash) bool { - for i, element := range ltHash.value { - if element != other.value[i] { - return false - } - } - return true + return ltHash.value == other.value } func (ltHash *LtHash) Checksum() []byte { diff --git a/pkg/lthash/mix.go b/pkg/lthash/mix.go new file mode 100644 index 000000000..6cce9a4d9 --- /dev/null +++ b/pkg/lthash/mix.go @@ -0,0 +1,16 @@ +package lthash + +// mixInGeneric and mixOutGeneric are the portable lane loops. Every +// architecture-specific implementation must produce identical results: the +// lanes are independent uint16 additions and subtractions modulo 2^16. +func mixInGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] += src[i] + } +} + +func mixOutGeneric(dst, src *[numElements]uint16) { + for i := range numElements { + dst[i] -= src[i] + } +} diff --git a/pkg/lthash/mix_amd64.go b/pkg/lthash/mix_amd64.go new file mode 100644 index 000000000..f4818245c --- /dev/null +++ b/pkg/lthash/mix_amd64.go @@ -0,0 +1,32 @@ +//go:build amd64 && !purego + +package lthash + +import "golang.org/x/sys/cpu" + +// useAVX2 selects the vector lane loops. cpu.X86.HasAVX2 already includes +// the operating-system XSAVE/YMM-state check. Tests flip it to compare the +// two implementations on the same machine. +var useAVX2 = cpu.X86.HasAVX2 + +func mixIn(dst, src *[numElements]uint16) { + if useAVX2 { + mixInAVX2(dst, src) + return + } + mixInGeneric(dst, src) +} + +func mixOut(dst, src *[numElements]uint16) { + if useAVX2 { + mixOutAVX2(dst, src) + return + } + mixOutGeneric(dst, src) +} + +//go:noescape +func mixInAVX2(dst, src *[numElements]uint16) + +//go:noescape +func mixOutAVX2(dst, src *[numElements]uint16) diff --git a/pkg/lthash/mix_amd64.s b/pkg/lthash/mix_amd64.s new file mode 100644 index 000000000..2233de57e --- /dev/null +++ b/pkg/lthash/mix_amd64.s @@ -0,0 +1,56 @@ +//go:build amd64 && !purego + +#include "textflag.h" + +// The LtHash value is 1024 uint16 lanes = 2048 bytes = 16 iterations of +// four 32-byte YMM vectors. VPADDW/VPSUBW operate on 16-bit lanes modulo +// 2^16, exactly like the generic Go loop. Loads and stores are unaligned +// (VMOVDQU): LtHash values live inside Go structs with 2-byte alignment. + +// func mixInAVX2(dst, src *[1024]uint16) +TEXT ·mixInAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixin_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPADDW (SI)(AX*1), Y0, Y0 + VPADDW 32(SI)(AX*1), Y1, Y1 + VPADDW 64(SI)(AX*1), Y2, Y2 + VPADDW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixin_loop + VZEROUPPER + RET + +// func mixOutAVX2(dst, src *[1024]uint16) +TEXT ·mixOutAVX2(SB), NOSPLIT, $0-16 + MOVQ dst+0(FP), DI + MOVQ src+8(FP), SI + XORQ AX, AX +mixout_loop: + VMOVDQU (DI)(AX*1), Y0 + VMOVDQU 32(DI)(AX*1), Y1 + VMOVDQU 64(DI)(AX*1), Y2 + VMOVDQU 96(DI)(AX*1), Y3 + VPSUBW (SI)(AX*1), Y0, Y0 + VPSUBW 32(SI)(AX*1), Y1, Y1 + VPSUBW 64(SI)(AX*1), Y2, Y2 + VPSUBW 96(SI)(AX*1), Y3, Y3 + VMOVDQU Y0, (DI)(AX*1) + VMOVDQU Y1, 32(DI)(AX*1) + VMOVDQU Y2, 64(DI)(AX*1) + VMOVDQU Y3, 96(DI)(AX*1) + ADDQ $128, AX + CMPQ AX, $2048 + JB mixout_loop + VZEROUPPER + RET diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go new file mode 100644 index 000000000..0cd378ea2 --- /dev/null +++ b/pkg/lthash/mix_amd64_test.go @@ -0,0 +1,51 @@ +//go:build amd64 && !purego + +package lthash + +import ( + "math/rand" + "testing" +) + +// TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has +// AVX2) against the portable loops so the comparison does not depend on the +// dispatch variable. +func TestMixAVX2AgainstGeneric(t *testing.T) { + if !useAVX2 { + t.Skip("no AVX2 on this machine") + } + rng := rand.New(rand.NewSource(6)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + want, got := *dst, *dst + mixInGeneric(&want, src) + mixInAVX2(&got, src) + if got != want { + t.Fatalf("mixInAVX2 diverges (iteration %d)", iter) + } + want, got = *dst, *dst + mixOutGeneric(&want, src) + mixOutAVX2(&got, src) + if got != want { + t.Fatalf("mixOutAVX2 diverges (iteration %d)", iter) + } + } +} + +// TestMixGenericFallbackSelectable makes sure the dispatch honours the flag, +// so a machine without AVX2 takes the portable path. +func TestMixGenericFallbackSelectable(t *testing.T) { + saved := useAVX2 + defer func() { useAVX2 = saved }() + useAVX2 = false + rng := rand.New(rand.NewSource(8)) + dst := randomLanes(rng) + src := randomLanes(rng) + want := *dst + mixInGeneric(&want, src) + mixIn(dst, src) + if *dst != want { + t.Fatal("generic fallback must be used when AVX2 is disabled") + } +} diff --git a/pkg/lthash/mix_generic.go b/pkg/lthash/mix_generic.go new file mode 100644 index 000000000..79e7b0bb4 --- /dev/null +++ b/pkg/lthash/mix_generic.go @@ -0,0 +1,6 @@ +//go:build !amd64 || purego + +package lthash + +func mixIn(dst, src *[numElements]uint16) { mixInGeneric(dst, src) } +func mixOut(dst, src *[numElements]uint16) { mixOutGeneric(dst, src) } diff --git a/pkg/lthash/mix_test.go b/pkg/lthash/mix_test.go new file mode 100644 index 000000000..272051689 --- /dev/null +++ b/pkg/lthash/mix_test.go @@ -0,0 +1,116 @@ +package lthash + +import ( + "math/rand" + "testing" +) + +func randomLanes(rng *rand.Rand) *[numElements]uint16 { + var lanes [numElements]uint16 + for i := range lanes { + switch rng.Intn(8) { + case 0: + lanes[i] = 0 + case 1: + lanes[i] = 0xffff + case 2: + lanes[i] = 0x8000 + default: + lanes[i] = uint16(rng.Uint32()) + } + } + return &lanes +} + +// TestMixMatchesGeneric checks the platform mixIn/mixOut against the +// portable loops, including wrap-around lanes, and that MixOut inverts MixIn. +func TestMixMatchesGeneric(t *testing.T) { + rng := rand.New(rand.NewSource(3)) + for iter := 0; iter < 2000; iter++ { + dst := randomLanes(rng) + src := randomLanes(rng) + wantIn := *dst + mixInGeneric(&wantIn, src) + gotIn := *dst + mixIn(&gotIn, src) + if gotIn != wantIn { + t.Fatalf("mixIn diverges from the generic loop (iteration %d)", iter) + } + wantOut := *dst + mixOutGeneric(&wantOut, src) + gotOut := *dst + mixOut(&gotOut, src) + if gotOut != wantOut { + t.Fatalf("mixOut diverges from the generic loop (iteration %d)", iter) + } + roundTrip := gotIn + mixOut(&roundTrip, src) + if roundTrip != *dst { + t.Fatalf("mixOut does not invert mixIn (iteration %d)", iter) + } + } + // In-place: mixing a value into itself doubles every lane. + dst := randomLanes(rng) + want := *dst + for i := range want { + want[i] *= 2 + } + mixIn(dst, dst) + if *dst != want { + t.Fatal("mixIn with aliased operands must double every lane") + } + mixOut(dst, dst) + if *dst != [numElements]uint16{} { + t.Fatal("mixOut with aliased operands must clear every lane") + } +} + +func TestLtHashMixInMixOutAndEquals(t *testing.T) { + rng := rand.New(rand.NewSource(4)) + var a, b, c LtHash + a.value = *randomLanes(rng) + b.value = *randomLanes(rng) + c = *a.Clone() + c.MixIn(&b) + if c.Equals(&a) { + t.Fatal("mixing in a random value must change the hash") + } + c.MixOut(&b) + if !c.Equals(&a) { + t.Fatal("MixOut must undo MixIn") + } + c.value[numElements-1]++ + if c.Equals(&a) { + t.Fatal("Equals must see a last-lane difference") + } +} + +func BenchmarkMixIn(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixIn(dst, src) + } +} + +func BenchmarkMixInGeneric(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixInGeneric(dst, src) + } +} + +func BenchmarkMixOut(b *testing.B) { + rng := rand.New(rand.NewSource(5)) + dst := randomLanes(rng) + src := randomLanes(rng) + b.SetBytes(numElements * 2) + for i := 0; i < b.N; i++ { + mixOut(dst, src) + } +} From bfddc2930b4df3469a6f41953fc8c6fd93e3e4a6 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 14/22] test: preserve zero-length memory validation and repair VM fixtures --- pkg/sealevel/sealevel_test.go | 12 +++++++----- pkg/sealevel/syscalls_mem.go | 8 ++++++++ pkg/sealevel/syscalls_mem_test.go | 21 ++++++++++++++++++++- 3 files changed, 35 insertions(+), 6 deletions(-) diff --git a/pkg/sealevel/sealevel_test.go b/pkg/sealevel/sealevel_test.go index 6ad64f3a9..18309927b 100644 --- a/pkg/sealevel/sealevel_test.go +++ b/pkg/sealevel/sealevel_test.go @@ -65,12 +65,14 @@ func TestInterpreter_Noop(t *testing.T) { var log LogRecorder + execCtx := &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()} interpreter := sbpf.NewInterpreter(program, &sbpf.VMOpts{ - HeapMax: 32 * 1024, - Input: nil, - MaxCU: 10000, - Syscalls: ToFunc(syscalls), - Context: &ExecutionCtx{Log: &log, ComputeMeter: cu.NewComputeMeterDefault()}, + HeapMax: 32 * 1024, + Input: nil, + MaxCU: 10000, + Syscalls: ToFunc(syscalls), + Context: execCtx, + ComputeMeter: &execCtx.ComputeMeter, }) require.NotNil(t, interpreter) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 29834a9ce..73f96fc67 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,6 +24,14 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { + // Translate intentionally bypasses address validation for zero-length slices; + // Read/Write did not. Preserve the old syscall validation and error order. + if n == 0 { + if err := vm.Read(src, nil); err != nil { + return err + } + return vm.Write(dst, nil) + } srcMem, err := vm.Translate(src, n, false) if err != nil { return err diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 3b02daea0..6cf5b5ba9 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -137,10 +137,11 @@ func TestSyscallMemcmpAndMemset(t *testing.T) { require.Equal(t, int32(0xff)-int32(31), int32(binary.LittleEndian.Uint32(input[96:]))) // memset 0xab over 33 bytes, then zero over 9 bytes. + sentinel := input[97] // The preceding memcmp result overwrote bytes 96..99. _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+64, 0x1ab, 33) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 33), input[64:97]) - require.Equal(t, byte(97), input[97]) + require.Equal(t, sentinel, input[97]) _, err = SyscallMemsetImpl(vm, sbpf.VaddrInput+70, 0, 9) require.NoError(t, err) require.Equal(t, bytes.Repeat([]byte{0xab}, 6), input[64:70]) @@ -197,3 +198,21 @@ func TestMemsetBytes(t *testing.T) { } } } + +func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { + for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { + for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { + vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) + want := vm.Read(src, nil) + if want == nil { + want = vm.Write(dst, nil) + } + got := memmoveImplInternal(vm, dst, src, 0) + if want == nil { + require.NoError(t, got) + } else { + require.EqualError(t, got, want.Error()) + } + } + } +} From 4ad98058428c78c2e0b77aca19bcb2b205174176 Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:04:12 -0500 Subject: [PATCH 15/22] test: cover unaligned and aliased AVX2 hash lanes --- pkg/lthash/mix_amd64_test.go | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/pkg/lthash/mix_amd64_test.go b/pkg/lthash/mix_amd64_test.go index 0cd378ea2..110c1b39a 100644 --- a/pkg/lthash/mix_amd64_test.go +++ b/pkg/lthash/mix_amd64_test.go @@ -5,6 +5,7 @@ package lthash import ( "math/rand" "testing" + "unsafe" ) // TestMixAVX2AgainstGeneric runs the assembly directly (when the CPU has @@ -49,3 +50,25 @@ func TestMixGenericFallbackSelectable(t *testing.T) { t.Fatal("generic fallback must be used when AVX2 is disabled") } } + +func TestMixAVX2UnalignedAndAliased(t *testing.T) { + if !useAVX2 { + t.Skip("AVX2 unavailable") + } + rng := rand.New(rand.NewSource(19)) + for off := 0; off < 32; off += 2 { + storage := make([]byte, numElements*2+32) + dst := (*[numElements]uint16)(unsafe.Pointer(&storage[off])) + *dst = *randomLanes(rng) + want := *dst + mixInGeneric(&want, &want) + mixInAVX2(dst, dst) + if *dst != want { + t.Fatalf("aliased addition offset %d", off) + } + mixOutAVX2(dst, dst) + if *dst != ([numElements]uint16{}) { + t.Fatalf("aliased subtraction offset %d", off) + } + } +} From 1ca0fb3240e254280edd35586cb85ffdebe88f9b Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:15:39 -0500 Subject: [PATCH 16/22] sbpf: account for resolved call targets in program cache cost --- pkg/sbpf/program.go | 2 +- pkg/sbpf/program_memory_test.go | 12 ++++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) create mode 100644 pkg/sbpf/program_memory_test.go diff --git a/pkg/sbpf/program.go b/pkg/sbpf/program.go index 969a15469..a353a4594 100644 --- a/pkg/sbpf/program.go +++ b/pkg/sbpf/program.go @@ -42,7 +42,7 @@ func (p *Program) MemoryBytes() uint64 { if p == nil { return 0 } - total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + total := uint64(len(p.RO)) + uint64(len(p.Text))*8 + uint64(len(p.CallTargets))*8 if len(p.RO) == 0 { total += uint64(len(p.TextBytes)) } diff --git a/pkg/sbpf/program_memory_test.go b/pkg/sbpf/program_memory_test.go new file mode 100644 index 000000000..3360cf45d --- /dev/null +++ b/pkg/sbpf/program_memory_test.go @@ -0,0 +1,12 @@ +package sbpf + +import "testing" + +func TestProgramMemoryIncludesResolvedCalls(t *testing.T) { + p := &Program{Text: make([]Slot, 20)} + before := p.MemoryBytes() + p.ResolveCallTargets() + if got := p.MemoryBytes() - before; got != 20*8 { + t.Fatalf("resolved-call cache bytes = %d, want 160", got) + } +} From a6f35861ae563be314c72b921487aedb9e50713c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:25:52 -0500 Subject: [PATCH 17/22] test: retain syscall differential coverage and execution reproduction guide --- docs/sbpf-interpreter-benchmarks.md | 80 +++++++++++++---------------- pkg/sealevel/syscalls_mem_test.go | 31 +++++++++++ 2 files changed, 68 insertions(+), 43 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 73eb970e5..d78653991 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -58,46 +58,40 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ -benchtime=1s -count=5 -benchmem ``` -## Recorded Alpenglow blocks - -Each run bootstrapped a fresh isolated AccountsDB from the same public full -snapshot at 4,150,503 and incremental snapshot at 4,231,162. Non-voting RPC replay -covered slots **4,231,163–4,231,418**: 233 replayed blocks, 23 skipped slots, and -142 blocks with sBPF execution. It ran with `--txpar 1`, `GOMAXPROCS=1`, CPU 15, -nice 19. Two paired runs used baseline/candidate then candidate/baseline order. -The live validator's AccountsDB and configuration were not used or changed. - -All 233 per-slot bank hashes matched between baseline and candidate in both -pairs, excluding the run-specific comment header in `bankhash.log`. - -| Work measured | Pair 1 before → after | Pair 2 before → after | -|---|---:|---:| -| All-block median ProcessBlock | 210.34 → 139.08 ms | 206.12 → 142.83 ms | -| All-block total ProcessBlock | 46.12 → 35.49 s | 45.16 → 35.94 s | -| sBPF-block median ProcessBlock | 253.17 → 158.95 ms | 232.03 → 164.95 ms | -| All-block p95 ProcessBlock | 433.13 → 437.99 ms | 420.52 → 442.43 ms | -| All-block p99 ProcessBlock | 556.05 → 552.07 ms | 607.83 → 568.22 ms | - -The sample includes blocks around 40–46 million CU. Three inspected non-empty -blocks used the System program, the AogGeA81 hash-loop workload, SPL Token, and -Memo. This is not evidence for DEX or lending workloads. RPC per-program summaries -attribute whole-transaction CU to every participating program and must not be -summed as if they were exclusive per-program execution costs. - -Whole-block p95 did not improve, and the p99 changes are small/variable. The -slowest candidate blocks in the first pair contained no sBPF execution; their -large timers were dispatch and signature verification. These single-CPU replay -results do not establish a live voting/FAST improvement or production parallel -replay latency. They exclude network wait from ProcessBlock and are not elapsed -end-to-end catch-up times. - -## Correctness and limits - -- Native Zen 5 baseline and candidate differential outputs match for 100,000 - deterministic generated programs; candidate pool-zero checks pass. -- Baseline/candidate workload effects and CU charges match. Targeted race tests - for the interpreter, loader, and workload harness pass; vet passes. -- An older `TestInterpreter_Noop` test panics on both the baseline and candidate; - therefore no complete sealevel test-suite pass is claimed. Broader conformance - testing remains separate from this performance experiment. -- No candidate was deployed and no validator restart was needed. +## Correctness and comparison boundaries + +The generated-program harness compares return values, errors, CU usage and memory +for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate +and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and +verifier semantics changes are excluded from this performance work. + +For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, +identical transaction parallelism, and the same slot interval. Compare normalized +per-slot bank hashes and slot sets before interpreting timings. Compare exact +`ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate +run order and retain raw outputs plus commit IDs outside the merge diff. + +The PR description links the recorded Alpenglow replay results and raw evidence. +Single-core shared-host results do not establish multicore contention or live FAST +inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe +a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now +supplies its execution context's compute meter. + +## Memory syscalls and LtHash + +Memory syscalls retain CU charges, source-before-destination error order, +zero-length behavior, memcpy overlap rejection and memmove overlap support. +Tests cover copy-on-write/growing regions and differential memory/CU results. + +LtHash uses AVX2 only when supported by both CPU and OS; other architectures and +`-tags purego` use portable loops. The vector path preserves 16-bit wraparound and +in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cover +both paths. Component speedups are not block-latency speedups. + +```sh +go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test -tags purego ./pkg/lthash +go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' +SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 +go test ./pkg/lthash -run '^$' -bench BenchmarkMix -benchmem -count=5 +``` diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 6cf5b5ba9..9fe7d22b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -216,3 +216,34 @@ func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { } } } + +// Compare the syscall with the old read-then-write path, including charged CU. +func TestMemoryCopyDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(127)) + for i := 0; i < 300; i++ { + data := make([]byte, 128) + rng.Read(data) + actual, want := append([]byte(nil), data...), append([]byte(nil), data...) + vm, ctx := newMemSyscallVM(t, actual, nil) + ref, refCtx := newMemSyscallVM(t, want, nil) + src, dst, n := uint64(rng.Intn(145)), uint64(rng.Intn(145)), uint64(rng.Intn(80)) + src += sbpf.VaddrInput + dst += sbpf.VaddrInput + _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) + wantErr := MemOpConsume(refCtx, n) + if wantErr == nil { + buf := make([]byte, n) + wantErr = ref.Read(src, buf) + if wantErr == nil { + wantErr = ref.Write(dst, buf) + } + } + if wantErr == nil { + require.NoError(t, gotErr) + } else { + require.EqualError(t, gotErr, wantErr.Error()) + } + require.Equal(t, want, actual) + require.Equal(t, refCtx.ComputeMeter.Remaining(), ctx.ComputeMeter.Remaining()) + } +} From 14c54048b2b0c41a41ef2f25487dbc53245b28fd Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:00:42 -0500 Subject: [PATCH 18/22] test: trim execution benchmark experiments and documentation --- docs/sbpf-interpreter-benchmarks.md | 49 ++------- docs/sha256-syscall.md | 107 ++++++------------- pkg/sealevel/program_workloads_bench_test.go | 2 +- pkg/sealevel/syscalls_sha256_bench_test.go | 81 +------------- 4 files changed, 43 insertions(+), 196 deletions(-) diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index d78653991..095eb0b06 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -1,40 +1,11 @@ -# Interpreter performance validation +# Execution performance validation -This experiment compares `9db7dcb` plus identical benchmark code with the same -base plus the interpreter changes in `2ed5e533`. The separate arithmetic-shift -semantics patch is excluded. Three VASA tests were updated to pass the new -`*[16]uint64` register type; no further production optimization was added during -this measurement pass. +## Program workloads -## Individual programs - -Zen 5 / Ryzen 7 9700X, Go 1.26.4, `GOMAXPROCS=1`, CPU 15, nice 19, five -one-second samples per variant with alternating run order. The existing validator -continued running. CPU 15 shares a physical core with CPU 7; these are shared-host -measurements rather than an isolated-machine throughput ceiling. - -The harness has a preloaded program cache and asserts a cache hit before timing. -Every invocation gets fresh account data and execution context. It measures -instruction setup, serialization, execution and publication together; it is not -just the interpreter loop. Timing instrumentation and instruction recording are -enabled identically on both versions. Tests check arithmetic return values, -post-transfer balances, allocation results, and recorded CPI. Requested budgets -are equal and the measured CU charges match between variants. - -| Workload | CU | Median before → after | Speedup | -|---|---:|---:|---:| -| Token-2022 TransferChecked, no extensions | 1,720 | 19.52 → 13.70 µs | 1.42× | -| Same, VASA | 1,720 | 21.78 → 16.11 µs | 1.35× | -| Arithmetic, 500 iterations | 5,631 | 20.91 → 12.71 µs | 1.65× | -| Arithmetic, 5,000 iterations | 55,131 | 167.24 → 94.55 µs | 1.77× | -| Rust CPI to System Allocate | 2,346 | 16.75 → 13.77 µs | 1.22× | -| Same, VASA | 2,346 | 18.97 → 15.50 µs | 1.22× | -| BPF lamport-transfer fixture | 2,895 | 22.09 → 17.57 µs | 1.26× | - -The arithmetic VASA cases measured 20.09 → 12.07 µs and 170.24 → 97.90 µs. -The BPF lamport-transfer VASA case measured 23.92 → 20.33 µs. These differ from -the earlier SPL Token loader-only benchmark: they use different program binaries, -instructions, and include execution-context setup. +The program harness measures instruction setup, serialization, execution and +publication with a warm program cache and fresh account data per invocation. +It checks return values, account updates, CPI and CU consumption. Loader-only +benchmarks separately measure VM execution and program loading. Set `MITHRIL_PROGRAM_BENCH_DIR` to a directory containing `rotation_compute.so` and `token2022.so` to enable those external fixtures. Without it, the in-repository @@ -71,11 +42,9 @@ per-slot bank hashes and slot sets before interpreting timings. Compare exact `ProcessBlock` wall-clock timers, not summed instruction/worker timers. Alternate run order and retain raw outputs plus commit IDs outside the merge diff. -The PR description links the recorded Alpenglow replay results and raw evidence. +Record tested commit IDs, hardware, Go version, affinity and parallelism with results. Single-core shared-host results do not establish multicore contention or live FAST -inclusion gains. The baseline has failing legacy BPF-loader tests; do not describe -a targeted test pass as a complete sealevel-suite pass. `TestInterpreter_Noop` now -supplies its execution context's compute meter. +inclusion gains. ## Memory syscalls and LtHash @@ -89,7 +58,7 @@ in-place operand aliasing. Randomized, unaligned, inverse and fallback tests cov both paths. Component speedups are not block-latency speedups. ```sh -go test ./pkg/metrics ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader ./pkg/replay +go test ./pkg/lthash ./pkg/sbpf ./pkg/sbpf/loader go test -tags purego ./pkg/lthash go test -race ./pkg/sealevel -run 'TestSyscallMem|TestMemoryCopyDifferential|TestProgramWorkloadResults' SBPF_DIFF_OUT=/tmp/candidate-diff.txt SBPF_CHECK_POOL_ZERO=1 go test ./pkg/sbpf -run TestDifferentialDump -count=1 diff --git a/docs/sha256-syscall.md b/docs/sha256-syscall.md index 10c1b59b2..4045b2bac 100644 --- a/docs/sha256-syscall.md +++ b/docs/sha256-syscall.md @@ -1,89 +1,44 @@ -# SHA-256 syscall overhead +# SHA-256 syscall validation -The syscall decodes the already-translated slice descriptor array directly and -writes the final digest into the translated output buffer. It retains streaming -SHA-256, slice order, memory translations, CU charges and validation order. Output -is written only after all inputs have been read, preserving overlapping-buffer -behavior. No special case for a particular on-chain program is introduced. +The syscall decodes the translated slice descriptors directly and writes the +final digest into the translated output buffer. It retains streaming SHA-256, +slice order, memory translations, CU charges and validation order. Output is +written only after all inputs have been read, preserving overlapping-buffer +behavior. -A bounded 55-byte input-buffer prototype was slower than this simpler path and -is retained only as a benchmark comparison. The baseline reference is copied -from combined review commit `bd17683a`. - -Local Apple M4 Pro, Go 1.26.4, GOMAXPROCS=2, five 200 ms samples per case; -medians below. Each benchmark runs serially through a real interpreter's memory -translation and CU meter, with VM creation outside the timed region. This does -not include VM instruction dispatch, a complete program, or block replay. - -| Input | Original | Direct decoding/output | Buffered prototype | -|---|---:|---:|---:| -| 36 contiguous bytes | 74.69 ns | 43.84 ns | 51.98 ns | -| 32 + 4 bytes, two slices | 89.87 ns | 47.22 ns | 56.38 ns | -| 1,232 bytes | 428.4 ns | 382.1 ns | 396.8 ns | -| 4,096 bytes | 1,292 ns | 1,242 ns | 1,258 ns | - -The two-slice case removes four allocations (112 bytes) per call. This is an -ARM64 component result, not a Zen 5 or full-block speedup claim. Measure native -Zen 5 and captured heavy-block replay before deployment decisions. - -Reproduce with: - -```sh -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256Syscall$' -benchtime=200ms -count=5 -go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' -``` +The frozen reference implementation in the test harness supports differential +checks and before/after benchmarks. Both variants use the same VM, including +its contiguous-region bounds-overflow fix. Differential tests compare hashes, return/error values, remaining CU and all input/output memory over valid inputs, invalid descriptors/addresses, depleted budgets and output aliasing. The existing SHA program fixture also executes. -Testing exposed pre-existing overflow in contiguous VM region bounds checks; -that correction and its regression test are a separate preceding commit. Both -benchmark variants use the corrected VM. The old SHA fixture also needed its -compute-meter pointer initialized for the current interpreter API. +## Benchmarks -## Zen 5 acceleration and captured loop +`BenchmarkSha256Syscall` measures the syscall through a real interpreter's memory +translation and CU meter, with VM creation outside the timed region. It covers +empty, single-slice, multiple-slice and larger inputs; it excludes instruction +dispatch and transaction execution. -The live Go 1.26.4 validator on Ryzen 7 9700X had its actual -`crypto/internal/fips140/sha256.useSHANI` flag set to true. SHA acceleration is -already active. No validator runtime setting was changed. +`BenchmarkSha256CapturedLoop` uses a captured SBF v0 hash loop with 1,000 +iterations and zero initial state. It retains descriptor setup, stack accesses, +digest copying, counter updates and branching. Its test checks the result against +a Go hash chain and checks equal CU consumption for both syscall implementations. +The captured ELF hash and extracted instruction range are recorded in the test. +Transaction loading, CPI and the rest of the original program are excluded. -The isolated loop harness copies text slots 518–545 from the captured SBF v0 -program, resolves the SHA syscall relocation, and supplies 1,000 iterations and -zero initial state. It retains descriptor setup, stack accesses, digest copying, -counter update and loop branching. A test checks its result against a Go hash -chain and checks equal CU consumption for both syscall implementations. This -excludes transaction loading, account dependencies, CPI and the remaining program. - -A locally cross-compiled Go 1.26.4 Linux/amd64 test binary ran with GOMAXPROCS=1, -affinity to CPU 15 and nice=19 on Zen 5. No build or deployment ran on that host. -Three 150 ms samples (medians, per hash iteration): - -| Isolated loop | Time | -|---|---:| -| Original syscall | 234.5 ns | -| Optimized syscall | 165.9 ns | -| Dispatch-only diagnostic control | 109.2 ns | -| Go hash chain without VM | 54.69 ns | - -The optimized loop takes about 29% less time. The dispatch-only control replaces -the syscall with a no-op: it omits hashing, translations and syscall CU charging, -and is only an overhead diagnostic, never a valid execution implementation. -The direct two-slice syscall measured 126–157 ns before and 59–63 ns after; -the buffered-input prototype remained slower at 71–73 ns. These short tests -share a host with other processes; they are not isolated-core latency guarantees. - -A separate short CPU profile of the optimized loop attributed 36.2% cumulative -sampled CPU to the entire SHA syscall, including 14.8% of total CPU in the SHA-NI -compression routine. Most remaining sampled work was VM execution: instruction -dispatch/decoding, stack address translation, loads/stores and compute metering. -Cumulative and flat percentages overlap and must not be added. This profile is -of the harness, not of full-block replay or the live validator. - -The next execution experiment should target measured VM overhead and then replay -captured blocks; these results do not justify a claimed 29% block-time improvement. +The dispatch-only control omits hashing, translations and syscall CU charging; +it is an overhead diagnostic, not a valid execution implementation. The raw Go +hash chain provides another comparison outside the VM. Neither control can +establish a full-block speedup. ```sh -go test ./pkg/sealevel -run '^TestSha256CapturedLoop$' -go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256CapturedLoop$' -benchtime=150ms -count=3 +go test ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256|TestSha256CapturedLoop' +go test -race ./pkg/sealevel -run 'TestSha256SyscallDifferential|TestInterpreter_Sha256' +go test ./pkg/sealevel -run '^$' -bench '^BenchmarkSha256(Syscall|CapturedLoop)$' -benchtime=1s -count=5 -benchmem ``` + +Alternate baseline/candidate run order and record commit IDs, Go version, +hardware and CPU affinity. Report component timings separately from full-program +and replay timings; hardware SHA acceleration also affects these results. diff --git a/pkg/sealevel/program_workloads_bench_test.go b/pkg/sealevel/program_workloads_bench_test.go index cc6b83e78..649d35ddd 100644 --- a/pkg/sealevel/program_workloads_bench_test.go +++ b/pkg/sealevel/program_workloads_bench_test.go @@ -19,7 +19,7 @@ import ( "github.com/stretchr/testify/require" ) -// External ELF inputs are pinned by SHA-256 in the benchmark result manifest. +// External ELF inputs are pinned by SHA-256 in docs/sbpf-interpreter-benchmarks.md. // No live account writes or network calls occur in this harness. Each invocation // gets fresh account data; the program cache is warm and shared between runs. type programWorkload struct { diff --git a/pkg/sealevel/syscalls_sha256_bench_test.go b/pkg/sealevel/syscalls_sha256_bench_test.go index b4240f54e..6c972f37f 100644 --- a/pkg/sealevel/syscalls_sha256_bench_test.go +++ b/pkg/sealevel/syscalls_sha256_bench_test.go @@ -14,8 +14,6 @@ import ( // Frozen syscall implementation from bd17683a; keep independent for differential tests. func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256BaselineReference") - if valsLen > cu.CUSha256MaxSlices { return syscallErr(SyscallErrTooManySlices) } @@ -73,81 +71,6 @@ func sha256BaselineReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) return syscallSuccess(0) } -// Experimental bounded-buffer variant retained only for benchmark comparison. -func sha256SmallInputReference(vm sbpf.VM, valsAddr, valsLen, resultsAddr uint64) (uint64, error) { - //mlog.Log.Debugf("sha256SmallInputReference") - - if valsLen > cu.CUSha256MaxSlices { - return syscallErr(SyscallErrTooManySlices) - } - - execCtx := executionCtx(vm) - err := execCtx.ComputeMeter.Consume(cu.CUSha256BaseCost) - if err != nil { - return syscallCuErr() - } - - hashResult, err := vm.Translate(resultsAddr, 32, true) - if err != nil { - return syscallErr(err) - } - - hasher := sha256.New() - // Inputs up to 55 bytes fit in one padded SHA-256 block. Buffer only - // this bounded case; larger inputs retain streaming hashing. - var small [55]byte - buffered := 0 - streaming := false - if valsLen > 0 { - var vals []byte - - // The data at 'valsAddr' consists of an array of 'slice references', which consists - // of: [ptr (u64)] [size (u64)], hence 16 bytes for each of the slice references that - // refers to an input value to hash. - // Safety: valsLen*16 cannot overflow because of the check versus CUSha256MaxSlices above - vals, err = vm.Translate(valsAddr, valsLen*16, false) - if err != nil { - return syscallErr(err) - } - - var data []byte - - for count := uint64(0); count < valsLen; count++ { - - offset := count * 16 - vec := VectorDescrC{Addr: binary.LittleEndian.Uint64(vals[offset:]), Len: binary.LittleEndian.Uint64(vals[offset+8:])} - - data, err = vm.Translate(vec.Addr, vec.Len, false) - if err != nil { - return syscallErr(err) - } - - cost := max(vec.Len/2, cu.CUMemOpBaseCost) - err = execCtx.ComputeMeter.Consume(cost) - if err != nil { - return syscallCuErr() - } - - if !streaming && len(data) <= len(small)-buffered { - buffered += copy(small[buffered:], data) - } else { - if !streaming { - hasher.Write(small[:buffered]) - streaming = true - } - hasher.Write(data) - } - } - } - if streaming { - hasher.Sum(hashResult[:0]) - } else { - digest := sha256.Sum256(small[:buffered]) - copy(hashResult, digest[:]) - } - return syscallSuccess(0) -} - type sha256Call func(sbpf.VM, uint64, uint64, uint64) (uint64, error) func sha256Fixture(sizes []int) ([]byte, uint64, uint64, uint64) { @@ -211,7 +134,7 @@ func TestSha256SyscallDifferential(t *testing.T) { var wantMem []byte var wantRet, wantCU uint64 var wantErr string - for k, fn := range []sha256Call{sha256BaselineReference, sha256SmallInputReference, SyscallSha256Impl} { + for k, fn := range []sha256Call{sha256BaselineReference, SyscallSha256Impl} { buf := append([]byte(nil), mem...) vm, ctx := sha256VM(buf, budget) ret, err := fn(vm, a, n, out) @@ -242,7 +165,7 @@ func BenchmarkSha256Syscall(b *testing.B) { for _, variant := range []struct { name string fn sha256Call - }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}, {"small", sha256SmallInputReference}} { + }{{"baseline", sha256BaselineReference}, {"lean", SyscallSha256Impl}} { b.Run(tc.name+"/"+variant.name, func(b *testing.B) { mem, a, n, out := sha256Fixture(tc.sizes) vm, ctx := sha256VM(mem, ^uint64(0)) From 1992ab43be6c65afbcef3336fb6e8cd344a9cf4c Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:46:59 -0500 Subject: [PATCH 19/22] sbpf: restore v2 memory opcodes and differential coverage --- docs/sbpf-interpreter-benchmarks.md | 11 ++- pkg/sbpf/interpreter.go | 80 ++++++++++++++++++++ pkg/sbpf/interpreter_v2_test.go | 113 ++++++++++++++++++++++++++++ pkg/sbpf/perf_differential_test.go | 90 +++++++++++++++++++--- 4 files changed, 281 insertions(+), 13 deletions(-) create mode 100644 pkg/sbpf/interpreter_v2_test.go diff --git a/docs/sbpf-interpreter-benchmarks.md b/docs/sbpf-interpreter-benchmarks.md index 095eb0b06..c399f4da5 100644 --- a/docs/sbpf-interpreter-benchmarks.md +++ b/docs/sbpf-interpreter-benchmarks.md @@ -32,9 +32,14 @@ MITHRIL_PROGRAM_BENCH_DIR=/path/to/pinned-fixtures GOMAXPROCS=1 \ ## Correctness and comparison boundaries The generated-program harness compares return values, errors, CU usage and memory -for 100,000 programs. Set `SBPF_DIFF_OUT` separately on the reference and candidate -and compare the files; `SBPF_CHECK_POOL_ZERO=1` also checks reused memory. ARSH and -verifier semantics changes are excluded from this performance work. +for 100,000 generated programs, evenly divided across SBPF v0–v3. The generator +uses v2-specific memory, arithmetic and constant-loading encodings; the dump +reports verifier rejection or execution results, and logs accepted counts per version. +Run the same harness on both trees: set `SBPF_DIFF_OUT` separately on the reference +and candidate and compare the files. `SBPF_CHECK_POOL_ZERO=1` additionally checks +the candidate’s clear-on-return pool invariant; older references may clear on +acquisition instead. ARSH and verifier semantics changes are excluded from this +performance work. For replay comparisons, use fresh isolated AccountsDBs from the same snapshots, identical transaction parallelism, and the same slot interval. Compare normalized diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e7e28fda0..e27392d7c 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -1299,6 +1299,86 @@ func (ip *Interpreter) Write64(addr uint64, x uint64) error { func (ip *Interpreter) executeCold(ins Slot, pc int64, r *[16]uint64) (int64, error) { var err error switch ins.Op() { + // In v2 these encodings are memory operations, not MUL/DIV/MOD. + // Run dispatches their non-v2 arithmetic forms on the hot path. + case OpLd1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint8 + v, err = ip.Read8(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt1BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(ins.Uimm())) + pc++ + case OpSt1BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write8(vma, uint8(r[ins.Src()])) + pc++ + case OpLd2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint16 + v, err = ip.Read16(vma) + r[ins.Dst()] = uint64(v) + pc++ + case OpSt2BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(ins.Uimm())) + pc++ + case OpSt2BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write16(vma, uint16(r[ins.Src()])) + pc++ + case OpLd8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Src()]) + int64(ins.Off())) + var v uint64 + v, err = ip.Read64(vma) + r[ins.Dst()] = v + pc++ + case OpSt8BImm: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, uint64(ins.Imm())) + pc++ + case OpSt8BReg: + if !ip.sbpfVersion.MoveMemoryInstructionClasses() { + err = ExcInvalidInstr + break + } + vma := uint64(int64(r[ins.Dst()]) + int64(ins.Off())) + err = ip.Write64(vma, r[ins.Src()]) + pc++ case OpDiv32Imm: r[ins.Dst()] = uint64(uint32(r[ins.Dst()]) / ins.Uimm()) pc++ diff --git a/pkg/sbpf/interpreter_v2_test.go b/pkg/sbpf/interpreter_v2_test.go new file mode 100644 index 000000000..39fd1e161 --- /dev/null +++ b/pkg/sbpf/interpreter_v2_test.go @@ -0,0 +1,113 @@ +package sbpf + +import ( + "bytes" + "encoding/binary" + "fmt" + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" + "github.com/stretchr/testify/require" +) + +// These byte values alias arithmetic instructions outside v2. Exercise all +// relocated memory instructions through Run, not just the cold handler. +func TestInterpreterV2MemoryOpcodes(t *testing.T) { + for _, width := range []int{1, 2, 4, 8} { + loads := map[int]uint8{1: OpLd1BReg, 2: OpLd2BReg, 4: OpLd4BReg, 8: OpLd8BReg} + immediates := map[int]uint8{1: OpSt1BImm, 2: OpSt2BImm, 4: OpSt4BImm, 8: OpSt8BImm} + registers := map[int]uint8{1: OpSt1BReg, 2: OpSt2BReg, 4: OpSt4BReg, 8: OpSt8BReg} + for _, kind := range []string{"load", "store_imm", "store_reg"} { + for _, region := range []string{"heap", "stack", "input", "rodata", "unmapped"} { + t.Run(fmt.Sprintf("%s/%d/%s", kind, width, region), func(t *testing.T) { + const value = uint64(0xfedcba9876543210) + immediate := uint32(0xf123abcd) + var addr uint64 + switch region { + case "heap": + addr = VaddrHeap + 9 + case "stack": + addr = VaddrStack + 9 + case "input": + addr = VaddrInput + 9 + case "rodata": + addr = VaddrProgram + 9 + case "unmapped": + addr = 0x600000009 + } + // Negative offsets and unaligned addresses must behave identically to + // the original interpreter, including the post-instruction exception PC. + text := diffLoadImm64(5, addr+3, sbpfver.SbpfVersionV2) + text = append(text, diffLoadImm64(6, value, sbpfver.SbpfVersionV2)...) + var op Slot + switch kind { + case "load": + op = slot(loads[width], 0, 5, -3, 0) + case "store_imm": + op = slot(immediates[width], 5, 0, -3, immediate) + case "store_reg": + op = slot(registers[width], 5, 6, -3, 0) + } + text = append(text, op, slot(OpExit, 0, 0, 0, 0)) + p := mkProgram(text, sbpfver.SbpfVersionV2) + p.RO = bytes.Repeat([]byte{0xa5}, 32) + require.NoError(t, p.Verify()) + cm := cu.NewComputeMeter(100) + ip := NewInterpreter(p, &VMOpts{HeapMax: 32, Input: bytes.Repeat([]byte{0xa5}, 32), ComputeMeter: &cm, Syscalls: noSyscalls}) + defer ip.Finish() + var memory []byte + switch region { + case "heap": + memory = ip.heap + case "stack": + memory = ip.stack.mem + case "input": + memory = ip.input + case "rodata": + memory = p.RO + } + if memory != nil { + // Use Write for writable VM storage so pooled-memory tracking is kept. + if region != "rodata" { + require.NoError(t, ip.Write(addr-9, bytes.Repeat([]byte{0xa5}, 32))) + } + } + before := append([]byte(nil), memory...) + ret, used, err := ip.Run() + if region == "unmapped" || (region == "rodata" && kind != "load") { + require.Error(t, err) + var exc *Exception + require.ErrorAs(t, err, &exc) + require.Equal(t, int64(5), exc.PC) + var access ExcBadAccess + require.ErrorAs(t, err, &access) + require.Equal(t, addr, access.Addr) + require.Equal(t, uint64(width), access.Size) + require.Equal(t, kind != "load", access.Write) + require.Equal(t, uint64(95), cm.Remaining()) + require.Equal(t, before, memory) + return + } + require.NoError(t, err) + require.Equal(t, uint64(6), used) + require.Equal(t, uint64(94), cm.Remaining()) + var encoded [8]byte + if kind == "load" { + copy(encoded[:], before[9:9+width]) + require.Equal(t, binary.LittleEndian.Uint64(encoded[:]), ret) + } else { + v := value + if kind == "store_imm" { + signed := int32(immediate) + v = uint64(int64(signed)) + } + binary.LittleEndian.PutUint64(encoded[:], v) + copy(before[9:9+width], encoded[:width]) + } + require.Equal(t, before, memory) + }) + } + } + } +} diff --git a/pkg/sbpf/perf_differential_test.go b/pkg/sbpf/perf_differential_test.go index 1d679d4e4..0d072f564 100644 --- a/pkg/sbpf/perf_differential_test.go +++ b/pkg/sbpf/perf_differential_test.go @@ -2,7 +2,6 @@ package sbpf import ( "bufio" - "encoding/binary" "fmt" "hash/fnv" "math/rand" @@ -97,6 +96,15 @@ func diffRegistry(h uint32) (Syscall, bool) { return nil, false } +// v2 replaces LDDW with MOV32 + HOR64. MOV32 avoids sign-extending the +// low word before ORing in the high word. +func diffLoadImm64(dst uint8, value uint64, ver uint32) []Slot { + if ver == sbpfver.SbpfVersionV2 { + return []Slot{slot(OpMov32Imm, dst, 0, 0, uint32(value)), slot(OpHor64Imm, dst, 0, 0, uint32(value>>32))} + } + return []Slot{slot(OpLddw, dst, 0, 0, uint32(value)), slot(0, 0, 0, 0, uint32(value>>32))} +} + func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { reg := func() uint8 { return uint8(1 + rng.Intn(9)) } // r1..r9 imm := func() uint32 { @@ -117,6 +125,29 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { OpAdd32Imm, OpAdd32Reg, OpSub32Imm, OpSub32Reg, OpMul32Imm, OpMul32Reg, OpDiv32Imm, OpDiv32Reg, OpOr32Imm, OpOr32Reg, OpAnd32Imm, OpAnd32Reg, OpLsh32Imm, OpLsh32Reg, OpRsh32Imm, OpRsh32Reg, OpMod32Imm, OpMod32Reg, OpXor32Imm, OpXor32Reg, OpMov32Imm, OpMov32Reg, OpArsh32Imm, OpArsh32Reg, OpNeg32, OpLe, OpBe} + if ver == sbpfver.SbpfVersionV2 { + // Arithmetic and memory encodings both change in v2. Keep generated ALU + // operations arithmetic rather than generating unintended memory accesses. + replacements := map[uint8]uint8{ + OpMul32Imm: OpLmul32Imm, OpMul32Reg: OpLmul32Reg, + OpMul64Imm: OpLmul64Imm, OpMul64Reg: OpLmul64Reg, + OpDiv32Imm: OpUdiv32Imm, OpDiv32Reg: OpUdiv32Reg, + OpDiv64Imm: OpUdiv64Imm, OpDiv64Reg: OpUdiv64Reg, + OpMod32Imm: OpUrem32Imm, OpMod32Reg: OpUrem32Reg, + OpMod64Imm: OpUrem64Imm, OpMod64Reg: OpUrem64Reg, + } + filtered := alu64[:0] + for _, op := range alu64 { + if op == OpNeg32 || op == OpNeg64 || op == OpLe { + continue + } + if replacement, ok := replacements[op]; ok { + op = replacement + } + filtered = append(filtered, op) + } + alu64 = filtered + } jmp := []uint8{OpJeqImm, OpJeqReg, OpJgtImm, OpJgtReg, OpJgeImm, OpJgeReg, OpJltImm, OpJltReg, OpJleImm, OpJleReg, OpJsetImm, OpJsetReg, OpJneImm, OpJneReg, OpJsgtImm, OpJsgtReg, OpJsgeImm, OpJsgeReg, OpJsltImm, OpJsltReg, OpJsleImm, OpJsleReg} switch rng.Intn(10) { @@ -126,7 +157,7 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { if op == OpLe || op == OpBe { i = []uint32{16, 32, 64}[rng.Intn(3)] } - if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm) && i == 0 { + if (op == OpDiv64Imm || op == OpMod64Imm || op == OpDiv32Imm || op == OpMod32Imm || op == OpUdiv32Imm || op == OpUdiv64Imm || op == OpUrem32Imm || op == OpUrem64Imm) && i == 0 { i = 3 } switch op { @@ -138,6 +169,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(op, reg(), reg(), 0, i)} case 4: // load ops := []uint8{OpLdxb, OpLdxh, OpLdxw, OpLdxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg} + } // base register: r10 (stack) or r5 (heap ptr) or r1 (input ptr) or random var base uint8 var off int16 @@ -154,6 +188,9 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { return []Slot{slot(ops[rng.Intn(4)], reg(), base, off, 0)} case 5: // store ops := []uint8{OpStb, OpSth, OpStw, OpStdw, OpStxb, OpStxh, OpStxw, OpStxdw} + if ver == sbpfver.SbpfVersionV2 { + ops = []uint8{OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} + } var base uint8 var off int16 switch rng.Intn(5) { @@ -184,11 +221,11 @@ func randSlot(rng *rand.Rand, pc, n int, ver uint32, fnPC int64) []Slot { default: // set up pointer registers switch rng.Intn(3) { case 0: // r5 = heap - return []Slot{slot(OpLddw, 5, 0, 0, uint32(VaddrHeap&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrHeap>>32))} + return diffLoadImm64(5, VaddrHeap, ver) case 1: // r1 = input + small - return []Slot{slot(OpLddw, 1, 0, 0, uint32((VaddrInput+uint64(rng.Intn(64)))&0xffffffff)), slot(0, 0, 0, 0, uint32(VaddrInput>>32))} + return diffLoadImm64(1, VaddrInput+uint64(rng.Intn(64)), ver) default: // r9 = random 64-bit - return []Slot{slot(OpLddw, 9, 0, 0, rng.Uint32()), slot(0, 0, 0, 0, uint32(rng.Intn(6)))} + return diffLoadImm64(9, uint64(rng.Uint32())|uint64(rng.Intn(6))<<32, ver) } } } @@ -215,10 +252,14 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { } } // function + store := uint8(OpStxdw) + if ver == sbpfver.SbpfVersionV2 { + store = OpSt8BReg + } body = append(body, slot(OpAdd64Imm, 6, 0, 0, uint32(rng.Intn(100))), slot(OpXor64Reg, 7, 6, 0, 0), - slot(OpStxdw, 10, 7, int16(-8-rng.Intn(64)), 0), + slot(store, 10, 7, int16(-8-rng.Intn(64)), 0), slot(OpExit, 0, 0, 0, 0)) p := mkProgram(body, ver) if ver < sbpfver.SbpfVersionV3 { @@ -231,6 +272,29 @@ func genProgram(rng *rand.Rand, ver uint32) *Program { return p } +// Keep this check in the ordinary suite: merely selecting v2 is insufficient +// if its programs still contain legacy memory opcodes or LDDW and never run. +func TestDifferentialV2Generator(t *testing.T) { + rng := rand.New(rand.NewSource(12345)) + seen := make(map[uint8]bool) + for i := 0; i < 1000; i++ { + p := genProgram(rng, sbpfver.SbpfVersionV2) + if err := p.Verify(); err != nil { + continue + } + for _, ins := range p.Text { + seen[ins.Op()] = true + } + } + for _, op := range []uint8{OpLd1BReg, OpLd2BReg, OpLd4BReg, OpLd8BReg, + OpSt1BImm, OpSt2BImm, OpSt4BImm, OpSt8BImm, + OpSt1BReg, OpSt2BReg, OpSt4BReg, OpSt8BReg} { + if !seen[op] { + t.Errorf("no verifier-accepted v2 program contains opcode %#x", op) + } + } +} + func memHash(bs ...[]byte) uint64 { h := fnv.New64a() for _, b := range bs { @@ -255,8 +319,10 @@ func TestDifferentialDump(t *testing.T) { rng := rand.New(rand.NewSource(12345)) const N = 100000 generated, verified := 0, 0 + var verifiedByVersion [4]int for i := 0; i < N; i++ { - ver := []uint32{0, 0, 3, 1}[rng.Intn(4)] + // Equal representation of every version, independent of RNG consumption. + ver := uint32(i % 4) p := genProgram(rng, ver) generated++ if err := p.Verify(); err != nil { @@ -264,6 +330,7 @@ func TestDifferentialDump(t *testing.T) { continue } verified++ + verifiedByVersion[ver]++ resolveCallTargetsIfSupported(p) input := make([]byte, 700) for j := range input { @@ -332,7 +399,10 @@ func TestDifferentialDump(t *testing.T) { heapPool.Put(hp) } } - t.Logf("generated=%d verified=%d", generated, verified) + for ver, count := range verifiedByVersion { + if count == 0 { + t.Errorf("no verifier-accepted programs for v%d", ver) + } + } + t.Logf("generated=%d verified=%d verified_by_version=%v", generated, verified, verifiedByVersion) } - -var _ = binary.LittleEndian From a52026442e4a9e7fadbd0e72de7fc3af21c03b1e Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:29:23 -0500 Subject: [PATCH 20/22] sbpf: guard pooled write tracking beyond bitmap capacity --- pkg/sbpf/fastmem.go | 3 +++ pkg/sbpf/interpreter.go | 9 ++++++++ pkg/sbpf/pooling_test.go | 46 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+) diff --git a/pkg/sbpf/fastmem.go b/pkg/sbpf/fastmem.go index 421f3ed42..1f15dab3b 100644 --- a/pkg/sbpf/fastmem.go +++ b/pkg/sbpf/fastmem.go @@ -25,6 +25,9 @@ type memRegion struct { // emptyRegion never matches any access. var emptyRegion = memRegion{gapShift: 63} +// A uint64 dirty bitmap can describe exactly 64 pages of 4 KiB. +const fastDirtyBytes = 64 * 4096 + const numFastRegions = 6 // index 5 is a permanently empty catch-all // fastRead returns a host pointer for a size-byte read at vma, or nil if the diff --git a/pkg/sbpf/interpreter.go b/pkg/sbpf/interpreter.go index e27392d7c..96d3cb25b 100644 --- a/pkg/sbpf/interpreter.go +++ b/pkg/sbpf/interpreter.go @@ -162,6 +162,15 @@ func (ip *Interpreter) initRegions() { if len(ip.heap) != 0 { ip.regions[VaddrHeap>>32] = memRegion{base: unsafe.Pointer(&ip.heap[0]), rlen: uint64(len(ip.heap)), wlen: uint64(len(ip.heap)), gapShift: 63} } + // Finish relies on complete write tracking before returning pooled storage. + // Larger heaps (or a future larger stack) must use translateInternal's byte + // ranges: shifting the fast-path bitmap beyond page 63 silently loses writes. + // Reads remain fast; current <=256 KiB writable mappings are unchanged. + for _, idx := range []uint64{VaddrStack >> 32, VaddrHeap >> 32} { + if ip.regions[idx].wlen > fastDirtyBytes { + ip.regions[idx].wlen = 0 + } + } if len(ip.inputRegions) == 0 && len(ip.input) != 0 { ip.regions[VaddrInput>>32] = memRegion{base: unsafe.Pointer(&ip.input[0]), rlen: uint64(len(ip.input)), wlen: uint64(len(ip.input)), gapShift: 63} } diff --git a/pkg/sbpf/pooling_test.go b/pkg/sbpf/pooling_test.go index c4bc5eb0d..4a922a70f 100644 --- a/pkg/sbpf/pooling_test.go +++ b/pkg/sbpf/pooling_test.go @@ -2,10 +2,13 @@ package sbpf import ( "bytes" + "encoding/binary" + "fmt" "sync" "testing" "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf/sbpfver" "github.com/stretchr/testify/require" ) @@ -87,3 +90,46 @@ func BenchmarkVMCreateAndFinish(b *testing.B) { }) } } + +// Exercise actual stores at and beyond the bitmap boundary. Inspect the returned +// buffer directly: sync.Pool is permitted to discard entries, so a subsequent Get +// alone would not reliably detect a missed clear. +func TestPooledHeapDirtyBitmapBoundary(t *testing.T) { + oldUsePool, oldPool := UsePool, heapPool + UsePool = true + heapPool = &sync.Pool{New: func() any { return newHeap() }} + t.Cleanup(func() { UsePool, heapPool = oldUsePool, oldPool }) + for _, size := range []int{fastDirtyBytes, fastDirtyBytes + 1, 2 * fastDirtyBytes} { + for _, ver := range []uint32{sbpfver.SbpfVersionV0, sbpfver.SbpfVersionV2, sbpfver.SbpfVersionV3} { + t.Run(fmt.Sprintf("%d/v%d", size, ver), func(t *testing.T) { + offsets := []int{0, size - 8} + if size >= fastDirtyBytes+8 { + offsets = append(offsets, fastDirtyBytes-4, fastDirtyBytes) + } + var text []Slot + op := uint8(OpStdw) + if ver == sbpfver.SbpfVersionV2 { + op = OpSt8BImm + } + for _, off := range offsets { + text = append(text, diffLoadImm64(5, VaddrHeap+uint64(off), ver)...) + text = append(text, slot(op, 5, 0, 0, 0x12345678)) + } + text = append(text, slot(OpExit, 0, 0, 0, 0)) + program := mkProgram(text, ver) + require.NoError(t, program.Verify()) + meter := cu.NewComputeMeter(100) + ip := NewInterpreter(program, &VMOpts{HeapMax: size, ComputeMeter: &meter, Syscalls: noSyscalls}) + // Always clear test storage on failure so later cases cannot inherit dirt. + defer func() { clear(ip.heap) }() + require.NotNil(t, ip.fastRead(VaddrHeap+uint64(size-8), 8)) + _, _, err := ip.Run() + require.NoError(t, err) + require.Equal(t, uint64(0x12345678), binary.LittleEndian.Uint64(ip.heap[offsets[len(offsets)-1]:])) + ip.Finish() + require.True(t, bytes.Equal(make([]byte, size), ip.heap), "Finish must clear every written byte before pooling") + require.Equal(t, size <= fastDirtyBytes, ip.regions[VaddrHeap>>32].wlen != 0) + }) + } + } +} From 3830a184f68486ebfbea7d4b5dbfb3377b60e02e Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:59:32 -0500 Subject: [PATCH 21/22] sealevel: match Agave zero-length memory copy behavior --- pkg/sealevel/syscalls_mem.go | 9 +++------ pkg/sealevel/syscalls_mem_test.go | 23 +++++++++++------------ 2 files changed, 14 insertions(+), 18 deletions(-) diff --git a/pkg/sealevel/syscalls_mem.go b/pkg/sealevel/syscalls_mem.go index 73f96fc67..14a1e39b0 100644 --- a/pkg/sealevel/syscalls_mem.go +++ b/pkg/sealevel/syscalls_mem.go @@ -24,13 +24,10 @@ func MemOpConsume(execCtx *ExecutionCtx, n uint64) error { // source slice still refers to the previous backing buffer, whose bytes are // exactly what the old read-then-write sequence would have copied. func memmoveImplInternal(vm sbpf.VM, dst, src, n uint64) error { - // Translate intentionally bypasses address validation for zero-length slices; - // Read/Write did not. Preserve the old syscall validation and error order. + // Agave's touch_slice_mut / translate_slice return empty slices before + // address lookup for zero length. CU was already charged by the syscall. if n == 0 { - if err := vm.Read(src, nil); err != nil { - return err - } - return vm.Write(dst, nil) + return nil } srcMem, err := vm.Translate(src, n, false) if err != nil { diff --git a/pkg/sealevel/syscalls_mem_test.go b/pkg/sealevel/syscalls_mem_test.go index 9fe7d22b7..d375a42b7 100644 --- a/pkg/sealevel/syscalls_mem_test.go +++ b/pkg/sealevel/syscalls_mem_test.go @@ -199,19 +199,18 @@ func TestMemsetBytes(t *testing.T) { } } -func TestSyscallMemoryZeroLengthPreservesValidation(t *testing.T) { +func TestSyscallMemoryZeroLengthSkipsAddressValidation(t *testing.T) { for _, src := range []uint64{sbpf.VaddrInput, 0, ^uint64(0)} { for _, dst := range []uint64{sbpf.VaddrInput, sbpf.VaddrProgram, ^uint64(0)} { - vm, _ := newMemSyscallVM(t, make([]byte, 32), nil) - want := vm.Read(src, nil) - if want == nil { - want = vm.Write(dst, nil) - } - got := memmoveImplInternal(vm, dst, src, 0) - if want == nil { - require.NoError(t, got) - } else { - require.EqualError(t, got, want.Error()) + for _, fn := range []func(sbpf.VM, uint64, uint64, uint64) (uint64, error){SyscallMemmoveImpl, SyscallMemcpyImpl} { + input := bytes.Repeat([]byte{0x42}, 32) + vm, ctx := newMemSyscallVM(t, input, nil) + before := ctx.ComputeMeter.Remaining() + ret, err := fn(vm, dst, src, 0) + require.NoError(t, err) + require.Zero(t, ret) + require.Equal(t, before-cu.CUMemOpBaseCost, ctx.ComputeMeter.Remaining()) + require.Equal(t, bytes.Repeat([]byte{0x42}, 32), input) } } } @@ -231,7 +230,7 @@ func TestMemoryCopyDifferential(t *testing.T) { dst += sbpf.VaddrInput _, gotErr := SyscallMemmoveImpl(vm, dst, src, n) wantErr := MemOpConsume(refCtx, n) - if wantErr == nil { + if wantErr == nil && n > 0 { buf := make([]byte, n) wantErr = ref.Read(src, buf) if wantErr == nil { From a317eb6d72994f7b93f548325b584756bd54f02f Mon Sep 17 00:00:00 2001 From: 7layermagik <7layermagik@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:00:21 -0500 Subject: [PATCH 22/22] sealevel: track sibling header writes in pooled VM memory --- pkg/sealevel/syscalls_call.go | 2 +- pkg/sealevel/syscalls_sibling_pool_test.go | 42 ++++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) create mode 100644 pkg/sealevel/syscalls_sibling_pool_test.go diff --git a/pkg/sealevel/syscalls_call.go b/pkg/sealevel/syscalls_call.go index 0ffe7b0c3..761069485 100644 --- a/pkg/sealevel/syscalls_call.go +++ b/pkg/sealevel/syscalls_call.go @@ -170,7 +170,7 @@ func SyscallGetProcessedSiblingInstructionImpl(vm sbpf.VM, index, metaAddr, prog } if instrCtxFound != nil { - resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, false) + resultsHeaderBytes, err := vm.Translate(metaAddr, ProcessedSiblingInstructionSize, true) if err != nil { return syscallErr(err) } diff --git a/pkg/sealevel/syscalls_sibling_pool_test.go b/pkg/sealevel/syscalls_sibling_pool_test.go new file mode 100644 index 000000000..a3b3b9e22 --- /dev/null +++ b/pkg/sealevel/syscalls_sibling_pool_test.go @@ -0,0 +1,42 @@ +package sealevel + +import ( + "testing" + + "github.com/Overclock-Validator/mithril/pkg/cu" + "github.com/Overclock-Validator/mithril/pkg/sbpf" + "github.com/stretchr/testify/require" +) + +func TestSiblingHeaderWriteIsClearedOnFinish(t *testing.T) { + old := sbpf.UsePool + sbpf.UsePool = true + t.Cleanup(func() { sbpf.UsePool = old }) + for _, addr := range []uint64{sbpf.VaddrStack + 128, sbpf.VaddrHeap + 128} { + ctx := &ExecutionCtx{ComputeMeter: cu.NewComputeMeter(100000), TransactionContext: &TransactionCtx{ + InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}, + }} + vm := sbpf.NewInterpreter(&sbpf.Program{TextVA: sbpf.VaddrProgram}, &sbpf.VMOpts{HeapMax: 32768, Context: ctx, ComputeMeter: &ctx.ComputeMeter}) + // Keep a read-only view so the test itself does not mark the header dirty. + header, err := vm.Translate(addr, ProcessedSiblingInstructionSize, false) + require.NoError(t, err) + require.Equal(t, make([]byte, 16), header) + found, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, addr, 0, 0, 0) + require.NoError(t, err) + require.Equal(t, uint64(1), found) + require.Equal(t, byte(3), header[0]) + vm.Finish() + // Inspect before allocating another VM: this is deterministic and does not + // depend on sync.Pool choosing a particular backing buffer on the next Get. + require.Equal(t, make([]byte, 16), header) + } +} + +func TestSiblingHeaderRejectsReadOnlyDestination(t *testing.T) { + data := make([]byte, 16) + vm, ctx := newMemSyscallVM(t, nil, []sbpf.InputRegion{{RegionSize: 16, AddressSpaceReserved: 16, Data: data}}) + ctx.TransactionContext = &TransactionCtx{InstructionTrace: []InstructionCtx{{Data: []byte{1, 2, 3}}, {}, {}}, InstructionStack: []uint64{1}} + _, err := SyscallGetProcessedSiblingInstructionImpl(vm, 0, sbpf.VaddrInput, 0, 0, 0) + require.Error(t, err) + require.Equal(t, make([]byte, 16), data) +}