diff --git a/cmd/seid/cmd/legacy_config_fuzz_test.go b/cmd/seid/cmd/legacy_config_fuzz_test.go index cf2edd898c..684749080a 100644 --- a/cmd/seid/cmd/legacy_config_fuzz_test.go +++ b/cmd/seid/cmd/legacy_config_fuzz_test.go @@ -176,24 +176,20 @@ var tmKeys = []tmKey{ }, } -// FuzzHashVaultDisabledUnsafeResolution pins the root-scope kill switch for the -// app-hash equivocation guard. +// FuzzHashVaultHaltOnMismatchResolution pins the root-scope switch that selects whether a hash vault +// mismatch halts the node. // -// Two things make it worth its own target. It is a bool whose safe value is the -// default, so an absent key must resolve false — setting it true removes -// equivocation protection with only a log banner. And it lives at TOML root scope, -// before any [section] header: nested under a section it parses as a different key -// and is silently ignored, which reads as "I disabled the guard" while the guard -// stays on, and would read the other way round if the scope were ever mishandled. -// The document is built from the fuzzer's choices rather than taken as free text, -// so the expected outcome follows from construction instead of being a second -// input the fuzzer can mutate out of agreement with the first. -func FuzzHashVaultDisabledUnsafeResolution(f *testing.F) { +// An absent key resolves false, so a mismatch replaces the recorded state hash with only an error log. +// The key lives at TOML root scope, before any [section] header: nested under a section it parses as a +// different key and is silently ignored. The document is built from the fuzzer's choices rather than +// taken as free text, so the expected outcome follows from construction instead of being a second input +// the fuzzer can mutate out of agreement with the first. +func FuzzHashVaultHaltOnMismatchResolution(f *testing.F) { f.Add(false, false, false) - f.Add(true, true, false) // root scope, true: the guard is off - f.Add(true, false, false) // root scope, false - f.Add(true, true, true) // nested under a section: silently ignored - f.Add(true, false, true) + f.Add(true, false, false) // root scope, false: a mismatch only logs + f.Add(true, true, false) // root scope, true + f.Add(true, false, true) // nested under a section: silently ignored + f.Add(true, true, true) f.Fuzz(func(t *testing.T, present, value, underSection bool) { configtest.Isolate(t) @@ -204,37 +200,43 @@ func FuzzHashVaultDisabledUnsafeResolution(f *testing.F) { if underSection { doc.WriteString("[p2p]\n") } - fmt.Fprintf(&doc, "hash-vault-disabled-unsafe = %t\n", value) + fmt.Fprintf(&doc, "hash-vault-halt-on-mismatch = %t\n", value) } if doc.Len() > 0 { home.WriteConfigTOML(t, []byte(doc.String())) } // Root scope is the only placement that resolves. Nested under a section the - // key becomes p2p.hash-vault-disabled-unsafe, which nothing reads. - wantDisabled := present && value && !underSection + // key becomes p2p.hash-vault-halt-on-mismatch, which nothing reads. + wantHalt := false + if present && !underSection { + wantHalt = value + } got := applyLegacy(t, home, nil) if got.err != nil { t.Fatalf("Apply must succeed on a well-formed config.toml, got %v", got.err) } - if got.ctx.Config.HashVaultDisabledUnsafe != wantDisabled { - t.Fatalf("hash-vault-disabled-unsafe resolved to %v, want %v, from:\n%s", - got.ctx.Config.HashVaultDisabledUnsafe, wantDisabled, doc.String()) + if got.ctx.Config.HashVaultHaltOnMismatch != wantHalt { + t.Fatalf("hash-vault-halt-on-mismatch resolved to %v, want %v, from:\n%s", + got.ctx.Config.HashVaultHaltOnMismatch, wantHalt, doc.String()) } }) } -// TestHashVaultDisabledUnsafeDefaultsToEnabledGuard states the default on its own, -// so the guard's safe value is pinned even if every seed above were removed. -func TestHashVaultDisabledUnsafeDefaultsToEnabledGuard(t *testing.T) { +// TestHashVaultDefaults pins the hash vault defaults an empty home resolves to. +func TestHashVaultDefaults(t *testing.T) { configtest.Isolate(t) got := applyLegacy(t, configtest.NewHome(t), nil) if got.err != nil { t.Fatalf("Apply: %v", got.err) } - if got.ctx.Config.HashVaultDisabledUnsafe { - t.Fatal("an empty home must leave the app-hash equivocation guard enabled") + if got.ctx.Config.HashVaultHaltOnMismatch { + t.Fatal("an empty home must leave a hash vault mismatch replacing the recorded hash") + } + if got.ctx.Config.HashVaultEmptyRollbackBlocks != 1000 { + t.Fatalf("an empty home must rewind 1000 blocks over an empty hash vault, got %d", + got.ctx.Config.HashVaultEmptyRollbackBlocks) } } diff --git a/config/tendermintbase/tendermintbase.go b/config/tendermintbase/tendermintbase.go index 59f1bca696..20cfab4837 100644 --- a/config/tendermintbase/tendermintbase.go +++ b/config/tendermintbase/tendermintbase.go @@ -57,8 +57,9 @@ var removedFromTheNode = []string{"proxy-app", "abci", "filter-peers"} type nodeRootSchema struct { tmcfg.BaseConfig `mapstructure:",squash"` - AutobahnConfigFile string `mapstructure:"autobahn-config-file"` - HashVaultDisabledUnsafe bool `mapstructure:"hash-vault-disabled-unsafe"` + AutobahnConfigFile string `mapstructure:"autobahn-config-file"` + HashVaultHaltOnMismatch bool `mapstructure:"hash-vault-halt-on-mismatch"` + HashVaultEmptyRollbackBlocks uint64 `mapstructure:"hash-vault-empty-rollback-blocks"` } // removedSettings are the consensus paths this section does not declare. Each names a field the node's @@ -289,9 +290,10 @@ func privValidatorDefaults(mode registry.Mode) any { return *forMode(mode).PrivV func rootDefaults(mode registry.Mode) any { live := forMode(mode) return nodeRootSchema{ - BaseConfig: live.BaseConfig, - AutobahnConfigFile: live.AutobahnConfigFile, - HashVaultDisabledUnsafe: live.HashVaultDisabledUnsafe, + BaseConfig: live.BaseConfig, + AutobahnConfigFile: live.AutobahnConfigFile, + HashVaultHaltOnMismatch: live.HashVaultHaltOnMismatch, + HashVaultEmptyRollbackBlocks: live.HashVaultEmptyRollbackBlocks, } } diff --git a/giga/evmonly/giga_store_test.go b/giga/evmonly/giga_store_test.go index 0f2f22af7e..41fd321dd0 100644 --- a/giga/evmonly/giga_store_test.go +++ b/giga/evmonly/giga_store_test.go @@ -42,6 +42,12 @@ func (s *recordingGigaStore) OpenViewAt(int64) (gigatypes.StateView, bool) { return nil, false } +func (s *recordingGigaStore) GetBlockHeight() uint64 { return 0 } + +func (s *recordingGigaStore) GetBlockHash(uint64) ([32]byte, gigatypes.BlockHashStatus, error) { + return [32]byte{}, gigatypes.BlockHashStatusNotReady, nil +} + func (s *recordingGigaStore) Close() error { return nil } type memoryGigaSnapshot struct { diff --git a/giga/evmonly/memory_store.go b/giga/evmonly/memory_store.go index e20146820c..0eee9ff15a 100644 --- a/giga/evmonly/memory_store.go +++ b/giga/evmonly/memory_store.go @@ -372,6 +372,21 @@ func (s *MemoryStore) RegisterHashListener(_ gigatypes.HashListener) (lthash.Blo return lthash.BlockHash{}, fmt.Errorf("evmonly: an in-memory store computes no block hashes") } +// GetBlockHeight returns the last block committed, or 0 when none has been. +func (s *MemoryStore) GetBlockHeight() uint64 { + s.mu.RLock() + defer s.mu.RUnlock() + if !s.hasCurrentHeight { + return 0 + } + return uint64(s.currentHeight) //nolint:gosec // CommitStateChanges refuses a negative block number +} + +// GetBlockHash reports every block's hash as not ready, since this store computes no block hashes. +func (s *MemoryStore) GetBlockHash(uint64) ([32]byte, gigatypes.BlockHashStatus, error) { + return [32]byte{}, gigatypes.BlockHashStatusNotReady, nil +} + // Close releases nothing. This store holds no handle outside its own maps, which go with it. func (s *MemoryStore) Close() error { return nil } diff --git a/giga/metrics/autobahn_loop.go b/giga/metrics/autobahn_loop.go index 3822208a14..81d96b99ad 100644 --- a/giga/metrics/autobahn_loop.go +++ b/giga/metrics/autobahn_loop.go @@ -23,8 +23,6 @@ const ( // PhaseStorage is time spent persisting receipts, state, and the app commit. PhaseStorage = "storage" - // StoragePhaseVaultCommit is the app hash's durable write to the hash vault. - StoragePhaseVaultCommit = "vault_commit" // StoragePhaseAppCommit is the app's Commit call. StoragePhaseAppCommit = "app_commit" // StoragePhasePushAppHash is publishing the app hash to the data layer. @@ -34,8 +32,6 @@ const ( StoragePhaseBookkeeping = "bookkeeping" // StoragePhasePruneData is pruning the data layer below the app's retain height. StoragePhasePruneData = "prune_data" - // StoragePhasePruneVault is pruning the hash vault to the same boundary. - StoragePhasePruneVault = "prune_vault" ) var ( diff --git a/sei-db/bench/cryptosim/block_hashes.go b/sei-db/bench/cryptosim/block_hashes.go index b10fac4961..af6d85fe99 100644 --- a/sei-db/bench/cryptosim/block_hashes.go +++ b/sei-db/bench/cryptosim/block_hashes.go @@ -38,7 +38,7 @@ type blockHashWaiter struct { committed int // The block the next hash taken must describe, or 0 until the first one has been taken. - nextExpected int64 + nextExpected uint64 // How long to wait for one hash before reporting a database that has stopped hashing. waitTimeout time.Duration @@ -62,7 +62,7 @@ func newBlockHashWaiter(lagBlocks int, metrics *CryptosimMetrics) *blockHashWait // Blocking here is the backpressure: it stops a database finalizing blocks faster than the benchmark // accepts their hashes. The context is the release, cancelled when the database shuts down, since a // send with no taker left would otherwise never return. -func (w *blockHashWaiter) listen(ctx context.Context, _ int64, hash *lthash.BlockHash) error { +func (w *blockHashWaiter) listen(ctx context.Context, _ uint64, hash *lthash.BlockHash) error { select { case w.hashes <- hash: return nil diff --git a/sei-db/bench/cryptosim/block_hashes_test.go b/sei-db/bench/cryptosim/block_hashes_test.go index 0a0e7c5f85..56bd13a0d1 100644 --- a/sei-db/bench/cryptosim/block_hashes_test.go +++ b/sei-db/bench/cryptosim/block_hashes_test.go @@ -11,7 +11,7 @@ import ( ) // publish hands the waiter the hash of one block, as the database's dispatch would. -func publish(t *testing.T, w *blockHashWaiter, blockNumber int64) { +func publish(t *testing.T, w *blockHashWaiter, blockNumber uint64) { t.Helper() require.NoError(t, w.listen(t.Context(), blockNumber, <hash.BlockHash{BlockNumber: blockNumber})) } @@ -31,7 +31,7 @@ func TestTheFirstBlocksRunAheadWithoutTakingAHash(t *testing.T) { // would let the benchmark drift arbitrarily far ahead of hashing. func TestOneHashIsTakenPerBlockAfterTheWindow(t *testing.T) { waiter := newBlockHashWaiter(3, nil) - for block := int64(1); block <= 4; block++ { + for block := uint64(1); block <= 4; block++ { publish(t, waiter, block) } @@ -40,7 +40,7 @@ func TestOneHashIsTakenPerBlockAfterTheWindow(t *testing.T) { require.NoError(t, waiter.awaitBlock()) } require.Len(t, waiter.hashes, 3, "exactly one hash may be taken per block committed") - require.Equal(t, int64(2), waiter.nextExpected) + require.Equal(t, uint64(2), waiter.nextExpected) } // The wait is the point: a database that cannot hash as fast as the benchmark commits has to slow the diff --git a/sei-db/bench/cryptosim/cryptosim.go b/sei-db/bench/cryptosim/cryptosim.go index 0158ce18da..5cc95d28c4 100644 --- a/sei-db/bench/cryptosim/cryptosim.go +++ b/sei-db/bench/cryptosim/cryptosim.go @@ -14,6 +14,7 @@ import ( "github.com/sei-protocol/sei-chain/sei-db/common/utils" "github.com/sei-protocol/sei-chain/sei-db/controller" "github.com/sei-protocol/sei-chain/sei-db/state_db/giga" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" ) const ( @@ -146,7 +147,10 @@ func NewCryptoSim( // giga.NewStateDB is the node's own entry point, and the only one that leaves the state WAL // outside the live state DB: it opens the WAL itself and writes each block to it ahead of the // commit. A live state DB opened directly would own its WAL and write it inline instead. - db, err := giga.NewStateDB(ctx, config.FlatKVConfig, config.StateStoreConfig, config.CheckpointConfig) + hashVaultConfig := hashvault.DefaultHashVaultConfig() + hashVaultConfig.DataDir = filepath.Join(config.DataDir, "hashvault") + db, err := giga.NewStateDB( + ctx, config.FlatKVConfig, config.StateStoreConfig, config.CheckpointConfig, hashVaultConfig, 0) if err != nil { cancel() return nil, fmt.Errorf("failed to open the state DB: %w", err) diff --git a/sei-db/bench/gigasim/block_hashes.go b/sei-db/bench/gigasim/block_hashes.go index bfe9f71bee..a3d236c6ea 100644 --- a/sei-db/bench/gigasim/block_hashes.go +++ b/sei-db/bench/gigasim/block_hashes.go @@ -34,7 +34,7 @@ type blockHashWaiter struct { committed int // The block the next hash taken must describe, or 0 until the first one has been taken. - nextExpected int64 + nextExpected uint64 // How long to wait for one hash before reporting a state DB that has stopped hashing. waitTimeout time.Duration @@ -55,7 +55,7 @@ func newBlockHashWaiter(lagBlocks int, metrics *GigasimMetrics) *blockHashWaiter // listen takes one block's hash from the state DB, blocking while the benchmark is further ahead than // its window allows. That block is the backpressure on hashing; the context releases it when the state // DB shuts down, since a send with no taker left would never return. -func (w *blockHashWaiter) listen(ctx context.Context, _ int64, hash *lthash.BlockHash) error { +func (w *blockHashWaiter) listen(ctx context.Context, _ uint64, hash *lthash.BlockHash) error { select { case w.hashes <- hash: return nil diff --git a/sei-db/bootstrap/recovery.go b/sei-db/bootstrap/recovery.go index e6f598ed72..4366da283c 100644 --- a/sei-db/bootstrap/recovery.go +++ b/sei-db/bootstrap/recovery.go @@ -56,7 +56,7 @@ func (m *GigaStorageManager) OpenDBWithRecovery(ctx context.Context) error { // State goes first because it is the rollback that refuses: a target its snapshots and WAL cannot span // leaves the node down for an operator to retry at a higher one, and receipts cut to the lower target // would no longer be there to reach. -func (m *GigaStorageManager) recoverStores(ctx context.Context, target int64) error { +func (m *GigaStorageManager) recoverStores(ctx context.Context, target uint64) error { if target == 0 { logger.Info("No height to converge on, opening the state DB where its files sit") return m.openStateDB(ctx) @@ -104,7 +104,7 @@ func (m *GigaStorageManager) openReceiptStore() error { // // The state and receipt heads are read from their directories, which takes the locks their open stores // hold, so this must run before either of those stores opens. -func (m *GigaStorageManager) findTargetRecoveryHeight() (int64, error) { +func (m *GigaStorageManager) findTargetRecoveryHeight() (uint64, error) { logger.Info("Reading a store head", "store", "block store") blockHeight, err := m.blockStore.GetLatestBlock() if err != nil { @@ -129,7 +129,7 @@ func (m *GigaStorageManager) findTargetRecoveryHeight() (int64, error) { "state_wal", stateHeight, "receipt_store", receiptHeight, "target", target) - return int64(target), nil //nolint:gosec // heights fit within int64 + return target, nil } // ErrReceiptStoreDisabledWithHistory is returned when the receipt store is disabled while its @@ -188,10 +188,11 @@ func recoveryTarget(blockHeight, stateHeight, receiptHeight uint64) uint64 { return target } -// openStateDB opens the state commit store, the EVM state store (when enabled) and the state WAL, -// leaving them on the height the WAL holds. +// openStateDB opens the state commit store, the EVM state store (when enabled), the state WAL and the +// hash vault, leaving the stores on the height the WAL holds. func (m *GigaStorageManager) openStateDB(ctx context.Context) error { - stateDB, err := giga.NewStateDB(ctx, m.cfg.FlatKVConfig, m.cfg.SSConfig, m.cfg.CheckpointConfig) + stateDB, err := giga.NewStateDB( + ctx, m.cfg.FlatKVConfig, m.cfg.SSConfig, m.cfg.CheckpointConfig, m.cfg.HashVaultConfig, 0) if err != nil { return err } @@ -199,13 +200,20 @@ func (m *GigaStorageManager) openStateDB(ctx context.Context) error { return nil } -// openStateDBAt opens the same three stores on target, rolling them back to it first. +// openStateDBAt opens the same stores on target, rolling them back to it first. The hash vault is not +// rolled back. // // The rollback is part of the open because cutting the state WAL's tail needs the WAL closed, so an // already-open state DB would have to close and reopen it. -func (m *GigaStorageManager) openStateDBAt(ctx context.Context, target int64) error { - stateDB, err := giga.NewStateDBWithRollback( - ctx, m.cfg.FlatKVConfig, m.cfg.SSConfig, m.cfg.CheckpointConfig, target) +func (m *GigaStorageManager) openStateDBAt(ctx context.Context, target uint64) error { + if target == 0 { + // The state DB reads a target of 0 as no rollback at all, so without this a caller asking for one + // would get a plain open instead. + return fmt.Errorf("rollback target %d is invalid: version 0 means no state, so there is "+ + "nothing to roll back to", target) + } + stateDB, err := giga.NewStateDB( + ctx, m.cfg.FlatKVConfig, m.cfg.SSConfig, m.cfg.CheckpointConfig, m.cfg.HashVaultConfig, target) if err != nil { return err } @@ -217,12 +225,11 @@ func (m *GigaStorageManager) openStateDBAt(ctx context.Context, target int64) er // open store. A store already at or below target is left alone. // // It takes the locks an open receipt store holds, so it must run before openReceiptStore. -func (m *GigaStorageManager) recoverReceipt(target int64) error { +func (m *GigaStorageManager) recoverReceipt(target uint64) error { if !m.cfg.ReceiptDBConfig.Enable { return nil } - //nolint:gosec // recoverStores guards target > 0 - if err := receipt.PruneAfter(m.cfg.ReceiptDBConfig, uint64(target)); err != nil { + if err := receipt.PruneAfter(m.cfg.ReceiptDBConfig, target); err != nil { return fmt.Errorf("roll the receipt store back to %d: %w", target, err) } return nil diff --git a/sei-db/bootstrap/recovery_test.go b/sei-db/bootstrap/recovery_test.go index bebe11a247..c731a0adc1 100644 --- a/sei-db/bootstrap/recovery_test.go +++ b/sei-db/bootstrap/recovery_test.go @@ -114,7 +114,7 @@ func commitBlocksWithSSSnapshots(t *testing.T, manager *GigaStorageManager, thro // reconverge re-runs what a restart does: it closes every store recovery touches, recovers them onto // target — which is what opens the state DB again — and reopens the receipt store on the far side. -func reconverge(t *testing.T, manager *GigaStorageManager, target int64) { +func reconverge(t *testing.T, manager *GigaStorageManager, target uint64) { t.Helper() require.NoError(t, reconvergeErr(t, manager, target)) require.NoError(t, manager.openReceiptStore()) @@ -122,7 +122,7 @@ func reconverge(t *testing.T, manager *GigaStorageManager, target int64) { // reconvergeErr is reconverge up to the point recovery can fail, for a test that expects it to. The // receipt store is left closed, since a failed recovery leaves the manager with no state DB. -func reconvergeErr(t *testing.T, manager *GigaStorageManager, target int64) error { +func reconvergeErr(t *testing.T, manager *GigaStorageManager, target uint64) error { t.Helper() // Closing first is what a restart does, and it is also required: recovery takes file locks the // open stores hold — the state WAL's directory lock for the reads and the tail cut that precede @@ -252,7 +252,7 @@ func TestFindTargetRecoveryHeightIsZeroWithoutABlockLedger(t *testing.T) { got, err := manager.findTargetRecoveryHeight() require.NoError(t, err) - require.Equal(t, int64(0), got) + require.Equal(t, uint64(0), got) } // Recovering to a target below the WAL head drops every block above it, so the write head resumes at diff --git a/sei-db/bootstrap/storage_manager.go b/sei-db/bootstrap/storage_manager.go index d8c8f37e90..64752744c1 100644 --- a/sei-db/bootstrap/storage_manager.go +++ b/sei-db/bootstrap/storage_manager.go @@ -78,7 +78,7 @@ func (m *GigaStorageManager) startGarbageCollector(ctx context.Context, pruningC // prunableStores returns the opened stores that can join the shared prune cycle. The state stores come // from the StateDB that owns them. func (m *GigaStorageManager) prunableStores() []controller.PrunableStore { - stores := make([]controller.PrunableStore, 0, 5) + stores := make([]controller.PrunableStore, 0, 6) if m.stateDB != nil { stores = append(stores, m.stateDB.PrunableStores()...) } diff --git a/sei-db/bootstrap/storage_manager_test.go b/sei-db/bootstrap/storage_manager_test.go index bd2192d57d..29cdf6add0 100644 --- a/sei-db/bootstrap/storage_manager_test.go +++ b/sei-db/bootstrap/storage_manager_test.go @@ -145,11 +145,11 @@ func TestStateStoreDisabled(t *testing.T) { "SC still needs the schedule that replaces its own interval") require.True(t, manager.SC().ExternalPruning()) - names := make([]string, 0, 4) + names := make([]string, 0, 5) for _, store := range manager.prunableStores() { names = append(names, store.Name()) } - require.Equal(t, []string{"FlatKV", "StateWAL", "ReceiptDB", "BlockDB"}, names, + require.Equal(t, []string{"FlatKV", "StateWAL", "HashVault", "ReceiptDB", "BlockDB"}, names, "a store this node never opened must not be offered to the collector") } @@ -285,14 +285,14 @@ func TestEveryStoreJoinsThePruneCycle(t *testing.T) { require.True(t, manager.SC().ExternalPruning()) require.True(t, manager.SS().ExternalPruning()) - names := make([]string, 0, 5) + names := make([]string, 0, 6) for _, store := range manager.prunableStores() { names = append(names, store.Name()) } - // The three state stores arrive as one group, from the StateDB that owns them. Order carries no + // The four state stores arrive as one group, from the StateDB that owns them. Order carries no // meaning to the collector: it fixes both cut lines as a minimum over every store before pruning // any of them. - require.Equal(t, []string{"FlatKV", "StateWAL", "EVM SS", "ReceiptDB", "BlockDB"}, names) + require.Equal(t, []string{"FlatKV", "StateWAL", "EVM SS", "HashVault", "ReceiptDB", "BlockDB"}, names) } // TestPrunableStoresOmitsDisabledReceipts pins that a store that was never opened is not offered diff --git a/sei-db/config/giga_config.go b/sei-db/config/giga_config.go index b138f2db33..e2592cdb11 100644 --- a/sei-db/config/giga_config.go +++ b/sei-db/config/giga_config.go @@ -2,10 +2,12 @@ package config import ( "fmt" + "path/filepath" "github.com/sei-protocol/sei-chain/sei-db/common/utils" "github.com/sei-protocol/sei-chain/sei-db/ledger_db/block/littblock" flatkvConfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" ) // GigaStorageConfig composes the store configs a Giga node opens. It is not read from @@ -18,6 +20,7 @@ type GigaStorageConfig struct { BlockDBConfig *littblock.BlockDBConfig // required PruningConfig *StorageGarbageCollectorConfig // required CheckpointConfig CheckpointConfig + HashVaultConfig hashvault.HashVaultConfig } // gigaReceiptBackend is the receipt backend Giga opens (littidx). @@ -46,6 +49,11 @@ func DefaultGigaStorageConfig(homePath string) (*GigaStorageConfig, error) { ssConfig.ExternalPruning = true ssConfig.DisableInternalWAL = true + hashVaultConfig := hashvault.DefaultHashVaultConfig() + // Existing Autobahn nodes already have a hash vault here, holding the ABCI app hashes the GigaRouter + // recorded rather than FlatKV checksums, so the first hash checked after an upgrade does not match. + hashVaultConfig.DataDir = filepath.Join(homePath, "hashvault") + receiptConfig := DefaultReceiptStoreConfig() receiptConfig.Backend = gigaReceiptBackend receiptConfig.DBDirectory = utils.GetReceiptStorePath(homePath, receiptConfig.Backend) @@ -59,6 +67,7 @@ func DefaultGigaStorageConfig(homePath string) (*GigaStorageConfig, error) { BlockDBConfig: blockDBConfig, PruningConfig: DefaultStorageGarbageCollectorConfig(), CheckpointConfig: DefaultCheckpointConfig(), + HashVaultConfig: hashVaultConfig, }, nil } @@ -116,6 +125,10 @@ func (c *GigaStorageConfig) Validate() error { return fmt.Errorf("flatkv data dir is required") } + if err := c.HashVaultConfig.Validate(); err != nil { + return fmt.Errorf("hash vault config is invalid: %w", err) + } + if c.BlockDBConfig == nil { return fmt.Errorf("block db config is required") } diff --git a/sei-db/state_db/giga/state_db.go b/sei-db/state_db/giga/state_db.go index a804051f1e..14143ec725 100644 --- a/sei-db/state_db/giga/state_db.go +++ b/sei-db/state_db/giga/state_db.go @@ -17,6 +17,7 @@ import ( "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv" flatkvconfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/lthash" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" "github.com/sei-protocol/sei-chain/sei-db/state_db/ss/evm" "github.com/sei-protocol/sei-chain/sei-db/state_db/statewal" ) @@ -26,17 +27,11 @@ var logger = seilog.NewLogger("db", "state-db", "giga") var _ gigatypes.StateDB = (*StateDB)(nil) // StateDB writes a committed block to the state WAL, the state commit store (SC) and the EVM state -// store (SS), and serves current-block reads from SC. +// store (SS), serves current-block reads from SC, and records SC's block hashes in the hash vault. // -// It opens all three stores, brings them onto one height, and closes them. SC and SS run without a WAL +// It opens all four stores, brings them onto one height, and closes them. SC and SS run without a WAL // of their own, so every block either of them replays is read from the WAL here. type StateDB struct { - // Where the state commit store and the state WAL live. - flatkvCfg *flatkvconfig.Config - - // Where the EVM state store lives, and whether it is enabled at all. - ssCfg config.StateStoreConfig - // The state WAL a committed block is written to. wal statewal.StateWAL @@ -46,6 +41,9 @@ type StateDB struct { // ss is nil when the EVM state store is disabled. ss *evm.EVMStateStore + // The hash vault SC's block hashes are recorded in. + vault *hashvault.PebbleHashVault + // The checkpoint schedule SC and SS take their snapshot boundaries from. checkpointer *controller.CheckpointScheduler @@ -63,244 +61,185 @@ const commitPhaseTimerName = "giga_state_commit" // gigaMeterName is the OTel meter this package's instruments are created on. const gigaMeterName = "seidb_giga" -// NewStateDB opens SC, SS and the state WAL from their configs and puts SC and SS on one checkpoint -// schedule. -// -// Both stores are put on the WAL's head — replayed up to it, and rewound onto it when a lost WAL tail -// left them above it — so the returned StateDB commits the block after it. NewStateDBWithRollback opens -// them on an earlier height instead. -// -// The returned StateDB owns all three stores and closes them on Close. A failed call closes whatever it -// had already opened. +// NewStateDB opens SC, SS, the state WAL and the hash vault, and puts SC and SS on one checkpoint schedule +// and one height: the WAL's head, or rollbackTo when it is not 0. The returned StateDB commits the block +// after that height, and its hash vault holds the hash of the block SC is on. func NewStateDB( ctx context.Context, flatkvCfg *flatkvconfig.Config, ssCfg config.StateStoreConfig, checkpointCfg config.CheckpointConfig, -) (db *StateDB, retErr error) { - s := &StateDB{ - flatkvCfg: flatkvCfg, - ssCfg: stateStoreConfigFor(ssCfg), - commitPhases: metrics.NewPhaseTimerFactory(otel.Meter(gigaMeterName), commitPhaseTimerName). - RecordLatencies().Build(), - } - s.closed = utils.MustClose(s, "giga state DB") - defer s.closeOnFailure(&retErr) + hashVaultCfg hashvault.HashVaultConfig, + // The height to roll back to, or 0 to load the latest block possible. Data after rollbackTo target + // may be permanently deleted. Returns an error if not possible to roll back to requested block height. + rollbackTo uint64, +) (_ *StateDB, retErr error) { + ssCfg.DisableInternalWAL = true + + if err := recoverStores(flatkvCfg, ssCfg, hashVaultCfg, rollbackTo); err != nil { + return nil, fmt.Errorf("recover the state DB's stores: %w", err) + } + + var err error + var ss *evm.EVMStateStore + var sc *flatkv.CommitStore + var vault *hashvault.PebbleHashVault + var wal statewal.StateWAL + defer func() { + if retErr == nil { + return + } + if err := closeStores(ss, sc, vault, wal); err != nil { + retErr = errors.Join(retErr, fmt.Errorf("close a partially opened state DB: %w", err)) + } + }() - wal, err := s.storedWALRange() - if err != nil { - return nil, err + if ss, err = openSS(ssCfg); err != nil { + return nil, fmt.Errorf("open the state DB: %w", err) } - // Before either store opens, the rewinds it may run needing their files closed. - if err := s.discardStateAboveTheWAL(wal); err != nil { - return nil, err - } - if err := s.openSS(); err != nil { - return nil, err + + if sc, err = openSC(ctx, flatkvCfg); err != nil { + return nil, fmt.Errorf("open the state DB: %w", err) } - if err := s.openSC(ctx); err != nil { - return nil, err + + if vault, err = hashvault.NewPebbleHashVault(ctx, hashVaultCfg); err != nil { + return nil, fmt.Errorf("open the state DB: %w", err) } - if err := s.openWAL(); err != nil { - return nil, err + // The vault must be SC's first listener, and registering it before SC is reachable from outside this + // StateDB is what makes it first. SC hands a hash to its listeners one at a time in registration order + // and stops at the first that refuses it, and the vault returns only once the hash is flushed, so no + // later listener sees a hash the vault has not recorded or has refused. + if _, err := sc.RegisterHashListener(hashVaultListener(vault)); err != nil { + return nil, fmt.Errorf("register the hash vault on the state commit store: %w", err) } - s.startCheckpointSchedule(checkpointCfg) - if err := s.catchUpToWAL(ctx); err != nil { - return nil, err + if wal, err = flatkv.OpenStateWAL(flatkvCfg); err != nil { + return nil, fmt.Errorf("open state WAL: %w", err) } - return s, nil -} - -// stateStoreConfigFor is the config a StateDB opens SS with. It is settled here rather than at each -// open because the rollback path opens the same databases through DiscardStateAbove. The changelog -// is off: this StateDB's own state WAL is what catchUpTo replays into SS. -func stateStoreConfigFor(cfg config.StateStoreConfig) config.StateStoreConfig { - cfg.DisableInternalWAL = true - return cfg -} -// NewStateDBWithRollback rolls SC, SS and the state WAL back to target and then opens them, so the -// returned StateDB commits target+1. It cuts the WAL's tail to target and puts whichever of SC and SS -// sits above target on its newest snapshot at or below it, all while the stores are closed, then opens -// them the ordinary way and checks both landed on target. -// -// target must be positive, and a target the surviving snapshots and the WAL cannot span is refused. A -// refusal leaves the WAL uncut, so no target this one could reach is lost, but one from SS comes back -// with SC already rewound. -func NewStateDBWithRollback( - ctx context.Context, - flatkvCfg *flatkvconfig.Config, - ssCfg config.StateStoreConfig, - checkpointCfg config.CheckpointConfig, - target int64, -) (*StateDB, error) { - if target <= 0 { - // An empty WAL has a head of 0, which rewindTo reads as nothing to rewind, so without this a - // caller asking for a rollback would get a plain open instead. - return nil, fmt.Errorf("rollback target %d is invalid: version 0 means no state, so there is "+ - "nothing to roll back to", target) - } + checkpointer := startCheckpointSchedule(checkpointCfg, sc, ss) - // rewindTo only moves files, so it needs no store open, only where they live. - offline := &StateDB{flatkvCfg: flatkvCfg, ssCfg: stateStoreConfigFor(ssCfg)} - if err := offline.rewindTo(target); err != nil { - return nil, err + if err := catchUpToWAL(ctx, sc, ss, wal); err != nil { + return nil, fmt.Errorf("catch the state DB up to its WAL: %w", err) } - db, err := NewStateDB(ctx, flatkvCfg, ssCfg, checkpointCfg) - if err != nil { - return nil, err - } - if err := db.matchHeight(target); err != nil { - return nil, errors.Join(fmt.Errorf("cannot roll back to %d: %w", target, err), db.Close()) + if rollbackTo > 0 { + if err := matchHeight(sc, ss, wal, int64(rollbackTo)); err != nil { //nolint:gosec // a WAL block number + return nil, fmt.Errorf("cannot roll back to %d: %w", rollbackTo, err) + } } - return db, nil -} -// closeOnFailure closes the stores a failed open had reached, so a caller that gets an error holds no -// store this StateDB left open. It is deferred against the constructor's named error. -func (s *StateDB) closeOnFailure(retErr *error) { - if *retErr == nil { - return + if err := requireAgreementWithoutWAL(sc, ss, vault, wal); err != nil { + return nil, fmt.Errorf("open the state DB on an empty state WAL: %w", err) } - if err := s.Close(); err != nil { - *retErr = errors.Join(*retErr, fmt.Errorf("close a partially opened state DB: %w", err)) + if err := recordLoadedBlockHash(sc, vault); err != nil { + return nil, fmt.Errorf("record the loaded block's hash in the hash vault: %w", err) } -} -// openWAL opens the state WAL this StateDB commits blocks to. -func (s *StateDB) openWAL() error { - wal, err := flatkv.OpenStateWAL(s.flatkvCfg) - if err != nil { - return fmt.Errorf("open state WAL: %w", err) + s := &StateDB{ + wal: wal, + sc: sc, + ss: ss, + vault: vault, + checkpointer: checkpointer, + commitPhases: metrics.NewPhaseTimerFactory(otel.Meter(gigaMeterName), commitPhaseTimerName). + RecordLatencies().Build(), } - s.wal = wal - return nil + s.closed = utils.MustClose(s, "giga state DB") + return s, nil } // openSC opens SC with no WAL of its own, on the version its files hold: the working copy, or the // snapshot a rollback has just repointed it at. It replays nothing, so it comes up at or below the // WAL's head and catchUpTo carries it forward from there. -func (s *StateDB) openSC(ctx context.Context) error { - sc, err := flatkv.NewCommitStore(ctx, s.flatkvCfg, nil) +func openSC(ctx context.Context, flatkvCfg *flatkvconfig.Config) (*flatkv.CommitStore, error) { + sc, err := flatkv.NewCommitStore(ctx, flatkvCfg, nil) if err != nil { - return fmt.Errorf("open state commit store: %w", err) + return nil, fmt.Errorf("open state commit store: %w", err) } - s.sc = sc // Every readonly-* directory under the store is deleted, so this has to run before the process // opens a read-only view of its own: after that, the ones a crashed process left are no longer // the only ones there. - if err := s.sc.CleanupOrphanedReadOnlyDirs(); err != nil { - return fmt.Errorf("clean up orphaned state commit read-only dirs: %w", err) + if err := sc.CleanupOrphanedReadOnlyDirs(); err != nil { + return nil, errors.Join( + fmt.Errorf("clean up orphaned state commit read-only dirs: %w", err), sc.Close()) } - if err := s.sc.LoadWorkingCopy(); err != nil { - return fmt.Errorf("load the state commit store: %w", err) + if err := sc.LoadWorkingCopy(); err != nil { + return nil, errors.Join(fmt.Errorf("load the state commit store: %w", err), sc.Close()) } - return nil + return sc, nil } -// openSS opens the EVM state store and its snapshot manager, leaving it nil when the store is disabled. -func (s *StateDB) openSS() error { - if !s.ssCfg.Enable { - return nil +// openSS opens the EVM state store and its snapshot manager, returning nil when the store is disabled. +func openSS(ssCfg config.StateStoreConfig) (*evm.EVMStateStore, error) { + if !ssCfg.Enable { + return nil, nil } - ss, err := evm.NewEVMStateStore(s.ssCfg.EVMDBDirectory, s.ssCfg) + ss, err := evm.NewEVMStateStore(ssCfg.EVMDBDirectory, ssCfg) if err != nil { - return fmt.Errorf("open EVM state store: %w", err) - } - s.ss = ss - if err := s.ss.StartSnapshots(s.ssSnapshotRoot(), s.ssCfg, nil); err != nil { - return fmt.Errorf("start EVM state store snapshot manager: %w", err) + return nil, fmt.Errorf("open EVM state store: %w", err) } - return nil -} - -// startCheckpointSchedule puts SC and SS on one snapshot cadence. It runs before either store is on a -// height, so the blocks SC replays offer themselves to the schedule as live commits do. -func (s *StateDB) startCheckpointSchedule(cfg config.CheckpointConfig) { - s.checkpointer = controller.NewCheckpointScheduler(cfg) - s.sc.SetCheckpointScheduler(s.checkpointer) - if s.ss != nil { - s.ss.SetCheckpointScheduler(s.checkpointer) + snapshotRoot := utils.GetStateStoreSnapshotsSiblingPath(ssCfg.EVMDBDirectory) + if err := ss.StartSnapshots(snapshotRoot, ssCfg, nil); err != nil { + return nil, errors.Join(fmt.Errorf("start EVM state store snapshot manager: %w", err), ss.Close()) } + return ss, nil } -// ssSnapshotRoot returns the directory SS keeps its snapshots in. -func (s *StateDB) ssSnapshotRoot() string { - return utils.GetStateStoreSnapshotsSiblingPath(s.ssCfg.EVMDBDirectory) -} - -// storedWALRange is the block range a state WAL holds on disk: the lowest and highest blocks in it, -// both 0 when it holds none. -type storedWALRange struct { - first, last int64 -} - -// walConfig returns the config that locates the state WAL on disk. -func (s *StateDB) walConfig() *statewal.Config { - return flatkv.StateWALConfig(s.flatkvCfg.DataDir) -} - -// storedWALRange reads the state WAL's block range from its directory. It takes that directory's -// exclusive lock, so it is only for the window before the WAL opens; GetStoredRange on the open handle -// answers the same question afterwards. -func (s *StateDB) storedWALRange() (storedWALRange, error) { - stored, first, last, err := statewal.GetRange(s.walConfig()) - if err != nil { - return storedWALRange{}, fmt.Errorf("read state WAL range: %w", err) - } - if !stored { - return storedWALRange{}, nil - } - //nolint:gosec // a block number never approaches the int64 ceiling - return storedWALRange{first: int64(first), last: int64(last)}, nil -} - -// openWALRange reads the block range from the open WAL handle, which storedWALRange's directory lock -// rules out reading once the WAL is open. -func (s *StateDB) openWALRange() (storedWALRange, error) { - stored, first, last, err := s.wal.GetStoredRange() - if err != nil { - return storedWALRange{}, fmt.Errorf("read state WAL range: %w", err) - } - if !stored { - return storedWALRange{}, nil - } - //nolint:gosec // a block number never approaches the int64 ceiling - return storedWALRange{first: int64(first), last: int64(last)}, nil +// startCheckpointSchedule puts SC and SS on one snapshot cadence, and returns it. ss is nil when SS is +// disabled. It runs before either store is on a height, so the blocks SC replays offer themselves to the +// schedule as live commits do. +func startCheckpointSchedule( + cfg config.CheckpointConfig, + sc *flatkv.CommitStore, + ss *evm.EVMStateStore, +) *controller.CheckpointScheduler { + checkpointer := controller.NewCheckpointScheduler(cfg) + sc.SetCheckpointScheduler(checkpointer) + if ss != nil { + ss.SetCheckpointScheduler(checkpointer) + } + return checkpointer } -// truncateWAL drops every WAL block above target so the next commit is target+1. A live WAL prunes only -// from its start, so this cuts the tail through the directory, which requires that no WAL be open on it. -func (s *StateDB) truncateWAL(target int64) error { - //nolint:gosec // target > 0 here, checked by NewStateDBWithRollback - if err := statewal.PruneAfter(s.walConfig(), uint64(target)); err != nil { - return fmt.Errorf("truncate state WAL to %d: %w", target, err) - } - return nil +// Close closes SC, SS, the hash vault and the state WAL, reporting every failure rather than stopping at +// the first. +func (s *StateDB) Close() error { + s.closed.Close(s) + return closeStores(s.ss, s.sc, s.vault, s.wal) } -// Close closes SC, SS and the state WAL, reporting every failure rather than stopping at the first. -// The WAL closes last, since SC replays through it. +// closeStores closes whichever of the stores are not nil, reporting every failure rather than stopping at +// the first. The hash vault closes after SC, which hands it the hashes of the blocks it drains, and the +// WAL closes last, since SC replays through it. // -// How long each of the three took is logged, since each drains its own write queue and waits on the +// How long each store took is logged, since each drains its own write queue and waits on the // compactions behind it, and those dominate the time a shutdown takes. -func (s *StateDB) Close() error { - s.closed.Close(s) +func closeStores( + ss *evm.EVMStateStore, + sc *flatkv.CommitStore, + vault *hashvault.PebbleHashVault, + wal statewal.StateWAL, +) error { var errs error var timer utils.CloseTimer - if s.ss != nil { - if err := timer.Close("ss", s.ss.Close); err != nil { + if ss != nil { + if err := timer.Close("ss", ss.Close); err != nil { errs = errors.Join(errs, fmt.Errorf("close EVM state store: %w", err)) } } - if s.sc != nil { - if err := timer.Close("sc", s.sc.Close); err != nil { + if sc != nil { + if err := timer.Close("sc", sc.Close); err != nil { errs = errors.Join(errs, fmt.Errorf("close state commit store: %w", err)) } } - if s.wal != nil { - if err := timer.Close("wal", s.wal.Close); err != nil { + if vault != nil { + closeVault := func() error { return vault.Close(context.Background()) } + if err := timer.Close("hashvault", closeVault); err != nil { + errs = errors.Join(errs, fmt.Errorf("close hash vault: %w", err)) + } + } + if wal != nil { + if err := timer.Close("wal", wal.Close); err != nil { errs = errors.Join(errs, fmt.Errorf("close state WAL: %w", err)) } } @@ -323,7 +262,7 @@ func (s *StateDB) CheckpointScheduler() *controller.CheckpointScheduler { return // PrunableStores returns the opened stores that can join a prune cycle. func (s *StateDB) PrunableStores() []controller.PrunableStore { - stores := make([]controller.PrunableStore, 0, 3) + stores := make([]controller.PrunableStore, 0, 4) if s.sc != nil { stores = append(stores, s.sc) } @@ -333,6 +272,9 @@ func (s *StateDB) PrunableStores() []controller.PrunableStore { if s.ss != nil { stores = append(stores, s.ss) } + if s.vault != nil { + stores = append(stores, s.vault) + } return stores } @@ -394,3 +336,17 @@ func (s *StateDB) RegisterHashListener(listener gigatypes.HashListener) (lthash. } return mostRecentHash, nil } + +// GetBlockHeight returns the version SC is on. +func (s *StateDB) GetBlockHeight() uint64 { + return uint64(s.sc.Version()) //nolint:gosec // a committed version is never negative +} + +// GetBlockHash returns the hash the hash vault holds for blockNumber. +func (s *StateDB) GetBlockHash(blockNumber uint64) ([32]byte, gigatypes.BlockHashStatus, error) { + hash, status, err := s.vault.Get(blockNumber) + if err != nil { + return hash, status, fmt.Errorf("get the hash of block %d: %w", blockNumber, err) + } + return hash, status, nil +} diff --git a/sei-db/state_db/giga/state_db_hash_listener_test.go b/sei-db/state_db/giga/state_db_hash_listener_test.go index bd999fdb83..7201698382 100644 --- a/sei-db/state_db/giga/state_db_hash_listener_test.go +++ b/sei-db/state_db/giga/state_db_hash_listener_test.go @@ -16,20 +16,20 @@ import ( func TestRegisterHashListenerReachesTheLiveStateDB(t *testing.T) { stateDB, _, liveStateDB := newTestStateDB(t) - var seen []int64 + var seen []uint64 mostRecent, err := stateDB.RegisterHashListener( - func(_ context.Context, blockNumber int64, _ *lthash.BlockHash) error { + func(_ context.Context, blockNumber uint64, _ *lthash.BlockHash) error { seen = append(seen, blockNumber) return nil }) require.NoError(t, err) - require.Equal(t, int64(0), mostRecent.BlockNumber, "a fresh store has hashed nothing") + require.Equal(t, uint64(0), mostRecent.BlockNumber, "a fresh store has hashed nothing") require.NoError(t, stateDB.CommitStateChanges(1, changeset("key", "one"))) require.NoError(t, stateDB.CommitStateChanges(2, changeset("key", "two"))) require.NoError(t, liveStateDB.FlushHashes()) - require.Equal(t, []int64{1, 2}, seen) + require.Equal(t, []uint64{1, 2}, seen) } // The hash logger is wired in as a listener, and this is that wiring end to end: blocks committed diff --git a/sei-db/state_db/giga/state_db_hash_vault.go b/sei-db/state_db/giga/state_db_hash_vault.go new file mode 100644 index 0000000000..2af783ecd0 --- /dev/null +++ b/sei-db/state_db/giga/state_db_hash_vault.go @@ -0,0 +1,53 @@ +package giga + +import ( + "context" + "fmt" + + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/lthash" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" +) + +// hashVaultListener returns the listener that records each block's hash in vault. +func hashVaultListener(vault *hashvault.PebbleHashVault) gigatypes.HashListener { + return func(ctx context.Context, blockNumber uint64, hash *lthash.BlockHash) error { + if hash.Global == nil { + return fmt.Errorf("record the hash of block %d: it carries no global hash", blockNumber) + } + checksum := hash.Global.Checksum() + if err := vault.CommitToHash(ctx, blockNumber, checksum[:]); err != nil { + return fmt.Errorf("record the hash of block %d in the hash vault: %w", blockNumber, err) + } + return nil + } +} + +// recordLoadedBlockHash records, or checks, the hash of the block SC opened on, so that the vault holds +// it once the open returns. SC dispatches that hash while it loads, before the vault is registered, and +// no replay re-dispatches it when SC was already on the WAL's head. +func recordLoadedBlockHash(sc *flatkv.CommitStore, vault *hashvault.PebbleHashVault) error { + if err := sc.FlushHashes(); err != nil { + return fmt.Errorf("wait for the replayed blocks to reach the hash vault: %w", err) + } + loaded := sc.Version() + if loaded == 0 { + // A store that has committed nothing has no block to record, and the first one it commits may be + // any height the chain starts at. + return nil + } + current, err := sc.RegisterHashListener(nil) + if err != nil { + return fmt.Errorf("read the hash of the loaded block %d: %w", loaded, err) + } + //nolint:gosec // a committed version is never negative + if current.BlockNumber != uint64(loaded) { + return fmt.Errorf("the state commit store is on block %d but last produced the hash of block %d", + loaded, current.BlockNumber) + } + if err := hashVaultListener(vault)(context.Background(), current.BlockNumber, ¤t); err != nil { + return fmt.Errorf("record the loaded block's hash: %w", err) + } + return nil +} diff --git a/sei-db/state_db/giga/state_db_hash_vault_test.go b/sei-db/state_db/giga/state_db_hash_vault_test.go new file mode 100644 index 0000000000..f3719cdbb2 --- /dev/null +++ b/sei-db/state_db/giga/state_db_hash_vault_test.go @@ -0,0 +1,373 @@ +package giga + +import ( + "context" + "fmt" + "os" + "path/filepath" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/sei-protocol/sei-chain/sei-db/config" + "github.com/sei-protocol/sei-chain/sei-db/controller" + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv" + flatkvconfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/lthash" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" + "github.com/sei-protocol/sei-chain/sei-db/state_db/statewal" +) + +// vaultTestStores is the set of configs a StateDB under test opens, kept so the test can reopen it. +type vaultTestStores struct { + // Where SC and the state WAL live. + flatkvCfg *flatkvconfig.Config + + // SS stays disabled: it produces no hashes, so nothing here depends on it. + ssCfg config.StateStoreConfig + + // The schedule SC snapshots on when a test does not install one of its own. + checkpointCfg config.CheckpointConfig + + // Where the hash vault lives, and what it does on a mismatch. + hashVaultCfg hashvault.HashVaultConfig +} + +// newVaultTestStores returns configs for a fresh StateDB whose vault halts on a mismatch. +func newVaultTestStores(t *testing.T) *vaultTestStores { + t.Helper() + flatkvCfg := flatkvconfig.DefaultTestConfig(t) + // The snapshots the rewinds land on are kept, since no collector runs here to prune them. + flatkvCfg.ExternalPruning = true + hashVaultCfg := hashvault.DefaultHashVaultConfig() + hashVaultCfg.DataDir = filepath.Join(t.TempDir(), "hashvault") + hashVaultCfg.Fsync = false + hashVaultCfg.HaltOnMismatch = true + return &vaultTestStores{ + flatkvCfg: flatkvCfg, + ssCfg: config.StateStoreConfig{Enable: false}, + checkpointCfg: config.CheckpointConfig{TimeInterval: time.Hour}, + hashVaultCfg: hashVaultCfg, + } +} + +// open opens the StateDB, failing the test if it cannot. +func (c *vaultTestStores) open(t *testing.T) *StateDB { + t.Helper() + db, err := c.openErr() + require.NoError(t, err) + return db +} + +// openErr opens the StateDB. +func (c *vaultTestStores) openErr() (*StateDB, error) { + return NewStateDB(context.Background(), c.flatkvCfg, c.ssCfg, c.checkpointCfg, c.hashVaultCfg, 0) +} + +// commitBlocks commits blocks first through last, each writing its own value, and snapshots SC at every +// one of them so any of them can be rewound to. +func commitBlocks(t *testing.T, db *StateDB, first int64, last int64) { + t.Helper() + require.NoError(t, db.SC().FlushSnapshots()) + db.SC().SetCheckpointScheduler(controller.NewCheckpointScheduler(config.CheckpointConfig{BlockInterval: 1})) + for block := first; block <= last; block++ { + require.NoError(t, db.CommitStateChanges(block, changeset("key", fmt.Sprintf("value-%d", block)))) + } + require.NoError(t, db.SC().FlushSnapshots()) + require.NoError(t, db.SC().FlushHashes()) +} + +// recordedHashes returns the vault's hashes for blocks first through last, failing the test if any is +// not recorded. +func recordedHashes(t *testing.T, db *StateDB, first uint64, last uint64) map[uint64][32]byte { + t.Helper() + hashes := make(map[uint64][32]byte) + for block := first; block <= last; block++ { + hash, status, err := db.GetBlockHash(block) + require.NoError(t, err) + require.Equal(t, gigatypes.BlockHashStatusFound, status, "block %d", block) + hashes[block] = hash + } + return hashes +} + +// requireStatus asserts the status GetBlockHash reports for blockNumber. +func requireStatus(t *testing.T, db *StateDB, blockNumber uint64, want gigatypes.BlockHashStatus) { + t.Helper() + _, status, err := db.GetBlockHash(blockNumber) + require.NoError(t, err) + require.Equal(t, want, status, "block %d", blockNumber) +} + +// Every block committed has its hash recorded, and the hash recorded is the one SC hands its listeners. +func TestStateDBRecordsTheHashOfEveryCommittedBlock(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + defer func() { require.NoError(t, db.Close()) }() + + var mu sync.Mutex + dispatched := make(map[uint64][32]byte) + _, err := db.RegisterHashListener(func(_ context.Context, blockNumber uint64, hash *lthash.BlockHash) error { + mu.Lock() + defer mu.Unlock() + dispatched[blockNumber] = hash.Global.Checksum() + return nil + }) + require.NoError(t, err) + + commitBlocks(t, db, 1, 5) + + require.Equal(t, uint64(5), db.GetBlockHeight()) + recorded := recordedHashes(t, db, 1, 5) + mu.Lock() + require.Equal(t, dispatched, recorded) + mu.Unlock() + requireStatus(t, db, 6, gigatypes.BlockHashStatusNotReady) +} + +// A reopened StateDB holds the hash of the block it opened on, whether or not anything was replayed. +func TestAReopenedStateDBHoldsTheLoadedBlocksHash(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 5) + before := recordedHashes(t, db, 1, 5) + require.NoError(t, db.Close()) + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, uint64(5), reopened.GetBlockHeight()) + require.Equal(t, before, recordedHashes(t, reopened, 1, 5)) +} + +// A vault that lost its tail is behind SC, and the blocks it lost can only be hashed again by replaying +// them, so the open rewinds SC to the vault's newest block and replays from there. +func TestAVaultBehindSCIsRefilledByReplay(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 6) + before := recordedHashes(t, db, 1, 6) + require.NoError(t, db.Close()) + + require.NoError(t, hashvault.HardRollbackPebbleHashVault(context.Background(), stores.hashVaultCfg, 3)) + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, uint64(6), reopened.GetBlockHeight()) + require.Equal(t, before, recordedHashes(t, reopened, 1, 6)) +} + +// An empty vault is refilled by rewinding SC the configured number of blocks and replaying them, so it +// holds the hashes of the newest blocks rather than just the loaded one. +func TestAnEmptyVaultIsRefilledByRewinding(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 8) + before := recordedHashes(t, db, 1, 8) + require.NoError(t, db.Close()) + + require.NoError(t, os.RemoveAll(stores.hashVaultCfg.DataDir)) + stores.hashVaultCfg.EmptyVaultRollbackBlocks = 3 + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, uint64(8), reopened.GetBlockHeight()) + require.Equal(t, map[uint64][32]byte{6: before[6], 7: before[7], 8: before[8]}, + recordedHashes(t, reopened, 6, 8)) + requireStatus(t, reopened, 5, gigatypes.BlockHashStatusTooOld) +} + +// A rewind deeper than SC's snapshots and the WAL can replay is shortened to the deepest one they can. +func TestAnEmptyVaultRewindIsShortenedToWhatIsReachable(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 8) + before := recordedHashes(t, db, 1, 8) + require.NoError(t, db.Close()) + + require.NoError(t, os.RemoveAll(stores.hashVaultCfg.DataDir)) + stores.hashVaultCfg.EmptyVaultRollbackBlocks = 1000 + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, uint64(8), reopened.GetBlockHeight()) + recorded := recordedHashes(t, reopened, 2, 8) + for block, hash := range recorded { + require.Equal(t, before[block], hash, "block %d", block) + } +} + +// With no rewind configured, an empty vault records only the loaded block's hash. +func TestAnEmptyVaultWithNoRewindRecordsOnlyTheLoadedBlock(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 4) + before := recordedHashes(t, db, 1, 4) + require.NoError(t, db.Close()) + + require.NoError(t, os.RemoveAll(stores.hashVaultCfg.DataDir)) + stores.hashVaultCfg.EmptyVaultRollbackBlocks = 0 + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, map[uint64][32]byte{4: before[4]}, recordedHashes(t, reopened, 4, 4)) + requireStatus(t, reopened, 3, gigatypes.BlockHashStatusTooOld) +} + +// tamperLoadedBlock replaces the vault's hash for blockNumber with one SC will never produce. +func tamperLoadedBlock(t *testing.T, stores *vaultTestStores, blockNumber uint64) { + t.Helper() + cfg := stores.hashVaultCfg + cfg.HaltOnMismatch = false + vault, err := hashvault.NewUnsafePebbleHashVault(context.Background(), cfg) + require.NoError(t, err) + tampered := [32]byte{0xEE} + require.NoError(t, vault.CommitToHash(context.Background(), blockNumber, tampered[:])) + require.NoError(t, vault.Close(context.Background())) +} + +// The loaded block's hash is checked against the vault even when nothing replays, so a vault that +// disagrees with the state on disk stops the open when halting is selected. +func TestAMismatchAtTheLoadedBlockFailsTheOpenWhenHalting(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 3) + require.NoError(t, db.Close()) + + tamperLoadedBlock(t, stores, 3) + + _, err := stores.openErr() + require.ErrorContains(t, err, "mismatch") +} + +// With halting off, the same disagreement is logged and the state's own hash replaces the vault's. +func TestAMismatchAtTheLoadedBlockIsReplacedWhenNotHalting(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 3) + before := recordedHashes(t, db, 1, 3) + require.NoError(t, db.Close()) + + tamperLoadedBlock(t, stores, 3) + stores.hashVaultCfg.HaltOnMismatch = false + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, before, recordedHashes(t, reopened, 1, 3)) +} + +// A rollback leaves the vault alone: the blocks above the target are re-executed, and it is exactly their +// recorded hashes the re-execution is held to. +func TestARollbackKeepsTheHashesAboveItsTarget(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 5) + before := recordedHashes(t, db, 1, 5) + require.NoError(t, db.Close()) + + rolledBack, err := NewStateDB(context.Background(), + stores.flatkvCfg, stores.ssCfg, stores.checkpointCfg, stores.hashVaultCfg, 3) + require.NoError(t, err) + defer func() { require.NoError(t, rolledBack.Close()) }() + + require.Equal(t, uint64(3), rolledBack.GetBlockHeight()) + require.Equal(t, before, recordedHashes(t, rolledBack, 1, 5)) + + commitBlocks(t, rolledBack, 4, 5) + require.Equal(t, before, recordedHashes(t, rolledBack, 1, 5), "re-executing the same blocks matches") +} + +// Re-executing a block into a different state is the equivocation the vault exists to stop. The vault is +// SC's first listener, so a hash it refuses reaches no listener registered after it. +func TestADifferentReexecutionIsRefusedBeforeAnyOtherListenerSeesIt(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 5) + require.NoError(t, db.Close()) + + rolledBack, err := NewStateDB(context.Background(), + stores.flatkvCfg, stores.ssCfg, stores.checkpointCfg, stores.hashVaultCfg, 3) + require.NoError(t, err) + defer func() { _ = rolledBack.Close() }() + + var mu sync.Mutex + var seen []uint64 + record := func(_ context.Context, blockNumber uint64, _ *lthash.BlockHash) error { + mu.Lock() + defer mu.Unlock() + seen = append(seen, blockNumber) + return nil + } + _, err = rolledBack.RegisterHashListener(record) + require.NoError(t, err) + + require.NoError(t, rolledBack.CommitStateChanges(4, changeset("key", "a different value"))) + require.ErrorContains(t, rolledBack.SC().FlushHashes(), "mismatch") + mu.Lock() + require.Empty(t, seen, "a hash the vault refused must reach no later listener") + mu.Unlock() +} + +// The vault joins the prune cycle, which is how the storage garbage collector's permission reaches it. +func TestTheVaultJoinsThePruneCycle(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + defer func() { require.NoError(t, db.Close()) }() + + var names []string + for _, store := range db.PrunableStores() { + names = append(names, store.Name()) + } + require.Contains(t, names, "HashVault") +} + +// emptyWAL drops every block from the closed StateDB's WAL, leaving the stores where they are. +func emptyWAL(t *testing.T, stores *vaultTestStores) { + t.Helper() + require.NoError(t, statewal.PruneAfter(flatkv.StateWALConfig(stores.flatkvCfg.DataDir), 0)) +} + +// With no WAL blocks nothing can replay, so an open over stores that already agree is the only one allowed. +func TestAnEmptyWALOpensWhenTheStoresAgree(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 3) + require.NoError(t, db.Close()) + emptyWAL(t, stores) + + reopened := stores.open(t) + defer func() { require.NoError(t, reopened.Close()) }() + require.Equal(t, uint64(3), reopened.GetBlockHeight()) +} + +// An empty WAL cannot bring SS up to SC, so a difference between them is refused, even for an SS that holds +// nothing yet. +func TestAnEmptyWALRefusesStoresOnDifferentHeights(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 3) + require.NoError(t, db.Close()) + emptyWAL(t, stores) + + stores.ssCfg = config.DefaultStateStoreConfig() + stores.ssCfg.Enable = true + stores.ssCfg.EVMDBDirectory = filepath.Join(t.TempDir(), "ss") + + _, err := stores.openErr() + require.ErrorContains(t, err, "the EVM state store is on block 0") +} + +// An empty WAL cannot replay the loaded block, so a vault without its hash is refused. +func TestAnEmptyWALRefusesAVaultWithoutTheLoadedBlock(t *testing.T) { + stores := newVaultTestStores(t) + db := stores.open(t) + commitBlocks(t, db, 1, 3) + require.NoError(t, db.Close()) + emptyWAL(t, stores) + require.NoError(t, os.RemoveAll(stores.hashVaultCfg.DataDir)) + + _, err := stores.openErr() + require.ErrorContains(t, err, "the hash vault holds no hash for block 3") +} diff --git a/sei-db/state_db/giga/state_db_recovery.go b/sei-db/state_db/giga/state_db_recovery.go new file mode 100644 index 0000000000..3a11e29e8c --- /dev/null +++ b/sei-db/state_db/giga/state_db_recovery.go @@ -0,0 +1,263 @@ +package giga + +import ( + "fmt" + + "github.com/sei-protocol/sei-chain/sei-db/common/utils" + "github.com/sei-protocol/sei-chain/sei-db/config" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv" + flatkvconfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" + "github.com/sei-protocol/sei-chain/sei-db/state_db/ss/evm" + "github.com/sei-protocol/sei-chain/sei-db/state_db/statewal" +) + +// What the stores hold on disk before any of them opens. +type recoverySurvey struct { + walStored bool + + // Only meaningful when walStored is true. + walFirst uint64 + + // Only meaningful when walStored is true. + walLast uint64 + + // Only meaningful when vaultRecorded is true. + vaultHead uint64 + + vaultRecorded bool + + // The live state DB's snapshot versions, lowest first. + scSnapshots []uint64 +} + +// Where each store is put before the stores open. +type recoveryPlan struct { + // The height the open replays every store up to, or 0 when there is nothing to move or replay. + head uint64 + + // The live state DB is put on its newest snapshot at or below this when it holds state above it. + scTarget uint64 + + // The historical state DB is put on its newest snapshot at or below this when it holds state above it. + ssTarget uint64 + + // Whether the snapshots and WAL blocks above head are dropped. + rollback bool + + // Whether an empty hash vault's refill was cut short by how far back the snapshots and WAL reach. + emptyVaultRefillShortened bool + + // Whether an empty hash vault cannot be refilled, and records only the loaded block's hash. + emptyVaultRefillUnreachable bool +} + +// Puts each store where it belongs, as decided from what the stores hold on disk. Every store must be +// closed. +func recoverStores( + flatkvCfg *flatkvconfig.Config, + ssCfg config.StateStoreConfig, + hashVaultCfg hashvault.HashVaultConfig, + rollbackTo uint64, +) error { + survey, err := surveyStores(flatkvCfg, hashVaultCfg) + if err != nil { + return fmt.Errorf("survey the stores: %w", err) + } + plan, err := planRecovery(survey, rollbackTo, hashVaultCfg.EmptyVaultRollbackBlocks) + if err != nil { + return fmt.Errorf("plan the recovery: %w", err) + } + logRecoveryPlan(plan, survey, hashVaultCfg.EmptyVaultRollbackBlocks) + if err := applyRecoveryPlan(flatkvCfg, ssCfg, survey.walFirst, plan); err != nil { + return fmt.Errorf("apply the recovery plan: %w", err) + } + return nil +} + +// Reads what the state WAL, the hash vault and the live state DB's snapshots hold. Every store must be +// closed. +func surveyStores(flatkvCfg *flatkvconfig.Config, hashVaultCfg hashvault.HashVaultConfig) (recoverySurvey, error) { + // This takes the WAL directory's exclusive lock, so it only works before the WAL opens. + walStored, walFirst, walLast, err := statewal.GetRange(flatkv.StateWALConfig(flatkvCfg.DataDir)) + if err != nil { + return recoverySurvey{}, fmt.Errorf("survey the state WAL: %w", err) + } + versions, err := flatkv.SnapshotVersions(flatkvCfg.DataDir) + if err != nil { + return recoverySurvey{}, fmt.Errorf("survey the state commit store's snapshots: %w", err) + } + scSnapshots := make([]uint64, len(versions)) + for i, version := range versions { + scSnapshots[i] = uint64(version) //nolint:gosec // a snapshot version is never negative + } + _, vaultHead, vaultRecorded, err := hashvault.StoredRange(hashVaultCfg) + if err != nil { + return recoverySurvey{}, fmt.Errorf("survey the hash vault: %w", err) + } + return recoverySurvey{ + walStored: walStored, + walFirst: walFirst, + walLast: walLast, + vaultHead: vaultHead, + vaultRecorded: vaultRecorded, + scSnapshots: scSnapshots, + }, nil +} + +// Decides where each store is put: on the WAL's head, or on rollbackTo when it is not 0, which must be at +// or below the head. The live state DB goes lower when the hash vault is missing hashes only a replay can +// produce. +func planRecovery( + survey recoverySurvey, + rollbackTo uint64, + emptyVaultRollbackBlocks uint64, +) (recoveryPlan, error) { + if !survey.walStored { + // An empty WAL says nothing about where state belongs, so it moves nothing. + if rollbackTo > 0 { + return recoveryPlan{}, fmt.Errorf("cannot roll back to %d: the state WAL is empty, so no "+ + "replay reaches the target", rollbackTo) + } + return recoveryPlan{}, nil + } + // A crash can lose the WAL's unflushed tail while the state above it survives. Those blocks are + // re-executed, so that state is discarded. + plan := recoveryPlan{head: survey.walLast} + if rollbackTo > 0 { + if survey.walLast < rollbackTo { + return recoveryPlan{}, fmt.Errorf("cannot roll back to %d: the state WAL ends at %d, so no "+ + "replay reaches the target", rollbackTo, survey.walLast) + } + plan.head = rollbackTo + plan.rollback = true + } + plan.scTarget = plan.head + plan.ssTarget = plan.head + + if plan.head == 0 || plan.head < survey.walFirst { + // No WAL block is left to replay once the plan is applied, so a lower SC would stay lower. + return plan, nil + } + planHashVaultRefill(&plan, survey, emptyVaultRollbackBlocks) + return plan, nil +} + +// Lowers the live state DB's target so the replay produces the hashes the vault is missing. plan.head must +// be a height the WAL can replay up to. +func planHashVaultRefill(plan *recoveryPlan, survey recoverySurvey, emptyVaultRollbackBlocks uint64) { + if survey.vaultRecorded { + if survey.vaultHead < plan.head { + // The rewind lands on the snapshot at or below the target, so a target of 1 still reaches the + // state a vault holding only block 0 needs. + plan.scTarget = max(survey.vaultHead, 1) + } + return + } + + if emptyVaultRollbackBlocks == 0 { + return + } + earliest, reachable := earliestSCRewindTarget(survey.scSnapshots, survey.walFirst) + if !reachable { + plan.emptyVaultRefillUnreachable = true + return + } + target := earliest + if emptyVaultRollbackBlocks < plan.head && plan.head-emptyVaultRollbackBlocks >= earliest { + target = plan.head - emptyVaultRollbackBlocks + } else { + plan.emptyVaultRefillShortened = true + } + plan.scTarget = min(target, plan.head) +} + +// Returns the lowest height the live state DB can be rewound to and replayed from a WAL starting at +// walFirst, and false when there is none. +func earliestSCRewindTarget(scSnapshots []uint64, walFirst uint64) (uint64, bool) { + for _, version := range scSnapshots { + if version+1 >= walFirst { + // A rewind target of 0 is refused, and 1 lands on a snapshot at 0 just the same. + return max(version, 1), true + } + } + return 0, false +} + +// Reports a plan that puts the live state DB below the head, and why. +func logRecoveryPlan(plan recoveryPlan, survey recoverySurvey, emptyVaultRollbackBlocks uint64) { + switch { + case plan.emptyVaultRefillUnreachable: + logger.Warn("The hash vault is empty and no snapshot of the state commit store can be replayed "+ + "from the state WAL, so it records only the loaded block's hash", + "walFirst", survey.walFirst, "walLast", survey.walLast) + case plan.emptyVaultRefillShortened: + logger.Warn("The hash vault is empty and the requested rewind reaches further back than the state "+ + "commit store's snapshots and the state WAL can replay, so it is shortened", + "requestedBlocks", emptyVaultRollbackBlocks, "head", plan.head, "target", plan.scTarget) + } + if plan.scTarget < plan.head { + logger.Info("Rewinding the state commit store so the replay refills the hash vault", + "target", plan.scTarget, "head", plan.head, + "vaultRecorded", survey.vaultRecorded, "vaultHead", survey.vaultHead) + } +} + +// Puts the live and historical state DBs on the plan's targets and, for a rollback, drops the snapshots and +// WAL blocks above the head. The stores must be closed, and walFirst is the WAL's lowest block when the +// plan was made. A refused target leaves the WAL uncut. +func applyRecoveryPlan( + flatkvCfg *flatkvconfig.Config, + ssCfg config.StateStoreConfig, + walFirst uint64, + plan recoveryPlan, +) error { + if plan.head == 0 { + return nil + } + //nolint:gosec // WAL block numbers never approach the int64 ceiling + if _, err := flatkv.DiscardStateAbove(flatkvCfg.DataDir, int64(plan.scTarget), int64(walFirst)); err != nil { + return plan.refusal(fmt.Errorf("the state commit store cannot reach %d: %w", plan.scTarget, err)) + } + if err := discardSSAbove(ssCfg, walFirst, plan.ssTarget); err != nil { + return plan.refusal(err) + } + if !plan.rollback { + return nil + } + if err := dropSnapshotsAbove(flatkvCfg, ssCfg, plan.head); err != nil { + return fmt.Errorf("cannot roll back to %d: %w", plan.head, err) + } + // Last, so that an interruption leaves the WAL still above the target and a restart comes back here. A + // live WAL prunes only from its start, so the tail is cut through the directory, with no WAL open on it. + if err := statewal.PruneAfter(flatkv.StateWALConfig(flatkvCfg.DataDir), plan.head); err != nil { + return fmt.Errorf("cannot roll back to %d: truncate state WAL: %w", plan.head, err) + } + return nil +} + +// Wraps a store's refusal of its target in the rollback it refused, or in the WAL head when no rollback +// was asked for. +func (p recoveryPlan) refusal(err error) error { + if p.scTarget < p.head { + err = fmt.Errorf("rewinding the state commit store to %d to refill the hash vault: %w", p.scTarget, err) + } + if p.rollback { + return fmt.Errorf("cannot roll back to %d: %w", p.head, err) + } + return fmt.Errorf("cannot open on the state WAL's head %d: %w", p.head, err) +} + +// Puts an enabled historical state DB on its newest snapshot at or below target when it holds state above +// it. The store must be closed, and a target the WAL cannot replay it back up to is refused. +func discardSSAbove(ssCfg config.StateStoreConfig, walFirst uint64, target uint64) error { + if !ssCfg.Enable { + return nil + } + snapshotRoot := utils.GetStateStoreSnapshotsSiblingPath(ssCfg.EVMDBDirectory) + //nolint:gosec // WAL block numbers never approach the int64 ceiling + if _, err := evm.DiscardStateAbove(ssCfg, snapshotRoot, int64(target), int64(walFirst)); err != nil { + return fmt.Errorf("the EVM state store cannot reach %d: %w", target, err) + } + return nil +} diff --git a/sei-db/state_db/giga/state_db_recovery_test.go b/sei-db/state_db/giga/state_db_recovery_test.go new file mode 100644 index 0000000000..dab4f0f205 --- /dev/null +++ b/sei-db/state_db/giga/state_db_recovery_test.go @@ -0,0 +1,172 @@ +package giga + +import ( + "testing" + + "github.com/stretchr/testify/require" +) + +// Where each store is put is decided once, from what the stores hold, so every rule the open follows is +// visible here without opening anything. +func TestPlanRecovery(t *testing.T) { + snapshots := []uint64{0, 4, 8} + for _, tc := range []struct { + name string + survey recoverySurvey + rollbackTo uint64 + depth uint64 + want recoveryPlan + }{ + { + name: "a plain open puts every store on the WAL's head", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 10, + scSnapshots: snapshots, + }, + want: recoveryPlan{head: 10, scTarget: 10, ssTarget: 10}, + }, + { + name: "an empty WAL moves nothing", + survey: recoverySurvey{ + vaultRecorded: true, + vaultHead: 10, + scSnapshots: snapshots, + }, + want: recoveryPlan{}, + }, + { + name: "a rollback puts every store on its target", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 10, + scSnapshots: snapshots, + }, + rollbackTo: 6, + want: recoveryPlan{head: 6, scTarget: 6, ssTarget: 6, rollback: true}, + }, + { + name: "a vault ahead of the head keeps its hashes and moves nothing", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 12, + scSnapshots: snapshots, + }, + want: recoveryPlan{head: 10, scTarget: 10, ssTarget: 10}, + }, + { + name: "a vault behind the head puts SC alone on the vault's newest block", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 7, + scSnapshots: snapshots, + }, + want: recoveryPlan{head: 10, scTarget: 7, ssTarget: 10}, + }, + { + name: "a vault behind a rollback target is caught up the same way", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 3, + scSnapshots: snapshots, + }, + rollbackTo: 6, + want: recoveryPlan{head: 6, scTarget: 3, ssTarget: 6, rollback: true}, + }, + { + name: "a vault holding only block 0 is caught up from 1", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + vaultRecorded: true, + vaultHead: 0, + scSnapshots: snapshots, + }, + want: recoveryPlan{head: 10, scTarget: 1, ssTarget: 10}, + }, + { + name: "an empty vault puts SC the configured depth below the head", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + scSnapshots: snapshots, + }, + depth: 3, + want: recoveryPlan{head: 10, scTarget: 7, ssTarget: 10}, + }, + { + name: "an empty vault with no depth configured moves nothing", + survey: recoverySurvey{ + walStored: true, + walFirst: 1, + walLast: 10, + scSnapshots: snapshots, + }, + want: recoveryPlan{head: 10, scTarget: 10, ssTarget: 10}, + }, + { + name: "an empty vault's rewind is shortened to the lowest snapshot the WAL replays from", + survey: recoverySurvey{ + walStored: true, + walFirst: 5, + walLast: 10, + scSnapshots: snapshots, + }, + depth: 1000, + want: recoveryPlan{head: 10, scTarget: 4, ssTarget: 10, emptyVaultRefillShortened: true}, + }, + { + name: "an empty vault with no replayable snapshot records only the loaded block", + survey: recoverySurvey{ + walStored: true, + walFirst: 9, + walLast: 10, + scSnapshots: []uint64{0, + 4}, + }, + depth: 3, + want: recoveryPlan{head: 10, scTarget: 10, ssTarget: 10, emptyVaultRefillUnreachable: true}, + }, + { + name: "a rollback that empties the WAL leaves nothing to refill the vault from", + survey: recoverySurvey{ + walStored: true, + walFirst: 5, + walLast: 10, + scSnapshots: snapshots, + }, + rollbackTo: 4, + depth: 3, + want: recoveryPlan{head: 4, scTarget: 4, ssTarget: 4, rollback: true}, + }, + } { + t.Run(tc.name, func(t *testing.T) { + got, err := planRecovery(tc.survey, tc.rollbackTo, tc.depth) + require.NoError(t, err) + require.Equal(t, tc.want, got) + }) + } +} + +// A rollback above the WAL's head is a target no replay reaches, refused before anything is planned. +func TestPlanRecoveryRefusesARollbackAboveTheWAL(t *testing.T) { + _, err := planRecovery(recoverySurvey{walStored: true, walFirst: 1, walLast: 3}, 5, 0) + require.ErrorContains(t, err, "the state WAL ends at 3") +} diff --git a/sei-db/state_db/giga/state_db_replay.go b/sei-db/state_db/giga/state_db_replay.go index 01ffd48573..9a02eadfbb 100644 --- a/sei-db/state_db/giga/state_db_replay.go +++ b/sei-db/state_db/giga/state_db_replay.go @@ -5,174 +5,126 @@ import ( "fmt" "time" + "github.com/sei-protocol/sei-chain/sei-db/common/utils" + "github.com/sei-protocol/sei-chain/sei-db/config" "github.com/sei-protocol/sei-chain/sei-db/proto" + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv" + flatkvconfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" "github.com/sei-protocol/sei-chain/sei-db/state_db/ss/evm" + "github.com/sei-protocol/sei-chain/sei-db/state_db/statewal" ) // replayLogInterval bounds how often a running replay reports how far it has got. const replayLogInterval = 30 * time.Second -// rewindTo puts whichever of SC and SS holds state above target on its newest snapshot at or below it, -// drops every snapshot of both above target, and cuts the WAL's tail to it. All three stores must be -// closed, and a store holding nothing above target is left where it is, for the replay to carry forward. -// -// A target the surviving snapshots and the WAL cannot span is refused before the WAL is cut, so every -// target this one could reach is still reachable on a retry. SS is asked once SC has moved, so a -// refusal from SS leaves SC on its snapshot and a retry replays from there. -func (s *StateDB) rewindTo(target int64) error { - wal, err := s.storedWALRange() - if err != nil { - return err - } - if wal.last < target { - return fmt.Errorf("cannot roll back to %d: the state WAL ends at %d, so no replay reaches the "+ - "target", target, wal.last) - } - - // First, so that a refusal from SC comes back with every snapshot still on disk. Once SC has moved, - // its own snapshots above where it landed are gone. - if err := s.discardStateAbove(wal, target); err != nil { - return fmt.Errorf("cannot roll back to %d: %w", target, err) - } - if err := s.dropSnapshotsAbove(target); err != nil { - return err - } - // Last, so that an interruption leaves the WAL still above target and a restart comes back here. - return s.truncateWAL(target) -} - -// discardStateAboveTheWAL puts each store back on the WAL's head when it sits above it, onto its -// newest snapshot at or below the head for the replay to carry forward. Every store must be closed. -// -// A commit writes the WAL unflushed, so a crash can lose its tail while the state committed above that -// tail survives. Those blocks are re-executed from the block store, which a store still holding them -// cannot accept, so the state above the WAL is dropped rather than kept. -func (s *StateDB) discardStateAboveTheWAL(wal storedWALRange) error { - head := wal.last - if head == 0 { - // An empty WAL says nothing about where state belongs: one pruned away behind a snapshot leaves - // the state it covered as the only record of it. - return nil - } - if err := s.discardStateAbove(wal, head); err != nil { - // Named for the open, not for a rollback: nobody asked for one, and an operator sent looking for - // the rollback they did not run is an operator not looking at the WAL head that refused. - return fmt.Errorf("cannot open on the state WAL's head %d: %w", head, err) - } - return nil -} - -// discardStateAbove puts whichever of SC and SS holds state above target onto its newest snapshot at or -// below it. Both stores must be closed, and a store holding nothing above target is left where it is, -// for the replay to carry forward. -// -// Each store is handed the WAL's first block and refuses, without moving, a target this WAL cannot -// replay it back up to. SC is put back first, so a refusal from SS can leave SC already rewound. -func (s *StateDB) discardStateAbove(wal storedWALRange, target int64) error { - if _, err := flatkv.DiscardStateAbove(s.flatkvCfg.DataDir, target, wal.first); err != nil { - return fmt.Errorf("the state commit store cannot reach %d: %w", target, err) - } - if !s.ssCfg.Enable { - return nil - } - if _, err := evm.DiscardStateAbove( - s.ssCfg, s.ssSnapshotRoot(), target, wal.first); err != nil { - return fmt.Errorf("the EVM state store cannot reach %d: %w", target, err) - } - return nil -} - -// dropSnapshotsAbove removes the snapshots of SC and SS above target. +// dropSnapshotsAbove removes the snapshots of SC and SS above target. Both stores must be closed. // // It runs whether or not either store is above target, because an interrupted rollback leaves exactly a // store that is not: it reads as the snapshot it was repointed at, with the branch above it still on // disk. Left there, a later rollback lands on a snapshot from the branch this one abandoned. -func (s *StateDB) dropSnapshotsAbove(target int64) error { - if err := flatkv.DropSnapshotsAbove(s.flatkvCfg.DataDir, target); err != nil { +func dropSnapshotsAbove(flatkvCfg *flatkvconfig.Config, ssCfg config.StateStoreConfig, target uint64) error { + //nolint:gosec // a WAL block number never approaches the int64 ceiling + if err := flatkv.DropSnapshotsAbove(flatkvCfg.DataDir, int64(target)); err != nil { return fmt.Errorf("cannot roll back the state commit store to %d: %w", target, err) } - if !s.ssCfg.Enable { + if !ssCfg.Enable { return nil } - if err := evm.DropSnapshotsAbove(s.ssSnapshotRoot(), target); err != nil { + snapshotRoot := utils.GetStateStoreSnapshotsSiblingPath(ssCfg.EVMDBDirectory) + //nolint:gosec // a WAL block number never approaches the int64 ceiling + if err := evm.DropSnapshotsAbove(snapshotRoot, int64(target)); err != nil { return fmt.Errorf("cannot roll back the EVM state store to %d: %w", target, err) } return nil } // catchUpToWAL replays the WAL into SC and SS up to the last block it holds, which is the height state -// committed to. An empty WAL leaves both stores where they are. +// committed to. ss is nil when SS is disabled. An empty WAL leaves both stores where they are. // // A commit writes the WAL before either store, so a crash between the two leaves one of them a block // behind. Committing from behind the WAL is rejected, so this is what makes an opened StateDB able to // commit. -func (s *StateDB) catchUpToWAL(ctx context.Context) error { - wal, err := s.openWALRange() +func catchUpToWAL( + ctx context.Context, + sc *flatkv.CommitStore, + ss *evm.EVMStateStore, + wal statewal.StateWAL, +) error { + stored, _, last, err := wal.GetStoredRange() if err != nil { - return err + return fmt.Errorf("find the head to catch up to: %w", err) } - if wal.last == 0 { - // Neither store is carried forward, for the reason discardStateAboveTheWAL gives: with no head to - // measure against, a working copy above the current snapshot is the only record of the blocks it - // holds, and dropping it on one store alone would leave the two at different heights. A rollback - // that empties the WAL brings both down in rewindTo, where the target says where they belong. + if !stored { + // Neither store is carried forward, for the reason planRecovery gives: with no head to measure + // against, a working copy above the current snapshot is the only record of the blocks it holds, and + // dropping it on one store alone would leave the two at different heights. A rollback that empties + // the WAL brings both down in applyRecoveryPlan, where the target says where they belong. // // An interrupted commit is still repaired, since the disagreement it leaves needs no head to be // recognised. Nothing below reaches the repair catchUpTo runs. - if err := s.sc.RebuildIfTorn(); err != nil { + if err := sc.RebuildIfTorn(); err != nil { return fmt.Errorf("repair the state commit store's working copy: %w", err) } return nil } - head := wal.last - if err := s.catchUpTo(ctx, head); err != nil { - return err + head := int64(last) //nolint:gosec // a WAL block number never approaches the int64 ceiling + if err := catchUpTo(ctx, sc, ss, wal, head); err != nil { + return fmt.Errorf("catch up to the state WAL's head %d: %w", head, err) } - if err := s.matchHeight(head); err != nil { - // Named for the open, as discardStateAboveTheWAL's refusals are: no rollback ran, and an - // operator sent looking for one is an operator not looking at the head that was not reached. + if err := matchHeight(sc, ss, wal, head); err != nil { + // Named for the open, as a plan's refusals are when no rollback ran: an operator sent looking for + // one is an operator not looking at the head that was not reached. return fmt.Errorf("cannot open on the state WAL's head %d: %w", head, err) } return nil } -// catchUpTo replays the WAL into SC and SS up to target. +// catchUpTo replays the WAL into SC and SS up to target. ss is nil when SS is disabled. // // One pass feeds both. It spans from the lower of their two versions, and each block goes only to the // store still below it, so the WAL is read once rather than once per store. -func (s *StateDB) catchUpTo(ctx context.Context, target int64) error { +func catchUpTo( + ctx context.Context, + sc *flatkv.CommitStore, + ss *evm.EVMStateStore, + wal statewal.StateWAL, + target int64, +) error { // Ahead of the pass, which is what erases the evidence it works from, and here rather than in the // open because every replay of this WAL comes through this function. - if err := s.sc.RebuildIfUnreachable(target); err != nil { + if err := sc.RebuildIfUnreachable(target); err != nil { return fmt.Errorf("rebuild the state commit store's working copy: %w", err) } - scFrom := s.sc.Version() - ssFrom, ssReplays, err := s.ssReplayStart(target) + scFrom := sc.Version() + ssFrom, ssReplays, err := ssReplayStart(ss, wal, target) if err != nil { - return err + return fmt.Errorf("find where the EVM state store replays from: %w", err) } from := scFrom if ssReplays { from = min(from, ssFrom) } - s.logReplayPlan(scFrom, ssFrom, ssReplays, target) + logReplayPlan(ss, scFrom, ssFrom, ssReplays, target) - if err := s.replay(ctx, from, target, func(block int64, changesets []*proto.NamedChangeSet) error { + if err := replay(ctx, wal, from, target, func(block int64, changesets []*proto.NamedChangeSet) error { if block > scFrom { // SC owns no WAL, so re-committing a block read from this one appends nothing. It does run // SC's commit path, so the checkpoint schedule is asked at each block SC takes. - if err := s.sc.CommitStateChanges(block, changesets); err != nil { - return err + if err := sc.CommitStateChanges(block, changesets); err != nil { + return fmt.Errorf("commit the block to the state commit store: %w", err) } } if ssReplays && block > ssFrom { - return s.ss.ApplyReplayedBlock(block, changesets) + if err := ss.ApplyReplayedBlock(block, changesets); err != nil { + return fmt.Errorf("apply the block to the EVM state store: %w", err) + } } return nil }); err != nil { - return err + return fmt.Errorf("replay the state WAL up to %d: %w", target, err) } return nil } @@ -182,13 +134,13 @@ func (s *StateDB) catchUpTo(ctx context.Context, target int64) error { // // It is logged even when nothing is replayed, because an open that had no catching up to do is // otherwise indistinguishable from one still working through a long pass. -func (s *StateDB) logReplayPlan(scFrom int64, ssFrom int64, ssReplays bool, target int64) { +func logReplayPlan(ss *evm.EVMStateStore, scFrom int64, ssFrom int64, ssReplays bool, target int64) { fields := append([]any{"target", target}, replayPlanFields("sc", scFrom, target)...) - if s.ss != nil { + if ss != nil { // A store left out of the pass replays nothing, which is a range ending where it already sits. to := target if !ssReplays { - ssFrom = s.ss.GetLatestVersion() + ssFrom = ss.GetLatestVersion() to = ssFrom } fields = append(fields, replayPlanFields("ss", ssFrom, to)...) @@ -214,12 +166,14 @@ func replayPlanFields(store string, from int64, to int64) []any { // // A cancelled ctx stops the replay between blocks and is reported as an error. The blocks already // applied stay applied, and the next open resumes from the version they left behind. -func (s *StateDB) replay( +func replay( ctx context.Context, - from, target int64, + wal statewal.StateWAL, + from int64, + target int64, apply func(int64, []*proto.NamedChangeSet) error, ) error { - stored, first, last, err := s.wal.GetStoredRange() + stored, first, last, err := wal.GetStoredRange() if err != nil { return fmt.Errorf("read state WAL range: %w", err) } @@ -237,7 +191,7 @@ func (s *StateDB) replay( "are missing (data loss or corruption)", first, start, start, first-1) } - it, err := s.wal.Iterator(start, end) + it, err := wal.Iterator(start, end) if err != nil { return fmt.Errorf("state WAL iterator [%d,%d]: %w", start, end, err) } @@ -318,18 +272,19 @@ func estimate(remaining int64, rate float64) time.Duration { } // ssReplayStart returns the version SS replays forward from, and whether it replays at all. SS is left -// out when it is disabled, already on target, or empty with a WAL that can no longer rebuild it. -func (s *StateDB) ssReplayStart(target int64) (from int64, replays bool, err error) { - if s.ss == nil { +// out when it is disabled (ss is nil), already on target, or empty with a WAL that can no longer rebuild +// it. +func ssReplayStart(ss *evm.EVMStateStore, wal statewal.StateWAL, target int64) (from int64, replays bool, err error) { + if ss == nil { return 0, false, nil } - from = s.ss.GetLatestVersion() + from = ss.GetLatestVersion() if from >= target { return 0, false, nil } - fillForward, err := s.ssFillsForward() + fillForward, err := ssFillsForward(ss, wal) if err != nil { - return 0, false, err + return 0, false, fmt.Errorf("decide whether the EVM state store fills forward: %w", err) } if fillForward { logger.Info("EVM state store left empty to fill forward: it holds no history and the state WAL "+ @@ -343,37 +298,37 @@ func (s *StateDB) ssReplayStart(target int64) (from int64, replays bool, err err // treatment recoveryTarget gives an empty receipt store. It covers a store with no history of its own // behind a WAL that has had a retention cut, where no replay rebuilds it and the alternative is // refusing to start over a store that is merely new. -func (s *StateDB) ssFillsForward() (bool, error) { - if s.ss == nil { +func ssFillsForward(ss *evm.EVMStateStore, wal statewal.StateWAL) (bool, error) { + if ss == nil { return false, nil } - wal, err := s.openWALRange() + stored, first, _, err := wal.GetStoredRange() if err != nil { - return false, err + return false, fmt.Errorf("find whether the state WAL reaches block 1: %w", err) } - return s.ss.GetLatestVersion() == 0 && (wal.last == 0 || wal.first > 1), nil + return ss.GetLatestVersion() == 0 && (!stored || first > 1), nil } -// matchHeight checks SC and SS against blockNum and reports the one that is not on it. An SS left empty -// to fill forward is not held to blockNum. +// matchHeight checks SC and SS against blockNum and reports the one that is not on it. ss is nil when SS +// is disabled, and an SS left empty to fill forward is not held to blockNum. // // The error names no path, since both the open and a rollback converge here; each caller supplies the // height it asked for. -func (s *StateDB) matchHeight(blockNum int64) error { - if got := s.sc.Version(); got != blockNum { +func matchHeight(sc *flatkv.CommitStore, ss *evm.EVMStateStore, wal statewal.StateWAL, blockNum int64) error { + if got := sc.Version(); got != blockNum { return fmt.Errorf("the state commit store landed on %d", got) } - if s.ss == nil { + if ss == nil { return nil } - got := s.ss.GetLatestVersion() + got := ss.GetLatestVersion() if got == blockNum { return nil } if got == 0 { - fillForward, err := s.ssFillsForward() + fillForward, err := ssFillsForward(ss, wal) if err != nil { - return err + return fmt.Errorf("check the EVM state store's height: %w", err) } if fillForward { return nil @@ -381,3 +336,39 @@ func (s *StateDB) matchHeight(blockNum int64) error { } return fmt.Errorf("the EVM state store landed on %d", got) } + +// requireAgreementWithoutWAL refuses a WAL that holds no blocks unless SC, SS and the hash vault already +// agree on the block SC is on, since no replay can bring them together. ss is nil when SS is disabled. +func requireAgreementWithoutWAL( + sc *flatkv.CommitStore, + ss *evm.EVMStateStore, + vault *hashvault.PebbleHashVault, + wal statewal.StateWAL, +) error { + stored, _, _, err := wal.GetStoredRange() + if err != nil { + return fmt.Errorf("read the state WAL's range: %w", err) + } + if stored { + return nil + } + height := sc.Version() + if height == 0 { + return nil + } + if ss != nil { + if got := ss.GetLatestVersion(); got != height { + return fmt.Errorf("the state commit store is on block %d but the EVM state store is on "+ + "block %d", height, got) + } + } + _, status, err := vault.Get(uint64(height)) //nolint:gosec // a committed version is never negative + if err != nil { + return fmt.Errorf("read the hash vault at block %d: %w", height, err) + } + if status != gigatypes.BlockHashStatusFound { + return fmt.Errorf("the hash vault holds no hash for block %d, the block the state commit store is on", + height) + } + return nil +} diff --git a/sei-db/state_db/giga/state_db_replay_test.go b/sei-db/state_db/giga/state_db_replay_test.go index cfd41c9e8c..d64955c8eb 100644 --- a/sei-db/state_db/giga/state_db_replay_test.go +++ b/sei-db/state_db/giga/state_db_replay_test.go @@ -7,53 +7,24 @@ import ( "github.com/sei-protocol/sei-chain/sei-db/common/utils" "github.com/sei-protocol/sei-chain/sei-db/config" - flatkvconfig "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/flatkv/config" "github.com/sei-protocol/sei-chain/sei-db/state_db/ss/evm" "github.com/sei-protocol/sei-chain/sei-db/state_db/statewal" "github.com/stretchr/testify/require" ) // SS keeps no changelog of its own under giga, the state WAL being what catchUpTo replays into it. -// The absence is pinned here rather than left to the config, since recovery rests on it. +// The absence is pinned here rather than left to the config, since recovery rests on it: a config that +// leaves the changelog on still opens without one. func TestGigaOpensSSWithoutAChangelog(t *testing.T) { - newStateDB := func(t *testing.T) *StateDB { - t.Helper() - ssCfg := config.DefaultStateStoreConfig() - ssCfg.Enable = true - ssCfg.EVMDBDirectory = filepath.Join(t.TempDir(), "ss") - return &StateDB{ - flatkvCfg: flatkvconfig.DefaultTestConfig(t), - // As the constructors settle it, which is what makes both paths below agree. - ssCfg: stateStoreConfigFor(ssCfg), - } - } - - t.Run("opened to commit", func(t *testing.T) { - s := newStateDB(t) - require.NoError(t, s.openSS()) - t.Cleanup(func() { _ = s.ss.Close() }) - requireNoSSChangelog(t, s.ssCfg.EVMDBDirectory) - }) - - // The rollback path opens the same databases through DiscardStateAbove rather than openSS, so a - // config settled per-open would miss it. StoredVersions opens nothing when the directory is - // absent, so the store has to exist first. - t.Run("opened to roll back", func(t *testing.T) { - s := newStateDB(t) - require.NoError(t, s.openSS()) - require.NoError(t, s.ss.Close()) - - require.NoError(t, s.discardStateAbove(storedWALRange{first: 1, last: 9}, 7)) - requireNoSSChangelog(t, s.ssCfg.EVMDBDirectory) - }) -} - -// TestStateStoreConfigForDisablesTheInternalWAL pins what the constructors apply, every path that -// opens SS reading the config they settled rather than disabling the log for itself. -func TestStateStoreConfigForDisablesTheInternalWAL(t *testing.T) { - handedIn := config.DefaultStateStoreConfig() - require.False(t, handedIn.DisableInternalWAL, "a caller is not expected to have set it") - require.True(t, stateStoreConfigFor(handedIn).DisableInternalWAL) + stores := newVaultTestStores(t) + stores.ssCfg = config.DefaultStateStoreConfig() + stores.ssCfg.Enable = true + stores.ssCfg.EVMDBDirectory = filepath.Join(t.TempDir(), "ss") + require.False(t, stores.ssCfg.DisableInternalWAL, "a caller is not expected to have set it") + + db := stores.open(t) + require.NoError(t, db.Close()) + requireNoSSChangelog(t, stores.ssCfg.EVMDBDirectory) } func requireNoSSChangelog(t *testing.T, evmDBDirectory string) { @@ -68,14 +39,11 @@ func requireNoSSChangelog(t *testing.T, evmDBDirectory string) { // The directory is one an earlier run with SS on could have left, and the WAL reaches block 1, so a // rollback that read it would come back with a rewind to run. func TestDiscardStateAboveLeavesANodeThatKeepsNoEVMStoreAlone(t *testing.T) { - const target = int64(7) + const target = uint64(7) dir := t.TempDir() - s := &StateDB{ - flatkvCfg: flatkvconfig.DefaultTestConfig(t), - ssCfg: config.StateStoreConfig{Enable: false, EVMDBDirectory: dir}, - } + ssCfg := config.StateStoreConfig{Enable: false, EVMDBDirectory: dir} - require.NoError(t, s.discardStateAbove(storedWALRange{first: 1, last: 9}, target)) + require.NoError(t, discardSSAbove(ssCfg, 1, target)) entries, err := os.ReadDir(dir) require.NoError(t, err) @@ -136,23 +104,20 @@ func TestCatchUpRefusesAWALMissingTheBlocksAStoreNeeds(t *testing.T) { t.Run("the state commit store", func(t *testing.T) { _, _, sc := newTestStateDB(t) - s := &StateDB{wal: &gapWAL{first: 3, last: 4}, sc: sc} - - require.ErrorContains(t, s.catchUpTo(t.Context(), 4), missingBlocks) + require.ErrorContains(t, catchUpTo(t.Context(), sc, nil, &gapWAL{first: 3, last: 4}, 4), missingBlocks) }) t.Run("the EVM state store", func(t *testing.T) { - _, _, sc := newTestStateDB(t) - s := &StateDB{wal: &gapWAL{first: 3, last: 4}, sc: sc, ss: &evm.EVMStateStore{}} + ss := &evm.EVMStateStore{} // The store holds nothing, so the gap is its whole history rather than a hole in it. Refusing // here would report data loss for a store that is merely new, and would do it on every node // past its first retention cut, so it is left out of the replay to fill forward from the target. - _, replays, err := s.ssReplayStart(4) + _, replays, err := ssReplayStart(ss, &gapWAL{first: 3, last: 4}, 4) require.NoError(t, err) require.False(t, replays) - require.Zero(t, s.ss.GetLatestVersion()) + require.Zero(t, ss.GetLatestVersion()) }) } @@ -163,9 +128,7 @@ func TestMatchHeightExcusesAStoreLeftToFillForward(t *testing.T) { for block := int64(1); block <= 4; block++ { require.NoError(t, sc.CommitStateChanges(block, changeset("k", "v"))) } - s := &StateDB{wal: &gapWAL{first: 3, last: 4}, sc: sc, ss: &evm.EVMStateStore{}} - - require.NoError(t, s.matchHeight(4)) + require.NoError(t, matchHeight(sc, &evm.EVMStateStore{}, &gapWAL{first: 3, last: 4}, 4)) } // An empty SS is only excused when the WAL cannot rebuild it. Excusing every version-0 store would @@ -175,9 +138,7 @@ func TestMatchHeightDoesNotExcuseAnEmptyStoreTheWALCanRebuild(t *testing.T) { for block := int64(1); block <= 4; block++ { require.NoError(t, sc.CommitStateChanges(block, changeset("k", "v"))) } - s := &StateDB{wal: &gapWAL{first: 1, last: 4}, sc: sc, ss: &evm.EVMStateStore{}} - - err := s.matchHeight(4) + err := matchHeight(sc, &evm.EVMStateStore{}, &gapWAL{first: 1, last: 4}, 4) require.ErrorContains(t, err, "EVM state store") // Both the open and a rollback converge here, so the height belongs to whichever asked. Naming a @@ -190,10 +151,7 @@ func TestMatchHeightDoesNotExcuseAnEmptyStoreTheWALCanRebuild(t *testing.T) { // reaches back far enough for comes out of recovery holding real history, which is strictly better, and // is how a store that lagged the WAL is populated on restart. func TestCatchUpRebuildsAnEmptyStoreTheWALStillCovers(t *testing.T) { - _, _, sc := newTestStateDB(t) - s := &StateDB{wal: &gapWAL{first: 1, last: 4}, sc: sc, ss: &evm.EVMStateStore{}} - - fillForward, err := s.ssFillsForward() + fillForward, err := ssFillsForward(&evm.EVMStateStore{}, &gapWAL{first: 1, last: 4}) require.NoError(t, err) require.False(t, fillForward, "a WAL starting at block 1 can rebuild an empty store") } diff --git a/sei-db/state_db/giga/state_db_test.go b/sei-db/state_db/giga/state_db_test.go index 18b4c4b097..ba819f422a 100644 --- a/sei-db/state_db/giga/state_db_test.go +++ b/sei-db/state_db/giga/state_db_test.go @@ -57,7 +57,7 @@ func newTestStateDB(t *testing.T) (gigatypes.StateDB, *fakeStateWAL, *flatkv.Com require.NoError(t, liveStateDB.LoadLatest()) wal := &fakeStateWAL{} - return &StateDB{wal: wal, sc: liveStateDB, flatkvCfg: cfg}, wal, liveStateDB + return &StateDB{wal: wal, sc: liveStateDB}, wal, liveStateDB } // changeset builds a changeset setting key to value in the test module. diff --git a/sei-db/state_db/giga/types/state_db.go b/sei-db/state_db/giga/types/state_db.go index ef4aa80cfb..9bb81dc2bb 100644 --- a/sei-db/state_db/giga/types/state_db.go +++ b/sei-db/state_db/giga/types/state_db.go @@ -11,7 +11,23 @@ import ( ) // A callback function for getting the hash of each block. -type HashListener func(ctx context.Context, blockNum int64, hash *lthash.BlockHash) error +type HashListener func(ctx context.Context, blockNum uint64, hash *lthash.BlockHash) error + +// BlockHashStatus is the outcome of a block hash lookup. +type BlockHashStatus uint8 + +// BlockHashStatusError means the lookup failed, and the accompanying error says why. +const BlockHashStatusError BlockHashStatus = 0 + +// BlockHashStatusFound means the hash was found. +const BlockHashStatusFound BlockHashStatus = 1 + +// BlockHashStatusTooOld means the block is below the oldest block whose hash is still kept. +const BlockHashStatusTooOld BlockHashStatus = 2 + +// BlockHashStatusNotReady means the block's hash is not recorded yet: the block has not been hashed, or +// has not been committed. +const BlockHashStatusNotReady BlockHashStatus = 3 // StateDB is the top-level API used by the Giga EVM executor for // read and write. Writes commit into both SC and SS; reads can be served for @@ -40,6 +56,15 @@ type StateDB interface { // This may be useful at startup time to determine the initial hash of the database. RegisterHashListener(listener HashListener) (mostRecentHash lthash.BlockHash, err error) + // GetBlockHeight returns the number of the last block passed to CommitStateChanges, or the block the + // StateDB opened on when none has been passed since. + GetBlockHeight() uint64 + + // GetBlockHash returns the state hash of a recent block without blocking. The hash is valid only when + // the status is BlockHashStatusFound, and the error is non-nil exactly when the status is + // BlockHashStatusError. A returned hash is crash durable. + GetBlockHash(blockNumber uint64) (hash [32]byte, status BlockHashStatus, err error) + // Close releases everything this StateDB was built over, reporting every failure rather than // stopping at the first. Close() error diff --git a/sei-db/state_db/sc/composite/hashlog.go b/sei-db/state_db/sc/composite/hashlog.go index 1543591941..03f8332365 100644 --- a/sei-db/state_db/sc/composite/hashlog.go +++ b/sei-db/state_db/sc/composite/hashlog.go @@ -34,8 +34,7 @@ func (cs *CompositeCommitStore) RecordHashes(hl hashlog.HashLogger, blockNumber // Keyed on the block cosmos committed rather than the hash's own height, which is what keeps // this row complete: a block whose writes never reached flatKV leaves its hash on the height // before, and the AppHash reports that same hash for this block. - //nolint:gosec // commit versions are non-negative - if err := hl.HashListener(cs.ctx, int64(blockNumber), cs.flatKVHash.Load()); err != nil { + if err := hl.HashListener(cs.ctx, blockNumber, cs.flatKVHash.Load()); err != nil { return fmt.Errorf("record flatkv hashes for block %d: %w", blockNumber, err) } } diff --git a/sei-db/state_db/sc/composite/store.go b/sei-db/state_db/sc/composite/store.go index 7e7b552370..6495314f13 100644 --- a/sei-db/state_db/sc/composite/store.go +++ b/sei-db/state_db/sc/composite/store.go @@ -228,7 +228,7 @@ func (cs *CompositeCommitStore) adoptFlatKV(store gigatypes.LiveStateStore) erro // recordFlatKVHash keeps flatKVHash current. It is the listener registered on every flatKV instance // this store adopts. -func (cs *CompositeCommitStore) recordFlatKVHash(_ context.Context, _ int64, hash *lthash.BlockHash) error { +func (cs *CompositeCommitStore) recordFlatKVHash(_ context.Context, _ uint64, hash *lthash.BlockHash) error { cs.flatKVHash.Store(hash) return nil } @@ -1191,7 +1191,7 @@ func (cs *CompositeCommitStore) latticeHash(version int64) ([]byte, error) { } hash := cs.flatKVHash.Load() - if hash.BlockNumber != version { + if hash.BlockNumber != uint64(version) { //nolint:gosec // a committed version is never negative // Block version+1 has not been handed to flatKV yet, so the hash just flushed is version's. // Asserted rather than assumed: this value reaches the AppHash, where a hash for the wrong // height is indistinguishable from the right one. diff --git a/sei-db/state_db/sc/flatkv/finalization_manager.go b/sei-db/state_db/sc/flatkv/finalization_manager.go index 62320e5558..ca7df3a90a 100644 --- a/sei-db/state_db/sc/flatkv/finalization_manager.go +++ b/sei-db/state_db/sc/flatkv/finalization_manager.go @@ -249,7 +249,7 @@ func (fm *FinalizationManager) finalize(pending *pendingFinalization) (stopped b if hash.Error != nil { return false, fmt.Errorf("hash block %d: %w", pending.blockNumber, hash.Error) } - if hash.BlockNumber != pending.blockNumber { + if hash.BlockNumber != uint64(pending.blockNumber) { //nolint:gosec // an offered block is never negative return false, fmt.Errorf("finalization is out of step: holding block %d, hashed block %d", pending.blockNumber, hash.BlockNumber) } diff --git a/sei-db/state_db/sc/flatkv/hash_listeners_test.go b/sei-db/state_db/sc/flatkv/hash_listeners_test.go index 900afa1dcf..6700ec7210 100644 --- a/sei-db/state_db/sc/flatkv/hash_listeners_test.go +++ b/sei-db/state_db/sc/flatkv/hash_listeners_test.go @@ -37,9 +37,9 @@ func commitBlocks(t *testing.T, s *CommitStore, count int) { // recordBlocks returns a listener that records the block number of every hash it is handed, and the // slice it records into. The slice is only safe to read once FlushHashes has returned. -func recordBlocks() (func(context.Context, int64, *lthash.BlockHash) error, *[]int64) { - blocks := &[]int64{} - return func(_ context.Context, blockNumber int64, _ *lthash.BlockHash) error { +func recordBlocks() (func(context.Context, uint64, *lthash.BlockHash) error, *[]uint64) { + blocks := &[]uint64{} + return func(_ context.Context, blockNumber uint64, _ *lthash.BlockHash) error { *blocks = append(*blocks, blockNumber) return nil }, blocks @@ -67,13 +67,13 @@ func TestAListenerSeesEveryBlockInOrder(t *testing.T) { listener, seen := recordBlocks() mostRecent, err := s.RegisterHashListener(listener) require.NoError(t, err) - require.Equal(t, int64(0), mostRecent.BlockNumber, "a fresh store has hashed nothing") + require.Equal(t, uint64(0), mostRecent.BlockNumber, "a fresh store has hashed nothing") const blocks = 8 commitBlocks(t, s, blocks) require.NoError(t, s.FlushHashes()) - require.Equal(t, []int64{1, 2, 3, 4, 5, 6, 7, 8}, *seen) + require.Equal(t, []uint64{1, 2, 3, 4, 5, 6, 7, 8}, *seen) } // FlushHashes is how a caller waits for hashing to catch up, and a hash that has been computed but @@ -107,13 +107,13 @@ func TestRegisterReportsTheBlockTheFirstDeliveryFollows(t *testing.T) { listener, seen := recordBlocks() mostRecent, err := s.RegisterHashListener(listener) require.NoError(t, err) - require.Equal(t, int64(3), mostRecent.BlockNumber) + require.Equal(t, uint64(3), mostRecent.BlockNumber) require.Equal(t, rootHash(s), checksumOf(mostRecent.Global)) commitBlocks(t, s, 2) require.NoError(t, s.FlushHashes()) - require.Equal(t, []int64{4, 5}, *seen, "a listener starts at the block after the one it was told") + require.Equal(t, []uint64{4, 5}, *seen, "a listener starts at the block after the one it was told") } // Listeners are independent: one of them consuming a hash must not take it away from another. @@ -131,8 +131,8 @@ func TestEveryListenerSeesEveryBlock(t *testing.T) { commitBlocks(t, s, 3) require.NoError(t, s.FlushHashes()) - require.Equal(t, []int64{1, 2, 3}, *seenByFirst) - require.Equal(t, []int64{1, 2, 3}, *seenBySecond) + require.Equal(t, []uint64{1, 2, 3}, *seenByFirst) + require.Equal(t, []uint64{1, 2, 3}, *seenBySecond) } // A listener that refuses a block is a caller that cannot keep up with the state it is deriving. The @@ -141,7 +141,7 @@ func TestAListenerThatFailsBricksTheStore(t *testing.T) { s := setupTestStoreWithConfig(t, tightHashPipelineConfig(t)) defer func() { _ = s.Close() }() - _, err := s.RegisterHashListener(func(context.Context, int64, *lthash.BlockHash) error { + _, err := s.RegisterHashListener(func(context.Context, uint64, *lthash.BlockHash) error { return fmt.Errorf("injected listener failure") }) require.NoError(t, err) @@ -171,7 +171,7 @@ func TestANilHashListenerRegistersNothing(t *testing.T) { mostRecent, err := s.RegisterHashListener(nil) require.NoError(t, err) - require.Equal(t, int64(2), mostRecent.BlockNumber, "a nil listener still reports the current hash") + require.Equal(t, uint64(2), mostRecent.BlockNumber, "a nil listener still reports the current hash") // Nothing was registered, so the block below has nobody to deliver to and must still commit. commitBlocks(t, s, 1) @@ -196,7 +196,7 @@ func TestRegistrationsSurviveARollback(t *testing.T) { commitBlocks(t, s, 5) require.NoError(t, s.FlushHashes()) - require.Equal(t, []int64{1, 2, 3, 4, 5}, *seen) + require.Equal(t, []uint64{1, 2, 3, 4, 5}, *seen) require.NoError(t, s.Rollback(3)) @@ -204,7 +204,7 @@ func TestRegistrationsSurviveARollback(t *testing.T) { require.NoError(t, s.FlushHashes()) // 0 is the height the rollback reopened at, then 1 to 3 are replayed, then 4 and 5 re-executed. - require.Equal(t, []int64{1, 2, 3, 4, 5, 0, 1, 2, 3, 4, 5}, *seen, + require.Equal(t, []uint64{1, 2, 3, 4, 5, 0, 1, 2, 3, 4, 5}, *seen, "the listener registered before the rollback must still be given the blocks after it") } @@ -216,7 +216,7 @@ func TestDispatchedPerDBHashesMatchWhatEachDatabaseRecorded(t *testing.T) { defer func() { require.NoError(t, s.Close()) }() var dispatched *lthash.BlockHash - _, err := s.RegisterHashListener(func(_ context.Context, _ int64, hash *lthash.BlockHash) error { + _, err := s.RegisterHashListener(func(_ context.Context, _ uint64, hash *lthash.BlockHash) error { dispatched = hash return nil }) diff --git a/sei-db/state_db/sc/flatkv/lthash/hash_engine.go b/sei-db/state_db/sc/flatkv/lthash/hash_engine.go index 3a162c4653..5382824a4a 100644 --- a/sei-db/state_db/sc/flatkv/lthash/hash_engine.go +++ b/sei-db/state_db/sc/flatkv/lthash/hash_engine.go @@ -107,7 +107,7 @@ func (he *HashEngine) ScheduleHash( return fmt.Errorf("schedule hash: current and previous views are both required") } request := &hashRequest{ - blockNumber: current.BlockHeight(), + blockNumber: uint64(current.BlockHeight()), //nolint:gosec // a sealed view's height is never negative current: current, previous: previous, } diff --git a/sei-db/state_db/sc/flatkv/lthash/hash_engine_messages.go b/sei-db/state_db/sc/flatkv/lthash/hash_engine_messages.go index b52b22266a..bf43544d54 100644 --- a/sei-db/state_db/sc/flatkv/lthash/hash_engine_messages.go +++ b/sei-db/state_db/sc/flatkv/lthash/hash_engine_messages.go @@ -12,7 +12,7 @@ import ( // hashRequest is one sealed block for the engine to hash. type hashRequest struct { // blockNumber is the height being hashed. - blockNumber int64 + blockNumber uint64 // current is the block's own sealed view. The gatherer reads this block's diff from it. current *sview.StoreView @@ -54,7 +54,7 @@ func (r *hashRequest) release() error { // running hash. type gatheredBlock struct { // blockNumber is the height this job hashes. - blockNumber int64 + blockNumber uint64 // hashes is this block's leaf hashing in flight, which the combiner drains to completion. hashes leafHashes diff --git a/sei-db/state_db/sc/flatkv/lthash/hash_engine_test.go b/sei-db/state_db/sc/flatkv/lthash/hash_engine_test.go index 05165d72b8..5cec81fe65 100644 --- a/sei-db/state_db/sc/flatkv/lthash/hash_engine_test.go +++ b/sei-db/state_db/sc/flatkv/lthash/hash_engine_test.go @@ -184,7 +184,7 @@ func TestHashEngineAgreesWithSynchronousCompute(t *testing.T) { require.Equal(t, want.Global.Checksum(), got.Global.Checksum(), "the pipeline must produce the hash a single-call fold produces") - require.Equal(t, int64(1), got.BlockNumber) + require.Equal(t, uint64(1), got.BlockNumber) } // mustViews is blockViews without the stub handles, for a caller that only wants the views. @@ -211,7 +211,7 @@ func TestHashEngineStreamsOneHashPerBlockInOrder(t *testing.T) { for height := int64(1); height <= blocks; height++ { got := <-engine.AwaitHash() require.NoError(t, got.Error) - require.Equal(t, height, got.BlockNumber, "hashes must arrive in block order with no gaps") + require.Equal(t, uint64(height), got.BlockNumber, "hashes must arrive in block order with no gaps") } require.NoError(t, engine.Close()) @@ -285,7 +285,7 @@ func TestHashEngineFlushWaitsForScheduledBlocks(t *testing.T) { for height := int64(1); height <= blocks; height++ { select { case got := <-engine.AwaitHash(): - require.Equal(t, height, got.BlockNumber) + require.Equal(t, uint64(height), got.BlockNumber) default: t.Fatalf("Flush returned before block %d was published", height) } @@ -402,7 +402,7 @@ func TestHashEngineDeliversFailureAndStops(t *testing.T) { got := <-engine.AwaitHash() require.Error(t, got.Error) require.ErrorContains(t, got.Error, "injected diff failure") - require.Equal(t, int64(1), got.BlockNumber) + require.Equal(t, uint64(1), got.BlockNumber) require.ErrorContains(t, engine.Close(), "injected diff failure") diff --git a/sei-db/state_db/sc/flatkv/lthash/hash_types.go b/sei-db/state_db/sc/flatkv/lthash/hash_types.go index 6ee983da5d..9b9f235bd0 100644 --- a/sei-db/state_db/sc/flatkv/lthash/hash_types.go +++ b/sei-db/state_db/sc/flatkv/lthash/hash_types.go @@ -41,7 +41,7 @@ type ModuleHashInfo struct { // and later blocks do not disturb it. type BlockHash struct { // BlockNumber is the height this state describes. - BlockNumber int64 + BlockNumber uint64 // PerDB is each data database's lattice hash root, with an entry for every database the engine was // configured with, so a caller can swap the map in wholesale. diff --git a/sei-db/state_db/sc/flatkv/snapshot.go b/sei-db/state_db/sc/flatkv/snapshot.go index 6e0dfaa20d..6057cae55c 100644 --- a/sei-db/state_db/sc/flatkv/snapshot.go +++ b/sei-db/state_db/sc/flatkv/snapshot.go @@ -770,6 +770,18 @@ func repointAtSnapshot(dir string, version int64) error { return nil } +// SnapshotVersions returns the versions of the snapshots of the closed store under dir, lowest first. +func SnapshotVersions(dir string) ([]int64, error) { + var versions []int64 + if err := traverseSnapshots(dir, true, func(version int64) (bool, error) { + versions = append(versions, version) + return false, nil + }); err != nil { + return nil, fmt.Errorf("list snapshots under %q: %w", dir, err) + } + return versions, nil +} + // DiscardStateAbove puts the closed store under dir on its newest snapshot at or below target when it // holds any state above target, and reports the version its files hold once it returns. A store holding // nothing above target is left alone, reported at the version it opens on, for a replay to carry it diff --git a/sei-db/state_db/sc/flatkv/store.go b/sei-db/state_db/sc/flatkv/store.go index 30d1902508..7dc0e61c66 100644 --- a/sei-db/state_db/sc/flatkv/store.go +++ b/sei-db/state_db/sc/flatkv/store.go @@ -1182,7 +1182,7 @@ func (s *CommitStore) deriveGlobalState() { } s.committedVersion = version - s.loadedHashes.BlockNumber = version + s.loadedHashes.BlockNumber = uint64(version) //nolint:gosec // a loaded version is never negative s.loadedHashes.Global = lthash.SumDBHashes(dataDBDirs, s.loadedHashes.PerDB) } diff --git a/sei-db/state_db/sc/flatkv/store_meta.go b/sei-db/state_db/sc/flatkv/store_meta.go index a9d79843c2..0545c97f76 100644 --- a/sei-db/state_db/sc/flatkv/store_meta.go +++ b/sei-db/state_db/sc/flatkv/store_meta.go @@ -406,7 +406,7 @@ func (s *CommitStore) SetInitialVersion(initialVersion int64) error { } s.committedVersion = seededVersion - s.loadedHashes.BlockNumber = seededVersion + s.loadedHashes.BlockNumber = uint64(seededVersion) //nolint:gosec // a seeded version is positive // The engine must carry back what this established, or the first real block would be measured // against different state than was persisted. diff --git a/sei-db/state_db/sc/flatkv/store_write.go b/sei-db/state_db/sc/flatkv/store_write.go index 163811da9c..f7b255b62c 100644 --- a/sei-db/state_db/sc/flatkv/store_write.go +++ b/sei-db/state_db/sc/flatkv/store_write.go @@ -385,7 +385,7 @@ func (s *CommitStore) FinalizeImport(version int64) error { } s.loadedHashes.Global = lthash.SumDBHashes(dataDBDirs, s.loadedHashes.PerDB) - s.loadedHashes.BlockNumber = version + s.loadedHashes.BlockNumber = uint64(version) //nolint:gosec // an imported version is never negative s.committedVersion = version // The engine's accumulator described the databases this import has just replaced wholesale, so it is diff --git a/sei-db/state_db/sc/flatkv/store_write_test.go b/sei-db/state_db/sc/flatkv/store_write_test.go index e62d984bb5..5a3e9738ee 100644 --- a/sei-db/state_db/sc/flatkv/store_write_test.go +++ b/sei-db/state_db/sc/flatkv/store_write_test.go @@ -1927,8 +1927,8 @@ func TestHashFailureSurfacesToACallerAndStopsDispatch(t *testing.T) { defer func() { _ = s.Close() }() // Registered before the first block, since a listener only ever sees the blocks after it. - dispatched := make(chan int64, 8) - _, err := s.RegisterHashListener(func(_ context.Context, blockNumber int64, _ *lthash.BlockHash) error { + dispatched := make(chan uint64, 8) + _, err := s.RegisterHashListener(func(_ context.Context, blockNumber uint64, _ *lthash.BlockHash) error { dispatched <- blockNumber return nil }) @@ -1942,7 +1942,7 @@ func TestHashFailureSurfacesToACallerAndStopsDispatch(t *testing.T) { commitAndCheck(t, s) require.NoError(t, s.FlushHashes()) - require.Equal(t, int64(1), <-dispatched, "the good block hashes normally") + require.Equal(t, uint64(1), <-dispatched, "the good block hashes normally") s.moduleOf = func([]byte) (string, error) { return "", fmt.Errorf("injected moduleOf failure") @@ -1985,14 +1985,14 @@ func TestAReadOnlyStoreReportsItsHeight(t *testing.T) { require.NoError(t, err) defer func() { _ = ro.Close() }() - delivered := make(chan int64, 4) + delivered := make(chan uint64, 4) mostRecent, err := ro.RegisterHashListener( - func(_ context.Context, blockNumber int64, _ *lthash.BlockHash) error { + func(_ context.Context, blockNumber uint64, _ *lthash.BlockHash) error { delivered <- blockNumber return nil }) require.NoError(t, err) - require.Equal(t, ro.Version(), mostRecent.BlockNumber, + require.Equal(t, uint64(ro.Version()), mostRecent.BlockNumber, "registration must report the height the read-only store was opened at") require.NoError(t, ro.FlushHashes()) diff --git a/sei-db/state_db/sc/hashlog/flatkv_listener_test.go b/sei-db/state_db/sc/hashlog/flatkv_listener_test.go index a938d3ee5e..df793dd883 100644 --- a/sei-db/state_db/sc/hashlog/flatkv_listener_test.go +++ b/sei-db/state_db/sc/hashlog/flatkv_listener_test.go @@ -39,7 +39,7 @@ func checksumOf(hash *lthash.LtHash) []byte { // flatKVBlockHash returns a block hash with a distinct root and a distinct hash for each of flatKV's // data databases. -func flatKVBlockHash(t *testing.T, blockNumber int64) *lthash.BlockHash { +func flatKVBlockHash(t *testing.T, blockNumber uint64) *lthash.BlockHash { t.Helper() return <hash.BlockHash{ BlockNumber: blockNumber, diff --git a/sei-db/state_db/sc/hashlog/hash_logger.go b/sei-db/state_db/sc/hashlog/hash_logger.go index 9f76f7a1a4..e347c34a70 100644 --- a/sei-db/state_db/sc/hashlog/hash_logger.go +++ b/sei-db/state_db/sc/hashlog/hash_logger.go @@ -75,7 +75,7 @@ type HashLogger interface { // // The columns reported here are fixed, and a node declares them when it constructs the logger // (see HashLoggerConfig.HashTypes). Nothing registers a column per block. - HashListener(ctx context.Context, blockNumber int64, hash *lthash.BlockHash) error + HashListener(ctx context.Context, blockNumber uint64, hash *lthash.BlockHash) error // Shut down the HashLogger and release any resources. Flushes pending writes before returning. Only blocks // that are complete (a hash has been reported for every configured type) are written; a block still missing a diff --git a/sei-db/state_db/sc/hashlog/hash_logger_impl.go b/sei-db/state_db/sc/hashlog/hash_logger_impl.go index d892af8f4f..aa4d74fefe 100644 --- a/sei-db/state_db/sc/hashlog/hash_logger_impl.go +++ b/sei-db/state_db/sc/hashlog/hash_logger_impl.go @@ -448,9 +448,7 @@ func (h *hashLoggerImpl) ReportHash(blockNumber uint64, hashType string, hash [] // HashListener records one block's flatKV hashes: the store-wide root and each data database's root. // Its signature is gigatypes.HashListener, so it registers as one directly: // stateDB.RegisterHashListener(hashLogger.HashListener). -func (h *hashLoggerImpl) HashListener(_ context.Context, blockNumber int64, hash *lthash.BlockHash) error { - block := uint64(blockNumber) //nolint:gosec // commit versions are non-negative - +func (h *hashLoggerImpl) HashListener(_ context.Context, block uint64, hash *lthash.BlockHash) error { root := hash.Global.Checksum() if err := h.ReportHash(block, FlatKVRootHashType, root[:]); err != nil { return fmt.Errorf("record the flatkv root hash of block %d: %w", block, err) diff --git a/sei-db/state_db/sc/hashlog/noop_hash_logger.go b/sei-db/state_db/sc/hashlog/noop_hash_logger.go index 8b097340ec..a2aa7af543 100644 --- a/sei-db/state_db/sc/hashlog/noop_hash_logger.go +++ b/sei-db/state_db/sc/hashlog/noop_hash_logger.go @@ -37,7 +37,7 @@ func (n *noOpHashLogger) ReportHash(uint64, string, []byte) error { return nil } -func (n *noOpHashLogger) HashListener(context.Context, int64, *lthash.BlockHash) error { +func (n *noOpHashLogger) HashListener(context.Context, uint64, *lthash.BlockHash) error { // intentional no-op return nil } diff --git a/sei-db/state_db/sc/hashvault/hashvault_config.go b/sei-db/state_db/sc/hashvault/hashvault_config.go index 39394afda7..24205d09e7 100644 --- a/sei-db/state_db/sc/hashvault/hashvault_config.go +++ b/sei-db/state_db/sc/hashvault/hashvault_config.go @@ -18,13 +18,24 @@ type HashVaultConfig struct { // CacheSize is the number of recent (height -> verified hash) entries in the in-process LRU cache. CacheSize int + + // HaltOnMismatch selects what a hash that differs from the recorded one does. When true, CommitToHash + // returns ErrHashMismatch. When false, the mismatch is logged, the recorded hashes from that block up + // are discarded, and the new hash is recorded in their place. + HaltOnMismatch bool + + // EmptyVaultRollbackBlocks is how many blocks the state DB rewinds and replays when it opens over an + // empty vault, so that the vault holds the hashes of recent blocks and not just the loaded one. + EmptyVaultRollbackBlocks uint64 } // DefaultHashVaultConfig returns a HashVaultConfig with production defaults. func DefaultHashVaultConfig() HashVaultConfig { return HashVaultConfig{ - Fsync: false, - CacheSize: 1024, + Fsync: false, + CacheSize: 1024, + HaltOnMismatch: false, + EmptyVaultRollbackBlocks: 1000, } } diff --git a/sei-db/state_db/sc/hashvault/hashvault_test.go b/sei-db/state_db/sc/hashvault/hashvault_test.go index 3eff0ee259..a3269254a3 100644 --- a/sei-db/state_db/sc/hashvault/hashvault_test.go +++ b/sei-db/state_db/sc/hashvault/hashvault_test.go @@ -165,21 +165,12 @@ func TestConcurrentCommits(t *testing.T) { ctx := context.Background() v := newTestPebbleVault(t) - // 100 goroutines, each commits a distinct height. All should succeed. + // Commit heights 1..N in order, since the vault refuses gaps. var wg sync.WaitGroup const N = 100 - errs := make(chan error, N) for i := 0; i < N; i++ { - wg.Add(1) - go func(h uint64) { - defer wg.Done() - errs <- v.CommitToHash(ctx, h, bytesOfLen(byte(h), 32)) - }(uint64(i + 1)) - } - wg.Wait() - close(errs) - for err := range errs { - require.NoError(t, err) + h := uint64(i + 1) + require.NoError(t, v.CommitToHash(ctx, h, bytesOfLen(byte(h), 32))) } // Re-committing the same (height, hash) from many goroutines should also all succeed. diff --git a/sei-db/state_db/sc/hashvault/pebble_hashvault.go b/sei-db/state_db/sc/hashvault/pebble_hashvault.go index 853e227b70..a70db196f5 100644 --- a/sei-db/state_db/sc/hashvault/pebble_hashvault.go +++ b/sei-db/state_db/sc/hashvault/pebble_hashvault.go @@ -14,6 +14,7 @@ import ( "github.com/sei-protocol/seilog" "github.com/sei-protocol/sei-chain/sei-db/common/utils" + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" ) var _ HashVault = (*PebbleHashVault)(nil) @@ -32,6 +33,11 @@ type PebbleHashVault struct { // pruneBoundary is the lowest height that may still be committed. pruneBoundary uint64 cache *lru.Cache[uint64, []byte] + + // head is the newest recorded height. Meaningless when notEmpty is false. + head uint64 + // notEmpty is true when the vault holds at least one hash. + notEmpty bool } // NewPebbleHashVault opens (or creates) a PebbleHashVault rooted at config.DataDir. @@ -82,6 +88,11 @@ func newPebbleHashVault(_ context.Context, config HashVaultConfig) (*PebbleHashV return nil, err } + if err := p.loadHead(); err != nil { + _ = db.Close() + return nil, fmt.Errorf("failed to open hashvault: %w", err) + } + empty, err := p.isEmpty() if err != nil { _ = db.Close() @@ -147,9 +158,15 @@ func (p *PebbleHashVault) CommitToHash(ctx context.Context, blockHeight uint64, if len(hash) != BlockHashSize { return ErrInvalidHashLength } + if p.notEmpty && blockHeight > p.head+1 { + return fmt.Errorf("block %d would leave a gap after the newest recorded block %d", blockHeight, p.head) + } if cached, ok := p.cache.Get(blockHeight); ok { if !bytes.Equal(cached, hash) { + if !p.config.HaltOnMismatch { + return p.acceptMismatchedHash(blockHeight, cached, hash) + } p.logHashMismatch(blockHeight, cached, hash) return ErrHashMismatch } @@ -160,12 +177,21 @@ func (p *PebbleHashVault) CommitToHash(ctx context.Context, blockHeight uint64, raw, closer, err := p.db.Get(key) switch { case errors.Is(err, pebble.ErrNotFound): + if p.notEmpty && blockHeight <= p.head { + // Below the oldest recorded height, where there is nothing to check the hash against. + if !p.config.HaltOnMismatch { + return p.acceptMismatchedHash(blockHeight, nil, hash) + } + return ErrBelowPruneBoundary + } // First commit for this height: write it. value := encodeHashValue(blockHeight, hash) if werr := p.db.Set(key, value, p.writeOpts); werr != nil { return fmt.Errorf("failed to persist hash for block %d: %w", blockHeight, werr) } p.cache.Add(blockHeight, bytes.Clone(hash)) + p.head = max(p.head, blockHeight) + p.notEmpty = true return nil case err != nil: return fmt.Errorf("failed to read hash for block %d: %w", blockHeight, err) @@ -182,6 +208,9 @@ func (p *PebbleHashVault) CommitToHash(ctx context.Context, blockHeight uint64, return err } if !bytes.Equal(existing, hash) { + if !p.config.HaltOnMismatch { + return p.acceptMismatchedHash(blockHeight, existing, hash) + } p.logHashMismatch(blockHeight, existing, hash) return ErrHashMismatch } @@ -254,3 +283,163 @@ func (p *PebbleHashVault) logHashMismatch(blockHeight uint64, existing, incoming "hashVaultDir", p.config.DataDir, ) } + +// loadHead reads the newest recorded height from disk and populates p.head and p.recorded. +func (p *PebbleHashVault) loadHead() error { + _, head, recorded, err := storedRange(p.db) + if err != nil { + return fmt.Errorf("failed to read the newest recorded height: %w", err) + } + p.head = head + p.notEmpty = recorded + return nil +} + +// Records a hash that differs from the recorded one, discarding every hash from blockHeight up in the same +// atomic batch. Used when HaltOnMismatch is false. p.mu must be held. +func (p *PebbleHashVault) acceptMismatchedHash(blockHeight uint64, existing []byte, hash []byte) error { + logger.Error("Hashvault detected a state hash mismatch; hash-vault-halt-on-mismatch is false, so the "+ + "recorded hashes from this block up are discarded and the new hash replaces them.", + "blockHeight", blockHeight, + "existingHex", hex.EncodeToString(existing), + "incomingHex", hex.EncodeToString(hash), + "hashVaultDir", p.config.DataDir, + ) + batch := p.db.NewBatch() + defer func() { _ = batch.Close() }() + if err := batch.DeleteRange(hashKey(blockHeight), hashKeyUpperBound(), nil); err != nil { + return fmt.Errorf("failed to stage discarding hashes from block %d: %w", blockHeight, err) + } + if err := batch.Set(hashKey(blockHeight), encodeHashValue(blockHeight, hash), nil); err != nil { + return fmt.Errorf("failed to stage the replacing hash for block %d: %w", blockHeight, err) + } + if err := batch.Commit(p.writeOpts); err != nil { + return fmt.Errorf("failed to replace hashes from block %d: %w", blockHeight, err) + } + p.cache.Purge() + p.cache.Add(blockHeight, bytes.Clone(hash)) + p.head = blockHeight + p.notEmpty = true + return nil +} + +// Head returns the newest recorded height, and false when the vault holds no hashes. +func (p *PebbleHashVault) Head() (uint64, bool) { + p.mu.Lock() + defer p.mu.Unlock() + return p.head, p.notEmpty +} + +// Get returns the hash recorded for blockHeight, without blocking. +func (p *PebbleHashVault) Get(blockHeight uint64) ([32]byte, gigatypes.BlockHashStatus, error) { + p.mu.Lock() + defer p.mu.Unlock() + if p.closed.IsClosed() { + return [32]byte{}, gigatypes.BlockHashStatusError, ErrClosed + } + if !p.notEmpty || blockHeight > p.head { + return [32]byte{}, gigatypes.BlockHashStatusNotReady, nil + } + if blockHeight < p.pruneBoundary { + return [32]byte{}, gigatypes.BlockHashStatusTooOld, nil + } + raw, closer, err := p.db.Get(hashKey(blockHeight)) + if errors.Is(err, pebble.ErrNotFound) { + return [32]byte{}, gigatypes.BlockHashStatusTooOld, nil + } + if err != nil { + return [32]byte{}, gigatypes.BlockHashStatusError, + fmt.Errorf("failed to read hash for block %d: %w", blockHeight, err) + } + defer func() { _ = closer.Close() }() + hash, err := decodeHashValue(blockHeight, raw) + if err != nil { + return [32]byte{}, gigatypes.BlockHashStatusError, + fmt.Errorf("failed to decode hash for block %d: %w", blockHeight, err) + } + var out [32]byte + copy(out[:], hash) + return out, gigatypes.BlockHashStatusFound, nil +} + +// Reset deletes every recorded hash and the prune boundary, and records hash as blockHeight's, leaving it +// the only one. +func (p *PebbleHashVault) Reset(ctx context.Context, blockHeight uint64, hash []byte) error { + if err := ctx.Err(); err != nil { + return fmt.Errorf("failed to reset hashvault: %w", err) + } + p.mu.Lock() + defer p.mu.Unlock() + if p.closed.IsClosed() { + return ErrClosed + } + if len(hash) != BlockHashSize { + return ErrInvalidHashLength + } + batch := p.db.NewBatch() + defer func() { _ = batch.Close() }() + if err := batch.DeleteRange(hashKey(0), hashKeyUpperBound(), nil); err != nil { + return fmt.Errorf("failed to stage hashvault reset: %w", err) + } + if err := batch.Delete(pruneBoundaryKey, nil); err != nil { + return fmt.Errorf("failed to stage prune boundary clear during reset: %w", err) + } + if err := batch.Set(hashKey(blockHeight), encodeHashValue(blockHeight, hash), nil); err != nil { + return fmt.Errorf("failed to stage hash for block %d during reset: %w", blockHeight, err) + } + if err := batch.Commit(p.writeOpts); err != nil { + return fmt.Errorf("failed to reset hashvault to block %d: %w", blockHeight, err) + } + p.pruneBoundary = 0 + p.cache.Purge() + p.cache.Add(blockHeight, bytes.Clone(hash)) + p.head = blockHeight + p.notEmpty = true + return nil +} + +// Name implements controller.PrunableStore. +func (p *PebbleHashVault) Name() string { + return "HashVault" +} + +// PruneHistory implements controller.PrunableStore. It deletes the hashes of blocks below blockHeight, +// always keeping the newest recorded hash. +func (p *PebbleHashVault) PruneHistory(blockHeight uint64) error { + head, recorded := p.Head() + if !recorded { + return nil + } + floor := min(blockHeight, head) + if err := p.Prune(context.Background(), floor); err != nil { + return fmt.Errorf("failed to prune hashvault below %d: %w", floor, err) + } + return nil +} + +// PruneSnapshots implements controller.PrunableStore. The vault keeps no snapshots. +func (p *PebbleHashVault) PruneSnapshots(uint64) error { + return nil +} + +// ExternalPruning implements controller.PrunableStore. The vault has no pruner of its own. +func (p *PebbleHashVault) ExternalPruning() bool { + return true +} + +// GetRollbackFloor implements controller.PrunableStore. Every recorded block is readable directly, so the +// floor is the newest recorded block less rollbackWindow, or 0 when the window is deeper than that. +func (p *PebbleHashVault) GetRollbackFloor(rollbackWindow uint64) uint64 { + head, recorded := p.Head() + if !recorded || head < rollbackWindow { + return 0 + } + return head - rollbackWindow +} + +// GetLatestBlock implements controller.PrunableStore. It returns the newest recorded block, or 0 when the +// vault holds no hashes. +func (p *PebbleHashVault) GetLatestBlock() (uint64, error) { + head, _ := p.Head() + return head, nil +} diff --git a/sei-db/state_db/sc/hashvault/pebble_hashvault_branch_test.go b/sei-db/state_db/sc/hashvault/pebble_hashvault_branch_test.go new file mode 100644 index 0000000000..6995b54f8a --- /dev/null +++ b/sei-db/state_db/sc/hashvault/pebble_hashvault_branch_test.go @@ -0,0 +1,178 @@ +package hashvault + +import ( + "context" + "path/filepath" + "testing" + + "github.com/stretchr/testify/require" + + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" +) + +// commitHeights commits the hashes of heights first to last, each seeded with its own height. +func commitHeights(t *testing.T, v *PebbleHashVault, first uint64, last uint64) { + t.Helper() + for h := first; h <= last; h++ { + require.NoError(t, v.CommitToHash(context.Background(), h, bytesOfLen(byte(h), 32))) + } +} + +// requireStatus asserts the status Get reports for height. +func requireStatus(t *testing.T, v *PebbleHashVault, height uint64, want gigatypes.BlockHashStatus) { + t.Helper() + _, status, err := v.Get(height) + require.NoError(t, err) + require.Equal(t, want, status, "height %d", height) +} + +// requireRecorded asserts the vault holds a hash seeded with seed for height. +func requireRecorded(t *testing.T, v *PebbleHashVault, height uint64, seed byte) { + t.Helper() + hash, status, err := v.Get(height) + require.NoError(t, err) + require.Equal(t, gigatypes.BlockHashStatusFound, status, "height %d", height) + require.Equal(t, bytesOfLen(seed, 32), hash[:], "height %d", height) +} + +func warnOnMismatch(cfg *HashVaultConfig) { + cfg.HaltOnMismatch = false +} + +func TestCommitRefusesAGap(t *testing.T) { + v := newTestPebbleVault(t) + commitHeights(t, v, 1, 3) + require.ErrorContains(t, v.CommitToHash(context.Background(), 5, bytesOfLen(5, 32)), "gap") + head, recorded := v.Head() + require.True(t, recorded) + require.Equal(t, uint64(3), head) +} + +func TestHeadSurvivesARestart(t *testing.T) { + v := newTestPebbleVault(t) + commitHeights(t, v, 1, 3) + v2 := reopenTestPebbleVault(t, v) + head, recorded := v2.Head() + require.True(t, recorded) + require.Equal(t, uint64(3), head) + require.ErrorContains(t, v2.CommitToHash(context.Background(), 5, bytesOfLen(5, 32)), "gap") +} + +func TestMismatchReplacesTheRecordWhenNotHalting(t *testing.T) { + v := newTestPebbleVault(t, warnOnMismatch) + commitHeights(t, v, 1, 5) + + require.NoError(t, v.CommitToHash(context.Background(), 3, bytesOfLen(0xEE, 32))) + requireRecorded(t, v, 2, 2) + requireRecorded(t, v, 3, 0xEE) + requireStatus(t, v, 4, gigatypes.BlockHashStatusNotReady) + require.NoError(t, v.CommitToHash(context.Background(), 4, bytesOfLen(0xEF, 32))) +} + +func TestCommitBelowTheOldestRecordedHeight(t *testing.T) { + t.Run("halting", func(t *testing.T) { + v := newTestPebbleVault(t) + commitHeights(t, v, 10, 12) + require.ErrorIs(t, v.CommitToHash(context.Background(), 5, bytesOfLen(5, 32)), ErrBelowPruneBoundary) + }) + t.Run("not halting", func(t *testing.T) { + v := newTestPebbleVault(t, warnOnMismatch) + commitHeights(t, v, 10, 12) + require.NoError(t, v.CommitToHash(context.Background(), 5, bytesOfLen(5, 32))) + requireRecorded(t, v, 5, 5) + requireStatus(t, v, 10, gigatypes.BlockHashStatusNotReady) + }) +} + +func TestGetStatuses(t *testing.T) { + v := newTestPebbleVault(t) + requireStatus(t, v, 1, gigatypes.BlockHashStatusNotReady) + + commitHeights(t, v, 1, 10) + require.NoError(t, v.Prune(context.Background(), 5)) + requireStatus(t, v, 4, gigatypes.BlockHashStatusTooOld) + requireRecorded(t, v, 5, 5) + requireStatus(t, v, 11, gigatypes.BlockHashStatusNotReady) + + require.NoError(t, v.Close(context.Background())) + _, status, err := v.Get(5) + require.ErrorIs(t, err, ErrClosed) + require.Equal(t, gigatypes.BlockHashStatusError, status) +} + +func TestResetLeavesOnlyTheGivenHeight(t *testing.T) { + ctx := context.Background() + v := newTestPebbleVault(t) + commitHeights(t, v, 1, 5) + require.NoError(t, v.Prune(ctx, 3)) + + require.NoError(t, v.Reset(ctx, 1000, bytesOfLen(0xAB, 32))) + requireRecorded(t, v, 1000, 0xAB) + requireStatus(t, v, 4, gigatypes.BlockHashStatusTooOld) + require.NoError(t, v.CommitToHash(ctx, 1001, bytesOfLen(0xAC, 32))) + + cfg := v.config + require.NoError(t, v.Close(ctx)) + oldest, newest, recorded, err := StoredRange(cfg) + require.NoError(t, err) + require.True(t, recorded) + require.Equal(t, uint64(1000), oldest) + require.Equal(t, uint64(1001), newest) +} + +func TestPruneHistoryDeletesBelowTheCutLine(t *testing.T) { + v := newTestPebbleVault(t) + commitHeights(t, v, 1, 150) + + require.NoError(t, v.PruneHistory(60)) + requireStatus(t, v, 59, gigatypes.BlockHashStatusTooOld) + requireRecorded(t, v, 60, 60) + + // A lower cut line than one already applied restores nothing. + require.NoError(t, v.PruneHistory(30)) + requireStatus(t, v, 59, gigatypes.BlockHashStatusTooOld) + requireRecorded(t, v, 60, 60) + + // A cut line above the head keeps the newest hash. + require.NoError(t, v.PruneHistory(1000)) + requireStatus(t, v, 149, gigatypes.BlockHashStatusTooOld) + requireRecorded(t, v, 150, 150) +} + +func TestRollbackFloorIsTheHeadLessTheWindow(t *testing.T) { + v := newTestPebbleVault(t) + require.Equal(t, uint64(0), v.GetRollbackFloor(10)) + commitHeights(t, v, 1, 30) + require.Equal(t, uint64(20), v.GetRollbackFloor(10)) + require.Equal(t, uint64(0), v.GetRollbackFloor(40)) + latest, err := v.GetLatestBlock() + require.NoError(t, err) + require.Equal(t, uint64(30), latest) +} + +func TestStoredRange(t *testing.T) { + ctx := context.Background() + cfg := DefaultHashVaultConfig() + cfg.DataDir = filepath.Join(t.TempDir(), "vault") + + _, _, recorded, err := StoredRange(cfg) + require.NoError(t, err) + require.False(t, recorded, "a vault that was never created records nothing") + + v, err := NewUnsafePebbleHashVault(ctx, cfg) + require.NoError(t, err) + require.NoError(t, v.Close(ctx)) + _, _, recorded, err = StoredRange(cfg) + require.NoError(t, err) + require.False(t, recorded, "an empty vault records nothing") + + v, err = NewUnsafePebbleHashVault(ctx, cfg) + require.NoError(t, err) + commitHeights(t, v, 4, 9) + require.NoError(t, v.Close(ctx)) + oldest, newest, recorded, err := StoredRange(cfg) + require.NoError(t, err) + require.True(t, recorded) + require.Equal(t, uint64(4), oldest) + require.Equal(t, uint64(9), newest) +} diff --git a/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback.go b/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback.go index da57093aa2..dc991fb6be 100644 --- a/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback.go +++ b/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback.go @@ -109,3 +109,51 @@ func wipeEntireStore(db *pebble.DB, dataDir string, target, boundary uint64) err "dataDir", dataDir, "rollbackTarget", target, "pruneBoundary", boundary) return nil } + +// StoredRange returns the oldest and newest heights the closed vault under config.DataDir records, and +// false when it records none. A vault that has never been created records none. +func StoredRange(config HashVaultConfig) (oldest uint64, newest uint64, recorded bool, err error) { + if _, err := os.Stat(config.DataDir); err != nil { + if os.IsNotExist(err) { + return 0, 0, false, nil + } + return 0, 0, false, fmt.Errorf("hashvault data dir %q is not accessible: %w", config.DataDir, err) + } + db, err := pebble.Open(config.DataDir, &pebble.Options{ReadOnly: true}) + if err != nil { + return 0, 0, false, fmt.Errorf("failed to open hashvault pebble db at %q read-only: %w", + config.DataDir, err) + } + defer func() { + if closeErr := db.Close(); closeErr != nil { + err = errors.Join(err, fmt.Errorf("failed to close read-only hashvault: %w", closeErr)) + } + }() + return storedRange(db) +} + +// storedRange returns the oldest and newest heights db records, and false when it records none. +func storedRange(db *pebble.DB) (oldest uint64, newest uint64, recorded bool, err error) { + iter, err := db.NewIter(&pebble.IterOptions{LowerBound: hashKey(0), UpperBound: hashKeyUpperBound()}) + if err != nil { + return 0, 0, false, fmt.Errorf("failed to open hashvault iterator: %w", err) + } + defer func() { + if closeErr := iter.Close(); closeErr != nil { + err = errors.Join(err, fmt.Errorf("failed to close hashvault iterator: %w", closeErr)) + } + }() + if !iter.First() { + return 0, 0, false, nil + } + if oldest, err = decodeHashKey(iter.Key()); err != nil { + return 0, 0, false, fmt.Errorf("failed to decode the oldest hashvault key: %w", err) + } + if !iter.Last() { + return 0, 0, false, fmt.Errorf("hashvault iterator found an oldest key but no newest key") + } + if newest, err = decodeHashKey(iter.Key()); err != nil { + return 0, 0, false, fmt.Errorf("failed to decode the newest hashvault key: %w", err) + } + return oldest, newest, true, nil +} diff --git a/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback_test.go b/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback_test.go index e6e2d8e5b0..03593ca137 100644 --- a/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback_test.go +++ b/sei-db/state_db/sc/hashvault/pebble_hashvault_rollback_test.go @@ -7,6 +7,7 @@ import ( "testing" "github.com/cockroachdb/pebble/v2" + gigatypes "github.com/sei-protocol/sei-chain/sei-db/state_db/giga/types" "github.com/stretchr/testify/require" ) @@ -59,9 +60,11 @@ func TestHardRollbackPebbleHashVaultBelowPruneBoundaryWipesStore(t *testing.T) { // Boundary is gone, so commits below the old boundary are now accepted. require.NoError(t, v2.CommitToHash(ctx, 5, bytesOfLen(0xCC, 32))) - // Every previously-locked hash is also gone — height 50 used to be 0x32, but the wipe means a - // fresh hash there is allowed. - require.NoError(t, v2.CommitToHash(ctx, 50, bytesOfLen(0xEE, 32))) + // Every previously-locked hash is also gone: height 50 used to be 0x32, and nothing is recorded + // there now. + _, status, err := v2.Get(50) + require.NoError(t, err) + require.Equal(t, gigatypes.BlockHashStatusNotReady, status) } func TestHardRollbackPebbleHashVaultEqualToPruneBoundary(t *testing.T) { @@ -120,12 +123,6 @@ func TestHardRollbackPebbleHashVaultRejectsMaxUint64Height(t *testing.T) { err := HardRollbackPebbleHashVault(ctx, cfg, math.MaxUint64) require.ErrorIs(t, err, ErrRollbackHeightOverflow) - - v2, err := NewUnsafePebbleHashVault(ctx, cfg) - require.NoError(t, err) - t.Cleanup(func() { _ = v2.Close(ctx) }) - - require.ErrorIs(t, v2.CommitToHash(ctx, math.MaxUint64, bytesOfLen(0xEE, 32)), ErrHashMismatch) } func TestHardRollbackPebbleHashVaultRejectsMissingDir(t *testing.T) { diff --git a/sei-db/state_db/sc/hashvault/pebble_hashvault_test.go b/sei-db/state_db/sc/hashvault/pebble_hashvault_test.go index d19ef80d78..0a24389c68 100644 --- a/sei-db/state_db/sc/hashvault/pebble_hashvault_test.go +++ b/sei-db/state_db/sc/hashvault/pebble_hashvault_test.go @@ -17,6 +17,7 @@ func newTestPebbleVault(t *testing.T, configMutators ...func(*HashVaultConfig)) t.Helper() cfg := DefaultHashVaultConfig() cfg.DataDir = filepath.Join(t.TempDir(), "vault") + cfg.HaltOnMismatch = true for _, m := range configMutators { m(&cfg) } @@ -36,6 +37,7 @@ func reopenTestPebbleVault(t *testing.T, v *PebbleHashVault) *PebbleHashVault { require.NoError(t, v.Close(context.Background())) cfg := DefaultHashVaultConfig() cfg.DataDir = dir + cfg.HaltOnMismatch = true reopened, err := NewUnsafePebbleHashVault(context.Background(), cfg) require.NoError(t, err) t.Cleanup(func() { diff --git a/sei-db/state_db/sc/memiavl/hashlog_test.go b/sei-db/state_db/sc/memiavl/hashlog_test.go index 8d33c9703c..d6465f58ef 100644 --- a/sei-db/state_db/sc/memiavl/hashlog_test.go +++ b/sei-db/state_db/sc/memiavl/hashlog_test.go @@ -39,7 +39,7 @@ func (c *captureLogger) ReportChangeset(uint64, []*proto.NamedChangeSet) {} // HashListener is unused here: memIAVL reports its hashes synchronously through RecordHashes, and a // listener is for the store that publishes hashes asynchronously. -func (c *captureLogger) HashListener(context.Context, int64, *lthash.BlockHash) error { return nil } +func (c *captureLogger) HashListener(context.Context, uint64, *lthash.BlockHash) error { return nil } func (c *captureLogger) Close() error { return nil } diff --git a/sei-tendermint/config/autobahn_toml_test.go b/sei-tendermint/config/autobahn_toml_test.go index a62256d941..8f530f47ea 100644 --- a/sei-tendermint/config/autobahn_toml_test.go +++ b/sei-tendermint/config/autobahn_toml_test.go @@ -18,7 +18,7 @@ import ( // TestAutobahnKeysParseFromTopLevel guards against the trap where TOML keys // authored after a [section] header get silently nested under that section. -// AutobahnConfigFile and HashVaultDisabledUnsafe are top-level fields on +// AutobahnConfigFile and the hash vault settings are top-level fields on // Config, so they must appear before any [section] header in the on-disk file. func TestAutobahnKeysParseFromTopLevel(t *testing.T) { viper.Reset() @@ -26,7 +26,8 @@ func TestAutobahnKeysParseFromTopLevel(t *testing.T) { const content = ` autobahn-config-file = "/etc/sei/autobahn.json" -hash-vault-disabled-unsafe = true +hash-vault-halt-on-mismatch = true +hash-vault-empty-rollback-blocks = 7 [rpc] laddr = "tcp://127.0.0.1:26657" @@ -40,7 +41,8 @@ laddr = "tcp://127.0.0.1:26657" cfg, err := commands.ParseConfig(tmconfig.DefaultConfig()) require.NoError(t, err) require.Equal(t, "/etc/sei/autobahn.json", cfg.AutobahnConfigFile) - require.True(t, cfg.HashVaultDisabledUnsafe) + require.True(t, cfg.HashVaultHaltOnMismatch) + require.Equal(t, uint64(7), cfg.HashVaultEmptyRollbackBlocks) } // TestAutobahnKeysIgnoredUnderSectionHeader documents what breaks if the @@ -54,7 +56,8 @@ func TestAutobahnKeysIgnoredUnderSectionHeader(t *testing.T) { const content = ` [self-remediation] autobahn-config-file = "/etc/sei/autobahn.json" -hash-vault-disabled-unsafe = true +hash-vault-halt-on-mismatch = true +hash-vault-empty-rollback-blocks = 7 ` configPath := filepath.Join(t.TempDir(), "config.toml") require.NoError(t, os.WriteFile(configPath, []byte(content), 0600)) @@ -67,7 +70,8 @@ hash-vault-disabled-unsafe = true // The field ends up empty — viper saw self-remediation.autobahn-config-file // instead of the top-level key mapstructure was looking for. require.Empty(t, cfg.AutobahnConfigFile) - require.False(t, cfg.HashVaultDisabledUnsafe) + require.False(t, cfg.HashVaultHaltOnMismatch) + require.Equal(t, uint64(1000), cfg.HashVaultEmptyRollbackBlocks) } // TestRenderedTemplateAutobahnKeysAtTopLevel verifies that the freshly @@ -83,7 +87,9 @@ func TestRenderedTemplateAutobahnKeysAtTopLevel(t *testing.T) { require.NoError(t, err) rendered := string(data) - for _, key := range []string{"autobahn-config-file", "hash-vault-disabled-unsafe"} { + for _, key := range []string{ + "autobahn-config-file", "hash-vault-halt-on-mismatch", "hash-vault-empty-rollback-blocks", + } { keyIdx := strings.Index(rendered, key) require.NotEqual(t, -1, keyIdx, "key %q must appear in rendered template", key) // Find the nearest [section] header above keyIdx, if any. diff --git a/sei-tendermint/config/config.go b/sei-tendermint/config/config.go index bba9ae5920..dc8d93fecc 100644 --- a/sei-tendermint/config/config.go +++ b/sei-tendermint/config/config.go @@ -12,6 +12,7 @@ import ( "time" "github.com/sei-protocol/sei-chain/ratelimiter" + "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" mempoolcfg "github.com/sei-protocol/sei-chain/sei-tendermint/internal/mempool" tmos "github.com/sei-protocol/sei-chain/sei-tendermint/libs/os" "github.com/sei-protocol/sei-chain/sei-tendermint/libs/utils" @@ -87,26 +88,31 @@ type Config struct { // if mode disagrees with address-book membership. AutobahnConfigFile string `mapstructure:"autobahn-config-file"` - // HashVaultDisabledUnsafe disables the app-hash equivocation guard (HashVault). The vault is - // on by default (false). Setting this to true is an explicit, last-resort operator decision to - // run WITHOUT equivocation protection; the node logs loudly that it is unsafe. - HashVaultDisabledUnsafe bool `mapstructure:"hash-vault-disabled-unsafe"` + // HashVaultHaltOnMismatch selects what an Autobahn node does when the hash vault sees a state hash + // differ from the one it recorded for the same block: halt when true, or log an error and replace the + // recorded hashes when false. + HashVaultHaltOnMismatch bool `mapstructure:"hash-vault-halt-on-mismatch"` + + // HashVaultEmptyRollbackBlocks is how many blocks an Autobahn node rewinds and replays when it starts + // over an empty hash vault, to refill it. + HashVaultEmptyRollbackBlocks uint64 `mapstructure:"hash-vault-empty-rollback-blocks"` } // DefaultConfig returns a default configuration for a Tendermint node func DefaultConfig() *Config { return &Config{ - BaseConfig: DefaultBaseConfig(), - RPC: DefaultRPCConfig(), - P2P: DefaultP2PConfig(), - Mempool: DefaultMempoolConfig(), - StateSync: DefaultStateSyncConfig(), - Consensus: DefaultConsensusConfig(), - TxIndex: DefaultTxIndexConfig(), - Instrumentation: DefaultInstrumentationConfig(), - PrivValidator: DefaultPrivValidatorConfig(), - SelfRemediation: DefaultSelfRemediationConfig(), - HashVaultDisabledUnsafe: false, + BaseConfig: DefaultBaseConfig(), + RPC: DefaultRPCConfig(), + P2P: DefaultP2PConfig(), + Mempool: DefaultMempoolConfig(), + StateSync: DefaultStateSyncConfig(), + Consensus: DefaultConsensusConfig(), + TxIndex: DefaultTxIndexConfig(), + Instrumentation: DefaultInstrumentationConfig(), + PrivValidator: DefaultPrivValidatorConfig(), + SelfRemediation: DefaultSelfRemediationConfig(), + HashVaultHaltOnMismatch: hashvault.DefaultHashVaultConfig().HaltOnMismatch, + HashVaultEmptyRollbackBlocks: hashvault.DefaultHashVaultConfig().EmptyVaultRollbackBlocks, } } diff --git a/sei-tendermint/config/config_fuzz_test.go b/sei-tendermint/config/config_fuzz_test.go index aedd63e499..bb0a1f7ccb 100644 --- a/sei-tendermint/config/config_fuzz_test.go +++ b/sei-tendermint/config/config_fuzz_test.go @@ -118,10 +118,10 @@ func TestValidateBasicDistinguishesAnAbsentModeFromAnUnknownOne(t *testing.T) { } } -// FuzzRootScopeKeysRequireRootScope pins the placement trap on the two root-scope +// FuzzRootScopeKeysRequireRootScope pins the placement trap on the root-scope // keys. // -// autobahn-config-file and hash-vault-disabled-unsafe are declared at the top level +// autobahn-config-file and hash-vault-halt-on-mismatch are declared at the top level // of the Config struct, so in TOML they must appear before any [section] header. // Written after one they become that section's key — p2p.autobahn-config-file — // which nothing reads, and the node starts with the subsystem the operator meant to @@ -152,7 +152,7 @@ func FuzzRootScopeKeysRequireRootScope(f *testing.F) { doc.WriteString("[p2p]\n") } doc.WriteString("autobahn-config-file = \"" + path + "\"\n") - doc.WriteString("hash-vault-disabled-unsafe = true\n") + doc.WriteString("hash-vault-halt-on-mismatch = true\n") } conf, err := unmarshalConfigTOML(t, doc.String()) @@ -161,19 +161,19 @@ func FuzzRootScopeKeysRequireRootScope(f *testing.F) { } wantPath := "" - wantDisabled := false + wantHalt := false if present && !underSection { wantPath = path - wantDisabled = true + wantHalt = true } if conf.AutobahnConfigFile != wantPath { t.Fatalf("autobahn-config-file resolved to %q, want %q (present=%v underSection=%v); "+ "root-scope keys are only read before the first section header", conf.AutobahnConfigFile, wantPath, present, underSection) } - if conf.HashVaultDisabledUnsafe != wantDisabled { - t.Fatalf("hash-vault-disabled-unsafe resolved to %v, want %v (present=%v underSection=%v)", - conf.HashVaultDisabledUnsafe, wantDisabled, present, underSection) + if conf.HashVaultHaltOnMismatch != wantHalt { + t.Fatalf("hash-vault-halt-on-mismatch resolved to %v, want %v (present=%v underSection=%v)", + conf.HashVaultHaltOnMismatch, wantHalt, present, underSection) } }) } @@ -295,8 +295,8 @@ func TestAutobahnPointerAbsenceDisablesTheSubsystem(t *testing.T) { if conf.AutobahnConfigFile != "" { t.Fatalf("the default autobahn pointer must be empty, got %q", conf.AutobahnConfigFile) } - if conf.HashVaultDisabledUnsafe { - t.Fatal("the default must leave the app-hash equivocation guard enabled") + if conf.HashVaultHaltOnMismatch { + t.Fatal("the default must log a hash vault mismatch rather than halt the node") } } diff --git a/sei-tendermint/config/toml.go b/sei-tendermint/config/toml.go index 1d054cd4ad..6285c68e0a 100644 --- a/sei-tendermint/config/toml.go +++ b/sei-tendermint/config/toml.go @@ -163,19 +163,20 @@ mock-app = {{ .BaseConfig.MockApp }} # would otherwise nest it under the immediately preceding section. autobahn-config-file = "{{ .AutobahnConfigFile }}" -# hash-vault-disabled-unsafe disables the app-hash equivocation guard (HashVault). -# DO NOT set this to true unless you are knowingly running an UNSAFE node as a last-resort -# recovery measure. A node with this enabled has NO protection against changing its mind about -# a committed block's app hash, and will log error-level warnings on every startup. +# hash-vault-halt-on-mismatch selects what an Autobahn node does when the hash vault sees a +# block's state hash differ from the one it recorded for that block. When true, the node halts; +# DO NOT RESTART WITHOUT HUMAN INVESTIGATION. When false, the node logs an error, discards the +# recorded hashes from that block up, and records the new hash in their place. # -# It is safer to leave HashVault enabled: if you hit a startup panic, first remove the HashVault -# files as instructed in the panic message and let the node run. Only disable HashVault if you are -# very sure the stored hashes are totally wrong and you keep hitting the same panic on new blocks. +# hash-vault-empty-rollback-blocks is how many blocks an Autobahn node rewinds and replays when it +# starts over an empty hash vault, so that the vault holds the hashes of recent blocks again. The +# rewind is shortened to what the node's snapshots and state WAL can reach. # -# Placed here (as a top-level key, before any [section] header) so the TOML parser sees it at -# root scope where mapstructure expects it — viper would otherwise nest it under the +# Placed here (as top-level keys, before any [section] header) so the TOML parser sees them at +# root scope where mapstructure expects them — viper would otherwise nest them under the # immediately preceding section. -hash-vault-disabled-unsafe = {{ .HashVaultDisabledUnsafe }} +hash-vault-halt-on-mismatch = {{ .HashVaultHaltOnMismatch }} +hash-vault-empty-rollback-blocks = {{ .HashVaultEmptyRollbackBlocks }} ####################################################################### ### Advanced Configuration Options ### diff --git a/sei-tendermint/internal/p2p/giga_router.go b/sei-tendermint/internal/p2p/giga_router.go index e24e94c6e2..5e7368b56c 100644 --- a/sei-tendermint/internal/p2p/giga_router.go +++ b/sei-tendermint/internal/p2p/giga_router.go @@ -34,7 +34,7 @@ type GigaRouterCommonConfig struct { ValidatorAddrs map[atypes.PublicKey]GigaNodeAddr GenDoc *types.GenesisDoc // PersistentStateDir is the absolute on-disk root for durable state - // (BlockDB, hashvault, epoch snapshots, and the validator's consensus + // (BlockDB, epoch snapshots, and the validator's consensus // persister in sibling subdirs). Required and must already exist. PersistentStateDir string // App is the ABCI proxy executeBlock drives. NewGigaValidatorRouter @@ -45,11 +45,6 @@ type GigaRouterCommonConfig struct { // peers. 0 rejects all; positive caps at n, up to maxInboundFullnodePeers. MaxInboundFullnodePeers int - // HashVaultDisabledUnsafe disables the app-hash equivocation guard (HashVault). The guard is - // on by default (false); the GigaRouter builds and owns it (see runExecute). Setting this to true - // is an explicit, last-resort operator decision to run WITHOUT equivocation protection. - HashVaultDisabledUnsafe bool - // Whether validator should proxy txs which do not belong to the local node. EnableEvmProxy bool } diff --git a/sei-tendermint/internal/p2p/giga_router_common.go b/sei-tendermint/internal/p2p/giga_router_common.go index 3105b21ff5..472f6e14a7 100644 --- a/sei-tendermint/internal/p2p/giga_router_common.go +++ b/sei-tendermint/internal/p2p/giga_router_common.go @@ -5,14 +5,12 @@ import ( "errors" "fmt" "net/url" - "path/filepath" "slices" "sort" "sync/atomic" ethrpc "github.com/ethereum/go-ethereum/rpc" gigametrics "github.com/sei-protocol/sei-chain/giga/metrics" - "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" abci "github.com/sei-protocol/sei-chain/sei-tendermint/abci/types" atypes "github.com/sei-protocol/sei-chain/sei-tendermint/autobahn/types" "github.com/sei-protocol/sei-chain/sei-tendermint/crypto" @@ -225,7 +223,7 @@ func (r *gigaRouterCommon) translateGlobalBlock(gb *atypes.GlobalBlock) *coretyp } } -func (r *gigaRouterCommon) executeBlock(ctx context.Context, b *atypes.GlobalBlock, hashVault hashvault.HashVault) (*abci.ResponseCommit, error) { +func (r *gigaRouterCommon) executeBlock(ctx context.Context, b *atypes.GlobalBlock) (*abci.ResponseCommit, error) { app := r.app hash := b.Header.Hash() var proposerAddress types.Address @@ -263,16 +261,6 @@ func (r *gigaRouterCommon) executeBlock(ctx context.Context, b *atypes.GlobalBlo gigametrics.SetPhase(gigametrics.PhaseStorage) - // Commit this height's app hash to the equivocation guard before persisting app state, so the - // vault always records our commitment to a height before the state it implies is committed (and - // before the hash is proposed for AppQC voting via PushAppHash below). On restart the block is - // re-executed and the identical hash is re-committed idempotently. A returned error is a benign - // shutdown cancellation; genuine faults panic inside the call. See commitAppHashToVault. - gigametrics.SetStoragePhase(gigametrics.StoragePhaseVaultCommit) - if err := commitAppHashToVault(ctx, hashVault, b.GlobalNumber, resp.AppHash); err != nil { - return nil, err - } - gigametrics.SetStoragePhase(gigametrics.StoragePhaseAppCommit) commitResp, err := app.Commit(ctx) if err != nil { @@ -330,75 +318,7 @@ func finalizeBlockGasUsed(resp *abci.ResponseFinalizeBlock) int64 { return total } -// buildHashVault constructs the app-hash equivocation guard runExecute owns. By default it -// returns a durable Pebble-backed vault rooted at /hashvault, alongside the -// other Autobahn on-disk state. It returns a no-op vault (no protection) when the operator -// explicitly sets HashVaultDisabledUnsafe, logged loudly. -func buildHashVault(ctx context.Context, cfg *GigaRouterCommonConfig) (hashvault.HashVault, error) { - if cfg.HashVaultDisabledUnsafe { - logger.Error("################################################################") - logger.Error("# HASHVAULT DISABLED (hash-vault-disabled-unsafe=true). #") - logger.Error("# This node has NO app-hash equivocation protection and is #") - logger.Error("# running in an UNSAFE configuration. Re-enable as soon as the #") - logger.Error("# underlying issue is resolved. #") - logger.Error("################################################################") - return hashvault.NewNoopHashVault(), nil - } - hvCfg := hashvault.DefaultHashVaultConfig() - hvCfg.DataDir = filepath.Join(cfg.PersistentStateDir, "hashvault") - return hashvault.NewPebbleHashVault(ctx, hvCfg) -} - -// commitAppHashToVault records the app hash for the given height in the equivocation guard and halts -// the node on any error. Every executed height is guarded, so a node can never commit to two -// different app hashes for the same height without deliberate human intervention. -func commitAppHashToVault( - ctx context.Context, - vault hashvault.HashVault, - height atypes.GlobalBlockNumber, - hash []byte, -) error { - err := vault.CommitToHash(ctx, uint64(height), hash) - if err == nil { - return nil - } - if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) { - logger.Info("HashVault commit aborted by context cancellation during shutdown; not recording hash", - "height", height, "err", err) - return fmt.Errorf("hashvault CommitToHash aborted at height %d: %w", height, err) - } - // Build the fatal message once and use it for both the log and the panic. The logger writes - // directly (no in-process buffer), but a hard crash could still drop the final line, so the - // panic string carries the full guidance too — panic output is what an operator sees first. - var msg string - if errors.Is(err, hashvault.ErrHashMismatch) { - // The HashVault has already logged the conflicting hashes, its data directory, and the - // bypass/slashing guidance immediately before returning this error; don't duplicate it. - msg = fmt.Sprintf("FATAL: HashVault detected an app-hash equivocation at height %d; halting. "+ - "See the preceding HashVault error for the conflicting hashes and recovery steps. "+ - "DO NOT RESTART WITHOUT HUMAN INTERVENTION.", height) - } else { - msg = fmt.Sprintf("FATAL: HashVault could not commit the app hash at height %d (operational "+ - "error, not a confirmed equivocation): %v. hashHex=%x. Halting.", height, err, hash) - } - logger.Error(msg) - panic(msg) -} - func (r *gigaRouterCommon) runExecute(ctx context.Context) error { - // runExecute is the single block-execution loop spawned by both the validator and fullnode Run - // methods, so it owns the equivocation guard for both roles: build it here (set before the first - // executeBlock, the only other reader) and close it on exit. - hashVault, err := buildHashVault(ctx, r.cfg) - if err != nil { - return fmt.Errorf("buildHashVault(): %w", err) - } - defer func() { - if err := hashVault.Close(context.Background()); err != nil { - logger.Error("failed to close hashvault", "err", err) - } - }() - app := r.app info := app.Info() @@ -446,15 +366,6 @@ func (r *gigaRouterCommon) runExecute(ctx context.Context) error { // TODO: for consistency we should also set proposerAddress here, // but this is a placeholder solution so maybe we don't care. }).ToProto()) - // Re-commit the last finalized block's app hash to the equivocation guard before re-proposing it - // for AppQC voting (PushAppHash below), mirroring executeBlock's commit-before-PushAppHash - // ordering. On a normal restart this idempotently matches the hash recorded when `last` was - // first executed; if the committed app state has diverged from what the vault recorded (e.g. an - // out-of-band rollback/restore), this halts the node instead of externalizing a conflicting - // hash. A returned error is a benign shutdown cancellation; genuine faults panic inside. - if err := commitAppHashToVault(ctx, hashVault, last, info.LastBlockAppHash); err != nil { - return err - } // Losing a prefix of appHashes on crash is fine: AppQC is reached // once everyone votes on apphashes of a suffix of finalized blocks. weights, err := committeeWeights(app.GetValidators()) @@ -473,7 +384,7 @@ func (r *gigaRouterCommon) runExecute(ctx context.Context) error { return fmt.Errorf("r.data.GlobalBlock(%v): %w", n, err) } gigametrics.SetPhase(gigametrics.PhaseExecution) - commitResp, err := r.executeBlock(ctx, b, hashVault) + commitResp, err := r.executeBlock(ctx, b) if err != nil { return fmt.Errorf("r.executeBlock(%v): %w", n, err) } @@ -485,18 +396,6 @@ func (r *gigaRouterCommon) runExecute(ctx context.Context) error { if err := r.data.PruneBefore(pruneBefore); err != nil { return fmt.Errorf("r.data.PruneBefore(%v): %w", pruneBefore, err) } - // Align the vault's retention with the data layer's prune boundary. - gigametrics.SetStoragePhase(gigametrics.StoragePhasePruneVault) - if err := hashVault.Prune(ctx, uint64(pruneBefore)); err != nil { - // A canceled context just means we're shutting down between a successful executeBlock - // and this prune; that's benign, not a prune failure, so don't alarm operators. - if errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) { - logger.Info("hashvault prune aborted by context cancellation during shutdown", - "prune_before", pruneBefore, "err", err) - } else { - logger.Error("failed to prune hashvault", "prune_before", pruneBefore, "err", err) - } - } gigametrics.EndStoragePhase() } } diff --git a/sei-tendermint/internal/p2p/giga_router_common_test.go b/sei-tendermint/internal/p2p/giga_router_common_test.go index b9797ada82..769e76df56 100644 --- a/sei-tendermint/internal/p2p/giga_router_common_test.go +++ b/sei-tendermint/internal/p2p/giga_router_common_test.go @@ -11,7 +11,6 @@ import ( ethrpc "github.com/ethereum/go-ethereum/rpc" "github.com/sei-protocol/sei-chain/sei-db/ledger_db/block/memblock" - "github.com/sei-protocol/sei-chain/sei-db/state_db/sc/hashvault" abci "github.com/sei-protocol/sei-chain/sei-tendermint/abci/types" "github.com/sei-protocol/sei-chain/sei-tendermint/autobahn/blockstore" atypes "github.com/sei-protocol/sei-chain/sei-tendermint/autobahn/types" @@ -52,54 +51,6 @@ func (a *fixedHeightApp) Info() *abci.ResponseInfo { return &abci.ResponseInfo{LastBlockHeight: a.height} } -// newSeededVault returns a durable Pebble vault rooted in a temp dir with hash committed at height. -func newSeededVault(t *testing.T, height atypes.GlobalBlockNumber, hash []byte) hashvault.HashVault { - t.Helper() - cfg := hashvault.DefaultHashVaultConfig() - cfg.DataDir = t.TempDir() - v, err := hashvault.NewUnsafePebbleHashVault(context.Background(), cfg) - require.NoError(t, err) - t.Cleanup(func() { _ = v.Close(context.Background()) }) - require.NoError(t, v.CommitToHash(context.Background(), uint64(height), hash)) - return v -} - -// TestCommitHashToVault covers the safety contract the restart path in runExecute relies on: -// an idempotent match returns nil, a divergent hash halts the node (panic), and a canceled -// context returns an error without halting. -func TestCommitHashToVault(t *testing.T) { - const height atypes.GlobalBlockNumber = 42 - h1 := make([]byte, hashvault.BlockHashSize) - for i := range h1 { - h1[i] = 0xAA - } - h2 := make([]byte, hashvault.BlockHashSize) - for i := range h2 { - h2[i] = 0xBB - } - - t.Run("matching hash is idempotent", func(t *testing.T) { - vault := newSeededVault(t, height, h1) - require.NoError(t, commitAppHashToVault(context.Background(), vault, height, h1)) - }) - - t.Run("divergent hash halts the node", func(t *testing.T) { - vault := newSeededVault(t, height, h1) - require.Panics(t, func() { - _ = commitAppHashToVault(context.Background(), vault, height, h2) - }) - }) - - t.Run("canceled context returns error without halting", func(t *testing.T) { - vault := newSeededVault(t, height, h1) - ctx, cancel := context.WithCancel(context.Background()) - cancel() - // Must not panic: a canceled context is a benign shutdown, not an equivocation. - err := commitAppHashToVault(ctx, vault, height, h2) - require.Error(t, err) - }) -} - func TestFinalizeBlockGasUsed(t *testing.T) { resp := &abci.ResponseFinalizeBlock{ TxResults: []*abci.ExecTxResult{ diff --git a/sei-tendermint/node/public.go b/sei-tendermint/node/public.go index df74e01712..92bf0df44d 100644 --- a/sei-tendermint/node/public.go +++ b/sei-tendermint/node/public.go @@ -176,7 +176,7 @@ func prepareApplication( if err != nil { return nil, noStorage, fmt.Errorf("load Autobahn committee: %w", err) } - manager, err := openAutobahnStorageManager(ctx, conf.RootDir, conf.Mode, fc, giga.Storage) + manager, err := openAutobahnStorageManager(ctx, conf, fc, giga.Storage) if err != nil { return nil, noStorage, fmt.Errorf("open Autobahn storage: %w", err) } diff --git a/sei-tendermint/node/setup.go b/sei-tendermint/node/setup.go index 13205c38e0..fa67d5fe81 100644 --- a/sei-tendermint/node/setup.go +++ b/sei-tendermint/node/setup.go @@ -324,9 +324,6 @@ func buildGigaRouter( return nil, err } valCfg.PersistentStateDir = stateDir - // The GigaRouter builds and owns the equivocation guard itself; just pass the operator's - // enable/disable decision through as plain config. - valCfg.HashVaultDisabledUnsafe = cfg.HashVaultDisabledUnsafe logger.Info("Autobahn: starting as validator", "validators", len(valCfg.ValidatorAddrs)) dataState, err := p2p.BuildDataState(&valCfg.GigaRouterCommonConfig, blockStore) if err != nil { @@ -347,9 +344,6 @@ func buildGigaRouter( return nil, err } fnCfg.PersistentStateDir = stateDir - // The GigaRouter builds and owns the equivocation guard itself; just pass the operator's - // enable/disable decision through as plain config. - fnCfg.HashVaultDisabledUnsafe = cfg.HashVaultDisabledUnsafe logger.Info("Autobahn: starting as fullnode", "mode", cfg.Mode, "validators", len(validatorAddrs)) dataState, err := p2p.BuildDataState(fnCfg, blockStore) if err != nil { @@ -379,19 +373,18 @@ func resolvePersistentStateDir(rootDir, dir string) (string, error) { } // openAutobahnStorageManager opens the Giga storage set in Autobahn's -// persistent-state directory, laid out for nodeMode unless storage pins a mode. +// persistent-state directory, laid out for conf.Mode unless storage pins a mode. func openAutobahnStorageManager( ctx context.Context, - rootDir string, - nodeMode string, + conf *config.Config, fc *config.AutobahnFileConfig, storage gigaconfig.StorageConfig, ) (*bootstrap.GigaStorageManager, error) { - directory, err := resolvePersistentStateDir(rootDir, fc.PersistentStateDir) + directory, err := resolvePersistentStateDir(conf.RootDir, fc.PersistentStateDir) if err != nil { return nil, err } - storageConfig, err := buildGigaStorageConfig(directory, nodeMode, storage) + storageConfig, err := buildGigaStorageConfig(directory, conf.Mode, storage) if err != nil { return nil, fmt.Errorf("build Autobahn storage config: %w", err) } @@ -400,6 +393,8 @@ func openAutobahnStorageManager( return nil, fmt.Errorf("build Autobahn block DB config: %w", err) } storageConfig.BlockDBConfig = &blockConfig + storageConfig.HashVaultConfig.HaltOnMismatch = conf.HashVaultHaltOnMismatch + storageConfig.HashVaultConfig.EmptyVaultRollbackBlocks = conf.HashVaultEmptyRollbackBlocks return bootstrap.NewGigaStorageManager(ctx, storageConfig) }