Recovery tail-truncation had two compounding bugs in wal/recover.go:
C5: truncateSegment only called os.Truncate. Missing per design §3.2
line 787-794:
- Step 2: fsync the truncated segment
- Step 3: delete empty trailing segments
- Step 4: fsync WAL directory
And all errors were swallowed into result.TruncateError with recovery
still returning success, violating design line 799: "若 ftruncate、
segment fsync、空 segment 删除或 WAL directory fsync 任一步失败,
recovery 必须报错,DB 不得进入可写状态".
H8: findValidOffset only checked physical record CRCs, ignoring the
FragmentCollector state machine. For a tail of First + Middle*
without Last, it returned the offset AFTER the last Middle fragment
instead of the last COMPLETE batch end. Result: residual half-batch
fragments caused repeated tail-corruption reports on every restart.
Changes:
- wal/recover.go:
- Add findLastCompleteBatchEnd: batch-aware offset finder using
FragmentCollector state machine. Handles block-boundary padding
correctly (continue across full-block padding, return on short-block).
- Add truncateAndPersist: 4-step protocol (ftruncate + fsync segment +
delete empty trailing + fsync dir). Any step failure is fatal.
- Add segmentFsyncFn (package-level var for test injection, same
pattern as C6's dirFsyncFn).
- Refactor Recover failure path: use new functions, hard-error on
truncation persist failure (was: swallow to TruncateError).
- TruncateError field semantics: informational only ("tail corruption
was detected and repair attempted"). Persist failures return error.
- Delete findValidOffset and truncateSegment (replaced).
- wal/recover_offset_test.go (new): 8 unit tests for
findLastCompleteBatchEnd covering clean/partial-tail/no-batch/
physical-corruption/partial-only/block-boundary-padding/non-zero-tail/
zero-tail cases. 5 unit tests for truncateAndPersist covering success/
ftruncate-fail/dir-fsync-fail/segment-fsync-fail/retry-after-failure.
- wal/recover_test.go: add TestRecoverPartialFragmentTailIdempotent
(H8 e2e regression: truncation point must be at last complete batch),
TestRecoverTruncationFailureFailsRecovery (C5 e2e regression: any
step failure fails Recover), TestRecoverInvalidBatchNotTruncatable
(design line 778-781: invalid batch content hard-fails, NOT truncatable).
Injection note: segmentFsyncFn and dirFsyncFn (from C6) are package-level
vars; tests that override either must not use t.Parallel().
Verified: each new test fails on pre-fix code by logical analysis and
passes after the fix. Full suite green including go test -race ./... .
Phase 1 simplification: emptyTrailingSegments is always nil in Phase 1
(truncated segment is always segments[last]). The parameter is kept in
truncateAndPersist's signature for forward compatibility with the C4 fix.
Audit context: docs/audit-3.2.md C5 and H8 (H8 Oracle-verified bg_ef425776).
253 lines
8.1 KiB
Go
253 lines
8.1 KiB
Go
package wal
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
|
|
"github.com/dailz/go-kv/manifest"
|
|
)
|
|
|
|
// RecoveryResult holds the outcome of a WAL recovery pass.
|
|
type RecoveryResult struct {
|
|
NextSequence uint64
|
|
NextSegmentID uint64
|
|
ReplayedEntries int
|
|
Truncated bool
|
|
// TruncateError is informational only: non-nil means "tail corruption
|
|
// was found and repair was attempted". It does NOT report persistence
|
|
// failures — those cause Recover to return an error instead.
|
|
TruncateError error
|
|
}
|
|
|
|
// Recover performs a full WAL recovery: reads the recovery checkpoint from
|
|
// MANIFEST, scans segments, replays entries, and persists tail truncation
|
|
// per design §3.2 line 787-800.
|
|
func Recover(dir string, replayer BatchReplayer) (*RecoveryResult, error) {
|
|
if replayer == nil {
|
|
return nil, fmt.Errorf("wal: recover: replayer is nil")
|
|
}
|
|
|
|
// Step 1: Determine recovery segment ID from MANIFEST.
|
|
recoverySegmentID, err := resolveRecoverySegmentID(dir)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("wal: recover: resolve segment id: %w", err)
|
|
}
|
|
|
|
// Step 2: Scan and replay segments.
|
|
nextSequence, err := RecoverFromSegments(dir, recoverySegmentID, replayer)
|
|
if err != nil {
|
|
if !IsTailCorruption(err) {
|
|
return nil, fmt.Errorf("wal: recover: %w", err)
|
|
}
|
|
|
|
// Step 3: Tail corruption — truncate the last segment per design
|
|
// §3.2 line 787-800.
|
|
result := &RecoveryResult{
|
|
NextSequence: nextSequence,
|
|
Truncated: true,
|
|
TruncateError: err, // informational: tail corruption was detected
|
|
}
|
|
|
|
segments, scanErr := ScanSegments(dir, recoverySegmentID)
|
|
if scanErr != nil {
|
|
return nil, fmt.Errorf("wal: recover: scan after tail corruption: %w", scanErr)
|
|
}
|
|
if len(segments) > 0 {
|
|
result.NextSegmentID = segments[len(segments)-1].SegmentID + 1
|
|
} else {
|
|
result.NextSegmentID = recoverySegmentID
|
|
}
|
|
|
|
if len(segments) > 0 {
|
|
lastSeg := segments[len(segments)-1]
|
|
lastCompleteBatchEnd, findErr := findLastCompleteBatchEnd(lastSeg.FilePath)
|
|
if findErr != nil {
|
|
return nil, fmt.Errorf("wal: recover: find truncation offset: %w", findErr)
|
|
}
|
|
|
|
// Phase 1: truncated segment is always segments[last], no
|
|
// trailing empty segments to clean up. C4 fix will need to
|
|
// identify trailing empties based on the actually-corrupted
|
|
// segment's index, which is not the same as segments[last].
|
|
var emptyTrailing []string
|
|
|
|
if err := truncateAndPersist(lastSeg.FilePath, lastCompleteBatchEnd, dir, emptyTrailing); err != nil {
|
|
// Per design §3.2 line 799: DB must NOT enter writable state.
|
|
return nil, fmt.Errorf("wal: recover: persist tail truncation: %w", err)
|
|
}
|
|
}
|
|
|
|
result.ReplayedEntries = 0 // Phase 1: replayer interface doesn't expose count
|
|
|
|
// Per design §3.2 line 280, recovery repair must NOT update MANIFEST.
|
|
// The truncated WAL state is persisted via ftruncate + fsync segment
|
|
// + fsync dir (see truncateAndPersist). MANIFEST can only advance via
|
|
// checkpoint (MemTable flush) in future phases.
|
|
|
|
return result, nil
|
|
}
|
|
|
|
// Step 4: Successful recovery — compute result.
|
|
segments, scanErr := ScanSegments(dir, recoverySegmentID)
|
|
if scanErr != nil {
|
|
return nil, fmt.Errorf("wal: recover: scan after replay: %w", scanErr)
|
|
}
|
|
|
|
result := &RecoveryResult{
|
|
NextSequence: nextSequence,
|
|
NextSegmentID: recoverySegmentID,
|
|
Truncated: false,
|
|
}
|
|
if len(segments) > 0 {
|
|
result.NextSegmentID = segments[len(segments)-1].SegmentID + 1
|
|
}
|
|
|
|
// Per design §3.2 line 280, recovery must NOT update MANIFEST.
|
|
// RecoveryResult.NextSegmentID is in-memory only, consumed by DB.Open to
|
|
// seed the new WalWriter. MANIFEST stays at its pre-recovery value.
|
|
|
|
return result, nil
|
|
}
|
|
|
|
// resolveRecoverySegmentID returns the recovery start segment ID from MANIFEST.
|
|
// MANIFEST is the only authoritative source of recovery start per design §3.2
|
|
// line 600-06. CURRENT is a write-side hint and must NOT be used here.
|
|
func resolveRecoverySegmentID(dir string) (uint64, error) {
|
|
mf, err := manifest.Load(dir)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("load manifest: %w", err)
|
|
}
|
|
return mf.RecoverySegmentID, nil
|
|
}
|
|
|
|
// segmentFsyncFn is the package-level indirection for fsyncing a truncated
|
|
// segment file. Tests that override this must not use t.Parallel().
|
|
// Same pattern as dirFsyncFn (see wal/dir_fsync.go).
|
|
var segmentFsyncFn = segmentFsync
|
|
|
|
func segmentFsync(filePath string) error {
|
|
f, err := os.OpenFile(filePath, os.O_WRONLY, 0o644)
|
|
if err != nil {
|
|
return fmt.Errorf("open for fsync: %w", err)
|
|
}
|
|
defer f.Close()
|
|
if err := f.Sync(); err != nil {
|
|
return fmt.Errorf("fsync: %w", err)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// truncateAndPersist executes the 4-step tail-truncation protocol per
|
|
// design §3.2 line 787-794. Any step failure is fatal: per line 799,
|
|
// DB must NOT enter writable state if truncation cannot be persisted.
|
|
func truncateAndPersist(
|
|
filePath string,
|
|
lastCompleteBatchEnd int64,
|
|
dir string,
|
|
emptyTrailingSegments []string,
|
|
) error {
|
|
// Step 1: ftruncate
|
|
if lastCompleteBatchEnd < 0 {
|
|
return fmt.Errorf("wal: invalid truncate offset %d", lastCompleteBatchEnd)
|
|
}
|
|
if err := os.Truncate(filePath, lastCompleteBatchEnd); err != nil {
|
|
return fmt.Errorf("ftruncate %s to %d: %w", filePath, lastCompleteBatchEnd, err)
|
|
}
|
|
|
|
// Step 2: fsync the truncated segment
|
|
if err := segmentFsyncFn(filePath); err != nil {
|
|
return fmt.Errorf("fsync truncated segment: %w", err)
|
|
}
|
|
|
|
// Step 3: delete empty trailing segments
|
|
for _, segPath := range emptyTrailingSegments {
|
|
if err := os.Remove(segPath); err != nil {
|
|
return fmt.Errorf("remove empty segment %s: %w", segPath, err)
|
|
}
|
|
}
|
|
|
|
// Step 4: fsync WAL directory (reuses C6's dirFsyncFn)
|
|
if err := dirFsyncFn(dir); err != nil {
|
|
return fmt.Errorf("fsync WAL dir after truncation: %w", err)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// findLastCompleteBatchEnd walks the segment file, runs physical records
|
|
// through the FragmentCollector state machine, and returns the byte offset
|
|
// of the END of the last complete WAL Batch.
|
|
//
|
|
// This is the correct truncation target per design §3.2 line 786. A previous
|
|
// version (findValidOffset) only checked physical record CRCs, missing the
|
|
// case where a First + Middle* fragment chain has no Last (H8 bug): physical
|
|
// CRCs pass but no complete batch exists at that offset.
|
|
//
|
|
// Block-boundary handling: WAL format allows a full block to end with zero
|
|
// padding when the next record doesn't fit (see BlockWriter.paddingNeeded).
|
|
// This function CONTINUES to the next block on padding in a full block, and
|
|
// only RETURNS on padding in a short (final) block or actual corruption.
|
|
func findLastCompleteBatchEnd(filePath string) (int64, error) {
|
|
f, err := os.Open(filePath)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("open %s: %w", filePath, err)
|
|
}
|
|
defer f.Close()
|
|
|
|
if _, err := f.Seek(WalFileHeaderSize, 0); err != nil {
|
|
return 0, fmt.Errorf("seek past header: %w", err)
|
|
}
|
|
|
|
collector := NewFragmentCollector()
|
|
lastCompleteEnd := int64(WalFileHeaderSize)
|
|
blockStartOffset := int64(WalFileHeaderSize)
|
|
buf := make([]byte, WalBlockSize)
|
|
|
|
for {
|
|
n, readErr := f.Read(buf)
|
|
if n > 0 {
|
|
blockData := buf[:n]
|
|
isFullBlock := n == WalBlockSize && readErr == nil
|
|
pos := 0
|
|
for pos < len(blockData) {
|
|
remaining := len(blockData) - pos
|
|
|
|
if remaining < PhysicalRecordHeaderSize {
|
|
if isFullBlock {
|
|
break // padding in full block, continue to next block
|
|
}
|
|
return lastCompleteEnd, nil // tail padding in short block
|
|
}
|
|
|
|
if isAllZeros(blockData[pos : pos+PhysicalRecordHeaderSize]) {
|
|
if isFullBlock {
|
|
break // zero-led padding in full block, continue
|
|
}
|
|
return lastCompleteEnd, nil // tail padding
|
|
}
|
|
|
|
rec, consumed, err := DecodePhysicalRecord(blockData[pos:])
|
|
if err != nil {
|
|
return lastCompleteEnd, nil // physical corruption
|
|
}
|
|
if err := collector.Append(rec.Type, rec.Payload); err != nil {
|
|
return lastCompleteEnd, nil // fragment state machine rejected
|
|
}
|
|
|
|
pos += consumed
|
|
recordEndAbsolute := blockStartOffset + int64(pos)
|
|
|
|
if collector.IsComplete() {
|
|
lastCompleteEnd = recordEndAbsolute
|
|
collector.Reset()
|
|
}
|
|
}
|
|
blockStartOffset += int64(n)
|
|
}
|
|
if readErr != nil || n < WalBlockSize {
|
|
break
|
|
}
|
|
}
|
|
|
|
return lastCompleteEnd, nil
|
|
}
|