Files
seaweedfs/weed/storage/volume_io_error_test.go
T
Chris Lu 7c60407897 fix(volume): don't nuke local data on transient IO error (#9378) (#9382)
* fix(volume): don't nuke local data on transient IO error (#9378)

A single syscall.EIO from any read/write/delete set v.lastIoError, and
the next CollectHeartbeat then called Volume.Destroy on the replica —
removing the .dat/.idx/.vif/.sdx/.ldb/.rdb files. A brief NFS / fabric
/ controller blip hitting several replicas at once could cascade into
removal of the last healthy copy, with no recovery for non-tiered
volumes.

Now require IoErrorTolerance (3) consecutive EIOs before acting, and on
that threshold mark the volume read-only and stop announcing it to the
master so re-replication kicks in from healthy peers — never delete
the data files. The on-disk copy stays for operator inspection /
recovery.

* review: fix race, accounting, recovery, non-EIO streak break

Addressing PR #9382 review:

- Data race on lastIoError: guard lastIoError + lastIoErrorCount with a
  RWMutex and expose them through note/clear/get helpers so the
  heartbeat reader sees a consistent snapshot. Verified with -race.
- Collection-size accounting: when a volume is quarantined for sustained
  EIO, skip the entire per-volume bookkeeping (`continue`) instead of
  flipping shouldDeleteVolume — the old branch subtracted a size that
  was never added, dragging the collection gauge to zero / negative.
- Recoverability: MarkVolumeWritable now also calls clearIoError so an
  operator can rejoin a quarantined replica. The next failed op
  re-arms the streak if the disk is still bad.
- Non-EIO streak break: a non-EIO error (e.g. ENOSPC) now resets the
  consecutive-EIO counter, so a sequence EIO,EIO,ENOSPC,EIO is treated
  as a streak of one — the counter only tracks consecutive EIOs.

Reads already call checkReadWriteError (volume_read.go), so successful
reads also clear the streak — no change needed there.
2026-05-09 09:20:31 -07:00

128 lines
3.3 KiB
Go

package storage
import (
"errors"
"fmt"
"sync"
"syscall"
"testing"
)
func TestCheckReadWriteErrorTracksConsecutiveEIO(t *testing.T) {
v := &Volume{}
// each EIO bumps the counter.
for i := int32(1); i <= 5; i++ {
v.checkReadWriteError(fmt.Errorf("disk failed: %w", syscall.EIO))
_, count := v.getIoErrorState()
if count != i {
t.Fatalf("after %d EIO(s): counter = %d, want %d", i, count, i)
}
}
// a single success resets both fields.
v.checkReadWriteError(nil)
if err, count := v.getIoErrorState(); err != nil || count != 0 {
t.Fatalf("success did not reset state: err=%v count=%d", err, count)
}
}
func TestCheckReadWriteErrorNonEIOResetsStreak(t *testing.T) {
v := &Volume{}
// build up a 2-EIO streak.
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
if _, count := v.getIoErrorState(); count != 2 {
t.Fatalf("expected count=2 after two EIOs, got %d", count)
}
// a non-EIO error breaks the streak — only sustained EIOs are
// diagnostic of a failing disk.
v.checkReadWriteError(fmt.Errorf("other: %w", syscall.ENOSPC))
if err, count := v.getIoErrorState(); err != nil || count != 0 {
t.Fatalf("non-EIO did not reset streak: err=%v count=%d", err, count)
}
// a fresh EIO starts the streak from 1, not 3.
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
if _, count := v.getIoErrorState(); count != 1 {
t.Fatalf("EIO after non-EIO did not restart streak: count=%d, want 1", count)
}
}
func TestCheckReadWriteErrorIgnoresPlainError(t *testing.T) {
v := &Volume{}
// non-EIO error with no prior streak should be a no-op (count
// stays 0, no spurious lastIoError).
v.checkReadWriteError(errors.New("some other error"))
if err, count := v.getIoErrorState(); err != nil || count != 0 {
t.Fatalf("non-EIO with no prior streak set state: err=%v count=%d", err, count)
}
}
func TestIoErrorToleranceGate(t *testing.T) {
v := &Volume{}
// below tolerance: do not act.
for i := 0; i < IoErrorTolerance-1; i++ {
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
}
if _, count := v.getIoErrorState(); count >= IoErrorTolerance {
t.Fatalf("counter %d already crossed tolerance %d after %d errors",
count, IoErrorTolerance, IoErrorTolerance-1)
}
// one more crosses the threshold.
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
if _, count := v.getIoErrorState(); count < IoErrorTolerance {
t.Fatalf("counter %d below tolerance %d after %d errors",
count, IoErrorTolerance, IoErrorTolerance)
}
}
func TestIoErrorStateIsRaceFree(t *testing.T) {
// Drives both writers (checkReadWriteError) and a reader
// (getIoErrorState) concurrently; relies on `go test -race` to
// detect any unprotected access on lastIoError / lastIoErrorCount.
v := &Volume{}
var wg sync.WaitGroup
stop := make(chan struct{})
wg.Add(1)
go func() {
defer wg.Done()
for {
select {
case <-stop:
return
default:
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
}
}
}()
wg.Add(1)
go func() {
defer wg.Done()
for {
select {
case <-stop:
return
default:
v.checkReadWriteError(nil)
}
}
}()
wg.Add(1)
go func() {
defer wg.Done()
for i := 0; i < 1000; i++ {
v.getIoErrorState()
}
close(stop)
}()
wg.Wait()
}