mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* fix(volume): don't nuke local data on transient IO error (#9378) A single syscall.EIO from any read/write/delete set v.lastIoError, and the next CollectHeartbeat then called Volume.Destroy on the replica — removing the .dat/.idx/.vif/.sdx/.ldb/.rdb files. A brief NFS / fabric / controller blip hitting several replicas at once could cascade into removal of the last healthy copy, with no recovery for non-tiered volumes. Now require IoErrorTolerance (3) consecutive EIOs before acting, and on that threshold mark the volume read-only and stop announcing it to the master so re-replication kicks in from healthy peers — never delete the data files. The on-disk copy stays for operator inspection / recovery. * review: fix race, accounting, recovery, non-EIO streak break Addressing PR #9382 review: - Data race on lastIoError: guard lastIoError + lastIoErrorCount with a RWMutex and expose them through note/clear/get helpers so the heartbeat reader sees a consistent snapshot. Verified with -race. - Collection-size accounting: when a volume is quarantined for sustained EIO, skip the entire per-volume bookkeeping (`continue`) instead of flipping shouldDeleteVolume — the old branch subtracted a size that was never added, dragging the collection gauge to zero / negative. - Recoverability: MarkVolumeWritable now also calls clearIoError so an operator can rejoin a quarantined replica. The next failed op re-arms the streak if the disk is still bad. - Non-EIO streak break: a non-EIO error (e.g. ENOSPC) now resets the consecutive-EIO counter, so a sequence EIO,EIO,ENOSPC,EIO is treated as a streak of one — the counter only tracks consecutive EIOs. Reads already call checkReadWriteError (volume_read.go), so successful reads also clear the streak — no change needed there.
128 lines
3.3 KiB
Go
128 lines
3.3 KiB
Go
package storage
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"sync"
|
|
"syscall"
|
|
"testing"
|
|
)
|
|
|
|
func TestCheckReadWriteErrorTracksConsecutiveEIO(t *testing.T) {
|
|
v := &Volume{}
|
|
|
|
// each EIO bumps the counter.
|
|
for i := int32(1); i <= 5; i++ {
|
|
v.checkReadWriteError(fmt.Errorf("disk failed: %w", syscall.EIO))
|
|
_, count := v.getIoErrorState()
|
|
if count != i {
|
|
t.Fatalf("after %d EIO(s): counter = %d, want %d", i, count, i)
|
|
}
|
|
}
|
|
|
|
// a single success resets both fields.
|
|
v.checkReadWriteError(nil)
|
|
if err, count := v.getIoErrorState(); err != nil || count != 0 {
|
|
t.Fatalf("success did not reset state: err=%v count=%d", err, count)
|
|
}
|
|
}
|
|
|
|
func TestCheckReadWriteErrorNonEIOResetsStreak(t *testing.T) {
|
|
v := &Volume{}
|
|
|
|
// build up a 2-EIO streak.
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
if _, count := v.getIoErrorState(); count != 2 {
|
|
t.Fatalf("expected count=2 after two EIOs, got %d", count)
|
|
}
|
|
|
|
// a non-EIO error breaks the streak — only sustained EIOs are
|
|
// diagnostic of a failing disk.
|
|
v.checkReadWriteError(fmt.Errorf("other: %w", syscall.ENOSPC))
|
|
if err, count := v.getIoErrorState(); err != nil || count != 0 {
|
|
t.Fatalf("non-EIO did not reset streak: err=%v count=%d", err, count)
|
|
}
|
|
|
|
// a fresh EIO starts the streak from 1, not 3.
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
if _, count := v.getIoErrorState(); count != 1 {
|
|
t.Fatalf("EIO after non-EIO did not restart streak: count=%d, want 1", count)
|
|
}
|
|
}
|
|
|
|
func TestCheckReadWriteErrorIgnoresPlainError(t *testing.T) {
|
|
v := &Volume{}
|
|
|
|
// non-EIO error with no prior streak should be a no-op (count
|
|
// stays 0, no spurious lastIoError).
|
|
v.checkReadWriteError(errors.New("some other error"))
|
|
if err, count := v.getIoErrorState(); err != nil || count != 0 {
|
|
t.Fatalf("non-EIO with no prior streak set state: err=%v count=%d", err, count)
|
|
}
|
|
}
|
|
|
|
func TestIoErrorToleranceGate(t *testing.T) {
|
|
v := &Volume{}
|
|
|
|
// below tolerance: do not act.
|
|
for i := 0; i < IoErrorTolerance-1; i++ {
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
}
|
|
if _, count := v.getIoErrorState(); count >= IoErrorTolerance {
|
|
t.Fatalf("counter %d already crossed tolerance %d after %d errors",
|
|
count, IoErrorTolerance, IoErrorTolerance-1)
|
|
}
|
|
|
|
// one more crosses the threshold.
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
if _, count := v.getIoErrorState(); count < IoErrorTolerance {
|
|
t.Fatalf("counter %d below tolerance %d after %d errors",
|
|
count, IoErrorTolerance, IoErrorTolerance)
|
|
}
|
|
}
|
|
|
|
func TestIoErrorStateIsRaceFree(t *testing.T) {
|
|
// Drives both writers (checkReadWriteError) and a reader
|
|
// (getIoErrorState) concurrently; relies on `go test -race` to
|
|
// detect any unprotected access on lastIoError / lastIoErrorCount.
|
|
v := &Volume{}
|
|
|
|
var wg sync.WaitGroup
|
|
stop := make(chan struct{})
|
|
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
for {
|
|
select {
|
|
case <-stop:
|
|
return
|
|
default:
|
|
v.checkReadWriteError(fmt.Errorf("eio: %w", syscall.EIO))
|
|
}
|
|
}
|
|
}()
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
for {
|
|
select {
|
|
case <-stop:
|
|
return
|
|
default:
|
|
v.checkReadWriteError(nil)
|
|
}
|
|
}
|
|
}()
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
for i := 0; i < 1000; i++ {
|
|
v.getIoErrorState()
|
|
}
|
|
close(stop)
|
|
}()
|
|
wg.Wait()
|
|
}
|