Files
seaweedfs/weed/replication/sink/filersink/fetch_write_test.go
T
Jaehoon KimandClaude Fable 5 c17aebeecf fix(filer.backup): prevent silent backup data loss on transient not-found (#10295)
* fix(filer.backup): stop silently dropping events on transient not-found

Under a write burst, filer.backup could consume a metadata event (advancing
the persisted offset) without replicating the file, with no error logged:

1. filersink CreateEntry/UpdateEntry swallowed replicateChunks errors
   (glog.Warningf + return nil), so the offset advanced past entries that
   were never written.
2. The manifest-chunk branch of replicateChunks resolved via LookupFileId
   with no retry, unlike the data-chunk branch — transient lookup races
   dropped exactly the large manifest-backed files while small
   inline-content siblings landed.
3. isIgnorable404 matched "LookupFileId" / "volume id ... not found",
   misclassifying those races as genuine source 404s at the backup layer.

Fix: on a replicateChunks failure the filer sink now skips only when the
live source has moved past the replayed version (deleted or strictly-newer
mtime) — lossless, a later event carries the current content — and
propagates otherwise so the event is retried. The manifest resolve retries
transient errors like the data-chunk path. isIgnorable404 is narrowed to
genuine 404s; non-filer sinks and the initial-snapshot walk, which relied
on the broad match as their only lossless-skip valve, now make the same
live-source decision (filersink.SourceSupersedes) instead of retrying
forever on a permanently gone volume.

Tests cover propagation of unconfirmed lookup failures, the narrowed 404
classification, and the supersession guards.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(filer.backup): derive supersession path with directory fallback

eventSourceSuperseded built the source path from NewParentPath alone.
Legacy metadata events (persisted by older filers) carry an empty
NewParentPath, so the probe looked up "/<name>", read the miss as
"source gone", and skipped a live file on a transient lookup error —
the silent drop this change is meant to eliminate.

Derive the path via MetadataEventTargetFullPath (the same directory
fallback genProcessFunction uses) and cover both event shapes with
TestEventSupersessionProbe_PathDerivation.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(filersink): retry manifest resolve only for transient errors, bounded when unverifiable

The manifest-resolve retry stopped only when hasSourceNewerVersion proved
the source moved past the replayed version, which wedged the sink in two
cases: incremental sinks use dated target keys that cannot map back to a
source path (supersession never provable), and permanent resolve errors
(corrupt manifest data, bad file ids) fail forever while the source entry
stays live.

Gate the retry instead: keep retrying only transient errors (volume-lookup
races, network interruptions), stop after a few attempts when supersession
cannot be checked, and propagate everything else immediately so the
configured metadata error policy applies (-disableErrorRetry included).
Propagation is lossless: filer.backup's fallback decides with the event's
real source key, and both filer.backup and filer.sync re-deliver the event
(RetryForeverOnError) without advancing the offset.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(filer.backup): make error classifiers nil-safe

isIgnorable404, isSourceLookupError, and isTransientResolveError called
err.Error() without a nil guard. All current call sites pass a non-nil
error, but the guard is free and matches isRetryableNetworkError.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 01:47:46 -07:00

428 lines
14 KiB
Go

package filersink
import (
"errors"
"fmt"
"io"
"net/http"
"net/http/httptest"
"os"
"strings"
"sync/atomic"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/replication/source"
"github.com/seaweedfs/seaweedfs/weed/util"
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
)
func TestMain(m *testing.M) {
util_http.InitGlobalHttpClient()
os.Exit(m.Run())
}
func TestTargetPathToSourcePath(t *testing.T) {
tests := []struct {
name string
targetRoot string
sourceRoot string
targetPath string
incremental bool
wantPath util.FullPath
wantOK bool
}{
{
name: "basic mapping",
targetRoot: "/target",
sourceRoot: "/source",
targetPath: "/target/path/file.txt",
wantPath: "/source/path/file.txt",
wantOK: true,
},
{
// incremental keys carry a date prefix that can't be reversed; unmappable
name: "incremental sink is unmappable",
targetRoot: "/target",
sourceRoot: "/source",
targetPath: "/target/2026-06-09/path/file.txt",
incremental: true,
wantPath: "",
wantOK: false,
},
{
name: "trailing slash roots",
targetRoot: "/target/",
sourceRoot: "/source/",
targetPath: "/target/path/file.txt",
wantPath: "/source/path/file.txt",
wantOK: true,
},
{
name: "root target mapping",
targetRoot: "/",
sourceRoot: "/source",
targetPath: "/path/file.txt",
wantPath: "/source/path/file.txt",
wantOK: true,
},
{
name: "target root itself",
targetRoot: "/target",
sourceRoot: "/source",
targetPath: "/target",
wantPath: "/source",
wantOK: true,
},
{
name: "outside target root",
targetRoot: "/target",
sourceRoot: "/source",
targetPath: "/other/path/file.txt",
wantPath: "",
wantOK: false,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
fs := &FilerSink{
dir: tc.targetRoot,
isIncremental: tc.incremental,
filerSource: &source.FilerSource{
Dir: tc.sourceRoot,
},
}
gotPath, ok := fs.targetPathToSourcePath(tc.targetPath)
if ok != tc.wantOK {
t.Fatalf("ok mismatch: got %v, want %v", ok, tc.wantOK)
}
if gotPath != tc.wantPath {
t.Fatalf("path mismatch: got %q, want %q", gotPath, tc.wantPath)
}
})
}
}
// FilerSink must reject chunks whose received byte count disagrees with the
// source filer metadata, instead of silently writing 0-byte needles with the
// source size in the destination metadata.
func TestValidateReplicatedChunkSize(t *testing.T) {
const fid = "74,047d16a94aa581"
tests := []struct {
name string
expectedSize uint64
readSize int
wantErr bool
}{
{
name: "healthy",
expectedSize: 5171,
readSize: 5171,
wantErr: false,
},
{
name: "legitimately empty file",
expectedSize: 0,
readSize: 0,
wantErr: false,
},
{
name: "zero-byte read for non-empty source",
expectedSize: 5171,
readSize: 0,
wantErr: true,
},
{
name: "short read",
expectedSize: 5171,
readSize: 100,
wantErr: true,
},
{
name: "over-read (server returned more than metadata)",
expectedSize: 5171,
readSize: 8192,
wantErr: true,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
chunk := &filer_pb.FileChunk{FileId: fid, Size: tc.expectedSize}
gotErr := validateReplicatedReadSize(chunk, tc.readSize)
if tc.wantErr {
if gotErr == nil {
t.Fatalf("expected error, got nil (read=%d expected=%d)",
tc.readSize, tc.expectedSize)
}
if !errors.Is(gotErr, errChunkSizeMismatch) {
t.Fatalf("expected errChunkSizeMismatch, got %v", gotErr)
}
if !strings.Contains(gotErr.Error(), fid) {
t.Fatalf("error %q does not mention chunk id %q", gotErr, fid)
}
return
}
if gotErr != nil {
t.Fatalf("unexpected read-size error: %v", gotErr)
}
})
}
}
// End-to-end regression :
// a source volume that responds 200 OK with Content-Length: 0
// for a chunk that filer metadata claims is 5171 bytes must be rejected
// by fetchAndWrite with a (non-retriable) size mismatch error,
// instead of being silently propagated to the destination as a 0-byte needle.
func TestFetchAndWriteRejectsZeroByteSource(t *testing.T) {
const fid = "74,047d16a94aa581"
const expectedSize uint64 = 5171
// Shorten retry backoff so a fail-fast test that briefly enters the retry
// loop doesn't pay the production 1s+ wait. Scoped to this test so any
// future test in the package keeps the production constant.
prevRetryWaitTime := util.RetryWaitTime
util.RetryWaitTime = 100 * time.Millisecond
t.Cleanup(func() { util.RetryWaitTime = prevRetryWaitTime })
var hits atomic.Int32
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
hits.Add(1)
w.Header().Set("Content-Type", "application/octet-stream")
w.WriteHeader(http.StatusOK)
// Intentionally write no body — mimic the buggy volume response.
}))
defer sourceServer.Close()
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
filerSrc := &source.FilerSource{}
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
t.Fatalf("filerSource.DoInitialize: %v", err)
}
fs := &FilerSink{
filerSource: filerSrc,
address: serverAddr,
dir: "/dst",
executor: util.NewLimitedConcurrentExecutor(1),
}
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
sourceChunk := &filer_pb.FileChunk{
FileId: fid,
Size: expectedSize,
}
done := make(chan struct {
fileId string
err error
}, 1)
go func() {
gotFileId, gotErr := fs.fetchAndWrite(sourceChunk, "/dst/index.bin", 0)
done <- struct {
fileId string
err error
}{gotFileId, gotErr}
}()
select {
case result := <-done:
if result.err == nil {
t.Fatalf("expected size mismatch error, got nil (fileId=%q)", result.fileId)
}
if !errors.Is(result.err, errChunkSizeMismatch) {
t.Fatalf("expected errChunkSizeMismatch, got %v", result.err)
}
if !strings.Contains(result.err.Error(), "5171") {
t.Fatalf("error %q does not mention expected size 5171", result.err)
}
if !strings.Contains(result.err.Error(), fid) {
t.Fatalf("error %q does not mention chunk id %q", result.err, fid)
}
if h := hits.Load(); h != 1 {
t.Fatalf("expected exactly 1 source hit (fail-fast), got %d", h)
}
case <-time.After(5 * time.Second):
t.Fatalf("fetchAndWrite did not return within 5s (retry loop not aborted on size mismatch); hits=%d", hits.Load())
}
}
type timeoutErr struct{}
func (timeoutErr) Error() string { return "synthetic timeout" }
func (timeoutErr) Timeout() bool { return true }
func (timeoutErr) Temporary() bool { return true }
// A transient network failure (interrupted read, idle-deadline timeout while
// the destination reads the upload body, reset/broken pipe) must route through
// the escalating backoff so an overloaded destination can recover instead of
// being hammered. The volume server returns its idle timeout as a JSON error
// string, so the text path matters as much as the net.Error interface.
func TestIsRetryableNetworkError(t *testing.T) {
tests := []struct {
name string
err error
want bool
}{
{"nil", nil, false},
{"eof", io.EOF, true},
{"unexpected eof", io.ErrUnexpectedEOF, true},
{"volume idle timeout json", fmt.Errorf("upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: i/o timeout"), true},
{"volume idle timeout capitalized", fmt.Errorf("Upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: I/O timeout"), true},
{"connection reset", fmt.Errorf("upload data: write tcp ...: connection reset by peer"), true},
{"connection reset capitalized", fmt.Errorf("Connection reset by peer"), true},
{"broken pipe", fmt.Errorf("broken pipe"), true},
{"broken pipe capitalized", fmt.Errorf("Broken pipe"), true},
{"net.Error timeout", fmt.Errorf("dial: %w", timeoutErr{}), true},
{"size mismatch is permanent", errChunkSizeMismatch, false},
{"unrelated error", errors.New("not found"), false},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := isRetryableNetworkError(tc.err); got != tc.want {
t.Fatalf("isRetryableNetworkError(%v) = %v, want %v", tc.err, got, tc.want)
}
})
}
}
// Lock in that the errChunkSizeMismatch sentinel survives the wrap in
// replicateOneChunk + pass-through in util.Retry, so filer_sink.go's
// errors.Is check actually fires.
func TestReplicateChunksPreservesSizeMismatchSentinel(t *testing.T) {
const fid = "74,047d16a94aa581"
const expectedSize uint64 = 5171
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/octet-stream")
w.WriteHeader(http.StatusOK)
}))
defer sourceServer.Close()
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
filerSrc := &source.FilerSource{}
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
t.Fatalf("filerSource.DoInitialize: %v", err)
}
fs := &FilerSink{
filerSource: filerSrc,
address: serverAddr,
dir: "/dst",
executor: util.NewLimitedConcurrentExecutor(1),
}
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
sourceChunks := []*filer_pb.FileChunk{{FileId: fid, Size: expectedSize}}
_, err := fs.replicateChunks(nil, sourceChunks, "/dst/index.bin", 0)
if err == nil {
t.Fatal("expected error from replicateChunks, got nil")
}
if !errors.Is(err, errChunkSizeMismatch) {
t.Fatalf("error chain broken: errors.Is(err, errChunkSizeMismatch) = false; got %v", err)
}
}
// sourceSupersedes decides whether to skip a stale replayed event. The replayed
// mtime is fixed; the table varies what the source lookup returned.
func TestSourceSupersedes(t *testing.T) {
const eventNs int64 = 5_000_000_500 // the version being replayed (sec 5, ns 500)
withMtime := func(sec int64, ns int32) *filer_pb.Entry {
return &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: sec, MtimeNs: ns}}
}
tests := []struct {
name string
entry *filer_pb.Entry
lookupErr error
want bool
}{
// deleted on source: ErrNotFound in several shapes, all read as gone -> skip
{"not-found sentinel", nil, filer_pb.ErrNotFound, true},
{"not-found wrapped", nil, fmt.Errorf("lookup /x: %w", filer_pb.ErrNotFound), true},
{"not-found as string (gRPC)", nil, errors.New("rpc error: " + filer_pb.ErrNotFound.Error()), true},
{"nil entry, nil error", nil, nil, true},
// transient lookup failure must NOT skip a possibly-live file
{"network error", nil, errors.New("dial tcp: i/o timeout"), false},
// live entry: compare full-ns mtime against the replayed version
{"source strictly newer", withMtime(5, 600), nil, true},
{"source same version", withMtime(5, 500), nil, false},
{"source older (out-of-order replay)", withMtime(5, 400), nil, false},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
got := sourceSupersedes("/source/x/config", tc.entry, tc.lookupErr, eventNs)
if got != tc.want {
t.Fatalf("sourceSupersedes = %v, want %v", got, tc.want)
}
})
}
}
// An epoch/unset replayed mtime (0) must not block "gone" detection: a deleted
// source still reports superseded so the event is skipped instead of wedging on
// permanent retries. A live source stays not-superseded — no valid mtime to compare.
func TestSourceSupersedesEpochMtime(t *testing.T) {
live := &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: 5, MtimeNs: 600}}
if !sourceSupersedes("/source/x", nil, filer_pb.ErrNotFound, 0) {
t.Fatal("epoch-mtime deleted source must be reported gone")
}
if sourceSupersedes("/source/x", live, nil, 0) {
t.Fatal("epoch-mtime live source must not be reported superseded")
}
}
// An incremental sink's dated target keys cannot be mapped back to a source
// path, so supersession is unverifiable: the gate must stop after the bounded
// attempts and propagate instead of spinning forever.
func TestManifestResolveRetryGateUnverifiableSupersessionBounded(t *testing.T) {
fs := &FilerSink{isIncremental: true, dir: "/backup"}
gate := fs.manifestResolveRetryGate("/backup/2026-07-10/buckets/x/f.pt", 123, "3,01abc")
resolveErr := errors.New("LookupFileId volume id 3: not found")
for i := 1; i < maxUnverifiableResolveAttempts; i++ {
if !gate(resolveErr) {
t.Fatalf("attempt %d: gate must keep retrying before the bound", i)
}
}
if gate(resolveErr) {
t.Error("gate must propagate once the bound is reached with supersession unverifiable")
}
}
// Non-transient resolve errors (corrupt manifest data, bad file ids) must
// propagate immediately — before any attempt counting or supersession
// mapping — so the configured metadata error policy applies, instead of
// retrying until the source is superseded.
func TestManifestResolveRetryGateNonTransientPropagates(t *testing.T) {
fs := &FilerSink{dir: "/backup"}
gate := fs.manifestResolveRetryGate("/backup/buckets/x/f.pt", 123, "3,01abc")
permanentErrs := []error{
errors.New("fail to unmarshal manifest 3,01abc: proto: cannot parse invalid wire-format data"),
errors.New("invalid fileId abc"),
}
for _, err := range permanentErrs {
if gate(err) {
t.Errorf("non-transient error must propagate immediately: %v", err)
}
}
if !gate(errors.New("LookupFileId volume id 3: not found")) {
t.Error("transient lookup race must keep retrying on the first attempt")
}
if isTransientResolveError(nil) {
t.Error("isTransientResolveError(nil) must be false")
}
}