mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 14:31:57 +02:00
* fix(filer.backup): stop silently dropping events on transient not-found Under a write burst, filer.backup could consume a metadata event (advancing the persisted offset) without replicating the file, with no error logged: 1. filersink CreateEntry/UpdateEntry swallowed replicateChunks errors (glog.Warningf + return nil), so the offset advanced past entries that were never written. 2. The manifest-chunk branch of replicateChunks resolved via LookupFileId with no retry, unlike the data-chunk branch — transient lookup races dropped exactly the large manifest-backed files while small inline-content siblings landed. 3. isIgnorable404 matched "LookupFileId" / "volume id ... not found", misclassifying those races as genuine source 404s at the backup layer. Fix: on a replicateChunks failure the filer sink now skips only when the live source has moved past the replayed version (deleted or strictly-newer mtime) — lossless, a later event carries the current content — and propagates otherwise so the event is retried. The manifest resolve retries transient errors like the data-chunk path. isIgnorable404 is narrowed to genuine 404s; non-filer sinks and the initial-snapshot walk, which relied on the broad match as their only lossless-skip valve, now make the same live-source decision (filersink.SourceSupersedes) instead of retrying forever on a permanently gone volume. Tests cover propagation of unconfirmed lookup failures, the narrowed 404 classification, and the supersession guards. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(filer.backup): derive supersession path with directory fallback eventSourceSuperseded built the source path from NewParentPath alone. Legacy metadata events (persisted by older filers) carry an empty NewParentPath, so the probe looked up "/<name>", read the miss as "source gone", and skipped a live file on a transient lookup error — the silent drop this change is meant to eliminate. Derive the path via MetadataEventTargetFullPath (the same directory fallback genProcessFunction uses) and cover both event shapes with TestEventSupersessionProbe_PathDerivation. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(filersink): retry manifest resolve only for transient errors, bounded when unverifiable The manifest-resolve retry stopped only when hasSourceNewerVersion proved the source moved past the replayed version, which wedged the sink in two cases: incremental sinks use dated target keys that cannot map back to a source path (supersession never provable), and permanent resolve errors (corrupt manifest data, bad file ids) fail forever while the source entry stays live. Gate the retry instead: keep retrying only transient errors (volume-lookup races, network interruptions), stop after a few attempts when supersession cannot be checked, and propagate everything else immediately so the configured metadata error policy applies (-disableErrorRetry included). Propagation is lossless: filer.backup's fallback decides with the event's real source key, and both filer.backup and filer.sync re-deliver the event (RetryForeverOnError) without advancing the offset. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(filer.backup): make error classifiers nil-safe isIgnorable404, isSourceLookupError, and isTransientResolveError called err.Error() without a nil guard. All current call sites pass a non-nil error, but the guard is free and matches isRetryableNetworkError. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
428 lines
14 KiB
Go
428 lines
14 KiB
Go
package filersink
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"strings"
|
|
"sync/atomic"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/replication/source"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
|
|
)
|
|
|
|
func TestMain(m *testing.M) {
|
|
util_http.InitGlobalHttpClient()
|
|
os.Exit(m.Run())
|
|
}
|
|
|
|
func TestTargetPathToSourcePath(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
targetRoot string
|
|
sourceRoot string
|
|
targetPath string
|
|
incremental bool
|
|
wantPath util.FullPath
|
|
wantOK bool
|
|
}{
|
|
{
|
|
name: "basic mapping",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
// incremental keys carry a date prefix that can't be reversed; unmappable
|
|
name: "incremental sink is unmappable",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target/2026-06-09/path/file.txt",
|
|
incremental: true,
|
|
wantPath: "",
|
|
wantOK: false,
|
|
},
|
|
{
|
|
name: "trailing slash roots",
|
|
targetRoot: "/target/",
|
|
sourceRoot: "/source/",
|
|
targetPath: "/target/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "root target mapping",
|
|
targetRoot: "/",
|
|
sourceRoot: "/source",
|
|
targetPath: "/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "target root itself",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target",
|
|
wantPath: "/source",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "outside target root",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/other/path/file.txt",
|
|
wantPath: "",
|
|
wantOK: false,
|
|
},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
fs := &FilerSink{
|
|
dir: tc.targetRoot,
|
|
isIncremental: tc.incremental,
|
|
filerSource: &source.FilerSource{
|
|
Dir: tc.sourceRoot,
|
|
},
|
|
}
|
|
|
|
gotPath, ok := fs.targetPathToSourcePath(tc.targetPath)
|
|
if ok != tc.wantOK {
|
|
t.Fatalf("ok mismatch: got %v, want %v", ok, tc.wantOK)
|
|
}
|
|
if gotPath != tc.wantPath {
|
|
t.Fatalf("path mismatch: got %q, want %q", gotPath, tc.wantPath)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// FilerSink must reject chunks whose received byte count disagrees with the
|
|
// source filer metadata, instead of silently writing 0-byte needles with the
|
|
// source size in the destination metadata.
|
|
func TestValidateReplicatedChunkSize(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
|
|
tests := []struct {
|
|
name string
|
|
expectedSize uint64
|
|
readSize int
|
|
wantErr bool
|
|
}{
|
|
{
|
|
name: "healthy",
|
|
expectedSize: 5171,
|
|
readSize: 5171,
|
|
wantErr: false,
|
|
},
|
|
{
|
|
name: "legitimately empty file",
|
|
expectedSize: 0,
|
|
readSize: 0,
|
|
wantErr: false,
|
|
},
|
|
{
|
|
name: "zero-byte read for non-empty source",
|
|
expectedSize: 5171,
|
|
readSize: 0,
|
|
wantErr: true,
|
|
},
|
|
{
|
|
name: "short read",
|
|
expectedSize: 5171,
|
|
readSize: 100,
|
|
wantErr: true,
|
|
},
|
|
{
|
|
name: "over-read (server returned more than metadata)",
|
|
expectedSize: 5171,
|
|
readSize: 8192,
|
|
wantErr: true,
|
|
},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
chunk := &filer_pb.FileChunk{FileId: fid, Size: tc.expectedSize}
|
|
|
|
gotErr := validateReplicatedReadSize(chunk, tc.readSize)
|
|
|
|
if tc.wantErr {
|
|
if gotErr == nil {
|
|
t.Fatalf("expected error, got nil (read=%d expected=%d)",
|
|
tc.readSize, tc.expectedSize)
|
|
}
|
|
if !errors.Is(gotErr, errChunkSizeMismatch) {
|
|
t.Fatalf("expected errChunkSizeMismatch, got %v", gotErr)
|
|
}
|
|
if !strings.Contains(gotErr.Error(), fid) {
|
|
t.Fatalf("error %q does not mention chunk id %q", gotErr, fid)
|
|
}
|
|
return
|
|
}
|
|
if gotErr != nil {
|
|
t.Fatalf("unexpected read-size error: %v", gotErr)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// End-to-end regression :
|
|
// a source volume that responds 200 OK with Content-Length: 0
|
|
// for a chunk that filer metadata claims is 5171 bytes must be rejected
|
|
// by fetchAndWrite with a (non-retriable) size mismatch error,
|
|
// instead of being silently propagated to the destination as a 0-byte needle.
|
|
func TestFetchAndWriteRejectsZeroByteSource(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
const expectedSize uint64 = 5171
|
|
|
|
// Shorten retry backoff so a fail-fast test that briefly enters the retry
|
|
// loop doesn't pay the production 1s+ wait. Scoped to this test so any
|
|
// future test in the package keeps the production constant.
|
|
prevRetryWaitTime := util.RetryWaitTime
|
|
util.RetryWaitTime = 100 * time.Millisecond
|
|
t.Cleanup(func() { util.RetryWaitTime = prevRetryWaitTime })
|
|
|
|
var hits atomic.Int32
|
|
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
hits.Add(1)
|
|
w.Header().Set("Content-Type", "application/octet-stream")
|
|
w.WriteHeader(http.StatusOK)
|
|
// Intentionally write no body — mimic the buggy volume response.
|
|
}))
|
|
defer sourceServer.Close()
|
|
|
|
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
|
|
|
filerSrc := &source.FilerSource{}
|
|
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
|
t.Fatalf("filerSource.DoInitialize: %v", err)
|
|
}
|
|
|
|
fs := &FilerSink{
|
|
filerSource: filerSrc,
|
|
address: serverAddr,
|
|
dir: "/dst",
|
|
executor: util.NewLimitedConcurrentExecutor(1),
|
|
}
|
|
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
|
|
|
sourceChunk := &filer_pb.FileChunk{
|
|
FileId: fid,
|
|
Size: expectedSize,
|
|
}
|
|
|
|
done := make(chan struct {
|
|
fileId string
|
|
err error
|
|
}, 1)
|
|
go func() {
|
|
gotFileId, gotErr := fs.fetchAndWrite(sourceChunk, "/dst/index.bin", 0)
|
|
done <- struct {
|
|
fileId string
|
|
err error
|
|
}{gotFileId, gotErr}
|
|
}()
|
|
|
|
select {
|
|
case result := <-done:
|
|
if result.err == nil {
|
|
t.Fatalf("expected size mismatch error, got nil (fileId=%q)", result.fileId)
|
|
}
|
|
if !errors.Is(result.err, errChunkSizeMismatch) {
|
|
t.Fatalf("expected errChunkSizeMismatch, got %v", result.err)
|
|
}
|
|
if !strings.Contains(result.err.Error(), "5171") {
|
|
t.Fatalf("error %q does not mention expected size 5171", result.err)
|
|
}
|
|
if !strings.Contains(result.err.Error(), fid) {
|
|
t.Fatalf("error %q does not mention chunk id %q", result.err, fid)
|
|
}
|
|
if h := hits.Load(); h != 1 {
|
|
t.Fatalf("expected exactly 1 source hit (fail-fast), got %d", h)
|
|
}
|
|
case <-time.After(5 * time.Second):
|
|
t.Fatalf("fetchAndWrite did not return within 5s (retry loop not aborted on size mismatch); hits=%d", hits.Load())
|
|
}
|
|
}
|
|
|
|
type timeoutErr struct{}
|
|
|
|
func (timeoutErr) Error() string { return "synthetic timeout" }
|
|
func (timeoutErr) Timeout() bool { return true }
|
|
func (timeoutErr) Temporary() bool { return true }
|
|
|
|
// A transient network failure (interrupted read, idle-deadline timeout while
|
|
// the destination reads the upload body, reset/broken pipe) must route through
|
|
// the escalating backoff so an overloaded destination can recover instead of
|
|
// being hammered. The volume server returns its idle timeout as a JSON error
|
|
// string, so the text path matters as much as the net.Error interface.
|
|
func TestIsRetryableNetworkError(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
err error
|
|
want bool
|
|
}{
|
|
{"nil", nil, false},
|
|
{"eof", io.EOF, true},
|
|
{"unexpected eof", io.ErrUnexpectedEOF, true},
|
|
{"volume idle timeout json", fmt.Errorf("upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: i/o timeout"), true},
|
|
{"volume idle timeout capitalized", fmt.Errorf("Upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: I/O timeout"), true},
|
|
{"connection reset", fmt.Errorf("upload data: write tcp ...: connection reset by peer"), true},
|
|
{"connection reset capitalized", fmt.Errorf("Connection reset by peer"), true},
|
|
{"broken pipe", fmt.Errorf("broken pipe"), true},
|
|
{"broken pipe capitalized", fmt.Errorf("Broken pipe"), true},
|
|
{"net.Error timeout", fmt.Errorf("dial: %w", timeoutErr{}), true},
|
|
{"size mismatch is permanent", errChunkSizeMismatch, false},
|
|
{"unrelated error", errors.New("not found"), false},
|
|
}
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
if got := isRetryableNetworkError(tc.err); got != tc.want {
|
|
t.Fatalf("isRetryableNetworkError(%v) = %v, want %v", tc.err, got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// Lock in that the errChunkSizeMismatch sentinel survives the wrap in
|
|
// replicateOneChunk + pass-through in util.Retry, so filer_sink.go's
|
|
// errors.Is check actually fires.
|
|
func TestReplicateChunksPreservesSizeMismatchSentinel(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
const expectedSize uint64 = 5171
|
|
|
|
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
w.Header().Set("Content-Type", "application/octet-stream")
|
|
w.WriteHeader(http.StatusOK)
|
|
}))
|
|
defer sourceServer.Close()
|
|
|
|
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
|
|
|
filerSrc := &source.FilerSource{}
|
|
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
|
t.Fatalf("filerSource.DoInitialize: %v", err)
|
|
}
|
|
|
|
fs := &FilerSink{
|
|
filerSource: filerSrc,
|
|
address: serverAddr,
|
|
dir: "/dst",
|
|
executor: util.NewLimitedConcurrentExecutor(1),
|
|
}
|
|
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
|
|
|
sourceChunks := []*filer_pb.FileChunk{{FileId: fid, Size: expectedSize}}
|
|
|
|
_, err := fs.replicateChunks(nil, sourceChunks, "/dst/index.bin", 0)
|
|
if err == nil {
|
|
t.Fatal("expected error from replicateChunks, got nil")
|
|
}
|
|
if !errors.Is(err, errChunkSizeMismatch) {
|
|
t.Fatalf("error chain broken: errors.Is(err, errChunkSizeMismatch) = false; got %v", err)
|
|
}
|
|
}
|
|
|
|
// sourceSupersedes decides whether to skip a stale replayed event. The replayed
|
|
// mtime is fixed; the table varies what the source lookup returned.
|
|
func TestSourceSupersedes(t *testing.T) {
|
|
const eventNs int64 = 5_000_000_500 // the version being replayed (sec 5, ns 500)
|
|
|
|
withMtime := func(sec int64, ns int32) *filer_pb.Entry {
|
|
return &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: sec, MtimeNs: ns}}
|
|
}
|
|
|
|
tests := []struct {
|
|
name string
|
|
entry *filer_pb.Entry
|
|
lookupErr error
|
|
want bool
|
|
}{
|
|
// deleted on source: ErrNotFound in several shapes, all read as gone -> skip
|
|
{"not-found sentinel", nil, filer_pb.ErrNotFound, true},
|
|
{"not-found wrapped", nil, fmt.Errorf("lookup /x: %w", filer_pb.ErrNotFound), true},
|
|
{"not-found as string (gRPC)", nil, errors.New("rpc error: " + filer_pb.ErrNotFound.Error()), true},
|
|
{"nil entry, nil error", nil, nil, true},
|
|
// transient lookup failure must NOT skip a possibly-live file
|
|
{"network error", nil, errors.New("dial tcp: i/o timeout"), false},
|
|
// live entry: compare full-ns mtime against the replayed version
|
|
{"source strictly newer", withMtime(5, 600), nil, true},
|
|
{"source same version", withMtime(5, 500), nil, false},
|
|
{"source older (out-of-order replay)", withMtime(5, 400), nil, false},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
got := sourceSupersedes("/source/x/config", tc.entry, tc.lookupErr, eventNs)
|
|
if got != tc.want {
|
|
t.Fatalf("sourceSupersedes = %v, want %v", got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// An epoch/unset replayed mtime (0) must not block "gone" detection: a deleted
|
|
// source still reports superseded so the event is skipped instead of wedging on
|
|
// permanent retries. A live source stays not-superseded — no valid mtime to compare.
|
|
func TestSourceSupersedesEpochMtime(t *testing.T) {
|
|
live := &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: 5, MtimeNs: 600}}
|
|
if !sourceSupersedes("/source/x", nil, filer_pb.ErrNotFound, 0) {
|
|
t.Fatal("epoch-mtime deleted source must be reported gone")
|
|
}
|
|
if sourceSupersedes("/source/x", live, nil, 0) {
|
|
t.Fatal("epoch-mtime live source must not be reported superseded")
|
|
}
|
|
}
|
|
|
|
// An incremental sink's dated target keys cannot be mapped back to a source
|
|
// path, so supersession is unverifiable: the gate must stop after the bounded
|
|
// attempts and propagate instead of spinning forever.
|
|
func TestManifestResolveRetryGateUnverifiableSupersessionBounded(t *testing.T) {
|
|
fs := &FilerSink{isIncremental: true, dir: "/backup"}
|
|
gate := fs.manifestResolveRetryGate("/backup/2026-07-10/buckets/x/f.pt", 123, "3,01abc")
|
|
resolveErr := errors.New("LookupFileId volume id 3: not found")
|
|
for i := 1; i < maxUnverifiableResolveAttempts; i++ {
|
|
if !gate(resolveErr) {
|
|
t.Fatalf("attempt %d: gate must keep retrying before the bound", i)
|
|
}
|
|
}
|
|
if gate(resolveErr) {
|
|
t.Error("gate must propagate once the bound is reached with supersession unverifiable")
|
|
}
|
|
}
|
|
|
|
// Non-transient resolve errors (corrupt manifest data, bad file ids) must
|
|
// propagate immediately — before any attempt counting or supersession
|
|
// mapping — so the configured metadata error policy applies, instead of
|
|
// retrying until the source is superseded.
|
|
func TestManifestResolveRetryGateNonTransientPropagates(t *testing.T) {
|
|
fs := &FilerSink{dir: "/backup"}
|
|
gate := fs.manifestResolveRetryGate("/backup/buckets/x/f.pt", 123, "3,01abc")
|
|
permanentErrs := []error{
|
|
errors.New("fail to unmarshal manifest 3,01abc: proto: cannot parse invalid wire-format data"),
|
|
errors.New("invalid fileId abc"),
|
|
}
|
|
for _, err := range permanentErrs {
|
|
if gate(err) {
|
|
t.Errorf("non-transient error must propagate immediately: %v", err)
|
|
}
|
|
}
|
|
if !gate(errors.New("LookupFileId volume id 3: not found")) {
|
|
t.Error("transient lookup race must keep retrying on the first attempt")
|
|
}
|
|
if isTransientResolveError(nil) {
|
|
t.Error("isTransientResolveError(nil) must be false")
|
|
}
|
|
}
|