mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-09 07:47:54 +02:00
* fix(filer.backup): skip replay events whose chunk no longer exists on the source "Source" is the filer we replicate FROM (e.g. green in a green->blue backup). Replaying the metadata log from a checkpoint can hit an event whose chunk was since overwritten/deleted and garbage-collected on the source volume. Fetching it returns 0 bytes (a permanent size mismatch), which the sink propagated to the subscription — so the same offset retried forever and replication stalled. Skip the event only when proven stale; otherwise keep refusing so genuine loss of a live file still halts loudly: - onCorruptChunk centralizes the three errChunkSizeMismatch sites. - getEntryMtimeNs compares mtime at nanosecond precision so same-second rewrites (git's config.lock dance) are ordered correctly. - sourceSupersedes re-reads the entry's current state on the source: gone (ErrNotFound) or a strictly-newer mtime than the replayed version -> skip; any other lookup error keeps the entry. Skipping is lossless: events are full-entry snapshots, so a later event re-carries the current chunks and a delete event reconciles a removed file. * test(filer.backup): cover the superseded-chunk skip decision - TestSourceSupersedes: not-found (sentinel / wrapped / gRPC string) and nil entry -> skip; network error -> keep; source newer -> skip; same/older -> keep. - TestGetEntryMtimeNs: nanosecond precision, same-second ordering, nil safety. - TestOnCorruptChunkRefusesWhenSupersessionUnconfirmed: never skip silently when supersession cannot be confirmed. * fix(filer.backup): don't infer supersession for incremental sinks In incremental mode the sink key carries a date prefix (sinkDir/YYYY-MM-DD/relPath) that cannot be reversed to a real source path, so a source lookup would always be ErrNotFound and wrongly classify a live entry as deleted — skipping it. Make targetPathToSourcePath report "unmappable" in incremental mode; hasSourceNewerVersion already declines to skip when the source path cannot be mapped. Found in code review. Non-incremental sinks (filer.backup green->blue) are unaffected. * refactor(filer.backup): name the mtime param sourceMtimeNs; note ns overflow bound - Rename the threaded sourceMtime parameter to sourceMtimeNs across the internal replicate/fetch helpers so the unit is explicit (it only feeds hasSourceNewerVersion, which compares in nanoseconds). - Document that getEntryMtimeNs's int64 ns arithmetic is safe until ~year 2262. No behavior change. * fix(filer.backup): order same-second versions in the CreateEntry skip and update gates The CreateEntry already-replicated short-circuit and chooseUpdateAction still compared second-grained mtime, so a newer version written within the same second could be skipped as already-replicated or overwritten by an older same-second replay. Route both through getEntryMtimeNs, matching the precision the chunk-replication path already uses. * test(filer.backup): cover same-second update-action ordering * docs(filer.backup): trim verbose comments to terse why * fix(filer.backup): check supersession against the rename's new path For a rename the filer sink updates in place (the delete+create branch is skipped for sink name "filer"), so the corrupt-chunk supersession check queried the pre-rename key. Its source-side ErrNotFound was read as "superseded", silently advancing the checkpoint without applying the rename. Map the incoming entry's new path (newParentPath/newEntry.Name) for both update branches. * fix(filer.backup): detect a deleted source even when the replayed mtime is epoch hasSourceNewerVersion returned early when sourceMtimeNs <= 0, skipping the source lookup, so a deleted entry with mtime 0 (a valid epoch timestamp) never got the gone verdict and wedged on permanent retries. Always look up; gate only the newer-mtime comparison on a valid replayed mtime. --------- Co-authored-by: Chris Lu <chris.lu@gmail.com>
387 lines
12 KiB
Go
387 lines
12 KiB
Go
package filersink
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"strings"
|
|
"sync/atomic"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/replication/source"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
|
|
)
|
|
|
|
func TestMain(m *testing.M) {
|
|
util_http.InitGlobalHttpClient()
|
|
os.Exit(m.Run())
|
|
}
|
|
|
|
func TestTargetPathToSourcePath(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
targetRoot string
|
|
sourceRoot string
|
|
targetPath string
|
|
incremental bool
|
|
wantPath util.FullPath
|
|
wantOK bool
|
|
}{
|
|
{
|
|
name: "basic mapping",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
// incremental keys carry a date prefix that can't be reversed; unmappable
|
|
name: "incremental sink is unmappable",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target/2026-06-09/path/file.txt",
|
|
incremental: true,
|
|
wantPath: "",
|
|
wantOK: false,
|
|
},
|
|
{
|
|
name: "trailing slash roots",
|
|
targetRoot: "/target/",
|
|
sourceRoot: "/source/",
|
|
targetPath: "/target/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "root target mapping",
|
|
targetRoot: "/",
|
|
sourceRoot: "/source",
|
|
targetPath: "/path/file.txt",
|
|
wantPath: "/source/path/file.txt",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "target root itself",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/target",
|
|
wantPath: "/source",
|
|
wantOK: true,
|
|
},
|
|
{
|
|
name: "outside target root",
|
|
targetRoot: "/target",
|
|
sourceRoot: "/source",
|
|
targetPath: "/other/path/file.txt",
|
|
wantPath: "",
|
|
wantOK: false,
|
|
},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
fs := &FilerSink{
|
|
dir: tc.targetRoot,
|
|
isIncremental: tc.incremental,
|
|
filerSource: &source.FilerSource{
|
|
Dir: tc.sourceRoot,
|
|
},
|
|
}
|
|
|
|
gotPath, ok := fs.targetPathToSourcePath(tc.targetPath)
|
|
if ok != tc.wantOK {
|
|
t.Fatalf("ok mismatch: got %v, want %v", ok, tc.wantOK)
|
|
}
|
|
if gotPath != tc.wantPath {
|
|
t.Fatalf("path mismatch: got %q, want %q", gotPath, tc.wantPath)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// FilerSink must reject chunks whose received byte count disagrees with the
|
|
// source filer metadata, instead of silently writing 0-byte needles with the
|
|
// source size in the destination metadata.
|
|
func TestValidateReplicatedChunkSize(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
|
|
tests := []struct {
|
|
name string
|
|
expectedSize uint64
|
|
readSize int
|
|
wantErr bool
|
|
}{
|
|
{
|
|
name: "healthy",
|
|
expectedSize: 5171,
|
|
readSize: 5171,
|
|
wantErr: false,
|
|
},
|
|
{
|
|
name: "legitimately empty file",
|
|
expectedSize: 0,
|
|
readSize: 0,
|
|
wantErr: false,
|
|
},
|
|
{
|
|
name: "zero-byte read for non-empty source",
|
|
expectedSize: 5171,
|
|
readSize: 0,
|
|
wantErr: true,
|
|
},
|
|
{
|
|
name: "short read",
|
|
expectedSize: 5171,
|
|
readSize: 100,
|
|
wantErr: true,
|
|
},
|
|
{
|
|
name: "over-read (server returned more than metadata)",
|
|
expectedSize: 5171,
|
|
readSize: 8192,
|
|
wantErr: true,
|
|
},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
chunk := &filer_pb.FileChunk{FileId: fid, Size: tc.expectedSize}
|
|
|
|
gotErr := validateReplicatedReadSize(chunk, tc.readSize)
|
|
|
|
if tc.wantErr {
|
|
if gotErr == nil {
|
|
t.Fatalf("expected error, got nil (read=%d expected=%d)",
|
|
tc.readSize, tc.expectedSize)
|
|
}
|
|
if !errors.Is(gotErr, errChunkSizeMismatch) {
|
|
t.Fatalf("expected errChunkSizeMismatch, got %v", gotErr)
|
|
}
|
|
if !strings.Contains(gotErr.Error(), fid) {
|
|
t.Fatalf("error %q does not mention chunk id %q", gotErr, fid)
|
|
}
|
|
return
|
|
}
|
|
if gotErr != nil {
|
|
t.Fatalf("unexpected read-size error: %v", gotErr)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// End-to-end regression :
|
|
// a source volume that responds 200 OK with Content-Length: 0
|
|
// for a chunk that filer metadata claims is 5171 bytes must be rejected
|
|
// by fetchAndWrite with a (non-retriable) size mismatch error,
|
|
// instead of being silently propagated to the destination as a 0-byte needle.
|
|
func TestFetchAndWriteRejectsZeroByteSource(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
const expectedSize uint64 = 5171
|
|
|
|
// Shorten retry backoff so a fail-fast test that briefly enters the retry
|
|
// loop doesn't pay the production 1s+ wait. Scoped to this test so any
|
|
// future test in the package keeps the production constant.
|
|
prevRetryWaitTime := util.RetryWaitTime
|
|
util.RetryWaitTime = 100 * time.Millisecond
|
|
t.Cleanup(func() { util.RetryWaitTime = prevRetryWaitTime })
|
|
|
|
var hits atomic.Int32
|
|
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
hits.Add(1)
|
|
w.Header().Set("Content-Type", "application/octet-stream")
|
|
w.WriteHeader(http.StatusOK)
|
|
// Intentionally write no body — mimic the buggy volume response.
|
|
}))
|
|
defer sourceServer.Close()
|
|
|
|
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
|
|
|
filerSrc := &source.FilerSource{}
|
|
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
|
t.Fatalf("filerSource.DoInitialize: %v", err)
|
|
}
|
|
|
|
fs := &FilerSink{
|
|
filerSource: filerSrc,
|
|
address: serverAddr,
|
|
dir: "/dst",
|
|
executor: util.NewLimitedConcurrentExecutor(1),
|
|
}
|
|
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
|
|
|
sourceChunk := &filer_pb.FileChunk{
|
|
FileId: fid,
|
|
Size: expectedSize,
|
|
}
|
|
|
|
done := make(chan struct {
|
|
fileId string
|
|
err error
|
|
}, 1)
|
|
go func() {
|
|
gotFileId, gotErr := fs.fetchAndWrite(sourceChunk, "/dst/index.bin", 0)
|
|
done <- struct {
|
|
fileId string
|
|
err error
|
|
}{gotFileId, gotErr}
|
|
}()
|
|
|
|
select {
|
|
case result := <-done:
|
|
if result.err == nil {
|
|
t.Fatalf("expected size mismatch error, got nil (fileId=%q)", result.fileId)
|
|
}
|
|
if !errors.Is(result.err, errChunkSizeMismatch) {
|
|
t.Fatalf("expected errChunkSizeMismatch, got %v", result.err)
|
|
}
|
|
if !strings.Contains(result.err.Error(), "5171") {
|
|
t.Fatalf("error %q does not mention expected size 5171", result.err)
|
|
}
|
|
if !strings.Contains(result.err.Error(), fid) {
|
|
t.Fatalf("error %q does not mention chunk id %q", result.err, fid)
|
|
}
|
|
if h := hits.Load(); h != 1 {
|
|
t.Fatalf("expected exactly 1 source hit (fail-fast), got %d", h)
|
|
}
|
|
case <-time.After(5 * time.Second):
|
|
t.Fatalf("fetchAndWrite did not return within 5s (retry loop not aborted on size mismatch); hits=%d", hits.Load())
|
|
}
|
|
}
|
|
|
|
type timeoutErr struct{}
|
|
|
|
func (timeoutErr) Error() string { return "synthetic timeout" }
|
|
func (timeoutErr) Timeout() bool { return true }
|
|
func (timeoutErr) Temporary() bool { return true }
|
|
|
|
// A transient network failure (interrupted read, idle-deadline timeout while
|
|
// the destination reads the upload body, reset/broken pipe) must route through
|
|
// the escalating backoff so an overloaded destination can recover instead of
|
|
// being hammered. The volume server returns its idle timeout as a JSON error
|
|
// string, so the text path matters as much as the net.Error interface.
|
|
func TestIsRetryableNetworkError(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
err error
|
|
want bool
|
|
}{
|
|
{"nil", nil, false},
|
|
{"eof", io.EOF, true},
|
|
{"unexpected eof", io.ErrUnexpectedEOF, true},
|
|
{"volume idle timeout json", fmt.Errorf("upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: i/o timeout"), true},
|
|
{"volume idle timeout capitalized", fmt.Errorf("Upload result: read tcp 10.0.0.1:8082->10.0.0.1:54848: I/O timeout"), true},
|
|
{"connection reset", fmt.Errorf("upload data: write tcp ...: connection reset by peer"), true},
|
|
{"connection reset capitalized", fmt.Errorf("Connection reset by peer"), true},
|
|
{"broken pipe", fmt.Errorf("broken pipe"), true},
|
|
{"broken pipe capitalized", fmt.Errorf("Broken pipe"), true},
|
|
{"net.Error timeout", fmt.Errorf("dial: %w", timeoutErr{}), true},
|
|
{"size mismatch is permanent", errChunkSizeMismatch, false},
|
|
{"unrelated error", errors.New("not found"), false},
|
|
}
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
if got := isRetryableNetworkError(tc.err); got != tc.want {
|
|
t.Fatalf("isRetryableNetworkError(%v) = %v, want %v", tc.err, got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// Lock in that the errChunkSizeMismatch sentinel survives the wrap in
|
|
// replicateOneChunk + pass-through in util.Retry, so filer_sink.go's
|
|
// errors.Is check actually fires.
|
|
func TestReplicateChunksPreservesSizeMismatchSentinel(t *testing.T) {
|
|
const fid = "74,047d16a94aa581"
|
|
const expectedSize uint64 = 5171
|
|
|
|
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
w.Header().Set("Content-Type", "application/octet-stream")
|
|
w.WriteHeader(http.StatusOK)
|
|
}))
|
|
defer sourceServer.Close()
|
|
|
|
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
|
|
|
filerSrc := &source.FilerSource{}
|
|
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
|
t.Fatalf("filerSource.DoInitialize: %v", err)
|
|
}
|
|
|
|
fs := &FilerSink{
|
|
filerSource: filerSrc,
|
|
address: serverAddr,
|
|
dir: "/dst",
|
|
executor: util.NewLimitedConcurrentExecutor(1),
|
|
}
|
|
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
|
|
|
sourceChunks := []*filer_pb.FileChunk{{FileId: fid, Size: expectedSize}}
|
|
|
|
_, err := fs.replicateChunks(nil, sourceChunks, "/dst/index.bin", 0)
|
|
if err == nil {
|
|
t.Fatal("expected error from replicateChunks, got nil")
|
|
}
|
|
if !errors.Is(err, errChunkSizeMismatch) {
|
|
t.Fatalf("error chain broken: errors.Is(err, errChunkSizeMismatch) = false; got %v", err)
|
|
}
|
|
}
|
|
|
|
// sourceSupersedes decides whether to skip a stale replayed event. The replayed
|
|
// mtime is fixed; the table varies what the source lookup returned.
|
|
func TestSourceSupersedes(t *testing.T) {
|
|
const eventNs int64 = 5_000_000_500 // the version being replayed (sec 5, ns 500)
|
|
|
|
withMtime := func(sec int64, ns int32) *filer_pb.Entry {
|
|
return &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: sec, MtimeNs: ns}}
|
|
}
|
|
|
|
tests := []struct {
|
|
name string
|
|
entry *filer_pb.Entry
|
|
lookupErr error
|
|
want bool
|
|
}{
|
|
// deleted on source: ErrNotFound in several shapes, all read as gone -> skip
|
|
{"not-found sentinel", nil, filer_pb.ErrNotFound, true},
|
|
{"not-found wrapped", nil, fmt.Errorf("lookup /x: %w", filer_pb.ErrNotFound), true},
|
|
{"not-found as string (gRPC)", nil, errors.New("rpc error: " + filer_pb.ErrNotFound.Error()), true},
|
|
{"nil entry, nil error", nil, nil, true},
|
|
// transient lookup failure must NOT skip a possibly-live file
|
|
{"network error", nil, errors.New("dial tcp: i/o timeout"), false},
|
|
// live entry: compare full-ns mtime against the replayed version
|
|
{"source strictly newer", withMtime(5, 600), nil, true},
|
|
{"source same version", withMtime(5, 500), nil, false},
|
|
{"source older (out-of-order replay)", withMtime(5, 400), nil, false},
|
|
}
|
|
|
|
for _, tc := range tests {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
got := sourceSupersedes("/source/x/config", tc.entry, tc.lookupErr, eventNs)
|
|
if got != tc.want {
|
|
t.Fatalf("sourceSupersedes = %v, want %v", got, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// An epoch/unset replayed mtime (0) must not block "gone" detection: a deleted
|
|
// source still reports superseded so the event is skipped instead of wedging on
|
|
// permanent retries. A live source stays not-superseded — no valid mtime to compare.
|
|
func TestSourceSupersedesEpochMtime(t *testing.T) {
|
|
live := &filer_pb.Entry{Attributes: &filer_pb.FuseAttributes{Mtime: 5, MtimeNs: 600}}
|
|
if !sourceSupersedes("/source/x", nil, filer_pb.ErrNotFound, 0) {
|
|
t.Fatal("epoch-mtime deleted source must be reported gone")
|
|
}
|
|
if sourceSupersedes("/source/x", live, nil, 0) {
|
|
t.Fatal("epoch-mtime live source must not be reported superseded")
|
|
}
|
|
}
|