Files
seaweedfs/weed/s3api/s3api_object_versioning_heal_test.go
T
576837f1b2 fix(s3): make self-heal pointer persist CAS-bound against concurrent writers (#11627)
* fix(s3): make self-heal pointer persist CAS-bound against concurrent writers

Follow-up to #11618: pointerless reads of a slash key whose regular-path
entry is a physical parent (or a bare-key object) now fall through to
healStaleLatestVersionPointer, which rescans .versions and persists a
repaired pointer. The persist was an unconditional upsert off the pre-scan
snapshot, so a PUT or delete that atomically advanced the pointer on the
owner filer while the heal was rescanning could be rolled back, making
older content or ACLs current again.

Mirror the CAS discipline clearStaleLatestVersionPointer already applies:
re-fetch the live .versions entry, require its pointer fields to still
match the ones the heal observed, and abandon the persist (still returning
the rescanned entry) when a concurrent writer has moved them. Write the
live Extended map so concurrently updated fields are preserved.

* fix(s3): close the check-then-act window in the self-heal pointer persist

The CAS re-fetch added in the previous commit narrows the race but leaves
a gateway-side window: after the live .versions entry is re-read and the
pointer compared, the repair is still written back through an
unconditional RPC, so a PUT or delete committing between the re-fetch and
the persist still ends up rolled back by the stale repair.

Bind the persist to the live image the heal just re-read with an
IF_ENTRY_EQUAL precondition, the same discipline routedSelfCopy applies
to stale self-copies: the filer evaluates the condition under the entry's
path lock and conditional writes route to the owner filer, so a writer
committing inside the window fails the precondition and the winner's
pointer stands. FailedPrecondition and NotFound are authoritative replies
and are not replayed by the failover layer.

The test now also covers a writer committing during the persist, which
reverts the pointer on the previous unconditional write-back.

* s3api: CAS-bind the stale-pointer clear against concurrent writers

The pointer clear re-read the live .versions entry and then wrote it
back unconditionally through mkFile, so a writer committing between the
re-fetch and the persist was rolled back to a cleared pointer. Persist
through the same IF_ENTRY_EQUAL conditional update as the repair path.

* s3api: test the CAS contract on the stale-pointer clear

* s3api: never clear a pointer the clear did not observe as stale

The CAS clear skipped its live-pointer match when the caller's snapshot
carried an empty latest-version id, so a writer promoting a version
between the clear's rescan and its re-fetch had the fresh pointer
CAS-cleared away (expected = the writer's own live entry), briefly
making the just-written version appear absent. With an empty observed
id, reaching the persist at all implies a concurrent promotion (an idle
key short-circuits as already-clear), so make the pointer match
unconditional and abort instead.

Extend TestClearStaleLatestVersionPointerConcurrentWriter with
pointerless-snapshot cases: a post-rescan promotion must survive, and an
idle pointerless key must short-circuit as already-clear. The fake
filer's proto round-trip drops empty Extended maps, so the snapshot is
padded the way real callers do.

---------

Co-authored-by: zhaoyuchen <yc.zhao@yinzon.com>
Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com>
2026-10-07 20:04:09 +08:00

215 lines
9.4 KiB
Go

package s3api
import (
"context"
"net/http"
"path"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/stretchr/testify/require"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
"google.golang.org/protobuf/proto"
)
// anonymousReadHealFiler drives the versioned self-heal path for "folder/": it
// serves a real version file from .versions, runs hooks at the heal's rescan
// and at the persist (simulating a writer committing at either point), and
// evaluates the heal's IF_ENTRY_EQUAL precondition like a filer would, so
// tests can inspect the resulting .versions metadata.
type anonymousReadHealFiler struct {
*anonymousReadFiler
onScan func(*anonymousReadHealFiler)
onWrite func(*anonymousReadHealFiler)
}
// ListEntries serves the version files under .versions for the heal's rescan.
func (f *anonymousReadHealFiler) ListEntries(req *filer_pb.ListEntriesRequest, stream filer_pb.SeaweedFiler_ListEntriesServer) error {
f.mu.Lock()
if f.onScan != nil {
onScan := f.onScan
f.onScan = nil
f.mu.Unlock()
onScan(f)
} else {
f.mu.Unlock()
}
f.mu.Lock()
defer f.mu.Unlock()
for _, entry := range f.entries {
if path.Join(req.Directory, entry.Name) == "/buckets/b/folder/.versions/v_v1" {
if err := stream.Send(&filer_pb.ListEntriesResponse{Entry: proto.Clone(entry).(*filer_pb.Entry)}); err != nil {
return err
}
}
}
return nil
}
// UpdateEntry evaluates the heal's IF_ENTRY_EQUAL precondition the way a
// filer does under the path lock: the stored entry must still equal the live
// image the heal re-read, otherwise the stale repair is rejected.
func (f *anonymousReadHealFiler) UpdateEntry(_ context.Context, req *filer_pb.UpdateEntryRequest) (*filer_pb.UpdateEntryResponse, error) {
f.mu.Lock()
if f.onWrite != nil {
onWrite := f.onWrite
f.onWrite = nil
f.mu.Unlock()
// A writer commits between the heal's re-fetch and the persist.
onWrite(f)
} else {
f.mu.Unlock()
}
f.mu.Lock()
defer f.mu.Unlock()
fullPath := path.Join(req.Directory, req.Entry.Name)
current := f.entries[fullPath]
for _, clause := range req.Condition.Clauses {
if clause.Kind != filer_pb.WriteCondition_IF_ENTRY_EQUAL {
return nil, status.Error(codes.Unimplemented, "unsupported clause")
}
expected := proto.Clone(clause.ExpectedEntry).(*filer_pb.Entry)
actual := proto.Clone(current).(*filer_pb.Entry)
filer_pb.BeforeEntrySerialization(expected.Chunks)
filer_pb.BeforeEntrySerialization(actual.Chunks)
if !proto.Equal(actual, expected) {
return nil, status.Error(codes.FailedPrecondition, "entry changed")
}
}
f.entries[fullPath] = proto.Clone(req.Entry).(*filer_pb.Entry)
return &filer_pb.UpdateEntryResponse{}, nil
}
// TestAnonymousObjectACLHealPointerConcurrentWriter pins the CAS contract of
// the self-heal persist: a pointer that a concurrent writer promoted while the
// heal was rescanning, or between the heal's re-fetch and its persist, must
// survive, instead of being rolled back to the scanned version, which would
// make older content or ACLs current again.
func TestAnonymousObjectACLHealPointerConcurrentWriter(t *testing.T) {
for _, tc := range []struct {
name string
writerDuringScan bool
writerDuringWrite bool
wantPointer string
wantPointerVersion string
}{
{name: "idle key persists the rescanned pointer", wantPointer: "v1", wantPointerVersion: "v_v1"},
{name: "concurrent writer during scan keeps its promotion", writerDuringScan: true, wantPointer: "v2", wantPointerVersion: "v_v2"},
{name: "concurrent writer during persist keeps its promotion", writerDuringWrite: true, wantPointer: "v3", wantPointerVersion: "v_v3"},
} {
t.Run(tc.name, func(t *testing.T) {
f := &anonymousReadHealFiler{anonymousReadFiler: &anonymousReadFiler{entries: make(map[string]*filer_pb.Entry)}}
// The object is "folder/"; the regular path is a physical parent, so a
// pointerless read falls through to the persistent self-heal.
f.entries["/buckets/b/folder"] = &filer_pb.Entry{Name: "folder", IsDirectory: true, Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000}}
f.entries["/buckets/b/folder/child"] = anonymousReadEntry([]byte(`[]`))
f.entries["/buckets/b/folder/.versions"] = &filer_pb.Entry{Name: ".versions", IsDirectory: true, Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000}, Extended: map[string][]byte{s3_constants.ExtLatestVersionIdKey: []byte("")}}
v1 := anonymousReadEntry([]byte(`[]`))
v1.Name = "v_v1"
v1.Extended[s3_constants.ExtVersionIdKey] = []byte("v1")
f.entries["/buckets/b/folder/.versions/v_v1"] = v1
promotePointer := func(versionId, fileName string) func(*anonymousReadHealFiler) {
return func(f *anonymousReadHealFiler) {
// A concurrent PUT commits a newer version.
f.mu.Lock()
defer f.mu.Unlock()
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey] = []byte(versionId)
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey] = []byte(fileName)
}
}
if tc.writerDuringScan {
f.onScan = promotePointer("v2", "v_v2")
}
if tc.writerDuringWrite {
f.onWrite = promotePointer("v3", "v_v3")
}
s3a := newPutTestServer(t, startFakeFiler(t, f))
s3a.iam = &IdentityAccessManagement{isAuthEnabled: true}
s3a.bucketConfigCache = NewBucketConfigCache(time.Minute)
s3a.bucketConfigCache.Set("b", &BucketConfig{Name: "b", Ownership: s3_constants.OwnershipObjectWriter, Versioning: "Enabled"})
s3a.policyEngine = NewBucketPolicyEngine()
s3a.iam.policyEngine = s3a.policyEngine
require.NoError(t, s3a.policyEngine.engine.SetBucketPolicy("b", `{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Principal":"*","Action":"s3:GetObject*","Resource":"arn:aws:s3:::b/folder/"}]}`))
rr := serveAnonymousRead(s3a, http.MethodGet, "folder/", "", nil)
require.Equal(t, http.StatusOK, rr.Code, rr.Body.String())
// This read returns the rescanned entry either way.
require.Equal(t, "hello world", rr.Body.String())
// The persisted pointer must reflect the concurrent writer, not the scan.
pointer := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey]
require.Equal(t, tc.wantPointer, string(pointer))
pointerFile := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey]
require.Equal(t, tc.wantPointerVersion, string(pointerFile))
})
}
}
// TestClearStaleLatestVersionPointerConcurrentWriter pins the same CAS
// contract on the pointer clear: a writer that promotes the pointer between
// the clear's re-fetch and its persist must not be rolled back to a cleared
// pointer. A pointerless caller snapshot must not clear a pointer a writer
// promoted after the clear's rescan either — the only live pointer such a
// clear can observe is one that appeared concurrently, never the stale one
// the clear set out to remove.
func TestClearStaleLatestVersionPointerConcurrentWriter(t *testing.T) {
for _, tc := range []struct {
name string
observedEmpty bool
writerDuringScan bool
writerDuringWrite bool
wantCleared bool
wantPointer string
}{
{name: "idle key clears the stale pointer", wantCleared: true, wantPointer: ""},
{name: "concurrent writer during persist keeps its promotion", writerDuringWrite: true, wantCleared: false, wantPointer: "v2"},
{name: "pointerless snapshot keeps a post-rescan promotion", observedEmpty: true, writerDuringScan: true, wantCleared: false, wantPointer: "v2"},
{name: "pointerless snapshot on an idle key is already clear", observedEmpty: true, wantCleared: true, wantPointer: ""},
} {
t.Run(tc.name, func(t *testing.T) {
f := &anonymousReadHealFiler{anonymousReadFiler: &anonymousReadFiler{entries: make(map[string]*filer_pb.Entry)}}
extended := map[string][]byte{
s3_constants.ExtLatestVersionIdKey: []byte("v1"),
s3_constants.ExtLatestVersionFileNameKey: []byte("v_v1"),
}
if tc.observedEmpty {
extended = map[string][]byte{}
}
f.entries["/buckets/b/folder/.versions"] = &filer_pb.Entry{
Name: ".versions",
IsDirectory: true,
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000},
Extended: extended,
}
promotePointer := func(f *anonymousReadHealFiler) {
f.mu.Lock()
defer f.mu.Unlock()
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey] = []byte("v2")
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey] = []byte("v_v2")
}
if tc.writerDuringScan {
f.onScan = promotePointer
}
if tc.writerDuringWrite {
f.onWrite = promotePointer
}
s3a := newPutTestServer(t, startFakeFiler(t, f))
versionsEntry := proto.Clone(f.entries["/buckets/b/folder/.versions"]).(*filer_pb.Entry)
if versionsEntry.Extended == nil {
// proto.Clone turns an empty Extended map into nil; real callers
// (updateLatestVersionAfterDeletion) hand the clear a non-nil
// snapshot, so mirror that here.
versionsEntry.Extended = map[string][]byte{}
}
cleared := s3a.clearStaleLatestVersionPointer("b", "folder", "/buckets/b", "folder/.versions", versionsEntry, "test")
require.Equal(t, tc.wantCleared, cleared)
pointer := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey]
require.Equal(t, tc.wantPointer, string(pointer))
})
}
}