mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 23:07:48 +02:00
* fix(s3): make self-heal pointer persist CAS-bound against concurrent writers Follow-up to #11618: pointerless reads of a slash key whose regular-path entry is a physical parent (or a bare-key object) now fall through to healStaleLatestVersionPointer, which rescans .versions and persists a repaired pointer. The persist was an unconditional upsert off the pre-scan snapshot, so a PUT or delete that atomically advanced the pointer on the owner filer while the heal was rescanning could be rolled back, making older content or ACLs current again. Mirror the CAS discipline clearStaleLatestVersionPointer already applies: re-fetch the live .versions entry, require its pointer fields to still match the ones the heal observed, and abandon the persist (still returning the rescanned entry) when a concurrent writer has moved them. Write the live Extended map so concurrently updated fields are preserved. * fix(s3): close the check-then-act window in the self-heal pointer persist The CAS re-fetch added in the previous commit narrows the race but leaves a gateway-side window: after the live .versions entry is re-read and the pointer compared, the repair is still written back through an unconditional RPC, so a PUT or delete committing between the re-fetch and the persist still ends up rolled back by the stale repair. Bind the persist to the live image the heal just re-read with an IF_ENTRY_EQUAL precondition, the same discipline routedSelfCopy applies to stale self-copies: the filer evaluates the condition under the entry's path lock and conditional writes route to the owner filer, so a writer committing inside the window fails the precondition and the winner's pointer stands. FailedPrecondition and NotFound are authoritative replies and are not replayed by the failover layer. The test now also covers a writer committing during the persist, which reverts the pointer on the previous unconditional write-back. * s3api: CAS-bind the stale-pointer clear against concurrent writers The pointer clear re-read the live .versions entry and then wrote it back unconditionally through mkFile, so a writer committing between the re-fetch and the persist was rolled back to a cleared pointer. Persist through the same IF_ENTRY_EQUAL conditional update as the repair path. * s3api: test the CAS contract on the stale-pointer clear * s3api: never clear a pointer the clear did not observe as stale The CAS clear skipped its live-pointer match when the caller's snapshot carried an empty latest-version id, so a writer promoting a version between the clear's rescan and its re-fetch had the fresh pointer CAS-cleared away (expected = the writer's own live entry), briefly making the just-written version appear absent. With an empty observed id, reaching the persist at all implies a concurrent promotion (an idle key short-circuits as already-clear), so make the pointer match unconditional and abort instead. Extend TestClearStaleLatestVersionPointerConcurrentWriter with pointerless-snapshot cases: a post-rescan promotion must survive, and an idle pointerless key must short-circuit as already-clear. The fake filer's proto round-trip drops empty Extended maps, so the snapshot is padded the way real callers do. --------- Co-authored-by: zhaoyuchen <yc.zhao@yinzon.com> Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com>
215 lines
9.4 KiB
Go
215 lines
9.4 KiB
Go
package s3api
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"path"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
|
"github.com/stretchr/testify/require"
|
|
"google.golang.org/grpc/codes"
|
|
"google.golang.org/grpc/status"
|
|
"google.golang.org/protobuf/proto"
|
|
)
|
|
|
|
// anonymousReadHealFiler drives the versioned self-heal path for "folder/": it
|
|
// serves a real version file from .versions, runs hooks at the heal's rescan
|
|
// and at the persist (simulating a writer committing at either point), and
|
|
// evaluates the heal's IF_ENTRY_EQUAL precondition like a filer would, so
|
|
// tests can inspect the resulting .versions metadata.
|
|
type anonymousReadHealFiler struct {
|
|
*anonymousReadFiler
|
|
onScan func(*anonymousReadHealFiler)
|
|
onWrite func(*anonymousReadHealFiler)
|
|
}
|
|
|
|
// ListEntries serves the version files under .versions for the heal's rescan.
|
|
func (f *anonymousReadHealFiler) ListEntries(req *filer_pb.ListEntriesRequest, stream filer_pb.SeaweedFiler_ListEntriesServer) error {
|
|
f.mu.Lock()
|
|
if f.onScan != nil {
|
|
onScan := f.onScan
|
|
f.onScan = nil
|
|
f.mu.Unlock()
|
|
onScan(f)
|
|
} else {
|
|
f.mu.Unlock()
|
|
}
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
for _, entry := range f.entries {
|
|
if path.Join(req.Directory, entry.Name) == "/buckets/b/folder/.versions/v_v1" {
|
|
if err := stream.Send(&filer_pb.ListEntriesResponse{Entry: proto.Clone(entry).(*filer_pb.Entry)}); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// UpdateEntry evaluates the heal's IF_ENTRY_EQUAL precondition the way a
|
|
// filer does under the path lock: the stored entry must still equal the live
|
|
// image the heal re-read, otherwise the stale repair is rejected.
|
|
func (f *anonymousReadHealFiler) UpdateEntry(_ context.Context, req *filer_pb.UpdateEntryRequest) (*filer_pb.UpdateEntryResponse, error) {
|
|
f.mu.Lock()
|
|
if f.onWrite != nil {
|
|
onWrite := f.onWrite
|
|
f.onWrite = nil
|
|
f.mu.Unlock()
|
|
// A writer commits between the heal's re-fetch and the persist.
|
|
onWrite(f)
|
|
} else {
|
|
f.mu.Unlock()
|
|
}
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
fullPath := path.Join(req.Directory, req.Entry.Name)
|
|
current := f.entries[fullPath]
|
|
for _, clause := range req.Condition.Clauses {
|
|
if clause.Kind != filer_pb.WriteCondition_IF_ENTRY_EQUAL {
|
|
return nil, status.Error(codes.Unimplemented, "unsupported clause")
|
|
}
|
|
expected := proto.Clone(clause.ExpectedEntry).(*filer_pb.Entry)
|
|
actual := proto.Clone(current).(*filer_pb.Entry)
|
|
filer_pb.BeforeEntrySerialization(expected.Chunks)
|
|
filer_pb.BeforeEntrySerialization(actual.Chunks)
|
|
if !proto.Equal(actual, expected) {
|
|
return nil, status.Error(codes.FailedPrecondition, "entry changed")
|
|
}
|
|
}
|
|
f.entries[fullPath] = proto.Clone(req.Entry).(*filer_pb.Entry)
|
|
return &filer_pb.UpdateEntryResponse{}, nil
|
|
}
|
|
|
|
// TestAnonymousObjectACLHealPointerConcurrentWriter pins the CAS contract of
|
|
// the self-heal persist: a pointer that a concurrent writer promoted while the
|
|
// heal was rescanning, or between the heal's re-fetch and its persist, must
|
|
// survive, instead of being rolled back to the scanned version, which would
|
|
// make older content or ACLs current again.
|
|
func TestAnonymousObjectACLHealPointerConcurrentWriter(t *testing.T) {
|
|
for _, tc := range []struct {
|
|
name string
|
|
writerDuringScan bool
|
|
writerDuringWrite bool
|
|
wantPointer string
|
|
wantPointerVersion string
|
|
}{
|
|
{name: "idle key persists the rescanned pointer", wantPointer: "v1", wantPointerVersion: "v_v1"},
|
|
{name: "concurrent writer during scan keeps its promotion", writerDuringScan: true, wantPointer: "v2", wantPointerVersion: "v_v2"},
|
|
{name: "concurrent writer during persist keeps its promotion", writerDuringWrite: true, wantPointer: "v3", wantPointerVersion: "v_v3"},
|
|
} {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
f := &anonymousReadHealFiler{anonymousReadFiler: &anonymousReadFiler{entries: make(map[string]*filer_pb.Entry)}}
|
|
// The object is "folder/"; the regular path is a physical parent, so a
|
|
// pointerless read falls through to the persistent self-heal.
|
|
f.entries["/buckets/b/folder"] = &filer_pb.Entry{Name: "folder", IsDirectory: true, Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000}}
|
|
f.entries["/buckets/b/folder/child"] = anonymousReadEntry([]byte(`[]`))
|
|
f.entries["/buckets/b/folder/.versions"] = &filer_pb.Entry{Name: ".versions", IsDirectory: true, Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000}, Extended: map[string][]byte{s3_constants.ExtLatestVersionIdKey: []byte("")}}
|
|
v1 := anonymousReadEntry([]byte(`[]`))
|
|
v1.Name = "v_v1"
|
|
v1.Extended[s3_constants.ExtVersionIdKey] = []byte("v1")
|
|
f.entries["/buckets/b/folder/.versions/v_v1"] = v1
|
|
promotePointer := func(versionId, fileName string) func(*anonymousReadHealFiler) {
|
|
return func(f *anonymousReadHealFiler) {
|
|
// A concurrent PUT commits a newer version.
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey] = []byte(versionId)
|
|
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey] = []byte(fileName)
|
|
}
|
|
}
|
|
if tc.writerDuringScan {
|
|
f.onScan = promotePointer("v2", "v_v2")
|
|
}
|
|
if tc.writerDuringWrite {
|
|
f.onWrite = promotePointer("v3", "v_v3")
|
|
}
|
|
s3a := newPutTestServer(t, startFakeFiler(t, f))
|
|
s3a.iam = &IdentityAccessManagement{isAuthEnabled: true}
|
|
s3a.bucketConfigCache = NewBucketConfigCache(time.Minute)
|
|
s3a.bucketConfigCache.Set("b", &BucketConfig{Name: "b", Ownership: s3_constants.OwnershipObjectWriter, Versioning: "Enabled"})
|
|
s3a.policyEngine = NewBucketPolicyEngine()
|
|
s3a.iam.policyEngine = s3a.policyEngine
|
|
require.NoError(t, s3a.policyEngine.engine.SetBucketPolicy("b", `{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Principal":"*","Action":"s3:GetObject*","Resource":"arn:aws:s3:::b/folder/"}]}`))
|
|
|
|
rr := serveAnonymousRead(s3a, http.MethodGet, "folder/", "", nil)
|
|
require.Equal(t, http.StatusOK, rr.Code, rr.Body.String())
|
|
// This read returns the rescanned entry either way.
|
|
require.Equal(t, "hello world", rr.Body.String())
|
|
|
|
// The persisted pointer must reflect the concurrent writer, not the scan.
|
|
pointer := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey]
|
|
require.Equal(t, tc.wantPointer, string(pointer))
|
|
pointerFile := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey]
|
|
require.Equal(t, tc.wantPointerVersion, string(pointerFile))
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestClearStaleLatestVersionPointerConcurrentWriter pins the same CAS
|
|
// contract on the pointer clear: a writer that promotes the pointer between
|
|
// the clear's re-fetch and its persist must not be rolled back to a cleared
|
|
// pointer. A pointerless caller snapshot must not clear a pointer a writer
|
|
// promoted after the clear's rescan either — the only live pointer such a
|
|
// clear can observe is one that appeared concurrently, never the stale one
|
|
// the clear set out to remove.
|
|
func TestClearStaleLatestVersionPointerConcurrentWriter(t *testing.T) {
|
|
for _, tc := range []struct {
|
|
name string
|
|
observedEmpty bool
|
|
writerDuringScan bool
|
|
writerDuringWrite bool
|
|
wantCleared bool
|
|
wantPointer string
|
|
}{
|
|
{name: "idle key clears the stale pointer", wantCleared: true, wantPointer: ""},
|
|
{name: "concurrent writer during persist keeps its promotion", writerDuringWrite: true, wantCleared: false, wantPointer: "v2"},
|
|
{name: "pointerless snapshot keeps a post-rescan promotion", observedEmpty: true, writerDuringScan: true, wantCleared: false, wantPointer: "v2"},
|
|
{name: "pointerless snapshot on an idle key is already clear", observedEmpty: true, wantCleared: true, wantPointer: ""},
|
|
} {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
f := &anonymousReadHealFiler{anonymousReadFiler: &anonymousReadFiler{entries: make(map[string]*filer_pb.Entry)}}
|
|
extended := map[string][]byte{
|
|
s3_constants.ExtLatestVersionIdKey: []byte("v1"),
|
|
s3_constants.ExtLatestVersionFileNameKey: []byte("v_v1"),
|
|
}
|
|
if tc.observedEmpty {
|
|
extended = map[string][]byte{}
|
|
}
|
|
f.entries["/buckets/b/folder/.versions"] = &filer_pb.Entry{
|
|
Name: ".versions",
|
|
IsDirectory: true,
|
|
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000},
|
|
Extended: extended,
|
|
}
|
|
promotePointer := func(f *anonymousReadHealFiler) {
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey] = []byte("v2")
|
|
f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionFileNameKey] = []byte("v_v2")
|
|
}
|
|
if tc.writerDuringScan {
|
|
f.onScan = promotePointer
|
|
}
|
|
if tc.writerDuringWrite {
|
|
f.onWrite = promotePointer
|
|
}
|
|
s3a := newPutTestServer(t, startFakeFiler(t, f))
|
|
versionsEntry := proto.Clone(f.entries["/buckets/b/folder/.versions"]).(*filer_pb.Entry)
|
|
if versionsEntry.Extended == nil {
|
|
// proto.Clone turns an empty Extended map into nil; real callers
|
|
// (updateLatestVersionAfterDeletion) hand the clear a non-nil
|
|
// snapshot, so mirror that here.
|
|
versionsEntry.Extended = map[string][]byte{}
|
|
}
|
|
|
|
cleared := s3a.clearStaleLatestVersionPointer("b", "folder", "/buckets/b", "folder/.versions", versionsEntry, "test")
|
|
require.Equal(t, tc.wantCleared, cleared)
|
|
pointer := f.entries["/buckets/b/folder/.versions"].Extended[s3_constants.ExtLatestVersionIdKey]
|
|
require.Equal(t, tc.wantPointer, string(pointer))
|
|
})
|
|
}
|
|
}
|