Files
seaweedfs/weed/s3api/s3api_object_routed_write.go
T
zhao-ycandChris Lu 16e66b1bad fix(s3): initialize destination ACLs for CopyObject (#11599)
* fix(s3): initialize destination ACLs for CopyObject

Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com>

* fix(s3): re-check routed self-copy eligibility on the locked read

routeInPlace was decided on the pre-lock entry, but the PATCH body
re-reads the entry. A concurrent write changing file mode or MIME in
between left the routed PATCH installing new ACL keys while Attributes
kept the stale mode. Evaluate eligibility against the re-read entry and
retry the self-copy under the distributed lock when it no longer
qualifies.

* fix(s3): guard metadata self-copies against concurrent writes

Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com>

* ci: raise s3api unit-test timeout to 9m

The suite crossed the 5m binary timeout on the hosted runner (local run
is ~4.3m and still growing). The job-level limit is already 10m.

* ci: allow setup time before the S3 API test suite

Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com>

---------

Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com>
Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com>
2026-10-05 18:22:16 +08:00

347 lines
14 KiB
Go

package s3api
import (
"context"
"errors"
"fmt"
"net/http"
"path"
"strings"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
"github.com/seaweedfs/seaweedfs/weed/util"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// objectWriteRouteKeyPrefix namespaces an object's full path into the ring key
// used to resolve and forward its writes. Shared by every routed builder so the
// gateway, the admin dashboard and the filer hash the same key.
const objectWriteRouteKeyPrefix = s3_constants.ObjectWriteRouteKeyPrefix
// objectRouteKey is the ring key the gateway hashes to resolve an object's owner
// filer. It is also sent as route_key on each routed transaction, so a non-owner
// filer (reached because the gateway's ring view was stale) forwards the
// transaction to the owner. All of an object's writes share this key.
func (s3a *S3ApiServer) objectRouteKey(bucket, object string) string {
return objectWriteRouteKeyPrefix + s3a.toFilerPath(bucket, object)
}
// routableWriteOwner returns the owner filer for an object's writes, or "" to
// keep them on the distributed lock. All writes to one object (versioned,
// suspended, non-versioned) share the owner. Any lookup error falls back.
func (s3a *S3ApiServer) routableWriteOwner(bucket, object string) pb.ServerAddress {
if object == "" || s3a.objectWriteLockClient == nil {
return ""
}
// Object-lock PUTs route: a versioned PUT creates a new version (never an
// overwrite of a locked one), and a non-versioned overwrite is WORM-checked
// gateway-side before dispatch. WORM-checked deletes use routedObjectOwner.
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
}
// routedObjectOwner is routableWriteOwner restricted to non-versioned,
// non-object-lock buckets, for the unversioned DELETE fast path.
func (s3a *S3ApiServer) routedObjectOwner(bucket, object string) (pb.ServerAddress, bool) {
if configured, err := s3a.isVersioningConfigured(bucket); err != nil || configured {
return "", false
}
// An unversioned object-lock delete enforces WORM in the lock path; keep it
// on the lock rather than routing past the check.
if locked, err := s3a.isObjectLockEnabled(bucket); err != nil || locked {
return "", false
}
owner := s3a.routableWriteOwner(bucket, object)
return owner, owner != ""
}
// routeWriteCondition reduces the request's conditional headers for a routed
// create. A unique version path carries no precondition and only routes when the
// request is unconditional (a conditional versioned write must check the latest,
// which the lock path does); an overwrite carries the reduced condition.
func routeWriteCondition(r *http.Request, uniqueWritePath bool) (*filer_pb.WriteCondition, bool) {
cond, ok := buildWriteCondition(r)
if !ok {
return nil, false
}
if uniqueWritePath && cond != nil {
return nil, false
}
return cond, true
}
// buildWriteCondition reduces the request's conditional headers to a
// WriteCondition. ok=false (combined headers, time conditions, ETag lists, weak
// ETags) keeps gateway-side evaluation under the lock; a nil condition with
// ok=true means unconditional.
func buildWriteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
headers, errCode := parseConditionalHeaders(r)
if errCode != s3err.ErrNone {
return nil, false
}
if !headers.isSet {
return nil, true
}
if !headers.ifModifiedSince.IsZero() || !headers.ifUnmodifiedSince.IsZero() {
return nil, false
}
hasMatch := headers.ifMatch != ""
hasNoneMatch := headers.ifNoneMatch != ""
switch {
case hasMatch && !hasNoneMatch:
if headers.ifMatch == "*" {
return clause(filer_pb.WriteCondition_IF_EXISTS), true
}
if etag, single := singleStrongETag(headers.ifMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
}
return nil, false
case hasNoneMatch && !hasMatch:
if headers.ifNoneMatch == "*" {
return clause(filer_pb.WriteCondition_IF_NOT_EXISTS), true
}
if etag, single := singleStrongETag(headers.ifNoneMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, etag), true
}
return nil, false
default:
return nil, false
}
}
// buildDeleteCondition reduces a DeleteObject's If-Match header to a condition;
// DeleteObject honors only If-Match, matching checkDeleteIfMatch.
func buildDeleteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
ifMatch := strings.TrimSpace(r.Header.Get(s3_constants.IfMatch))
switch {
case ifMatch == "":
return nil, true
case ifMatch == "*":
return clause(filer_pb.WriteCondition_IF_EXISTS), true
default:
if etag, single := singleStrongETag(ifMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
}
return nil, false
}
}
func clause(kind filer_pb.WriteCondition_Kind) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind}}}
}
func etagClause(kind filer_pb.WriteCondition_Kind, etag string) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind, Etags: []string{etag}}}}
}
// singleStrongETag returns the normalized ETag when v carries exactly one strong
// ETag, and false for ETag lists or weak ("W/") ETags.
func singleStrongETag(v string) (string, bool) {
v = strings.TrimSpace(v)
if strings.Contains(v, ",") {
return "", false
}
if strings.HasPrefix(v, "W/") || strings.HasPrefix(v, "w/") {
return "", false
}
return strings.Trim(v, `"`), true
}
// objectTxnOnFiler applies a routed transaction, preferring the object's owner
// filer but failing over to a live filer when the owner is unreachable, so a
// restarted filer's stale ring address (e.g. a rolled K8s pod's old IP) can't hang
// the write. The reached filer forwards to the real owner by route_key, keeping its
// per-path lock authoritative. Mirrors getObjectEntryRoutedByKey on the read path.
func (s3a *S3ApiServer) objectTxnOnFiler(owner pb.ServerAddress, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
var resp *filer_pb.ObjectTransactionResponse
txn := func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.ObjectTransaction(context.Background(), req)
return e
}
if s3a.filerClient == nil {
err := pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, txn)
return resp, err
}
// Skip an owner whose recent write hit a transport error so a dead/stale
// address isn't re-dialed every request.
preferred := owner
if s3a.ownerRecentlyUnreachable(owner) {
preferred = ""
}
err := s3a.withFilerClientFailover(context.Background(), preferred, false, txn)
return resp, err
}
// routedPut writes an entry as a PUT, optionally followed by finalize mutations,
// as one ObjectTransaction applied in order under lockKey on the owner filer.
// lockKey is normally the entry's own path; a versioned add instead passes the
// object path plus a RECOMPUTE_LATEST finalize, so the version's PUT and its
// .versions pointer flip commit atomically (the recompute scans .versions/ after
// the PUT and sees the new version).
func (s3a *S3ApiServer) routedPut(owner pb.ServerAddress, routeKey, lockKey, filePath string, entry *filer_pb.Entry, cond *filer_pb.WriteCondition, conditionKey string, finalize []*filer_pb.ObjectMutation) (*filer_pb.ObjectTransactionResponse, error) {
mutations := make([]*filer_pb.ObjectMutation, 0, 1+len(finalize))
mutations = append(mutations, &filer_pb.ObjectMutation{
Type: filer_pb.ObjectMutation_PUT,
Directory: path.Dir(filePath),
Entry: entry,
})
mutations = append(mutations, finalize...)
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: lockKey,
RouteKey: routeKey,
Condition: cond,
ConditionKey: conditionKey,
Mutations: mutations,
})
}
// routedMkFile builds an entry like filer_pb.MkFile and writes it through a
// routed PUT on the owner filer, for callers that would otherwise mkFile to the
// default filer (e.g. multipart completion of a non-versioned object).
func (s3a *S3ApiServer) routedMkFile(owner pb.ServerAddress, routeKey, parentDir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry), removal *uploadRemovalTxn) error {
now := time.Now().Unix()
entry := &filer_pb.Entry{
Name: name,
Attributes: &filer_pb.FuseAttributes{
Mtime: now,
Crtime: now,
FileMode: uint32(0770),
Uid: filer_pb.OS_UID,
Gid: filer_pb.OS_GID,
},
Chunks: chunks,
}
if fn != nil {
fn(entry)
}
filePath := parentDir + "/" + name
var cond *filer_pb.WriteCondition
var conditionKey string
var finalize []*filer_pb.ObjectMutation
if removal != nil {
cond, conditionKey = removal.condition, removal.conditionKey
finalize = []*filer_pb.ObjectMutation{removal.mutation}
}
resp, err := s3a.routedPut(owner, routeKey, filePath, filePath, entry, cond, conditionKey, finalize)
if err != nil {
return err
}
if resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED {
return errUploadRemoved
}
if resp.Error != "" {
return fmt.Errorf("routed mkfile %s/%s: %s", parentDir, name, resp.Error)
}
return nil
}
// errUploadRemoved is a routed commit rejected because its precondition found
// the upload directory already deleted, mapped to NoSuchUpload by callers.
var errUploadRemoved = errors.New("upload directory already removed")
// writeMultipartObject writes a completed multipart object entry, routed to the
// owner when known (so it serializes with concurrent writes to the same key)
// and falling back to a plain mkFile otherwise. removal carries the upload
// precondition and deletion applied in the same transaction as the routed PUT.
// routeKey must be the same key the caller used to resolve owner, so owner
// selection and forwarding stay consistent.
func (s3a *S3ApiServer) writeMultipartObject(owner pb.ServerAddress, routeKey, dir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry), removal *uploadRemovalTxn) error {
if owner != "" {
return s3a.routedMkFile(owner, routeKey, dir, name, chunks, fn, removal)
}
return s3a.mkFile(dir, name, chunks, fn)
}
// uploadRemovalTxn carries the precondition and mutation that drop a completed
// upload's directory inside the object's commit transaction. The precondition
// fails the commit when a delete that does not take the object lock (abort,
// lifecycle, s3.clean.uploads) removed the directory first, instead of
// publishing the object over freed chunks.
type uploadRemovalTxn struct {
condition *filer_pb.WriteCondition
conditionKey string
mutation *filer_pb.ObjectMutation
}
// routedUploadRemoval returns the transaction parts that remove a completed
// upload's directory inside the object's commit, after freeing the part
// entries the object does not reference. It is nil when the write is not
// routed and the upload directory still needs post-commit cleanup.
func (s3a *S3ApiServer) routedUploadRemoval(ctx context.Context, owner pb.ServerAddress, uploadDirectory, bucket, uploadId string, completionState *multipartCompletionState) (*uploadRemovalTxn, error) {
if owner == "" {
return nil, nil
}
if err := s3a.deleteUnusedPartEntries(ctx, uploadDirectory, bucket, uploadId, completionState); err != nil {
return nil, err
}
conditionKey, condition := uploadExistsCondition(uploadDirectory)
return &uploadRemovalTxn{
condition: condition,
conditionKey: conditionKey,
mutation: s3a.removeUploadDirMutation(bucket, uploadId),
}, nil
}
func (s3a *S3ApiServer) routedDelete(owner pb.ServerAddress, bucket, object string, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
// NewFullPath normalizes a trailing-slash directory-marker key (e.g. "dir/")
// to the entry name "dir", matching deleteUnversionedObjectWithClient.
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
dir, name := fullpath.DirAndName()
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: string(fullpath),
RouteKey: s3a.objectRouteKey(bucket, object),
Condition: cond,
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_DELETE,
Directory: dir,
Name: name,
IsDeleteData: true,
}},
})
}
// routedSelfCopy updates metadata and attributes together, guarded by the raw
// source snapshot. UpdateEntry routes conditional writes to the owner and checks
// the condition under its path lock, so concurrent content, ACL or retention
// updates cannot be overwritten by a stale self-copy. Chunks are reused in place.
func (s3a *S3ApiServer) routedSelfCopy(ctx context.Context, owner pb.ServerAddress, bucket, object string, current, updated *filer_pb.Entry) error {
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
dir, _ := fullpath.DirAndName()
req := &filer_pb.UpdateEntryRequest{
Directory: dir,
Entry: updated,
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{
Kind: filer_pb.WriteCondition_IF_ENTRY_EQUAL,
ExpectedEntry: current,
}}},
}
var conditionErr error
update := func(client filer_pb.SeaweedFilerClient) error {
err := filer_pb.UpdateEntry(ctx, client, req)
switch status.Code(err) {
case codes.FailedPrecondition, codes.NotFound:
// These are authoritative replies from a healthy owner, not a
// transport failure. Do not replay them or mark the owner down.
conditionErr = err
return nil
default:
return err
}
}
var err error
if s3a.filerClient == nil {
err = pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, update)
} else {
err = s3a.withFilerClientFailover(ctx, owner, false, update)
}
if err != nil {
return err
}
return conditionErr
}