mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 14:31:57 +02:00
* fix(s3): initialize destination ACLs for CopyObject Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com> * fix(s3): re-check routed self-copy eligibility on the locked read routeInPlace was decided on the pre-lock entry, but the PATCH body re-reads the entry. A concurrent write changing file mode or MIME in between left the routed PATCH installing new ACL keys while Attributes kept the stale mode. Evaluate eligibility against the re-read entry and retry the self-copy under the distributed lock when it no longer qualifies. * fix(s3): guard metadata self-copies against concurrent writes Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com> * ci: raise s3api unit-test timeout to 9m The suite crossed the 5m binary timeout on the hosted runner (local run is ~4.3m and still growing). The job-level limit is already 10m. * ci: allow setup time before the S3 API test suite Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com> --------- Signed-off-by: zhaoyuchen <43179751+zhao-yc@users.noreply.github.com> Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com>
347 lines
14 KiB
Go
347 lines
14 KiB
Go
package s3api
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"net/http"
|
|
"path"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
"google.golang.org/grpc/codes"
|
|
"google.golang.org/grpc/status"
|
|
)
|
|
|
|
// objectWriteRouteKeyPrefix namespaces an object's full path into the ring key
|
|
// used to resolve and forward its writes. Shared by every routed builder so the
|
|
// gateway, the admin dashboard and the filer hash the same key.
|
|
const objectWriteRouteKeyPrefix = s3_constants.ObjectWriteRouteKeyPrefix
|
|
|
|
// objectRouteKey is the ring key the gateway hashes to resolve an object's owner
|
|
// filer. It is also sent as route_key on each routed transaction, so a non-owner
|
|
// filer (reached because the gateway's ring view was stale) forwards the
|
|
// transaction to the owner. All of an object's writes share this key.
|
|
func (s3a *S3ApiServer) objectRouteKey(bucket, object string) string {
|
|
return objectWriteRouteKeyPrefix + s3a.toFilerPath(bucket, object)
|
|
}
|
|
|
|
// routableWriteOwner returns the owner filer for an object's writes, or "" to
|
|
// keep them on the distributed lock. All writes to one object (versioned,
|
|
// suspended, non-versioned) share the owner. Any lookup error falls back.
|
|
func (s3a *S3ApiServer) routableWriteOwner(bucket, object string) pb.ServerAddress {
|
|
if object == "" || s3a.objectWriteLockClient == nil {
|
|
return ""
|
|
}
|
|
// Object-lock PUTs route: a versioned PUT creates a new version (never an
|
|
// overwrite of a locked one), and a non-versioned overwrite is WORM-checked
|
|
// gateway-side before dispatch. WORM-checked deletes use routedObjectOwner.
|
|
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
|
|
}
|
|
|
|
// routedObjectOwner is routableWriteOwner restricted to non-versioned,
|
|
// non-object-lock buckets, for the unversioned DELETE fast path.
|
|
func (s3a *S3ApiServer) routedObjectOwner(bucket, object string) (pb.ServerAddress, bool) {
|
|
if configured, err := s3a.isVersioningConfigured(bucket); err != nil || configured {
|
|
return "", false
|
|
}
|
|
// An unversioned object-lock delete enforces WORM in the lock path; keep it
|
|
// on the lock rather than routing past the check.
|
|
if locked, err := s3a.isObjectLockEnabled(bucket); err != nil || locked {
|
|
return "", false
|
|
}
|
|
owner := s3a.routableWriteOwner(bucket, object)
|
|
return owner, owner != ""
|
|
}
|
|
|
|
// routeWriteCondition reduces the request's conditional headers for a routed
|
|
// create. A unique version path carries no precondition and only routes when the
|
|
// request is unconditional (a conditional versioned write must check the latest,
|
|
// which the lock path does); an overwrite carries the reduced condition.
|
|
func routeWriteCondition(r *http.Request, uniqueWritePath bool) (*filer_pb.WriteCondition, bool) {
|
|
cond, ok := buildWriteCondition(r)
|
|
if !ok {
|
|
return nil, false
|
|
}
|
|
if uniqueWritePath && cond != nil {
|
|
return nil, false
|
|
}
|
|
return cond, true
|
|
}
|
|
|
|
// buildWriteCondition reduces the request's conditional headers to a
|
|
// WriteCondition. ok=false (combined headers, time conditions, ETag lists, weak
|
|
// ETags) keeps gateway-side evaluation under the lock; a nil condition with
|
|
// ok=true means unconditional.
|
|
func buildWriteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
|
|
headers, errCode := parseConditionalHeaders(r)
|
|
if errCode != s3err.ErrNone {
|
|
return nil, false
|
|
}
|
|
if !headers.isSet {
|
|
return nil, true
|
|
}
|
|
if !headers.ifModifiedSince.IsZero() || !headers.ifUnmodifiedSince.IsZero() {
|
|
return nil, false
|
|
}
|
|
hasMatch := headers.ifMatch != ""
|
|
hasNoneMatch := headers.ifNoneMatch != ""
|
|
switch {
|
|
case hasMatch && !hasNoneMatch:
|
|
if headers.ifMatch == "*" {
|
|
return clause(filer_pb.WriteCondition_IF_EXISTS), true
|
|
}
|
|
if etag, single := singleStrongETag(headers.ifMatch); single {
|
|
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
|
|
}
|
|
return nil, false
|
|
case hasNoneMatch && !hasMatch:
|
|
if headers.ifNoneMatch == "*" {
|
|
return clause(filer_pb.WriteCondition_IF_NOT_EXISTS), true
|
|
}
|
|
if etag, single := singleStrongETag(headers.ifNoneMatch); single {
|
|
return etagClause(filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, etag), true
|
|
}
|
|
return nil, false
|
|
default:
|
|
return nil, false
|
|
}
|
|
}
|
|
|
|
// buildDeleteCondition reduces a DeleteObject's If-Match header to a condition;
|
|
// DeleteObject honors only If-Match, matching checkDeleteIfMatch.
|
|
func buildDeleteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
|
|
ifMatch := strings.TrimSpace(r.Header.Get(s3_constants.IfMatch))
|
|
switch {
|
|
case ifMatch == "":
|
|
return nil, true
|
|
case ifMatch == "*":
|
|
return clause(filer_pb.WriteCondition_IF_EXISTS), true
|
|
default:
|
|
if etag, single := singleStrongETag(ifMatch); single {
|
|
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
|
|
}
|
|
return nil, false
|
|
}
|
|
}
|
|
|
|
func clause(kind filer_pb.WriteCondition_Kind) *filer_pb.WriteCondition {
|
|
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind}}}
|
|
}
|
|
|
|
func etagClause(kind filer_pb.WriteCondition_Kind, etag string) *filer_pb.WriteCondition {
|
|
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind, Etags: []string{etag}}}}
|
|
}
|
|
|
|
// singleStrongETag returns the normalized ETag when v carries exactly one strong
|
|
// ETag, and false for ETag lists or weak ("W/") ETags.
|
|
func singleStrongETag(v string) (string, bool) {
|
|
v = strings.TrimSpace(v)
|
|
if strings.Contains(v, ",") {
|
|
return "", false
|
|
}
|
|
if strings.HasPrefix(v, "W/") || strings.HasPrefix(v, "w/") {
|
|
return "", false
|
|
}
|
|
return strings.Trim(v, `"`), true
|
|
}
|
|
|
|
// objectTxnOnFiler applies a routed transaction, preferring the object's owner
|
|
// filer but failing over to a live filer when the owner is unreachable, so a
|
|
// restarted filer's stale ring address (e.g. a rolled K8s pod's old IP) can't hang
|
|
// the write. The reached filer forwards to the real owner by route_key, keeping its
|
|
// per-path lock authoritative. Mirrors getObjectEntryRoutedByKey on the read path.
|
|
func (s3a *S3ApiServer) objectTxnOnFiler(owner pb.ServerAddress, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
|
|
var resp *filer_pb.ObjectTransactionResponse
|
|
txn := func(client filer_pb.SeaweedFilerClient) error {
|
|
var e error
|
|
resp, e = client.ObjectTransaction(context.Background(), req)
|
|
return e
|
|
}
|
|
if s3a.filerClient == nil {
|
|
err := pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, txn)
|
|
return resp, err
|
|
}
|
|
// Skip an owner whose recent write hit a transport error so a dead/stale
|
|
// address isn't re-dialed every request.
|
|
preferred := owner
|
|
if s3a.ownerRecentlyUnreachable(owner) {
|
|
preferred = ""
|
|
}
|
|
err := s3a.withFilerClientFailover(context.Background(), preferred, false, txn)
|
|
return resp, err
|
|
}
|
|
|
|
// routedPut writes an entry as a PUT, optionally followed by finalize mutations,
|
|
// as one ObjectTransaction applied in order under lockKey on the owner filer.
|
|
// lockKey is normally the entry's own path; a versioned add instead passes the
|
|
// object path plus a RECOMPUTE_LATEST finalize, so the version's PUT and its
|
|
// .versions pointer flip commit atomically (the recompute scans .versions/ after
|
|
// the PUT and sees the new version).
|
|
func (s3a *S3ApiServer) routedPut(owner pb.ServerAddress, routeKey, lockKey, filePath string, entry *filer_pb.Entry, cond *filer_pb.WriteCondition, conditionKey string, finalize []*filer_pb.ObjectMutation) (*filer_pb.ObjectTransactionResponse, error) {
|
|
mutations := make([]*filer_pb.ObjectMutation, 0, 1+len(finalize))
|
|
mutations = append(mutations, &filer_pb.ObjectMutation{
|
|
Type: filer_pb.ObjectMutation_PUT,
|
|
Directory: path.Dir(filePath),
|
|
Entry: entry,
|
|
})
|
|
mutations = append(mutations, finalize...)
|
|
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
|
LockKey: lockKey,
|
|
RouteKey: routeKey,
|
|
Condition: cond,
|
|
ConditionKey: conditionKey,
|
|
Mutations: mutations,
|
|
})
|
|
}
|
|
|
|
// routedMkFile builds an entry like filer_pb.MkFile and writes it through a
|
|
// routed PUT on the owner filer, for callers that would otherwise mkFile to the
|
|
// default filer (e.g. multipart completion of a non-versioned object).
|
|
func (s3a *S3ApiServer) routedMkFile(owner pb.ServerAddress, routeKey, parentDir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry), removal *uploadRemovalTxn) error {
|
|
now := time.Now().Unix()
|
|
entry := &filer_pb.Entry{
|
|
Name: name,
|
|
Attributes: &filer_pb.FuseAttributes{
|
|
Mtime: now,
|
|
Crtime: now,
|
|
FileMode: uint32(0770),
|
|
Uid: filer_pb.OS_UID,
|
|
Gid: filer_pb.OS_GID,
|
|
},
|
|
Chunks: chunks,
|
|
}
|
|
if fn != nil {
|
|
fn(entry)
|
|
}
|
|
filePath := parentDir + "/" + name
|
|
var cond *filer_pb.WriteCondition
|
|
var conditionKey string
|
|
var finalize []*filer_pb.ObjectMutation
|
|
if removal != nil {
|
|
cond, conditionKey = removal.condition, removal.conditionKey
|
|
finalize = []*filer_pb.ObjectMutation{removal.mutation}
|
|
}
|
|
resp, err := s3a.routedPut(owner, routeKey, filePath, filePath, entry, cond, conditionKey, finalize)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
if resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED {
|
|
return errUploadRemoved
|
|
}
|
|
if resp.Error != "" {
|
|
return fmt.Errorf("routed mkfile %s/%s: %s", parentDir, name, resp.Error)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// errUploadRemoved is a routed commit rejected because its precondition found
|
|
// the upload directory already deleted, mapped to NoSuchUpload by callers.
|
|
var errUploadRemoved = errors.New("upload directory already removed")
|
|
|
|
// writeMultipartObject writes a completed multipart object entry, routed to the
|
|
// owner when known (so it serializes with concurrent writes to the same key)
|
|
// and falling back to a plain mkFile otherwise. removal carries the upload
|
|
// precondition and deletion applied in the same transaction as the routed PUT.
|
|
// routeKey must be the same key the caller used to resolve owner, so owner
|
|
// selection and forwarding stay consistent.
|
|
func (s3a *S3ApiServer) writeMultipartObject(owner pb.ServerAddress, routeKey, dir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry), removal *uploadRemovalTxn) error {
|
|
if owner != "" {
|
|
return s3a.routedMkFile(owner, routeKey, dir, name, chunks, fn, removal)
|
|
}
|
|
return s3a.mkFile(dir, name, chunks, fn)
|
|
}
|
|
|
|
// uploadRemovalTxn carries the precondition and mutation that drop a completed
|
|
// upload's directory inside the object's commit transaction. The precondition
|
|
// fails the commit when a delete that does not take the object lock (abort,
|
|
// lifecycle, s3.clean.uploads) removed the directory first, instead of
|
|
// publishing the object over freed chunks.
|
|
type uploadRemovalTxn struct {
|
|
condition *filer_pb.WriteCondition
|
|
conditionKey string
|
|
mutation *filer_pb.ObjectMutation
|
|
}
|
|
|
|
// routedUploadRemoval returns the transaction parts that remove a completed
|
|
// upload's directory inside the object's commit, after freeing the part
|
|
// entries the object does not reference. It is nil when the write is not
|
|
// routed and the upload directory still needs post-commit cleanup.
|
|
func (s3a *S3ApiServer) routedUploadRemoval(ctx context.Context, owner pb.ServerAddress, uploadDirectory, bucket, uploadId string, completionState *multipartCompletionState) (*uploadRemovalTxn, error) {
|
|
if owner == "" {
|
|
return nil, nil
|
|
}
|
|
if err := s3a.deleteUnusedPartEntries(ctx, uploadDirectory, bucket, uploadId, completionState); err != nil {
|
|
return nil, err
|
|
}
|
|
conditionKey, condition := uploadExistsCondition(uploadDirectory)
|
|
return &uploadRemovalTxn{
|
|
condition: condition,
|
|
conditionKey: conditionKey,
|
|
mutation: s3a.removeUploadDirMutation(bucket, uploadId),
|
|
}, nil
|
|
}
|
|
|
|
func (s3a *S3ApiServer) routedDelete(owner pb.ServerAddress, bucket, object string, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
|
|
// NewFullPath normalizes a trailing-slash directory-marker key (e.g. "dir/")
|
|
// to the entry name "dir", matching deleteUnversionedObjectWithClient.
|
|
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
|
|
dir, name := fullpath.DirAndName()
|
|
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
|
LockKey: string(fullpath),
|
|
RouteKey: s3a.objectRouteKey(bucket, object),
|
|
Condition: cond,
|
|
Mutations: []*filer_pb.ObjectMutation{{
|
|
Type: filer_pb.ObjectMutation_DELETE,
|
|
Directory: dir,
|
|
Name: name,
|
|
IsDeleteData: true,
|
|
}},
|
|
})
|
|
}
|
|
|
|
// routedSelfCopy updates metadata and attributes together, guarded by the raw
|
|
// source snapshot. UpdateEntry routes conditional writes to the owner and checks
|
|
// the condition under its path lock, so concurrent content, ACL or retention
|
|
// updates cannot be overwritten by a stale self-copy. Chunks are reused in place.
|
|
func (s3a *S3ApiServer) routedSelfCopy(ctx context.Context, owner pb.ServerAddress, bucket, object string, current, updated *filer_pb.Entry) error {
|
|
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
|
|
dir, _ := fullpath.DirAndName()
|
|
req := &filer_pb.UpdateEntryRequest{
|
|
Directory: dir,
|
|
Entry: updated,
|
|
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{
|
|
Kind: filer_pb.WriteCondition_IF_ENTRY_EQUAL,
|
|
ExpectedEntry: current,
|
|
}}},
|
|
}
|
|
var conditionErr error
|
|
update := func(client filer_pb.SeaweedFilerClient) error {
|
|
err := filer_pb.UpdateEntry(ctx, client, req)
|
|
switch status.Code(err) {
|
|
case codes.FailedPrecondition, codes.NotFound:
|
|
// These are authoritative replies from a healthy owner, not a
|
|
// transport failure. Do not replay them or mark the owner down.
|
|
conditionErr = err
|
|
return nil
|
|
default:
|
|
return err
|
|
}
|
|
}
|
|
var err error
|
|
if s3a.filerClient == nil {
|
|
err = pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, update)
|
|
} else {
|
|
err = s3a.withFilerClientFailover(ctx, owner, false, update)
|
|
}
|
|
if err != nil {
|
|
return err
|
|
}
|
|
return conditionErr
|
|
}
|