Files
seaweedfs/weed/s3api/s3api_object_handlers_rename.go
T
Chris Lu 863fec6c3f S3: let a key that is a prefix of other keys be an object (#10912)
* filer: keep the sentinel when CreateEntry reports an update failure

CreateEntry flattened the error UpdateEntry wraps, so errors.Is stopped
matching and ErrExistingIsDirectory and ErrExistingIsFile never reached
the S3 mapper, which answered a retryable 500 instead.

* s3: let a key that is a prefix of other keys be an object

S3 keys are flat, so "a/b" and "a/b/c" are independent objects that
coexist in either write order. The filer stores a key as a path, so one
of them has to live on the directory the other is nested under.

Writing the nested key first refused the prefix key outright. Writing it
second promoted the file to a directory, which kept its data but lost the
key: an empty object left nothing to recognise it by and disappeared, and
one with data listed under a trailing slash it never had.

Mark the directory that carries such a key, and write the object onto it
when the path is already a directory. The mark makes an empty prefix
object visible to listings and readable by GET and HEAD, keeps the empty
folder cleaner off it, and lists it under the key it was written with.
Deleting the key strips the mark back off along with the data.

* filer: keep a TTL off a directory that stands for an object

An expired entry is deleted a row at a time, so expiring a directory
removes it and leaves everything under it unreachable. Promoting a file
to a directory carried its TTL across, and a promoted file is exactly the
one that has keys nested under it.

Drop the TTL on promotion, and leave one an older build wrote alone. The
lifecycle worker still expires the object, through the delete that leaves
the directory behind.

* s3: delete the null version of a key other keys are nested under

The routed delete cannot remove an entry that other keys live under, and
answered a retryable 500 rather than falling back to the lock path the
unversioned delete already falls back to. That path then looked the entry
up under the bucket with the whole key as its name, so the demote wrote it
back one directory too high and failed as not found.

Fall back on any non-precondition error, and split the key before deleting
it. Trailing-slash directory markers with children reach the same delete.

* filer: keep the sentinel when MkFile and Mkdir report a create failure

Same flattening one layer out: every mkFile caller lost the sentinel, so
a CopyObject onto a key that other keys are nested under answered a
retryable 500 where a PutObject of the same key answers 409.

* s3: copy and rename a key that other keys are nested under

Such a key is stored on the directory those keys live in, and copy and
rename both refused it: the source lookup maps every directory entry to
NoSuchKey, so a key a plain GET serves could not be copied or moved, and
the destination side refused it as a directory conflict.

The source is read through a view of the entry as the object it names.
The destination is written the way a PutObject of that key writes it. A
rename at either end copies the object's own data across and strips it off
the source key rather than going through AtomicRenameEntry, which moves a
directory by moving everything under it - the nested keys are not part of
what is being renamed.
2026-08-24 15:10:34 -07:00

298 lines
12 KiB
Go

package s3api
import (
"context"
"errors"
"net/http"
"net/url"
"strings"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/util"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
var renameSourceConditionalHeaders = sourceConditionalHeaderNames{
ifMatch: s3_constants.AmzRenameSourceIfMatch,
ifNoneMatch: s3_constants.AmzRenameSourceIfNoneMatch,
ifModifiedSince: s3_constants.AmzRenameSourceIfModifiedSince,
ifUnmodifiedSince: s3_constants.AmzRenameSourceIfUnmodifiedSince,
}
// RenameObjectHandler implements RenameObject:
//
// PUT /{bucket}/{destination key}?renameObject
// x-amz-rename-source: /{bucket}/{source key}
//
// The object is moved by the filer's AtomicRenameEntry, so its bytes are never
// read or rewritten and its metadata (ETag, tags, SSE keys) travels unchanged.
// Versioned buckets are rejected: the move would have to rebuild the .versions
// chain, and AWS itself only offers RenameObject on directory buckets, which
// cannot be versioned.
func (s3a *S3ApiServer) RenameObjectHandler(w http.ResponseWriter, r *http.Request) {
bucket, dstObject := s3_constants.GetBucketAndObject(r)
candidates, errCode := renameSourceCandidates(r, bucket)
if errCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, errCode)
return
}
srcObject := s3a.pickRenameSource(bucket, candidates)
glog.V(3).Infof("RenameObjectHandler %s: %s => %s", bucket, srcObject, dstObject)
if len(dstObject) > s3_constants.MaxS3ObjectKeyLength {
s3err.WriteErrorResponse(w, r, s3err.ErrKeyTooLongError)
return
}
if err := s3a.validateTableBucketObjectPath(bucket, dstObject); err != nil {
s3err.WriteErrorResponse(w, r, s3err.ErrAccessDenied)
return
}
// A trailing slash names a directory, and renaming one would move a whole
// subtree rather than an object.
if strings.HasSuffix(dstObject, "/") {
s3err.WriteErrorResponse(w, r, s3err.ErrInvalidRequest)
return
}
if strings.HasSuffix(srcObject, "/") {
s3err.WriteErrorResponse(w, r, s3err.ErrNoSuchKey)
return
}
if srcObject == dstObject {
s3err.WriteErrorResponse(w, r, s3err.ErrRenameDestinationSameAsSource)
return
}
// The route's Auth middleware only authorized the destination, because that
// is what the request URL names. The source arrives in a header and loses
// its key, so it needs both read and delete permission checked here.
if errCode := s3a.authorizeRenameSource(r, bucket, srcObject); errCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, errCode)
return
}
versioningState, err := s3a.getVersioningState(bucket)
if err != nil {
if errors.Is(err, filer_pb.ErrNotFound) {
s3err.WriteErrorResponse(w, r, s3err.ErrNoSuchBucket)
return
}
glog.Errorf("RenameObjectHandler: versioning state for bucket %s: %v", bucket, err)
s3err.WriteErrorResponse(w, r, s3err.ErrInternalError)
return
}
if versioningState != "" {
s3err.WriteErrorResponse(w, r, s3err.ErrNotImplemented)
return
}
errCode = s3a.withRenameWriteLocks(bucket, srcObject, dstObject, func() s3err.ErrorCode {
entry, err := s3a.resolveCopySourceEntry(bucket, srcObject, "", "")
srcIsPrefixObject := entry.IsPrefixObject()
entry = prefixObjectSource(entry)
if errCode := classifyCopySourceError(entry, err); errCode != s3err.ErrNone {
return errCode
}
if errCode := validateSourceConditionalHeaders(r, entry, renameSourceConditionalHeaders); errCode != s3err.ErrNone {
return errCode
}
if errCode := s3a.checkConditionalHeaders(r, bucket, dstObject); errCode != s3err.ErrNone {
return errCode
}
return s3a.renameObjectEntry(r.Context(), bucket, srcObject, dstObject, entry, srcIsPrefixObject)
})
if errCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, errCode)
return
}
stats_collect.RecordBucketActiveTime(bucket)
writeSuccessResponseEmpty(w, r)
}
// renameSourceCandidates reads x-amz-rename-source into the source keys it may
// mean, best guess first.
//
// AWS spells the source both ways: its CLI, Java and Rust examples pass a bare
// key, while a second CLI example and the boto3 conditional example pass
// bucket/key. A value is therefore read as a literal key first — that is the
// form AWS leads with, and it is the only reading that can never name the wrong
// object — and, when it is prefixed with the request's own bucket, as that
// bucket-qualified form second. There is no cross-bucket reading: RenameObject
// moves within one bucket, and the filer refuses to move an entry between two.
func renameSourceCandidates(r *http.Request, bucket string) ([]string, s3err.ErrorCode) {
rawSource := r.Header.Get(s3_constants.AmzRenameSource)
if rawSource == "" {
return nil, s3err.ErrInvalidRenameSource
}
// PathUnescape, not QueryUnescape: the value is a path, where '+' is a
// literal plus and not a space.
source, err := url.PathUnescape(rawSource)
if err != nil {
source = rawSource
}
// NormalizeObjectKey drops the leading slash both forms may carry.
source = s3_constants.NormalizeObjectKey(source)
if source == "" {
return nil, s3err.ErrInvalidRenameSource
}
candidates := []string{source}
if qualified := strings.TrimPrefix(source, bucket+"/"); qualified != source && qualified != "" {
candidates = append(candidates, qualified)
}
// `.`/`..` segments are collapsed by the filer's path join, so reject them
// here as the request URL's own key already is.
for _, candidate := range candidates {
if !s3_constants.IsValidObjectKey(candidate) {
return nil, s3err.ErrInvalidRenameSource
}
}
return candidates, s3err.ErrNone
}
// pickRenameSource resolves which reading of the source header the bucket
// actually holds. A single candidate is returned unprobed, so the common bare
// key costs no extra lookup; when both readings are possible the one the bucket
// holds wins, and when neither does the last is reported missing.
//
// Only a proven absence moves on to the next reading. A path that holds
// something the rename cannot move — a directory, say — is still the path the
// caller named, and answering for it beats renaming a different object under
// the other reading; so is a path whose lookup merely failed, since a blip must
// not be able to redirect a rename.
func (s3a *S3ApiServer) pickRenameSource(bucket string, candidates []string) string {
for _, candidate := range candidates[:len(candidates)-1] {
// A trailing slash never names an object, and never reaches a usable
// directory/name split either.
if strings.HasSuffix(candidate, "/") {
continue
}
if !renameSourceAbsent(s3a.resolveCopySourceEntry(bucket, candidate, "", "")) {
return candidate
}
}
return candidates[len(candidates)-1]
}
// renameSourceAbsent reports whether a lookup proved the candidate absent. Only
// the filer saying so counts; a lookup that failed for any other reason is not
// a proof of absence.
func renameSourceAbsent(entry *filer_pb.Entry, err error) bool {
if entry != nil {
return false
}
return err == nil || errors.Is(err, filer_pb.ErrNotFound) || status.Code(err) == codes.NotFound
}
func (s3a *S3ApiServer) authorizeRenameSource(r *http.Request, bucket, srcObject string) s3err.ErrorCode {
if s3a.iam == nil || !s3a.iam.isEnabled() {
return s3err.ErrNone
}
var identity *Identity
if id, ok := s3_constants.GetIdentityFromContext(r).(*Identity); ok {
identity = id
}
// The rename both reads the source object and removes it from its key.
if errCode := s3a.iam.AuthorizeCopySource(r, identity, bucket, srcObject, ""); errCode != s3err.ErrNone {
return errCode
}
return s3a.iam.AuthorizeObjectDelete(r, identity, bucket, srcObject, "")
}
// withRenameWriteLocks holds the object write lock of both keys across the
// precondition checks and the move. The keys are locked in a fixed order so a
// rename in the opposite direction cannot deadlock against this one.
func (s3a *S3ApiServer) withRenameWriteLocks(bucket, srcObject, dstObject string, fn func() s3err.ErrorCode) s3err.ErrorCode {
first, second := srcObject, dstObject
if second < first {
first, second = second, first
}
return s3a.withObjectWriteLock(bucket, first, nil, func() s3err.ErrorCode {
return s3a.withObjectWriteLock(bucket, second, nil, fn)
})
}
func (s3a *S3ApiServer) renameObjectEntry(ctx context.Context, bucket, srcObject, dstObject string, srcEntry *filer_pb.Entry, srcIsPrefixObject bool) s3err.ErrorCode {
srcDir, srcName := util.FullPath(s3a.toFilerPath(bucket, srcObject)).DirAndName()
dstDir, dstName := util.FullPath(s3a.toFilerPath(bucket, dstObject)).DirAndName()
// The move overwrites an existing destination object. A directory there is not a
// conflict: it means other keys are nested under the destination key, and the
// object goes onto the directory they live in, the way a PutObject of that key
// would put it there.
dstHoldsNestedKeys := false
if existing, err := s3a.getEntry(dstDir, dstName); err == nil {
dstHoldsNestedKeys = existing.IsDirectory
} else if !errors.Is(err, filer_pb.ErrNotFound) {
glog.Errorf("RenameObject %s: destination %s: %v", bucket, dstObject, err)
return s3err.ErrInternalError
}
// AtomicRenameEntry moves a directory by moving everything under it, and the keys
// nested under either end of this rename are not part of what is being renamed.
if srcIsPrefixObject || dstHoldsNestedKeys {
return s3a.renameKeyHoldingNestedKeys(bucket, srcObject, dstObject, srcEntry)
}
err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
_, err := client.AtomicRenameEntry(ctx, &filer_pb.AtomicRenameEntryRequest{
OldDirectory: srcDir,
OldName: srcName,
NewDirectory: dstDir,
NewName: dstName,
})
return err
})
if err != nil {
glog.Errorf("RenameObject %s: %s => %s: %v", bucket, srcObject, dstObject, err)
if isTransientFilerError(err) {
return s3err.ErrServiceUnavailable
}
return s3err.ErrInternalError
}
return s3err.ErrNone
}
// renameKeyHoldingNestedKeys moves an object when either key of the rename is one
// other keys are nested under. Such a key is stored on the directory those keys live
// in, which has to stay where it is, so the object's own data is written at the
// destination and then stripped off the source key - the entry survives as the plain
// directory it also is. Both keys are held under their write locks for the whole
// move, so no other S3 write interleaves; a crash between the two steps leaves the
// destination written and the source still there, which a retry settles.
func (s3a *S3ApiServer) renameKeyHoldingNestedKeys(bucket, srcObject, dstObject string, srcEntry *filer_pb.Entry) s3err.ErrorCode {
dstPath := util.FullPath(s3a.toFilerPath(bucket, dstObject))
dstDir, dstName := dstPath.DirAndName()
chunks, err := s3a.copyChunks(srcEntry, string(dstPath))
if err != nil {
glog.Errorf("RenameObject %s: copy chunks of %s: %v", bucket, srcObject, err)
return s3err.ErrInternalError
}
if err := s3a.mkFile(dstDir, dstName, chunks, func(entry *filer_pb.Entry) {
copyEntryToTarget(entry, srcEntry)
entry.Chunks = chunks
}); err != nil {
glog.Errorf("RenameObject %s: write %s: %v", bucket, dstObject, err)
s3a.deleteOrphanedChunks(chunks)
return filerErrorToS3Error(err)
}
// The destination holds copies now, so the source's own chunks go with it.
srcDir, srcName := util.FullPath(s3a.toFilerPath(bucket, srcObject)).DirAndName()
if err := s3a.rmObject(srcDir, srcName, true, false); err != nil {
glog.Errorf("RenameObject %s: strip %s: %v", bucket, srcObject, err)
return s3err.ErrInternalError
}
return s3err.ErrNone
}