mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 14:31:57 +02:00
filer.remote.sync: do not pin the sync offset on completed work (#11569)
* filer.remote.sync: do not pin the sync offset on completed work * filer.remote.sync: a superseded rename uploads the current entry; typed NotFound for a stamp on a deleted entry * filer.remote.sync: a superseded rename keeps the old key when it is the only copy and uploads once * filer.remote.sync: a rename whose content is now remote-only fails the event instead of completing it * filer.remote.sync: a remote-only rename copies the old object to the destination before deleting it * filer.remote.sync: the remote-only rename path follows the filer's current entry and verifies the destination object * filer.remote.sync: an event that described an entry without data is superseded once the filer wrote to it * filer.remote.sync: a superseded rename does only the work left to do uploadCurrentEntry met a remote-only current entry with a fixed error, but a sync plus remote.uncache in the meantime leaves the destination holding the stamped object; that state is complete, not lost. The remote-only case now finishes through completeRemoteOnlyRename, which verifies the destination against the entry stamp and fails only when neither key holds the content. A current entry whose stamp covers its content was already uploaded by the superseding event; skip it instead of writing the same bytes again. * filer.remote.sync: an inherited stamp does not prove the content synced The stamp-coverage skip in uploadCurrentEntry read LastLocalSyncTsNs as proof the current content was uploaded, but a rename carries the source entry's stamp to the destination: a rewrite hidden by that stamp (the case the fallback upload exists for) carries a LastLocalSyncTsNs at or after its mtime and would have been skipped. Drop the check; the remote-only path verifies content at the destination itself through describes. --------- Co-authored-by: James Sas <james@medable.com> Co-authored-by: Chris Lu <chris.lu@gmail.com>
This commit is contained in:
5 files changed
+653
-41
No files matched your search
@@ -5,6 +5,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -171,7 +172,6 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
glog.V(0).Infof("mkdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.WriteDirectory(dest, remoteWriteEntry(message.NewEntry, *option.storageClass))
|
||||
}
|
||||
glog.V(0).Infof("create %s", remote_storage.FormatLocation(dest))
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, *option.storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
@@ -191,20 +191,7 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return updateLocalEntry(option, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
}
|
||||
if filer_pb.IsDelete(resp) {
|
||||
// Skip deletion of internal version files; individual version
|
||||
// deletes should not propagate to the remote object
|
||||
if isVersionedPath(resp.Directory, message.OldEntry.Name, message.OldEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping delete of internal version path: %s/%s", resp.Directory, message.OldEntry.Name)
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("delete: %+v", resp)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
if message.OldEntry.IsDirectory {
|
||||
glog.V(0).Infof("rmdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.RemoveDirectory(dest)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(dest))
|
||||
return client.DeleteFile(dest)
|
||||
return processDeleteEvent(client, mountedDir, remoteStorageMountLocation, resp)
|
||||
}
|
||||
if message.OldEntry != nil && message.NewEntry != nil {
|
||||
return processUpdateEvent(option, filerSource, *option.storageClass, client, mountedDir, remoteStorageMountLocation, resp)
|
||||
@@ -215,6 +202,45 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return eachEntryFunc, nil
|
||||
}
|
||||
|
||||
// processDeleteEvent removes the remote object an entry mapped to. A remote
|
||||
// object that is already absent counts as deleted: the entry was never
|
||||
// uploaded (created and deleted faster than the sync ran, or a replay of an
|
||||
// event the inline delete already handled), and GCS reports that case as
|
||||
// ErrRemoteObjectNotFound where S3 and Azure answer an idempotent success.
|
||||
// Returning the error would pin the sync offset on an event that has nothing
|
||||
// left to do.
|
||||
func processDeleteEvent(
|
||||
client remote_storage.RemoteStorageClient,
|
||||
mountedDir string,
|
||||
remoteStorageMountLocation *remote_pb.RemoteStorageLocation,
|
||||
resp *filer_pb.SubscribeMetadataResponse,
|
||||
) error {
|
||||
message := resp.EventNotification
|
||||
// Skip deletion of internal version files; individual version
|
||||
// deletes should not propagate to the remote object
|
||||
if isVersionedPath(resp.Directory, message.OldEntry.Name, message.OldEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping delete of internal version path: %s/%s", resp.Directory, message.OldEntry.Name)
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("delete: %+v", resp)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
if message.OldEntry.IsDirectory {
|
||||
glog.V(0).Infof("rmdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.RemoveDirectory(dest)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(dest))
|
||||
return deleteRemoteFile(client, dest)
|
||||
}
|
||||
|
||||
// deleteRemoteFile deletes the object and treats an already-absent object as
|
||||
// deleted.
|
||||
func deleteRemoteFile(client remote_storage.RemoteStorageClient, dest *remote_pb.RemoteStorageLocation) error {
|
||||
if err := client.DeleteFile(dest); err != nil && !errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func processUpdateEvent(
|
||||
filerClient filer_pb.FilerClient,
|
||||
filerSource filer_pb.FilerClient,
|
||||
@@ -256,12 +282,22 @@ func processUpdateEvent(
|
||||
}
|
||||
glog.V(0).Infof("never replicated, uploading %s", remote_storage.FormatLocation(dest))
|
||||
}
|
||||
if !proto.Equal(oldDest, dest) && !filer.HasData(message.NewEntry) && message.NewEntry.IsInRemoteOnly() {
|
||||
glog.V(0).Infof("skip uploading renamed remote-only entry %s: content is only on the deleted remote object", remote_storage.FormatLocation(dest))
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("update: %+v", resp)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
// A renamed entry that holds no local data now (the snapshot was
|
||||
// remote-only, or remote.uncache ran since) has its content only as a
|
||||
// remote object. Remote-only reads resolve by the entry's own path, so
|
||||
// the rename completes only once the destination object holds it. The
|
||||
// filer's current state decides, not the snapshot: an entry rewritten
|
||||
// since the event has local data again, and the rewrite's bytes are
|
||||
// what the destination must hold.
|
||||
current, err := currentEntry(filerSource, message.NewParentPath, message.NewEntry.Name)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isRemoteOnly(current) {
|
||||
return completeRemoteOnlyRename(filerClient, client, message.NewParentPath, current, oldDest, dest, storageClass)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(oldDest))
|
||||
if err := client.DeleteFile(oldDest); err != nil {
|
||||
if isMultipartUploadFile(resp.Directory, message.OldEntry.Name) {
|
||||
@@ -275,6 +311,9 @@ func processUpdateEvent(
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
return uploadCurrentEntry(filerClient, filerSource, client, message.NewParentPath, message.NewEntry.Name, oldDest, dest, storageClass)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
@@ -283,6 +322,128 @@ func processUpdateEvent(
|
||||
return updateLocalEntry(filerClient, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
}
|
||||
|
||||
// currentEntry returns what the filer holds at dir/name now, nil when the
|
||||
// entry is gone.
|
||||
func currentEntry(filerSource filer_pb.FilerClient, dir, name string) (*filer_pb.Entry, error) {
|
||||
current, _, _, err := filer_pb.GetEntry(context.Background(), filerSource, util.NewFullPath(dir, name))
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return current, nil
|
||||
}
|
||||
|
||||
// isRemoteOnly reports an entry whose content exists only as its remote
|
||||
// object: no local data, a RemoteEntry with a size.
|
||||
func isRemoteOnly(entry *filer_pb.Entry) bool {
|
||||
return entry != nil && !filer.HasData(entry) && entry.IsInRemoteOnly()
|
||||
}
|
||||
|
||||
// completeRemoteOnlyRename finishes a rename whose content exists only on the
|
||||
// remote. When the destination object already holds what the entry's stamp
|
||||
// describes (the rename ran before, its offset was not persisted, and
|
||||
// remote.uncache followed), the old key goes. Otherwise the old object is
|
||||
// copied over the destination, the entry is stamped with the copy, and the
|
||||
// old key goes. With neither object present the content is lost: the event
|
||||
// fails and holds the offset for recovery.
|
||||
func completeRemoteOnlyRename(filerClient filer_pb.FilerClient, client remote_storage.RemoteStorageClient, dir string, current *filer_pb.Entry, oldDest, dest *remote_pb.RemoteStorageLocation, storageClass string) error {
|
||||
if existing, err := client.StatFile(dest); err == nil {
|
||||
if describes(current.RemoteEntry, existing) {
|
||||
glog.V(0).Infof("%s already holds the renamed content", remote_storage.FormatLocation(dest))
|
||||
return deleteRemoteFile(client, oldDest)
|
||||
}
|
||||
glog.V(0).Infof("%s holds an object the entry does not describe (size %d, want %d); replacing it from %s", remote_storage.FormatLocation(dest), existing.RemoteSize, current.RemoteEntry.RemoteSize, remote_storage.FormatLocation(oldDest))
|
||||
} else if !errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return err
|
||||
}
|
||||
stat, err := client.StatFile(oldDest)
|
||||
if errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return fmt.Errorf("%s: content is on neither %s nor %s", util.NewFullPath(dir, current.Name), remote_storage.FormatLocation(oldDest), remote_storage.FormatLocation(dest))
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
reader, err := openRemoteObject(client, oldDest, stat.RemoteSize)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer reader.Close()
|
||||
glog.V(0).Infof("copy %s -> %s", remote_storage.FormatLocation(oldDest), remote_storage.FormatLocation(dest))
|
||||
remoteEntry, err := client.WriteFile(dest, remoteWriteEntry(current, storageClass), reader)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := updateLocalEntry(filerClient, dir, current, remoteEntry); err != nil {
|
||||
return err
|
||||
}
|
||||
return deleteRemoteFile(client, oldDest)
|
||||
}
|
||||
|
||||
// describes reports whether the object a stat returned is the one the entry's
|
||||
// stamp describes: same size, and the same ETag when both sides carry one (a
|
||||
// copy written as one stream can legitimately carry a different ETag from a
|
||||
// multipart original).
|
||||
func describes(stamp, object *filer_pb.RemoteEntry) bool {
|
||||
if stamp == nil || object == nil || stamp.RemoteSize != object.RemoteSize {
|
||||
return false
|
||||
}
|
||||
return stamp.RemoteETag == "" || object.RemoteETag == "" || stamp.RemoteETag == object.RemoteETag
|
||||
}
|
||||
|
||||
// openRemoteObject streams the object when the client can, and reads it whole
|
||||
// otherwise.
|
||||
func openRemoteObject(client remote_storage.RemoteStorageClient, loc *remote_pb.RemoteStorageLocation, size int64) (io.ReadCloser, error) {
|
||||
if streamer, ok := client.(remote_storage.RemoteStorageStreamReader); ok {
|
||||
return streamer.ReadFileAsStream(context.Background(), loc, 0, size)
|
||||
}
|
||||
data, err := client.ReadFile(loc, 0, size)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return io.NopCloser(bytes.NewReader(data)), nil
|
||||
}
|
||||
|
||||
// uploadCurrentEntry uploads what the filer holds at dir/name now, when the
|
||||
// event that superseded this rename will not. A rename whose snapshot is
|
||||
// superseded has already deleted the old key; the rewrite behind it in the
|
||||
// log uploads the destination itself unless shouldSendToRemote skips it on
|
||||
// the inherited RemoteEntry, whose RemoteMtime can equal the rewrite's mtime
|
||||
// within the same second. Only that case uploads here, so the content goes
|
||||
// up once. A remote-only entry is finished the way completeRemoteOnlyRename
|
||||
// finishes a rename: its stamp can already describe the destination (a sync
|
||||
// plus remote.uncache in the meantime), and only then is it complete.
|
||||
func uploadCurrentEntry(filerClient filer_pb.FilerClient, filerSource filer_pb.FilerClient, client remote_storage.RemoteStorageClient, dir, name string, oldDest, dest *remote_pb.RemoteStorageLocation, storageClass string) error {
|
||||
current, _, _, err := filer_pb.GetEntry(context.Background(), filerSource, util.NewFullPath(dir, name))
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if current.IsDirectory {
|
||||
return nil
|
||||
}
|
||||
if isRemoteOnly(current) {
|
||||
return completeRemoteOnlyRename(filerClient, client, dir, current, oldDest, dest, storageClass)
|
||||
}
|
||||
if shouldSendToRemote(current) {
|
||||
glog.V(0).Infof("leaving %s to the rewrite that superseded the rename", remote_storage.FormatLocation(dest))
|
||||
return nil
|
||||
}
|
||||
glog.V(0).Infof("uploading the current %s in place of the superseded rename", remote_storage.FormatLocation(dest))
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, dir, remoteWriteEntry(current, storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
return writeErr
|
||||
}
|
||||
return updateLocalEntry(filerClient, dir, current, remoteEntry)
|
||||
}
|
||||
|
||||
// isSuperseded reports whether the filer has moved past the entry an event
|
||||
// described: it is deleted, or it no longer references every chunk the event
|
||||
// named. Those are the chunks the filer deletes when an entry is updated, so
|
||||
@@ -304,6 +465,12 @@ func isSuperseded(filerClient filer_pb.FilerClient, dir string, entry *filer_pb.
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if !filer.HasData(entry) && filer.HasData(current) {
|
||||
// The event described an entry without data (remote-only, or empty);
|
||||
// the filer has written to it since. Uploading the snapshot would put
|
||||
// an empty object where the write belongs.
|
||||
return true
|
||||
}
|
||||
if len(entry.Content) > 0 || len(current.Content) > 0 {
|
||||
return !bytes.Equal(entry.Content, current.Content)
|
||||
}
|
||||
@@ -314,11 +481,18 @@ func isSuperseded(filerClient filer_pb.FilerClient, dir string, entry *filer_pb.
|
||||
// moved past (isSuperseded). The caller skips the event instead of failing it.
|
||||
var errSuperseded = errors.New("deleted or rewritten since the event was logged")
|
||||
|
||||
// retriedWriteFile uploads the entry, retrying transient failures. Every failed
|
||||
// attempt first asks the filer whether the entry is superseded, and stops at
|
||||
// once when it is: a dead chunk reads as a transient "RequestError" from the
|
||||
// SDK, and waiting out the backoff on it buys nothing.
|
||||
// retriedWriteFile uploads the entry, retrying transient failures. The filer is
|
||||
// asked whether the entry is superseded before the first attempt and after
|
||||
// every failed one, and the upload stops at once when it is. Before: a backlog
|
||||
// (a restart resuming from an old offset) carries every intermediate version
|
||||
// of a hot file, and uploading each one in turn is wasted bandwidth that can
|
||||
// trip the remote's per-object mutation rate limit; only the version the filer
|
||||
// still holds is worth sending. After: a dead chunk reads as a transient
|
||||
// "RequestError" from the SDK, and waiting out the backoff on it buys nothing.
|
||||
func retriedWriteFile(client remote_storage.RemoteStorageClient, filerSource filer_pb.FilerClient, dir string, newEntry *filer_pb.Entry, dest *remote_pb.RemoteStorageLocation) (remoteEntry *filer_pb.RemoteEntry, err error) {
|
||||
if isSuperseded(filerSource, dir, newEntry) {
|
||||
return nil, fmt.Errorf("%s %w", util.NewFullPath(dir, newEntry.Name), errSuperseded)
|
||||
}
|
||||
err = util.RetryOnError("writeFile", func(err error) bool {
|
||||
return !errors.Is(err, errSuperseded) && util.IsTransientError(err)
|
||||
}, func() error {
|
||||
@@ -453,9 +627,29 @@ func updateLocalEntry(filerClient filer_pb.FilerClient, dir string, entry *filer
|
||||
glog.Errorf("skipping stale stamp of %s: %v", util.NewFullPath(dir, entry.Name), err)
|
||||
return nil
|
||||
}
|
||||
if isEntryGone(err) {
|
||||
glog.Errorf("skipping stamp of %s: deleted since the event was logged: %v", util.NewFullPath(dir, entry.Name), err)
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// isEntryGone reports an UpdateEntry the filer refused because the entry no
|
||||
// longer exists: a delete that followed the event superseded the stamp, and
|
||||
// the delete's own event follows in the log. A filer with the typed answer
|
||||
// returns codes.NotFound; an older filer returns a plain error of the form
|
||||
// "not found <path>: <cause>", so the cause (the message's tail, never the
|
||||
// path) is matched against filer_pb.ErrNotFound's text.
|
||||
func isEntryGone(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
if st, ok := status.FromError(err); ok && st.Code() == codes.NotFound {
|
||||
return true
|
||||
}
|
||||
return strings.HasSuffix(strings.TrimSpace(err.Error()), filer_pb.ErrNotFound.Error())
|
||||
}
|
||||
|
||||
// ifEntryEqual builds the precondition that the stored entry still equals the
|
||||
// one the event described: chunk fids, inline content, and metadata alike.
|
||||
func ifEntryEqual(entry *filer_pb.Entry) *filer_pb.WriteCondition {
|
||||
@@ -501,7 +695,7 @@ func syncDeleteMarker(
|
||||
dest *remote_pb.RemoteStorageLocation,
|
||||
) error {
|
||||
glog.V(0).Infof("delete (marker) %s", remote_storage.FormatLocation(dest))
|
||||
if err := client.DeleteFile(dest); err != nil {
|
||||
if err := deleteRemoteFile(client, dest); err != nil {
|
||||
return err
|
||||
}
|
||||
return updateLocalEntry(filerClient, message.NewParentPath, message.NewEntry, &filer_pb.RemoteEntry{
|
||||
|
||||
Reference in new issue
Block a user