mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 14:31:57 +02:00
* fix(filer): persist pending chunk deletions across restarts The in-memory FileIdDeletionQueue and DeletionRetryQueue lose every queued-but-unconfirmed deletion when the filer process restarts. Because deletions only enter the pipeline through that queue, a crash between enqueue and the volume confirming the delete leaks the chunk permanently: nothing remembers it. In a multi-filer deployment this was observed as growing collections of orphaned chunks after filer restarts, and — via meta-replay from a peer that still had the entry — orphans being "resurrected" as live references on the recovered filer. This implements the "periodic snapshot with recovery on startup" option noted in the existing DeletionRetryQueue TODO, using the store's KV layer (no new iterator API required across the 15+ store backends): - queueDeletions() is the single entry point that keeps the hot in-memory queue and the durable ledger in sync. - Only terminal outcomes (success / not-found / permanent) remove an id from the ledger; retryable failures keep it, which is the point. - A timer and Shutdown() snapshot the pending set to a single KV key. - On startup, reloadDeletionLedger() re-queues recovered ids after a grace window so the initial peer meta-aggregation settles first. This avoids a new hazard: purging a chunk that a lagging peer is about to re-reference as live data (stale replay turns a stale read into a dangling read otherwise). - Volume deletes are idempotent (not-found == success), so re-deleting after a crash never double-frees. - Kill switch via viper: filer.deleteQueue.persist=false opts out entirely (reload also refuses to recover so a stale ledger never comes back). Tunables: filer.deleteQueue.persistInterval, .recoveryGrace. Adds unit tests covering snapshot+recover, retry-keeps-entry, disabled switch, and zero-value Filer safety (run green under -race). Co-Authored-By: Athena 🏛️ <hermes-agent@local> (custom / Qwen3.8-Flash-Next-ROCmFP4) * filer: harden the deletion ledger - Scope the ledger key by filer address so filers sharing one store do not overwrite each other's pending sets; ledgers written under the old unscoped key are claimed once on startup. - Serialize snapshots on deletionSnapshotLock so an in-flight timer snapshot cannot overwrite a newer shutdown snapshot, and wake the snapshotter on every queue/forget so a queued id persists within milliseconds instead of a full interval. - Merge recovered ids into the pending set immediately on reload; only the queue push waits out the grace window, so an early snapshot rewrites the recovered ids rather than dropping them. - A failed or unparseable ledger read blocks persistence for the run instead of letting snapshots overwrite the unread ledger. - Split the ledger into part keys when it exceeds one 64KB value so stores with a size cap (FoundationDB) do not strand the backlog. - GetReadyItems reports retry-exhausted ids so they are forgotten in the ledger instead of replaying after every restart. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: close the remaining deletion-ledger durability gaps - A manifest referencing a missing part is corruption: surface a wrapped error and block persistence instead of treating the ledger as absent. - Multipart snapshots write generation-scoped part keys and publish the manifest last, so a crash never mixes old and new part contents. - Orphaned parts are tracked in a persisted .stale sidecar and retried. - Legacy/index ledgers are republished under the scoped key before the old keys are removed. - A ledger index lets a filer restart under a new address claim the ledger its previous incarnation left behind. - Expired and permanently-failed retry items only forget the ledger epoch they recorded, so they cannot erase a re-queued id. - A failed startup read no longer disables persistence: every snapshot retries the reload until the store reads again. * filer: tighten ledger claiming, index updates, and retry epochs - touchLedgerIndex verifies its write and retries so a concurrent filer's merge cannot silently drop this key from the index. - Foreign-ledger claims abort on any unreadable source instead of leaving it stranded once the new scoped key exists. - A source that republished during the claim is left in place and its newer ids merge into the claimant's pending set. - AddOrUpdate no longer overwrites the ledger epoch of an in-flight retry item, so its expiry or permanent outcome cannot forget a record that was re-queued after the attempt began. - The recovery grace wait exits on shutdown instead of re-queueing after the filer has stopped. * filer: requeue surviving records, persist claim deltas, guard index writes - A dropped retry item (expired or permanent) whose ledger record was re-enqueued now pushes the id back through the hot queue instead of leaving it pending with nothing scheduled. - Ids merged from a claim source that republished mid-claim are rewritten under our ledger immediately, so they are durable even if the claimant crashes before the next snapshot. - touchLedgerIndex aborts when the index read fails for a real error; only ErrKvNotFound means the index is empty, so a transient failure can no longer wipe peer entries with a one-key write. --------- Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com> Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
846 lines
32 KiB
Go
846 lines
32 KiB
Go
package filer
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"sort"
|
|
"strings"
|
|
"sync"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/remote_storage"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3bucket"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
|
"github.com/seaweedfs/seaweedfs/weed/filer/empty_folder_cleanup"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/remote_pb"
|
|
|
|
"google.golang.org/grpc"
|
|
"google.golang.org/protobuf/proto"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
|
|
"github.com/seaweedfs/seaweedfs/weed/wdclient"
|
|
"golang.org/x/sync/singleflight"
|
|
)
|
|
|
|
const (
|
|
LogFlushInterval = time.Minute
|
|
PaginationSize = 1024
|
|
FilerStoreId = "filer.store.id"
|
|
)
|
|
|
|
var (
|
|
OS_UID = uint32(os.Getuid())
|
|
OS_GID = uint32(os.Getgid())
|
|
)
|
|
|
|
type Filer struct {
|
|
UniqueFilerId int32
|
|
UniqueFilerEpoch int32
|
|
Store VirtualFilerStore
|
|
MasterClient *wdclient.MasterClient
|
|
FileIdDeletionQueue *util.UnboundedQueue
|
|
GrpcDialOption grpc.DialOption
|
|
DirBucketsPath string
|
|
Cipher bool
|
|
LocalMetaLogBuffer *log_buffer.LogBuffer
|
|
metaLogCollection string
|
|
metaLogReplication string
|
|
// Override where system metadata-log chunks are assigned, keeping the
|
|
// internal log out of the default collection; empty keeps today's
|
|
// behaviour. Set via viper: filer.options.metaLog.collection / .replication.
|
|
metaLogTargetCollection string
|
|
metaLogTargetReplication string
|
|
DefaultDiskType string
|
|
MetaAggregator *MetaAggregator
|
|
Signature int32
|
|
FilerConf *FilerConf
|
|
placementOverlay PlacementOverlay
|
|
RemoteStorage *FilerRemoteStorage
|
|
BuildGuardedRemoteClient RemoteStorageClientBuilder
|
|
AllowUntrustedRemoteEndpoints bool
|
|
lazyFetchGroup singleflight.Group
|
|
lazyListGroup singleflight.Group
|
|
Dlm *lock_manager.DistributedLockManager
|
|
MaxFilenameLength uint32
|
|
deletionQuit chan struct{}
|
|
DeletionRetryQueue *DeletionRetryQueue
|
|
EmptyFolderCleaner *empty_folder_cleanup.EmptyFolderCleaner
|
|
EmptyFolderCleanupDelay time.Duration
|
|
persistedLogCache *persistedLogCache
|
|
metaLogInflight metaLogInflight
|
|
remoteTombstones *remoteDeletionTombstones
|
|
// remoteTombstonesDone, when non-nil, is closed once the startup tombstone
|
|
// rebuild finishes; lazy remote reads wait on it so a pending delete
|
|
// cannot resurrect in the gap.
|
|
remoteTombstonesDone atomic.Pointer[chan struct{}]
|
|
|
|
// Durable deletion ledger (see filer_deletion_persist.go). The set of
|
|
// fileIds that still need deleting but are not yet confirmed gone, mirrored
|
|
// to the store so a restart does not leak chunks. Guarded by
|
|
// deletionLedgerLock; nil-safe for Filer literals in tests.
|
|
// pendingDeletions maps each id to its enqueue epoch so an expiry-forget
|
|
// cannot erase a re-queued id.
|
|
deletionLedgerLock sync.Mutex
|
|
pendingDeletions map[string]uint64
|
|
deletionSeq uint64
|
|
deletionLedgerDirty bool
|
|
deletionLedgerParts int
|
|
deletionLedgerGen int
|
|
// deletionLedgerStale holds orphan part keys from abandoned multipart
|
|
// writes, retried on the next snapshot.
|
|
deletionLedgerStale []string
|
|
// deletionSnapshotLock serializes ledger writes in copy order so an
|
|
// in-flight timer snapshot cannot overwrite a newer shutdown snapshot.
|
|
deletionSnapshotLock sync.Mutex
|
|
// deletionLedgerBlocked is set when the startup ledger read fails: the
|
|
// persisted set is then unknown, so snapshots retry the read instead of
|
|
// overwriting the unread ledger with a partial set.
|
|
deletionLedgerBlocked atomic.Bool
|
|
deletionLedgerFlush chan struct{}
|
|
}
|
|
|
|
func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerHost pb.ServerAddress, filerGroup string, collection string, replication string, dataCenter string, maxFilenameLength uint32, notifyFn func()) *Filer {
|
|
f := &Filer{
|
|
MasterClient: wdclient.NewMasterClient(grpcDialOption, filerGroup, cluster.FilerType, filerHost, dataCenter, "", masters),
|
|
FileIdDeletionQueue: util.NewUnboundedQueue(),
|
|
GrpcDialOption: grpcDialOption,
|
|
FilerConf: NewFilerConf(),
|
|
RemoteStorage: NewFilerRemoteStorage(),
|
|
UniqueFilerId: util.RandomInt32(),
|
|
Dlm: lock_manager.NewDistributedLockManager(filerHost),
|
|
MaxFilenameLength: maxFilenameLength,
|
|
deletionQuit: make(chan struct{}),
|
|
deletionLedgerFlush: make(chan struct{}, 1),
|
|
DeletionRetryQueue: NewDeletionRetryQueue(),
|
|
persistedLogCache: newPersistedLogCache(persistedLogCacheMaxBytes),
|
|
remoteTombstones: newRemoteDeletionTombstones(),
|
|
}
|
|
if f.UniqueFilerId < 0 {
|
|
f.UniqueFilerId = -f.UniqueFilerId
|
|
}
|
|
|
|
// ReadFromDiskFn is intentionally nil here. SubscribeLocalMetadata already
|
|
// manages disk reads explicitly with shouldReadFromDisk / lastCheckedFlushTsNs
|
|
// tracking. Setting ReadFromDiskFn would cause LoopProcessLogData to issue a
|
|
// redundant ReadPersistedLogBuffer call (ListDirectoryEntries + readahead
|
|
// goroutine) on every 250ms health-check tick when a subscriber encounters
|
|
// ResumeFromDiskError, adding significant CPU and GC pressure even when idle.
|
|
f.LocalMetaLogBuffer = log_buffer.NewLogBuffer("local", LogFlushInterval, f.logFlushFunc, nil, notifyFn)
|
|
f.metaLogCollection = collection
|
|
f.metaLogReplication = replication
|
|
|
|
// Optional override for where the system metadata-log chunks land, so
|
|
// operators can keep internal log volumes out of the default collection.
|
|
// Unset (""), this changes nothing: meta logs keep following the filer
|
|
// default exactly as before.
|
|
v := util.GetViper()
|
|
v.SetDefault("filer.options.metaLog.collection", "")
|
|
v.SetDefault("filer.options.metaLog.replication", "")
|
|
f.metaLogTargetCollection = v.GetString("filer.options.metaLog.collection")
|
|
f.metaLogTargetReplication = v.GetString("filer.options.metaLog.replication")
|
|
if f.metaLogTargetCollection != "" {
|
|
glog.V(0).Infof("system metadata logs will be stored in collection %q", f.metaLogTargetCollection)
|
|
}
|
|
|
|
if newPlacementOverlay != nil {
|
|
f.placementOverlay = newPlacementOverlay(f)
|
|
}
|
|
|
|
go f.loopProcessingDeletion()
|
|
|
|
return f
|
|
}
|
|
|
|
func (f *Filer) buildRemoteStorageClient(ctx context.Context, remoteConf *remote_pb.RemoteConf) (remote_storage.RemoteStorageClient, error) {
|
|
if f.BuildGuardedRemoteClient != nil {
|
|
return f.BuildGuardedRemoteClient(ctx, remoteConf, f.AllowUntrustedRemoteEndpoints)
|
|
}
|
|
return remote_storage.GetRemoteStorage(remoteConf)
|
|
}
|
|
|
|
func (f *Filer) MaybeBootstrapFromOnePeer(self pb.ServerAddress, existingNodes []*master_pb.ClusterNodeUpdate, snapshotTime time.Time) (err error) {
|
|
if len(existingNodes) == 0 {
|
|
return
|
|
}
|
|
sort.Slice(existingNodes, func(i, j int) bool {
|
|
return existingNodes[i].CreatedAtNs < existingNodes[j].CreatedAtNs
|
|
})
|
|
earliestNode := existingNodes[0]
|
|
if pb.ServerAddress(earliestNode.Address).Equals(self) {
|
|
return
|
|
}
|
|
|
|
glog.V(0).Infof("bootstrap from %v clientId:%d", earliestNode.Address, f.UniqueFilerId)
|
|
|
|
return pb.WithFilerClient(false, f.UniqueFilerId, pb.ServerAddress(earliestNode.Address), f.GrpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
|
return filer_pb.StreamBfs(client, "/", snapshotTime.UnixNano(), func(parentPath util.FullPath, entry *filer_pb.Entry) error {
|
|
return f.Store.InsertEntry(context.Background(), FromPbEntry(string(parentPath), entry))
|
|
})
|
|
})
|
|
|
|
}
|
|
|
|
func (f *Filer) AggregateFromPeers(self pb.ServerAddress, existingNodes []*master_pb.ClusterNodeUpdate, startFrom time.Time) {
|
|
|
|
var snapshot []pb.ServerAddress
|
|
for _, node := range existingNodes {
|
|
address := pb.ServerAddress(node.Address)
|
|
snapshot = append(snapshot, address)
|
|
}
|
|
f.Dlm.LockRing.SetSnapshot(snapshot, 0)
|
|
glog.V(0).Infof("%s aggregate from peers %+v", self, snapshot)
|
|
|
|
// Initialize the empty folder cleaner using the same LockRing as Dlm for consistent hashing
|
|
f.EmptyFolderCleaner = empty_folder_cleanup.NewEmptyFolderCleaner(f, f.Dlm.LockRing, self, f.DirBucketsPath, f.EmptyFolderCleanupDelay)
|
|
|
|
f.MetaAggregator = NewMetaAggregator(f, self, f.GrpcDialOption)
|
|
// The ring starts empty while peer history sits on disk: mark the pre-startFrom
|
|
// range evicted so a cursor there reads disk, not the ring's earliest entry.
|
|
f.MetaAggregator.MetaLogBuffer.MarkEvictedThrough(startFrom.UnixNano())
|
|
f.MasterClient.SetOnPeerUpdateFn(func(update *master_pb.ClusterNodeUpdate, startFrom time.Time) {
|
|
if update.NodeType != cluster.FilerType {
|
|
return
|
|
}
|
|
// Lock ring is now managed by the master via LockRingUpdate,
|
|
// so we no longer call AddServer/RemoveServer here.
|
|
f.MetaAggregator.OnPeerUpdate(update, startFrom)
|
|
})
|
|
f.MasterClient.SetOnLockRingUpdateFn(func(update *master_pb.LockRingUpdate) {
|
|
var servers []pb.ServerAddress
|
|
for _, s := range update.Servers {
|
|
servers = append(servers, pb.ServerAddress(s))
|
|
}
|
|
glog.V(0).Infof("LockRing: applying master ring update v%d: %v", update.Version, servers)
|
|
f.Dlm.LockRing.SetSnapshot(servers, update.Version)
|
|
})
|
|
f.MasterClient.SetOnMasterChangeFn(func(previous, current pb.ServerAddress) {
|
|
glog.V(0).Infof("LockRing: master changed %s -> %s, resetting ring", previous, current)
|
|
f.Dlm.LockRing.Reset()
|
|
})
|
|
|
|
// Subscribe to the local filer first: its events reach the aggregated
|
|
// buffer only through this subscription, and the peer watermarks must
|
|
// account for it before any remote peer - a remotes-only watermark set
|
|
// would claim completeness without self. existingNodes can omit self
|
|
// (master registration races this bootstrap); duplicate adds are no-ops.
|
|
f.MetaAggregator.OnPeerUpdate(&master_pb.ClusterNodeUpdate{
|
|
NodeType: cluster.FilerType,
|
|
Address: string(self),
|
|
IsAdd: true,
|
|
}, startFrom)
|
|
for _, peerUpdate := range existingNodes {
|
|
f.MetaAggregator.OnPeerUpdate(peerUpdate, startFrom)
|
|
}
|
|
|
|
}
|
|
|
|
func (f *Filer) ListExistingPeerUpdates(ctx context.Context) (existingNodes []*master_pb.ClusterNodeUpdate) {
|
|
return cluster.ListExistingPeerUpdates(f.GetMaster(ctx), f.GrpcDialOption, f.MasterClient.FilerGroup, cluster.FilerType)
|
|
}
|
|
|
|
func (f *Filer) SetStore(store FilerStore) (isFresh bool) {
|
|
f.Store = NewFilerStoreWrapper(store)
|
|
|
|
isFresh = f.setOrLoadFilerStoreSignature(store)
|
|
|
|
// Recover deletions that were pending when a previous process died, and keep
|
|
// the durable ledger snapshotted while running (see filer_deletion_persist.go).
|
|
// A failed ledger read leaves writes blocked until a snapshot retries and
|
|
// the read succeeds, so the unread ledger is never overwritten.
|
|
f.reloadDeletionLedger()
|
|
f.startDeletionLedgerSnapshotter()
|
|
|
|
return isFresh
|
|
}
|
|
|
|
func (f *Filer) setOrLoadFilerStoreSignature(store FilerStore) (isFresh bool) {
|
|
storeIdBytes, err := store.KvGet(context.Background(), []byte(FilerStoreId))
|
|
if err == ErrKvNotFound || err == nil && len(storeIdBytes) == 0 {
|
|
f.Signature = util.RandomInt32()
|
|
storeIdBytes = make([]byte, 4)
|
|
util.Uint32toBytes(storeIdBytes, uint32(f.Signature))
|
|
if err = store.KvPut(context.Background(), []byte(FilerStoreId), storeIdBytes); err != nil {
|
|
glog.Fatalf("set %s=%d : %v", FilerStoreId, f.Signature, err)
|
|
}
|
|
glog.V(0).Infof("create %s to %d", FilerStoreId, f.Signature)
|
|
return true
|
|
} else if err == nil && len(storeIdBytes) == 4 {
|
|
f.Signature = int32(util.BytesToUint32(storeIdBytes))
|
|
glog.V(0).Infof("existing %s = %d", FilerStoreId, f.Signature)
|
|
} else {
|
|
glog.Fatalf("read %v=%v : %v", FilerStoreId, string(storeIdBytes), err)
|
|
}
|
|
return false
|
|
}
|
|
|
|
func (f *Filer) GetStore() (store FilerStore) {
|
|
return f.Store
|
|
}
|
|
|
|
func (fs *Filer) GetMaster(ctx context.Context) pb.ServerAddress {
|
|
return fs.MasterClient.GetMaster(ctx)
|
|
}
|
|
|
|
func (f *Filer) BeginTransaction(ctx context.Context) (context.Context, error) {
|
|
return f.Store.BeginTransaction(ctx)
|
|
}
|
|
|
|
func (f *Filer) CommitTransaction(ctx context.Context) error {
|
|
return f.Store.CommitTransaction(ctx)
|
|
}
|
|
|
|
func (f *Filer) RollbackTransaction(ctx context.Context) error {
|
|
return f.Store.RollbackTransaction(ctx)
|
|
}
|
|
|
|
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
|
|
// has already fetched the current entry at this path under a path lock, and it
|
|
// is reused instead of looking the store up again; pass nil to have CreateEntry
|
|
// look it up itself.
|
|
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
|
|
|
if string(entry.FullPath) == "/" {
|
|
return nil
|
|
}
|
|
|
|
if entry.FullPath.IsLongerFileName(maxFilenameLength) {
|
|
return filer_pb.ErrEntryNameTooLong
|
|
}
|
|
|
|
if entry.IsDirectory() {
|
|
entry.Attr.TtlSec = 0
|
|
}
|
|
|
|
if entry.Attr.Atime.IsZero() {
|
|
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
|
}
|
|
|
|
oldEntry := existing
|
|
if oldEntry == nil {
|
|
var findErr error
|
|
oldEntry, findErr = f.FindEntry(ctx, entry.FullPath)
|
|
if o_excl && findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
|
// An exclusive create cannot decide whether the path exists when
|
|
// the lookup itself failed; proceeding would upsert over it.
|
|
return fmt.Errorf("find entry %s: %w", entry.FullPath, findErr)
|
|
}
|
|
}
|
|
|
|
/*
|
|
if !hasWritePermission(lastDirectoryEntry, entry) {
|
|
glog.V(0).Infof("directory %s: %v, entry: uid=%d gid=%d",
|
|
lastDirectoryEntry.FullPath, lastDirectoryEntry.Attr, entry.Uid, entry.Gid)
|
|
return fmt.Errorf("no write permission in folder %v", lastDirectoryEntry.FullPath)
|
|
}
|
|
*/
|
|
|
|
if oldEntry == nil {
|
|
f.ensureEntryInode(entry)
|
|
|
|
glog.V(4).InfofCtx(ctx, "InsertEntry %s: new entry: %v", entry.FullPath, entry.Name())
|
|
if err := f.Store.InsertEntry(ctx, entry); err != nil {
|
|
glog.ErrorfCtx(ctx, "insert entry %s: %v", entry.FullPath, err)
|
|
return fmt.Errorf("insert entry %s: %v", entry.FullPath, err)
|
|
}
|
|
|
|
// Parents go after the entry: one checked first can be taken by the
|
|
// empty-folder cleaner before the entry lands, and nothing would look again.
|
|
if !skipCreateParentDir {
|
|
dirParts := strings.Split(string(entry.FullPath), "/")
|
|
if err := f.ensureParentDirectoryEntry(ctx, entry, dirParts, len(dirParts)-1, isFromOtherCluster); err != nil {
|
|
// The entry stays: deleting by path would destroy a concurrent create
|
|
// that already succeeded through the update branch, and the update keeps
|
|
// the inode and crtime while mtime only survives the store to the second.
|
|
// It is not announced - the aggregator replicates creates into peer
|
|
// stores, so a failed write would land on all of them rather than on the
|
|
// one filer whose next write into that folder repairs it.
|
|
glog.ErrorfCtx(ctx, "create parent directories of %s: %v", entry.FullPath, err)
|
|
return err
|
|
}
|
|
}
|
|
|
|
if !entry.IsDirectory() {
|
|
stats.FilerObjectSizeBytesHistogram.Observe(float64(entry.Size()))
|
|
}
|
|
} else {
|
|
if o_excl {
|
|
glog.V(3).InfofCtx(ctx, "EEXIST: entry %s already exists", entry.FullPath)
|
|
return fmt.Errorf("%s: %w", entry.FullPath, filer_pb.ErrEntryAlreadyExists)
|
|
}
|
|
glog.V(4).InfofCtx(ctx, "UpdateEntry %s: old entry: %v", entry.FullPath, oldEntry.Name())
|
|
if err := f.UpdateEntry(ctx, oldEntry, entry, isFromOtherCluster); err != nil {
|
|
if errors.Is(err, filer_pb.ErrExistingIsDirectory) || errors.Is(err, filer_pb.ErrExistingIsFile) {
|
|
glog.V(2).InfofCtx(ctx, "update entry %s: %v", entry.FullPath, err)
|
|
} else {
|
|
glog.ErrorfCtx(ctx, "update entry %s: %v", entry.FullPath, err)
|
|
}
|
|
return fmt.Errorf("update entry %s: %w", entry.FullPath, err)
|
|
}
|
|
}
|
|
|
|
f.NotifyUpdateEvent(ctx, oldEntry, entry, true, isFromOtherCluster, signatures)
|
|
|
|
f.deleteChunksIfNotNew(ctx, oldEntry, entry)
|
|
|
|
glog.V(4).InfofCtx(ctx, "CreateEntry %s: created", entry.FullPath)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) ensureParentDirectoryEntry(ctx context.Context, entry *Entry, dirParts []string, level int, isFromOtherCluster bool) (err error) {
|
|
|
|
if level == 0 {
|
|
return nil
|
|
}
|
|
|
|
dirPath := "/" + util.Join(dirParts[:level]...)
|
|
// fmt.Printf("%d dirPath: %+v\n", level, dirPath)
|
|
|
|
// check the store directly
|
|
glog.V(4).InfofCtx(ctx, "find uncached directory: %s", dirPath)
|
|
dirEntry, findErr := f.FindEntry(ctx, util.FullPath(dirPath))
|
|
if findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
|
return findErr
|
|
}
|
|
|
|
// no such existing directory
|
|
if dirEntry == nil {
|
|
|
|
// fmt.Printf("dirParts: %v %v %v\n", dirParts[0], dirParts[1], dirParts[2])
|
|
// dirParts[0] == "" and dirParts[1] == "buckets"
|
|
isUnderBuckets := len(dirParts) >= 3 && dirParts[1] == "buckets"
|
|
if isUnderBuckets {
|
|
if !strings.HasPrefix(dirParts[2], ".") {
|
|
if err := s3bucket.VerifyS3BucketName(dirParts[2]); err != nil {
|
|
return fmt.Errorf("invalid bucket name %s: %v", dirParts[2], err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// ensure parent directory
|
|
if err = f.ensureParentDirectoryEntry(ctx, entry, dirParts, level-1, isFromOtherCluster); err != nil {
|
|
return err
|
|
}
|
|
|
|
// create the directory
|
|
now := time.Now()
|
|
|
|
dirEntry = &Entry{
|
|
FullPath: util.FullPath(dirPath),
|
|
Attr: Attr{
|
|
Mtime: now,
|
|
Crtime: now,
|
|
Mode: os.ModeDir | entry.Mode | 0111,
|
|
Uid: entry.Uid,
|
|
Gid: entry.Gid,
|
|
UserName: entry.UserName,
|
|
GroupNames: entry.GroupNames,
|
|
},
|
|
}
|
|
f.ensureEntryInode(dirEntry)
|
|
if isUnderBuckets && level > 3 {
|
|
// Parent directories under buckets are created automatically; no additional logging.
|
|
}
|
|
|
|
glog.V(2).InfofCtx(ctx, "create directory: %s %v", dirPath, dirEntry.Mode)
|
|
mkdirErr := f.Store.InsertEntry(ctx, dirEntry)
|
|
if mkdirErr != nil {
|
|
if fEntry, err := f.FindEntry(ctx, util.FullPath(dirPath)); err == filer_pb.ErrNotFound || fEntry == nil {
|
|
glog.V(3).InfofCtx(ctx, "mkdir %s: %v", dirPath, mkdirErr)
|
|
return fmt.Errorf("mkdir %s: %v", dirPath, mkdirErr)
|
|
}
|
|
} else {
|
|
if !strings.HasPrefix("/"+util.Join(dirParts[:]...), SystemLogDir) {
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, isFromOtherCluster, nil)
|
|
}
|
|
}
|
|
|
|
} else if !dirEntry.IsDirectory() {
|
|
// S3 allows both "foo/bar" (object) and "foo/bar/xyzzy" (another
|
|
// object) to coexist because S3 has a flat key space. Promote the
|
|
// existing file to a directory, preserving its content/chunks so
|
|
// the original object data remains accessible.
|
|
glog.V(2).InfofCtx(ctx, "promoting %s from file to directory for %s", dirPath, entry.FullPath)
|
|
dirEntry.Attr.Mode |= os.ModeDir | 0111
|
|
// Expiring the entry now deletes the directory row and strands the keys under
|
|
// it, so the prefix object gives up its lazy TTL. The lifecycle worker still
|
|
// expires it, through the delete that leaves the directory behind.
|
|
dirEntry.Attr.TtlSec = 0
|
|
// An empty object leaves no chunks, content or mime behind, so without the
|
|
// mark the promotion would hide it.
|
|
if dirEntry.Extended == nil {
|
|
dirEntry.Extended = make(map[string][]byte)
|
|
}
|
|
dirEntry.Extended[s3_constants.SeaweedFSPrefixObject] = []byte("true")
|
|
if updateErr := f.Store.UpdateEntry(ctx, dirEntry); updateErr != nil {
|
|
return fmt.Errorf("promote %s to directory: %v", dirPath, updateErr)
|
|
}
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, isFromOtherCluster, nil)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// restorableModeBits is everything but the type bits: ModePerm alone would drop
|
|
// setgid, setuid and sticky, quietly changing group inheritance and delete semantics.
|
|
const restorableModeBits = os.ModePerm | os.ModeSetuid | os.ModeSetgid | os.ModeSticky
|
|
|
|
// DirectoryAttributes reads what a directory would need to be recreated as it is.
|
|
func (f *Filer) DirectoryAttributes(ctx context.Context, dirPath util.FullPath) (attrs empty_folder_cleanup.DirectoryAttributes, err error) {
|
|
entry, err := f.FindEntry(ctx, dirPath)
|
|
if err != nil {
|
|
return attrs, err
|
|
}
|
|
if entry == nil {
|
|
return attrs, filer_pb.ErrNotFound
|
|
}
|
|
return empty_folder_cleanup.DirectoryAttributes{
|
|
Mode: entry.Mode & restorableModeBits,
|
|
Uid: entry.Uid,
|
|
Gid: entry.Gid,
|
|
UserName: entry.UserName,
|
|
GroupNames: entry.GroupNames,
|
|
}, nil
|
|
}
|
|
|
|
// EnsureDirectoryEntry recreates dirPath, and any missing ancestor, for entries that
|
|
// outlived the directory holding them. dirPath comes back with the attributes it was
|
|
// deleted with, so a restore cannot hand back a directory more permissive than the one
|
|
// it replaces - including when someone else already put it back with wider ones.
|
|
func (f *Filer) EnsureDirectoryEntry(ctx context.Context, dirPath util.FullPath, attrs empty_folder_cleanup.DirectoryAttributes) error {
|
|
existing, err := f.FindEntry(ctx, dirPath)
|
|
if err != nil && !errors.Is(err, filer_pb.ErrNotFound) {
|
|
return err
|
|
}
|
|
if existing != nil {
|
|
// A writer recreating its own missing parent infers the mode from the entry it
|
|
// is inserting, so it can come back wider. Intersect rather than replace, or a
|
|
// mode holding a bit the saved one lacks would be granted the ones it lacks.
|
|
kept := existing.Mode & attrs.Mode & restorableModeBits
|
|
if existing.Mode&restorableModeBits == kept {
|
|
return nil
|
|
}
|
|
narrowed := existing.ShallowClone()
|
|
narrowed.Mode = existing.Mode&^restorableModeBits | kept
|
|
glog.V(1).InfofCtx(ctx, "restore directory %s: narrowing %v to %v", dirPath, existing.Mode, narrowed.Mode)
|
|
if err := f.UpdateEntry(ctx, existing, narrowed, false); err != nil {
|
|
return err
|
|
}
|
|
f.NotifyUpdateEvent(ctx, existing, narrowed, false, false, nil)
|
|
return nil
|
|
}
|
|
|
|
holder := &Entry{FullPath: dirPath, Attr: Attr{
|
|
Mode: attrs.Mode, Uid: attrs.Uid, Gid: attrs.Gid,
|
|
UserName: attrs.UserName, GroupNames: attrs.GroupNames,
|
|
}}
|
|
dirParts := strings.Split(string(dirPath), "/")
|
|
if err := f.ensureParentDirectoryEntry(ctx, holder, dirParts, len(dirParts)-1, false); err != nil {
|
|
return err
|
|
}
|
|
|
|
now := time.Now()
|
|
dirEntry := &Entry{FullPath: dirPath, Attr: Attr{
|
|
Mtime: now, Crtime: now,
|
|
Mode: os.ModeDir | attrs.Mode,
|
|
Uid: attrs.Uid,
|
|
Gid: attrs.Gid,
|
|
UserName: attrs.UserName, GroupNames: attrs.GroupNames,
|
|
}}
|
|
f.ensureEntryInode(dirEntry)
|
|
if err := f.Store.InsertEntry(ctx, dirEntry); err != nil {
|
|
return fmt.Errorf("restore directory %s: %v", dirPath, err)
|
|
}
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, false, nil)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) UpdateEntry(ctx context.Context, oldEntry, entry *Entry, isFromOtherCluster bool) (err error) {
|
|
if oldEntry != nil {
|
|
entry.Attr.Crtime = oldEntry.Attr.Crtime
|
|
if oldEntry.Attr.Inode != 0 {
|
|
// Object identity must not change on in-place updates.
|
|
entry.Attr.Inode = oldEntry.Attr.Inode
|
|
} else {
|
|
f.ensureEntryInode(entry)
|
|
}
|
|
// A type conflict is reported through the sentinel, and callers act on it -
|
|
// an S3 write of a key other keys are nested under retries as a prefix object -
|
|
// so it is the caller's outcome that decides whether anything went wrong.
|
|
if oldEntry.IsDirectory() && !entry.IsDirectory() {
|
|
glog.V(2).InfofCtx(ctx, "existing %s is a directory", oldEntry.FullPath)
|
|
return fmt.Errorf("%s: %w", oldEntry.FullPath, filer_pb.ErrExistingIsDirectory)
|
|
}
|
|
if !oldEntry.IsDirectory() && entry.IsDirectory() {
|
|
glog.V(2).InfofCtx(ctx, "existing %s is a file", oldEntry.FullPath)
|
|
return fmt.Errorf("%s: %w", oldEntry.FullPath, filer_pb.ErrExistingIsFile)
|
|
}
|
|
// A local write to a remote-backed entry leaves the copy on remote
|
|
// stale until a sync re-uploads it. Content changes that arrive
|
|
// without a fresh sync stamp are unsynced; clear the stamp so nothing
|
|
// mistakes the still-local chunks for a re-fetchable cache copy.
|
|
// Replicated updates carry the writer's authoritative stamp.
|
|
if !isFromOtherCluster && oldEntry.Remote != nil && entry.Remote != nil &&
|
|
oldEntry.Remote.LastLocalSyncTsNs == entry.Remote.LastLocalSyncTsNs &&
|
|
!chunksEqual(oldEntry.Chunks, entry.Chunks) {
|
|
entry.Remote = proto.Clone(entry.Remote).(*filer_pb.RemoteEntry)
|
|
entry.Remote.LastLocalSyncTsNs = 0
|
|
}
|
|
}
|
|
if entry.Attr.Atime.IsZero() {
|
|
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
|
}
|
|
return f.Store.UpdateEntry(ctx, entry)
|
|
}
|
|
|
|
func entryInitialAtime(attr Attr) time.Time {
|
|
if !attr.Mtime.IsZero() {
|
|
return attr.Mtime
|
|
}
|
|
return attr.Crtime
|
|
}
|
|
|
|
var (
|
|
Root = &Entry{
|
|
FullPath: "/",
|
|
Attr: Attr{
|
|
Mtime: time.Now(),
|
|
Crtime: time.Now(),
|
|
Mode: os.ModeDir | 0755,
|
|
Uid: OS_UID,
|
|
Gid: OS_GID,
|
|
},
|
|
}
|
|
)
|
|
|
|
func (f *Filer) FindEntry(ctx context.Context, p util.FullPath) (entry *Entry, err error) {
|
|
|
|
if string(p) == "/" {
|
|
return Root, nil
|
|
}
|
|
entry, err = f.Store.FindEntry(ctx, p)
|
|
// A directory is deleted here one row at a time, which would strand whatever is
|
|
// under it, so a TTL an older build left on one is not acted on.
|
|
if entry != nil && entry.TtlSec > 0 && !entry.IsDirectory() {
|
|
if entry.IsExpireS3Enabled() {
|
|
if entry.GetS3ExpireTime().Before(time.Now()) && !entry.IsS3Versioning() {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, entry, true, false, nil); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "FindEntry doDeleteEntryMetaAndData %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
return nil, filer_pb.ErrNotFound
|
|
}
|
|
} else if entry.Crtime.Add(time.Duration(entry.TtlSec) * time.Second).Before(time.Now()) {
|
|
f.Store.DeleteOneEntry(ctx, entry)
|
|
return nil, filer_pb.ErrNotFound
|
|
}
|
|
}
|
|
|
|
if entry == nil && (err == nil || errors.Is(err, filer_pb.ErrNotFound)) {
|
|
if lazy, lazyErr := f.maybeLazyFetchFromRemote(ctx, p); lazyErr != nil {
|
|
glog.V(1).InfofCtx(ctx, "FindEntry lazy fetch %s: %v", p, lazyErr)
|
|
} else if lazy != nil {
|
|
return lazy, nil
|
|
}
|
|
}
|
|
|
|
return entry, err
|
|
}
|
|
|
|
func (f *Filer) doListDirectoryEntries(ctx context.Context, p util.FullPath, startFileName string, inclusive bool, limit int64, prefix string, eachEntryFunc ListEachEntryFunc) (expiredCount int64, lastFileName string, err error) {
|
|
f.maybeLazyListFromRemote(ctx, p)
|
|
|
|
// Collect expired entries during iteration to avoid deadlock with DB connection pool
|
|
var expiredEntries []*Entry
|
|
var s3ExpiredEntries []*Entry
|
|
var hasValidEntries bool
|
|
|
|
lastFileName, err = f.Store.ListDirectoryPrefixedEntries(ctx, p, startFileName, inclusive, limit, prefix, func(entry *Entry) (bool, error) {
|
|
select {
|
|
case <-ctx.Done():
|
|
glog.V(1).InfofCtx(ctx, "listing %q canceled: %v", p, ctx.Err())
|
|
return false, fmt.Errorf("context canceled: %w", ctx.Err())
|
|
default:
|
|
if entry.TtlSec > 0 && !entry.IsDirectory() {
|
|
if entry.IsExpireS3Enabled() {
|
|
if entry.GetS3ExpireTime().Before(time.Now()) && !entry.IsS3Versioning() {
|
|
// Collect for deletion after iteration completes to avoid DB deadlock
|
|
s3ExpiredEntries = append(s3ExpiredEntries, entry)
|
|
expiredCount++
|
|
return true, nil
|
|
}
|
|
} else if entry.Crtime.Add(time.Duration(entry.TtlSec) * time.Second).Before(time.Now()) {
|
|
// Collect for deletion after iteration completes to avoid DB deadlock
|
|
expiredEntries = append(expiredEntries, entry)
|
|
expiredCount++
|
|
return true, nil
|
|
}
|
|
}
|
|
// Track that we found at least one valid (non-expired) entry
|
|
hasValidEntries = true
|
|
return eachEntryFunc(entry)
|
|
}
|
|
})
|
|
if err != nil {
|
|
return expiredCount, lastFileName, err
|
|
}
|
|
|
|
// Delete expired entries after iteration completes to avoid DB connection deadlock
|
|
if len(s3ExpiredEntries) > 0 || len(expiredEntries) > 0 {
|
|
for _, entry := range s3ExpiredEntries {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, entry, true, false, nil); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "doListDirectoryEntries doDeleteEntryMetaAndData %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
}
|
|
for _, entry := range expiredEntries {
|
|
if delErr := f.Store.DeleteOneEntry(ctx, entry); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "doListDirectoryEntries DeleteOneEntry %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
}
|
|
|
|
// After expiring entries, the directory might be empty.
|
|
// Attempt to clean it up and any empty parent directories.
|
|
if !hasValidEntries && p != "/" && startFileName == "" {
|
|
stopAtPath := util.FullPath(f.DirBucketsPath)
|
|
f.DeleteEmptyParentDirectories(ctx, p, stopAtPath)
|
|
}
|
|
}
|
|
|
|
return
|
|
}
|
|
|
|
// DeleteEmptyParentDirectories recursively checks and deletes parent directories if they become empty.
|
|
// It stops at root "/" or at stopAtPath (if provided).
|
|
// This is useful for cleaning up directories after deleting files or expired entries.
|
|
//
|
|
// IMPORTANT: For safety, dirPath must be under stopAtPath (when stopAtPath is provided).
|
|
// This prevents accidental deletion of directories outside the intended scope (e.g., outside bucket paths).
|
|
//
|
|
// Example usage:
|
|
//
|
|
// // After deleting /bucket/dir/subdir/file.txt, clean up empty parent directories
|
|
// // but stop at the bucket path
|
|
// parentPath := util.FullPath("/bucket/dir/subdir")
|
|
// filer.DeleteEmptyParentDirectories(ctx, parentPath, util.FullPath("/bucket"))
|
|
//
|
|
// Example with gRPC client:
|
|
//
|
|
// if err := pb_filer_client.WithFilerClient(ctx, func(client filer_pb.SeaweedFilerClient) error {
|
|
// return filer_pb.Traverse(ctx, filer, parentPath, "", func(entry *filer_pb.Entry) error {
|
|
// // Process entries...
|
|
// })
|
|
// }); err == nil {
|
|
// filer.DeleteEmptyParentDirectories(ctx, parentPath, stopPath)
|
|
// }
|
|
func (f *Filer) DeleteEmptyParentDirectories(ctx context.Context, dirPath util.FullPath, stopAtPath util.FullPath) {
|
|
if dirPath == "/" || dirPath == stopAtPath {
|
|
return
|
|
}
|
|
|
|
// Safety check: if stopAtPath is provided, dirPath must be under it (root "/" allows everything)
|
|
stopStr := string(stopAtPath)
|
|
if stopAtPath != "" && stopStr != "/" && !strings.HasPrefix(string(dirPath)+"/", stopStr+"/") {
|
|
glog.V(1).InfofCtx(ctx, "DeleteEmptyParentDirectories: %s is not under %s, skipping", dirPath, stopAtPath)
|
|
return
|
|
}
|
|
|
|
// Additional safety: prevent deletion of bucket-level directories
|
|
// This protects /buckets/mybucket from being deleted even if empty
|
|
baseDepth := strings.Count(f.DirBucketsPath, "/")
|
|
dirDepth := strings.Count(string(dirPath), "/")
|
|
if dirDepth <= baseDepth+1 {
|
|
glog.V(2).InfofCtx(ctx, "DeleteEmptyParentDirectories: skipping deletion of bucket-level directory %s", dirPath)
|
|
return
|
|
}
|
|
|
|
// Check if directory is empty
|
|
isEmpty, err := f.IsDirectoryEmpty(ctx, dirPath)
|
|
if err != nil {
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: error checking %s: %v", dirPath, err)
|
|
return
|
|
}
|
|
|
|
if !isEmpty {
|
|
// Directory is not empty, stop checking upward
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: directory %s is not empty, stopping cleanup", dirPath)
|
|
return
|
|
}
|
|
|
|
// Directory is empty, try to delete it
|
|
glog.V(2).InfofCtx(ctx, "DeleteEmptyParentDirectories: deleting empty directory %s", dirPath)
|
|
parentDir, _ := dirPath.DirAndName()
|
|
if dirEntry, findErr := f.FindEntry(ctx, dirPath); findErr == nil {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, dirEntry, false, false, nil); delErr == nil {
|
|
// Successfully deleted, continue checking upwards
|
|
f.DeleteEmptyParentDirectories(ctx, util.FullPath(parentDir), stopAtPath)
|
|
} else {
|
|
// Failed to delete, stop cleanup
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: failed to delete %s: %v", dirPath, delErr)
|
|
}
|
|
}
|
|
}
|
|
|
|
// IsDirectoryEmpty checks if a directory contains any entries
|
|
func (f *Filer) IsDirectoryEmpty(ctx context.Context, dirPath util.FullPath) (bool, error) {
|
|
isEmpty := true
|
|
_, err := f.Store.ListDirectoryPrefixedEntries(ctx, dirPath, "", true, 1, "", func(entry *Entry) (bool, error) {
|
|
isEmpty = false
|
|
return false, nil // Stop after first entry
|
|
})
|
|
return isEmpty, err
|
|
}
|
|
|
|
func (f *Filer) Shutdown() {
|
|
close(f.deletionQuit)
|
|
if f.EmptyFolderCleaner != nil {
|
|
f.EmptyFolderCleaner.Stop()
|
|
}
|
|
f.LocalMetaLogBuffer.ShutdownLogBuffer()
|
|
// The final metadata-log flush still needs the store to append its entry.
|
|
f.LocalMetaLogBuffer.WaitForShutdown()
|
|
// Persist the deletion ledger one last time before the store closes, so a
|
|
// clean shutdown leaves the recovery set exactly consistent with reality.
|
|
f.snapshotDeletionLedger()
|
|
f.Store.Shutdown()
|
|
}
|
|
|
|
func (f *Filer) GetEntryAttributes(ctx context.Context, p util.FullPath) (map[string][]byte, error) {
|
|
entry, err := f.FindEntry(ctx, p)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if entry == nil {
|
|
return nil, nil
|
|
}
|
|
return entry.Extended, nil
|
|
}
|
|
|
|
func (f *Filer) IsDirectoryKeyObject(ctx context.Context, p util.FullPath) (bool, error) {
|
|
entry, err := f.FindEntry(ctx, p)
|
|
if err != nil {
|
|
if errors.Is(err, filer_pb.ErrNotFound) {
|
|
return false, nil
|
|
}
|
|
return false, err
|
|
}
|
|
if entry == nil {
|
|
return false, nil
|
|
}
|
|
// Mirror filer_pb.Entry.IsDirectoryKeyObject so the cleaner keeps a promoted file's data.
|
|
_, isPrefixObject := entry.Extended[s3_constants.SeaweedFSPrefixObject]
|
|
return entry.IsDirectory() && (entry.Mime != "" || len(entry.GetChunks()) > 0 || len(entry.Content) > 0 || entry.IsInRemoteOnly() || isPrefixObject), nil
|
|
}
|