mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 23:07:48 +02:00
* filer: bound metadata log flush retries during shutdown On SIGTERM the filer could hang forever in Shutdown: the final LocalMetaLogBuffer flush retries appendToFile indefinitely, and with the master already down each AssignVolume attempt just kept failing. WaitForShutdown never returned, the interrupt hook never reached os.Exit, and the half-dead filer kept its ports bound. Thread a context through appendToFile/assignAndUpload and switch to a 15s-bounded context once the filer is stopping: the flush abandons with a log line instead of retrying forever. Normal operation keeps the unbounded retry so no metadata is dropped while running. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: share one shutdown deadline across all pending meta log flushes Review feedback on the per-flush timeout: a deadline armed at flush start could already be expired when shutdown arrived, the first append of a flush still ran unbounded, each queued window got a fresh budget (16 windows * 15s), and an abandoned window still advanced the flushed watermark as if it had landed. Rework to a single shared flush context on the Filer, cancelled once by Shutdown via AfterFunc. Every append - the in-flight one and every queued window - observes the same deadline, so the whole drain is bounded at 15s. A flush that gives up reports its dropped bytes through the new LogBuffer.NoteFlushDropped, and loopFlush then skips the offset/timestamp advance and subscriber notifications for that window. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: bound append attempts and commit uploaded pieces detached Store operations check then drop request cancellation, so a stalled backend could still hold flushFn past the shutdown deadline; run each append attempt on its own goroutine and give up on it at the deadline. Once a piece is uploaded, commit its entry on a detached context so the expired deadline cannot strand the chunk. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: keep shutdown flushes synchronous Detaching the append attempt let flushFn return while the goroutine still held the pooled flush buffer and could commit after the metadata store closed; abandonment is only safe for the cancelable assign/upload phase, which the shared flush context already bounds. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
860 lines
33 KiB
Go
860 lines
33 KiB
Go
package filer
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"os"
|
|
"sort"
|
|
"strings"
|
|
"sync"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/remote_storage"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3bucket"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
|
"github.com/seaweedfs/seaweedfs/weed/filer/empty_folder_cleanup"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/remote_pb"
|
|
|
|
"google.golang.org/grpc"
|
|
"google.golang.org/protobuf/proto"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
|
|
"github.com/seaweedfs/seaweedfs/weed/wdclient"
|
|
"golang.org/x/sync/singleflight"
|
|
)
|
|
|
|
const (
|
|
LogFlushInterval = time.Minute
|
|
PaginationSize = 1024
|
|
FilerStoreId = "filer.store.id"
|
|
)
|
|
|
|
var (
|
|
OS_UID = uint32(os.Getuid())
|
|
OS_GID = uint32(os.Getgid())
|
|
)
|
|
|
|
type Filer struct {
|
|
UniqueFilerId int32
|
|
UniqueFilerEpoch int32
|
|
Store VirtualFilerStore
|
|
MasterClient *wdclient.MasterClient
|
|
FileIdDeletionQueue *util.UnboundedQueue
|
|
GrpcDialOption grpc.DialOption
|
|
DirBucketsPath string
|
|
Cipher bool
|
|
LocalMetaLogBuffer *log_buffer.LogBuffer
|
|
metaLogCollection string
|
|
metaLogReplication string
|
|
// Override where system metadata-log chunks are assigned, keeping the
|
|
// internal log out of the default collection; empty keeps today's
|
|
// behaviour. Set via viper: filer.options.metaLog.collection / .replication.
|
|
metaLogTargetCollection string
|
|
metaLogTargetReplication string
|
|
DefaultDiskType string
|
|
MetaAggregator *MetaAggregator
|
|
Signature int32
|
|
FilerConf *FilerConf
|
|
placementOverlay PlacementOverlay
|
|
RemoteStorage *FilerRemoteStorage
|
|
BuildGuardedRemoteClient RemoteStorageClientBuilder
|
|
AllowUntrustedRemoteEndpoints bool
|
|
lazyFetchGroup singleflight.Group
|
|
lazyListGroup singleflight.Group
|
|
Dlm *lock_manager.DistributedLockManager
|
|
MaxFilenameLength uint32
|
|
deletionQuit chan struct{}
|
|
DeletionRetryQueue *DeletionRetryQueue
|
|
EmptyFolderCleaner *empty_folder_cleanup.EmptyFolderCleaner
|
|
EmptyFolderCleanupDelay time.Duration
|
|
persistedLogCache *persistedLogCache
|
|
metaLogInflight metaLogInflight
|
|
remoteTombstones *remoteDeletionTombstones
|
|
// remoteTombstonesDone, when non-nil, is closed once the startup tombstone
|
|
// rebuild finishes; lazy remote reads wait on it so a pending delete
|
|
// cannot resurrect in the gap.
|
|
remoteTombstonesDone atomic.Pointer[chan struct{}]
|
|
|
|
// Durable deletion ledger (see filer_deletion_persist.go). The set of
|
|
// fileIds that still need deleting but are not yet confirmed gone, mirrored
|
|
// to the store so a restart does not leak chunks. Guarded by
|
|
// deletionLedgerLock; nil-safe for Filer literals in tests.
|
|
// pendingDeletions maps each id to its enqueue epoch so an expiry-forget
|
|
// cannot erase a re-queued id.
|
|
deletionLedgerLock sync.Mutex
|
|
pendingDeletions map[string]uint64
|
|
deletionSeq uint64
|
|
deletionLedgerDirty bool
|
|
deletionLedgerParts int
|
|
deletionLedgerGen int
|
|
// deletionLedgerStale holds orphan part keys from abandoned multipart
|
|
// writes, retried on the next snapshot.
|
|
deletionLedgerStale []string
|
|
// deletionSnapshotLock serializes ledger writes in copy order so an
|
|
// in-flight timer snapshot cannot overwrite a newer shutdown snapshot.
|
|
deletionSnapshotLock sync.Mutex
|
|
// deletionLedgerBlocked is set when the startup ledger read fails: the
|
|
// persisted set is then unknown, so snapshots retry the read instead of
|
|
// overwriting the unread ledger with a partial set.
|
|
deletionLedgerBlocked atomic.Bool
|
|
deletionLedgerFlush chan struct{}
|
|
// flushCtx is the context metadata-log flushes append under. It stays
|
|
// live for the process lifetime and is cancelled once during Shutdown,
|
|
// bounding every retry - in-flight and queued alike - to one deadline.
|
|
// Nil for Filer literals built without NewFiler.
|
|
flushCtx context.Context
|
|
flushCancel context.CancelFunc
|
|
}
|
|
|
|
func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerHost pb.ServerAddress, filerGroup string, collection string, replication string, dataCenter string, maxFilenameLength uint32, notifyFn func()) *Filer {
|
|
f := &Filer{
|
|
MasterClient: wdclient.NewMasterClient(grpcDialOption, filerGroup, cluster.FilerType, filerHost, dataCenter, "", masters),
|
|
FileIdDeletionQueue: util.NewUnboundedQueue(),
|
|
GrpcDialOption: grpcDialOption,
|
|
FilerConf: NewFilerConf(),
|
|
RemoteStorage: NewFilerRemoteStorage(),
|
|
UniqueFilerId: util.RandomInt32(),
|
|
Dlm: lock_manager.NewDistributedLockManager(filerHost),
|
|
MaxFilenameLength: maxFilenameLength,
|
|
deletionQuit: make(chan struct{}),
|
|
deletionLedgerFlush: make(chan struct{}, 1),
|
|
DeletionRetryQueue: NewDeletionRetryQueue(),
|
|
persistedLogCache: newPersistedLogCache(persistedLogCacheMaxBytes),
|
|
remoteTombstones: newRemoteDeletionTombstones(),
|
|
}
|
|
f.flushCtx, f.flushCancel = context.WithCancel(context.Background())
|
|
if f.UniqueFilerId < 0 {
|
|
f.UniqueFilerId = -f.UniqueFilerId
|
|
}
|
|
|
|
// ReadFromDiskFn is intentionally nil here. SubscribeLocalMetadata already
|
|
// manages disk reads explicitly with shouldReadFromDisk / lastCheckedFlushTsNs
|
|
// tracking. Setting ReadFromDiskFn would cause LoopProcessLogData to issue a
|
|
// redundant ReadPersistedLogBuffer call (ListDirectoryEntries + readahead
|
|
// goroutine) on every 250ms health-check tick when a subscriber encounters
|
|
// ResumeFromDiskError, adding significant CPU and GC pressure even when idle.
|
|
f.LocalMetaLogBuffer = log_buffer.NewLogBuffer("local", LogFlushInterval, f.logFlushFunc, nil, notifyFn)
|
|
f.metaLogCollection = collection
|
|
f.metaLogReplication = replication
|
|
|
|
// Optional override for where the system metadata-log chunks land, so
|
|
// operators can keep internal log volumes out of the default collection.
|
|
// Unset (""), this changes nothing: meta logs keep following the filer
|
|
// default exactly as before.
|
|
v := util.GetViper()
|
|
v.SetDefault("filer.options.metaLog.collection", "")
|
|
v.SetDefault("filer.options.metaLog.replication", "")
|
|
f.metaLogTargetCollection = v.GetString("filer.options.metaLog.collection")
|
|
f.metaLogTargetReplication = v.GetString("filer.options.metaLog.replication")
|
|
if f.metaLogTargetCollection != "" {
|
|
glog.V(0).Infof("system metadata logs will be stored in collection %q", f.metaLogTargetCollection)
|
|
}
|
|
|
|
if newPlacementOverlay != nil {
|
|
f.placementOverlay = newPlacementOverlay(f)
|
|
}
|
|
|
|
go f.loopProcessingDeletion()
|
|
|
|
return f
|
|
}
|
|
|
|
func (f *Filer) buildRemoteStorageClient(ctx context.Context, remoteConf *remote_pb.RemoteConf) (remote_storage.RemoteStorageClient, error) {
|
|
if f.BuildGuardedRemoteClient != nil {
|
|
return f.BuildGuardedRemoteClient(ctx, remoteConf, f.AllowUntrustedRemoteEndpoints)
|
|
}
|
|
return remote_storage.GetRemoteStorage(remoteConf)
|
|
}
|
|
|
|
func (f *Filer) MaybeBootstrapFromOnePeer(self pb.ServerAddress, existingNodes []*master_pb.ClusterNodeUpdate, snapshotTime time.Time) (err error) {
|
|
if len(existingNodes) == 0 {
|
|
return
|
|
}
|
|
sort.Slice(existingNodes, func(i, j int) bool {
|
|
return existingNodes[i].CreatedAtNs < existingNodes[j].CreatedAtNs
|
|
})
|
|
earliestNode := existingNodes[0]
|
|
if pb.ServerAddress(earliestNode.Address).Equals(self) {
|
|
return
|
|
}
|
|
|
|
glog.V(0).Infof("bootstrap from %v clientId:%d", earliestNode.Address, f.UniqueFilerId)
|
|
|
|
return pb.WithFilerClient(false, f.UniqueFilerId, pb.ServerAddress(earliestNode.Address), f.GrpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
|
return filer_pb.StreamBfs(client, "/", snapshotTime.UnixNano(), func(parentPath util.FullPath, entry *filer_pb.Entry) error {
|
|
return f.Store.InsertEntry(context.Background(), FromPbEntry(string(parentPath), entry))
|
|
})
|
|
})
|
|
|
|
}
|
|
|
|
func (f *Filer) AggregateFromPeers(self pb.ServerAddress, existingNodes []*master_pb.ClusterNodeUpdate, startFrom time.Time) {
|
|
|
|
var snapshot []pb.ServerAddress
|
|
for _, node := range existingNodes {
|
|
address := pb.ServerAddress(node.Address)
|
|
snapshot = append(snapshot, address)
|
|
}
|
|
f.Dlm.LockRing.SetSnapshot(snapshot, 0)
|
|
glog.V(0).Infof("%s aggregate from peers %+v", self, snapshot)
|
|
|
|
// Initialize the empty folder cleaner using the same LockRing as Dlm for consistent hashing
|
|
f.EmptyFolderCleaner = empty_folder_cleanup.NewEmptyFolderCleaner(f, f.Dlm.LockRing, self, f.DirBucketsPath, f.EmptyFolderCleanupDelay)
|
|
|
|
f.MetaAggregator = NewMetaAggregator(f, self, f.GrpcDialOption)
|
|
// The ring starts empty while peer history sits on disk: mark the pre-startFrom
|
|
// range evicted so a cursor there reads disk, not the ring's earliest entry.
|
|
f.MetaAggregator.MetaLogBuffer.MarkEvictedThrough(startFrom.UnixNano())
|
|
f.MasterClient.SetOnPeerUpdateFn(func(update *master_pb.ClusterNodeUpdate, startFrom time.Time) {
|
|
if update.NodeType != cluster.FilerType {
|
|
return
|
|
}
|
|
// Lock ring is now managed by the master via LockRingUpdate,
|
|
// so we no longer call AddServer/RemoveServer here.
|
|
f.MetaAggregator.OnPeerUpdate(update, startFrom)
|
|
})
|
|
f.MasterClient.SetOnLockRingUpdateFn(func(update *master_pb.LockRingUpdate) {
|
|
var servers []pb.ServerAddress
|
|
for _, s := range update.Servers {
|
|
servers = append(servers, pb.ServerAddress(s))
|
|
}
|
|
glog.V(0).Infof("LockRing: applying master ring update v%d: %v", update.Version, servers)
|
|
f.Dlm.LockRing.SetSnapshot(servers, update.Version)
|
|
})
|
|
f.MasterClient.SetOnMasterChangeFn(func(previous, current pb.ServerAddress) {
|
|
glog.V(0).Infof("LockRing: master changed %s -> %s, resetting ring", previous, current)
|
|
f.Dlm.LockRing.Reset()
|
|
})
|
|
|
|
// Subscribe to the local filer first: its events reach the aggregated
|
|
// buffer only through this subscription, and the peer watermarks must
|
|
// account for it before any remote peer - a remotes-only watermark set
|
|
// would claim completeness without self. existingNodes can omit self
|
|
// (master registration races this bootstrap); duplicate adds are no-ops.
|
|
f.MetaAggregator.OnPeerUpdate(&master_pb.ClusterNodeUpdate{
|
|
NodeType: cluster.FilerType,
|
|
Address: string(self),
|
|
IsAdd: true,
|
|
}, startFrom)
|
|
for _, peerUpdate := range existingNodes {
|
|
f.MetaAggregator.OnPeerUpdate(peerUpdate, startFrom)
|
|
}
|
|
|
|
}
|
|
|
|
func (f *Filer) ListExistingPeerUpdates(ctx context.Context) (existingNodes []*master_pb.ClusterNodeUpdate) {
|
|
return cluster.ListExistingPeerUpdates(f.GetMaster(ctx), f.GrpcDialOption, f.MasterClient.FilerGroup, cluster.FilerType)
|
|
}
|
|
|
|
func (f *Filer) SetStore(store FilerStore) (isFresh bool) {
|
|
f.Store = NewFilerStoreWrapper(store)
|
|
|
|
isFresh = f.setOrLoadFilerStoreSignature(store)
|
|
|
|
// Recover deletions that were pending when a previous process died, and keep
|
|
// the durable ledger snapshotted while running (see filer_deletion_persist.go).
|
|
// A failed ledger read leaves writes blocked until a snapshot retries and
|
|
// the read succeeds, so the unread ledger is never overwritten.
|
|
f.reloadDeletionLedger()
|
|
f.startDeletionLedgerSnapshotter()
|
|
|
|
return isFresh
|
|
}
|
|
|
|
func (f *Filer) setOrLoadFilerStoreSignature(store FilerStore) (isFresh bool) {
|
|
storeIdBytes, err := store.KvGet(context.Background(), []byte(FilerStoreId))
|
|
if err == ErrKvNotFound || err == nil && len(storeIdBytes) == 0 {
|
|
f.Signature = util.RandomInt32()
|
|
storeIdBytes = make([]byte, 4)
|
|
util.Uint32toBytes(storeIdBytes, uint32(f.Signature))
|
|
if err = store.KvPut(context.Background(), []byte(FilerStoreId), storeIdBytes); err != nil {
|
|
glog.Fatalf("set %s=%d : %v", FilerStoreId, f.Signature, err)
|
|
}
|
|
glog.V(0).Infof("create %s to %d", FilerStoreId, f.Signature)
|
|
return true
|
|
} else if err == nil && len(storeIdBytes) == 4 {
|
|
f.Signature = int32(util.BytesToUint32(storeIdBytes))
|
|
glog.V(0).Infof("existing %s = %d", FilerStoreId, f.Signature)
|
|
} else {
|
|
glog.Fatalf("read %v=%v : %v", FilerStoreId, string(storeIdBytes), err)
|
|
}
|
|
return false
|
|
}
|
|
|
|
func (f *Filer) GetStore() (store FilerStore) {
|
|
return f.Store
|
|
}
|
|
|
|
func (fs *Filer) GetMaster(ctx context.Context) pb.ServerAddress {
|
|
return fs.MasterClient.GetMaster(ctx)
|
|
}
|
|
|
|
func (f *Filer) BeginTransaction(ctx context.Context) (context.Context, error) {
|
|
return f.Store.BeginTransaction(ctx)
|
|
}
|
|
|
|
func (f *Filer) CommitTransaction(ctx context.Context) error {
|
|
return f.Store.CommitTransaction(ctx)
|
|
}
|
|
|
|
func (f *Filer) RollbackTransaction(ctx context.Context) error {
|
|
return f.Store.RollbackTransaction(ctx)
|
|
}
|
|
|
|
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
|
|
// has already fetched the current entry at this path under a path lock, and it
|
|
// is reused instead of looking the store up again; pass nil to have CreateEntry
|
|
// look it up itself.
|
|
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
|
|
|
if string(entry.FullPath) == "/" {
|
|
return nil
|
|
}
|
|
|
|
if entry.FullPath.IsLongerFileName(maxFilenameLength) {
|
|
return filer_pb.ErrEntryNameTooLong
|
|
}
|
|
|
|
if entry.IsDirectory() {
|
|
entry.Attr.TtlSec = 0
|
|
}
|
|
|
|
if entry.Attr.Atime.IsZero() {
|
|
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
|
}
|
|
|
|
oldEntry := existing
|
|
if oldEntry == nil {
|
|
var findErr error
|
|
oldEntry, findErr = f.FindEntry(ctx, entry.FullPath)
|
|
if o_excl && findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
|
// An exclusive create cannot decide whether the path exists when
|
|
// the lookup itself failed; proceeding would upsert over it.
|
|
return fmt.Errorf("find entry %s: %w", entry.FullPath, findErr)
|
|
}
|
|
}
|
|
|
|
/*
|
|
if !hasWritePermission(lastDirectoryEntry, entry) {
|
|
glog.V(0).Infof("directory %s: %v, entry: uid=%d gid=%d",
|
|
lastDirectoryEntry.FullPath, lastDirectoryEntry.Attr, entry.Uid, entry.Gid)
|
|
return fmt.Errorf("no write permission in folder %v", lastDirectoryEntry.FullPath)
|
|
}
|
|
*/
|
|
|
|
if oldEntry == nil {
|
|
f.ensureEntryInode(entry)
|
|
|
|
glog.V(4).InfofCtx(ctx, "InsertEntry %s: new entry: %v", entry.FullPath, entry.Name())
|
|
if err := f.Store.InsertEntry(ctx, entry); err != nil {
|
|
glog.ErrorfCtx(ctx, "insert entry %s: %v", entry.FullPath, err)
|
|
return fmt.Errorf("insert entry %s: %v", entry.FullPath, err)
|
|
}
|
|
|
|
// Parents go after the entry: one checked first can be taken by the
|
|
// empty-folder cleaner before the entry lands, and nothing would look again.
|
|
if !skipCreateParentDir {
|
|
dirParts := strings.Split(string(entry.FullPath), "/")
|
|
if err := f.ensureParentDirectoryEntry(ctx, entry, dirParts, len(dirParts)-1, isFromOtherCluster); err != nil {
|
|
// The entry stays: deleting by path would destroy a concurrent create
|
|
// that already succeeded through the update branch, and the update keeps
|
|
// the inode and crtime while mtime only survives the store to the second.
|
|
// It is not announced - the aggregator replicates creates into peer
|
|
// stores, so a failed write would land on all of them rather than on the
|
|
// one filer whose next write into that folder repairs it.
|
|
glog.ErrorfCtx(ctx, "create parent directories of %s: %v", entry.FullPath, err)
|
|
return err
|
|
}
|
|
}
|
|
|
|
if !entry.IsDirectory() {
|
|
stats.FilerObjectSizeBytesHistogram.Observe(float64(entry.Size()))
|
|
}
|
|
} else {
|
|
if o_excl {
|
|
glog.V(3).InfofCtx(ctx, "EEXIST: entry %s already exists", entry.FullPath)
|
|
return fmt.Errorf("%s: %w", entry.FullPath, filer_pb.ErrEntryAlreadyExists)
|
|
}
|
|
glog.V(4).InfofCtx(ctx, "UpdateEntry %s: old entry: %v", entry.FullPath, oldEntry.Name())
|
|
if err := f.UpdateEntry(ctx, oldEntry, entry, isFromOtherCluster); err != nil {
|
|
if errors.Is(err, filer_pb.ErrExistingIsDirectory) || errors.Is(err, filer_pb.ErrExistingIsFile) {
|
|
glog.V(2).InfofCtx(ctx, "update entry %s: %v", entry.FullPath, err)
|
|
} else {
|
|
glog.ErrorfCtx(ctx, "update entry %s: %v", entry.FullPath, err)
|
|
}
|
|
return fmt.Errorf("update entry %s: %w", entry.FullPath, err)
|
|
}
|
|
}
|
|
|
|
f.NotifyUpdateEvent(ctx, oldEntry, entry, true, isFromOtherCluster, signatures)
|
|
|
|
f.deleteChunksIfNotNew(ctx, oldEntry, entry)
|
|
|
|
glog.V(4).InfofCtx(ctx, "CreateEntry %s: created", entry.FullPath)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) ensureParentDirectoryEntry(ctx context.Context, entry *Entry, dirParts []string, level int, isFromOtherCluster bool) (err error) {
|
|
|
|
if level == 0 {
|
|
return nil
|
|
}
|
|
|
|
dirPath := "/" + util.Join(dirParts[:level]...)
|
|
// fmt.Printf("%d dirPath: %+v\n", level, dirPath)
|
|
|
|
// check the store directly
|
|
glog.V(4).InfofCtx(ctx, "find uncached directory: %s", dirPath)
|
|
dirEntry, findErr := f.FindEntry(ctx, util.FullPath(dirPath))
|
|
if findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
|
return findErr
|
|
}
|
|
|
|
// no such existing directory
|
|
if dirEntry == nil {
|
|
|
|
// fmt.Printf("dirParts: %v %v %v\n", dirParts[0], dirParts[1], dirParts[2])
|
|
// dirParts[0] == "" and dirParts[1] == "buckets"
|
|
isUnderBuckets := len(dirParts) >= 3 && dirParts[1] == "buckets"
|
|
if isUnderBuckets {
|
|
if !strings.HasPrefix(dirParts[2], ".") {
|
|
if err := s3bucket.VerifyS3BucketName(dirParts[2]); err != nil {
|
|
return fmt.Errorf("invalid bucket name %s: %v", dirParts[2], err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// ensure parent directory
|
|
if err = f.ensureParentDirectoryEntry(ctx, entry, dirParts, level-1, isFromOtherCluster); err != nil {
|
|
return err
|
|
}
|
|
|
|
// create the directory
|
|
now := time.Now()
|
|
|
|
dirEntry = &Entry{
|
|
FullPath: util.FullPath(dirPath),
|
|
Attr: Attr{
|
|
Mtime: now,
|
|
Crtime: now,
|
|
Mode: os.ModeDir | entry.Mode | 0111,
|
|
Uid: entry.Uid,
|
|
Gid: entry.Gid,
|
|
UserName: entry.UserName,
|
|
GroupNames: entry.GroupNames,
|
|
},
|
|
}
|
|
f.ensureEntryInode(dirEntry)
|
|
if isUnderBuckets && level > 3 {
|
|
// Parent directories under buckets are created automatically; no additional logging.
|
|
}
|
|
|
|
glog.V(2).InfofCtx(ctx, "create directory: %s %v", dirPath, dirEntry.Mode)
|
|
mkdirErr := f.Store.InsertEntry(ctx, dirEntry)
|
|
if mkdirErr != nil {
|
|
if fEntry, err := f.FindEntry(ctx, util.FullPath(dirPath)); err == filer_pb.ErrNotFound || fEntry == nil {
|
|
glog.V(3).InfofCtx(ctx, "mkdir %s: %v", dirPath, mkdirErr)
|
|
return fmt.Errorf("mkdir %s: %v", dirPath, mkdirErr)
|
|
}
|
|
} else {
|
|
if !strings.HasPrefix("/"+util.Join(dirParts[:]...), SystemLogDir) {
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, isFromOtherCluster, nil)
|
|
}
|
|
}
|
|
|
|
} else if !dirEntry.IsDirectory() {
|
|
// S3 allows both "foo/bar" (object) and "foo/bar/xyzzy" (another
|
|
// object) to coexist because S3 has a flat key space. Promote the
|
|
// existing file to a directory, preserving its content/chunks so
|
|
// the original object data remains accessible.
|
|
glog.V(2).InfofCtx(ctx, "promoting %s from file to directory for %s", dirPath, entry.FullPath)
|
|
dirEntry.Attr.Mode |= os.ModeDir | 0111
|
|
// Expiring the entry now deletes the directory row and strands the keys under
|
|
// it, so the prefix object gives up its lazy TTL. The lifecycle worker still
|
|
// expires it, through the delete that leaves the directory behind.
|
|
dirEntry.Attr.TtlSec = 0
|
|
// An empty object leaves no chunks, content or mime behind, so without the
|
|
// mark the promotion would hide it.
|
|
if dirEntry.Extended == nil {
|
|
dirEntry.Extended = make(map[string][]byte)
|
|
}
|
|
dirEntry.Extended[s3_constants.SeaweedFSPrefixObject] = []byte("true")
|
|
if updateErr := f.Store.UpdateEntry(ctx, dirEntry); updateErr != nil {
|
|
return fmt.Errorf("promote %s to directory: %v", dirPath, updateErr)
|
|
}
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, isFromOtherCluster, nil)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// restorableModeBits is everything but the type bits: ModePerm alone would drop
|
|
// setgid, setuid and sticky, quietly changing group inheritance and delete semantics.
|
|
const restorableModeBits = os.ModePerm | os.ModeSetuid | os.ModeSetgid | os.ModeSticky
|
|
|
|
// DirectoryAttributes reads what a directory would need to be recreated as it is.
|
|
func (f *Filer) DirectoryAttributes(ctx context.Context, dirPath util.FullPath) (attrs empty_folder_cleanup.DirectoryAttributes, err error) {
|
|
entry, err := f.FindEntry(ctx, dirPath)
|
|
if err != nil {
|
|
return attrs, err
|
|
}
|
|
if entry == nil {
|
|
return attrs, filer_pb.ErrNotFound
|
|
}
|
|
return empty_folder_cleanup.DirectoryAttributes{
|
|
Mode: entry.Mode & restorableModeBits,
|
|
Uid: entry.Uid,
|
|
Gid: entry.Gid,
|
|
UserName: entry.UserName,
|
|
GroupNames: entry.GroupNames,
|
|
}, nil
|
|
}
|
|
|
|
// EnsureDirectoryEntry recreates dirPath, and any missing ancestor, for entries that
|
|
// outlived the directory holding them. dirPath comes back with the attributes it was
|
|
// deleted with, so a restore cannot hand back a directory more permissive than the one
|
|
// it replaces - including when someone else already put it back with wider ones.
|
|
func (f *Filer) EnsureDirectoryEntry(ctx context.Context, dirPath util.FullPath, attrs empty_folder_cleanup.DirectoryAttributes) error {
|
|
existing, err := f.FindEntry(ctx, dirPath)
|
|
if err != nil && !errors.Is(err, filer_pb.ErrNotFound) {
|
|
return err
|
|
}
|
|
if existing != nil {
|
|
// A writer recreating its own missing parent infers the mode from the entry it
|
|
// is inserting, so it can come back wider. Intersect rather than replace, or a
|
|
// mode holding a bit the saved one lacks would be granted the ones it lacks.
|
|
kept := existing.Mode & attrs.Mode & restorableModeBits
|
|
if existing.Mode&restorableModeBits == kept {
|
|
return nil
|
|
}
|
|
narrowed := existing.ShallowClone()
|
|
narrowed.Mode = existing.Mode&^restorableModeBits | kept
|
|
glog.V(1).InfofCtx(ctx, "restore directory %s: narrowing %v to %v", dirPath, existing.Mode, narrowed.Mode)
|
|
if err := f.UpdateEntry(ctx, existing, narrowed, false); err != nil {
|
|
return err
|
|
}
|
|
f.NotifyUpdateEvent(ctx, existing, narrowed, false, false, nil)
|
|
return nil
|
|
}
|
|
|
|
holder := &Entry{FullPath: dirPath, Attr: Attr{
|
|
Mode: attrs.Mode, Uid: attrs.Uid, Gid: attrs.Gid,
|
|
UserName: attrs.UserName, GroupNames: attrs.GroupNames,
|
|
}}
|
|
dirParts := strings.Split(string(dirPath), "/")
|
|
if err := f.ensureParentDirectoryEntry(ctx, holder, dirParts, len(dirParts)-1, false); err != nil {
|
|
return err
|
|
}
|
|
|
|
now := time.Now()
|
|
dirEntry := &Entry{FullPath: dirPath, Attr: Attr{
|
|
Mtime: now, Crtime: now,
|
|
Mode: os.ModeDir | attrs.Mode,
|
|
Uid: attrs.Uid,
|
|
Gid: attrs.Gid,
|
|
UserName: attrs.UserName, GroupNames: attrs.GroupNames,
|
|
}}
|
|
f.ensureEntryInode(dirEntry)
|
|
if err := f.Store.InsertEntry(ctx, dirEntry); err != nil {
|
|
return fmt.Errorf("restore directory %s: %v", dirPath, err)
|
|
}
|
|
f.NotifyUpdateEvent(ctx, nil, dirEntry, false, false, nil)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) UpdateEntry(ctx context.Context, oldEntry, entry *Entry, isFromOtherCluster bool) (err error) {
|
|
if oldEntry != nil {
|
|
entry.Attr.Crtime = oldEntry.Attr.Crtime
|
|
if oldEntry.Attr.Inode != 0 {
|
|
// Object identity must not change on in-place updates.
|
|
entry.Attr.Inode = oldEntry.Attr.Inode
|
|
} else {
|
|
f.ensureEntryInode(entry)
|
|
}
|
|
// A type conflict is reported through the sentinel, and callers act on it -
|
|
// an S3 write of a key other keys are nested under retries as a prefix object -
|
|
// so it is the caller's outcome that decides whether anything went wrong.
|
|
if oldEntry.IsDirectory() && !entry.IsDirectory() {
|
|
glog.V(2).InfofCtx(ctx, "existing %s is a directory", oldEntry.FullPath)
|
|
return fmt.Errorf("%s: %w", oldEntry.FullPath, filer_pb.ErrExistingIsDirectory)
|
|
}
|
|
if !oldEntry.IsDirectory() && entry.IsDirectory() {
|
|
glog.V(2).InfofCtx(ctx, "existing %s is a file", oldEntry.FullPath)
|
|
return fmt.Errorf("%s: %w", oldEntry.FullPath, filer_pb.ErrExistingIsFile)
|
|
}
|
|
// A local write to a remote-backed entry leaves the copy on remote
|
|
// stale until a sync re-uploads it. Content changes that arrive
|
|
// without a fresh sync stamp are unsynced; clear the stamp so nothing
|
|
// mistakes the still-local chunks for a re-fetchable cache copy.
|
|
// Replicated updates carry the writer's authoritative stamp.
|
|
if !isFromOtherCluster && oldEntry.Remote != nil && entry.Remote != nil &&
|
|
oldEntry.Remote.LastLocalSyncTsNs == entry.Remote.LastLocalSyncTsNs &&
|
|
!chunksEqual(oldEntry.Chunks, entry.Chunks) {
|
|
entry.Remote = proto.Clone(entry.Remote).(*filer_pb.RemoteEntry)
|
|
entry.Remote.LastLocalSyncTsNs = 0
|
|
}
|
|
}
|
|
if entry.Attr.Atime.IsZero() {
|
|
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
|
}
|
|
return f.Store.UpdateEntry(ctx, entry)
|
|
}
|
|
|
|
func entryInitialAtime(attr Attr) time.Time {
|
|
if !attr.Mtime.IsZero() {
|
|
return attr.Mtime
|
|
}
|
|
return attr.Crtime
|
|
}
|
|
|
|
var (
|
|
Root = &Entry{
|
|
FullPath: "/",
|
|
Attr: Attr{
|
|
Mtime: time.Now(),
|
|
Crtime: time.Now(),
|
|
Mode: os.ModeDir | 0755,
|
|
Uid: OS_UID,
|
|
Gid: OS_GID,
|
|
},
|
|
}
|
|
)
|
|
|
|
func (f *Filer) FindEntry(ctx context.Context, p util.FullPath) (entry *Entry, err error) {
|
|
|
|
if string(p) == "/" {
|
|
return Root, nil
|
|
}
|
|
entry, err = f.Store.FindEntry(ctx, p)
|
|
// A directory is deleted here one row at a time, which would strand whatever is
|
|
// under it, so a TTL an older build left on one is not acted on.
|
|
if entry != nil && entry.TtlSec > 0 && !entry.IsDirectory() {
|
|
if entry.IsExpireS3Enabled() {
|
|
if entry.GetS3ExpireTime().Before(time.Now()) && !entry.IsS3Versioning() {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, entry, true, false, nil); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "FindEntry doDeleteEntryMetaAndData %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
return nil, filer_pb.ErrNotFound
|
|
}
|
|
} else if entry.Crtime.Add(time.Duration(entry.TtlSec) * time.Second).Before(time.Now()) {
|
|
f.Store.DeleteOneEntry(ctx, entry)
|
|
return nil, filer_pb.ErrNotFound
|
|
}
|
|
}
|
|
|
|
if entry == nil && (err == nil || errors.Is(err, filer_pb.ErrNotFound)) {
|
|
if lazy, lazyErr := f.maybeLazyFetchFromRemote(ctx, p); lazyErr != nil {
|
|
glog.V(1).InfofCtx(ctx, "FindEntry lazy fetch %s: %v", p, lazyErr)
|
|
} else if lazy != nil {
|
|
return lazy, nil
|
|
}
|
|
}
|
|
|
|
return entry, err
|
|
}
|
|
|
|
func (f *Filer) doListDirectoryEntries(ctx context.Context, p util.FullPath, startFileName string, inclusive bool, limit int64, prefix string, eachEntryFunc ListEachEntryFunc) (expiredCount int64, lastFileName string, err error) {
|
|
f.maybeLazyListFromRemote(ctx, p)
|
|
|
|
// Collect expired entries during iteration to avoid deadlock with DB connection pool
|
|
var expiredEntries []*Entry
|
|
var s3ExpiredEntries []*Entry
|
|
var hasValidEntries bool
|
|
|
|
lastFileName, err = f.Store.ListDirectoryPrefixedEntries(ctx, p, startFileName, inclusive, limit, prefix, func(entry *Entry) (bool, error) {
|
|
select {
|
|
case <-ctx.Done():
|
|
glog.V(1).InfofCtx(ctx, "listing %q canceled: %v", p, ctx.Err())
|
|
return false, fmt.Errorf("context canceled: %w", ctx.Err())
|
|
default:
|
|
if entry.TtlSec > 0 && !entry.IsDirectory() {
|
|
if entry.IsExpireS3Enabled() {
|
|
if entry.GetS3ExpireTime().Before(time.Now()) && !entry.IsS3Versioning() {
|
|
// Collect for deletion after iteration completes to avoid DB deadlock
|
|
s3ExpiredEntries = append(s3ExpiredEntries, entry)
|
|
expiredCount++
|
|
return true, nil
|
|
}
|
|
} else if entry.Crtime.Add(time.Duration(entry.TtlSec) * time.Second).Before(time.Now()) {
|
|
// Collect for deletion after iteration completes to avoid DB deadlock
|
|
expiredEntries = append(expiredEntries, entry)
|
|
expiredCount++
|
|
return true, nil
|
|
}
|
|
}
|
|
// Track that we found at least one valid (non-expired) entry
|
|
hasValidEntries = true
|
|
return eachEntryFunc(entry)
|
|
}
|
|
})
|
|
if err != nil {
|
|
return expiredCount, lastFileName, err
|
|
}
|
|
|
|
// Delete expired entries after iteration completes to avoid DB connection deadlock
|
|
if len(s3ExpiredEntries) > 0 || len(expiredEntries) > 0 {
|
|
for _, entry := range s3ExpiredEntries {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, entry, true, false, nil); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "doListDirectoryEntries doDeleteEntryMetaAndData %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
}
|
|
for _, entry := range expiredEntries {
|
|
if delErr := f.Store.DeleteOneEntry(ctx, entry); delErr != nil {
|
|
glog.ErrorfCtx(ctx, "doListDirectoryEntries DeleteOneEntry %s failed: %v", entry.FullPath, delErr)
|
|
}
|
|
}
|
|
|
|
// After expiring entries, the directory might be empty.
|
|
// Attempt to clean it up and any empty parent directories.
|
|
if !hasValidEntries && p != "/" && startFileName == "" {
|
|
stopAtPath := util.FullPath(f.DirBucketsPath)
|
|
f.DeleteEmptyParentDirectories(ctx, p, stopAtPath)
|
|
}
|
|
}
|
|
|
|
return
|
|
}
|
|
|
|
// DeleteEmptyParentDirectories recursively checks and deletes parent directories if they become empty.
|
|
// It stops at root "/" or at stopAtPath (if provided).
|
|
// This is useful for cleaning up directories after deleting files or expired entries.
|
|
//
|
|
// IMPORTANT: For safety, dirPath must be under stopAtPath (when stopAtPath is provided).
|
|
// This prevents accidental deletion of directories outside the intended scope (e.g., outside bucket paths).
|
|
//
|
|
// Example usage:
|
|
//
|
|
// // After deleting /bucket/dir/subdir/file.txt, clean up empty parent directories
|
|
// // but stop at the bucket path
|
|
// parentPath := util.FullPath("/bucket/dir/subdir")
|
|
// filer.DeleteEmptyParentDirectories(ctx, parentPath, util.FullPath("/bucket"))
|
|
//
|
|
// Example with gRPC client:
|
|
//
|
|
// if err := pb_filer_client.WithFilerClient(ctx, func(client filer_pb.SeaweedFilerClient) error {
|
|
// return filer_pb.Traverse(ctx, filer, parentPath, "", func(entry *filer_pb.Entry) error {
|
|
// // Process entries...
|
|
// })
|
|
// }); err == nil {
|
|
// filer.DeleteEmptyParentDirectories(ctx, parentPath, stopPath)
|
|
// }
|
|
func (f *Filer) DeleteEmptyParentDirectories(ctx context.Context, dirPath util.FullPath, stopAtPath util.FullPath) {
|
|
if dirPath == "/" || dirPath == stopAtPath {
|
|
return
|
|
}
|
|
|
|
// Safety check: if stopAtPath is provided, dirPath must be under it (root "/" allows everything)
|
|
stopStr := string(stopAtPath)
|
|
if stopAtPath != "" && stopStr != "/" && !strings.HasPrefix(string(dirPath)+"/", stopStr+"/") {
|
|
glog.V(1).InfofCtx(ctx, "DeleteEmptyParentDirectories: %s is not under %s, skipping", dirPath, stopAtPath)
|
|
return
|
|
}
|
|
|
|
// Additional safety: prevent deletion of bucket-level directories
|
|
// This protects /buckets/mybucket from being deleted even if empty
|
|
baseDepth := strings.Count(f.DirBucketsPath, "/")
|
|
dirDepth := strings.Count(string(dirPath), "/")
|
|
if dirDepth <= baseDepth+1 {
|
|
glog.V(2).InfofCtx(ctx, "DeleteEmptyParentDirectories: skipping deletion of bucket-level directory %s", dirPath)
|
|
return
|
|
}
|
|
|
|
// Check if directory is empty
|
|
isEmpty, err := f.IsDirectoryEmpty(ctx, dirPath)
|
|
if err != nil {
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: error checking %s: %v", dirPath, err)
|
|
return
|
|
}
|
|
|
|
if !isEmpty {
|
|
// Directory is not empty, stop checking upward
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: directory %s is not empty, stopping cleanup", dirPath)
|
|
return
|
|
}
|
|
|
|
// Directory is empty, try to delete it
|
|
glog.V(2).InfofCtx(ctx, "DeleteEmptyParentDirectories: deleting empty directory %s", dirPath)
|
|
parentDir, _ := dirPath.DirAndName()
|
|
if dirEntry, findErr := f.FindEntry(ctx, dirPath); findErr == nil {
|
|
if delErr := f.doDeleteEntryMetaAndData(ctx, dirEntry, false, false, nil); delErr == nil {
|
|
// Successfully deleted, continue checking upwards
|
|
f.DeleteEmptyParentDirectories(ctx, util.FullPath(parentDir), stopAtPath)
|
|
} else {
|
|
// Failed to delete, stop cleanup
|
|
glog.V(3).InfofCtx(ctx, "DeleteEmptyParentDirectories: failed to delete %s: %v", dirPath, delErr)
|
|
}
|
|
}
|
|
}
|
|
|
|
// IsDirectoryEmpty checks if a directory contains any entries
|
|
func (f *Filer) IsDirectoryEmpty(ctx context.Context, dirPath util.FullPath) (bool, error) {
|
|
isEmpty := true
|
|
_, err := f.Store.ListDirectoryPrefixedEntries(ctx, dirPath, "", true, 1, "", func(entry *Entry) (bool, error) {
|
|
isEmpty = false
|
|
return false, nil // Stop after first entry
|
|
})
|
|
return isEmpty, err
|
|
}
|
|
|
|
func (f *Filer) Shutdown() {
|
|
close(f.deletionQuit)
|
|
if f.EmptyFolderCleaner != nil {
|
|
f.EmptyFolderCleaner.Stop()
|
|
}
|
|
// Bound the remaining flush retries: with the cluster already gone each
|
|
// append keeps failing, and an unbounded retry would hold the shutdown
|
|
// drain open indefinitely. One shared deadline covers every in-flight and
|
|
// queued window; the flush loop drops what it cannot write.
|
|
if f.flushCancel != nil {
|
|
time.AfterFunc(shutdownMetadataLogFlushBudget, f.flushCancel)
|
|
}
|
|
f.LocalMetaLogBuffer.ShutdownLogBuffer()
|
|
// The final metadata-log flush still needs the store to append its entry.
|
|
f.LocalMetaLogBuffer.WaitForShutdown()
|
|
// Persist the deletion ledger one last time before the store closes, so a
|
|
// clean shutdown leaves the recovery set exactly consistent with reality.
|
|
f.snapshotDeletionLedger()
|
|
f.Store.Shutdown()
|
|
}
|
|
|
|
func (f *Filer) GetEntryAttributes(ctx context.Context, p util.FullPath) (map[string][]byte, error) {
|
|
entry, err := f.FindEntry(ctx, p)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if entry == nil {
|
|
return nil, nil
|
|
}
|
|
return entry.Extended, nil
|
|
}
|
|
|
|
func (f *Filer) IsDirectoryKeyObject(ctx context.Context, p util.FullPath) (bool, error) {
|
|
entry, err := f.FindEntry(ctx, p)
|
|
if err != nil {
|
|
if errors.Is(err, filer_pb.ErrNotFound) {
|
|
return false, nil
|
|
}
|
|
return false, err
|
|
}
|
|
if entry == nil {
|
|
return false, nil
|
|
}
|
|
// Mirror filer_pb.Entry.IsDirectoryKeyObject so the cleaner keeps a promoted file's data.
|
|
_, isPrefixObject := entry.Extended[s3_constants.SeaweedFSPrefixObject]
|
|
return entry.IsDirectory() && (entry.Mime != "" || len(entry.GetChunks()) > 0 || len(entry.Content) > 0 || entry.IsInRemoteOnly() || isPrefixObject), nil
|
|
}
|