mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 06:22:05 +02:00
* feat(filer): option to store system metadata logs in their own collection The filer's internal /topics/.system/log chunks are assigned to the filer's default collection (-collection). In a multi-filer deployment that default is often empty, so every restart flap, full-sync, or event-buffered flush grows the default collection with system chunks that are indistinguishable from user data in collection.list. This is a large part of what makes the default collection balloon and confuses orphan analysis. This keeps the internal log in a dedicated collection when the operator asks for one, without changing where user data goes: - New optional override, filer.options.metaLog.collection (and .replication), read in NewFiler so both `weed filer` and `weed server -filer` honour it. Default "" => exactly today's behaviour (log follows the filer default), fully backward compatible. - Resolution is a small helper: override first, then the filer default, then a storage rule matched on the log path. Kept separate from the user write path so the internal log targets itself. - bucketCollection() is hardened the same way it already protects the filer's default collection: a bucket that happens to resolve to the redirected meta-log collection must not drop it on delete, because it backs internal log volumes. - Scaffold filer.toml documents the new knobs under [filer.options]. Related to the persisted deletion ledger branch (fix/persist-deletion-queue): together they cut the two sources of post-flap junk in the default collection — that PR stops orphaned user-chunk leak on filer crash, this one stops the internal log from living in default at all. They are independent: no file overlap, no functional dependency; either can merge first. They are paired only in the narrative of cleaning up default. Adds unit tests for the collection/replication resolution chain, the viper keys, and the bucket-delete guard (run green under -race). Co-Authored-By: Athena 🏛️ <hermes-agent@local> (custom / Qwen3.8-Flash-Next-ROCmFP4) * filer: collect bucket chunks when its collection survives the delete bucketCollection returning "" preserves the collection, but the bucket path still skipped per-entry chunk collection and could skip listing the children entirely, so a bucket sharing the meta-log (or any preserved) collection left its object chunks orphaned with no entry pointing at them. Only the wholesale drop of a deleted collection skips those now. Note in filer.toml that the meta-log target should stay stable: chunks written under an older collection are not migrated. * filer: exercise the metaLog override wiring through NewFiler The viper test only echoed back the keys it set, so a wrong key in NewFiler would still pass. It now asserts the fields NewFiler fills from those keys. Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: tighten comments around the metaLog collection override Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Chris Lu <chris.lu@gmail.com> Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com> Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
344 lines
12 KiB
Go
344 lines
12 KiB
Go
package filer
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
const (
|
|
MsgFailDelNonEmptyFolder = "fail to delete non-empty folder"
|
|
)
|
|
|
|
// ErrNonEmptyFolder is a non-recursive delete refused because the folder still
|
|
// has children. The marker leads the message the filer builds for it and is
|
|
// never wrapped on the way out, so the path that follows it, which the client
|
|
// chose, cannot forge one.
|
|
var ErrNonEmptyFolder = errors.New(MsgFailDelNonEmptyFolder)
|
|
|
|
// DeleteEntryError turns the text of DeleteEntryResponse.Error back into an
|
|
// error carrying the condition the filer reported. Call it on the response
|
|
// field, before formatting a path around it.
|
|
func DeleteEntryError(msg string) error {
|
|
if strings.HasPrefix(msg, MsgFailDelNonEmptyFolder) {
|
|
return &deleteEntryError{msg: msg, cause: ErrNonEmptyFolder}
|
|
}
|
|
return errors.New(msg)
|
|
}
|
|
|
|
// IsNonEmptyFolderError is for callers holding a delete failure that has not
|
|
// been wrapped yet: the sentinel when it survived, the leading marker when the
|
|
// error only crossed the wire as text.
|
|
func IsNonEmptyFolderError(err error) bool {
|
|
if err == nil {
|
|
return false
|
|
}
|
|
return errors.Is(err, ErrNonEmptyFolder) || strings.HasPrefix(err.Error(), MsgFailDelNonEmptyFolder)
|
|
}
|
|
|
|
type deleteEntryError struct {
|
|
msg string
|
|
cause error
|
|
}
|
|
|
|
func (e *deleteEntryError) Error() string { return e.msg }
|
|
func (e *deleteEntryError) Unwrap() error { return e.cause }
|
|
|
|
type OnChunksFunc func([]*filer_pb.FileChunk) error
|
|
type OnHardLinkIdsFunc func([]HardLinkId) error
|
|
|
|
func (f *Filer) DeleteEntryMetaAndData(ctx context.Context, p util.FullPath, isRecursive, ignoreRecursiveError, shouldDeleteChunks, isFromOtherCluster bool, signatures []int32, ifNotModifiedAfter int64) (err error) {
|
|
if p == "/" {
|
|
return nil
|
|
}
|
|
|
|
entry, findErr := f.FindEntry(ctx, p)
|
|
if findErr != nil {
|
|
return findErr
|
|
}
|
|
if ifNotModifiedAfter > 0 && entry.Attr.Mtime.Unix() > ifNotModifiedAfter {
|
|
return nil
|
|
}
|
|
isDeleteCollection := f.IsBucket(entry)
|
|
collectionName := ""
|
|
if isDeleteCollection {
|
|
collectionName = f.bucketCollection(ctx, entry.Name())
|
|
}
|
|
// A preserved collection outlives the bucket, so its chunks are collected
|
|
// per entry rather than dropped wholesale with it.
|
|
dropsCollection := isDeleteCollection && collectionName != ""
|
|
if entry.IsDirectory() {
|
|
// delete the folder children, not including the folder itself
|
|
err = f.doBatchDeleteFolderMetaAndData(ctx, entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks && !dropsCollection, isDeleteCollection && (dropsCollection || !shouldDeleteChunks), isFromOtherCluster, signatures, func(hardLinkIds []HardLinkId) error {
|
|
// A case not handled:
|
|
// what if the chunk is in a different collection?
|
|
if shouldDeleteChunks {
|
|
f.maybeDeleteHardLinks(ctx, hardLinkIds)
|
|
}
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
glog.V(2).InfofCtx(ctx, "delete directory %s: %v", p, err)
|
|
if errors.Is(err, ErrNonEmptyFolder) {
|
|
return err
|
|
}
|
|
return fmt.Errorf("delete directory %s: %v", p, err)
|
|
}
|
|
}
|
|
|
|
// delete the file or folder
|
|
err = f.doDeleteEntryMetaAndData(ctx, entry, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
if err != nil {
|
|
return fmt.Errorf("delete file %s: %v", p, err)
|
|
}
|
|
|
|
if shouldDeleteChunks && !dropsCollection {
|
|
if len(entry.HardLinkId) != 0 && entry.HardLinkCounter > 1 {
|
|
// if the file is a hard link and there are other hard links, do not delete the chunks
|
|
} else {
|
|
f.DeleteChunks(ctx, p, entry.GetChunks())
|
|
}
|
|
}
|
|
|
|
if isDeleteCollection {
|
|
if collectionName != "" {
|
|
// the entry is already gone: a caller that hung up must not leave the
|
|
// collection behind, so this cleanup outlives the request -- bounded all
|
|
// the same, or a master that is down parks this handler indefinitely and
|
|
// every client retry behind it parks another
|
|
collectionCtx, cancelCollection := context.WithTimeout(context.WithoutCancel(ctx), collectionDeleteTimeout)
|
|
f.DoDeleteCollection(collectionCtx, collectionName)
|
|
cancelCollection()
|
|
}
|
|
// drop bucket-labeled series held by this process; the S3 gateway
|
|
// only cleans its own registry
|
|
stats.DeleteBucketMetrics(entry.Name())
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) doBatchDeleteFolderMetaAndData(ctx context.Context, entry *Entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks, isDeletingBucket, isFromOtherCluster bool, signatures []int32, onHardLinkIdsFn OnHardLinkIdsFunc) (err error) {
|
|
|
|
var dirTombstoneTs int64
|
|
if isRecursive {
|
|
// Tombstone the directory before its children: when the store drops
|
|
// the subtree without listing it, or a child error aborts the walk,
|
|
// the ancestor tombstone still covers every descendant.
|
|
dirTombstoneTs = time.Now().UnixNano()
|
|
f.noteRemoteDeletion(entry.FullPath, true, dirTombstoneTs)
|
|
}
|
|
|
|
//collect all the chunks of this layer and delete them together at the end
|
|
var chunksToDelete []*filer_pb.FileChunk
|
|
lastFileName := ""
|
|
includeLastFile := false
|
|
listedChildren := !isDeletingBucket || !f.Store.CanDropWholeBucket()
|
|
if listedChildren {
|
|
for {
|
|
entries, _, err := f.ListDirectoryEntries(ctx, entry.FullPath, lastFileName, includeLastFile, PaginationSize, "", "", "")
|
|
if err != nil {
|
|
// nothing was deleted; a leftover tombstone would hide the
|
|
// still-existing remote children
|
|
f.unnoteRemoteDeletion(entry.FullPath, dirTombstoneTs)
|
|
glog.ErrorfCtx(ctx, "list folder %s: %v", entry.FullPath, err)
|
|
return fmt.Errorf("list folder %s: %v", entry.FullPath, err)
|
|
}
|
|
if lastFileName == "" && !isRecursive && len(entries) > 0 {
|
|
// only for first iteration in the loop
|
|
glog.V(2).InfofCtx(ctx, "deleting a folder %s has children: %+v ...", entry.FullPath, entries[0].Name())
|
|
return fmt.Errorf("%w: %s", ErrNonEmptyFolder, entry.FullPath)
|
|
}
|
|
|
|
for _, sub := range entries {
|
|
lastFileName = sub.Name()
|
|
f.noteRemoteDeletion(sub.FullPath, sub.IsDirectory(), time.Now().UnixNano())
|
|
if sub.IsDirectory() {
|
|
subIsDeletingBucket := f.IsBucket(sub)
|
|
err = f.doBatchDeleteFolderMetaAndData(ctx, sub, isRecursive, ignoreRecursiveError, shouldDeleteChunks, subIsDeletingBucket, isFromOtherCluster, nil, onHardLinkIdsFn)
|
|
} else {
|
|
if !isFromOtherCluster {
|
|
if _, remoteErr := f.maybeDeleteFromRemote(ctx, sub); remoteErr != nil {
|
|
glog.Warningf("remote delete child %s: %v", sub.FullPath, remoteErr)
|
|
if !ignoreRecursiveError {
|
|
err = remoteErr
|
|
}
|
|
}
|
|
}
|
|
if err != nil && !ignoreRecursiveError {
|
|
break
|
|
}
|
|
f.NotifyUpdateEvent(ctx, sub, nil, shouldDeleteChunks, isFromOtherCluster, nil)
|
|
if len(sub.HardLinkId) != 0 {
|
|
// hard link chunk data are deleted separately
|
|
err = onHardLinkIdsFn([]HardLinkId{sub.HardLinkId})
|
|
} else {
|
|
if shouldDeleteChunks {
|
|
chunksToDelete = append(chunksToDelete, sub.GetChunks()...)
|
|
}
|
|
}
|
|
}
|
|
if err != nil && !ignoreRecursiveError {
|
|
return err
|
|
}
|
|
}
|
|
|
|
if len(entries) < PaginationSize {
|
|
break
|
|
}
|
|
}
|
|
}
|
|
|
|
glog.V(3).InfofCtx(ctx, "deleting directory %v delete chunks: %v", entry.FullPath, shouldDeleteChunks)
|
|
|
|
// a non-recursive delete already proved the folder empty above, so sweeping the
|
|
// children now can only remove entries that raced in after that listing
|
|
if isRecursive || !listedChildren {
|
|
if storeDeletionErr := f.Store.DeleteFolderChildren(ctx, entry.FullPath); storeDeletionErr != nil {
|
|
return fmt.Errorf("filer store delete: %w", storeDeletionErr)
|
|
}
|
|
}
|
|
|
|
f.NotifyUpdateEvent(ctx, entry, nil, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
f.DeleteChunks(ctx, entry.FullPath, chunksToDelete)
|
|
|
|
return nil
|
|
}
|
|
|
|
func (f *Filer) doDeleteEntryMetaAndData(ctx context.Context, entry *Entry, shouldDeleteChunks bool, isFromOtherCluster bool, signatures []int32) (err error) {
|
|
|
|
glog.V(3).InfofCtx(ctx, "deleting entry %v, delete chunks: %v", entry.FullPath, shouldDeleteChunks)
|
|
|
|
if !isFromOtherCluster {
|
|
if _, remoteDeletionErr := f.maybeDeleteFromRemote(ctx, entry); remoteDeletionErr != nil {
|
|
return remoteDeletionErr
|
|
}
|
|
}
|
|
|
|
f.noteRemoteDeletion(entry.FullPath, entry.IsDirectory(), time.Now().UnixNano())
|
|
|
|
if storeDeletionErr := f.Store.DeleteOneEntry(ctx, entry); storeDeletionErr != nil {
|
|
return fmt.Errorf("filer store delete: %w", storeDeletionErr)
|
|
}
|
|
|
|
if !entry.IsDirectory() {
|
|
f.NotifyUpdateEvent(ctx, entry, nil, shouldDeleteChunks, isFromOtherCluster, signatures)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// collectionDeleteTimeout bounds the collection delete a bucket entry's own
|
|
// delete leaves behind, which carries no deadline of its own. It bounds the wait
|
|
// and not the work: the master keeps deleting on its own fan-out once asked, so
|
|
// giving up costs the confirmation. Short enough that the S3 client waiting on
|
|
// the bucket delete, which pays this and then the gateway's own follow-up
|
|
// DeleteCollection, still has retry budget left.
|
|
const collectionDeleteTimeout = 15 * time.Second
|
|
|
|
// bucketCollection resolves the collection a bucket's objects land in
|
|
// through the same chain the write path uses -- a grouped gateway's explicit
|
|
// collection, then the storage rules, then the bucket name -- and reports it
|
|
// only when nothing outside the bucket can still route into it. A shared
|
|
// collection must survive the bucket delete: dropping it removes volumes
|
|
// other paths still write to. A listing failure keeps the collection, the
|
|
// safe side of an unknown.
|
|
func (f *Filer) bucketCollection(ctx context.Context, bucket string) (collection string) {
|
|
bucketDir := f.DirBucketsPath + "/" + bucket + "/"
|
|
resolve := func(dir, name string) string {
|
|
if f.MasterClient != nil {
|
|
if group := f.MasterClient.FilerGroup; group != "" {
|
|
return group + "_" + name
|
|
}
|
|
}
|
|
return util.Nvl(f.FilerConf.MatchStorageRule(dir).Collection, name)
|
|
}
|
|
collection = resolve(bucketDir, bucket)
|
|
|
|
// Rule-less writes outside buckets fall back to the filer's default
|
|
// collection, so a bucket resolving there shares it with them. The
|
|
// system metadata-log collection (when explicitly redirected via
|
|
// filer.options.metaLog.collection) is in the same boat: it backs internal
|
|
// log volumes, so a bucket that resolves there must never drop it
|
|
// either.
|
|
if collection == f.metaLogCollection {
|
|
return ""
|
|
}
|
|
if f.metaLogTargetCollection != "" && collection == f.metaLogTargetCollection {
|
|
return ""
|
|
}
|
|
|
|
// A rule whose prefix escapes the bucket can route other paths into the
|
|
// same collection, including prefixes nested under surviving buckets.
|
|
for _, rule := range f.FilerConf.ToProto().Locations {
|
|
prefix := strings.TrimSuffix(rule.LocationPrefix, "/") + "/"
|
|
if strings.HasPrefix(prefix, bucketDir) {
|
|
continue
|
|
}
|
|
if f.FilerConf.MatchStorageRule(prefix).Collection == collection {
|
|
return ""
|
|
}
|
|
}
|
|
|
|
siblings, err := f.listBuckets(ctx)
|
|
if err != nil {
|
|
glog.ErrorfCtx(ctx, "list buckets for collection check: %v", err)
|
|
return ""
|
|
}
|
|
for _, sibling := range siblings {
|
|
if sibling != bucket && resolve(f.DirBucketsPath+"/"+sibling+"/", sibling) == collection {
|
|
return ""
|
|
}
|
|
}
|
|
return collection
|
|
}
|
|
|
|
func (f *Filer) listBuckets(ctx context.Context) (buckets []string, err error) {
|
|
lastFileName := ""
|
|
for {
|
|
entries, _, listErr := f.ListDirectoryEntries(ctx, util.FullPath(f.DirBucketsPath), lastFileName, false, PaginationSize, "", "", "")
|
|
if listErr != nil {
|
|
return nil, listErr
|
|
}
|
|
for _, entry := range entries {
|
|
lastFileName = entry.Name()
|
|
if f.IsBucket(entry) {
|
|
buckets = append(buckets, entry.Name())
|
|
}
|
|
}
|
|
if len(entries) < PaginationSize {
|
|
return buckets, nil
|
|
}
|
|
}
|
|
}
|
|
|
|
func (f *Filer) DoDeleteCollection(ctx context.Context, collectionName string) (err error) {
|
|
|
|
return f.MasterClient.WithClient(ctx, false, func(client master_pb.SeaweedClient) error {
|
|
_, err := client.CollectionDelete(ctx, &master_pb.CollectionDeleteRequest{
|
|
Name: collectionName,
|
|
})
|
|
if err != nil {
|
|
glog.Infof("delete collection %s: %v", collectionName, err)
|
|
}
|
|
return err
|
|
})
|
|
|
|
}
|
|
|
|
func (f *Filer) maybeDeleteHardLinks(ctx context.Context, hardLinkIds []HardLinkId) {
|
|
for _, hardLinkId := range hardLinkIds {
|
|
if err := f.Store.DeleteHardLink(ctx, hardLinkId); err != nil {
|
|
glog.ErrorfCtx(ctx, "delete hard link id %d : %v", hardLinkId, err)
|
|
}
|
|
}
|
|
}
|