mount: read oversized directories through instead of caching them (#10631)

* mount: read oversized directories through instead of caching them

Visiting a directory pulls every child from the filer into the local
LevelDB before the first listing returns. For a directory of a few
million entries that is minutes of streaming, gigabytes of local store,
and gigabytes of decoded entries in flight -- paid by a mount that may
only walk the directory once.

A build that crosses -cacheDirMaxEntries (default ten thousand) now
stops, cleans up, and marks the directory read-through: listings stream
from the filer with pagination, the way update-hot directories already
do, and lookups in it consult the filer per entry as any uncached
directory does. The refusal is remembered, so the next visit fails fast
instead of streaming to the limit again, and an oversized ancestor is
stepped over when caching its subdirectories rather than wedging every
listing beneath it.

The direct path keeps the same pagination state on the handle, so a walk
that crosses the limit mid-flight carries on from where the cached walk
reached.

* mount: an ancestor found oversized must not fail its descendants

Visiting a directory builds its whole uncached ancestor chain in one
group, so the first discovery that an ancestor is oversized cancelled the
group and surfaced as the listed directory's own refusal: the descendant
build was aborted and the caller marked the descendant read-through,
leaving a perfectly cacheable directory streaming from the filer until
its inode was forgotten. The earlier test missed this by pre-marking the
ancestor, which exercises only the fast path.

The refusal of any directory other than the one being listed is now kept
out of the group's result; it is already remembered for the next visit.
This commit is contained in:
Chris Lu
2026-08-07 17:48:40 -07:00
committed by GitHub
parent 506ce0850b
commit dd73fee077
9 changed files with 238 additions and 12 deletions
+46 -4
View File
@@ -2,6 +2,7 @@ package meta_cache
import (
"context"
"errors"
"fmt"
"time"
@@ -13,7 +14,19 @@ import (
"github.com/seaweedfs/seaweedfs/weed/util"
)
func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.FullPath) error {
// DirectoryTooLargeError reports a directory the mount refuses to cache
// locally. Its listings read through to the filer instead.
type DirectoryTooLargeError struct {
Path util.FullPath
}
func (e *DirectoryTooLargeError) Error() string {
return fmt.Sprintf("directory %s is too large to cache locally", e.Path)
}
// maxCacheableEntries is the directory size above which a build gives up, or 0
// to cache everything.
func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.FullPath, maxCacheableEntries int) error {
// Collect all uncached paths from target directory up to root
var uncachedPaths []util.FullPath
currentPath := dirPath
@@ -23,7 +36,15 @@ func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.Full
if mc.isCachedFn(currentPath) {
break
}
uncachedPaths = append(uncachedPaths, currentPath)
if mc.isOversized(currentPath) {
// The directory itself reads through; an ancestor is stepped over,
// or it would wedge every listing beneath it forever.
if currentPath == dirPath {
return &DirectoryTooLargeError{Path: currentPath}
}
} else {
uncachedPaths = append(uncachedPaths, currentPath)
}
// Continue to parent directory
if currentPath != mc.root {
@@ -44,7 +65,16 @@ func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.Full
for _, p := range uncachedPaths {
path := p // capture for closure
g.Go(func() error {
return doEnsureVisited(ctx, mc, client, path)
err := doEnsureVisited(ctx, mc, client, path, maxCacheableEntries)
var tooLarge *DirectoryTooLargeError
if errors.As(err, &tooLarge) && path != dirPath {
// An ancestor found oversized just reads through; failing the
// group here would cancel the builds of its cacheable
// descendants, and the caller would treat the refusal as the
// listed directory's own.
return nil
}
return err
})
}
return g.Wait()
@@ -60,7 +90,7 @@ const (
emptyRebuildConfirmDelay = 50 * time.Millisecond
)
func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerClient, path util.FullPath) error {
func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerClient, path util.FullPath, maxCacheableEntries int) error {
// Use singleflight to deduplicate concurrent requests for the same path
_, err, _ := mc.visitGroup.Do(string(path), func() (interface{}, error) {
// Check for cancellation before starting
@@ -116,6 +146,9 @@ func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerCl
return nil
}
if maxCacheableEntries > 0 && entryCount >= maxCacheableEntries {
return &DirectoryTooLargeError{Path: path}
}
batch = append(batch, entry)
entryCount++
@@ -143,6 +176,15 @@ func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerCl
entryCount, snapshotTsNs, fetchErr := reloadFromFiler()
if fetchErr != nil {
var tooLarge *DirectoryTooLargeError
if errors.As(fetchErr, &tooLarge) {
// Remember the refusal so the next visit fails fast instead of
// streaming up to the limit again to rediscover it.
mc.markOversized(path)
glog.V(0).Infof("directory %s exceeds %d entries, reading it through instead of caching", path, maxCacheableEntries)
cleanupBuild("oversized")
return nil, fetchErr
}
cleanupBuild("failed")
return nil, fmt.Errorf("list %s: %w", path, fetchErr)
}