mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* feat(mount): pre-allocate file IDs in pool for writeback cache mode When writeback caching is enabled, chunk uploads no longer block on a per-chunk AssignVolume RPC. Instead, a FileIdPool pre-allocates file IDs in batches using a single AssignVolume(Count=N, ExpectedDataSize=ChunkSize) call and hands them out instantly to upload workers. Pool size is 2x ConcurrentWriters, refilled in background when it drops below ConcurrentWriters. Entries expire after 25s to respect JWT TTL. Sequential needle keys are generated from the base file ID returned by the master, so one Assign RPC produces N usable IDs. This cuts per-chunk upload latency from 2 RTTs (assign + upload) to 1 RTT (upload only), with the assign cost amortized across the batch. * test: add benchmarks for file ID pool vs direct assign Benchmarks measure: - Pool Get vs Direct AssignVolume at various simulated latencies - Batch assign scaling (Count=1 through Count=32) - Concurrent pool access with 1-64 workers Results on Apple M4: - Pool Get: constant ~3ns regardless of assign latency - Batch=16: 15.7x more IDs/sec than individual assigns - 64 concurrent workers: 19M IDs/sec throughput * fix(mount): address review feedback on file ID pool 1. Fix race condition in Get(): use sync.Cond so callers wait for an in-flight refill instead of returning an error when the pool is empty. 2. Match default pool size to async flush worker count (128, not 16) when ConcurrentWriters is unset. 3. Add logging to UploadWithAssignFunc for consistency with UploadWithRetry. 4. Document that pooled assigns omit the Path field, bypassing path-based storage rules (filer.conf). This is an intentional tradeoff for writeback cache performance. 5. Fix flaky expiry test: widen time margin from 50ms to 1s. 6. Add TestFileIdPoolGetWaitsForRefill to verify concurrent waiters. * fix(mount): use individual Count=1 assigns to get per-fid JWTs The master generates one JWT per AssignResponse, bound to the base file ID (master_grpc_server_assign.go:158). The volume server validates that the JWT's Fid matches the upload exactly (volume_server_handlers.go:367). Using Count=N and deriving sequential IDs would fail this check. Switch to individual Count=1 RPCs over a single gRPC connection. This still amortizes connection overhead while getting a correct per-fid JWT for each entry. Partial batches are accepted if some requests fail. Remove unused needle import now that sequential ID generation is gone. * fix(mount): separate pprof from FUSE protocol debug logging The -debug flag was enabling both the pprof HTTP server and the noisy go-fuse protocol logging (rx/tx lines for every FUSE operation). This makes profiling impractical as the log output dominates. Split into two flags: - -debug: enables pprof HTTP server only (for profiling) - -debug.fuse: enables raw FUSE protocol request/response logging * perf(mount): replace LevelDB read+write with in-memory overlay for dir mtime Profile showed TouchDirMtimeCtime at 0.22s — every create/rename/unlink in a directory did a LevelDB FindEntry (read) + UpdateEntry (write) just to bump the parent dir's mtime/ctime. Replace with an in-memory map (same pattern as existing atime overlay): - touchDirMtimeCtimeLocal now stores inode→timestamp in dirMtimeMap - applyInMemoryDirMtime overlays onto GetAttr/Lookup output - No LevelDB I/O on the mutation hot path The overlay only advances timestamps forward (max of stored vs overlay), so stale entries are harmless. Map is bounded at 8192 entries. * perf(mount): skip self-originated metadata subscription events in writeback mode With writeback caching, this mount is the single writer. All local mutations are already applied to the local meta cache (via applyLocalMetadataEvent or direct InsertEntry). The filer subscription then delivers the same event back, causing redundant work: proto.Clone, enqueue to apply loop, dedup ring check, and sometimes redundant LevelDB writes when the dedup ring misses (deferred creates). Check EventNotification.Signatures against selfSignature and skip events that originated from this mount. This eliminates the redundant processing for every self-originated mutation. * perf(mount): increase kernel FUSE cache TTL in writeback cache mode With writeback caching, this mount is the single writer — the local meta cache is authoritative. Increase EntryValid and AttrValid from 1s to 10s so the kernel doesn't re-issue Lookup/GetAttr for every path component and stat call. This reduces FUSE /dev/fuse round-trips which dominate the profile at 38% of CPU (syscall.rawsyscalln). Each saved round-trip eliminates a kernel→userspace→kernel transition. Normal (non-writeback) mode retains the 1s TTL for multi-mount consistency.
440 lines
12 KiB
Go
440 lines
12 KiB
Go
package mount
|
|
|
|
import (
|
|
"os"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/go-fuse/v2/fuse"
|
|
"github.com/seaweedfs/seaweedfs/weed/filer"
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse.AttrOut) (code fuse.Status) {
|
|
glog.V(4).Infof("GetAttr %v", input.NodeId)
|
|
if input.NodeId == 1 {
|
|
wfs.setRootAttr(out)
|
|
if wfs.option.PosixDirNlink {
|
|
wfs.applyDirNlink(&out.Attr, util.FullPath(wfs.option.FilerMountRootPath))
|
|
}
|
|
return fuse.OK
|
|
}
|
|
|
|
inode := input.NodeId
|
|
path, _, entry, status := wfs.maybeReadEntry(inode)
|
|
if status == fuse.OK {
|
|
out.AttrValid = wfs.attrValidSec
|
|
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
|
|
wfs.applyInMemoryAtime(&out.Attr, inode)
|
|
if entry.IsDirectory {
|
|
wfs.applyInMemoryDirMtime(&out.Attr, inode)
|
|
if wfs.option.PosixDirNlink {
|
|
wfs.applyDirNlink(&out.Attr, path)
|
|
}
|
|
}
|
|
return status
|
|
} else {
|
|
if fh, found := wfs.fhMap.FindFileHandle(inode); found {
|
|
out.AttrValid = wfs.attrValidSec
|
|
// Use shared lock to prevent race with Write operations
|
|
fhActiveLock := wfs.fhLockTable.AcquireLock("GetAttr", fh.fh, util.SharedLock)
|
|
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.GetEntry(), true)
|
|
wfs.fhLockTable.ReleaseLock(fh.fh, fhActiveLock)
|
|
wfs.applyInMemoryAtime(&out.Attr, inode)
|
|
out.Nlink = 0
|
|
return fuse.OK
|
|
}
|
|
}
|
|
|
|
return status
|
|
}
|
|
|
|
func (wfs *WFS) SetAttr(cancel <-chan struct{}, input *fuse.SetAttrIn, out *fuse.AttrOut) (code fuse.Status) {
|
|
|
|
// Check quota including uncommitted writes for real-time enforcement
|
|
if wfs.IsOverQuotaWithUncommitted() {
|
|
return fuse.Status(syscall.ENOSPC)
|
|
}
|
|
|
|
path, fh, entry, status := wfs.maybeReadEntry(input.NodeId)
|
|
if status != fuse.OK || entry == nil {
|
|
return status
|
|
}
|
|
if fh != nil {
|
|
fh.entryLock.Lock()
|
|
defer fh.entryLock.Unlock()
|
|
}
|
|
|
|
wormEnforced, wormEnabled := wfs.wormEnforcedForEntry(path, entry)
|
|
if wormEnforced {
|
|
return fuse.EPERM
|
|
}
|
|
|
|
if size, ok := input.GetSize(); ok {
|
|
glog.V(4).Infof("%v setattr set size=%v chunks=%d", path, size, len(entry.GetChunks()))
|
|
if size < filer.FileSize(entry) {
|
|
// fmt.Printf("truncate %v \n", fullPath)
|
|
var chunks []*filer_pb.FileChunk
|
|
var truncatedChunks []*filer_pb.FileChunk
|
|
for _, chunk := range entry.GetChunks() {
|
|
int64Size := int64(chunk.Size)
|
|
if chunk.Offset+int64Size > int64(size) {
|
|
// this chunk is truncated
|
|
int64Size = int64(size) - chunk.Offset
|
|
if int64Size > 0 {
|
|
chunks = append(chunks, chunk)
|
|
glog.V(4).Infof("truncated chunk %+v from %d to %d\n", chunk.GetFileIdString(), chunk.Size, int64Size)
|
|
chunk.Size = uint64(int64Size)
|
|
} else {
|
|
glog.V(4).Infof("truncated whole chunk %+v\n", chunk.GetFileIdString())
|
|
truncatedChunks = append(truncatedChunks, chunk)
|
|
}
|
|
} else {
|
|
chunks = append(chunks, chunk)
|
|
}
|
|
}
|
|
// set the new chunks and reset entry cache
|
|
entry.Chunks = chunks
|
|
if fh != nil {
|
|
fh.entryChunkGroup.SetChunks(chunks)
|
|
}
|
|
}
|
|
truncNow := time.Now()
|
|
entry.Attributes.Mtime = truncNow.Unix()
|
|
entry.Attributes.MtimeNs = int32(truncNow.Nanosecond())
|
|
entry.Attributes.FileSize = size
|
|
|
|
}
|
|
|
|
if mode, ok := input.GetMode(); ok {
|
|
// commit the file to worm when it is set to readonly at the first time
|
|
if entry.WormEnforcedAtTsNs == 0 && wormEnabled && !hasWritePermission(mode) {
|
|
entry.WormEnforcedAtTsNs = time.Now().UnixNano()
|
|
}
|
|
|
|
// glog.V(4).Infof("setAttr mode %o", mode)
|
|
entry.Attributes.FileMode = chmod(entry.Attributes.FileMode, mode)
|
|
if input.NodeId == 1 {
|
|
wfs.option.MountMode = os.FileMode(chmod(uint32(wfs.option.MountMode), mode))
|
|
}
|
|
}
|
|
|
|
ownerChanged := false
|
|
if uid, ok := input.GetUID(); ok {
|
|
entry.Attributes.Uid = uid
|
|
ownerChanged = true
|
|
if input.NodeId == 1 {
|
|
wfs.option.MountUid = uid
|
|
}
|
|
}
|
|
|
|
if gid, ok := input.GetGID(); ok {
|
|
entry.Attributes.Gid = gid
|
|
ownerChanged = true
|
|
if input.NodeId == 1 {
|
|
wfs.option.MountGid = gid
|
|
}
|
|
}
|
|
|
|
// POSIX: clear SUID/SGID bits when ownership changes (unless caller is root).
|
|
if ownerChanged && input.Uid != 0 {
|
|
entry.Attributes.FileMode &^= 0o6000
|
|
}
|
|
|
|
if atime, ok := input.GetATime(); ok {
|
|
wfs.setAtime(input.NodeId, atime)
|
|
}
|
|
|
|
if mtime, ok := input.GetMTime(); ok {
|
|
entry.Attributes.Mtime = mtime.Unix()
|
|
entry.Attributes.MtimeNs = int32(mtime.Nanosecond())
|
|
}
|
|
|
|
// POSIX: update ctime on any metadata change.
|
|
now := time.Now()
|
|
entry.Attributes.Ctime = now.Unix()
|
|
entry.Attributes.CtimeNs = int32(now.Nanosecond())
|
|
|
|
out.AttrValid = wfs.attrValidSec
|
|
size, includeSize := input.GetSize()
|
|
if includeSize {
|
|
out.Attr.Size = size
|
|
}
|
|
wfs.setAttrByPbEntry(&out.Attr, input.NodeId, entry, !includeSize)
|
|
wfs.applyInMemoryAtime(&out.Attr, input.NodeId)
|
|
|
|
if fh != nil {
|
|
fh.dirtyMetadata = true
|
|
return fuse.OK
|
|
}
|
|
|
|
return wfs.saveEntry(path, entry)
|
|
|
|
}
|
|
|
|
func (wfs *WFS) setRootAttr(out *fuse.AttrOut) {
|
|
now := uint64(time.Now().Unix())
|
|
out.AttrValid = 119
|
|
out.Ino = 1
|
|
setBlksize(&out.Attr, blockSize)
|
|
out.Uid = wfs.option.MountUid
|
|
out.Gid = wfs.option.MountGid
|
|
out.Mtime = now
|
|
out.Ctime = now
|
|
out.Atime = now
|
|
out.Mode = toSyscallType(os.ModeDir) | uint32(wfs.option.MountMode)
|
|
out.Nlink = 2
|
|
}
|
|
|
|
func (wfs *WFS) setAttrByPbEntry(out *fuse.Attr, inode uint64, entry *filer_pb.Entry, calculateSize bool) {
|
|
out.Ino = inode
|
|
setBlksize(out, blockSize)
|
|
if entry == nil {
|
|
return
|
|
}
|
|
if entry.Attributes != nil && entry.Attributes.Inode != 0 {
|
|
out.Ino = entry.Attributes.Inode
|
|
}
|
|
if calculateSize {
|
|
out.Size = filer.FileSize(entry)
|
|
}
|
|
if entry.FileMode()&os.ModeSymlink != 0 {
|
|
out.Size = uint64(len(entry.Attributes.SymlinkTarget))
|
|
}
|
|
out.Blocks = (out.Size + blockSize - 1) / blockSize
|
|
out.Mtime = uint64(entry.Attributes.Mtime)
|
|
out.Mtimensec = uint32(entry.Attributes.MtimeNs)
|
|
if entry.Attributes.Ctime != 0 {
|
|
out.Ctime = uint64(entry.Attributes.Ctime)
|
|
out.Ctimensec = uint32(entry.Attributes.CtimeNs)
|
|
} else {
|
|
out.Ctime = uint64(entry.Attributes.Mtime)
|
|
out.Ctimensec = uint32(entry.Attributes.MtimeNs)
|
|
}
|
|
out.Atime = uint64(entry.Attributes.Mtime)
|
|
out.Atimensec = uint32(entry.Attributes.MtimeNs)
|
|
// In-memory atime overlay is applied by the caller via applyInMemoryAtime.
|
|
out.Mode = toSyscallMode(os.FileMode(entry.Attributes.FileMode))
|
|
if entry.IsDirectory {
|
|
out.Nlink = 2
|
|
} else if entry.HardLinkCounter > 0 {
|
|
out.Nlink = uint32(entry.HardLinkCounter)
|
|
} else {
|
|
out.Nlink = 1
|
|
}
|
|
out.Uid = entry.Attributes.Uid
|
|
out.Gid = entry.Attributes.Gid
|
|
out.Rdev = entry.Attributes.Rdev
|
|
}
|
|
|
|
func (wfs *WFS) setAttrByFilerEntry(out *fuse.Attr, inode uint64, entry *filer.Entry) {
|
|
out.Ino = inode
|
|
out.Size = entry.FileSize
|
|
if entry.Mode&os.ModeSymlink != 0 {
|
|
out.Size = uint64(len(entry.SymlinkTarget))
|
|
}
|
|
out.Blocks = (out.Size + blockSize - 1) / blockSize
|
|
setBlksize(out, blockSize)
|
|
out.Atime = uint64(entry.Attr.Mtime.Unix())
|
|
out.Atimensec = uint32(entry.Attr.Mtime.Nanosecond())
|
|
out.Mtime = uint64(entry.Attr.Mtime.Unix())
|
|
out.Mtimensec = uint32(entry.Attr.Mtime.Nanosecond())
|
|
if !entry.Attr.Ctime.IsZero() {
|
|
out.Ctime = uint64(entry.Attr.Ctime.Unix())
|
|
out.Ctimensec = uint32(entry.Attr.Ctime.Nanosecond())
|
|
} else {
|
|
out.Ctime = uint64(entry.Attr.Mtime.Unix())
|
|
out.Ctimensec = uint32(entry.Attr.Mtime.Nanosecond())
|
|
}
|
|
out.Mode = toSyscallMode(entry.Attr.Mode)
|
|
if entry.IsDirectory() {
|
|
out.Nlink = 2
|
|
} else if entry.HardLinkCounter > 0 {
|
|
out.Nlink = uint32(entry.HardLinkCounter)
|
|
} else {
|
|
out.Nlink = 1
|
|
}
|
|
out.Uid = entry.Attr.Uid
|
|
out.Gid = entry.Attr.Gid
|
|
out.Rdev = entry.Attr.Rdev
|
|
}
|
|
|
|
func (wfs *WFS) outputPbEntry(out *fuse.EntryOut, inode uint64, entry *filer_pb.Entry) {
|
|
out.NodeId = inode
|
|
out.Generation = 1
|
|
out.EntryValid = wfs.entryValidSec
|
|
out.AttrValid = wfs.attrValidSec
|
|
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
|
|
}
|
|
|
|
func (wfs *WFS) outputFilerEntry(out *fuse.EntryOut, inode uint64, entry *filer.Entry) {
|
|
out.NodeId = inode
|
|
out.Generation = 1
|
|
out.EntryValid = wfs.entryValidSec
|
|
out.AttrValid = wfs.attrValidSec
|
|
wfs.setAttrByFilerEntry(&out.Attr, inode, entry)
|
|
}
|
|
|
|
// touchDirMtimeCtimeBest updates a directory's mtime and ctime using the
|
|
// best strategy for the current mode:
|
|
// - WritebackCache: local meta cache only (no filer RPC)
|
|
// - Normal mode: filer UpdateEntry RPC for POSIX correctness
|
|
func (wfs *WFS) touchDirMtimeCtimeBest(dirPath util.FullPath) {
|
|
if wfs.option.WritebackCache {
|
|
wfs.touchDirMtimeCtimeLocal(dirPath)
|
|
} else {
|
|
wfs.touchDirMtimeCtime(dirPath)
|
|
}
|
|
}
|
|
|
|
// touchDirMtimeCtime updates a directory's mtime and ctime on the filer.
|
|
// POSIX requires this when entries are created or removed in the directory.
|
|
func (wfs *WFS) touchDirMtimeCtime(dirPath util.FullPath) {
|
|
dirEntry, code := wfs.maybeLoadEntry(dirPath)
|
|
if code != fuse.OK || dirEntry == nil || dirEntry.Attributes == nil {
|
|
return
|
|
}
|
|
now := time.Now()
|
|
dirEntry.Attributes.Mtime = now.Unix()
|
|
dirEntry.Attributes.MtimeNs = int32(now.Nanosecond())
|
|
dirEntry.Attributes.Ctime = now.Unix()
|
|
dirEntry.Attributes.CtimeNs = int32(now.Nanosecond())
|
|
wfs.saveEntry(dirPath, dirEntry)
|
|
}
|
|
|
|
// touchDirMtimeCtimeLocal updates a directory's mtime and ctime in an in-memory
|
|
// overlay, avoiding LevelDB reads and writes entirely. The overlay is applied
|
|
// by applyInMemoryDirMtime when GetAttr/Lookup reads the directory's attributes.
|
|
func (wfs *WFS) touchDirMtimeCtimeLocal(dirPath util.FullPath) {
|
|
if inode, found := wfs.inodeToPath.GetInode(dirPath); found {
|
|
wfs.setDirMtime(inode, time.Now())
|
|
}
|
|
}
|
|
|
|
const dirMtimeMapMaxSize = 8192
|
|
|
|
func (wfs *WFS) setDirMtime(inode uint64, t time.Time) {
|
|
wfs.dirMtimeMu.Lock()
|
|
defer wfs.dirMtimeMu.Unlock()
|
|
if len(wfs.dirMtimeMap) >= dirMtimeMapMaxSize {
|
|
for k := range wfs.dirMtimeMap {
|
|
delete(wfs.dirMtimeMap, k)
|
|
break
|
|
}
|
|
}
|
|
wfs.dirMtimeMap[inode] = t
|
|
}
|
|
|
|
// applyInMemoryDirMtime overlays the in-memory mtime/ctime onto fuse.Attr
|
|
// for directories that had recent child mutations.
|
|
func (wfs *WFS) applyInMemoryDirMtime(out *fuse.Attr, inode uint64) {
|
|
wfs.dirMtimeMu.Lock()
|
|
if t, ok := wfs.dirMtimeMap[inode]; ok {
|
|
sec := uint64(t.Unix())
|
|
nsec := uint32(t.Nanosecond())
|
|
if sec > out.Mtime || (sec == out.Mtime && nsec > out.Mtimensec) {
|
|
out.Mtime = sec
|
|
out.Mtimensec = nsec
|
|
out.Ctime = sec
|
|
out.Ctimensec = nsec
|
|
}
|
|
}
|
|
wfs.dirMtimeMu.Unlock()
|
|
}
|
|
|
|
const atimeMapMaxSize = 8192
|
|
|
|
// setAtime stores an in-memory atime for an inode. The map is bounded;
|
|
// when full, a random entry is evicted.
|
|
func (wfs *WFS) setAtime(inode uint64, t time.Time) {
|
|
wfs.atimeMu.Lock()
|
|
defer wfs.atimeMu.Unlock()
|
|
if len(wfs.atimeMap) >= atimeMapMaxSize {
|
|
// evict one random entry
|
|
for k := range wfs.atimeMap {
|
|
delete(wfs.atimeMap, k)
|
|
break
|
|
}
|
|
}
|
|
wfs.atimeMap[inode] = t
|
|
}
|
|
|
|
// applyInMemoryAtime overlays the in-memory atime onto a fuse.Attr if present.
|
|
func (wfs *WFS) applyInMemoryAtime(out *fuse.Attr, inode uint64) {
|
|
wfs.atimeMu.Lock()
|
|
if t, ok := wfs.atimeMap[inode]; ok {
|
|
out.Atime = uint64(t.Unix())
|
|
out.Atimensec = uint32(t.Nanosecond())
|
|
}
|
|
wfs.atimeMu.Unlock()
|
|
}
|
|
|
|
// applyDirNlink sets nlink = 2 + number_of_subdirectories for a directory.
|
|
// Uses the in-memory subdirectory count tracked by mkdir/rmdir/rename.
|
|
func (wfs *WFS) applyDirNlink(out *fuse.Attr, dirPath util.FullPath) {
|
|
count := wfs.inodeToPath.GetSubdirCount(dirPath)
|
|
if count > 0 {
|
|
out.Nlink = 2 + uint32(count)
|
|
}
|
|
}
|
|
|
|
func chmod(existing uint32, mode uint32) uint32 {
|
|
return existing&^07777 | mode&07777
|
|
}
|
|
|
|
const ownerWrite = 0o200
|
|
const groupWrite = 0o020
|
|
const otherWrite = 0o002
|
|
|
|
func hasWritePermission(mode uint32) bool {
|
|
return (mode&ownerWrite != 0) || (mode&groupWrite != 0) || (mode&otherWrite != 0)
|
|
}
|
|
|
|
func toSyscallMode(mode os.FileMode) uint32 {
|
|
return toSyscallType(mode) | uint32(mode)
|
|
}
|
|
|
|
func toSyscallType(mode os.FileMode) uint32 {
|
|
switch mode & os.ModeType {
|
|
case os.ModeDir:
|
|
return syscall.S_IFDIR
|
|
case os.ModeSymlink:
|
|
return syscall.S_IFLNK
|
|
case os.ModeNamedPipe:
|
|
return syscall.S_IFIFO
|
|
case os.ModeSocket:
|
|
return syscall.S_IFSOCK
|
|
case os.ModeDevice:
|
|
return syscall.S_IFBLK
|
|
case os.ModeCharDevice:
|
|
return syscall.S_IFCHR
|
|
default:
|
|
return syscall.S_IFREG
|
|
}
|
|
}
|
|
|
|
func toOsFileType(mode uint32) os.FileMode {
|
|
switch mode & (syscall.S_IFMT & 0xffff) {
|
|
case syscall.S_IFDIR:
|
|
return os.ModeDir
|
|
case syscall.S_IFLNK:
|
|
return os.ModeSymlink
|
|
case syscall.S_IFIFO:
|
|
return os.ModeNamedPipe
|
|
case syscall.S_IFSOCK:
|
|
return os.ModeSocket
|
|
case syscall.S_IFBLK:
|
|
return os.ModeDevice
|
|
case syscall.S_IFCHR:
|
|
return os.ModeCharDevice
|
|
default:
|
|
return 0
|
|
}
|
|
}
|
|
|
|
func toOsFileMode(mode uint32) os.FileMode {
|
|
return toOsFileType(mode) | os.FileMode(mode&07777)
|
|
}
|