Files
seaweedfs/weed/mount/weedfs_attr.go
T
Chris Lu 9896eade51 feat(mount): set FOPEN_KEEP_CACHE on re-open of unchanged files (#9097)
* feat(mount): set FOPEN_KEEP_CACHE when file mtime is unchanged

On re-open of an unmodified file, signal the kernel to preserve its
existing page cache. This eliminates redundant volume server reads for
workloads that repeatedly open-read-close the same files (build systems,
config readers, etc.).

* fix(mount): use guarded type assertion for openMtimeCache load

Use the two-value form of type assertion when loading from sync.Map
to prevent potential panics if a non-int64 value is ever stored.

* fix(mount): skip redundant mtime store and invalidate on truncation

- Avoid redundant sync.Map Store when cached mtime already matches
  the current mtime, reducing contention on the hot open path.
- Invalidate openMtimeCache in SetAttr when file size changes
  (truncation), preventing stale kernel page cache after ftruncate.

* fix(mount): use nanosecond mtime precision and bounded cache for FOPEN_KEEP_CACHE

- Compare both Mtime (seconds) and MtimeNs (nanoseconds) to detect
  sub-second modifications common in automated workloads.
- Replace unbounded sync.Map with a bounded map + mutex (8192 entries,
  random eviction when full), following the existing atimeMap pattern.
- Extract applyKeepCacheFlag and invalidateOpenMtimeCache methods for
  clarity and testability.
- Add tests for nanosecond precision and cache eviction.

* fix(mount): invalidate mtime cache in truncateEntry for O_TRUNC consistency

Add invalidateOpenMtimeCache call to truncateEntry so the Create path
with O_TRUNC follows the same explicit invalidation pattern as SetAttr
and Write.
2026-04-16 11:37:52 -07:00

443 lines
13 KiB
Go

package mount
import (
"os"
"syscall"
"time"
"github.com/seaweedfs/go-fuse/v2/fuse"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
)
func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse.AttrOut) (code fuse.Status) {
glog.V(4).Infof("GetAttr %v", input.NodeId)
if input.NodeId == 1 {
wfs.setRootAttr(out)
if wfs.option.PosixDirNlink {
wfs.applyDirNlink(&out.Attr, util.FullPath(wfs.option.FilerMountRootPath))
}
return fuse.OK
}
inode := input.NodeId
path, _, entry, status := wfs.maybeReadEntry(inode)
if status == fuse.OK {
out.AttrValid = wfs.attrValidSec
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
wfs.applyInMemoryAtime(&out.Attr, inode)
if entry.IsDirectory {
wfs.applyInMemoryDirMtime(&out.Attr, inode)
if wfs.option.PosixDirNlink {
wfs.applyDirNlink(&out.Attr, path)
}
}
return status
} else {
if fh, found := wfs.fhMap.FindFileHandle(inode); found {
out.AttrValid = wfs.attrValidSec
// Use shared lock to prevent race with Write operations
fhActiveLock := wfs.fhLockTable.AcquireLock("GetAttr", fh.fh, util.SharedLock)
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.GetEntry(), true)
wfs.fhLockTable.ReleaseLock(fh.fh, fhActiveLock)
wfs.applyInMemoryAtime(&out.Attr, inode)
out.Nlink = 0
return fuse.OK
}
}
return status
}
func (wfs *WFS) SetAttr(cancel <-chan struct{}, input *fuse.SetAttrIn, out *fuse.AttrOut) (code fuse.Status) {
// Check quota including uncommitted writes for real-time enforcement
if wfs.IsOverQuotaWithUncommitted() {
return fuse.Status(syscall.ENOSPC)
}
path, fh, entry, status := wfs.maybeReadEntry(input.NodeId)
if status != fuse.OK || entry == nil {
return status
}
if fh != nil {
fh.entryLock.Lock()
defer fh.entryLock.Unlock()
}
wormEnforced, wormEnabled := wfs.wormEnforcedForEntry(path, entry)
if wormEnforced {
return fuse.EPERM
}
if size, ok := input.GetSize(); ok {
glog.V(4).Infof("%v setattr set size=%v chunks=%d", path, size, len(entry.GetChunks()))
// Invalidate the open-mtime cache so the next Open does not set
// FOPEN_KEEP_CACHE with stale kernel page cache data.
wfs.invalidateOpenMtimeCache(input.NodeId)
if size < filer.FileSize(entry) {
// fmt.Printf("truncate %v \n", fullPath)
var chunks []*filer_pb.FileChunk
var truncatedChunks []*filer_pb.FileChunk
for _, chunk := range entry.GetChunks() {
int64Size := int64(chunk.Size)
if chunk.Offset+int64Size > int64(size) {
// this chunk is truncated
int64Size = int64(size) - chunk.Offset
if int64Size > 0 {
chunks = append(chunks, chunk)
glog.V(4).Infof("truncated chunk %+v from %d to %d\n", chunk.GetFileIdString(), chunk.Size, int64Size)
chunk.Size = uint64(int64Size)
} else {
glog.V(4).Infof("truncated whole chunk %+v\n", chunk.GetFileIdString())
truncatedChunks = append(truncatedChunks, chunk)
}
} else {
chunks = append(chunks, chunk)
}
}
// set the new chunks and reset entry cache
entry.Chunks = chunks
if fh != nil {
fh.entryChunkGroup.SetChunks(chunks)
}
}
truncNow := time.Now()
entry.Attributes.Mtime = truncNow.Unix()
entry.Attributes.MtimeNs = int32(truncNow.Nanosecond())
entry.Attributes.FileSize = size
}
if mode, ok := input.GetMode(); ok {
// commit the file to worm when it is set to readonly at the first time
if entry.WormEnforcedAtTsNs == 0 && wormEnabled && !hasWritePermission(mode) {
entry.WormEnforcedAtTsNs = time.Now().UnixNano()
}
// glog.V(4).Infof("setAttr mode %o", mode)
entry.Attributes.FileMode = chmod(entry.Attributes.FileMode, mode)
if input.NodeId == 1 {
wfs.option.MountMode = os.FileMode(chmod(uint32(wfs.option.MountMode), mode))
}
}
ownerChanged := false
if uid, ok := input.GetUID(); ok {
entry.Attributes.Uid = uid
ownerChanged = true
if input.NodeId == 1 {
wfs.option.MountUid = uid
}
}
if gid, ok := input.GetGID(); ok {
entry.Attributes.Gid = gid
ownerChanged = true
if input.NodeId == 1 {
wfs.option.MountGid = gid
}
}
// POSIX: clear SUID/SGID bits when ownership changes (unless caller is root).
if ownerChanged && input.Uid != 0 {
entry.Attributes.FileMode &^= 0o6000
}
if atime, ok := input.GetATime(); ok {
wfs.setAtime(input.NodeId, atime)
}
if mtime, ok := input.GetMTime(); ok {
entry.Attributes.Mtime = mtime.Unix()
entry.Attributes.MtimeNs = int32(mtime.Nanosecond())
}
// POSIX: update ctime on any metadata change.
now := time.Now()
entry.Attributes.Ctime = now.Unix()
entry.Attributes.CtimeNs = int32(now.Nanosecond())
out.AttrValid = wfs.attrValidSec
size, includeSize := input.GetSize()
if includeSize {
out.Attr.Size = size
}
wfs.setAttrByPbEntry(&out.Attr, input.NodeId, entry, !includeSize)
wfs.applyInMemoryAtime(&out.Attr, input.NodeId)
if fh != nil {
fh.dirtyMetadata = true
return fuse.OK
}
return wfs.saveEntry(path, entry)
}
func (wfs *WFS) setRootAttr(out *fuse.AttrOut) {
now := uint64(time.Now().Unix())
out.AttrValid = 119
out.Ino = 1
setBlksize(&out.Attr, blockSize)
out.Uid = wfs.option.MountUid
out.Gid = wfs.option.MountGid
out.Mtime = now
out.Ctime = now
out.Atime = now
out.Mode = toSyscallType(os.ModeDir) | uint32(wfs.option.MountMode)
out.Nlink = 2
}
func (wfs *WFS) setAttrByPbEntry(out *fuse.Attr, inode uint64, entry *filer_pb.Entry, calculateSize bool) {
out.Ino = inode
setBlksize(out, blockSize)
if entry == nil {
return
}
if entry.Attributes != nil && entry.Attributes.Inode != 0 {
out.Ino = entry.Attributes.Inode
}
if calculateSize {
out.Size = filer.FileSize(entry)
}
if entry.FileMode()&os.ModeSymlink != 0 {
out.Size = uint64(len(entry.Attributes.SymlinkTarget))
}
out.Blocks = (out.Size + blockSize - 1) / blockSize
out.Mtime = uint64(entry.Attributes.Mtime)
out.Mtimensec = uint32(entry.Attributes.MtimeNs)
if entry.Attributes.Ctime != 0 {
out.Ctime = uint64(entry.Attributes.Ctime)
out.Ctimensec = uint32(entry.Attributes.CtimeNs)
} else {
out.Ctime = uint64(entry.Attributes.Mtime)
out.Ctimensec = uint32(entry.Attributes.MtimeNs)
}
out.Atime = uint64(entry.Attributes.Mtime)
out.Atimensec = uint32(entry.Attributes.MtimeNs)
// In-memory atime overlay is applied by the caller via applyInMemoryAtime.
out.Mode = toSyscallMode(os.FileMode(entry.Attributes.FileMode))
if entry.IsDirectory {
out.Nlink = 2
} else if entry.HardLinkCounter > 0 {
out.Nlink = uint32(entry.HardLinkCounter)
} else {
out.Nlink = 1
}
out.Uid = entry.Attributes.Uid
out.Gid = entry.Attributes.Gid
out.Rdev = entry.Attributes.Rdev
}
func (wfs *WFS) setAttrByFilerEntry(out *fuse.Attr, inode uint64, entry *filer.Entry) {
out.Ino = inode
out.Size = entry.FileSize
if entry.Mode&os.ModeSymlink != 0 {
out.Size = uint64(len(entry.SymlinkTarget))
}
out.Blocks = (out.Size + blockSize - 1) / blockSize
setBlksize(out, blockSize)
out.Atime = uint64(entry.Attr.Mtime.Unix())
out.Atimensec = uint32(entry.Attr.Mtime.Nanosecond())
out.Mtime = uint64(entry.Attr.Mtime.Unix())
out.Mtimensec = uint32(entry.Attr.Mtime.Nanosecond())
if !entry.Attr.Ctime.IsZero() {
out.Ctime = uint64(entry.Attr.Ctime.Unix())
out.Ctimensec = uint32(entry.Attr.Ctime.Nanosecond())
} else {
out.Ctime = uint64(entry.Attr.Mtime.Unix())
out.Ctimensec = uint32(entry.Attr.Mtime.Nanosecond())
}
out.Mode = toSyscallMode(entry.Attr.Mode)
if entry.IsDirectory() {
out.Nlink = 2
} else if entry.HardLinkCounter > 0 {
out.Nlink = uint32(entry.HardLinkCounter)
} else {
out.Nlink = 1
}
out.Uid = entry.Attr.Uid
out.Gid = entry.Attr.Gid
out.Rdev = entry.Attr.Rdev
}
func (wfs *WFS) outputPbEntry(out *fuse.EntryOut, inode uint64, entry *filer_pb.Entry) {
out.NodeId = inode
out.Generation = 1
out.EntryValid = wfs.entryValidSec
out.AttrValid = wfs.attrValidSec
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
}
func (wfs *WFS) outputFilerEntry(out *fuse.EntryOut, inode uint64, entry *filer.Entry) {
out.NodeId = inode
out.Generation = 1
out.EntryValid = wfs.entryValidSec
out.AttrValid = wfs.attrValidSec
wfs.setAttrByFilerEntry(&out.Attr, inode, entry)
}
// touchDirMtimeCtimeBest updates a directory's mtime and ctime using the
// best strategy for the current mode:
// - WritebackCache: local meta cache only (no filer RPC)
// - Normal mode: filer UpdateEntry RPC for POSIX correctness
func (wfs *WFS) touchDirMtimeCtimeBest(dirPath util.FullPath) {
if wfs.option.WritebackCache {
wfs.touchDirMtimeCtimeLocal(dirPath)
} else {
wfs.touchDirMtimeCtime(dirPath)
}
}
// touchDirMtimeCtime updates a directory's mtime and ctime on the filer.
// POSIX requires this when entries are created or removed in the directory.
func (wfs *WFS) touchDirMtimeCtime(dirPath util.FullPath) {
dirEntry, code := wfs.maybeLoadEntry(dirPath)
if code != fuse.OK || dirEntry == nil || dirEntry.Attributes == nil {
return
}
now := time.Now()
dirEntry.Attributes.Mtime = now.Unix()
dirEntry.Attributes.MtimeNs = int32(now.Nanosecond())
dirEntry.Attributes.Ctime = now.Unix()
dirEntry.Attributes.CtimeNs = int32(now.Nanosecond())
wfs.saveEntry(dirPath, dirEntry)
}
// touchDirMtimeCtimeLocal updates a directory's mtime and ctime in an in-memory
// overlay, avoiding LevelDB reads and writes entirely. The overlay is applied
// by applyInMemoryDirMtime when GetAttr/Lookup reads the directory's attributes.
func (wfs *WFS) touchDirMtimeCtimeLocal(dirPath util.FullPath) {
if inode, found := wfs.inodeToPath.GetInode(dirPath); found {
wfs.setDirMtime(inode, time.Now())
}
}
const dirMtimeMapMaxSize = 8192
func (wfs *WFS) setDirMtime(inode uint64, t time.Time) {
wfs.dirMtimeMu.Lock()
defer wfs.dirMtimeMu.Unlock()
if len(wfs.dirMtimeMap) >= dirMtimeMapMaxSize {
for k := range wfs.dirMtimeMap {
delete(wfs.dirMtimeMap, k)
break
}
}
wfs.dirMtimeMap[inode] = t
}
// applyInMemoryDirMtime overlays the in-memory mtime/ctime onto fuse.Attr
// for directories that had recent child mutations.
func (wfs *WFS) applyInMemoryDirMtime(out *fuse.Attr, inode uint64) {
wfs.dirMtimeMu.Lock()
if t, ok := wfs.dirMtimeMap[inode]; ok {
sec := uint64(t.Unix())
nsec := uint32(t.Nanosecond())
if sec > out.Mtime || (sec == out.Mtime && nsec > out.Mtimensec) {
out.Mtime = sec
out.Mtimensec = nsec
out.Ctime = sec
out.Ctimensec = nsec
}
}
wfs.dirMtimeMu.Unlock()
}
const atimeMapMaxSize = 8192
// setAtime stores an in-memory atime for an inode. The map is bounded;
// when full, a random entry is evicted.
func (wfs *WFS) setAtime(inode uint64, t time.Time) {
wfs.atimeMu.Lock()
defer wfs.atimeMu.Unlock()
if len(wfs.atimeMap) >= atimeMapMaxSize {
// evict one random entry
for k := range wfs.atimeMap {
delete(wfs.atimeMap, k)
break
}
}
wfs.atimeMap[inode] = t
}
// applyInMemoryAtime overlays the in-memory atime onto a fuse.Attr if present.
func (wfs *WFS) applyInMemoryAtime(out *fuse.Attr, inode uint64) {
wfs.atimeMu.Lock()
if t, ok := wfs.atimeMap[inode]; ok {
out.Atime = uint64(t.Unix())
out.Atimensec = uint32(t.Nanosecond())
}
wfs.atimeMu.Unlock()
}
// applyDirNlink sets nlink = 2 + number_of_subdirectories for a directory.
// Uses the in-memory subdirectory count tracked by mkdir/rmdir/rename.
func (wfs *WFS) applyDirNlink(out *fuse.Attr, dirPath util.FullPath) {
count := wfs.inodeToPath.GetSubdirCount(dirPath)
if count > 0 {
out.Nlink = 2 + uint32(count)
}
}
func chmod(existing uint32, mode uint32) uint32 {
return existing&^07777 | mode&07777
}
const ownerWrite = 0o200
const groupWrite = 0o020
const otherWrite = 0o002
func hasWritePermission(mode uint32) bool {
return (mode&ownerWrite != 0) || (mode&groupWrite != 0) || (mode&otherWrite != 0)
}
func toSyscallMode(mode os.FileMode) uint32 {
return toSyscallType(mode) | uint32(mode)
}
func toSyscallType(mode os.FileMode) uint32 {
switch mode & os.ModeType {
case os.ModeDir:
return syscall.S_IFDIR
case os.ModeSymlink:
return syscall.S_IFLNK
case os.ModeNamedPipe:
return syscall.S_IFIFO
case os.ModeSocket:
return syscall.S_IFSOCK
case os.ModeDevice:
return syscall.S_IFBLK
case os.ModeCharDevice:
return syscall.S_IFCHR
default:
return syscall.S_IFREG
}
}
func toOsFileType(mode uint32) os.FileMode {
switch mode & (syscall.S_IFMT & 0xffff) {
case syscall.S_IFDIR:
return os.ModeDir
case syscall.S_IFLNK:
return os.ModeSymlink
case syscall.S_IFIFO:
return os.ModeNamedPipe
case syscall.S_IFSOCK:
return os.ModeSocket
case syscall.S_IFBLK:
return os.ModeDevice
case syscall.S_IFCHR:
return os.ModeCharDevice
default:
return 0
}
}
func toOsFileMode(mode uint32) os.FileMode {
return toOsFileType(mode) | os.FileMode(mode&07777)
}