mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-08 15:41:15 +02:00
* mount: batched announcer + pooled peer conns for mount-to-mount RPCs * peer_announcer.go: non-blocking EnqueueAnnounce + ticker flush that groups fids by HRW owner, fans out one ChunkAnnounce per owner in parallel. announcedAt is pruned at 2× TTL so it stays bounded. * peer_dialer.go: PeerConnPool caches one grpc.ClientConn per peer address; the announcer and (next PR) the fetcher share it so steady-state owner RPCs skip the handshake cost entirely. Bounded at 4096 cached entries; shutdown conns are transparently replaced. * WFS starts both alongside the gRPC server; stops them on unmount. * mount: wire tryPeerRead via FetchChunk streaming gRPC Replaces the HTTP GET byte-transfer path with a gRPC server-stream FetchChunk call. Same fall-through semantics: any failure drops through to entryChunkGroup.ReadDataAt, so reads never slow below status quo. * peer_fetcher.go: tryPeerRead resolves the offset to a leaf chunk (flattening manifests), asks the HRW owner for holders via ChunkLookup, then opens FetchChunk on each holder in LRU order (PR #5) until one succeeds. Assembled bytes are verified against FileChunk.ETag end-to-end — the peer is still treated as untrusted. Reuses the shared PeerConnPool from PR #6 for all outbound gRPC. * peer_grpc.go: expose SelfAddr() so the fetcher can avoid dialing itself on a self-owned fid. * filehandle_read.go: tryPeerRead slot between tryRDMARead and entryChunkGroup.ReadDataAt. Gated by option.PeerEnabled and the presence of peerGrpcServer (the single identity test). Read ordering with the feature enabled is now: local cache -> RDMA sidecar -> peer mount (gRPC stream) -> volume server One port, one identity, one connection pool — no more HTTP bytecast. * test(fuse_p2p): end-to-end CI test for peer chunk sharing Adds a FUSE-backed integration test that proves mount B can satisfy a read from mount A's chunk cache instead of the volume tier. Layout (modelled on test/fuse_dlm): test/fuse_p2p/framework_test.go — cluster harness (1 master, 1 volume, 1 filer, N mounts, all with -peer.enable) test/fuse_p2p/peer_chunk_sharing_test.go — writer-reader scenario The test (TestPeerChunkSharing_ReadersPullFromPeerCache): 1. Starts 3 mounts. Three is the sweet spot: with 2 mounts, HRW owner of a chunk is self ~50 % of the time (peer path short-circuits); with 3+ it drops to ≤ 1/3, so a multi-chunk file almost certainly exercises the remote-owner fan-out. 2. Mount 0 writes a ~8 MiB file, then reads it back through its own FUSE to warm its chunk cache. 3. Waits for seed convergence (one full MountList refresh) plus an announcer flush cycle, so chunk-holder entries have reached each HRW owner. 4. Mount 1 reads the same file. 5. Verifies byte-for-byte equality AND greps mount 1's log for "peer read successful" — content matching alone is not proof (the volume fallback would also succeed), so the log marker is what distinguishes p2p from fallback. Workflow .github/workflows/fuse-p2p-integration.yml triggers on any change to mount/filer peer code, the p2p protos, or the test itself. Failure artifacts (server + mount logs) are uploaded for 3 days. Mounts run with -v=4 so the tryPeerRead success/failure glog messages land in the log file the test greps.
195 lines
6.4 KiB
Go
195 lines
6.4 KiB
Go
package mount
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"sort"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/filer"
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
)
|
|
|
|
func (fh *FileHandle) lockForRead(startOffset int64, size int) {
|
|
fh.dirtyPages.LockForRead(startOffset, startOffset+int64(size))
|
|
}
|
|
func (fh *FileHandle) unlockForRead(startOffset int64, size int) {
|
|
fh.dirtyPages.UnlockForRead(startOffset, startOffset+int64(size))
|
|
}
|
|
|
|
func (fh *FileHandle) readFromDirtyPages(buff []byte, startOffset int64, tsNs int64) (maxStop int64) {
|
|
maxStop = fh.dirtyPages.ReadDirtyDataAt(buff, startOffset, tsNs)
|
|
return
|
|
}
|
|
|
|
func (fh *FileHandle) readFromChunks(buff []byte, offset int64) (int64, int64, error) {
|
|
return fh.readFromChunksWithContext(context.Background(), buff, offset)
|
|
}
|
|
|
|
func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte, offset int64) (int64, int64, error) {
|
|
fh.entryLock.RLock()
|
|
defer fh.entryLock.RUnlock()
|
|
|
|
fileFullPath := fh.FullPath()
|
|
|
|
entry := fh.GetEntry()
|
|
|
|
if entry.IsInRemoteOnly() {
|
|
glog.V(4).Infof("download remote entry %s", fileFullPath)
|
|
err := fh.downloadRemoteEntry(entry)
|
|
if err != nil {
|
|
glog.V(1).Infof("download remote entry %s: %v", fileFullPath, err)
|
|
return 0, 0, err
|
|
}
|
|
}
|
|
|
|
fileSize := int64(entry.Attributes.FileSize)
|
|
if fileSize == 0 {
|
|
fileSize = int64(filer.FileSize(entry.GetEntry()))
|
|
}
|
|
|
|
if fileSize == 0 {
|
|
glog.V(1).Infof("empty fh %v", fileFullPath)
|
|
return 0, 0, io.EOF
|
|
} else if offset == fileSize {
|
|
return 0, 0, io.EOF
|
|
} else if offset >= fileSize {
|
|
glog.V(1).Infof("invalid read, fileSize %d, offset %d for %s", fileSize, offset, fileFullPath)
|
|
return 0, 0, io.EOF
|
|
}
|
|
|
|
if offset < int64(len(entry.Content)) {
|
|
totalRead := copy(buff, entry.Content[offset:])
|
|
glog.V(4).Infof("file handle read cached %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
|
return int64(totalRead), 0, nil
|
|
}
|
|
|
|
// Try RDMA acceleration first if available
|
|
if fh.wfs.rdmaClient != nil && fh.wfs.option.RdmaEnabled {
|
|
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, entry)
|
|
if err == nil {
|
|
glog.V(4).Infof("RDMA read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
|
return int64(totalRead), ts, nil
|
|
}
|
|
glog.V(4).Infof("RDMA read failed for %s, falling back to HTTP: %v", fileFullPath, err)
|
|
}
|
|
|
|
// Peer chunk sharing: try a peer mount's cache before the volume tier.
|
|
// Any failure falls through transparently. See design-weed-mount-
|
|
// peer-chunk-sharing.md §4.3.
|
|
if fh.wfs.option.PeerEnabled && fh.wfs.peerGrpcServer != nil {
|
|
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, entry)
|
|
if err == nil {
|
|
glog.V(4).Infof("peer read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
|
return int64(totalRead), ts, nil
|
|
}
|
|
// Skip the "failed" log for benign skip reasons (local cache
|
|
// hit, no peer owner yet, etc.) — the cache/volume fallback is
|
|
// the expected outcome, not a failure.
|
|
if err != errPeerReadSkipped {
|
|
glog.V(4).Infof("peer read failed for %s, falling back to volume: %v", fileFullPath, err)
|
|
}
|
|
}
|
|
|
|
// Fall back to normal chunk reading
|
|
totalRead, ts, err := fh.entryChunkGroup.ReadDataAt(ctx, fileSize, buff, offset)
|
|
|
|
if err != nil && err != io.EOF {
|
|
glog.Errorf("file handle read %s: %v", fileFullPath, err)
|
|
}
|
|
|
|
// glog.V(4).Infof("file handle read %s [%d,%d] %d : %v", fileFullPath, offset, offset+int64(totalRead), totalRead, err)
|
|
|
|
return int64(totalRead), ts, err
|
|
}
|
|
|
|
// tryRDMARead attempts to read file data using RDMA acceleration
|
|
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
|
|
// For now, we'll try to read the chunks directly using RDMA
|
|
// This is a simplified approach - in a full implementation, we'd need to
|
|
// handle chunk boundaries, multiple chunks, etc.
|
|
|
|
chunks := entry.GetEntry().Chunks
|
|
if len(chunks) == 0 {
|
|
return 0, 0, fmt.Errorf("no chunks available for RDMA read")
|
|
}
|
|
|
|
// Find the chunk that contains our offset using binary search
|
|
var targetChunk *filer_pb.FileChunk
|
|
var chunkOffset int64
|
|
|
|
// Get cached cumulative offsets for efficient binary search
|
|
cumulativeOffsets := fh.getCumulativeOffsets(chunks)
|
|
|
|
// Use binary search to find the chunk containing the offset
|
|
chunkIndex := sort.Search(len(chunks), func(i int) bool {
|
|
return offset < cumulativeOffsets[i+1]
|
|
})
|
|
|
|
// Verify the chunk actually contains our offset
|
|
if chunkIndex < len(chunks) && offset >= cumulativeOffsets[chunkIndex] {
|
|
targetChunk = chunks[chunkIndex]
|
|
chunkOffset = offset - cumulativeOffsets[chunkIndex]
|
|
}
|
|
|
|
if targetChunk == nil {
|
|
return 0, 0, fmt.Errorf("no chunk found for offset %d", offset)
|
|
}
|
|
|
|
// Calculate how much to read from this chunk
|
|
remainingInChunk := int64(targetChunk.Size) - chunkOffset
|
|
readSize := min(int64(len(buff)), remainingInChunk)
|
|
|
|
glog.V(4).Infof("RDMA read attempt: chunk=%s (fileId=%s), chunkOffset=%d, readSize=%d",
|
|
targetChunk.FileId, targetChunk.FileId, chunkOffset, readSize)
|
|
|
|
// Try RDMA read using file ID directly (more efficient)
|
|
data, isRDMA, err := fh.wfs.rdmaClient.ReadNeedle(ctx, targetChunk.FileId, uint64(chunkOffset), uint64(readSize))
|
|
if err != nil {
|
|
return 0, 0, fmt.Errorf("RDMA read failed: %w", err)
|
|
}
|
|
|
|
if !isRDMA {
|
|
return 0, 0, fmt.Errorf("RDMA not available for chunk")
|
|
}
|
|
|
|
// Copy data to buffer
|
|
copied := copy(buff, data)
|
|
return int64(copied), targetChunk.ModifiedTsNs, nil
|
|
}
|
|
|
|
func (fh *FileHandle) downloadRemoteEntry(entry *LockedEntry) error {
|
|
|
|
fileFullPath := fh.FullPath()
|
|
dir, _ := fileFullPath.DirAndName()
|
|
|
|
err := fh.wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
|
|
|
request := &filer_pb.CacheRemoteObjectToLocalClusterRequest{
|
|
Directory: string(dir),
|
|
Name: entry.Name,
|
|
}
|
|
|
|
glog.V(4).Infof("download entry: %v", request)
|
|
resp, err := client.CacheRemoteObjectToLocalCluster(context.Background(), request)
|
|
if err != nil {
|
|
return fmt.Errorf("CacheRemoteObjectToLocalCluster file %s: %v", fileFullPath, err)
|
|
}
|
|
|
|
fh.SetEntry(resp.Entry)
|
|
|
|
event := resp.GetMetadataEvent()
|
|
if event == nil {
|
|
event = metadataUpdateEvent(request.Directory, resp.Entry)
|
|
}
|
|
if applyErr := fh.wfs.applyLocalMetadataEvent(context.Background(), event); applyErr != nil {
|
|
glog.Warningf("CacheRemoteObject %s: best-effort metadata apply failed: %v", fileFullPath, applyErr)
|
|
}
|
|
|
|
return nil
|
|
})
|
|
|
|
return err
|
|
}
|