Files
seaweedfs/weed/filer/reader_cache_budget.go
T
Chris LuandDevin 6c07a5fdd0 s3: keep small ranged GETs on range reads, no whole-chunk downloads (#11577)
* filer: keep a ranged read in random mode through its contiguous tail

A far ReadAt on a fresh ReaderPattern left the sequential counter at -1,
so the next buffer of the same ranged request landed on the frontier and
flipped the verdict straight back to sequential — readChunkSliceAt then
paid a whole-chunk fetch for the remainder of the range. Drop the
counter to -ModeChangeLimit when random mode is entered so the verdict
needs sustained sequential evidence to undo, matching the hysteresis an
established sequential stream already gets.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* s3: pin small ranged GETs to range reads

A ranged GET whose first read lands within SeqTolerance of offset 0 is
judged sequential immediately, and even a far-starting range could flip
back mid-request; either way readChunkSliceAt downloads each covered
chunk in full, multiplying disk reads for small ranged reads (measured
~7x). Pin random mode for ranged requests no larger than SeqTolerance so
all of the request's buffer reads stay range fetches. Larger ranges keep
the dynamic pattern, where whole-chunk fetches amortize.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* filer: fetch only the part of a chunk the view covers

Replaces the PinRandomMode size heuristic with a per-chunk coverage rule.
ViewFromVisibleIntervals already clips chunk views to the request window,
so a view that is not IsFullChunk() is one the request only partially
needs; fetch it as a range regardless of the detected read pattern.

This closes the holes a request-size pin left open: ranges larger than
SeqTolerance no longer revert to whole-chunk downloads once their buffers
look sequential, and ranges that fully cover a chunk keep the shared
whole-chunk path instead of fetching 256KiB slices piecemeal. Prefetch
(MaybeCache) skips clipped views so it cannot amplify a range read either.

PinRandomMode is dropped: no caller needs it once coverage drives the
fetch choice. Range fetches route through fetchChunkDataFn so tests
observe them the same way as whole-chunk downloads.

* filer: keep ciphered chunks on the whole-chunk path

A range fetch cannot save bytes for a ciphered chunk: readEncryptedUrl
always downloads and decrypts the whole blob before slicing. Sending
partial views of ciphered chunks through fetchChunkRange would repeat the
full download per buffer, so they keep the shared whole-chunk path where
one download serves every buffer. Prefetch stays enabled for them for
the same reason.

* filer: keep compressed chunks on the whole-chunk path

Like ciphered chunks, a range request on a compressed chunk makes the
volume server read and decompress the whole needle, so range-per-buffer
would repeat the full backend read for each 256KiB window. Route them
through the shared whole-chunk path via ChunkView.CanRangeFetch.

* filer: fall back to range fetch when a chunk exceeds the reader budget

A ciphered or compressed chunk larger than readerCacheSizeMB can never
be read through the whole-chunk path — the budget rejects the buffer —
so its partial views must still range-fetch or the GET fails outright.

---------

Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
2026-10-03 15:01:04 +08:00

133 lines
3.3 KiB
Go

package filer
import (
"container/list"
"fmt"
"sync"
"sync/atomic"
"github.com/seaweedfs/seaweedfs/weed/util/mem"
)
const DefaultReaderCacheMemoryLimit = 256 << 20
type ReaderCacheBudget struct {
sync.Mutex
limit int64
used int64
reservations map[*SingleChunkCacher]int64
idle list.List
idleEntries map[*SingleChunkCacher]*list.Element
changed chan struct{}
}
func NewReaderCacheBudget(limit int64) *ReaderCacheBudget {
if limit <= 0 {
limit = DefaultReaderCacheMemoryLimit
}
return &ReaderCacheBudget{
limit: limit,
reservations: make(map[*SingleChunkCacher]int64),
idleEntries: make(map[*SingleChunkCacher]*list.Element),
changed: make(chan struct{}),
}
}
func (b *ReaderCacheBudget) reserve(s *SingleChunkCacher) error {
if s.chunkSize < 0 {
return fmt.Errorf("invalid chunk size %d", s.chunkSize)
}
if b == nil {
return nil
}
size := int64(mem.AllocationSize(s.chunkSize))
if size > b.limit {
return fmt.Errorf("chunk buffer needs %d bytes, exceeding reader cache budget %d; increase the readerCacheSizeMB budget", size, b.limit)
}
for {
b.Lock()
if size <= b.limit-b.used {
b.used += size
b.reservations[s] = size
b.Unlock()
return nil
}
// Prefer evicting an idle chunk no stream is positioned in; fall back
// to the oldest pinned one so abandoned pins cannot block the budget.
var victim *SingleChunkCacher
var entry *list.Element
pinnedVictim := false
for e := b.idle.Front(); e != nil; e = e.Next() {
c := e.Value.(*SingleChunkCacher)
if atomic.LoadInt32(&c.pins) == 0 {
victim, entry = c, e
pinnedVictim = false
break
}
if victim == nil {
victim, entry = c, e
pinnedVictim = true
}
}
if entry != nil {
b.idle.Remove(entry)
delete(b.idleEntries, victim)
b.Unlock()
if pinnedVictim {
victim.parent.remove(victim)
} else if !victim.parent.removeUnpinned(victim) {
// The victim was pinned between selection and removal: keep it
// evictable so a pin abandoned in that gap cannot wedge the
// budget, then retry the selection.
b.Lock()
if _, ok := b.reservations[victim]; ok && b.idleEntries[victim] == nil {
b.idleEntries[victim] = b.idle.PushBack(victim)
}
b.Unlock()
}
continue
}
changed := b.changed
b.Unlock()
<-changed
}
}
// canFit reports whether a whole-chunk buffer of this size can ever be
// reserved. A chunk bigger than the budget cannot be read through the
// whole-chunk path at all, so callers must fall back to range fetches.
func (b *ReaderCacheBudget) canFit(size int) bool {
return b == nil || int64(mem.AllocationSize(size)) <= b.limit
}
func (b *ReaderCacheBudget) complete(s *SingleChunkCacher) {
if b == nil {
return
}
b.Lock()
defer b.Unlock()
if _, found := b.reservations[s]; found && b.idleEntries[s] == nil {
b.idleEntries[s] = b.idle.PushBack(s)
close(b.changed)
b.changed = make(chan struct{})
}
}
func (b *ReaderCacheBudget) release(s *SingleChunkCacher) {
if b == nil {
return
}
b.Lock()
defer b.Unlock()
if size, found := b.reservations[s]; found {
b.used -= size
delete(b.reservations, s)
if entry := b.idleEntries[s]; entry != nil {
b.idle.Remove(entry)
delete(b.idleEntries, s)
}
close(b.changed)
b.changed = make(chan struct{})
}
}