mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-10 16:40:46 +02:00
* ec: pin auto-selected shard placement to the disk that already owns the shard A multi-disk server legitimately mounts one EC volume on several disks, so FindEcShardTargetLocation's per-volume tiers tie at "mounted" and the free-shard-count tie-break decides — pointing at whichever disk is emptier, not at the disk that already holds the shard being placed. A re-copy of a shard the server already has (a retried ec.balance / ec.rebuild move) then lands on a sibling disk, and both disks register the same (volume, shard id): the shard is reported to the master from two disk ids, and which claimant serves reads or survives a later unmount/delete becomes an accident of Locations order. Add a tier above "mounted": a disk that already claims one of the shard ids being placed wins, ahead of the space filters too — re-copying in place needs no new shard slot, and a genuinely full disk should fail the write rather than silently split the claim. Applied to the Go selector and the VolumeEcShardsCopy auto-select (ReceiveFile refuses mounted EC volumes, so no claim can exist there) and mirrored in the Rust volume server. Claude-Session: https://claude.ai/code/session_01AWpefvdi4U3HLng18x5CJ9 * ec: refuse a copy batch whose shards are already owned by different disks Review follow-up: ownership-aware selection ranks a mixed-owner batch (shard 0 on disk A, shard 2 on disk B — the legitimate multi-disk spread) into one destination, so the copy would still duplicate the losing disk's claim. No production caller sends such a batch (balance moves one shard, rebuild and encode copy shards the target lacks), so fail closed: report every owning disk via Store.EcShardOwnerDisks and refuse the copy with an error naming them, telling the caller to split per shard or pass disk_id. Go and Rust, with unit tests for the owner-reporting contract. Claude-Session: https://claude.ai/code/session_01AWpefvdi4U3HLng18x5CJ9
1062 lines
41 KiB
Go
1062 lines
41 KiB
Go
package storage
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"slices"
|
|
"sync"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"github.com/klauspost/reedsolomon"
|
|
"golang.org/x/sync/semaphore"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
|
)
|
|
|
|
// errShardNotLocal indicates that the requested EC shard is simply not
|
|
// stored on this volume server. It is expected during normal reads when
|
|
// shards are spread across multiple servers, so callers should not log
|
|
// it as an error.
|
|
var errShardNotLocal = errors.New("ec shard not on this server")
|
|
|
|
// FindEcShardTargetLocation returns the disk that should receive a new
|
|
// shard / index file for (collection, vid). The selection order is:
|
|
//
|
|
// 0. a disk that already owns one of shardIds (in-memory claim),
|
|
// 1. a disk that already has the EC volume mounted (in-memory state),
|
|
// 2. a disk that owns the .ecx file on disk (volume not mounted yet),
|
|
// 3. any HDD with free space,
|
|
// 4. any disk with free space.
|
|
//
|
|
// Step 0 keeps the per-server invariant that a shard id is owned by at
|
|
// most one disk. Steps 1-4 only know the volume, and a multi-disk server
|
|
// legitimately mounts the same vid on several disks, so the free-count
|
|
// tie-break alone can send a re-copy of a shard the server already holds
|
|
// (a retried ec.balance / ec.rebuild move) to a sibling disk. Both disks
|
|
// then claim the same (vid, shard) and report it to the master from two
|
|
// disk ids, and which claimant serves reads or survives a later
|
|
// unmount/delete of the shard id becomes an accident of Locations order.
|
|
// Overwriting in place is what the caller meant, so an owning disk wins
|
|
// ahead of the space filters too — a re-copy of a shard already on that
|
|
// disk needs no new shard slot, and a genuinely full disk fails the
|
|
// write with ENOSPC rather than silently splitting the claim.
|
|
//
|
|
// Step 2 is the missing primitive that pinned subsequent shards to the
|
|
// first-shard disk during ec.rebuild. ec.rebuild only sets CopyEcxFile=true
|
|
// for the first shard, then relies on auto-select to land later shards on
|
|
// the same disk. Without an on-disk check, FindEcVolume returns nothing
|
|
// (no mount yet) and the fallback picks "any HDD with free space" — which
|
|
// can split shards from their index files across disks of the same node
|
|
// and lose them at startup. See issue #9212 and the orphan-shard
|
|
// reconciliation in #9244.
|
|
//
|
|
// dataShardCount is the data-shard count for this volume's EC layout (10
|
|
// for the OSS default, but custom ratios are supported via .vif). Callers
|
|
// pass it explicitly so this helper stays free of package-level constants
|
|
// — easier to mirror into builds that ship a different default ratio.
|
|
//
|
|
// Implementation walks s.Locations once and scores each disk by tier; the
|
|
// highest-tier disk wins, ties broken by free count. The earlier waterfall
|
|
// across four FindFreeLocation passes was equivalent but acquired
|
|
// volumesLock and ecVolumesLock RLocks (via VolumesLen / EcShardCount) up
|
|
// to four times per disk per call.
|
|
func (s *Store) FindEcShardTargetLocation(collection string, vid needle.VolumeId, dataShardCount int, shardIds ...erasure_coding.ShardId) *DiskLocation {
|
|
const (
|
|
tierAnyDisk = iota + 1
|
|
tierHDD
|
|
tierEcxOnDisk
|
|
tierMounted
|
|
tierOwnsShard
|
|
)
|
|
|
|
var (
|
|
best *DiskLocation
|
|
bestTier int
|
|
bestOwned int
|
|
bestFree int32
|
|
)
|
|
for _, loc := range s.Locations {
|
|
owned := ownedEcShardCount(loc, vid, shardIds)
|
|
freeCount := ecFreeShardCount(loc, dataShardCount)
|
|
if owned == 0 {
|
|
if loc.isDiskSpaceLow.Load() {
|
|
continue
|
|
}
|
|
if freeCount <= 0 {
|
|
continue
|
|
}
|
|
}
|
|
tier := tierAnyDisk
|
|
if loc.DiskType == types.HardDriveType {
|
|
tier = tierHDD
|
|
}
|
|
if loc.HasEcxFileOnDisk(collection, vid) {
|
|
tier = tierEcxOnDisk
|
|
}
|
|
if _, mounted := loc.FindEcVolume(vid); mounted {
|
|
tier = tierMounted
|
|
}
|
|
if owned > 0 {
|
|
tier = tierOwnsShard
|
|
}
|
|
better := best == nil || tier > bestTier
|
|
if !better && tier == bestTier {
|
|
// owned only separates disks inside tierOwnsShard; it is 0
|
|
// everywhere else, so this falls through to the free-count
|
|
// tie-break for the other tiers.
|
|
better = owned > bestOwned || (owned == bestOwned && freeCount > bestFree)
|
|
}
|
|
if better {
|
|
best = loc
|
|
bestTier = tier
|
|
bestOwned = owned
|
|
bestFree = freeCount
|
|
}
|
|
}
|
|
return best
|
|
}
|
|
|
|
// EcShardOwnerDisks returns the distinct disks that already own one of
|
|
// shardIds for vid, in Locations order. More than one owner means the batch
|
|
// has no single correct destination — whichever disk receives it would
|
|
// duplicate a sibling disk's claim — so batch callers must split by owner
|
|
// (VolumeEcShardsCopy refuses such a batch instead of guessing).
|
|
func (s *Store) EcShardOwnerDisks(vid needle.VolumeId, shardIds []erasure_coding.ShardId) []*DiskLocation {
|
|
var owners []*DiskLocation
|
|
for _, loc := range s.Locations {
|
|
if ownedEcShardCount(loc, vid, shardIds) > 0 {
|
|
owners = append(owners, loc)
|
|
}
|
|
}
|
|
return owners
|
|
}
|
|
|
|
// ownedEcShardCount reports how many of shardIds this disk already claims
|
|
// for vid, per the in-memory registration the read path and heartbeats use.
|
|
func ownedEcShardCount(loc *DiskLocation, vid needle.VolumeId, shardIds []erasure_coding.ShardId) int {
|
|
if len(shardIds) == 0 {
|
|
return 0
|
|
}
|
|
loc.ecVolumesLock.RLock()
|
|
defer loc.ecVolumesLock.RUnlock()
|
|
ecVolume, found := loc.ecVolumes[vid]
|
|
if !found {
|
|
return 0
|
|
}
|
|
owned := 0
|
|
for _, shard := range ecVolume.Shards {
|
|
for _, shardId := range shardIds {
|
|
if shard.ShardId == shardId {
|
|
owned++
|
|
break
|
|
}
|
|
}
|
|
}
|
|
return owned
|
|
}
|
|
|
|
// ecFreeShardCount returns the free EC shard capacity of loc, expressed
|
|
// in shard slots (not volume-equivalent slots). dataShardCount is the
|
|
// data-shard count of the EC layout being placed — see
|
|
// FindEcShardTargetLocation's docstring for why it's a parameter.
|
|
//
|
|
// FindFreeLocation in store.go does the same math but divides by
|
|
// DataShardsCount at the end. That truncation can exclude a disk that
|
|
// still has room for several individual shards (e.g. MaxVolumeCount=1,
|
|
// EcShardCount=1, dataShardCount=10 → reports 0 despite 9 free shard
|
|
// slots), which in this helper would re-route subsequent shards off the
|
|
// .ecx-owning disk and re-introduce the orphan-shard layout #9212 is
|
|
// trying to prevent. So we keep the result in shard slots throughout.
|
|
//
|
|
// MaxVolumeCount == 0 is the "unlimited" sentinel used elsewhere in the
|
|
// store (see hasFreeDiskLocation). Reporting a synthetic large free
|
|
// count keeps unlimited disks eligible while still letting tie-breaks
|
|
// prefer the less-loaded one.
|
|
func ecFreeShardCount(loc *DiskLocation, dataShardCount int) int32 {
|
|
if dataShardCount <= 0 {
|
|
return 0
|
|
}
|
|
if loc.MaxVolumeCount <= 0 {
|
|
const unlimitedFree = int32(1 << 30)
|
|
used := int32(loc.VolumesLen())*int32(dataShardCount) + int32(loc.EcShardCount())
|
|
if used >= unlimitedFree {
|
|
return 1
|
|
}
|
|
return unlimitedFree - used
|
|
}
|
|
free := (loc.MaxVolumeCount - int32(loc.VolumesLen())) * int32(dataShardCount)
|
|
free -= int32(loc.EcShardCount())
|
|
if free < 0 {
|
|
return 0
|
|
}
|
|
return free
|
|
}
|
|
|
|
func (s *Store) CollectErasureCodingHeartbeat() *master_pb.Heartbeat {
|
|
var ecShardMessages []*master_pb.VolumeEcShardInformationMessage
|
|
collectionEcShardSize := make(map[string]int64)
|
|
for diskId, location := range s.Locations {
|
|
location.ecVolumesLock.RLock()
|
|
for _, ecShards := range location.ecVolumes {
|
|
ecShardMessages = append(ecShardMessages, ecShards.ToVolumeEcShardInformationMessage(uint32(diskId))...)
|
|
|
|
for _, ecShard := range ecShards.Shards {
|
|
collectionEcShardSize[ecShards.Collection] += ecShard.Size()
|
|
}
|
|
}
|
|
location.ecVolumesLock.RUnlock()
|
|
}
|
|
|
|
for col, size := range collectionEcShardSize {
|
|
stats.VolumeServerDiskSizeGauge.WithLabelValues(col, "ec").Set(float64(size))
|
|
}
|
|
|
|
for col := range s.reportedEcCollections {
|
|
if _, stillHere := collectionEcShardSize[col]; !stillHere {
|
|
stats.VolumeServerDiskSizeGauge.DeleteLabelValues(col, "ec")
|
|
}
|
|
}
|
|
s.reportedEcCollections = make(map[string]struct{}, len(collectionEcShardSize))
|
|
for col := range collectionEcShardSize {
|
|
s.reportedEcCollections[col] = struct{}{}
|
|
}
|
|
|
|
return &master_pb.Heartbeat{
|
|
EcShards: ecShardMessages,
|
|
HasNoEcShards: len(ecShardMessages) == 0,
|
|
}
|
|
|
|
}
|
|
|
|
func (s *Store) MountEcShards(collection string, vid needle.VolumeId, shardId erasure_coding.ShardId, sourceDiskType string) error {
|
|
// The .ecx index file may live on a different disk than the one
|
|
// holding the .ec?? shard being mounted: ec.balance / ec.rebuild can
|
|
// place the .ecx on one local disk while later distributing shards
|
|
// across sibling disks of the same volume server. The per-disk
|
|
// IdxDirectory used by LoadEcShard would ENOENT the .ecx, so look up
|
|
// the .ecx owner across all DiskLocations once and route NewEcVolume
|
|
// at the directory that actually has the file. A 0-byte .ecx is
|
|
// treated as missing here (writeToFile can leave a stub on a failed
|
|
// EC distribute) so we still scan the rest of the disks.
|
|
ecxIdxDir, ecxFound := s.findEcxIdxDirForVolume(collection, vid)
|
|
|
|
// Collect failures so an all-disks-fail return reports every disk we
|
|
// tried rather than just the first one. Before this loop reordered
|
|
// itself to keep going after the first non-ENOENT error, a single
|
|
// shard-on-disk-without-.ecx situation would bail the loop and the
|
|
// operator saw "cannot open ec volume index" naming exactly one disk
|
|
// even when others held a valid index.
|
|
type diskError struct {
|
|
dir string
|
|
err error
|
|
}
|
|
var failures []diskError
|
|
|
|
for diskId, location := range s.Locations {
|
|
idxDir := location.IdxDirectory
|
|
if ecxFound {
|
|
// Fast path: if findEcxIdxDirForVolume already pointed at
|
|
// one of this disk's directories, the disk owns the .ecx
|
|
// and the local IdxDirectory is the right answer — skip
|
|
// the HasEcxFileOnDisk stat. Only fall back to the sibling
|
|
// disk's idxDir when this disk's directories are neither.
|
|
if location.IdxDirectory != ecxIdxDir && location.Directory != ecxIdxDir {
|
|
if !location.HasEcxFileOnDisk(collection, vid) {
|
|
idxDir = ecxIdxDir
|
|
}
|
|
}
|
|
}
|
|
ecVolume, err := location.loadEcShardWithIdxDir(collection, vid, shardId, idxDir)
|
|
if err == nil {
|
|
glog.V(0).Infof("MountEcShards %d.%d on disk ID %d", vid, shardId, diskId)
|
|
|
|
// Apply the orchestrator-supplied source disk type so the EC
|
|
// volume reports under it instead of the location's. Empty means
|
|
// "fall back to location's disk type" (#9423).
|
|
if sourceDiskType != "" {
|
|
ecVolume.SetDiskType(types.ToDiskType(sourceDiskType))
|
|
}
|
|
|
|
si := erasure_coding.NewShardsInfo()
|
|
si.Set(erasure_coding.NewShardInfo(shardId, erasure_coding.ShardSize(ecVolume.ShardSize())))
|
|
s.NewEcShardsChan <- &master_pb.VolumeEcShardInformationMessage{
|
|
Id: uint32(vid),
|
|
Collection: collection,
|
|
EcIndexBits: uint32(si.Bitmap()),
|
|
ShardSizes: si.SizesInt64(),
|
|
DiskType: string(ecVolume.DiskType()),
|
|
ExpireAtSec: ecVolume.ExpireAtSec,
|
|
DiskId: uint32(diskId),
|
|
EncodeTsNs: ecVolume.EncodeTsNs,
|
|
}
|
|
return nil
|
|
}
|
|
if errors.Is(err, os.ErrNotExist) {
|
|
// Shard or index not on this disk; another disk may own it.
|
|
continue
|
|
}
|
|
failures = append(failures, diskError{dir: location.Directory, err: err})
|
|
}
|
|
|
|
if len(failures) == 0 {
|
|
// No disk had the shard or the index; this volume server is not
|
|
// holding any artefacts for the requested shard. Name what we
|
|
// scanned for so the operator can tell "no .ecx anywhere" apart
|
|
// from "shard not on this server".
|
|
if !ecxFound {
|
|
return fmt.Errorf("MountEcShards %d.%d: no .ecx index found on any local disk", vid, shardId)
|
|
}
|
|
return fmt.Errorf("MountEcShards %d.%d not found on disk", vid, shardId)
|
|
}
|
|
// Some disks returned a real (non-ENOENT) error. Report them all so
|
|
// the caller can see whether the failures cluster around one disk
|
|
// (likely hardware) or are spread out (likely a config problem).
|
|
var b []byte
|
|
for i, f := range failures {
|
|
if i > 0 {
|
|
b = append(b, "; "...)
|
|
}
|
|
b = append(b, fmt.Sprintf("%s: %v", f.dir, f.err)...)
|
|
}
|
|
return fmt.Errorf("MountEcShards %d.%d load failures: %s", vid, shardId, string(b))
|
|
}
|
|
|
|
func (s *Store) UnmountEcShards(vid needle.VolumeId, shardId erasure_coding.ShardId, reqEncodeTsNs int64) error {
|
|
// Walk every disk: a split-disk reconciled volume can mount the same vid on
|
|
// more than one disk, so a first-match unmount would leave a sibling copy
|
|
// mounted and heartbeating. Emit one deletion delta per disk.
|
|
unmountedAny := false
|
|
var lastErr error
|
|
for diskId, location := range s.Locations {
|
|
ecShard, found := location.FindEcShard(vid, shardId)
|
|
if !found {
|
|
continue
|
|
}
|
|
// Capture the encode generation before unloading so the deletion delta
|
|
// carries it like the mount delta does.
|
|
var encodeTsNs int64
|
|
if ecVolume, ok := location.FindEcVolume(vid); ok {
|
|
encodeTsNs = ecVolume.EncodeTsNs
|
|
}
|
|
// Generation fence: when the caller carries a generation (stale-worker
|
|
// cleanup), only unmount a strictly-older generation; preserve a disk whose
|
|
// generation is same-or-newer, 0, or unknown, so a stale run cannot unmount
|
|
// a newer run's live shards. reqEncodeTsNs==0 (legacy/shell) unmounts all.
|
|
if reqEncodeTsNs > 0 && !(encodeTsNs > 0 && encodeTsNs < reqEncodeTsNs) {
|
|
glog.V(1).Infof("UnmountEcShards %d.%d disk_id:%d skipped: disk gen %d not older than request gen %d", vid, shardId, diskId, encodeTsNs, reqEncodeTsNs)
|
|
continue
|
|
}
|
|
if deleted := location.UnloadEcShard(vid, shardId); deleted {
|
|
si := erasure_coding.NewShardsInfo()
|
|
si.Set(erasure_coding.NewShardInfo(shardId, 0))
|
|
s.DeletedEcShardsChan <- &master_pb.VolumeEcShardInformationMessage{
|
|
Id: uint32(vid),
|
|
Collection: ecShard.Collection,
|
|
EcIndexBits: si.Bitmap(),
|
|
ShardSizes: si.SizesInt64(),
|
|
DiskType: string(ecShard.DiskType),
|
|
DiskId: uint32(diskId),
|
|
EncodeTsNs: encodeTsNs,
|
|
}
|
|
glog.V(0).Infof("UnmountEcShards %d.%d disk_id:%d", vid, shardId, diskId)
|
|
unmountedAny = true
|
|
} else {
|
|
lastErr = fmt.Errorf("UnmountEcShards %d.%d not found on disk %d", vid, shardId, diskId)
|
|
}
|
|
}
|
|
|
|
// nil when no disk held the shard (idempotent re-unmount).
|
|
if !unmountedAny {
|
|
return lastErr
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func (s *Store) findEcShard(vid needle.VolumeId, shardId erasure_coding.ShardId) (diskId uint32, shard *erasure_coding.EcVolumeShard, found bool) {
|
|
for diskId, location := range s.Locations {
|
|
if v, found := location.FindEcShard(vid, shardId); found {
|
|
return uint32(diskId), v, found
|
|
}
|
|
}
|
|
return 0, nil, false
|
|
}
|
|
|
|
// FindEcShard returns the shard if any DiskLocation on this server holds it,
|
|
// along with that disk's id.
|
|
func (s *Store) FindEcShard(vid needle.VolumeId, shardId erasure_coding.ShardId) (diskId uint32, shard *erasure_coding.EcVolumeShard, found bool) {
|
|
return s.findEcShard(vid, shardId)
|
|
}
|
|
|
|
// FindEcVolumeWithShard returns the EcVolume on the disk that owns the given
|
|
// shard, plus the shard. The read guard must check the identity of the volume
|
|
// that owns the bytes served: on a multi-disk server one vid can hold shards
|
|
// from different encode runs across disks, so a first-match volume can differ.
|
|
func (s *Store) FindEcVolumeWithShard(vid needle.VolumeId, shardId erasure_coding.ShardId) (*erasure_coding.EcVolume, *erasure_coding.EcVolumeShard, bool) {
|
|
for _, location := range s.Locations {
|
|
if shard, found := location.FindEcShard(vid, shardId); found {
|
|
if ev, ok := location.FindEcVolume(vid); ok {
|
|
return ev, shard, true
|
|
}
|
|
}
|
|
}
|
|
return nil, nil, false
|
|
}
|
|
|
|
// EcMetadataDirs lists every directory on this server that could hold an EC
|
|
// volume's metadata — each disk's data and index directory. Startup mirroring
|
|
// gives each shard-bearing disk its own .ecx/.ecj/.vif, but the checksum
|
|
// sidecar is not mirrored and a repair delivers exactly one copy, so a runtime
|
|
// looking only at its own two directories cannot see it. Handing this list to
|
|
// the sidecar resolution keeps one authoritative copy reachable from every
|
|
// runtime rather than duplicating a file that is rewritten as shards are
|
|
// repaired and generations published.
|
|
func (s *Store) EcMetadataDirs() []string {
|
|
var dirs []string
|
|
appendDir := func(dir string) {
|
|
if dir == "" {
|
|
return
|
|
}
|
|
for _, existing := range dirs {
|
|
if existing == dir {
|
|
return
|
|
}
|
|
}
|
|
dirs = append(dirs, dir)
|
|
}
|
|
for _, location := range s.Locations {
|
|
appendDir(location.Directory)
|
|
appendDir(location.IdxDirectory)
|
|
}
|
|
return dirs
|
|
}
|
|
|
|
// FindAllEcVolumes returns every per-disk *EcVolume the store maps for vid. A vid
|
|
// can mount on N disks as N distinct runtimes, and the first-match FindEcVolume
|
|
// hides the siblings — so anything that has to reach the whole volume, rather
|
|
// than any one runtime of it, iterates this instead. Order mirrors the
|
|
// deterministic s.Locations order; nil when no disk holds the vid.
|
|
func (s *Store) FindAllEcVolumes(vid needle.VolumeId) []*erasure_coding.EcVolume {
|
|
var evs []*erasure_coding.EcVolume
|
|
for _, location := range s.Locations {
|
|
if ev, found := location.FindEcVolume(vid); found {
|
|
evs = append(evs, ev)
|
|
}
|
|
}
|
|
return evs
|
|
}
|
|
|
|
func (s *Store) FindEcVolume(vid needle.VolumeId) (*erasure_coding.EcVolume, bool) {
|
|
for _, location := range s.Locations {
|
|
if s, found := location.FindEcVolume(vid); found {
|
|
return s, true
|
|
}
|
|
}
|
|
return nil, false
|
|
}
|
|
|
|
// FindEcVolumeDiskIds returns every disk_id on this store that has an
|
|
// EcVolume entry for the given volume. Useful for diagnostic logging
|
|
// when a single FindEcVolume hit hides which disk is actually holding
|
|
// the mount (e.g., the ReceiveFile mounted-volume guard).
|
|
func (s *Store) FindEcVolumeDiskIds(vid needle.VolumeId) []uint32 {
|
|
var ids []uint32
|
|
for diskId, location := range s.Locations {
|
|
if _, found := location.FindEcVolume(vid); found {
|
|
ids = append(ids, uint32(diskId))
|
|
}
|
|
}
|
|
return ids
|
|
}
|
|
|
|
// shardFiles is a list of shard files, which is used to return the shard locations
|
|
func (s *Store) CollectEcShards(vid needle.VolumeId, shardFileNames []string) (ecVolume *erasure_coding.EcVolume, found bool) {
|
|
for _, location := range s.Locations {
|
|
if s, foundShards := location.CollectEcShards(vid, shardFileNames); foundShards {
|
|
ecVolume = s
|
|
found = true
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
func (s *Store) DestroyEcVolume(vid needle.VolumeId) {
|
|
for _, location := range s.Locations {
|
|
location.DestroyEcVolume(vid)
|
|
}
|
|
}
|
|
|
|
// UnloadEcVolume drops any in-memory EcVolume for vid from every disk and closes
|
|
// its fds without deleting files, so a following unlink frees the inodes.
|
|
func (s *Store) UnloadEcVolume(vid needle.VolumeId) {
|
|
for _, location := range s.Locations {
|
|
location.unloadEcVolume(vid)
|
|
}
|
|
}
|
|
|
|
func (s *Store) ReadEcShardNeedle(vid needle.VolumeId, n *needle.Needle, onReadSizeFn func(size types.Size)) (int, error) {
|
|
for _, location := range s.Locations {
|
|
if localEcVolume, found := location.FindEcVolume(vid); found {
|
|
|
|
offset, size, intervals, err := localEcVolume.LocateEcShardNeedle(n.Id, localEcVolume.Version)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("locate in local ec volume: %w", err)
|
|
}
|
|
if size.IsDeleted() {
|
|
return 0, ErrorDeleted
|
|
}
|
|
|
|
if onReadSizeFn != nil {
|
|
onReadSizeFn(size)
|
|
}
|
|
|
|
glog.V(3).Infof("read ec volume %d offset %d size %d intervals:%+v", vid, offset.ToActualOffset(), size, intervals)
|
|
|
|
if len(intervals) > 1 {
|
|
glog.V(3).Infof("ReadEcShardNeedle needle id %s intervals:%+v", n.String(), intervals)
|
|
}
|
|
bytes, isDeleted, err := s.readEcShardIntervals(n.Id, localEcVolume, intervals)
|
|
// A holder reporting the needle deleted is authoritative -- deletes are
|
|
// never invented and never undone -- so answer that ahead of whatever
|
|
// error the shards it could not gather produced.
|
|
if isDeleted {
|
|
return 0, ErrorDeleted
|
|
}
|
|
if err != nil {
|
|
return 0, fmt.Errorf("ReadEcShardIntervals: %w", err)
|
|
}
|
|
|
|
err = n.ReadBytes(bytes, offset.ToActualOffset(), size, localEcVolume.Version)
|
|
if err != nil {
|
|
return 0, fmt.Errorf("ec volume %d needle %s offset %d size %d: %w", vid, n.String(), offset.ToActualOffset(), size, err)
|
|
}
|
|
|
|
return len(bytes), nil
|
|
}
|
|
}
|
|
return 0, fmt.Errorf("ec shard %d not found", vid)
|
|
}
|
|
|
|
var (
|
|
// ecRecoverBudget bounds the interval-sized buffers EC recovery holds across
|
|
// all concurrent reads: a peer that is slow to fail keeps a whole fan-out of
|
|
// them alive for the gRPC timeout, and a burst of those walked servers into an
|
|
// OOM. A var so a test can narrow it alongside ecRecoverSem.
|
|
ecRecoverBudget int64 = 256 << 20
|
|
ecRecoverSem = semaphore.NewWeighted(ecRecoverBudget)
|
|
)
|
|
|
|
// ecIntervalReadConcurrency bounds the fan-out of a single needle read. Blocks
|
|
// that follow each other in the .dat live on different shards, so a needle
|
|
// spanning several of them costs one round trip per block when read in sequence.
|
|
const ecIntervalReadConcurrency = 8
|
|
|
|
func (s *Store) readEcShardIntervals(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, intervals []erasure_coding.Interval) (data []byte, is_deleted bool, err error) {
|
|
if err = s.cachedLookupEcShardLocations(ecVolume); err != nil {
|
|
return nil, false, fmt.Errorf("failed to locate shard via master grpc %s: %v", s.MasterAddress, err)
|
|
}
|
|
|
|
var totalSize int
|
|
for _, interval := range intervals {
|
|
totalSize += int(interval.Size)
|
|
}
|
|
data = make([]byte, totalSize)
|
|
|
|
if len(intervals) <= 1 {
|
|
for _, interval := range intervals {
|
|
if is_deleted, err = s.readOneEcShardInterval(needleId, ecVolume, interval, data); err != nil {
|
|
return nil, is_deleted, err
|
|
}
|
|
}
|
|
return data, is_deleted, nil
|
|
}
|
|
|
|
errs := make([]error, len(intervals))
|
|
var deleted atomic.Bool
|
|
var wg sync.WaitGroup
|
|
sem := make(chan struct{}, ecIntervalReadConcurrency)
|
|
var pos int
|
|
for i, interval := range intervals {
|
|
buf := data[pos : pos+int(interval.Size)]
|
|
pos += int(interval.Size)
|
|
wg.Add(1)
|
|
sem <- struct{}{}
|
|
go func() {
|
|
defer wg.Done()
|
|
defer func() { <-sem }()
|
|
isDeleted, e := s.readOneEcShardInterval(needleId, ecVolume, interval, buf)
|
|
if isDeleted {
|
|
deleted.Store(true)
|
|
}
|
|
errs[i] = e
|
|
}()
|
|
}
|
|
wg.Wait()
|
|
|
|
is_deleted = deleted.Load()
|
|
for _, e := range errs {
|
|
if e != nil {
|
|
return nil, is_deleted, e
|
|
}
|
|
}
|
|
return data, is_deleted, nil
|
|
}
|
|
|
|
// readOneEcShardInterval fills data, which must be interval.Size long.
|
|
func (s *Store) readOneEcShardInterval(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, interval erasure_coding.Interval, data []byte) (is_deleted bool, err error) {
|
|
shardId, actualOffset := ecVolume.IntervalToShardIdAndOffset(interval)
|
|
|
|
// try local read
|
|
err = s.readLocalEcShardInterval(ecVolume, shardId, data, actualOffset)
|
|
if err == nil {
|
|
return
|
|
}
|
|
if errors.Is(err, errShardNotLocal) {
|
|
// expected when shards are spread across servers; fall through to remote read
|
|
glog.V(4).Infof("ec shard %d.%d not local, will try remote", ecVolume.VolumeId, shardId)
|
|
} else {
|
|
glog.V(0).Infof("read local ec shard %d.%d offset %d: %v", ecVolume.VolumeId, shardId, actualOffset, err)
|
|
}
|
|
|
|
ecVolume.ShardLocationsLock.RLock()
|
|
sourceDataNodes, hasShardIdLocation := ecVolume.ShardLocations[shardId]
|
|
ecVolume.ShardLocationsLock.RUnlock()
|
|
|
|
// try reading directly
|
|
if hasShardIdLocation {
|
|
_, is_deleted, err = s.readRemoteEcShardInterval(sourceDataNodes, needleId, ecVolume.VolumeId, shardId, data, actualOffset, ecVolume.EncodeTsNs)
|
|
if err == nil {
|
|
return
|
|
}
|
|
glog.V(0).Infof("read remote ec shard %d.%d locations: %v", ecVolume.VolumeId, shardId, err)
|
|
// Recovery below skips this very shard, so nothing else invalidates the
|
|
// location that just failed -- and a shard that has moved would otherwise
|
|
// be reconstructed on every read until the map's own window expires.
|
|
markShardLocationsStale(ecVolume)
|
|
}
|
|
|
|
// try reading by recovering from other shards
|
|
_, is_deleted, err = s.recoverOneRemoteEcShardInterval(needleId, ecVolume, shardId, data, actualOffset)
|
|
if err == nil {
|
|
return
|
|
}
|
|
glog.V(0).Infof("recover ec shard %d.%d : %v", ecVolume.VolumeId, shardId, err)
|
|
|
|
return
|
|
}
|
|
|
|
func forgetShardId(ecVolume *erasure_coding.EcVolume, shardId erasure_coding.ShardId) {
|
|
// failed to access the source data nodes, clear it up
|
|
ecVolume.ShardLocationsLock.Lock()
|
|
delete(ecVolume.ShardLocations, shardId)
|
|
ecVolume.ShardLocationsStale = true
|
|
ecVolume.ShardLocationsLock.Unlock()
|
|
}
|
|
|
|
// markShardLocationsStale flags the cached map for a prompt re-check after a
|
|
// read failed against one of its locations. Unlike forgetShardId it keeps the
|
|
// entry: a direct read is worth retrying, since a dead peer fails fast.
|
|
func markShardLocationsStale(ecVolume *erasure_coding.EcVolume) {
|
|
ecVolume.ShardLocationsLock.Lock()
|
|
ecVolume.ShardLocationsStale = true
|
|
ecVolume.ShardLocationsLock.Unlock()
|
|
}
|
|
|
|
// ecShardLocationsTTL is how long a cached shard map is trusted. A complete map
|
|
// is trusted longest. One short of DataShards, or one a failed read has just
|
|
// invalidated, is re-checked promptly: until it is, every read of the dropped
|
|
// shard skips the direct fetch and pays for a Reed-Solomon recovery instead.
|
|
func ecShardLocationsTTL(shardCount int, stale bool, ecCtx *erasure_coding.ECContext) time.Duration {
|
|
switch {
|
|
case stale || shardCount < ecCtx.DataShards:
|
|
return 11 * time.Second
|
|
case shardCount == ecCtx.Total():
|
|
return 37 * time.Minute
|
|
default:
|
|
return 7 * time.Minute
|
|
}
|
|
}
|
|
|
|
func (s *Store) cachedLookupEcShardLocations(ecVolume *erasure_coding.EcVolume) (err error) {
|
|
|
|
// Use the volume's own EC ratio so a custom-ratio volume (e.g. 9+3) is judged
|
|
// complete/recoverable against its real data-shard count, not the build default.
|
|
// In OSS the ratio is always 10+4, so this is a no-op.
|
|
ecCtx := ecVolume.ECContext
|
|
if ecCtx == nil {
|
|
ecCtx = erasure_coding.NewDefaultECContext(ecVolume.Collection, ecVolume.VolumeId)
|
|
}
|
|
|
|
// Judge the map and consume its mark in one critical section, so a mark raised
|
|
// from here on belongs to the next refresh rather than being cleared by this
|
|
// one -- the read that raised it has disproved the map this lookup is about to
|
|
// install. Recover goroutines mutate all three fields via forgetShardId, so an
|
|
// unguarded read would race a concurrent map write besides.
|
|
ecVolume.ShardLocationsLock.Lock()
|
|
stale := ecVolume.ShardLocationsStale
|
|
ttl := ecShardLocationsTTL(len(ecVolume.ShardLocations), stale, ecCtx)
|
|
fresh := ecVolume.ShardLocationsRefreshTime.Add(ttl).After(time.Now())
|
|
if !fresh {
|
|
ecVolume.ShardLocationsStale = false
|
|
}
|
|
ecVolume.ShardLocationsLock.Unlock()
|
|
if fresh {
|
|
return nil
|
|
}
|
|
|
|
glog.V(3).Infof("lookup and cache ec volume %d locations", ecVolume.VolumeId)
|
|
|
|
err = operation.WithMasterServerClient(context.Background(), false, s.MasterAddress, s.grpcDialOption, func(masterClient master_pb.SeaweedClient) error {
|
|
req := &master_pb.LookupEcVolumeRequest{
|
|
VolumeId: uint32(ecVolume.VolumeId),
|
|
}
|
|
resp, err := masterClient.LookupEcVolume(context.Background(), req)
|
|
if err != nil {
|
|
return fmt.Errorf("lookup ec volume %d: %v", ecVolume.VolumeId, err)
|
|
}
|
|
if len(resp.ShardIdLocations) < ecCtx.DataShards {
|
|
return fmt.Errorf("only %d shards found but %d required", len(resp.ShardIdLocations), ecCtx.DataShards)
|
|
}
|
|
|
|
ecVolume.ShardLocationsLock.Lock()
|
|
for _, shardIdLocations := range resp.ShardIdLocations {
|
|
shardId := erasure_coding.ShardId(shardIdLocations.ShardId)
|
|
delete(ecVolume.ShardLocations, shardId)
|
|
for _, loc := range shardIdLocations.Locations {
|
|
ecVolume.ShardLocations[shardId] = append(ecVolume.ShardLocations[shardId], pb.NewServerAddressFromLocation(loc))
|
|
}
|
|
}
|
|
ecVolume.ShardLocationsRefreshTime = time.Now()
|
|
ecVolume.ShardLocationsLock.Unlock()
|
|
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
// The lookup this mark was consumed for did not answer, and the refresh time
|
|
// is only advanced on success -- so without putting the mark back, a map a
|
|
// read had already disproved would be trusted for its full window again.
|
|
// Marking unconditionally is safe: where no mark was consumed the refresh
|
|
// time is stale anyway, and the next call looks up whatever the mark says.
|
|
markShardLocationsStale(ecVolume)
|
|
}
|
|
return
|
|
}
|
|
|
|
func (s *Store) readLocalEcShardInterval(ecVolume *erasure_coding.EcVolume, shardId erasure_coding.ShardId, buf []byte, offset int64) error {
|
|
// Resolve the shard together with the EcVolume on the disk that owns it; the
|
|
// shard may live on a sibling disk of this server.
|
|
ownerVolume, shard, found := s.FindEcVolumeWithShard(ecVolume.VolumeId, shardId)
|
|
if !found {
|
|
return fmt.Errorf("shard %d for volume %d: %w", shardId, ecVolume.VolumeId, errShardNotLocal)
|
|
}
|
|
// Skip a local shard whose identity doesn't match the caller's index, so the
|
|
// read recovers from the correct generation. Lenient only when the caller has
|
|
// no identity (pre-upgrade): a known caller must not accept an unstamped local
|
|
// shard, which would serve a stale pre-upgrade generation.
|
|
if ecVolume.EncodeTsNs != 0 && ecVolume.EncodeTsNs != ownerVolume.EncodeTsNs {
|
|
glog.V(1).Infof("skip local ec shard %d.%d from a different encode run: caller EncodeTsNs %d, local %d", ecVolume.VolumeId, shardId, ecVolume.EncodeTsNs, ownerVolume.EncodeTsNs)
|
|
return fmt.Errorf("shard %d for volume %d: %w", shardId, ecVolume.VolumeId, errShardNotLocal)
|
|
}
|
|
|
|
readBytes, err := shard.ReadAt(buf, offset)
|
|
if err != nil {
|
|
return fmt.Errorf("failed to read local EC shard %d for volume %d: %v", shardId, ecVolume.VolumeId, err)
|
|
}
|
|
if got, want := readBytes, len(buf); got != want {
|
|
return fmt.Errorf("expected %d bytes for local EC shard %d on volume %d, got %d", want, shardId, ecVolume.VolumeId, got)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (s *Store) readRemoteEcShardInterval(sourceDataNodes []pb.ServerAddress, needleId types.NeedleId, vid needle.VolumeId, shardId erasure_coding.ShardId, buf []byte, offset int64, expectedEncodeTsNs int64) (n int, is_deleted bool, err error) {
|
|
|
|
if len(sourceDataNodes) == 0 {
|
|
return 0, false, fmt.Errorf("failed to find ec shard %d.%d", vid, shardId)
|
|
}
|
|
|
|
for _, sourceDataNode := range sourceDataNodes {
|
|
glog.V(3).Infof("read remote ec shard %d.%d from %s", vid, shardId, sourceDataNode)
|
|
n, is_deleted, err = s.doReadRemoteEcShardInterval(sourceDataNode, needleId, vid, shardId, buf, offset, expectedEncodeTsNs)
|
|
if err == nil {
|
|
return
|
|
}
|
|
glog.V(1).Infof("read remote ec shard %d.%d from %s: %v", vid, shardId, sourceDataNode, err)
|
|
}
|
|
|
|
return
|
|
}
|
|
|
|
func (s *Store) doReadRemoteEcShardInterval(sourceDataNode pb.ServerAddress, needleId types.NeedleId, vid needle.VolumeId, shardId erasure_coding.ShardId, buf []byte, offset int64, expectedEncodeTsNs int64) (n int, is_deleted bool, err error) {
|
|
|
|
err = operation.WithVolumeServerClient(false, sourceDataNode, s.grpcDialOption, func(client volume_server_pb.VolumeServerClient) error {
|
|
|
|
// copy data slice
|
|
shardReadClient, err := client.VolumeEcShardRead(context.Background(), &volume_server_pb.VolumeEcShardReadRequest{
|
|
VolumeId: uint32(vid),
|
|
ShardId: uint32(shardId),
|
|
Offset: offset,
|
|
Size: int64(len(buf)),
|
|
FileKey: uint64(needleId),
|
|
EncodeTsNs: expectedEncodeTsNs,
|
|
})
|
|
if err != nil {
|
|
return fmt.Errorf("failed to start reading ec shard %d.%d from %s: %v", vid, shardId, sourceDataNode, err)
|
|
}
|
|
|
|
for {
|
|
resp, receiveErr := shardReadClient.Recv()
|
|
if receiveErr == io.EOF {
|
|
break
|
|
}
|
|
if receiveErr != nil {
|
|
return fmt.Errorf("receiving ec shard %d.%d from %s: %v", vid, shardId, sourceDataNode, receiveErr)
|
|
}
|
|
// Validate the served shard's identity client-side, so the guard holds
|
|
// even against a pre-upgrade server that ignored the request field (it
|
|
// returns 0). A mismatch fails the read; the caller recovers from parity.
|
|
if expectedEncodeTsNs != 0 && resp.EncodeTsNs != expectedEncodeTsNs {
|
|
return fmt.Errorf("ec shard %d.%d from %s belongs to a different encode run (want %d, got %d)", vid, shardId, sourceDataNode, expectedEncodeTsNs, resp.EncodeTsNs)
|
|
}
|
|
if resp.IsDeleted {
|
|
is_deleted = true
|
|
}
|
|
copy(buf[n:n+len(resp.Data)], resp.Data)
|
|
n += len(resp.Data)
|
|
}
|
|
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
return 0, is_deleted, fmt.Errorf("read ec shard %d.%d from %s: %v", vid, shardId, sourceDataNode, err)
|
|
}
|
|
|
|
// A non-deleted interval must arrive whole: the server stamps EncodeTsNs only
|
|
// on chunks that carry bytes, so a short or empty stream (e.g. immediate EOF
|
|
// from a pre-upgrade or stale server) leaves the buffer partly zero-filled and
|
|
// unvalidated. Reject it so the caller recovers from parity. The is_deleted
|
|
// short-circuit legitimately returns n=0 with no data and is exempt, matching
|
|
// readLocalEcShardInterval's got==len(buf) rule for the local path.
|
|
if !is_deleted && n != len(buf) {
|
|
return n, is_deleted, fmt.Errorf("short read ec shard %d.%d from %s: got %d want %d", vid, shardId, sourceDataNode, n, len(buf))
|
|
}
|
|
|
|
return
|
|
}
|
|
|
|
// gatherEcShardIntervals collects the same interval from every shard but shardIdToRecover,
|
|
// returning one buffer per shard and nil where the shard could not be gathered. It stops
|
|
// once DataShards of them are in hand: reconstruction consumes no more than that.
|
|
func (s *Store) gatherEcShardIntervals(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, ecCtx *erasure_coding.ECContext, shardIdToRecover erasure_coding.ShardId, size int, offset int64) (shardIntervals [][]byte, isDeleted bool) {
|
|
shardIntervals = make([][]byte, ecCtx.Total())
|
|
|
|
// A shard this server already holds costs no round trip and no peer buffer,
|
|
// so seed those before asking peers for the rest.
|
|
available := 0
|
|
for shardId := erasure_coding.ShardId(0); int(shardId) < ecCtx.Total() && available < ecCtx.DataShards; shardId++ {
|
|
if shardId == shardIdToRecover {
|
|
continue
|
|
}
|
|
if _, _, found := s.FindEcVolumeWithShard(ecVolume.VolumeId, shardId); !found {
|
|
continue
|
|
}
|
|
data := make([]byte, size)
|
|
if localErr := s.readLocalEcShardInterval(ecVolume, shardId, data, offset); localErr != nil {
|
|
glog.V(3).Infof("recover: read local ec shard %d.%d: %v", ecVolume.VolumeId, shardId, localErr)
|
|
continue
|
|
}
|
|
shardIntervals[shardId] = data
|
|
available++
|
|
}
|
|
|
|
var candidates []erasure_coding.ShardId
|
|
candidateLocations := make(map[erasure_coding.ShardId][]pb.ServerAddress)
|
|
ecVolume.ShardLocationsLock.RLock()
|
|
for shardId, locations := range ecVolume.ShardLocations {
|
|
|
|
// skip the shard being recovered, one already seeded locally, or an empty shard
|
|
if shardId == shardIdToRecover || int(shardId) >= ecCtx.Total() || shardIntervals[shardId] != nil {
|
|
continue
|
|
}
|
|
if len(locations) == 0 {
|
|
glog.V(3).Infof("readRemoteEcShardInterval missing %d.%d from %+v", ecVolume.VolumeId, shardId, locations)
|
|
continue
|
|
}
|
|
candidates = append(candidates, shardId)
|
|
candidateLocations[shardId] = locations
|
|
}
|
|
ecVolume.ShardLocationsLock.RUnlock()
|
|
|
|
// The recover goroutines run concurrently, so the deleted flag is collected
|
|
// atomically and folded into the named return after they join, rather than each
|
|
// goroutine writing the shared bool directly.
|
|
var isDeletedFlag atomic.Bool
|
|
|
|
// Reconstruction consumes DataShards buffers, so reading every remaining shard
|
|
// holds a third more than that and asks a third more of peers that are, by
|
|
// then, already struggling. Widen only when a wave falls short.
|
|
for len(candidates) > 0 && available < ecCtx.DataShards {
|
|
wave := min(ecCtx.DataShards-available, len(candidates))
|
|
var wg sync.WaitGroup
|
|
var fetched atomic.Int64
|
|
for _, shardId := range candidates[:wave] {
|
|
locations := candidateLocations[shardId]
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
data := make([]byte, size)
|
|
nRead, isDeleted, readErr := s.readRemoteEcShardInterval(locations, needleId, ecVolume.VolumeId, shardId, data, offset, ecVolume.EncodeTsNs)
|
|
if readErr != nil {
|
|
glog.V(3).Infof("recover: readRemoteEcShardInterval %d.%d %d bytes from %+v: %v", ecVolume.VolumeId, shardId, nRead, locations, readErr)
|
|
forgetShardId(ecVolume, shardId)
|
|
}
|
|
if isDeleted {
|
|
isDeletedFlag.Store(true)
|
|
}
|
|
if nRead == size {
|
|
shardIntervals[shardId] = data
|
|
fetched.Add(1)
|
|
}
|
|
}()
|
|
}
|
|
wg.Wait()
|
|
candidates = candidates[wave:]
|
|
available += int(fetched.Load())
|
|
if isDeletedFlag.Load() {
|
|
// every shard of a deleted needle answers deleted, so another wave cannot help
|
|
break
|
|
}
|
|
}
|
|
|
|
return shardIntervals, isDeletedFlag.Load()
|
|
}
|
|
|
|
// checkEcShardRebuildable rejects a parity target. ReconstructData rebuilds data shards
|
|
// only, so it leaves a parity slot nil and reports no error, and the caller would copy
|
|
// that out as a zero-filled buffer and report a successful read.
|
|
func checkEcShardRebuildable(ecVolume *erasure_coding.EcVolume, ecCtx *erasure_coding.ECContext, shardIdToRecover erasure_coding.ShardId) error {
|
|
if int(shardIdToRecover) >= ecCtx.DataShards {
|
|
return fmt.Errorf("cannot reconstruct shard %d.%d: only data shards can be rebuilt, %d of %d are parity",
|
|
ecVolume.VolumeId, shardIdToRecover, ecCtx.ParityShards, ecCtx.Total())
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// reconstructEcShardInterval rebuilds one shard's interval in place from the others.
|
|
func reconstructEcShardInterval(ecVolume *erasure_coding.EcVolume, ecCtx *erasure_coding.ECContext, shardIntervals [][]byte, shardIdToRecover erasure_coding.ShardId) error {
|
|
if err := checkEcShardRebuildable(ecVolume, ecCtx, shardIdToRecover); err != nil {
|
|
return err
|
|
}
|
|
|
|
enc, err := reedsolomon.New(ecCtx.DataShards, ecCtx.ParityShards)
|
|
if err != nil {
|
|
return fmt.Errorf("failed to create encoder: %w", err)
|
|
}
|
|
|
|
// Count and log available shards for diagnostics
|
|
availableShards := make([]erasure_coding.ShardId, 0, ecCtx.Total())
|
|
missingShards := make([]erasure_coding.ShardId, 0, ecCtx.ParityShards+1)
|
|
for shardId := 0; shardId < ecCtx.Total(); shardId++ {
|
|
if shardIntervals[shardId] != nil {
|
|
availableShards = append(availableShards, erasure_coding.ShardId(shardId))
|
|
} else {
|
|
missingShards = append(missingShards, erasure_coding.ShardId(shardId))
|
|
}
|
|
}
|
|
|
|
glog.V(3).Infof("recover ec shard %d.%d: %d shards available %v, %d missing %v",
|
|
ecVolume.VolumeId, shardIdToRecover,
|
|
len(availableShards), availableShards,
|
|
len(missingShards), missingShards)
|
|
|
|
if len(availableShards) < ecCtx.DataShards {
|
|
return fmt.Errorf("cannot recover shard %d.%d: only %d shards available %v, need at least %d (missing: %v)",
|
|
ecVolume.VolumeId, shardIdToRecover,
|
|
len(availableShards), availableShards,
|
|
ecCtx.DataShards, missingShards)
|
|
}
|
|
|
|
// Rebuild only what was asked for. ReconstructData rebuilds every missing data
|
|
// shard, and a gather that stopped at DataShards can leave up to ParityShards of
|
|
// them missing -- an interval-sized buffer and a decode each, discarded unread.
|
|
// The mask is Total() long, not DataShards: reedsolomon documents both lengths
|
|
// but indexes the short one past its end when a parity shard is absent, which
|
|
// here it usually is.
|
|
required := make([]bool, ecCtx.Total())
|
|
required[shardIdToRecover] = true
|
|
if err := enc.ReconstructSome(shardIntervals, required); err != nil {
|
|
return fmt.Errorf("failed to reconstruct data for shard %d.%d with %d available shards %v: %w",
|
|
ecVolume.VolumeId, shardIdToRecover, len(availableShards), availableShards, err)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (s *Store) recoverOneRemoteEcShardInterval(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, shardIdToRecover erasure_coding.ShardId, buf []byte, offset int64) (n int, is_deleted bool, err error) {
|
|
glog.V(3).Infof("recover ec shard %d.%d from other locations", ecVolume.VolumeId, shardIdToRecover)
|
|
|
|
// Reconstruct with the volume's OWN EC ratio (loaded from its .vif), not the
|
|
// build default, so a custom-ratio volume (e.g. 9+3) is decoded with the matrix
|
|
// that actually produced its shards -- decoding it as 10+4 would corrupt the
|
|
// recovered bytes. In OSS the ratio is always 10+4, so this is a no-op.
|
|
ecCtx := ecVolume.ECContext
|
|
if ecCtx == nil {
|
|
ecCtx = erasure_coding.NewDefaultECContext(ecVolume.Collection, ecVolume.VolumeId)
|
|
}
|
|
// checked before the gather: a doomed target should not cost a fan-out, nor drop a
|
|
// sibling's location through forgetShardId on the way to failing
|
|
if err := checkEcShardRebuildable(ecVolume, ecCtx, shardIdToRecover); err != nil {
|
|
return 0, false, err
|
|
}
|
|
|
|
// Charge the buffers this recovery is about to hold against the budget, so a
|
|
// burst of them queues here rather than on the heap: DataShards gathered, plus
|
|
// the one the rebuild allocates for the shard it recreates.
|
|
weight := int64(len(buf)) * int64(ecCtx.DataShards+1)
|
|
if weight > ecRecoverBudget {
|
|
// An interval whose fan-out outgrows the whole budget takes all of it and
|
|
// so runs alone, rather than blocking forever on an acquire that can never
|
|
// succeed. The cap is then one such recovery, not a burst of them.
|
|
weight = ecRecoverBudget
|
|
}
|
|
if err = ecRecoverSem.Acquire(context.Background(), weight); err != nil {
|
|
return 0, false, err
|
|
}
|
|
defer ecRecoverSem.Release(weight)
|
|
|
|
shardIntervals, is_deleted := s.gatherEcShardIntervals(needleId, ecVolume, ecCtx, shardIdToRecover, len(buf), offset)
|
|
if err := reconstructEcShardInterval(ecVolume, ecCtx, shardIntervals, shardIdToRecover); err != nil {
|
|
return 0, is_deleted, err
|
|
}
|
|
glog.V(4).Infof("recovered ec shard %d.%d from other locations", ecVolume.VolumeId, shardIdToRecover)
|
|
|
|
copy(buf, shardIntervals[shardIdToRecover])
|
|
|
|
return len(buf), is_deleted, nil
|
|
}
|
|
|
|
func (s *Store) EcVolumes() (ecVolumes []*erasure_coding.EcVolume) {
|
|
for _, location := range s.Locations {
|
|
location.ecVolumesLock.RLock()
|
|
for _, v := range location.ecVolumes {
|
|
ecVolumes = append(ecVolumes, v)
|
|
}
|
|
location.ecVolumesLock.RUnlock()
|
|
}
|
|
slices.SortFunc(ecVolumes, func(a, b *erasure_coding.EcVolume) int {
|
|
return int(a.VolumeId) - int(b.VolumeId)
|
|
})
|
|
return ecVolumes
|
|
}
|