ec: pin auto-selected shard placement to the disk that already owns the shard (#11029)

* ec: pin auto-selected shard placement to the disk that already owns the shard

A multi-disk server legitimately mounts one EC volume on several disks, so
FindEcShardTargetLocation's per-volume tiers tie at "mounted" and the
free-shard-count tie-break decides — pointing at whichever disk is emptier,
not at the disk that already holds the shard being placed. A re-copy of a
shard the server already has (a retried ec.balance / ec.rebuild move) then
lands on a sibling disk, and both disks register the same (volume, shard id):
the shard is reported to the master from two disk ids, and which claimant
serves reads or survives a later unmount/delete becomes an accident of
Locations order.

Add a tier above "mounted": a disk that already claims one of the shard ids
being placed wins, ahead of the space filters too — re-copying in place
needs no new shard slot, and a genuinely full disk should fail the write
rather than silently split the claim. Applied to the Go selector and the
VolumeEcShardsCopy auto-select (ReceiveFile refuses mounted EC volumes, so
no claim can exist there) and mirrored in the Rust volume server.

Claude-Session: https://claude.ai/code/session_01AWpefvdi4U3HLng18x5CJ9

* ec: refuse a copy batch whose shards are already owned by different disks

Review follow-up: ownership-aware selection ranks a mixed-owner batch
(shard 0 on disk A, shard 2 on disk B — the legitimate multi-disk spread)
into one destination, so the copy would still duplicate the losing disk's
claim. No production caller sends such a batch (balance moves one shard,
rebuild and encode copy shards the target lacks), so fail closed: report
every owning disk via Store.EcShardOwnerDisks and refuse the copy with an
error naming them, telling the caller to split per shard or pass disk_id.
Go and Rust, with unit tests for the owner-reporting contract.

Claude-Session: https://claude.ai/code/session_01AWpefvdi4U3HLng18x5CJ9
This commit is contained in:
Chris Lu
2026-08-29 12:12:41 -07:00
committed by GitHub
parent 83a929720a
commit 74b520113e
5 changed files with 436 additions and 40 deletions
+29 -7
View File
@@ -1820,11 +1820,14 @@ impl VolumeServer for VolumeGrpcService {
}
Some(store.locations[info.disk_id as usize].directory.clone())
} else {
// The mounted-volume refusal above means no disk
// holds an in-memory claim here.
store
.find_ec_shard_target_location(
&info.collection,
vid,
DATA_SHARDS_COUNT as u32,
&[],
)
.map(|i| store.locations[i].directory.clone())
};
@@ -2726,13 +2729,15 @@ impl VolumeServer for VolumeGrpcService {
// When disk_id > 0: use that specific location.
// When disk_id == 0 (unset): auto-select via
// find_ec_shard_target_location, which prefers a disk that
// already has the EC volume mounted, then a disk that owns the
// .ecx on disk (volume not yet mounted — relevant for
// ec.rebuild, where only the first shard carries .ecx and
// subsequent shards must land on the same disk; see #9212),
// then any HDD, then any disk. Pass the build's default
// data-shard count; the helper takes it as a parameter so
// custom-ratio builds can swap it.
// already owns one of the shards being copied (a retried move
// must overwrite in place, not leave two disks of this server
// claiming the same shard), then a disk that already has the EC
// volume mounted, then a disk that owns the .ecx on disk (volume
// not yet mounted — relevant for ec.rebuild, where only the
// first shard carries .ecx and subsequent shards must land on
// the same disk; see #9212), then any HDD, then any disk. Pass
// the build's default data-shard count; the helper takes it as a
// parameter so custom-ratio builds can swap it.
let (dest_dir, dest_idx_dir) = {
let store = self.state.store.read().unwrap();
let count = store.locations.len();
@@ -2748,10 +2753,27 @@ impl VolumeServer for VolumeGrpcService {
let loc = &store.locations[req.disk_id as usize];
(loc.directory.clone(), loc.idx_directory.clone())
} else {
// A batch whose requested shards are already owned by
// different local disks has no single correct destination:
// writing them all to one disk would duplicate the other
// disks' claims. Refuse so the caller splits the batch per
// shard (or chooses explicitly via disk_id).
let owners = store.ec_shard_owner_disks(vid, &req.shard_ids);
if owners.len() > 1 {
let dirs: Vec<&str> = owners
.iter()
.map(|&i| store.locations[i].directory.as_str())
.collect();
return Err(Status::failed_precondition(format!(
"volume {} shards {:?} are already owned by multiple local disks {:?}: no single destination; copy per shard or pass disk_id",
req.volume_id, req.shard_ids, dirs
)));
}
match store.find_ec_shard_target_location(
&req.collection,
vid,
DATA_SHARDS_COUNT as u32,
&req.shard_ids,
) {
Some(i) => {
let loc = &store.locations[i];
+200 -14
View File
@@ -259,11 +259,25 @@ impl Store {
/// Returns the index of the disk that should receive a new EC
/// shard / index file for `(collection, vid)`. Selection order:
///
/// 0. a disk that already owns one of `shard_ids` (in-memory claim),
/// 1. a disk that already has the EC volume mounted (in-memory state),
/// 2. a disk that owns the `.ecx` file on disk (volume not yet mounted),
/// 3. any HDD with free space,
/// 4. any disk with free space.
///
/// Step 0 keeps the per-server invariant that a shard id is owned by
/// at most one disk. Steps 1-4 only know the volume, and a multi-disk
/// server legitimately mounts the same vid on several disks, so the
/// free-count tie-break alone can send a re-copy of a shard the server
/// already holds (a retried `ec.balance` / `ec.rebuild` move) to a
/// sibling disk. Both disks then claim the same (vid, shard) and
/// report it to the master from two disk ids, and which claimant
/// serves reads or survives a later unmount/delete of the shard id
/// becomes an accident of location order. Overwriting in place is what
/// the caller meant, so an owning disk wins ahead of the space filters
/// too — a re-copy needs no new shard slot, and a genuinely full disk
/// fails the write rather than silently splitting the claim.
///
/// Step 2 is the missing primitive that pinned subsequent shards to
/// the first-shard disk during `ec.rebuild`: rebuild only sets
/// `CopyEcxFile=true` on the first shard, then relies on auto-select
@@ -286,20 +300,26 @@ impl Store {
collection: &str,
vid: VolumeId,
data_shard_count: u32,
shard_ids: &[u32],
) -> Option<usize> {
const TIER_ANY_DISK: u8 = 1;
const TIER_HDD: u8 = 2;
const TIER_ECX_ON_DISK: u8 = 3;
const TIER_MOUNTED: u8 = 4;
const TIER_OWNS_SHARD: u8 = 5;
let mut best: Option<(usize, u8, i64)> = None;
// (index, tier, owned shard count, free shard slots)
let mut best: Option<(usize, u8, usize, i64)> = None;
for (i, loc) in self.locations.iter().enumerate() {
if loc.is_disk_space_low.load(Ordering::Relaxed) {
continue;
}
let owned = owned_ec_shard_count(loc, vid, shard_ids);
let free = ec_free_shard_count(loc, data_shard_count);
if free <= 0 {
continue;
if owned == 0 {
if loc.is_disk_space_low.load(Ordering::Relaxed) {
continue;
}
if free <= 0 {
continue;
}
}
let mut tier = TIER_ANY_DISK;
if loc.disk_type == DiskType::HardDrive {
@@ -311,15 +331,41 @@ impl Store {
if loc.has_ec_volume(vid) {
tier = TIER_MOUNTED;
}
if owned > 0 {
tier = TIER_OWNS_SHARD;
}
let better = match best {
None => true,
Some((_, b_tier, b_free)) => tier > b_tier || (tier == b_tier && free > b_free),
// owned only separates disks inside TIER_OWNS_SHARD; it is 0
// everywhere else, so this falls through to the free-count
// tie-break for the other tiers.
Some((_, b_tier, b_owned, b_free)) => {
tier > b_tier
|| (tier == b_tier
&& (owned > b_owned || (owned == b_owned && free > b_free)))
}
};
if better {
best = Some((i, tier, free));
best = Some((i, tier, owned, free));
}
}
best.map(|(i, _, _)| i)
best.map(|(i, _, _, _)| i)
}
/// Returns the distinct disk indexes that already own one of `shard_ids`
/// for `vid`, in location order. More than one owner means the batch has
/// no single correct destination — whichever disk receives it would
/// duplicate a sibling disk's claim — so batch callers must split by
/// owner (`volume_ec_shards_copy` refuses such a batch instead of
/// guessing). Mirrors `Store.EcShardOwnerDisks` in
/// `weed/storage/store_ec.go`.
pub fn ec_shard_owner_disks(&self, vid: VolumeId, shard_ids: &[u32]) -> Vec<usize> {
self.locations
.iter()
.enumerate()
.filter(|(_, loc)| owned_ec_shard_count(loc, vid, shard_ids) > 0)
.map(|(i, _)| i)
.collect()
}
/// Create a new volume, placing it on the location with the most free space.
@@ -1333,6 +1379,20 @@ fn ec_free_shard_count(loc: &DiskLocation, data_shard_count: u32) -> i64 {
free
}
/// Reports how many of `shard_ids` this disk already claims for `vid`, per
/// the in-memory registration the read path and heartbeats use.
///
/// Mirrors `ownedEcShardCount` in `weed/storage/store_ec.go`.
fn owned_ec_shard_count(loc: &DiskLocation, vid: VolumeId, shard_ids: &[u32]) -> usize {
let Some(ecv) = loc.find_ec_volume(vid) else {
return 0;
};
shard_ids
.iter()
.filter(|&&shard_id| ecv.has_shard(shard_id as u8))
.count()
}
// ============================================================================
// Tests
// ============================================================================
@@ -1871,7 +1931,7 @@ mod tests {
let base = volume_file_name(&store.locations[2].idx_directory, collection, vid);
std::fs::write(format!("{}.ecx", base), vec![0u8; 20]).unwrap();
let got = store.find_ec_shard_target_location(collection, vid, 10);
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
assert_eq!(
got,
Some(2),
@@ -2013,7 +2073,7 @@ mod tests {
let base = volume_file_name(&store.locations[2].idx_directory, collection, vid);
std::fs::write(format!("{}.ecx", base), vec![0u8; 20]).unwrap();
let got = store.find_ec_shard_target_location(collection, vid, 10);
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
assert_eq!(got, Some(1), "expected the mounted disk to win; got {:?}", got);
}
@@ -2022,7 +2082,7 @@ mod tests {
#[test]
fn test_find_ec_shard_target_location_falls_through_to_hdd_when_nothing_matches() {
let (store, _tmp) = make_ec_target_test_store(2);
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(3333), 10);
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(3333), 10, &[]);
assert!(got.is_some(), "expected an HDD fallback");
assert_eq!(store.locations[got.unwrap()].disk_type, DiskType::HardDrive);
}
@@ -2038,7 +2098,7 @@ mod tests {
.max_volume_count
.store(0, Ordering::Relaxed);
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(4444), 10);
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(4444), 10, &[]);
assert_eq!(
got,
Some(0),
@@ -2076,7 +2136,7 @@ mod tests {
.mount_ec_shards(vid, collection, &[0], "")
.unwrap();
let got = store.find_ec_shard_target_location(collection, vid, 10);
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
assert_eq!(
got,
Some(1),
@@ -2084,4 +2144,130 @@ mod tests {
got,
);
}
/// Per-server invariant: a shard id is owned by at most one disk.
///
/// A multi-disk server legitimately mounts one vid on several disks, each
/// holding a disjoint subset of the shards, so the mounted tier ties and
/// the free-count tie-break decides — and it points at whichever disk
/// happens to be emptier, not at the disk that already has this shard. A
/// re-copy of a shard the server already holds (a retried `ec.balance` /
/// `ec.rebuild` move) then lands a second copy on the sibling disk, and
/// both disks register the same shard id.
#[test]
fn test_find_ec_shard_target_location_pins_to_the_disk_owning_the_shard() {
let (mut store, _tmp) = make_ec_target_test_store(2);
let collection = "grafana-loki";
let vid = VolumeId(8888);
// Disk 0 owns shards 0 and 1, disk 1 owns shard 2 — so disk 1 is the
// emptier of the two and wins the free-count tie-break.
let base0 = volume_file_name(&store.locations[0].directory, collection, vid);
std::fs::write(format!("{}.ec00", base0), b"x").unwrap();
std::fs::write(format!("{}.ec01", base0), b"x").unwrap();
store.locations[0]
.mount_ec_shards(vid, collection, &[0, 1], "")
.unwrap();
let base1 = volume_file_name(&store.locations[1].directory, collection, vid);
std::fs::write(format!("{}.ec02", base1), b"x").unwrap();
store.locations[1]
.mount_ec_shards(vid, collection, &[2], "")
.unwrap();
assert_eq!(
store.find_ec_shard_target_location(collection, vid, 10, &[0]),
Some(0),
"a copy of shard 0 left its owning disk",
);
assert_eq!(
store.find_ec_shard_target_location(collection, vid, 10, &[2]),
Some(1),
"a copy of shard 2 left its owning disk",
);
// A shard no disk owns yet is placed by the unchanged waterfall: both
// disks have it mounted, so the emptier one wins.
assert_eq!(
store.find_ec_shard_target_location(collection, vid, 10, &[7]),
Some(1),
"placement of an unclaimed shard changed",
);
}
/// The second half of the invariant: a disk with no free shard slots
/// still wins for a shard it already owns. Re-copying that shard
/// overwrites bytes the disk is already accounted for, while routing to a
/// sibling splits the claim across two disks.
#[test]
fn test_find_ec_shard_target_location_owning_disk_wins_when_full() {
let (mut store, _tmp) = make_ec_target_test_store(2);
store.locations[0]
.max_volume_count
.store(1, Ordering::Relaxed);
let collection = "grafana-loki";
let vid = VolumeId(9999);
let base = volume_file_name(&store.locations[0].directory, collection, vid);
std::fs::write(format!("{}.ec00", base), b"x").unwrap();
store.locations[0]
.mount_ec_shards(vid, collection, &[0], "")
.unwrap();
// Fill disk 0 past its shard-slot budget so ec_free_shard_count is 0.
let filler = VolumeId(10000);
let filler_base = volume_file_name(&store.locations[0].directory, collection, filler);
let filler_shards: Vec<u32> = (0..10).collect();
for shard_id in &filler_shards {
std::fs::write(format!("{}.ec{:02}", filler_base, shard_id), b"x").unwrap();
}
store.locations[0]
.mount_ec_shards(filler, collection, &filler_shards, "")
.unwrap();
assert_eq!(
store.find_ec_shard_target_location(collection, vid, 10, &[0]),
Some(0),
"a copy of shard 0 left its owning disk when the disk was full",
);
// A shard it does not own still respects the space filter.
assert_eq!(
store.find_ec_shard_target_location(collection, vid, 10, &[7]),
Some(1),
"an unclaimed shard should go to the disk with free slots",
);
}
/// Mixed-owner batch contract: a batch whose requested shards are
/// already owned by different disks reports every owner, so
/// `volume_ec_shards_copy` can refuse it rather than rank the owners
/// into one destination and duplicate the loser's claim.
#[test]
fn test_ec_shard_owner_disks() {
let (mut store, _tmp) = make_ec_target_test_store(3);
let collection = "grafana-loki";
let vid = VolumeId(11111);
let base0 = volume_file_name(&store.locations[0].directory, collection, vid);
std::fs::write(format!("{}.ec00", base0), b"x").unwrap();
std::fs::write(format!("{}.ec01", base0), b"x").unwrap();
store.locations[0]
.mount_ec_shards(vid, collection, &[0, 1], "")
.unwrap();
let base1 = volume_file_name(&store.locations[1].directory, collection, vid);
std::fs::write(format!("{}.ec02", base1), b"x").unwrap();
store.locations[1]
.mount_ec_shards(vid, collection, &[2], "")
.unwrap();
assert_eq!(
store.ec_shard_owner_disks(vid, &[0, 2]),
vec![0, 1],
"a batch owned by two disks must report both owners",
);
// A batch on one disk, with or without unowned extras, has one owner.
assert_eq!(store.ec_shard_owner_disks(vid, &[0, 1, 7]), vec![0]);
// A wholly unowned batch reports none — fresh placement stays allowed.
assert!(store.ec_shard_owner_disks(vid, &[7, 8]).is_empty());
}
}
+27 -9
View File
@@ -343,15 +343,33 @@ func (vs *VolumeServer) VolumeEcShardsCopy(ctx context.Context, req *volume_serv
location = vs.store.Locations[req.DiskId]
glog.V(1).Infof("Using disk %d for EC shard copy: %s", req.DiskId, location.Directory)
} else {
// Auto-select the target disk: prefer a disk that already has the
// EC volume mounted, then a disk that owns the .ecx on disk (the
// volume hasn't been mounted yet — relevant for ec.rebuild, where
// only the first shard carries .ecx and subsequent shards must
// land on the same disk; see #9212), then any HDD, then any disk.
// Pass the build's default data-shard count for free-slot maths;
// the helper takes it as a parameter so custom-ratio builds (e.g.
// enterprise) can swap it without touching this file.
location = vs.store.FindEcShardTargetLocation(req.Collection, needle.VolumeId(req.VolumeId), erasure_coding.DataShardsCount)
// Auto-select the target disk: prefer a disk that already owns one
// of the shards being copied (a retried move must overwrite in
// place, not leave two disks of this server claiming the same
// shard), then a disk that already has the EC volume mounted, then
// a disk that owns the .ecx on disk (the volume hasn't been mounted
// yet — relevant for ec.rebuild, where only the first shard carries
// .ecx and subsequent shards must land on the same disk; see
// #9212), then any HDD, then any disk. Pass the build's default
// data-shard count for free-slot maths; the helper takes it as a
// parameter so custom-ratio builds (e.g. enterprise) can swap it
// without touching this file.
shardIds := make([]erasure_coding.ShardId, 0, len(req.ShardIds))
for _, shardId := range req.ShardIds {
shardIds = append(shardIds, erasure_coding.ShardId(shardId))
}
// A batch whose requested shards are already owned by different local
// disks has no single correct destination: writing them all to one
// disk would duplicate the other disks' claims. Refuse so the caller
// splits the batch per shard (or chooses explicitly via disk_id).
if owners := vs.store.EcShardOwnerDisks(needle.VolumeId(req.VolumeId), shardIds); len(owners) > 1 {
dirs := make([]string, 0, len(owners))
for _, owner := range owners {
dirs = append(dirs, owner.Directory)
}
return nil, fmt.Errorf("volume %d shards %v are already owned by multiple local disks %v: no single destination; copy per shard or pass disk_id", req.VolumeId, req.ShardIds, dirs)
}
location = vs.store.FindEcShardTargetLocation(req.Collection, needle.VolumeId(req.VolumeId), erasure_coding.DataShardsCount, shardIds...)
if location == nil {
return nil, fmt.Errorf("no space left")
}
+79 -10
View File
@@ -34,11 +34,25 @@ var errShardNotLocal = errors.New("ec shard not on this server")
// FindEcShardTargetLocation returns the disk that should receive a new
// shard / index file for (collection, vid). The selection order is:
//
// 0. a disk that already owns one of shardIds (in-memory claim),
// 1. a disk that already has the EC volume mounted (in-memory state),
// 2. a disk that owns the .ecx file on disk (volume not mounted yet),
// 3. any HDD with free space,
// 4. any disk with free space.
//
// Step 0 keeps the per-server invariant that a shard id is owned by at
// most one disk. Steps 1-4 only know the volume, and a multi-disk server
// legitimately mounts the same vid on several disks, so the free-count
// tie-break alone can send a re-copy of a shard the server already holds
// (a retried ec.balance / ec.rebuild move) to a sibling disk. Both disks
// then claim the same (vid, shard) and report it to the master from two
// disk ids, and which claimant serves reads or survives a later
// unmount/delete of the shard id becomes an accident of Locations order.
// Overwriting in place is what the caller meant, so an owning disk wins
// ahead of the space filters too — a re-copy of a shard already on that
// disk needs no new shard slot, and a genuinely full disk fails the
// write with ENOSPC rather than silently splitting the claim.
//
// Step 2 is the missing primitive that pinned subsequent shards to the
// first-shard disk during ec.rebuild. ec.rebuild only sets CopyEcxFile=true
// for the first shard, then relies on auto-select to land later shards on
@@ -58,26 +72,31 @@ var errShardNotLocal = errors.New("ec shard not on this server")
// across four FindFreeLocation passes was equivalent but acquired
// volumesLock and ecVolumesLock RLocks (via VolumesLen / EcShardCount) up
// to four times per disk per call.
func (s *Store) FindEcShardTargetLocation(collection string, vid needle.VolumeId, dataShardCount int) *DiskLocation {
func (s *Store) FindEcShardTargetLocation(collection string, vid needle.VolumeId, dataShardCount int, shardIds ...erasure_coding.ShardId) *DiskLocation {
const (
tierAnyDisk = iota + 1
tierHDD
tierEcxOnDisk
tierMounted
tierOwnsShard
)
var (
best *DiskLocation
bestTier int
bestFree int32
best *DiskLocation
bestTier int
bestOwned int
bestFree int32
)
for _, loc := range s.Locations {
if loc.isDiskSpaceLow.Load() {
continue
}
owned := ownedEcShardCount(loc, vid, shardIds)
freeCount := ecFreeShardCount(loc, dataShardCount)
if freeCount <= 0 {
continue
if owned == 0 {
if loc.isDiskSpaceLow.Load() {
continue
}
if freeCount <= 0 {
continue
}
}
tier := tierAnyDisk
if loc.DiskType == types.HardDriveType {
@@ -89,15 +108,65 @@ func (s *Store) FindEcShardTargetLocation(collection string, vid needle.VolumeId
if _, mounted := loc.FindEcVolume(vid); mounted {
tier = tierMounted
}
if best == nil || tier > bestTier || (tier == bestTier && freeCount > bestFree) {
if owned > 0 {
tier = tierOwnsShard
}
better := best == nil || tier > bestTier
if !better && tier == bestTier {
// owned only separates disks inside tierOwnsShard; it is 0
// everywhere else, so this falls through to the free-count
// tie-break for the other tiers.
better = owned > bestOwned || (owned == bestOwned && freeCount > bestFree)
}
if better {
best = loc
bestTier = tier
bestOwned = owned
bestFree = freeCount
}
}
return best
}
// EcShardOwnerDisks returns the distinct disks that already own one of
// shardIds for vid, in Locations order. More than one owner means the batch
// has no single correct destination — whichever disk receives it would
// duplicate a sibling disk's claim — so batch callers must split by owner
// (VolumeEcShardsCopy refuses such a batch instead of guessing).
func (s *Store) EcShardOwnerDisks(vid needle.VolumeId, shardIds []erasure_coding.ShardId) []*DiskLocation {
var owners []*DiskLocation
for _, loc := range s.Locations {
if ownedEcShardCount(loc, vid, shardIds) > 0 {
owners = append(owners, loc)
}
}
return owners
}
// ownedEcShardCount reports how many of shardIds this disk already claims
// for vid, per the in-memory registration the read path and heartbeats use.
func ownedEcShardCount(loc *DiskLocation, vid needle.VolumeId, shardIds []erasure_coding.ShardId) int {
if len(shardIds) == 0 {
return 0
}
loc.ecVolumesLock.RLock()
defer loc.ecVolumesLock.RUnlock()
ecVolume, found := loc.ecVolumes[vid]
if !found {
return 0
}
owned := 0
for _, shard := range ecVolume.Shards {
for _, shardId := range shardIds {
if shard.ShardId == shardId {
owned++
break
}
}
}
return owned
}
// ecFreeShardCount returns the free EC shard capacity of loc, expressed
// in shard slots (not volume-equivalent slots). dataShardCount is the
// data-shard count of the EC layout being placed — see
@@ -148,6 +148,107 @@ func TestFindEcShardTargetLocation_TightProvisioningKeepsEcxDisk(t *testing.T) {
}
}
// TestFindEcShardTargetLocation_PinsToTheDiskOwningTheShard covers the
// per-server invariant that a shard id is owned by at most one disk.
//
// A multi-disk server legitimately mounts one vid on several disks, each
// holding a disjoint subset of the shards, so the mounted tier ties and the
// free-count tie-break decides — and it points at whichever disk happens to
// be emptier, not at the disk that already has this shard. A re-copy of a
// shard the server already holds (a retried ec.balance / ec.rebuild move)
// then lands a second copy on the sibling disk, and both disks register the
// same shard id.
func TestFindEcShardTargetLocation_PinsToTheDiskOwningTheShard(t *testing.T) {
store := newEcTargetTestStore(t, 2)
collection := "grafana-loki"
vid := needle.VolumeId(8888)
// Disk 0 owns shards 0 and 1, disk 1 owns shard 2 — so disk 1 is the
// emptier of the two and wins the free-count tie-break.
mountEcShards(store.Locations[0], collection, vid, 0, 1)
mountEcShards(store.Locations[1], collection, vid, 2)
if got := store.FindEcShardTargetLocation(collection, vid, dataShardCount, 0); got != store.Locations[0] {
t.Errorf("a copy of shard 0 left its owning disk: got %v, want %s", got, store.Locations[0].Directory)
}
if got := store.FindEcShardTargetLocation(collection, vid, dataShardCount, 2); got != store.Locations[1] {
t.Errorf("a copy of shard 2 left its owning disk: got %v, want %s", got, store.Locations[1].Directory)
}
// A shard no disk owns yet is placed by the unchanged waterfall: both
// disks have it mounted, so the emptier one wins.
if got := store.FindEcShardTargetLocation(collection, vid, dataShardCount, 7); got != store.Locations[1] {
t.Errorf("placement of an unclaimed shard changed: got %v, want the emptier mounted disk", got)
}
}
// TestFindEcShardTargetLocation_OwningDiskWinsWhenFull pins the second half
// of the invariant: a disk with no free shard slots still wins for a shard it
// already owns. Re-copying that shard overwrites bytes the disk is already
// accounted for, while the alternative — routing to a sibling — splits the
// claim across two disks.
func TestFindEcShardTargetLocation_OwningDiskWinsWhenFull(t *testing.T) {
store := newEcTargetTestStore(t, 2)
collection := "grafana-loki"
vid := needle.VolumeId(9999)
store.Locations[0].MaxVolumeCount = 1
mountEcShards(store.Locations[0], collection, vid, 0)
// Fill disk 0 past its shard-slot budget so ecFreeShardCount reports 0.
filler := needle.VolumeId(10000)
fillerShards := make([]erasure_coding.ShardId, dataShardCount)
for i := range fillerShards {
fillerShards[i] = erasure_coding.ShardId(i)
}
mountEcShards(store.Locations[0], collection, filler, fillerShards...)
if got := store.FindEcShardTargetLocation(collection, vid, dataShardCount, 0); got != store.Locations[0] {
t.Errorf("a copy of shard 0 left its owning disk when the disk was full: got %v", got)
}
// A shard it does not own still respects the space filter.
if got := store.FindEcShardTargetLocation(collection, vid, dataShardCount, 7); got != store.Locations[1] {
t.Errorf("an unclaimed shard should go to the disk with free slots: got %v", got)
}
}
// TestEcShardOwnerDisks pins the mixed-owner batch contract: a batch whose
// requested shards are already owned by different disks reports every owner,
// so VolumeEcShardsCopy can refuse it rather than rank the owners into one
// destination and duplicate the loser's claim.
func TestEcShardOwnerDisks(t *testing.T) {
store := newEcTargetTestStore(t, 3)
collection := "grafana-loki"
vid := needle.VolumeId(11111)
mountEcShards(store.Locations[0], collection, vid, 0, 1)
mountEcShards(store.Locations[1], collection, vid, 2)
if owners := store.EcShardOwnerDisks(vid, []erasure_coding.ShardId{0, 2}); len(owners) != 2 {
t.Errorf("a batch owned by two disks reported %d owners; the copy handler would duplicate a claim", len(owners))
}
// A batch on one disk, with or without unowned extras, has one owner.
if owners := store.EcShardOwnerDisks(vid, []erasure_coding.ShardId{0, 1, 7}); len(owners) != 1 || owners[0] != store.Locations[0] {
t.Errorf("a single-owner batch reported owners %v; want just disk 0", owners)
}
// A wholly unowned batch reports none — fresh placement stays allowed.
if owners := store.EcShardOwnerDisks(vid, []erasure_coding.ShardId{7, 8}); len(owners) != 0 {
t.Errorf("an unowned batch reported owners %v; want none", owners)
}
}
// mountEcShards registers an EcVolume for vid on loc claiming shardIds.
func mountEcShards(loc *DiskLocation, collection string, vid needle.VolumeId, shardIds ...erasure_coding.ShardId) {
ecVolume := &erasure_coding.EcVolume{VolumeId: vid, Collection: collection}
for _, shardId := range shardIds {
ecVolume.Shards = append(ecVolume.Shards, &erasure_coding.EcVolumeShard{
VolumeId: vid, ShardId: shardId, Collection: collection,
})
}
loc.ecVolumesLock.Lock()
loc.ecVolumes[vid] = ecVolume
loc.ecVolumesLock.Unlock()
}
// newEcTargetTestStore is a leaner cousin of the helper in
// store_load_balancing_test.go: it spins up an in-memory Store with N
// HDD disk locations under a single t.TempDir and consumes any heartbeat