diff --git a/seaweed-volume/src/storage/store.rs b/seaweed-volume/src/storage/store.rs index 7039d553a..b7d09c9d6 100644 --- a/seaweed-volume/src/storage/store.rs +++ b/seaweed-volume/src/storage/store.rs @@ -1553,32 +1553,40 @@ impl Store { vid: VolumeId, preallocate: u64, ) -> Result, VolumeError> { - // Required space matches Go's CompactVolume check: the larger of the - // requested preallocation and the estimated compacted size — the live - // needles, not the .dat the garbage already occupies, so a full disk - // can still be reclaimed. - let (loc_idx, space_needed) = { - let (loc_idx, v) = self + // Mirrors Go's ensureCompactVolumeSpace, per filesystem. + let (dir, dir_idx, data_bytes, index_bytes) = { + let (_, v) = self .find_volume(vid) .ok_or(VolumeError::VolumeNotFound(vid))?; - let live_count = (v.file_count() - v.deleted_count()).max(0) as u64; - let live_bytes = v.content_size().saturating_sub(v.deleted_size()); - let per_needle = (get_actual_size(Size(0), v.version()) - + NEEDLE_PADDING_SIZE as i64 - + NEEDLE_MAP_ENTRY_SIZE as i64) as u64; - let estimated = SUPER_BLOCK_SIZE as u64 + live_count * per_needle + live_bytes; - let space_needed = std::cmp::max(preallocate, estimated); - (loc_idx, space_needed + space_needed / 10) + let (data_bytes, index_bytes) = compaction_space_needed(v, preallocate); + ( + v.dir().to_string(), + v.dir_idx().to_string(), + data_bytes, + index_bytes, + ) }; - let dir = self.locations[loc_idx].directory.clone(); - let (_, free) = crate::storage::disk_location::get_disk_stats(&dir); - if free < space_needed { - return Err(VolumeError::InsufficientSpace { - vid, - required: space_needed, - free, - }); + let check = |dir: &str, needed: u64| -> Result<(), VolumeError> { + let (_, free) = crate::storage::disk_location::get_disk_stats(dir); + if free < needed { + return Err(VolumeError::InsufficientSpace { + vid, + required: needed, + free, + }); + } + Ok(()) + }; + + if dir_idx.is_empty() || dir_idx == dir { + check(&dir, data_bytes + index_bytes)?; + } else if !same_filesystem(&dir, &dir_idx) { + check(&dir, data_bytes)?; + check(&dir_idx, index_bytes)?; + } else { + check(&dir, data_bytes + index_bytes)?; + check(&dir_idx, index_bytes)?; } let (_, v) = self @@ -1751,6 +1759,65 @@ fn owned_ec_shard_count(loc: &DiskLocation, vid: VolumeId, shard_ids: &[ShardId] .count() } +/// Mirrors Go's compactionSpaceNeeded: what compaction will write, split into +/// the new .dat and rebuilt .idx shares, each capped at its current file. +fn compaction_space_needed(v: &Volume, preallocate: u64) -> (u64, u64) { + let mut data_bytes = v.current_dat_file_size().unwrap_or(0); + let mut index_bytes = v.idx_file_size(); + + let live_count = v.file_count() - v.deleted_count(); + let live_content = v.content_size() as i64 - v.deleted_size() as i64; + // Unknown or inconsistent deleted sizes: the whole volume stays the + // estimate. + let deleted_size_known = v.deleted_count() == 0 || v.deleted_size() > 0; + if deleted_size_known && live_count >= 0 && live_content >= 0 { + // Empty-needle framing plus a padding unit covers the worst case. + let per_needle = + (get_actual_size(Size(0), v.version()) + NEEDLE_PADDING_SIZE as i64) as u64; + let estimate = with_headroom( + SUPER_BLOCK_SIZE as u64 + live_content as u64 + live_count as u64 * per_needle, + ); + if estimate < data_bytes { + data_bytes = estimate; + } + let estimate = with_headroom(live_count as u64 * NEEDLE_MAP_ENTRY_SIZE as u64); + if estimate < index_bytes { + index_bytes = estimate; + } + } + if preallocate > data_bytes { + data_bytes = preallocate; + } + (data_bytes, index_bytes) +} + +/// Headroom for Bloom-filter false positives in the live/deleted counters. +fn with_headroom(estimate: u64) -> u64 { + estimate + estimate / 16 +} + +/// Whether two directories draw on the same free-space pool; in doubt, yes. +fn same_filesystem(a: &str, b: &str) -> bool { + if a == b { + return true; + } + same_filesystem_impl(a, b) +} + +#[cfg(unix)] +fn same_filesystem_impl(a: &str, b: &str) -> bool { + use std::os::unix::fs::MetadataExt; + match (std::fs::metadata(a), std::fs::metadata(b)) { + (Ok(ma), Ok(mb)) => ma.dev() == mb.dev(), + _ => true, + } +} + +#[cfg(not(unix))] +fn same_filesystem_impl(_a: &str, _b: &str) -> bool { + true +} + // ============================================================================ // Tests // ============================================================================ diff --git a/seaweed-volume/src/storage/volume.rs b/seaweed-volume/src/storage/volume.rs index d2991a7bd..bfe04df4c 100644 --- a/seaweed-volume/src/storage/volume.rs +++ b/seaweed-volume/src/storage/volume.rs @@ -1860,7 +1860,7 @@ impl Volume { self.dat_file.is_some() || self.remote_dat_file.is_some() } - fn current_dat_file_size(&self) -> io::Result { + pub(crate) fn current_dat_file_size(&self) -> io::Result { if let Some(ref f) = self.dat_file { Ok(f.metadata()?.len()) } else if let Some(ref remote_dat_file) = self.remote_dat_file { @@ -3950,6 +3950,11 @@ impl Volume { &self.dir } + /// Get the directory this volume's index is stored in. + pub fn dir_idx(&self) -> &str { + &self.dir_idx + } + /// Throttle IO during compaction to avoid saturating disk. pub fn maybe_throttle_compaction(&self, bytes_written: u64) { if self.compaction_byte_per_second <= 0 || !self.is_compacting() { diff --git a/weed/storage/store_vacuum.go b/weed/storage/store_vacuum.go index 99878c952..aee7843c1 100644 --- a/weed/storage/store_vacuum.go +++ b/weed/storage/store_vacuum.go @@ -61,50 +61,93 @@ func (s *Store) CommitCleanupVolume(vid needle.VolumeId) error { return fmt.Errorf("volume id %d is not found during cleaning up: %w", vid, ErrVolumeNotFound) } -// estimatedCompactedSize is what compaction writes: a superblock, the live -// needles with their on-disk framing, and an index with live entries only. -// Deleted bytes do not carry over, so a mostly-garbage volume needs far less -// space than it occupies. -func estimatedCompactedSize(v *Volume) int64 { - liveCount := v.FileCount() - if deleted := v.DeletedCount(); deleted < liveCount { - liveCount -= deleted - } else { - liveCount = 0 +// compactionDiskFree reports the free bytes on the disk holding dir, and +// compactionSameFilesystem whether two directories draw on the same pool. +// Both are variables so a test can stand in for a full or a split disk. +var ( + compactionDiskFree = func(dir string) uint64 { + return stats.NewDiskStatus(dir).Free } - liveBytes := v.ContentSize() - if deleted := v.DeletedSize(); deleted < liveBytes { - liveBytes -= deleted - } else { - liveBytes = 0 + compactionSameFilesystem = sameFilesystem +) + +// compactionSpaceNeeded estimates what CompactByIndex will write for v: a new +// .dat holding the live needles with their on-disk framing behind a superblock +// (or preallocate, when that is larger, since the file is preallocated to it), +// and a rebuilt index with one entry per live needle. The volume's current size +// is the wrong yardstick: the more garbage a volume holds, the less its +// compaction writes, and a store that filled up until its volumes went +// read-only is exactly where the all-garbage volumes must still compact to +// give the space back (issue #11516). Neither estimate exceeds the current file. +func compactionSpaceNeeded(v *Volume, preallocate int64) (dataBytes, indexBytes int64) { + datSize, idxSize, _ := v.FileStat() + dataBytes, indexBytes = int64(datSize), int64(idxSize) + + liveCount := int64(v.FileCount()) - int64(v.DeletedCount()) + liveContent := int64(v.ContentSize()) - int64(v.DeletedSize()) + // A .sdx converted back to .idx carries no deleted sizes (see + // garbageLevel), and counters that disagree mean the metric is off; + // either way the whole volume stays the estimate. + deletedSizeKnown := v.DeletedCount() == 0 || v.DeletedSize() > 0 + if deletedSizeKnown && liveCount >= 0 && liveContent >= 0 { + // GetActualSize(0) is the framing of an empty needle; another + // padding unit covers the worst case for any other size. + perNeedle := needle.GetActualSize(0, v.Version()) + types.NeedlePaddingSize + // Counters rebuilt from an index file (LevelDB and sorted maps) go + // through a Bloom filter with a 0.1% false positive rate that can + // count a live needle as deleted. A few percent of headroom covers + // that many times over. + if estimate := withHeadroom(super_block.SuperBlockSize + liveContent + liveCount*perNeedle); estimate < dataBytes { + dataBytes = estimate + } + if estimate := withHeadroom(liveCount * types.NeedleMapEntrySize); estimate < indexBytes { + indexBytes = estimate + } } - perNeedle := needle.GetActualSize(0, v.Version()) + types.NeedlePaddingSize + types.NeedleMapEntrySize - return super_block.SuperBlockSize + int64(liveCount)*perNeedle + int64(liveBytes) + if preallocate > dataBytes { + dataBytes = preallocate + } + return dataBytes, indexBytes +} + +func withHeadroom(estimate int64) int64 { + return estimate + estimate/16 } func ensureCompactVolumeSpace(v *Volume, preallocate int64) error { + dataBytes, indexBytes := compactionSpaceNeeded(v, preallocate) volumeSize, indexSize, _ := v.FileStat() - - // The compacted output holds live needles only, so measure against the - // estimated compacted size — otherwise a disk full of garbage can never - // reclaim itself. - estimatedCompactSize := estimatedCompactedSize(v) - spaceNeeded := preallocate - if estimatedCompactSize > preallocate { - spaceNeeded = estimatedCompactSize + check := func(dir string, needed int64) error { + free := compactionDiskFree(dir) + if int64(free) < needed { + return fmt.Errorf("insufficient free space for compaction in %s: need %d bytes (data: %d, index: %d, current volume: %d, current index: %d), but only %d bytes available: %w", + dir, needed, dataBytes, indexBytes, volumeSize, indexSize, free, ErrInsufficientSpace) + } + glog.V(1).Infof("volume %d compaction space check in %s: data=%d, index=%d, current volume=%d, space_needed=%d, free_space=%d", + v.Id, dir, dataBytes, indexBytes, volumeSize, needed, free) + return nil } - spaceNeeded += spaceNeeded / 10 - - diskStatus := stats.NewDiskStatus(v.dir) - if int64(diskStatus.Free) < spaceNeeded { - return fmt.Errorf("insufficient free space for compaction: need %d bytes (volume: %d, index: %d), but only %d bytes available: %w", - spaceNeeded, volumeSize, indexSize, diskStatus.Free, ErrInsufficientSpace) + // The new .dat lands next to the old one and the new .idx next to the old + // index. When the index directory is on another filesystem each disk + // answers for its own share; two directories on one filesystem draw on + // the same free space and must cover the sum. + if v.dirIdx == "" || v.dirIdx == v.dir { + return check(v.dir, dataBytes+indexBytes) } - - glog.V(1).Infof("volume %d compaction space check: volume=%d, index=%d, space_needed=%d, free_space=%d", - v.Id, volumeSize, indexSize, spaceNeeded, diskStatus.Free) - - return nil + if !compactionSameFilesystem(v.dir, v.dirIdx) { + if err := check(v.dir, dataBytes); err != nil { + return err + } + return check(v.dirIdx, indexBytes) + } + // Same filesystem as far as the identity check can tell. The index + // directory is still asked for its own share, because a mount point the + // check cannot see (a volume mounted under one drive letter on Windows) + // would otherwise go unchecked. + if err := check(v.dir, dataBytes+indexBytes); err != nil { + return err + } + return check(v.dirIdx, indexBytes) } func (s *Store) CompactVolumeFiles(vid needle.VolumeId, collection string, location *DiskLocation, needleMapKind NeedleMapKind, ldbTimeout int64, preallocate int64, compactionBytePerSecond int64) (err error) { diff --git a/weed/storage/store_vacuum_fs_unix.go b/weed/storage/store_vacuum_fs_unix.go new file mode 100644 index 000000000..1147a6a49 --- /dev/null +++ b/weed/storage/store_vacuum_fs_unix.go @@ -0,0 +1,27 @@ +//go:build !windows + +package storage + +import ( + "os" + "syscall" +) + +// sameFilesystem reports whether two directories share one free-space pool. +// When in doubt it says yes, which makes the space check ask for the sum. +func sameFilesystem(a, b string) bool { + if a == b { + return true + } + sa, errA := os.Stat(a) + sb, errB := os.Stat(b) + if errA != nil || errB != nil { + return true + } + sta, okA := sa.Sys().(*syscall.Stat_t) + stb, okB := sb.Sys().(*syscall.Stat_t) + if !okA || !okB { + return true + } + return sta.Dev == stb.Dev +} diff --git a/weed/storage/store_vacuum_fs_windows.go b/weed/storage/store_vacuum_fs_windows.go new file mode 100644 index 000000000..cfbb12e72 --- /dev/null +++ b/weed/storage/store_vacuum_fs_windows.go @@ -0,0 +1,48 @@ +//go:build windows + +package storage + +import ( + "path/filepath" + "strings" + + "golang.org/x/sys/windows" +) + +// sameFilesystem reports whether two directories share one free-space pool. +// On Windows a volume can be reached through a drive letter and through a +// folder it is mounted on, so the path prefix says nothing; the volume GUID +// behind each path does. When in doubt it says yes, which makes the space +// check ask for the sum. +func sameFilesystem(a, b string) bool { + if a == b { + return true + } + va, errA := volumeGUID(a) + vb, errB := volumeGUID(b) + if errA != nil || errB != nil { + return true + } + return strings.EqualFold(va, vb) +} + +func volumeGUID(path string) (string, error) { + abs, err := filepath.Abs(path) + if err != nil { + return "", err + } + p, err := windows.UTF16PtrFromString(abs) + if err != nil { + return "", err + } + mountPoint := make([]uint16, windows.MAX_LONG_PATH) + if err := windows.GetVolumePathName(p, &mountPoint[0], uint32(len(mountPoint))); err != nil { + return "", err + } + // \\?\Volume{GUID}\ is 49 characters plus the terminator. + guid := make([]uint16, 50) + if err := windows.GetVolumeNameForVolumeMountPoint(&mountPoint[0], &guid[0], uint32(len(guid))); err != nil { + return "", err + } + return windows.UTF16ToString(guid), nil +} diff --git a/weed/storage/store_vacuum_test.go b/weed/storage/store_vacuum_test.go index 8d7bb7139..3adc16433 100644 --- a/weed/storage/store_vacuum_test.go +++ b/weed/storage/store_vacuum_test.go @@ -1,6 +1,8 @@ package storage import ( + "errors" + "path/filepath" "testing" "github.com/stretchr/testify/require" @@ -10,108 +12,180 @@ import ( "github.com/seaweedfs/seaweedfs/weed/storage/types" ) -func TestSpaceCalculation(t *testing.T) { - // Test the space calculation logic - testCases := []struct { - name string - volumeSize uint64 - indexSize uint64 - preallocate int64 - expectedMin int64 - }{ - { - name: "Large volume, small preallocate", - volumeSize: 244 * 1024 * 1024 * 1024, // 244GB - indexSize: 1024 * 1024, // 1MB - preallocate: 1024, // 1KB - expectedMin: int64((244*1024*1024*1024 + 1024*1024) * 11 / 10), // +10% buffer - }, - { - name: "Small volume, large preallocate", - volumeSize: 100 * 1024 * 1024, // 100MB - indexSize: 1024, // 1KB - preallocate: 1024 * 1024 * 1024, // 1GB - expectedMin: int64(1024 * 1024 * 1024 * 11 / 10), // preallocate + 10% - }, - } - - for _, tc := range testCases { - t.Run(tc.name, func(t *testing.T) { - // Calculate space needed using the same logic as our fix - estimatedCompactSize := int64(tc.volumeSize + tc.indexSize) - spaceNeeded := tc.preallocate - if estimatedCompactSize > tc.preallocate { - spaceNeeded = estimatedCompactSize - } - // Add 10% safety buffer - spaceNeeded = spaceNeeded + (spaceNeeded / 10) - - if spaceNeeded < tc.expectedMin { - t.Errorf("Space calculation too low: got %d, expected at least %d", spaceNeeded, tc.expectedMin) - } - - t.Logf("Volume size: %d bytes, Space needed: %d bytes (%.2f%% of volume size)", - tc.volumeSize, spaceNeeded, float64(spaceNeeded)/float64(tc.volumeSize)*100) - }) - } -} - -// Compaction writes live needles only, so the space check must be measured -// against the live size, not the .dat the garbage occupies — a full disk -// needs the estimate to shrink or it can never reclaim. -func TestEstimatedCompactedSizeCountsLiveNeedles(t *testing.T) { +// newVolumeWithGarbage writes live+deleted needles and deletes the first +// deleted ones, so the volume carries that much garbage on disk. +func newVolumeWithGarbage(t *testing.T, live, deleted int) *Volume { + t.Helper() dir := t.TempDir() - v, err := NewVolume(dir, dir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0) if err != nil { - t.Fatalf("volume creation: %v", err) + t.Fatalf("NewVolume: %v", err) } - defer v.Close() - - const count = 20 - for i := 1; i <= count; i++ { + t.Cleanup(v.Close) + for i := 1; i <= live+deleted; i++ { if _, _, _, err := v.writeNeedle2(newRandomNeedle(uint64(i)), true, false, false); err != nil { t.Fatalf("write needle %d: %v", i, err) } } - datSize, _, _ := v.FileStat() - - fullEstimate := estimatedCompactedSize(v) - if fullEstimate <= super_block.SuperBlockSize { - t.Fatalf("estimate for all-live volume = %d, want > superblock", fullEstimate) - } - - for i := 1; i < count; i++ { - if _, err := v.doDeleteRequest(newEmptyNeedle(uint64(i))); err != nil { + for i := 1; i <= deleted; i++ { + if _, err := v.deleteNeedle2(newEmptyNeedle(uint64(i))); err != nil { t.Fatalf("delete needle %d: %v", i, err) } } + return v +} - estimate := estimatedCompactedSize(v) - if estimate >= int64(datSize) { - t.Fatalf("estimate %d not below .dat size %d with 19/20 needles deleted", estimate, datSize) - } - if estimate <= super_block.SuperBlockSize { - t.Fatalf("estimate %d lost the one live needle", estimate) - } - live := int64(v.FileCount()-v.DeletedCount())*types.NeedleMapEntrySize + super_block.SuperBlockSize - if estimate < live { - t.Fatalf("estimate %d below superblock + live index entries %d", estimate, live) - } +func stubCompactionDiskFree(t *testing.T, free uint64) { + t.Helper() + prev := compactionDiskFree + compactionDiskFree = func(string) uint64 { return free } + t.Cleanup(func() { compactionDiskFree = prev }) +} - if _, err := v.doDeleteRequest(newEmptyNeedle(uint64(count))); err != nil { - t.Fatalf("delete last needle: %v", err) +func TestCompactionSpaceNeeded_CountsLiveBytesNotVolumeSize(t *testing.T) { + v := newVolumeWithGarbage(t, 20, 2000) + datSize, idxSize, _ := v.FileStat() + liveContent := int64(v.ContentSize() - v.DeletedSize()) + + dataBytes, indexBytes := compactionSpaceNeeded(v, 0) + + if dataBytes < liveContent+super_block.SuperBlockSize { + t.Fatalf("data estimate %d does not cover the %d live content bytes plus the superblock", dataBytes, liveContent) } - if estimate := estimatedCompactedSize(v); estimate != super_block.SuperBlockSize { - t.Fatalf("all-deleted estimate = %d, want superblock only (%d)", estimate, super_block.SuperBlockSize) + if indexBytes < 20*types.NeedleMapEntrySize { + t.Fatalf("index estimate %d does not cover 20 live entries", indexBytes) + } + if dataBytes+indexBytes >= int64(datSize+idxSize) { + t.Fatalf("space needed %d is not below the current volume size %d: a mostly-garbage volume must not require its own size to compact", dataBytes+indexBytes, datSize+idxSize) } } -// The estimate must cover what compaction writes on disk: each live needle's -// content plus its header, checksum, timestamp and padding. An all-live -// volume's compacted .dat is byte-for-byte its current one, so the estimate -// may not fall below the current file. -func TestEstimatedCompactedSizeCoversNeedleFraming(t *testing.T) { +func TestCompactionSpaceNeeded_AllGarbageNeedsAlmostNothing(t *testing.T) { + v := newVolumeWithGarbage(t, 0, 2000) + datSize, _, _ := v.FileStat() + + dataBytes, indexBytes := compactionSpaceNeeded(v, 0) + + if needed := dataBytes + indexBytes; needed > int64(datSize)/10 { + t.Fatalf("an all-garbage volume of %d bytes still asks for %d bytes", datSize, needed) + } +} + +func TestCompactionSpaceNeeded_NeverAboveCurrentVolume(t *testing.T) { + // Nothing deleted: the estimate may not exceed what is already on disk. + v := newVolumeWithGarbage(t, 200, 0) + datSize, idxSize, _ := v.FileStat() + + dataBytes, indexBytes := compactionSpaceNeeded(v, 0) + + if dataBytes > int64(datSize) || indexBytes > int64(idxSize) { + t.Fatalf("estimate data=%d index=%d exceeds the current files data=%d index=%d", dataBytes, indexBytes, datSize, idxSize) + } +} + +func TestCompactionSpaceNeeded_PreallocateWinsForDataOnly(t *testing.T) { + // The new .dat is preallocated to this size; the rebuilt index is a + // separate file and still needs its own room. + v := newVolumeWithGarbage(t, 5, 5) + const preallocate = int64(1) << 30 + + dataBytes, indexBytes := compactionSpaceNeeded(v, preallocate) + + if dataBytes != preallocate { + t.Fatalf("data estimate %d, want the preallocate size %d", dataBytes, preallocate) + } + if indexBytes < 5*types.NeedleMapEntrySize { + t.Fatalf("index estimate %d does not cover 5 live entries", indexBytes) + } +} + +func TestEnsureCompactVolumeSpace_FullDiskWithGarbage(t *testing.T) { + // The disk-full case from #11516: free space is far below the volume's + // size, but well above what compacting its live needles will write. + v := newVolumeWithGarbage(t, 20, 2000) + datSize, idxSize, _ := v.FileStat() + dataBytes, indexBytes := compactionSpaceNeeded(v, 0) + needed := dataBytes + indexBytes + if uint64(needed) >= datSize+idxSize { + t.Fatalf("test setup: estimate %d is not below volume size %d", needed, datSize+idxSize) + } + + stubCompactionDiskFree(t, uint64(needed)) + if err := ensureCompactVolumeSpace(v, 0); err != nil { + t.Fatalf("free %d covers the estimate %d, volume is %d: unexpected %v", needed, needed, datSize+idxSize, err) + } + + stubCompactionDiskFree(t, uint64(needed)-1) + err := ensureCompactVolumeSpace(v, 0) + if !errors.Is(err, ErrInsufficientSpace) { + t.Fatalf("free %d below the estimate %d: got %v, want ErrInsufficientSpace", needed-1, needed, err) + } +} + +func TestEnsureCompactVolumeSpace_SeparateIndexDisk(t *testing.T) { + dataDir, idxDir := t.TempDir(), t.TempDir() + v, err := NewVolume(dataDir, idxDir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0) + if err != nil { + t.Fatalf("NewVolume: %v", err) + } + t.Cleanup(v.Close) + for i := 1; i <= 50; i++ { + if _, _, _, err := v.writeNeedle2(newRandomNeedle(uint64(i)), true, false, false); err != nil { + t.Fatalf("write needle %d: %v", i, err) + } + } + dataBytes, indexBytes := compactionSpaceNeeded(v, 0) + + free := map[string]uint64{dataDir: uint64(dataBytes), idxDir: uint64(indexBytes)} + prevFree, prevSame := compactionDiskFree, compactionSameFilesystem + compactionDiskFree = func(dir string) uint64 { return free[dir] } + compactionSameFilesystem = func(string, string) bool { return false } + t.Cleanup(func() { compactionDiskFree, compactionSameFilesystem = prevFree, prevSame }) + + if err := ensureCompactVolumeSpace(v, 0); err != nil { + t.Fatalf("each disk covers its own share: unexpected %v", err) + } + // Plenty of room on the data disk cannot make up for a full index disk. + free[dataDir] = uint64(dataBytes) * 10 + free[idxDir] = uint64(indexBytes) - 1 + if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) { + t.Fatalf("full index disk: got %v, want ErrInsufficientSpace", err) + } + + // Two directories on one filesystem draw on the same free space, so the + // data and the index estimates must be covered together. + compactionSameFilesystem = func(string, string) bool { return true } + free[dataDir] = uint64(dataBytes+indexBytes) - 1 + free[idxDir] = uint64(dataBytes+indexBytes) - 1 + if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) { + t.Fatalf("shared filesystem short of the sum: got %v, want ErrInsufficientSpace", err) + } + free[dataDir] = uint64(dataBytes + indexBytes) + free[idxDir] = uint64(dataBytes + indexBytes) + if err := ensureCompactVolumeSpace(v, 0); err != nil { + t.Fatalf("shared filesystem covering the sum: unexpected %v", err) + } + // A mount point the identity check cannot see: the index directory + // reports its own, smaller pool and must still be checked. + free[idxDir] = uint64(indexBytes) - 1 + if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) { + t.Fatalf("index mount point short of the index: got %v, want ErrInsufficientSpace", err) + } +} + +func TestSameFilesystem(t *testing.T) { + dir := t.TempDir() + if !sameFilesystem(dir, dir) { + t.Fatal("a directory is on its own filesystem") + } + if !sameFilesystem(dir, filepath.Join(dir, "missing")) { + t.Fatal("an unreadable path must be treated as shared, so the check asks for the sum") + } +} + +// The estimate may not fall below the current .dat size: an all-live +// volume's compacted copy is byte-for-byte its current one. +func TestCompactionSpaceNeededCoversNeedleFraming(t *testing.T) { dir := t.TempDir() v, err := NewVolume(dir, dir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0) @@ -127,14 +201,13 @@ func TestEstimatedCompactedSizeCoversNeedleFraming(t *testing.T) { } datSize, _, _ := v.FileStat() - if estimate := estimatedCompactedSize(v); estimate < int64(datSize) { - t.Fatalf("estimate %d below .dat size %d for an all-live volume: missing per-needle framing", estimate, datSize) + dataBytes, _ := compactionSpaceNeeded(v, 0) + if dataBytes < int64(datSize) { + t.Fatalf("estimate %d below .dat size %d for an all-live volume: missing per-needle framing", dataBytes, datSize) } } -// disk_space_low is only reported when low space is the sole read-only cause, -// so a volume also marked read-only by an operator or quarantined by failed -// I/O stays out of the sweep. +// disk_space_low is only reported when low space is the sole read-only cause. func TestCheckCompactVolumeDiskLowSoleCauseOnly(t *testing.T) { dir := t.TempDir() store := newSingleDirStore(t, dir)