mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 06:22:05 +02:00
vacuum: check compaction space against live bytes, not volume size (#11524)
* vacuum: size the compaction space check by live bytes, not volume size ensureCompactVolumeSpace required the volume's current .dat and .idx size as free space before compacting. That is the size of the garbage, not of what compaction writes, so on a disk that filled up until its volumes went read-only every compaction was refused, including all-garbage volumes that would compact to a superblock and an empty index. The sweep then retried every volume each cycle and reclaimed nothing (issue #11516). Estimate the output from what the needle map already tracks: live content bytes plus a per-needle framing upper bound behind a superblock, and one index entry per live needle. The estimate never exceeds the current volume size and preallocate still wins when larger. Volumes whose deleted sizes are unknown (.sdx converted back to .idx) keep the whole volume as the estimate. The disk probe moves behind a package variable so the tests can stand in for a full disk; the tests build real volumes instead of re-implementing the formula. * vacuum: space check reserves the index on top of preallocate, checks a separate index disk Review follow-ups: preallocate only stands in for the new .dat, so the rebuilt index is added on top of it; with separate index directories the data disk is checked for the .cpd and the index disk for the .cpx; and the estimates carry 1/16 headroom because counters rebuilt from an index file pass through a Bloom filter with a 0.1% false positive rate. Neither estimate exceeds the current file. * vacuum: split the space check by filesystem, not by directory name Two directories can sit on one filesystem and share its free space, so the data and index estimates are checked separately only when the index directory is on another device; otherwise the sum must fit. Unknown is treated as shared. * vacuum: ask the index directory for its share even when it looks like the same filesystem A volume mounted under the data directory's drive letter on Windows has the same volume name, so the identity check calls it shared. Checking the index directory for the index estimate as well costs one statfs and catches a full index mount either way. * vacuum: identify a Windows volume by its GUID, not its path prefix A volume can be reached through a drive letter and through a folder it is mounted on, so filepath.VolumeName says nothing about the free-space pool. Resolve each directory to its mount point and compare the volume GUIDs; when that fails the two are treated as shared. * vacuum: keep the framing and disk_space_low coverage the rebase displaced * rust volume: split the compaction space check across data and index disks Mirror the Go check: estimate the new .dat and rebuilt .idx separately — live content plus per-needle framing capped at the current file, with preallocate standing in for the data file when larger — and check each directory against its own filesystem's free space. Two directories on one filesystem are asked for the sum. * vacuum: tighten comments on the compaction space check Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com> Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
6 files changed
+412
-149
No files matched your search
@@ -1553,32 +1553,40 @@ impl Store {
|
||||
vid: VolumeId,
|
||||
preallocate: u64,
|
||||
) -> Result<Option<CompactionJob>, VolumeError> {
|
||||
// Required space matches Go's CompactVolume check: the larger of the
|
||||
// requested preallocation and the estimated compacted size — the live
|
||||
// needles, not the .dat the garbage already occupies, so a full disk
|
||||
// can still be reclaimed.
|
||||
let (loc_idx, space_needed) = {
|
||||
let (loc_idx, v) = self
|
||||
// Mirrors Go's ensureCompactVolumeSpace, per filesystem.
|
||||
let (dir, dir_idx, data_bytes, index_bytes) = {
|
||||
let (_, v) = self
|
||||
.find_volume(vid)
|
||||
.ok_or(VolumeError::VolumeNotFound(vid))?;
|
||||
let live_count = (v.file_count() - v.deleted_count()).max(0) as u64;
|
||||
let live_bytes = v.content_size().saturating_sub(v.deleted_size());
|
||||
let per_needle = (get_actual_size(Size(0), v.version())
|
||||
+ NEEDLE_PADDING_SIZE as i64
|
||||
+ NEEDLE_MAP_ENTRY_SIZE as i64) as u64;
|
||||
let estimated = SUPER_BLOCK_SIZE as u64 + live_count * per_needle + live_bytes;
|
||||
let space_needed = std::cmp::max(preallocate, estimated);
|
||||
(loc_idx, space_needed + space_needed / 10)
|
||||
let (data_bytes, index_bytes) = compaction_space_needed(v, preallocate);
|
||||
(
|
||||
v.dir().to_string(),
|
||||
v.dir_idx().to_string(),
|
||||
data_bytes,
|
||||
index_bytes,
|
||||
)
|
||||
};
|
||||
|
||||
let dir = self.locations[loc_idx].directory.clone();
|
||||
let (_, free) = crate::storage::disk_location::get_disk_stats(&dir);
|
||||
if free < space_needed {
|
||||
return Err(VolumeError::InsufficientSpace {
|
||||
vid,
|
||||
required: space_needed,
|
||||
free,
|
||||
});
|
||||
let check = |dir: &str, needed: u64| -> Result<(), VolumeError> {
|
||||
let (_, free) = crate::storage::disk_location::get_disk_stats(dir);
|
||||
if free < needed {
|
||||
return Err(VolumeError::InsufficientSpace {
|
||||
vid,
|
||||
required: needed,
|
||||
free,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
};
|
||||
|
||||
if dir_idx.is_empty() || dir_idx == dir {
|
||||
check(&dir, data_bytes + index_bytes)?;
|
||||
} else if !same_filesystem(&dir, &dir_idx) {
|
||||
check(&dir, data_bytes)?;
|
||||
check(&dir_idx, index_bytes)?;
|
||||
} else {
|
||||
check(&dir, data_bytes + index_bytes)?;
|
||||
check(&dir_idx, index_bytes)?;
|
||||
}
|
||||
|
||||
let (_, v) = self
|
||||
@@ -1751,6 +1759,65 @@ fn owned_ec_shard_count(loc: &DiskLocation, vid: VolumeId, shard_ids: &[ShardId]
|
||||
.count()
|
||||
}
|
||||
|
||||
/// Mirrors Go's compactionSpaceNeeded: what compaction will write, split into
|
||||
/// the new .dat and rebuilt .idx shares, each capped at its current file.
|
||||
fn compaction_space_needed(v: &Volume, preallocate: u64) -> (u64, u64) {
|
||||
let mut data_bytes = v.current_dat_file_size().unwrap_or(0);
|
||||
let mut index_bytes = v.idx_file_size();
|
||||
|
||||
let live_count = v.file_count() - v.deleted_count();
|
||||
let live_content = v.content_size() as i64 - v.deleted_size() as i64;
|
||||
// Unknown or inconsistent deleted sizes: the whole volume stays the
|
||||
// estimate.
|
||||
let deleted_size_known = v.deleted_count() == 0 || v.deleted_size() > 0;
|
||||
if deleted_size_known && live_count >= 0 && live_content >= 0 {
|
||||
// Empty-needle framing plus a padding unit covers the worst case.
|
||||
let per_needle =
|
||||
(get_actual_size(Size(0), v.version()) + NEEDLE_PADDING_SIZE as i64) as u64;
|
||||
let estimate = with_headroom(
|
||||
SUPER_BLOCK_SIZE as u64 + live_content as u64 + live_count as u64 * per_needle,
|
||||
);
|
||||
if estimate < data_bytes {
|
||||
data_bytes = estimate;
|
||||
}
|
||||
let estimate = with_headroom(live_count as u64 * NEEDLE_MAP_ENTRY_SIZE as u64);
|
||||
if estimate < index_bytes {
|
||||
index_bytes = estimate;
|
||||
}
|
||||
}
|
||||
if preallocate > data_bytes {
|
||||
data_bytes = preallocate;
|
||||
}
|
||||
(data_bytes, index_bytes)
|
||||
}
|
||||
|
||||
/// Headroom for Bloom-filter false positives in the live/deleted counters.
|
||||
fn with_headroom(estimate: u64) -> u64 {
|
||||
estimate + estimate / 16
|
||||
}
|
||||
|
||||
/// Whether two directories draw on the same free-space pool; in doubt, yes.
|
||||
fn same_filesystem(a: &str, b: &str) -> bool {
|
||||
if a == b {
|
||||
return true;
|
||||
}
|
||||
same_filesystem_impl(a, b)
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn same_filesystem_impl(a: &str, b: &str) -> bool {
|
||||
use std::os::unix::fs::MetadataExt;
|
||||
match (std::fs::metadata(a), std::fs::metadata(b)) {
|
||||
(Ok(ma), Ok(mb)) => ma.dev() == mb.dev(),
|
||||
_ => true,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
fn same_filesystem_impl(_a: &str, _b: &str) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Tests
|
||||
// ============================================================================
|
||||
|
||||
@@ -1860,7 +1860,7 @@ impl Volume {
|
||||
self.dat_file.is_some() || self.remote_dat_file.is_some()
|
||||
}
|
||||
|
||||
fn current_dat_file_size(&self) -> io::Result<u64> {
|
||||
pub(crate) fn current_dat_file_size(&self) -> io::Result<u64> {
|
||||
if let Some(ref f) = self.dat_file {
|
||||
Ok(f.metadata()?.len())
|
||||
} else if let Some(ref remote_dat_file) = self.remote_dat_file {
|
||||
@@ -3950,6 +3950,11 @@ impl Volume {
|
||||
&self.dir
|
||||
}
|
||||
|
||||
/// Get the directory this volume's index is stored in.
|
||||
pub fn dir_idx(&self) -> &str {
|
||||
&self.dir_idx
|
||||
}
|
||||
|
||||
/// Throttle IO during compaction to avoid saturating disk.
|
||||
pub fn maybe_throttle_compaction(&self, bytes_written: u64) {
|
||||
if self.compaction_byte_per_second <= 0 || !self.is_compacting() {
|
||||
|
||||
@@ -61,50 +61,93 @@ func (s *Store) CommitCleanupVolume(vid needle.VolumeId) error {
|
||||
return fmt.Errorf("volume id %d is not found during cleaning up: %w", vid, ErrVolumeNotFound)
|
||||
}
|
||||
|
||||
// estimatedCompactedSize is what compaction writes: a superblock, the live
|
||||
// needles with their on-disk framing, and an index with live entries only.
|
||||
// Deleted bytes do not carry over, so a mostly-garbage volume needs far less
|
||||
// space than it occupies.
|
||||
func estimatedCompactedSize(v *Volume) int64 {
|
||||
liveCount := v.FileCount()
|
||||
if deleted := v.DeletedCount(); deleted < liveCount {
|
||||
liveCount -= deleted
|
||||
} else {
|
||||
liveCount = 0
|
||||
// compactionDiskFree reports the free bytes on the disk holding dir, and
|
||||
// compactionSameFilesystem whether two directories draw on the same pool.
|
||||
// Both are variables so a test can stand in for a full or a split disk.
|
||||
var (
|
||||
compactionDiskFree = func(dir string) uint64 {
|
||||
return stats.NewDiskStatus(dir).Free
|
||||
}
|
||||
liveBytes := v.ContentSize()
|
||||
if deleted := v.DeletedSize(); deleted < liveBytes {
|
||||
liveBytes -= deleted
|
||||
} else {
|
||||
liveBytes = 0
|
||||
compactionSameFilesystem = sameFilesystem
|
||||
)
|
||||
|
||||
// compactionSpaceNeeded estimates what CompactByIndex will write for v: a new
|
||||
// .dat holding the live needles with their on-disk framing behind a superblock
|
||||
// (or preallocate, when that is larger, since the file is preallocated to it),
|
||||
// and a rebuilt index with one entry per live needle. The volume's current size
|
||||
// is the wrong yardstick: the more garbage a volume holds, the less its
|
||||
// compaction writes, and a store that filled up until its volumes went
|
||||
// read-only is exactly where the all-garbage volumes must still compact to
|
||||
// give the space back (issue #11516). Neither estimate exceeds the current file.
|
||||
func compactionSpaceNeeded(v *Volume, preallocate int64) (dataBytes, indexBytes int64) {
|
||||
datSize, idxSize, _ := v.FileStat()
|
||||
dataBytes, indexBytes = int64(datSize), int64(idxSize)
|
||||
|
||||
liveCount := int64(v.FileCount()) - int64(v.DeletedCount())
|
||||
liveContent := int64(v.ContentSize()) - int64(v.DeletedSize())
|
||||
// A .sdx converted back to .idx carries no deleted sizes (see
|
||||
// garbageLevel), and counters that disagree mean the metric is off;
|
||||
// either way the whole volume stays the estimate.
|
||||
deletedSizeKnown := v.DeletedCount() == 0 || v.DeletedSize() > 0
|
||||
if deletedSizeKnown && liveCount >= 0 && liveContent >= 0 {
|
||||
// GetActualSize(0) is the framing of an empty needle; another
|
||||
// padding unit covers the worst case for any other size.
|
||||
perNeedle := needle.GetActualSize(0, v.Version()) + types.NeedlePaddingSize
|
||||
// Counters rebuilt from an index file (LevelDB and sorted maps) go
|
||||
// through a Bloom filter with a 0.1% false positive rate that can
|
||||
// count a live needle as deleted. A few percent of headroom covers
|
||||
// that many times over.
|
||||
if estimate := withHeadroom(super_block.SuperBlockSize + liveContent + liveCount*perNeedle); estimate < dataBytes {
|
||||
dataBytes = estimate
|
||||
}
|
||||
if estimate := withHeadroom(liveCount * types.NeedleMapEntrySize); estimate < indexBytes {
|
||||
indexBytes = estimate
|
||||
}
|
||||
}
|
||||
perNeedle := needle.GetActualSize(0, v.Version()) + types.NeedlePaddingSize + types.NeedleMapEntrySize
|
||||
return super_block.SuperBlockSize + int64(liveCount)*perNeedle + int64(liveBytes)
|
||||
if preallocate > dataBytes {
|
||||
dataBytes = preallocate
|
||||
}
|
||||
return dataBytes, indexBytes
|
||||
}
|
||||
|
||||
func withHeadroom(estimate int64) int64 {
|
||||
return estimate + estimate/16
|
||||
}
|
||||
|
||||
func ensureCompactVolumeSpace(v *Volume, preallocate int64) error {
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, preallocate)
|
||||
volumeSize, indexSize, _ := v.FileStat()
|
||||
|
||||
// The compacted output holds live needles only, so measure against the
|
||||
// estimated compacted size — otherwise a disk full of garbage can never
|
||||
// reclaim itself.
|
||||
estimatedCompactSize := estimatedCompactedSize(v)
|
||||
spaceNeeded := preallocate
|
||||
if estimatedCompactSize > preallocate {
|
||||
spaceNeeded = estimatedCompactSize
|
||||
check := func(dir string, needed int64) error {
|
||||
free := compactionDiskFree(dir)
|
||||
if int64(free) < needed {
|
||||
return fmt.Errorf("insufficient free space for compaction in %s: need %d bytes (data: %d, index: %d, current volume: %d, current index: %d), but only %d bytes available: %w",
|
||||
dir, needed, dataBytes, indexBytes, volumeSize, indexSize, free, ErrInsufficientSpace)
|
||||
}
|
||||
glog.V(1).Infof("volume %d compaction space check in %s: data=%d, index=%d, current volume=%d, space_needed=%d, free_space=%d",
|
||||
v.Id, dir, dataBytes, indexBytes, volumeSize, needed, free)
|
||||
return nil
|
||||
}
|
||||
spaceNeeded += spaceNeeded / 10
|
||||
|
||||
diskStatus := stats.NewDiskStatus(v.dir)
|
||||
if int64(diskStatus.Free) < spaceNeeded {
|
||||
return fmt.Errorf("insufficient free space for compaction: need %d bytes (volume: %d, index: %d), but only %d bytes available: %w",
|
||||
spaceNeeded, volumeSize, indexSize, diskStatus.Free, ErrInsufficientSpace)
|
||||
// The new .dat lands next to the old one and the new .idx next to the old
|
||||
// index. When the index directory is on another filesystem each disk
|
||||
// answers for its own share; two directories on one filesystem draw on
|
||||
// the same free space and must cover the sum.
|
||||
if v.dirIdx == "" || v.dirIdx == v.dir {
|
||||
return check(v.dir, dataBytes+indexBytes)
|
||||
}
|
||||
|
||||
glog.V(1).Infof("volume %d compaction space check: volume=%d, index=%d, space_needed=%d, free_space=%d",
|
||||
v.Id, volumeSize, indexSize, spaceNeeded, diskStatus.Free)
|
||||
|
||||
return nil
|
||||
if !compactionSameFilesystem(v.dir, v.dirIdx) {
|
||||
if err := check(v.dir, dataBytes); err != nil {
|
||||
return err
|
||||
}
|
||||
return check(v.dirIdx, indexBytes)
|
||||
}
|
||||
// Same filesystem as far as the identity check can tell. The index
|
||||
// directory is still asked for its own share, because a mount point the
|
||||
// check cannot see (a volume mounted under one drive letter on Windows)
|
||||
// would otherwise go unchecked.
|
||||
if err := check(v.dir, dataBytes+indexBytes); err != nil {
|
||||
return err
|
||||
}
|
||||
return check(v.dirIdx, indexBytes)
|
||||
}
|
||||
|
||||
func (s *Store) CompactVolumeFiles(vid needle.VolumeId, collection string, location *DiskLocation, needleMapKind NeedleMapKind, ldbTimeout int64, preallocate int64, compactionBytePerSecond int64) (err error) {
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
//go:build !windows
|
||||
|
||||
package storage
|
||||
|
||||
import (
|
||||
"os"
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// sameFilesystem reports whether two directories share one free-space pool.
|
||||
// When in doubt it says yes, which makes the space check ask for the sum.
|
||||
func sameFilesystem(a, b string) bool {
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
sa, errA := os.Stat(a)
|
||||
sb, errB := os.Stat(b)
|
||||
if errA != nil || errB != nil {
|
||||
return true
|
||||
}
|
||||
sta, okA := sa.Sys().(*syscall.Stat_t)
|
||||
stb, okB := sb.Sys().(*syscall.Stat_t)
|
||||
if !okA || !okB {
|
||||
return true
|
||||
}
|
||||
return sta.Dev == stb.Dev
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//go:build windows
|
||||
|
||||
package storage
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/sys/windows"
|
||||
)
|
||||
|
||||
// sameFilesystem reports whether two directories share one free-space pool.
|
||||
// On Windows a volume can be reached through a drive letter and through a
|
||||
// folder it is mounted on, so the path prefix says nothing; the volume GUID
|
||||
// behind each path does. When in doubt it says yes, which makes the space
|
||||
// check ask for the sum.
|
||||
func sameFilesystem(a, b string) bool {
|
||||
if a == b {
|
||||
return true
|
||||
}
|
||||
va, errA := volumeGUID(a)
|
||||
vb, errB := volumeGUID(b)
|
||||
if errA != nil || errB != nil {
|
||||
return true
|
||||
}
|
||||
return strings.EqualFold(va, vb)
|
||||
}
|
||||
|
||||
func volumeGUID(path string) (string, error) {
|
||||
abs, err := filepath.Abs(path)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
p, err := windows.UTF16PtrFromString(abs)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
mountPoint := make([]uint16, windows.MAX_LONG_PATH)
|
||||
if err := windows.GetVolumePathName(p, &mountPoint[0], uint32(len(mountPoint))); err != nil {
|
||||
return "", err
|
||||
}
|
||||
// \\?\Volume{GUID}\ is 49 characters plus the terminator.
|
||||
guid := make([]uint16, 50)
|
||||
if err := windows.GetVolumeNameForVolumeMountPoint(&mountPoint[0], &guid[0], uint32(len(guid))); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return windows.UTF16ToString(guid), nil
|
||||
}
|
||||
@@ -1,6 +1,8 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -10,108 +12,180 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
)
|
||||
|
||||
func TestSpaceCalculation(t *testing.T) {
|
||||
// Test the space calculation logic
|
||||
testCases := []struct {
|
||||
name string
|
||||
volumeSize uint64
|
||||
indexSize uint64
|
||||
preallocate int64
|
||||
expectedMin int64
|
||||
}{
|
||||
{
|
||||
name: "Large volume, small preallocate",
|
||||
volumeSize: 244 * 1024 * 1024 * 1024, // 244GB
|
||||
indexSize: 1024 * 1024, // 1MB
|
||||
preallocate: 1024, // 1KB
|
||||
expectedMin: int64((244*1024*1024*1024 + 1024*1024) * 11 / 10), // +10% buffer
|
||||
},
|
||||
{
|
||||
name: "Small volume, large preallocate",
|
||||
volumeSize: 100 * 1024 * 1024, // 100MB
|
||||
indexSize: 1024, // 1KB
|
||||
preallocate: 1024 * 1024 * 1024, // 1GB
|
||||
expectedMin: int64(1024 * 1024 * 1024 * 11 / 10), // preallocate + 10%
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// Calculate space needed using the same logic as our fix
|
||||
estimatedCompactSize := int64(tc.volumeSize + tc.indexSize)
|
||||
spaceNeeded := tc.preallocate
|
||||
if estimatedCompactSize > tc.preallocate {
|
||||
spaceNeeded = estimatedCompactSize
|
||||
}
|
||||
// Add 10% safety buffer
|
||||
spaceNeeded = spaceNeeded + (spaceNeeded / 10)
|
||||
|
||||
if spaceNeeded < tc.expectedMin {
|
||||
t.Errorf("Space calculation too low: got %d, expected at least %d", spaceNeeded, tc.expectedMin)
|
||||
}
|
||||
|
||||
t.Logf("Volume size: %d bytes, Space needed: %d bytes (%.2f%% of volume size)",
|
||||
tc.volumeSize, spaceNeeded, float64(spaceNeeded)/float64(tc.volumeSize)*100)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Compaction writes live needles only, so the space check must be measured
|
||||
// against the live size, not the .dat the garbage occupies — a full disk
|
||||
// needs the estimate to shrink or it can never reclaim.
|
||||
func TestEstimatedCompactedSizeCountsLiveNeedles(t *testing.T) {
|
||||
// newVolumeWithGarbage writes live+deleted needles and deletes the first
|
||||
// deleted ones, so the volume carries that much garbage on disk.
|
||||
func newVolumeWithGarbage(t *testing.T, live, deleted int) *Volume {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
|
||||
v, err := NewVolume(dir, dir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("volume creation: %v", err)
|
||||
t.Fatalf("NewVolume: %v", err)
|
||||
}
|
||||
defer v.Close()
|
||||
|
||||
const count = 20
|
||||
for i := 1; i <= count; i++ {
|
||||
t.Cleanup(v.Close)
|
||||
for i := 1; i <= live+deleted; i++ {
|
||||
if _, _, _, err := v.writeNeedle2(newRandomNeedle(uint64(i)), true, false, false); err != nil {
|
||||
t.Fatalf("write needle %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
datSize, _, _ := v.FileStat()
|
||||
|
||||
fullEstimate := estimatedCompactedSize(v)
|
||||
if fullEstimate <= super_block.SuperBlockSize {
|
||||
t.Fatalf("estimate for all-live volume = %d, want > superblock", fullEstimate)
|
||||
}
|
||||
|
||||
for i := 1; i < count; i++ {
|
||||
if _, err := v.doDeleteRequest(newEmptyNeedle(uint64(i))); err != nil {
|
||||
for i := 1; i <= deleted; i++ {
|
||||
if _, err := v.deleteNeedle2(newEmptyNeedle(uint64(i))); err != nil {
|
||||
t.Fatalf("delete needle %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
estimate := estimatedCompactedSize(v)
|
||||
if estimate >= int64(datSize) {
|
||||
t.Fatalf("estimate %d not below .dat size %d with 19/20 needles deleted", estimate, datSize)
|
||||
}
|
||||
if estimate <= super_block.SuperBlockSize {
|
||||
t.Fatalf("estimate %d lost the one live needle", estimate)
|
||||
}
|
||||
live := int64(v.FileCount()-v.DeletedCount())*types.NeedleMapEntrySize + super_block.SuperBlockSize
|
||||
if estimate < live {
|
||||
t.Fatalf("estimate %d below superblock + live index entries %d", estimate, live)
|
||||
}
|
||||
func stubCompactionDiskFree(t *testing.T, free uint64) {
|
||||
t.Helper()
|
||||
prev := compactionDiskFree
|
||||
compactionDiskFree = func(string) uint64 { return free }
|
||||
t.Cleanup(func() { compactionDiskFree = prev })
|
||||
}
|
||||
|
||||
if _, err := v.doDeleteRequest(newEmptyNeedle(uint64(count))); err != nil {
|
||||
t.Fatalf("delete last needle: %v", err)
|
||||
func TestCompactionSpaceNeeded_CountsLiveBytesNotVolumeSize(t *testing.T) {
|
||||
v := newVolumeWithGarbage(t, 20, 2000)
|
||||
datSize, idxSize, _ := v.FileStat()
|
||||
liveContent := int64(v.ContentSize() - v.DeletedSize())
|
||||
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, 0)
|
||||
|
||||
if dataBytes < liveContent+super_block.SuperBlockSize {
|
||||
t.Fatalf("data estimate %d does not cover the %d live content bytes plus the superblock", dataBytes, liveContent)
|
||||
}
|
||||
if estimate := estimatedCompactedSize(v); estimate != super_block.SuperBlockSize {
|
||||
t.Fatalf("all-deleted estimate = %d, want superblock only (%d)", estimate, super_block.SuperBlockSize)
|
||||
if indexBytes < 20*types.NeedleMapEntrySize {
|
||||
t.Fatalf("index estimate %d does not cover 20 live entries", indexBytes)
|
||||
}
|
||||
if dataBytes+indexBytes >= int64(datSize+idxSize) {
|
||||
t.Fatalf("space needed %d is not below the current volume size %d: a mostly-garbage volume must not require its own size to compact", dataBytes+indexBytes, datSize+idxSize)
|
||||
}
|
||||
}
|
||||
|
||||
// The estimate must cover what compaction writes on disk: each live needle's
|
||||
// content plus its header, checksum, timestamp and padding. An all-live
|
||||
// volume's compacted .dat is byte-for-byte its current one, so the estimate
|
||||
// may not fall below the current file.
|
||||
func TestEstimatedCompactedSizeCoversNeedleFraming(t *testing.T) {
|
||||
func TestCompactionSpaceNeeded_AllGarbageNeedsAlmostNothing(t *testing.T) {
|
||||
v := newVolumeWithGarbage(t, 0, 2000)
|
||||
datSize, _, _ := v.FileStat()
|
||||
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, 0)
|
||||
|
||||
if needed := dataBytes + indexBytes; needed > int64(datSize)/10 {
|
||||
t.Fatalf("an all-garbage volume of %d bytes still asks for %d bytes", datSize, needed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompactionSpaceNeeded_NeverAboveCurrentVolume(t *testing.T) {
|
||||
// Nothing deleted: the estimate may not exceed what is already on disk.
|
||||
v := newVolumeWithGarbage(t, 200, 0)
|
||||
datSize, idxSize, _ := v.FileStat()
|
||||
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, 0)
|
||||
|
||||
if dataBytes > int64(datSize) || indexBytes > int64(idxSize) {
|
||||
t.Fatalf("estimate data=%d index=%d exceeds the current files data=%d index=%d", dataBytes, indexBytes, datSize, idxSize)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompactionSpaceNeeded_PreallocateWinsForDataOnly(t *testing.T) {
|
||||
// The new .dat is preallocated to this size; the rebuilt index is a
|
||||
// separate file and still needs its own room.
|
||||
v := newVolumeWithGarbage(t, 5, 5)
|
||||
const preallocate = int64(1) << 30
|
||||
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, preallocate)
|
||||
|
||||
if dataBytes != preallocate {
|
||||
t.Fatalf("data estimate %d, want the preallocate size %d", dataBytes, preallocate)
|
||||
}
|
||||
if indexBytes < 5*types.NeedleMapEntrySize {
|
||||
t.Fatalf("index estimate %d does not cover 5 live entries", indexBytes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnsureCompactVolumeSpace_FullDiskWithGarbage(t *testing.T) {
|
||||
// The disk-full case from #11516: free space is far below the volume's
|
||||
// size, but well above what compacting its live needles will write.
|
||||
v := newVolumeWithGarbage(t, 20, 2000)
|
||||
datSize, idxSize, _ := v.FileStat()
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, 0)
|
||||
needed := dataBytes + indexBytes
|
||||
if uint64(needed) >= datSize+idxSize {
|
||||
t.Fatalf("test setup: estimate %d is not below volume size %d", needed, datSize+idxSize)
|
||||
}
|
||||
|
||||
stubCompactionDiskFree(t, uint64(needed))
|
||||
if err := ensureCompactVolumeSpace(v, 0); err != nil {
|
||||
t.Fatalf("free %d covers the estimate %d, volume is %d: unexpected %v", needed, needed, datSize+idxSize, err)
|
||||
}
|
||||
|
||||
stubCompactionDiskFree(t, uint64(needed)-1)
|
||||
err := ensureCompactVolumeSpace(v, 0)
|
||||
if !errors.Is(err, ErrInsufficientSpace) {
|
||||
t.Fatalf("free %d below the estimate %d: got %v, want ErrInsufficientSpace", needed-1, needed, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnsureCompactVolumeSpace_SeparateIndexDisk(t *testing.T) {
|
||||
dataDir, idxDir := t.TempDir(), t.TempDir()
|
||||
v, err := NewVolume(dataDir, idxDir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("NewVolume: %v", err)
|
||||
}
|
||||
t.Cleanup(v.Close)
|
||||
for i := 1; i <= 50; i++ {
|
||||
if _, _, _, err := v.writeNeedle2(newRandomNeedle(uint64(i)), true, false, false); err != nil {
|
||||
t.Fatalf("write needle %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
dataBytes, indexBytes := compactionSpaceNeeded(v, 0)
|
||||
|
||||
free := map[string]uint64{dataDir: uint64(dataBytes), idxDir: uint64(indexBytes)}
|
||||
prevFree, prevSame := compactionDiskFree, compactionSameFilesystem
|
||||
compactionDiskFree = func(dir string) uint64 { return free[dir] }
|
||||
compactionSameFilesystem = func(string, string) bool { return false }
|
||||
t.Cleanup(func() { compactionDiskFree, compactionSameFilesystem = prevFree, prevSame })
|
||||
|
||||
if err := ensureCompactVolumeSpace(v, 0); err != nil {
|
||||
t.Fatalf("each disk covers its own share: unexpected %v", err)
|
||||
}
|
||||
// Plenty of room on the data disk cannot make up for a full index disk.
|
||||
free[dataDir] = uint64(dataBytes) * 10
|
||||
free[idxDir] = uint64(indexBytes) - 1
|
||||
if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) {
|
||||
t.Fatalf("full index disk: got %v, want ErrInsufficientSpace", err)
|
||||
}
|
||||
|
||||
// Two directories on one filesystem draw on the same free space, so the
|
||||
// data and the index estimates must be covered together.
|
||||
compactionSameFilesystem = func(string, string) bool { return true }
|
||||
free[dataDir] = uint64(dataBytes+indexBytes) - 1
|
||||
free[idxDir] = uint64(dataBytes+indexBytes) - 1
|
||||
if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) {
|
||||
t.Fatalf("shared filesystem short of the sum: got %v, want ErrInsufficientSpace", err)
|
||||
}
|
||||
free[dataDir] = uint64(dataBytes + indexBytes)
|
||||
free[idxDir] = uint64(dataBytes + indexBytes)
|
||||
if err := ensureCompactVolumeSpace(v, 0); err != nil {
|
||||
t.Fatalf("shared filesystem covering the sum: unexpected %v", err)
|
||||
}
|
||||
// A mount point the identity check cannot see: the index directory
|
||||
// reports its own, smaller pool and must still be checked.
|
||||
free[idxDir] = uint64(indexBytes) - 1
|
||||
if err := ensureCompactVolumeSpace(v, 0); !errors.Is(err, ErrInsufficientSpace) {
|
||||
t.Fatalf("index mount point short of the index: got %v, want ErrInsufficientSpace", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameFilesystem(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
if !sameFilesystem(dir, dir) {
|
||||
t.Fatal("a directory is on its own filesystem")
|
||||
}
|
||||
if !sameFilesystem(dir, filepath.Join(dir, "missing")) {
|
||||
t.Fatal("an unreadable path must be treated as shared, so the check asks for the sum")
|
||||
}
|
||||
}
|
||||
|
||||
// The estimate may not fall below the current .dat size: an all-live
|
||||
// volume's compacted copy is byte-for-byte its current one.
|
||||
func TestCompactionSpaceNeededCoversNeedleFraming(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
|
||||
v, err := NewVolume(dir, dir, "", 1, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
|
||||
@@ -127,14 +201,13 @@ func TestEstimatedCompactedSizeCoversNeedleFraming(t *testing.T) {
|
||||
}
|
||||
datSize, _, _ := v.FileStat()
|
||||
|
||||
if estimate := estimatedCompactedSize(v); estimate < int64(datSize) {
|
||||
t.Fatalf("estimate %d below .dat size %d for an all-live volume: missing per-needle framing", estimate, datSize)
|
||||
dataBytes, _ := compactionSpaceNeeded(v, 0)
|
||||
if dataBytes < int64(datSize) {
|
||||
t.Fatalf("estimate %d below .dat size %d for an all-live volume: missing per-needle framing", dataBytes, datSize)
|
||||
}
|
||||
}
|
||||
|
||||
// disk_space_low is only reported when low space is the sole read-only cause,
|
||||
// so a volume also marked read-only by an operator or quarantined by failed
|
||||
// I/O stays out of the sweep.
|
||||
// disk_space_low is only reported when low space is the sole read-only cause.
|
||||
func TestCheckCompactVolumeDiskLowSoleCauseOnly(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
store := newSingleDirStore(t, dir)
|
||||
|
||||
Reference in new issue
Block a user