mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-11 16:57:45 +02:00
storage: unify needle-map counters across index loaders and warn on copy-count drift (#11701)
* storage: count index-load metrics only for live-key removals The in-memory and leveldb offset loaders counted a deletion for every tombstone row replayed, including re-deletes of already-deleted keys (runtime logDelete only counts a live removal), and added the returned old size to the byte counters even when it was a tombstone sentinel — uint64(-1) wraps DeletionByteCounter. The leveldb offset loader also counted every index row as a file and its bytes unconditionally. Gate the counters the same way the runtime paths do so a reload reports the same numbers the runtime counters hold. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * storage: derive index metrics from latest-row state in the metric loader The leveldb metric loader walked the index in reverse and incremented FileCounter for every row (inflating FileCount with tombstones) and DeletionCounter for every superseded row plus every deleted-key row (double counting both). Per key, the runtime deleted count is exactly (valid rows) - (1 when the latest row is live), so count live-latest rows via the same bloom filter and derive both deletion counters; this matches the runtime logPut/logDelete counters exactly instead of drifting with tombstone history. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * server: warn instead of failing volume copies on counter drift checkCopyCounts compared the source's in-memory counters against the target's fresh-load counters, but counters computed by different loaders drifted apart for identical .idx data, so a good copy was rejected (issue #11678). The file-size checks already gate integrity; log the mismatch instead of failing the copy. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * storage: assert MaxFileKey parity in the metric loader test MaxFileKey derives from row order alone, so unlike the bloom-filtered deletion counters it must always match the runtime counters exactly. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
1 parent
5f1a742726
commit
4bf5e92b64
6 files changed
+99
-37
No files matched your search
@@ -259,7 +259,13 @@ func (vs *VolumeServer) VolumeCopy(req *volume_server_pb.VolumeCopyRequest, stre
|
||||
if !shouldValidateCopyCounts {
|
||||
return nil
|
||||
}
|
||||
return checkCopyCounts(sourceVolumeStatusAfterCopy, targetVolume.FileCount(), targetVolume.DeletedCount())
|
||||
// Counter drift (bloom-filter estimates, rows appended between the
|
||||
// status snapshots) is not proof of a bad copy; the byte sizes
|
||||
// already match. Warn instead of rejecting the copy.
|
||||
if countErr := checkCopyCounts(sourceVolumeStatusAfterCopy, targetVolume.FileCount(), targetVolume.DeletedCount()); countErr != nil {
|
||||
glog.Warningf("copied volume %d counts differ from source in-memory counters (counter drift): %v", req.VolumeId, countErr)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to mount or validate volume %d: %w", req.VolumeId, err)
|
||||
@@ -302,7 +308,7 @@ func (vs *VolumeServer) doCopyFileWithThrottler(client volume_server_pb.VolumeSe
|
||||
|
||||
}
|
||||
|
||||
// checkCopyFiles verifies the copied file sizes. Record counts are checked
|
||||
// checkCopyFiles verifies the copied file sizes. Record counts are compared
|
||||
// after the target volume is mounted, when the target needle map is available.
|
||||
func checkCopyFiles(originFileInf *volume_server_pb.ReadVolumeFileStatusResponse, hasRemoteDatFile bool, idxFileName, datFileName string) error {
|
||||
stat, err := os.Stat(idxFileName)
|
||||
|
||||
@@ -396,13 +396,18 @@ func (m *LevelDbNeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startF
|
||||
}
|
||||
|
||||
err = idx.WalkIndexFile(indexFile, startFrom, func(key NeedleId, offset Offset, size Size) (e error) {
|
||||
m.mapMetric.FileCounter++
|
||||
m.mapMetric.MaybeSetMaxNeedleEnd(offset, size, version)
|
||||
bytes := make([]byte, NeedleIdSize)
|
||||
NeedleIdToBytes(bytes[0:NeedleIdSize], key)
|
||||
validRow := !offset.IsZero() && !size.IsDeleted()
|
||||
if validRow {
|
||||
m.mapMetric.FileCounter++
|
||||
}
|
||||
// fresh loading
|
||||
if startFrom == 0 {
|
||||
m.mapMetric.FileByteCounter += uint64(size)
|
||||
if validRow {
|
||||
m.mapMetric.FileByteCounter += uint64(size)
|
||||
}
|
||||
e = levelDbWrite(db, key, offset, size, false, 0)
|
||||
return e
|
||||
}
|
||||
@@ -414,24 +419,28 @@ func (m *LevelDbNeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startF
|
||||
return err
|
||||
}
|
||||
// new needle, unlikely happen
|
||||
m.mapMetric.FileByteCounter += uint64(size)
|
||||
if validRow {
|
||||
m.mapMetric.FileByteCounter += uint64(size)
|
||||
}
|
||||
e = levelDbWrite(db, key, offset, size, false, 0)
|
||||
} else {
|
||||
// needle is found
|
||||
oldSize := BytesToSize(data[OffsetSize : OffsetSize+SizeSize])
|
||||
oldOffset := BytesToOffset(data[0:OffsetSize])
|
||||
if !offset.IsZero() && !size.IsDeleted() {
|
||||
if validRow {
|
||||
// updated needle
|
||||
m.mapMetric.FileByteCounter += uint64(size)
|
||||
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
|
||||
if !oldOffset.IsZero() && oldSize.IsValid() {
|
||||
m.mapMetric.DeletionCounter++
|
||||
m.mapMetric.DeletionByteCounter += uint64(oldSize)
|
||||
}
|
||||
e = levelDbWrite(db, key, offset, size, false, 0)
|
||||
} else {
|
||||
// deleted needle
|
||||
m.mapMetric.DeletionCounter++
|
||||
m.mapMetric.DeletionByteCounter += uint64(oldSize)
|
||||
if oldSize > 0 {
|
||||
m.mapMetric.DeletionCounter++
|
||||
m.mapMetric.DeletionByteCounter += uint64(oldSize)
|
||||
}
|
||||
e = levelDbDelete(db, key)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,14 +42,16 @@ func doLoading(file *os.File, nm *NeedleMap, version needle.Version) (*NeedleMap
|
||||
nm.FileCounter++
|
||||
nm.FileByteCounter = nm.FileByteCounter + uint64(size)
|
||||
oldOffset, oldSize := nm.m.Set(NeedleId(key), offset, size)
|
||||
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
|
||||
if !oldOffset.IsZero() && oldSize.IsValid() {
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
}
|
||||
} else {
|
||||
oldSize := nm.m.Delete(NeedleId(key))
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
if oldSize > 0 {
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
})
|
||||
@@ -126,18 +128,20 @@ func (nm *NeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startFrom ui
|
||||
e := idx.WalkIndexFile(indexFile, startFrom, func(key NeedleId, offset Offset, size Size) error {
|
||||
nm.MaybeSetMaxFileKey(key)
|
||||
nm.MaybeSetMaxNeedleEnd(offset, size, version)
|
||||
nm.FileCounter++
|
||||
if !offset.IsZero() && !size.IsDeleted() {
|
||||
nm.FileCounter++
|
||||
nm.FileByteCounter = nm.FileByteCounter + uint64(size)
|
||||
oldOffset, oldSize := nm.m.Set(NeedleId(key), offset, size)
|
||||
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
|
||||
if !oldOffset.IsZero() && oldSize.IsValid() {
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
}
|
||||
} else {
|
||||
oldSize := nm.m.Delete(NeedleId(key))
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
if oldSize > 0 {
|
||||
nm.DeletionCounter++
|
||||
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
})
|
||||
|
||||
@@ -151,6 +151,8 @@ func (mm *mapMetric) MaxNeedleEnd() int64 {
|
||||
func needleMapMetricFromIndexFile(r *os.File, mm *mapMetric, version needle.Version) error {
|
||||
var bf *boom.BloomFilter
|
||||
buf := make([]byte, NeedleIdSize)
|
||||
var liveKeyCount uint32
|
||||
var liveKeyBytes uint64
|
||||
err := reverseWalkIndexFile(r, func(entryCount int64) {
|
||||
bf = boom.NewBloomFilter(uint(entryCount), 0.001)
|
||||
}, func(key NeedleId, offset Offset, size Size) error {
|
||||
@@ -158,27 +160,27 @@ func needleMapMetricFromIndexFile(r *os.File, mm *mapMetric, version needle.Vers
|
||||
mm.MaybeSetMaxFileKey(key)
|
||||
mm.MaybeSetMaxNeedleEnd(offset, size, version)
|
||||
NeedleIdToBytes(buf, key)
|
||||
if size.IsValid() {
|
||||
mm.FileByteCounter += uint64(size)
|
||||
seen := bf.TestAndAdd(buf)
|
||||
if offset.IsZero() || size.IsDeleted() {
|
||||
return nil
|
||||
}
|
||||
|
||||
mm.FileCounter++
|
||||
if !bf.TestAndAdd(buf) {
|
||||
// if !size.IsValid(), then this file is deleted already
|
||||
if !size.IsValid() {
|
||||
mm.DeletionCounter++
|
||||
}
|
||||
} else {
|
||||
// deleted file
|
||||
mm.DeletionCounter++
|
||||
if size.IsValid() {
|
||||
// previously already deleted file
|
||||
mm.DeletionByteCounter += uint64(size)
|
||||
}
|
||||
mm.FileByteCounter += uint64(size)
|
||||
if !seen {
|
||||
liveKeyCount++
|
||||
liveKeyBytes += uint64(size)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return err
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Runtime counters tally each live-key removal: an overwrite deletes the
|
||||
// superseded row and a tombstone deletes the live row. Replaying forward,
|
||||
// that is (file rows) - (keys whose latest row is live).
|
||||
mm.DeletionCounter = mm.FileCounter - liveKeyCount
|
||||
mm.DeletionByteCounter = mm.FileByteCounter - liveKeyBytes
|
||||
return nil
|
||||
}
|
||||
|
||||
func newNeedleMapMetricFromIndexFile(r *os.File, version needle.Version) (mm *mapMetric, err error) {
|
||||
|
||||
@@ -16,17 +16,38 @@ func TestFastLoadingNeedleMapMetrics(t *testing.T) {
|
||||
nm := NewCompactNeedleMap(idxFile)
|
||||
|
||||
for i := 0; i < 10000; i++ {
|
||||
nm.Put(Uint64ToNeedleId(uint64(i+1)), Uint32ToOffset(uint32(0)), Size(1))
|
||||
nm.Put(Uint64ToNeedleId(uint64(i+1)), Uint32ToOffset(uint32(i+1)), Size(1))
|
||||
if rand.Float32() < 0.2 && i > 0 {
|
||||
nm.Delete(Uint64ToNeedleId(uint64(rand.Int63n(int64(i))+1)), Uint32ToOffset(uint32(0)))
|
||||
}
|
||||
}
|
||||
|
||||
mm, _ := newNeedleMapMetricFromIndexFile(idxFile, needle.GetCurrentVersion())
|
||||
mm, err := newNeedleMapMetricFromIndexFile(idxFile, needle.GetCurrentVersion())
|
||||
if err != nil {
|
||||
t.Fatalf("newNeedleMapMetricFromIndexFile() error = %v", err)
|
||||
}
|
||||
|
||||
glog.V(0).Infof("FileCount expected %d actual %d", nm.FileCount(), mm.FileCount())
|
||||
glog.V(0).Infof("DeletedSize expected %d actual %d", nm.DeletedSize(), mm.DeletedSize())
|
||||
glog.V(0).Infof("ContentSize expected %d actual %d", nm.ContentSize(), mm.ContentSize())
|
||||
glog.V(0).Infof("DeletedCount expected %d actual %d", nm.DeletedCount(), mm.DeletedCount())
|
||||
glog.V(0).Infof("MaxFileKey expected %d actual %d", nm.MaxFileKey(), mm.MaxFileKey())
|
||||
|
||||
if mm.FileCount() != nm.FileCount() {
|
||||
t.Fatalf("FileCount = %d, want %d", mm.FileCount(), nm.FileCount())
|
||||
}
|
||||
if mm.ContentSize() != nm.ContentSize() {
|
||||
t.Fatalf("ContentSize = %d, want %d", mm.ContentSize(), nm.ContentSize())
|
||||
}
|
||||
if mm.MaxFileKey() != nm.MaxFileKey() {
|
||||
t.Fatalf("MaxFileKey = %d, want %d", mm.MaxFileKey(), nm.MaxFileKey())
|
||||
}
|
||||
// Bloom false positives can hide a key whose latest row is live, which
|
||||
// only inflates the deletion counters by a small bounded amount.
|
||||
if got, want := mm.DeletedCount(), nm.DeletedCount(); got < want || got > want+256 {
|
||||
t.Fatalf("DeletedCount = %d, want within [%d, %d]", got, want, want+256)
|
||||
}
|
||||
if got, want := mm.DeletedSize(), nm.DeletedSize(); got < want || got > want+256 {
|
||||
t.Fatalf("DeletedSize = %d, want within [%d, %d]", got, want, want+256)
|
||||
}
|
||||
}
|
||||
@@ -105,6 +105,7 @@ func testCompactionByIndex(t *testing.T, needleMapKind NeedleMapKind) {
|
||||
}
|
||||
v.CommitCompact()
|
||||
realRecordCount := v.nm.IndexFileSize() / types.NeedleMapEntrySize
|
||||
liveRowCount := uint64(countLiveIndexRows(t, v.FileName(".idx")))
|
||||
if needleMapKind == NeedleMapLevelDb {
|
||||
nm := reflect.ValueOf(v.nm).Interface().(*LevelDbNeedleMap)
|
||||
mm := nm.mapMetric
|
||||
@@ -115,10 +116,10 @@ func testCompactionByIndex(t *testing.T, needleMapKind NeedleMapKind) {
|
||||
t.Fatalf("testing watermark failed")
|
||||
}
|
||||
} else {
|
||||
t.Logf("realRecordCount:%d, v.FileCount():%d mm.DeletedCount():%d", realRecordCount, v.FileCount(), v.DeletedCount())
|
||||
t.Logf("realRecordCount:%d, liveRowCount:%d, v.FileCount():%d mm.DeletedCount():%d", realRecordCount, liveRowCount, v.FileCount(), v.DeletedCount())
|
||||
}
|
||||
if realRecordCount != v.FileCount() {
|
||||
t.Fatalf("testing file count failed")
|
||||
if liveRowCount != v.FileCount() {
|
||||
t.Fatalf("testing file count failed: live idx rows %d, FileCount %d", liveRowCount, v.FileCount())
|
||||
}
|
||||
|
||||
v.Close()
|
||||
@@ -566,6 +567,25 @@ func TestExceedsExpectedCompactedSize(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func countLiveIndexRows(t *testing.T, indexFileName string) int {
|
||||
t.Helper()
|
||||
f, err := os.Open(indexFileName)
|
||||
if err != nil {
|
||||
t.Fatalf("open index file %s: %v", indexFileName, err)
|
||||
}
|
||||
defer f.Close()
|
||||
count := 0
|
||||
if err := idx.WalkIndexFile(f, 0, func(key types.NeedleId, offset types.Offset, size types.Size) error {
|
||||
if !offset.IsZero() && !size.IsDeleted() {
|
||||
count++
|
||||
}
|
||||
return nil
|
||||
}); err != nil {
|
||||
t.Fatalf("walk index file %s: %v", indexFileName, err)
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
func doSomeWritesDeletes(i int, v *Volume, t *testing.T, infos []*needleInfo) {
|
||||
n := newRandomNeedle(uint64(i))
|
||||
_, size, _, err := v.writeNeedle2(n, true, false, false)
|
||||
|
||||
Reference in new issue
Block a user