storage: unify needle-map counters across index loaders and warn on copy-count drift (#11701)

* storage: count index-load metrics only for live-key removals

The in-memory and leveldb offset loaders counted a deletion for every
tombstone row replayed, including re-deletes of already-deleted keys
(runtime logDelete only counts a live removal), and added the returned
old size to the byte counters even when it was a tombstone sentinel —
uint64(-1) wraps DeletionByteCounter. The leveldb offset loader also
counted every index row as a file and its bytes unconditionally. Gate
the counters the same way the runtime paths do so a reload reports the
same numbers the runtime counters hold.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* storage: derive index metrics from latest-row state in the metric loader

The leveldb metric loader walked the index in reverse and incremented
FileCounter for every row (inflating FileCount with tombstones) and
DeletionCounter for every superseded row plus every deleted-key row
(double counting both). Per key, the runtime deleted count is exactly
(valid rows) - (1 when the latest row is live), so count live-latest
rows via the same bloom filter and derive both deletion counters; this
matches the runtime logPut/logDelete counters exactly instead of
drifting with tombstone history.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* server: warn instead of failing volume copies on counter drift

checkCopyCounts compared the source's in-memory counters against the
target's fresh-load counters, but counters computed by different
loaders drifted apart for identical .idx data, so a good copy was
rejected (issue #11678). The file-size checks already gate integrity;
log the mismatch instead of failing the copy.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* storage: assert MaxFileKey parity in the metric loader test

MaxFileKey derives from row order alone, so unlike the bloom-filtered
deletion counters it must always match the runtime counters exactly.

Generated with [Devin](https://devin.ai)

Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Chris LuandDevin authored and GitHub committed 2026-10-10 22:31:46 +08:00
1 parent 5f1a742726
commit 4bf5e92b64
6 files changed
+99 -37

No files matched your search

+8 -2
View File
@@ -259,7 +259,13 @@ func (vs *VolumeServer) VolumeCopy(req *volume_server_pb.VolumeCopyRequest, stre
if !shouldValidateCopyCounts {
return nil
}
return checkCopyCounts(sourceVolumeStatusAfterCopy, targetVolume.FileCount(), targetVolume.DeletedCount())
// Counter drift (bloom-filter estimates, rows appended between the
// status snapshots) is not proof of a bad copy; the byte sizes
// already match. Warn instead of rejecting the copy.
if countErr := checkCopyCounts(sourceVolumeStatusAfterCopy, targetVolume.FileCount(), targetVolume.DeletedCount()); countErr != nil {
glog.Warningf("copied volume %d counts differ from source in-memory counters (counter drift): %v", req.VolumeId, countErr)
}
return nil
})
if err != nil {
return fmt.Errorf("failed to mount or validate volume %d: %w", req.VolumeId, err)
@@ -302,7 +308,7 @@ func (vs *VolumeServer) doCopyFileWithThrottler(client volume_server_pb.VolumeSe
}
// checkCopyFiles verifies the copied file sizes. Record counts are checked
// checkCopyFiles verifies the copied file sizes. Record counts are compared
// after the target volume is mounted, when the target needle map is available.
func checkCopyFiles(originFileInf *volume_server_pb.ReadVolumeFileStatusResponse, hasRemoteDatFile bool, idxFileName, datFileName string) error {
stat, err := os.Stat(idxFileName)
+16 -7
View File
@@ -396,13 +396,18 @@ func (m *LevelDbNeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startF
}
err = idx.WalkIndexFile(indexFile, startFrom, func(key NeedleId, offset Offset, size Size) (e error) {
m.mapMetric.FileCounter++
m.mapMetric.MaybeSetMaxNeedleEnd(offset, size, version)
bytes := make([]byte, NeedleIdSize)
NeedleIdToBytes(bytes[0:NeedleIdSize], key)
validRow := !offset.IsZero() && !size.IsDeleted()
if validRow {
m.mapMetric.FileCounter++
}
// fresh loading
if startFrom == 0 {
m.mapMetric.FileByteCounter += uint64(size)
if validRow {
m.mapMetric.FileByteCounter += uint64(size)
}
e = levelDbWrite(db, key, offset, size, false, 0)
return e
}
@@ -414,24 +419,28 @@ func (m *LevelDbNeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startF
return err
}
// new needle, unlikely happen
m.mapMetric.FileByteCounter += uint64(size)
if validRow {
m.mapMetric.FileByteCounter += uint64(size)
}
e = levelDbWrite(db, key, offset, size, false, 0)
} else {
// needle is found
oldSize := BytesToSize(data[OffsetSize : OffsetSize+SizeSize])
oldOffset := BytesToOffset(data[0:OffsetSize])
if !offset.IsZero() && !size.IsDeleted() {
if validRow {
// updated needle
m.mapMetric.FileByteCounter += uint64(size)
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
if !oldOffset.IsZero() && oldSize.IsValid() {
m.mapMetric.DeletionCounter++
m.mapMetric.DeletionByteCounter += uint64(oldSize)
}
e = levelDbWrite(db, key, offset, size, false, 0)
} else {
// deleted needle
m.mapMetric.DeletionCounter++
m.mapMetric.DeletionByteCounter += uint64(oldSize)
if oldSize > 0 {
m.mapMetric.DeletionCounter++
m.mapMetric.DeletionByteCounter += uint64(oldSize)
}
e = levelDbDelete(db, key)
}
}
+11 -7
View File
@@ -42,14 +42,16 @@ func doLoading(file *os.File, nm *NeedleMap, version needle.Version) (*NeedleMap
nm.FileCounter++
nm.FileByteCounter = nm.FileByteCounter + uint64(size)
oldOffset, oldSize := nm.m.Set(NeedleId(key), offset, size)
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
if !oldOffset.IsZero() && oldSize.IsValid() {
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
}
} else {
oldSize := nm.m.Delete(NeedleId(key))
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
if oldSize > 0 {
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
}
}
return nil
})
@@ -126,18 +128,20 @@ func (nm *NeedleMap) DoOffsetLoading(v *Volume, indexFile *os.File, startFrom ui
e := idx.WalkIndexFile(indexFile, startFrom, func(key NeedleId, offset Offset, size Size) error {
nm.MaybeSetMaxFileKey(key)
nm.MaybeSetMaxNeedleEnd(offset, size, version)
nm.FileCounter++
if !offset.IsZero() && !size.IsDeleted() {
nm.FileCounter++
nm.FileByteCounter = nm.FileByteCounter + uint64(size)
oldOffset, oldSize := nm.m.Set(NeedleId(key), offset, size)
if !oldOffset.IsZero() && !oldSize.IsDeleted() {
if !oldOffset.IsZero() && oldSize.IsValid() {
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
}
} else {
oldSize := nm.m.Delete(NeedleId(key))
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
if oldSize > 0 {
nm.DeletionCounter++
nm.DeletionByteCounter = nm.DeletionByteCounter + uint64(oldSize)
}
}
return nil
})
+18 -16
View File
@@ -151,6 +151,8 @@ func (mm *mapMetric) MaxNeedleEnd() int64 {
func needleMapMetricFromIndexFile(r *os.File, mm *mapMetric, version needle.Version) error {
var bf *boom.BloomFilter
buf := make([]byte, NeedleIdSize)
var liveKeyCount uint32
var liveKeyBytes uint64
err := reverseWalkIndexFile(r, func(entryCount int64) {
bf = boom.NewBloomFilter(uint(entryCount), 0.001)
}, func(key NeedleId, offset Offset, size Size) error {
@@ -158,27 +160,27 @@ func needleMapMetricFromIndexFile(r *os.File, mm *mapMetric, version needle.Vers
mm.MaybeSetMaxFileKey(key)
mm.MaybeSetMaxNeedleEnd(offset, size, version)
NeedleIdToBytes(buf, key)
if size.IsValid() {
mm.FileByteCounter += uint64(size)
seen := bf.TestAndAdd(buf)
if offset.IsZero() || size.IsDeleted() {
return nil
}
mm.FileCounter++
if !bf.TestAndAdd(buf) {
// if !size.IsValid(), then this file is deleted already
if !size.IsValid() {
mm.DeletionCounter++
}
} else {
// deleted file
mm.DeletionCounter++
if size.IsValid() {
// previously already deleted file
mm.DeletionByteCounter += uint64(size)
}
mm.FileByteCounter += uint64(size)
if !seen {
liveKeyCount++
liveKeyBytes += uint64(size)
}
return nil
})
return err
if err != nil {
return err
}
// Runtime counters tally each live-key removal: an overwrite deletes the
// superseded row and a tombstone deletes the live row. Replaying forward,
// that is (file rows) - (keys whose latest row is live).
mm.DeletionCounter = mm.FileCounter - liveKeyCount
mm.DeletionByteCounter = mm.FileByteCounter - liveKeyBytes
return nil
}
func newNeedleMapMetricFromIndexFile(r *os.File, version needle.Version) (mm *mapMetric, err error) {
+23 -2
View File
@@ -16,17 +16,38 @@ func TestFastLoadingNeedleMapMetrics(t *testing.T) {
nm := NewCompactNeedleMap(idxFile)
for i := 0; i < 10000; i++ {
nm.Put(Uint64ToNeedleId(uint64(i+1)), Uint32ToOffset(uint32(0)), Size(1))
nm.Put(Uint64ToNeedleId(uint64(i+1)), Uint32ToOffset(uint32(i+1)), Size(1))
if rand.Float32() < 0.2 && i > 0 {
nm.Delete(Uint64ToNeedleId(uint64(rand.Int63n(int64(i))+1)), Uint32ToOffset(uint32(0)))
}
}
mm, _ := newNeedleMapMetricFromIndexFile(idxFile, needle.GetCurrentVersion())
mm, err := newNeedleMapMetricFromIndexFile(idxFile, needle.GetCurrentVersion())
if err != nil {
t.Fatalf("newNeedleMapMetricFromIndexFile() error = %v", err)
}
glog.V(0).Infof("FileCount expected %d actual %d", nm.FileCount(), mm.FileCount())
glog.V(0).Infof("DeletedSize expected %d actual %d", nm.DeletedSize(), mm.DeletedSize())
glog.V(0).Infof("ContentSize expected %d actual %d", nm.ContentSize(), mm.ContentSize())
glog.V(0).Infof("DeletedCount expected %d actual %d", nm.DeletedCount(), mm.DeletedCount())
glog.V(0).Infof("MaxFileKey expected %d actual %d", nm.MaxFileKey(), mm.MaxFileKey())
if mm.FileCount() != nm.FileCount() {
t.Fatalf("FileCount = %d, want %d", mm.FileCount(), nm.FileCount())
}
if mm.ContentSize() != nm.ContentSize() {
t.Fatalf("ContentSize = %d, want %d", mm.ContentSize(), nm.ContentSize())
}
if mm.MaxFileKey() != nm.MaxFileKey() {
t.Fatalf("MaxFileKey = %d, want %d", mm.MaxFileKey(), nm.MaxFileKey())
}
// Bloom false positives can hide a key whose latest row is live, which
// only inflates the deletion counters by a small bounded amount.
if got, want := mm.DeletedCount(), nm.DeletedCount(); got < want || got > want+256 {
t.Fatalf("DeletedCount = %d, want within [%d, %d]", got, want, want+256)
}
if got, want := mm.DeletedSize(), nm.DeletedSize(); got < want || got > want+256 {
t.Fatalf("DeletedSize = %d, want within [%d, %d]", got, want, want+256)
}
}
+23 -3
View File
@@ -105,6 +105,7 @@ func testCompactionByIndex(t *testing.T, needleMapKind NeedleMapKind) {
}
v.CommitCompact()
realRecordCount := v.nm.IndexFileSize() / types.NeedleMapEntrySize
liveRowCount := uint64(countLiveIndexRows(t, v.FileName(".idx")))
if needleMapKind == NeedleMapLevelDb {
nm := reflect.ValueOf(v.nm).Interface().(*LevelDbNeedleMap)
mm := nm.mapMetric
@@ -115,10 +116,10 @@ func testCompactionByIndex(t *testing.T, needleMapKind NeedleMapKind) {
t.Fatalf("testing watermark failed")
}
} else {
t.Logf("realRecordCount:%d, v.FileCount():%d mm.DeletedCount():%d", realRecordCount, v.FileCount(), v.DeletedCount())
t.Logf("realRecordCount:%d, liveRowCount:%d, v.FileCount():%d mm.DeletedCount():%d", realRecordCount, liveRowCount, v.FileCount(), v.DeletedCount())
}
if realRecordCount != v.FileCount() {
t.Fatalf("testing file count failed")
if liveRowCount != v.FileCount() {
t.Fatalf("testing file count failed: live idx rows %d, FileCount %d", liveRowCount, v.FileCount())
}
v.Close()
@@ -566,6 +567,25 @@ func TestExceedsExpectedCompactedSize(t *testing.T) {
}
}
func countLiveIndexRows(t *testing.T, indexFileName string) int {
t.Helper()
f, err := os.Open(indexFileName)
if err != nil {
t.Fatalf("open index file %s: %v", indexFileName, err)
}
defer f.Close()
count := 0
if err := idx.WalkIndexFile(f, 0, func(key types.NeedleId, offset types.Offset, size types.Size) error {
if !offset.IsZero() && !size.IsDeleted() {
count++
}
return nil
}); err != nil {
t.Fatalf("walk index file %s: %v", indexFileName, err)
}
return count
}
func doSomeWritesDeletes(i int, v *Volume, t *testing.T, infos []*needleInfo) {
n := newRandomNeedle(uint64(i))
_, size, _, err := v.writeNeedle2(n, true, false, false)