mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-08 07:17:48 +02:00
* volume: forward fsync=true to replicas in ReplicatedWrite When a write request carries fsync=true, only the primary volume server flushed to disk: the replica fan-out URL in ReplicatedWrite only carried type/ttl/ts/cm, so replicas always wrote without fsync even when the client explicitly requested a durable write. Forward the fsync request parameter to the replica volume servers so a durable write means every replica has flushed to disk, not just the primary. Replicas without fsync are untouched (zero behavior change). * storage: flush a durable write inline while stopping The fsync flag on the write path really selects the async batch worker, and it was switched off once the store is stopping. So a fsync=true write landing during the pre-stop drain got acked without ever being flushed - and now that ReplicatedWrite forwards fsync, that covers replicas too. Flush it inline instead of queueing it. The drain keeps accepting writes, which is the whole point of preStopSeconds, and the ack still means the .dat is on disk. If the fsync fails, the append comes back off the .dat and the needle map goes back to what it pointed at before, so nothing resolves to an offset past the truncated end. * storage: make the store's stopping flag atomic SetStopping runs on the signal handler goroutine while the write and vacuum paths read the flag, so every read of it was racy. Nothing about the shutdown ordering changes; only the flag itself is now safe to read. * topology: check the errors the replication test was dropping The mock replica ignored its response write and the mock master ignored whatever Serve returned, so a broken mock would have shown up as a confusing timeout rather than a failure. Also drops the explicit listener close: grpc.Server.Stop already closes the listener it was given. --------- Co-authored-by: hzsunchao <hzsunchao@corp.netease.com> Co-authored-by: Chris Lu <chris.lu@gmail.com>
453 lines
16 KiB
Go
453 lines
16 KiB
Go
package storage
|
|
|
|
import (
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"path/filepath"
|
|
"sync"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/stretchr/testify/require"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
// In-process integration tests for cloud-tiered ("remote") volumes.
|
|
//
|
|
// Cover the operations the user is likely to schedule against a tiered
|
|
// volume — balance/move, vacuum, EC encode, EC decode — exercising the real
|
|
// Volume / DiskLocation / Store code paths against a fake BackendStorage that
|
|
// stores objects in a temp dir. The fake stands in for S3/rclone/etc. so the
|
|
// tests stay hermetic and fast.
|
|
|
|
// localDirBackend is a BackendStorage that stores objects as files in a
|
|
// temp directory. It deletes from / writes to the dir under a mutex so the
|
|
// tests can observe ordering (e.g. that a remote object survives a move).
|
|
type localDirBackend struct {
|
|
root string
|
|
|
|
mu sync.Mutex
|
|
deletes []string // history of DeleteFile keys, for assertions
|
|
}
|
|
|
|
func newLocalDirBackend(t *testing.T) *localDirBackend {
|
|
t.Helper()
|
|
root := t.TempDir()
|
|
return &localDirBackend{root: root}
|
|
}
|
|
|
|
func (b *localDirBackend) ToProperties() map[string]string {
|
|
return map[string]string{"root": b.root}
|
|
}
|
|
|
|
func (b *localDirBackend) NewStorageFile(key string, tierInfo *volume_server_pb.VolumeInfo) backend.BackendStorageFile {
|
|
return &localDirBackendFile{backend: b, key: key, tierInfo: tierInfo}
|
|
}
|
|
|
|
func (b *localDirBackend) CopyFile(f *os.File, fn func(progressed int64, percentage float32) error) (key string, size int64, err error) {
|
|
key = fmt.Sprintf("obj-%d-%d", time.Now().UnixNano(), os.Getpid())
|
|
dst := filepath.Join(b.root, key)
|
|
out, err := os.Create(dst)
|
|
if err != nil {
|
|
return "", 0, err
|
|
}
|
|
defer out.Close()
|
|
if _, err = f.Seek(0, io.SeekStart); err != nil {
|
|
return "", 0, err
|
|
}
|
|
written, err := io.Copy(out, f)
|
|
if err != nil {
|
|
return "", 0, err
|
|
}
|
|
if fn != nil {
|
|
_ = fn(written, 100)
|
|
}
|
|
return key, written, nil
|
|
}
|
|
|
|
func (b *localDirBackend) DownloadFile(fileName string, key string, fn func(progressed int64, percentage float32) error) (size int64, err error) {
|
|
src := filepath.Join(b.root, key)
|
|
in, err := os.Open(src)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer in.Close()
|
|
out, err := os.Create(fileName)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer out.Close()
|
|
written, err := io.Copy(out, in)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
if fn != nil {
|
|
_ = fn(written, 100)
|
|
}
|
|
return written, nil
|
|
}
|
|
|
|
func (b *localDirBackend) DeleteFile(key string) error {
|
|
b.mu.Lock()
|
|
b.deletes = append(b.deletes, key)
|
|
b.mu.Unlock()
|
|
return os.Remove(filepath.Join(b.root, key))
|
|
}
|
|
|
|
func (b *localDirBackend) deleteHistory() []string {
|
|
b.mu.Lock()
|
|
defer b.mu.Unlock()
|
|
out := make([]string, len(b.deletes))
|
|
copy(out, b.deletes)
|
|
return out
|
|
}
|
|
|
|
func (b *localDirBackend) objectExists(key string) bool {
|
|
_, err := os.Stat(filepath.Join(b.root, key))
|
|
return err == nil
|
|
}
|
|
|
|
// localDirBackendFile satisfies BackendStorageFile by reading/writing through
|
|
// the file in the temp dir keyed by the .vif's stored object key. Size and
|
|
// modtime come from the cached tierInfo, mirroring the S3 backend's
|
|
// behavior of returning .vif metadata from GetStat.
|
|
type localDirBackendFile struct {
|
|
backend *localDirBackend
|
|
key string
|
|
tierInfo *volume_server_pb.VolumeInfo
|
|
}
|
|
|
|
func (f *localDirBackendFile) ReadAt(p []byte, off int64) (int, error) {
|
|
in, err := os.Open(filepath.Join(f.backend.root, f.key))
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer in.Close()
|
|
return in.ReadAt(p, off)
|
|
}
|
|
|
|
func (f *localDirBackendFile) WriteAt(p []byte, off int64) (int, error) {
|
|
out, err := os.OpenFile(filepath.Join(f.backend.root, f.key), os.O_RDWR|os.O_CREATE, 0o644)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer out.Close()
|
|
return out.WriteAt(p, off)
|
|
}
|
|
|
|
func (f *localDirBackendFile) Truncate(off int64) error {
|
|
return os.Truncate(filepath.Join(f.backend.root, f.key), off)
|
|
}
|
|
|
|
func (f *localDirBackendFile) Close() error { return nil }
|
|
func (f *localDirBackendFile) Name() string { return f.key }
|
|
func (f *localDirBackendFile) Sync() error { return nil }
|
|
func (f *localDirBackendFile) GetStat() (int64, time.Time, error) {
|
|
files := f.tierInfo.GetFiles()
|
|
if len(files) == 0 {
|
|
return 0, time.Time{}, fmt.Errorf("remote file info not found")
|
|
}
|
|
return int64(files[0].FileSize), time.Unix(int64(files[0].ModifiedTime), 0), nil
|
|
}
|
|
|
|
const (
|
|
testBackendName = "test_local_dir.default"
|
|
)
|
|
|
|
// registerTestBackend installs the fake backend in the global registry under
|
|
// testBackendName for the duration of one test. Volume.Destroy and tier
|
|
// upload look this map up by name, so the registration must outlive the
|
|
// volume operations exercised below.
|
|
func registerTestBackend(t *testing.T, b *localDirBackend) {
|
|
t.Helper()
|
|
backend.BackendStorages[testBackendName] = b
|
|
t.Cleanup(func() {
|
|
delete(backend.BackendStorages, testBackendName)
|
|
})
|
|
}
|
|
|
|
// tierUpVolumeLive creates a real on-disk volume, writes a few needles, then
|
|
// uploads the .dat to the fake backend and rewrites the volume in remote mode
|
|
// (mirrors the production flow in volume_grpc_tier_upload.go but in-process).
|
|
// It returns the still-open volume, exactly as a volume server holds it after a
|
|
// live `volume.tier.upload` — no reload. Callers that want the on-disk state a
|
|
// server sees after restart use tierUpVolume, which closes it.
|
|
func tierUpVolumeLive(t *testing.T, dir string, vid needle.VolumeId, b *localDirBackend) (v *Volume, key string) {
|
|
t.Helper()
|
|
v, err := NewVolume(dir, dir, "", vid, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
|
|
require.NoError(t, err)
|
|
|
|
for i := 1; i <= 5; i++ {
|
|
_, _, _, err := v.writeNeedle2(newRandomNeedle(uint64(i)), true, false, false)
|
|
require.NoError(t, err)
|
|
}
|
|
|
|
diskFile, ok := v.DataBackend.(*backend.DiskFile)
|
|
require.True(t, ok, "expected on-disk backend before tier-up")
|
|
|
|
uploadKey, size, err := b.CopyFile(diskFile.File, nil)
|
|
require.NoError(t, err)
|
|
|
|
bType, bId := backend.BackendNameToTypeId(testBackendName)
|
|
v.GetVolumeInfo().Files = append(v.GetVolumeInfo().GetFiles(), &volume_server_pb.RemoteFile{
|
|
BackendType: bType,
|
|
BackendId: bId,
|
|
Key: uploadKey,
|
|
Offset: 0,
|
|
FileSize: uint64(size),
|
|
ModifiedTime: uint64(time.Now().Unix()),
|
|
Extension: ".dat",
|
|
})
|
|
require.NoError(t, v.SaveVolumeInfo())
|
|
require.NoError(t, v.LoadRemoteFile())
|
|
require.NoError(t, os.Remove(v.FileName(".dat")))
|
|
|
|
return v, uploadKey
|
|
}
|
|
|
|
// tierUpVolume runs tierUpVolumeLive then closes the volume. Tests using it
|
|
// reload from disk to mirror what a volume server does on restart with a
|
|
// tiered volume.
|
|
func tierUpVolume(t *testing.T, dir string, vid needle.VolumeId, b *localDirBackend) (collection string, key string) {
|
|
t.Helper()
|
|
v, uploadKey := tierUpVolumeLive(t, dir, vid, b)
|
|
v.Close()
|
|
return v.Collection, uploadKey
|
|
}
|
|
|
|
// reloadVolume loads an existing volume from disk, the way a volume server
|
|
// does at startup. Returns the live volume with its async write worker
|
|
// running, so Destroy's channel close is valid.
|
|
func reloadVolume(t *testing.T, dir string, vid needle.VolumeId) *Volume {
|
|
t.Helper()
|
|
v, err := NewVolume(dir, dir, "", vid, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
|
|
require.NoError(t, err)
|
|
return v
|
|
}
|
|
|
|
// TestRemoteTier_DiskScanLoadsRemoteOnlyVolume locks in that the disk scan
|
|
// (loadExistingVolume) loads a remote-only volume — a .vif pointing at remote
|
|
// files with no local .dat — instead of skipping it as a lone sidecar. The
|
|
// phantom-.dat guard must let remote volumes through.
|
|
func TestRemoteTier_DiskScanLoadsRemoteOnlyVolume(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(44)
|
|
tierUpVolume(t, dir, vid, b) // leaves .vif (remote) + .idx, no .dat
|
|
|
|
require.False(t, util.FileExists(filepath.Join(dir, "44.dat")), "tier-up should have removed the local .dat")
|
|
|
|
loc := &DiskLocation{
|
|
Directory: dir,
|
|
DirectoryUuid: "test-uuid",
|
|
IdxDirectory: dir,
|
|
DiskType: types.HddType,
|
|
MaxVolumeCount: 100,
|
|
OriginalMaxVolumeCount: 100,
|
|
MinFreeSpace: util.MinFreeSpace{Type: util.AsPercent, Percent: 1, Raw: "1"},
|
|
}
|
|
loc.volumes = make(map[needle.VolumeId]*Volume)
|
|
loc.ecVolumes = make(map[needle.VolumeId]*erasure_coding.EcVolume)
|
|
|
|
loc.loadExistingVolumes(NeedleMapInMemory, 0)
|
|
|
|
v, ok := loc.volumes[vid]
|
|
require.True(t, ok, "remote-only volume must be loaded by the disk scan, not skipped by the phantom-.dat guard")
|
|
require.True(t, v.HasRemoteFile(), "loaded volume should be in remote mode")
|
|
v.Close()
|
|
}
|
|
|
|
// TestRemoteTier_LiveTierUpload_StillReportsToMaster covers a live
|
|
// `volume.tier.upload`: the .dat is removed and the volume serves from remote,
|
|
// but the same in-memory Volume keeps heartbeating with no reload. The
|
|
// phantom-.dat guard must not suppress it just because .dat is gone —
|
|
// LoadRemoteFile has flipped it into remote mode, so ToVolumeInformationMessage
|
|
// must still report it to the master. If HasRemoteFile stayed false the volume
|
|
// would vanish from the topology ("volume not found").
|
|
func TestRemoteTier_LiveTierUpload_StillReportsToMaster(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(67)
|
|
v, _ := tierUpVolumeLive(t, dir, vid, b)
|
|
defer v.Close()
|
|
// A store-owned volume always carries its DiskLocation; NewVolume leaves it
|
|
// nil, so give it one for the IsReadOnly disk-space check inside the heartbeat.
|
|
v.location = &DiskLocation{Directory: dir, DiskType: types.HddType}
|
|
|
|
require.True(t, v.HasRemoteFile(), "a tier-uploaded volume is in remote mode even before any reload")
|
|
require.False(t, util.FileExists(v.FileName(".dat")), "tier-up should have removed the local .dat")
|
|
|
|
_, msg := v.ToVolumeInformationMessage()
|
|
require.NotNil(t, msg, "tier-uploaded volume must still report to master")
|
|
require.NotEmpty(t, msg.RemoteStorageName, "reported volume must carry its remote backend name")
|
|
}
|
|
|
|
// TestRemoteTier_ReloadUnderDataLock_NoDeadlock guards the reload-under-lock
|
|
// path: CommitCompact holds dataFileAccessLock and calls v.load(), which for a
|
|
// remote-tiered volume swaps the data backend. That swap must go through the
|
|
// lock-free loadRemoteFileLocked; if load() instead used the public
|
|
// LoadRemoteFile (which takes dataFileAccessLock), it would re-enter the held
|
|
// lock and deadlock.
|
|
func TestRemoteTier_ReloadUnderDataLock_NoDeadlock(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(68)
|
|
v, _ := tierUpVolumeLive(t, dir, vid, b)
|
|
v.location = &DiskLocation{Directory: dir, DiskType: types.HddType}
|
|
require.True(t, v.HasRemoteFile())
|
|
|
|
done := make(chan error, 1)
|
|
go func() {
|
|
// Mirror CommitCompact: hold the data lock across the reload.
|
|
v.dataFileAccessLock.Lock()
|
|
defer v.dataFileAccessLock.Unlock()
|
|
done <- v.load(true, false, v.needleMapKind, 0, v.Version())
|
|
}()
|
|
|
|
select {
|
|
case err := <-done:
|
|
require.NoError(t, err)
|
|
require.True(t, v.HasRemoteFile(), "volume must stay remote-tiered after reload")
|
|
v.Close()
|
|
case <-time.After(10 * time.Second):
|
|
t.Fatal("reload under dataFileAccessLock deadlocked: load() re-entered the held lock via LoadRemoteFile")
|
|
}
|
|
}
|
|
|
|
// TestRemoteTier_Move_KeepsRemoteObject simulates the move-on-source-after-copy
|
|
// step of a balance: Destroy(onlyEmpty=false, keepRemoteData=true). The remote
|
|
// object must survive — the destination's freshly-copied .vif points at it.
|
|
func TestRemoteTier_Move_KeepsRemoteObject(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(31)
|
|
_, key := tierUpVolume(t, dir, vid, b)
|
|
require.True(t, b.objectExists(key), "remote object missing after tier-up")
|
|
|
|
v := reloadVolume(t, dir, vid)
|
|
require.True(t, v.HasRemoteFile())
|
|
|
|
require.NoError(t, v.Destroy(false, true))
|
|
|
|
require.True(t, b.objectExists(key), "Destroy(keepRemoteData=true) must not delete remote object")
|
|
require.Empty(t, b.deleteHistory(), "no DeleteFile call expected on a move-style destroy")
|
|
}
|
|
|
|
// TestRemoteTier_RealDelete_RemovesRemoteObject is the inverse: a true delete
|
|
// (keepRemoteData=false) must clean up the remote object. Locks in that we
|
|
// did not accidentally turn the new flag into a global skip.
|
|
func TestRemoteTier_RealDelete_RemovesRemoteObject(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(32)
|
|
_, key := tierUpVolume(t, dir, vid, b)
|
|
|
|
v := reloadVolume(t, dir, vid)
|
|
require.True(t, v.HasRemoteFile())
|
|
|
|
require.NoError(t, v.Destroy(false, false))
|
|
|
|
require.False(t, b.objectExists(key), "Destroy(keepRemoteData=false) must delete remote object")
|
|
require.Equal(t, []string{key}, b.deleteHistory())
|
|
}
|
|
|
|
// TestRemoteTier_Vacuum_DoesNotDeleteRemote runs the compact paths against a
|
|
// remote-tier volume and asserts the safety property: regardless of whether
|
|
// compact succeeds, errors, or no-ops, it must not delete the cloud object.
|
|
func TestRemoteTier_Vacuum_DoesNotDeleteRemote(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(33)
|
|
_, key := tierUpVolume(t, dir, vid, b)
|
|
require.True(t, b.objectExists(key))
|
|
|
|
v := reloadVolume(t, dir, vid)
|
|
defer v.Close()
|
|
require.True(t, v.HasRemoteFile())
|
|
|
|
_ = v.CompactByVolumeData(nil)
|
|
_ = v.CompactByIndex(nil)
|
|
require.True(t, b.objectExists(key), "Compact must not delete remote object")
|
|
require.Empty(t, b.deleteHistory())
|
|
}
|
|
|
|
// TestRemoteTier_ECEncode_RequiresLocalDat confirms the EC encoder runs
|
|
// against the local .dat path. For a remote-tier volume the .dat is gone,
|
|
// so encoding is expected to fail with a missing-file error — locks in that
|
|
// callers must download (tier_move_dat_from_remote) before encoding.
|
|
func TestRemoteTier_ECEncode_RequiresLocalDat(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(34)
|
|
tierUpVolume(t, dir, vid, b)
|
|
|
|
baseFileName := filepath.Join(dir, fmt.Sprintf("%d", uint32(vid)))
|
|
_, err := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext())
|
|
require.Error(t, err, "EC encoder must not run with .dat missing — caller is expected to download first")
|
|
require.Contains(t, err.Error(), ".dat")
|
|
}
|
|
|
|
// TestRemoteTier_ECEncodeDecode_AfterDownload exercises the encode/decode
|
|
// round-trip on a tiered volume after pulling the .dat back to local disk
|
|
// (the production sequence used by `volume.tier.move -dest=local`).
|
|
func TestRemoteTier_ECEncodeDecode_AfterDownload(t *testing.T) {
|
|
b := newLocalDirBackend(t)
|
|
registerTestBackend(t, b)
|
|
|
|
dir := t.TempDir()
|
|
const vid = needle.VolumeId(35)
|
|
_, key := tierUpVolume(t, dir, vid, b)
|
|
|
|
baseFileName := filepath.Join(dir, fmt.Sprintf("%d", uint32(vid)))
|
|
datPath := baseFileName + ".dat"
|
|
_, err := b.DownloadFile(datPath, key, nil)
|
|
require.NoError(t, err)
|
|
|
|
require.NoError(t, erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"))
|
|
_, ecErr := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext())
|
|
require.NoError(t, ecErr)
|
|
|
|
for i := 0; i < erasure_coding.TotalShardsCount; i++ {
|
|
shardPath := fmt.Sprintf("%s.ec%02d", baseFileName, i)
|
|
_, statErr := os.Stat(shardPath)
|
|
require.NoError(t, statErr, "shard %d missing after encode", i)
|
|
}
|
|
|
|
// Drop the parity-range shards (indices DataShardsCount..TotalShardsCount-1)
|
|
// and rebuild — exercises the recover-from-missing-parity path.
|
|
for i := erasure_coding.DataShardsCount; i < erasure_coding.TotalShardsCount; i++ {
|
|
shardPath := fmt.Sprintf("%s.ec%02d", baseFileName, i)
|
|
require.NoError(t, os.Remove(shardPath))
|
|
}
|
|
rebuilt, err := erasure_coding.RebuildEcFiles(baseFileName, erasure_coding.BackgroundECContext(), false)
|
|
require.NoError(t, err)
|
|
require.NotEmpty(t, rebuilt, "rebuild should report which parity shards were regenerated")
|
|
for i := erasure_coding.DataShardsCount; i < erasure_coding.TotalShardsCount; i++ {
|
|
shardPath := fmt.Sprintf("%s.ec%02d", baseFileName, i)
|
|
_, statErr := os.Stat(shardPath)
|
|
require.NoError(t, statErr, "parity shard %d missing after rebuild", i)
|
|
}
|
|
}
|