Files
seaweedfs/weed/server/volume_grpc_tier_test.go
T
87332eb60b Cloud/remote storage & tiering: configurable multipart upload/download concurrency (#11319)
* pb: add multipart concurrency fields to RemoteConf and tier move requests

RemoteConf gains upload_concurrency/download_concurrency (0 = client
default); VolumeTierMoveDatToRemote/FromRemote requests gain a
concurrency field (0 = backend default).

* remote storage: honor RemoteConf upload/download concurrency in s3 and azure clients

s3 client: ReadFile passes conf download_concurrency to the downloader,
WriteFile uses upload_concurrency for the uploader; previously
hard-coded 1 upload / 5 download parts. 0 keeps defaults. Same for
azure client.

* storage: plumb concurrency through backend interface and tier upload/download

BackendStorage.CopyFile/DownloadFile take a concurrency hint (<=0 =
backend configured default); s3 backend reads
upload_concurrency/download_concurrency from scaffold config with
parseConcurrency fallback, rclone updated to the new signature. Tier
move gRPC handlers forward the request concurrency to the backend.

* shell: -upload_concurrency/-download_concurrency for remote.configure, -concurrent for volume.tier

remote.configure exposes upload/download concurrency persisted into
RemoteConf; volume.tier move/evict commands forward -concurrent to the
tier move requests. Documented in master-cloud.toml scaffold.

* test: cover concurrency propagation in remote tier integration test

* remote.configure: merge existing config on partial update

Load the stored RemoteConf before saving so a partial update (e.g. only
-upload_concurrency) preserves credentials, endpoints, and type instead
of replacing them with new-config defaults. Only treat a confirmed
ErrNotFound as a new configuration; propagate all other load errors so a
transient filer failure does not overwrite stored settings.

On a type transition, reset backend-specific fields to the destination
type's new-config defaults rather than inheriting the old backend's
empty values. Bound configured concurrency to a sane maximum.

* remote storage: honor configured download concurrency in S3 and Azure

ReadFileWithConcurrency now resolves a zero request override against the
client's configured download_concurrency (new downloadConcurrency()
helpers), so the remote-mount/cache read path honors
RemoteConf.DownloadConcurrency instead of the hard-coded default.

Azure also clamps the resolved value to math.MaxUint16 regardless of
whether the fallback was used, preventing uint16 wraparound when a
configured value exceeds 65535.

* shell: rename -concurrent to -concurrency and validate tier transfer bounds

Rename the -concurrent flag to -concurrency across volume.tier.upload,
volume.tier.download, and volume.tier.compact to match the proto field and
RemoteConf field names. Add validateTierConcurrency to reject values that
would wrap int32 or exceed a 1024 cap before constructing the request.

* server: clamp tier move concurrency in gRPC handlers

Add clampTierConcurrency to both VolumeTierMoveDatToRemote and
VolumeTierMoveDatFromRemote handlers so a direct gRPC caller cannot spawn
an unbounded number of network workers.

* trim verbose comments added with concurrency feature

Remove redundant doc comments on the backend interface, rclone backend,
s3_backend parseConcurrency, and test helpers that restated the obvious.

* remote.configure: apply type defaults before re-parse so explicit flags win

applyTypeDefaults ran after the second flag parse, overwriting explicit
destination flags (e.g. -s3.region=eu-west-1) with new-config defaults.
Move the type-transition default reset before the re-parse so user-supplied
flags override the destination defaults.

* remote.configure: only treat explicit -type as a type transition

The first parse defaults -type to s3, so a concurrency-only update on an
existing non-S3 config captured requestedType=s3 and wrongly triggered a
type transition, resetting the stored backend to S3. Use fs.Visit to
detect whether -type was explicitly supplied; an omitted -type keeps the
stored backend.

---------

Co-authored-by: Jack Meredith <9480542+jackusm@users.noreply.github.com>
Co-authored-by: Chris Lu <chris.lu@gmail.com>
2026-09-14 22:09:08 -07:00

229 lines
6.3 KiB
Go

package weed_server
import (
"context"
"fmt"
"io"
"os"
"path/filepath"
"testing"
"time"
"google.golang.org/grpc"
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
"github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/storage"
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
"github.com/seaweedfs/seaweedfs/weed/storage/types"
"github.com/seaweedfs/seaweedfs/weed/util"
)
const tierTimestampTestBackendName = "tier_timestamp_test.default"
// discardServerStream drops everything sent to it. The embedded
// grpc.ServerStream is nil, so Context is implemented here rather than promoted
// -- the tier RPCs read it to authorize the caller.
type discardServerStream[T any] struct {
grpc.ServerStream
}
func (s *discardServerStream[T]) Send(*T) error {
return nil
}
func (s *discardServerStream[T]) Context() context.Context {
return context.Background()
}
type tierTimestampTestBackend struct {
root string
}
func (b *tierTimestampTestBackend) ToProperties() map[string]string {
return map[string]string{"root": b.root}
}
func (b *tierTimestampTestBackend) NewStorageFile(key string, volumeInfo *volume_server_pb.VolumeInfo) backend.BackendStorageFile {
return &tierTimestampTestBackendFile{
path: filepath.Join(b.root, key),
volumeInfo: volumeInfo,
}
}
func (b *tierTimestampTestBackend) CopyFile(file *os.File, fn func(progressed int64, percentage float32) error, concurrency int) (key string, size int64, err error) {
key = "remote.dat"
fileInfo, err := file.Stat()
if err != nil {
return "", 0, err
}
output, err := os.Create(filepath.Join(b.root, key))
if err != nil {
return "", 0, err
}
defer output.Close()
size, err = io.Copy(output, io.NewSectionReader(file, 0, fileInfo.Size()))
if err == nil && fn != nil {
err = fn(size, 100)
}
return key, size, err
}
func (b *tierTimestampTestBackend) DownloadFile(fileName string, key string, fn func(progressed int64, percentage float32) error, concurrency int) (size int64, err error) {
input, err := os.Open(filepath.Join(b.root, key))
if err != nil {
return 0, err
}
defer input.Close()
output, err := os.Create(fileName)
if err != nil {
return 0, err
}
defer output.Close()
size, err = io.Copy(output, input)
if err == nil && fn != nil {
err = fn(size, 100)
}
return size, err
}
func (b *tierTimestampTestBackend) DeleteFile(key string) error {
return os.Remove(filepath.Join(b.root, key))
}
type tierTimestampTestBackendFile struct {
path string
volumeInfo *volume_server_pb.VolumeInfo
}
func (f *tierTimestampTestBackendFile) ReadAt(p []byte, off int64) (int, error) {
file, err := os.Open(f.path)
if err != nil {
return 0, err
}
defer file.Close()
return file.ReadAt(p, off)
}
func (f *tierTimestampTestBackendFile) WriteAt(p []byte, off int64) (int, error) {
return 0, fmt.Errorf("remote test file is read-only")
}
func (f *tierTimestampTestBackendFile) Truncate(off int64) error {
return fmt.Errorf("remote test file is read-only")
}
func (f *tierTimestampTestBackendFile) Close() error {
return nil
}
func (f *tierTimestampTestBackendFile) GetStat() (datSize int64, modTime time.Time, err error) {
files := f.volumeInfo.GetFiles()
if len(files) == 0 {
return 0, time.Time{}, fmt.Errorf("remote file info not found")
}
return int64(files[0].GetFileSize()), time.Unix(int64(files[0].GetModifiedTime()), 0), nil
}
func (f *tierTimestampTestBackendFile) Name() string {
return f.path
}
func (f *tierTimestampTestBackendFile) Sync() error {
return nil
}
func TestVolumeTierMoveDatPreservesModifiedTime(t *testing.T) {
dataDir := t.TempDir()
remoteDir := t.TempDir()
testBackend := &tierTimestampTestBackend{root: remoteDir}
backend.BackendStorages[tierTimestampTestBackendName] = testBackend
t.Cleanup(func() {
delete(backend.BackendStorages, tierTimestampTestBackendName)
})
store := storage.NewStore(
nil,
"localhost",
8080,
18080,
"http://localhost:8080",
"store-id",
[]string{dataDir},
[]int32{10},
[]util.MinFreeSpace{{}},
"",
storage.NeedleMapInMemory,
[]types.DiskType{types.HardDriveType},
nil,
0,
stats.DefaultDiskIOProbeConfig(),
)
t.Cleanup(store.Close)
const volumeId = needle.VolumeId(1)
if err := store.AddVolume(volumeId, "", storage.NeedleMapInMemory, "000", "", 0, needle.Version3, 0, types.HardDriveType, 0); err != nil {
t.Fatalf("add volume: %v", err)
}
volume := store.GetVolume(volumeId)
dataFileName := volume.FileName(".dat")
sourceModifiedTime := time.Unix(1_700_000_000, 0)
if err := os.Chtimes(dataFileName, sourceModifiedTime, sourceModifiedTime); err != nil {
t.Fatalf("set source modified time: %v", err)
}
// Re-open the data backend so the DiskFile caches the on-disk mtime, the way
// a volume freshly loaded from disk does.
volume.DataBackend.Close()
reopened, err := os.OpenFile(dataFileName, os.O_RDWR, 0644)
if err != nil {
t.Fatalf("reopen data file: %v", err)
}
volume.DataBackend = backend.NewDiskFile(reopened)
volumeServer := &VolumeServer{store: store}
if err := volumeServer.VolumeTierMoveDatToRemote(
&volume_server_pb.VolumeTierMoveDatToRemoteRequest{
VolumeId: uint32(volumeId),
DestinationBackendName: tierTimestampTestBackendName,
},
&discardServerStream[volume_server_pb.VolumeTierMoveDatToRemoteResponse]{},
); err != nil {
t.Fatalf("move data to remote: %v", err)
}
remoteFiles := volume.GetVolumeInfo().GetFiles()
if len(remoteFiles) != 1 {
t.Fatalf("remote file count = %d, want 1", len(remoteFiles))
}
if got := remoteFiles[0].GetModifiedTime(); got != uint64(sourceModifiedTime.Unix()) {
t.Fatalf("remote modified time = %d, want %d", got, sourceModifiedTime.Unix())
}
if _, err := os.Stat(dataFileName); !os.IsNotExist(err) {
t.Fatalf("local data file still exists after upload: %v", err)
}
if err := volumeServer.VolumeTierMoveDatFromRemote(
&volume_server_pb.VolumeTierMoveDatFromRemoteRequest{
VolumeId: uint32(volumeId),
},
&discardServerStream[volume_server_pb.VolumeTierMoveDatFromRemoteResponse]{},
); err != nil {
t.Fatalf("move data from remote: %v", err)
}
fileInfo, err := os.Stat(dataFileName)
if err != nil {
t.Fatalf("stat downloaded data file: %v", err)
}
if got := fileInfo.ModTime().Unix(); got != sourceModifiedTime.Unix() {
t.Fatalf("downloaded modified time = %d, want %d", got, sourceModifiedTime.Unix())
}
}