mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* pb: add multipart concurrency fields to RemoteConf and tier move requests RemoteConf gains upload_concurrency/download_concurrency (0 = client default); VolumeTierMoveDatToRemote/FromRemote requests gain a concurrency field (0 = backend default). * remote storage: honor RemoteConf upload/download concurrency in s3 and azure clients s3 client: ReadFile passes conf download_concurrency to the downloader, WriteFile uses upload_concurrency for the uploader; previously hard-coded 1 upload / 5 download parts. 0 keeps defaults. Same for azure client. * storage: plumb concurrency through backend interface and tier upload/download BackendStorage.CopyFile/DownloadFile take a concurrency hint (<=0 = backend configured default); s3 backend reads upload_concurrency/download_concurrency from scaffold config with parseConcurrency fallback, rclone updated to the new signature. Tier move gRPC handlers forward the request concurrency to the backend. * shell: -upload_concurrency/-download_concurrency for remote.configure, -concurrent for volume.tier remote.configure exposes upload/download concurrency persisted into RemoteConf; volume.tier move/evict commands forward -concurrent to the tier move requests. Documented in master-cloud.toml scaffold. * test: cover concurrency propagation in remote tier integration test * remote.configure: merge existing config on partial update Load the stored RemoteConf before saving so a partial update (e.g. only -upload_concurrency) preserves credentials, endpoints, and type instead of replacing them with new-config defaults. Only treat a confirmed ErrNotFound as a new configuration; propagate all other load errors so a transient filer failure does not overwrite stored settings. On a type transition, reset backend-specific fields to the destination type's new-config defaults rather than inheriting the old backend's empty values. Bound configured concurrency to a sane maximum. * remote storage: honor configured download concurrency in S3 and Azure ReadFileWithConcurrency now resolves a zero request override against the client's configured download_concurrency (new downloadConcurrency() helpers), so the remote-mount/cache read path honors RemoteConf.DownloadConcurrency instead of the hard-coded default. Azure also clamps the resolved value to math.MaxUint16 regardless of whether the fallback was used, preventing uint16 wraparound when a configured value exceeds 65535. * shell: rename -concurrent to -concurrency and validate tier transfer bounds Rename the -concurrent flag to -concurrency across volume.tier.upload, volume.tier.download, and volume.tier.compact to match the proto field and RemoteConf field names. Add validateTierConcurrency to reject values that would wrap int32 or exceed a 1024 cap before constructing the request. * server: clamp tier move concurrency in gRPC handlers Add clampTierConcurrency to both VolumeTierMoveDatToRemote and VolumeTierMoveDatFromRemote handlers so a direct gRPC caller cannot spawn an unbounded number of network workers. * trim verbose comments added with concurrency feature Remove redundant doc comments on the backend interface, rclone backend, s3_backend parseConcurrency, and test helpers that restated the obvious. * remote.configure: apply type defaults before re-parse so explicit flags win applyTypeDefaults ran after the second flag parse, overwriting explicit destination flags (e.g. -s3.region=eu-west-1) with new-config defaults. Move the type-transition default reset before the re-parse so user-supplied flags override the destination defaults. * remote.configure: only treat explicit -type as a type transition The first parse defaults -type to s3, so a concurrency-only update on an existing non-S3 config captured requestedType=s3 and wrongly triggered a type transition, resetting the stored backend to S3. Use fs.Visit to detect whether -type was explicitly supplied; an omitted -type keeps the stored backend. --------- Co-authored-by: Jack Meredith <9480542+jackusm@users.noreply.github.com> Co-authored-by: Chris Lu <chris.lu@gmail.com>
229 lines
6.3 KiB
Go
229 lines
6.3 KiB
Go
package weed_server
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"path/filepath"
|
|
"testing"
|
|
"time"
|
|
|
|
"google.golang.org/grpc"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
|
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
const tierTimestampTestBackendName = "tier_timestamp_test.default"
|
|
|
|
// discardServerStream drops everything sent to it. The embedded
|
|
// grpc.ServerStream is nil, so Context is implemented here rather than promoted
|
|
// -- the tier RPCs read it to authorize the caller.
|
|
type discardServerStream[T any] struct {
|
|
grpc.ServerStream
|
|
}
|
|
|
|
func (s *discardServerStream[T]) Send(*T) error {
|
|
return nil
|
|
}
|
|
|
|
func (s *discardServerStream[T]) Context() context.Context {
|
|
return context.Background()
|
|
}
|
|
|
|
type tierTimestampTestBackend struct {
|
|
root string
|
|
}
|
|
|
|
func (b *tierTimestampTestBackend) ToProperties() map[string]string {
|
|
return map[string]string{"root": b.root}
|
|
}
|
|
|
|
func (b *tierTimestampTestBackend) NewStorageFile(key string, volumeInfo *volume_server_pb.VolumeInfo) backend.BackendStorageFile {
|
|
return &tierTimestampTestBackendFile{
|
|
path: filepath.Join(b.root, key),
|
|
volumeInfo: volumeInfo,
|
|
}
|
|
}
|
|
|
|
func (b *tierTimestampTestBackend) CopyFile(file *os.File, fn func(progressed int64, percentage float32) error, concurrency int) (key string, size int64, err error) {
|
|
key = "remote.dat"
|
|
fileInfo, err := file.Stat()
|
|
if err != nil {
|
|
return "", 0, err
|
|
}
|
|
|
|
output, err := os.Create(filepath.Join(b.root, key))
|
|
if err != nil {
|
|
return "", 0, err
|
|
}
|
|
defer output.Close()
|
|
|
|
size, err = io.Copy(output, io.NewSectionReader(file, 0, fileInfo.Size()))
|
|
if err == nil && fn != nil {
|
|
err = fn(size, 100)
|
|
}
|
|
return key, size, err
|
|
}
|
|
|
|
func (b *tierTimestampTestBackend) DownloadFile(fileName string, key string, fn func(progressed int64, percentage float32) error, concurrency int) (size int64, err error) {
|
|
input, err := os.Open(filepath.Join(b.root, key))
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer input.Close()
|
|
|
|
output, err := os.Create(fileName)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer output.Close()
|
|
|
|
size, err = io.Copy(output, input)
|
|
if err == nil && fn != nil {
|
|
err = fn(size, 100)
|
|
}
|
|
return size, err
|
|
}
|
|
|
|
func (b *tierTimestampTestBackend) DeleteFile(key string) error {
|
|
return os.Remove(filepath.Join(b.root, key))
|
|
}
|
|
|
|
type tierTimestampTestBackendFile struct {
|
|
path string
|
|
volumeInfo *volume_server_pb.VolumeInfo
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) ReadAt(p []byte, off int64) (int, error) {
|
|
file, err := os.Open(f.path)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer file.Close()
|
|
return file.ReadAt(p, off)
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) WriteAt(p []byte, off int64) (int, error) {
|
|
return 0, fmt.Errorf("remote test file is read-only")
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) Truncate(off int64) error {
|
|
return fmt.Errorf("remote test file is read-only")
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) Close() error {
|
|
return nil
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) GetStat() (datSize int64, modTime time.Time, err error) {
|
|
files := f.volumeInfo.GetFiles()
|
|
if len(files) == 0 {
|
|
return 0, time.Time{}, fmt.Errorf("remote file info not found")
|
|
}
|
|
return int64(files[0].GetFileSize()), time.Unix(int64(files[0].GetModifiedTime()), 0), nil
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) Name() string {
|
|
return f.path
|
|
}
|
|
|
|
func (f *tierTimestampTestBackendFile) Sync() error {
|
|
return nil
|
|
}
|
|
|
|
func TestVolumeTierMoveDatPreservesModifiedTime(t *testing.T) {
|
|
dataDir := t.TempDir()
|
|
remoteDir := t.TempDir()
|
|
testBackend := &tierTimestampTestBackend{root: remoteDir}
|
|
backend.BackendStorages[tierTimestampTestBackendName] = testBackend
|
|
t.Cleanup(func() {
|
|
delete(backend.BackendStorages, tierTimestampTestBackendName)
|
|
})
|
|
|
|
store := storage.NewStore(
|
|
nil,
|
|
"localhost",
|
|
8080,
|
|
18080,
|
|
"http://localhost:8080",
|
|
"store-id",
|
|
[]string{dataDir},
|
|
[]int32{10},
|
|
[]util.MinFreeSpace{{}},
|
|
"",
|
|
storage.NeedleMapInMemory,
|
|
[]types.DiskType{types.HardDriveType},
|
|
nil,
|
|
0,
|
|
stats.DefaultDiskIOProbeConfig(),
|
|
)
|
|
t.Cleanup(store.Close)
|
|
|
|
const volumeId = needle.VolumeId(1)
|
|
if err := store.AddVolume(volumeId, "", storage.NeedleMapInMemory, "000", "", 0, needle.Version3, 0, types.HardDriveType, 0); err != nil {
|
|
t.Fatalf("add volume: %v", err)
|
|
}
|
|
|
|
volume := store.GetVolume(volumeId)
|
|
dataFileName := volume.FileName(".dat")
|
|
sourceModifiedTime := time.Unix(1_700_000_000, 0)
|
|
if err := os.Chtimes(dataFileName, sourceModifiedTime, sourceModifiedTime); err != nil {
|
|
t.Fatalf("set source modified time: %v", err)
|
|
}
|
|
|
|
// Re-open the data backend so the DiskFile caches the on-disk mtime, the way
|
|
// a volume freshly loaded from disk does.
|
|
volume.DataBackend.Close()
|
|
reopened, err := os.OpenFile(dataFileName, os.O_RDWR, 0644)
|
|
if err != nil {
|
|
t.Fatalf("reopen data file: %v", err)
|
|
}
|
|
volume.DataBackend = backend.NewDiskFile(reopened)
|
|
|
|
volumeServer := &VolumeServer{store: store}
|
|
if err := volumeServer.VolumeTierMoveDatToRemote(
|
|
&volume_server_pb.VolumeTierMoveDatToRemoteRequest{
|
|
VolumeId: uint32(volumeId),
|
|
DestinationBackendName: tierTimestampTestBackendName,
|
|
},
|
|
&discardServerStream[volume_server_pb.VolumeTierMoveDatToRemoteResponse]{},
|
|
); err != nil {
|
|
t.Fatalf("move data to remote: %v", err)
|
|
}
|
|
|
|
remoteFiles := volume.GetVolumeInfo().GetFiles()
|
|
if len(remoteFiles) != 1 {
|
|
t.Fatalf("remote file count = %d, want 1", len(remoteFiles))
|
|
}
|
|
if got := remoteFiles[0].GetModifiedTime(); got != uint64(sourceModifiedTime.Unix()) {
|
|
t.Fatalf("remote modified time = %d, want %d", got, sourceModifiedTime.Unix())
|
|
}
|
|
if _, err := os.Stat(dataFileName); !os.IsNotExist(err) {
|
|
t.Fatalf("local data file still exists after upload: %v", err)
|
|
}
|
|
|
|
if err := volumeServer.VolumeTierMoveDatFromRemote(
|
|
&volume_server_pb.VolumeTierMoveDatFromRemoteRequest{
|
|
VolumeId: uint32(volumeId),
|
|
},
|
|
&discardServerStream[volume_server_pb.VolumeTierMoveDatFromRemoteResponse]{},
|
|
); err != nil {
|
|
t.Fatalf("move data from remote: %v", err)
|
|
}
|
|
|
|
fileInfo, err := os.Stat(dataFileName)
|
|
if err != nil {
|
|
t.Fatalf("stat downloaded data file: %v", err)
|
|
}
|
|
if got := fileInfo.ModTime().Unix(); got != sourceModifiedTime.Unix() {
|
|
t.Fatalf("downloaded modified time = %d, want %d", got, sourceModifiedTime.Unix())
|
|
}
|
|
}
|