mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 22:41:56 +02:00
* fix 11586 * Update filer_server_handlers_write_autochunk.go * filer: fix inline append races, empty files, and stale ETags Serialize the append read-modify-write on the entry lock so concurrent appends merge instead of losing content, keep small appends to empty files inline, tolerate legacy entries whose metadata size differs from their content, and set the entry digest so appended inline files keep a real ETag. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com> Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
322 lines
11 KiB
Go
322 lines
11 KiB
Go
package weed_server
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"crypto/md5"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
"hash"
|
|
"io"
|
|
"net/http"
|
|
"strconv"
|
|
"sync"
|
|
"time"
|
|
|
|
"slices"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/security"
|
|
"github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
var bufPool = sync.Pool{
|
|
New: func() interface{} {
|
|
return new(bytes.Buffer)
|
|
},
|
|
}
|
|
|
|
func (fs *FilerServer) uploadRequestToChunks(ctx context.Context, w http.ResponseWriter, r *http.Request, reader io.Reader, chunkSize int32, fileName, contentType string, contentLength int64, so *operation.StorageOption) (fileChunks []*filer_pb.FileChunk, md5Hash hash.Hash, chunkOffset int64, uploadErr error, smallContent []byte) {
|
|
query := r.URL.Query()
|
|
|
|
isAppend := isAppend(r)
|
|
if query.Has("offset") {
|
|
offset := query.Get("offset")
|
|
offsetInt, err := strconv.ParseInt(offset, 10, 64)
|
|
if err != nil || offsetInt < 0 {
|
|
err = fmt.Errorf("invalid 'offset': '%s'", offset)
|
|
return nil, nil, 0, err, nil
|
|
}
|
|
if isAppend && offsetInt > 0 {
|
|
err = fmt.Errorf("cannot set offset when op=append")
|
|
return nil, nil, 0, err, nil
|
|
}
|
|
chunkOffset = offsetInt
|
|
}
|
|
|
|
if isAppend && !so.SaveInside {
|
|
fullPath := fs.fixFilePath(ctx, r, fileName)
|
|
entry, findErr := fs.filer.FindEntry(ctx, util.FullPath(fullPath))
|
|
if findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
|
return nil, nil, 0, fmt.Errorf("find entry for append %q: %w", fullPath, findErr), nil
|
|
}
|
|
if findErr == nil && entry != nil && !entry.IsDirectory() && entry.Remote == nil && len(entry.HardLinkId) == 0 && len(entry.GetChunks()) == 0 && (len(entry.Content) > 0 || entry.FileSize == 0) {
|
|
if entry.FileSize > uint64(len(entry.Content)) {
|
|
return nil, nil, 0, fmt.Errorf("inline file %q has inconsistent size: metadata=%d content=%d", fullPath, entry.FileSize, len(entry.Content)), nil
|
|
}
|
|
|
|
remainingInlineBudget := fs.option.SaveToFilerLimit - int64(len(entry.Content))
|
|
if remainingInlineBudget > 0 {
|
|
// Read at most the remaining inline budget. Reaching the limit means
|
|
// the resulting file must be chunked, even if this is also EOF.
|
|
prefix, readErr := io.ReadAll(io.LimitReader(reader, remainingInlineBudget))
|
|
if readErr != nil {
|
|
return nil, nil, 0, fmt.Errorf("read input: %w", readErr), nil
|
|
}
|
|
if int64(len(prefix)) < remainingInlineBudget {
|
|
inlineAppend := append([]byte{}, prefix...)
|
|
md5Hash = md5.New()
|
|
_, _ = md5Hash.Write(inlineAppend)
|
|
if len(inlineAppend) > 0 {
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ContentSaveToFiler).Inc()
|
|
}
|
|
return nil, md5Hash, int64(len(inlineAppend)), nil, inlineAppend
|
|
}
|
|
if chunkSize <= 0 {
|
|
return nil, nil, 0, fmt.Errorf("invalid chunk size %d for inline append promotion", chunkSize), nil
|
|
}
|
|
reader = io.MultiReader(bytes.NewReader(prefix), reader)
|
|
} else {
|
|
// Probe one byte so an empty append can preserve the inline entry
|
|
// even when the configured threshold is disabled or below its size.
|
|
prefix, readErr := io.ReadAll(io.LimitReader(reader, 1))
|
|
if readErr != nil {
|
|
return nil, nil, 0, fmt.Errorf("read input: %w", readErr), nil
|
|
}
|
|
if len(prefix) == 0 {
|
|
return nil, md5.New(), 0, nil, []byte{}
|
|
}
|
|
if chunkSize <= 0 {
|
|
return nil, nil, 0, fmt.Errorf("invalid chunk size %d for inline append promotion", chunkSize), nil
|
|
}
|
|
reader = io.MultiReader(bytes.NewReader(prefix), reader)
|
|
}
|
|
}
|
|
}
|
|
|
|
return fs.uploadReaderToChunks(ctx, r, reader, chunkOffset, chunkSize, fileName, contentType, isAppend, so)
|
|
}
|
|
|
|
func (fs *FilerServer) uploadReaderToChunks(ctx context.Context, r *http.Request, reader io.Reader, startOffset int64, chunkSize int32, fileName, contentType string, isAppend bool, so *operation.StorageOption) (fileChunks []*filer_pb.FileChunk, md5Hash hash.Hash, chunkOffset int64, uploadErr error, smallContent []byte) {
|
|
|
|
md5Hash = md5.New()
|
|
chunkOffset = startOffset
|
|
var partReader = io.NopCloser(io.TeeReader(reader, md5Hash))
|
|
|
|
var wg sync.WaitGroup
|
|
var bytesBufferCounter int64 = 4
|
|
bytesBufferLimitChan := make(chan struct{}, bytesBufferCounter)
|
|
var fileChunksLock sync.Mutex
|
|
var uploadErrLock sync.Mutex
|
|
for {
|
|
|
|
// need to throttle used byte buffer
|
|
bytesBufferLimitChan <- struct{}{}
|
|
|
|
// As long as there is an error in the upload of one chunk, it can be terminated early
|
|
// uploadErr may be modified in other go routines, lock is needed to avoid race condition
|
|
uploadErrLock.Lock()
|
|
if uploadErr != nil {
|
|
<-bytesBufferLimitChan
|
|
uploadErrLock.Unlock()
|
|
break
|
|
}
|
|
uploadErrLock.Unlock()
|
|
|
|
bytesBuffer := bufPool.Get().(*bytes.Buffer)
|
|
|
|
limitedReader := io.LimitReader(partReader, int64(chunkSize))
|
|
|
|
bytesBuffer.Reset()
|
|
|
|
dataSize, err := bytesBuffer.ReadFrom(limitedReader)
|
|
|
|
// data, err := io.ReadAll(limitedReader)
|
|
if err != nil || dataSize == 0 {
|
|
bufPool.Put(bytesBuffer)
|
|
<-bytesBufferLimitChan
|
|
if err != nil {
|
|
uploadErrLock.Lock()
|
|
if uploadErr == nil {
|
|
uploadErr = err
|
|
}
|
|
uploadErrLock.Unlock()
|
|
}
|
|
break
|
|
}
|
|
if chunkOffset == 0 && !isAppend {
|
|
if dataSize < fs.option.SaveToFilerLimit {
|
|
chunkOffset += dataSize
|
|
smallContent = make([]byte, dataSize)
|
|
bytesBuffer.Read(smallContent)
|
|
bufPool.Put(bytesBuffer)
|
|
<-bytesBufferLimitChan
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ContentSaveToFiler).Inc()
|
|
break
|
|
}
|
|
} else {
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.AutoChunk).Inc()
|
|
}
|
|
|
|
wg.Add(1)
|
|
go func(offset int64, buf *bytes.Buffer) {
|
|
defer func() {
|
|
bufPool.Put(buf)
|
|
<-bytesBufferLimitChan
|
|
wg.Done()
|
|
}()
|
|
|
|
chunks, toChunkErr := fs.dataToChunkWithSSE(ctx, r, fileName, contentType, buf.Bytes(), offset, so)
|
|
if toChunkErr != nil {
|
|
uploadErrLock.Lock()
|
|
if uploadErr == nil {
|
|
uploadErr = toChunkErr
|
|
}
|
|
uploadErrLock.Unlock()
|
|
}
|
|
if chunks != nil {
|
|
fileChunksLock.Lock()
|
|
for _, chunk := range chunks {
|
|
fileChunks = append(fileChunks, chunk)
|
|
}
|
|
fileChunksLock.Unlock()
|
|
}
|
|
}(chunkOffset, bytesBuffer)
|
|
|
|
// reset variables for the next chunk
|
|
glog.V(4).Infof("uploadReaderToChunks read chunk at offset %d, size %d", chunkOffset, dataSize)
|
|
chunkOffset = chunkOffset + dataSize
|
|
|
|
// if last chunk was not at full chunk size, but already exhausted the reader
|
|
if dataSize < int64(chunkSize) {
|
|
break
|
|
}
|
|
}
|
|
|
|
wg.Wait()
|
|
|
|
if uploadErr != nil {
|
|
glog.V(0).InfofCtx(ctx, "upload file %s error: %v", fileName, uploadErr)
|
|
for _, chunk := range fileChunks {
|
|
glog.V(4).InfofCtx(ctx, "purging failed uploaded %s chunk %s [%d,%d)", fileName, chunk.FileId, chunk.Offset, chunk.Offset+int64(chunk.Size))
|
|
}
|
|
fs.filer.DeleteUncommittedChunks(ctx, fileChunks)
|
|
return nil, md5Hash, 0, uploadErr, nil
|
|
}
|
|
slices.SortFunc(fileChunks, func(a, b *filer_pb.FileChunk) int {
|
|
return int(a.Offset - b.Offset)
|
|
})
|
|
return fileChunks, md5Hash, chunkOffset, nil, smallContent
|
|
}
|
|
|
|
func (fs *FilerServer) doUpload(ctx context.Context, urlLocation string, limitedReader io.Reader, fileName string, contentType string, pairMap map[string]string, auth security.EncodedJwt, contentMd5 string) (*operation.UploadResult, error, []byte) {
|
|
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ChunkUpload).Inc()
|
|
start := time.Now()
|
|
defer func() {
|
|
stats.FilerRequestHistogram.WithLabelValues(stats.ChunkUpload).Observe(time.Since(start).Seconds())
|
|
}()
|
|
|
|
uploadOption := &operation.UploadOption{
|
|
UploadUrl: urlLocation,
|
|
Filename: fileName,
|
|
Cipher: fs.option.Cipher,
|
|
IsInputCompressed: false,
|
|
MimeType: contentType,
|
|
PairMap: pairMap,
|
|
Jwt: auth,
|
|
Md5: contentMd5,
|
|
}
|
|
|
|
uploader, err := operation.NewUploader()
|
|
if err != nil {
|
|
return nil, err, []byte{}
|
|
}
|
|
|
|
// Use a context that ignores cancellation from the request context
|
|
uploadCtx := context.WithoutCancel(ctx)
|
|
|
|
uploadResult, err, data := uploader.Upload(uploadCtx, limitedReader, uploadOption)
|
|
if uploadResult != nil && uploadResult.RetryCount > 0 {
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ChunkUploadRetry).Add(float64(uploadResult.RetryCount))
|
|
}
|
|
return uploadResult, err, data
|
|
}
|
|
|
|
func (fs *FilerServer) dataToChunkWithSSE(ctx context.Context, r *http.Request, fileName, contentType string, data []byte, chunkOffset int64, so *operation.StorageOption) ([]*filer_pb.FileChunk, error) {
|
|
dataReader := util.NewBytesReader(data)
|
|
|
|
// retry to assign a different file id
|
|
var fileId, urlLocation string
|
|
var auth security.EncodedJwt
|
|
var uploadErr error
|
|
var uploadResult *operation.UploadResult
|
|
var failedFileChunks []*filer_pb.FileChunk
|
|
|
|
// Each attempt assigns anew, so also retry the errors a fresh volume dodges:
|
|
// a target gone read-only or full mid-write 5xxs until the master notices.
|
|
shouldRetry := func(err error) bool {
|
|
return util.IsTransientError(err) || operation.ShouldReassignUpload(err)
|
|
}
|
|
err := util.RetryOnError("filerDataToChunk", shouldRetry, func() error {
|
|
// assign one file id for one chunk
|
|
fileId, urlLocation, auth, uploadErr = fs.assignNewFileInfo(ctx, so, uint64(len(data)))
|
|
if uploadErr != nil {
|
|
glog.V(4).InfofCtx(ctx, "retry later due to assign error: %v", uploadErr)
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ChunkAssignRetry).Inc()
|
|
return uploadErr
|
|
}
|
|
chunkMd5 := md5.Sum(data)
|
|
chunkMd5B64 := base64.StdEncoding.EncodeToString(chunkMd5[:])
|
|
// upload the chunk to the volume server
|
|
uploadResult, uploadErr, _ = fs.doUpload(ctx, urlLocation, dataReader, fileName, contentType, nil, auth, chunkMd5B64)
|
|
if uploadErr != nil {
|
|
glog.V(4).InfofCtx(ctx, "retry later due to upload error: %v", uploadErr)
|
|
stats.FilerHandlerCounter.WithLabelValues(stats.ChunkDoUploadRetry).Inc()
|
|
fid, _ := filer_pb.ToFileIdObject(fileId)
|
|
fileChunk := filer_pb.FileChunk{
|
|
FileId: fileId,
|
|
Offset: chunkOffset,
|
|
Fid: fid,
|
|
}
|
|
failedFileChunks = append(failedFileChunks, &fileChunk)
|
|
return uploadErr
|
|
}
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
glog.ErrorfCtx(ctx, "upload error: %v", err)
|
|
return failedFileChunks, err
|
|
}
|
|
|
|
// A retry that lands elsewhere strands the earlier attempts: a volume server
|
|
// 5xxs after storing the needle locally when replication fails, and each
|
|
// attempt used its own file id, so nothing references them now.
|
|
if len(failedFileChunks) > 0 {
|
|
fs.filer.DeleteUncommittedChunks(ctx, failedFileChunks)
|
|
}
|
|
|
|
// if last chunk exhausted the reader exactly at the border
|
|
if uploadResult.Size == 0 {
|
|
return nil, nil
|
|
}
|
|
|
|
// Extract SSE metadata from request headers if available
|
|
var sseType filer_pb.SSEType = filer_pb.SSEType_NONE
|
|
var sseMetadata []byte
|
|
|
|
// Create chunk with SSE metadata if available
|
|
var chunk *filer_pb.FileChunk
|
|
if sseType != filer_pb.SSEType_NONE {
|
|
chunk = uploadResult.ToPbFileChunkWithSSE(fileId, chunkOffset, time.Now().UnixNano(), sseType, sseMetadata)
|
|
} else {
|
|
chunk = uploadResult.ToPbFileChunk(fileId, chunkOffset, time.Now().UnixNano())
|
|
}
|
|
|
|
return []*filer_pb.FileChunk{chunk}, nil
|
|
}
|