mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-13 10:00:41 +02:00
* filer: self-heal fetchWholeChunk on stale volume locations Upstream #10156/#10800 wired cache invalidation into the buffer-based read paths, but manifest resolution still goes through fetchWholeChunk, which returns the raw error on failure. When cached volume locations are stale (volume tiered to remote storage, server rolled), resolving a large multipart file fails permanently even though other locations are healthy. Thread the ChunkGroup's cacheInvalidator through ResolveChunkManifest / ResolveOneChunkManifest / fetchWholeChunk, and on failure invalidate, re-lookup and retry once via the existing retryFetchWithFreshLocations helper. The streaming bytesBuffer is reset before the retry so partial bytes from the failed attempt cannot corrupt the manifest proto.Unmarshal. Non-mount callers pass nil and keep their semantics. * filer: move the manifest self-heal tests in with the other manifest tests Also make the stale server stream a prefix and then abort mid-body, which is what actually leaves partial bytes in the buffer: an HTTP error status returns before ReadUrlAsStream ever calls the writer, so a 500 never exercised the Reset the tests claimed to cover. Claude-Session: https://claude.ai/code/session_01FK3oGC5ZVeJYvNBWgb9JUD * filer: keep the cached volume locations when a manifest read is cancelled A cancelled or timed-out read says nothing about where the volume lives, so dropping the location and going back to the master only costs the next reader a round trip. PrepareStreamContentWithThrottler already guards its self-heal this way. The guard also goes inside retryFetchWithFreshLocations, since the caller can be cancelled between its own check and the invalidation, and that covers the reader cache and prefetch paths too. fetchWholeChunk returns the context error rather than the stream failure it provoked, and ResolveOneChunkManifest wraps with %w so errors.Is still sees it. That matters even where no invalidator is passed: volume.fsck resolves manifests with nil and tells its own abort from a corrupt manifest that way, so the cancellation check sits ahead of the nil-invalidator return. Claude-Session: https://claude.ai/code/session_01FK3oGC5ZVeJYvNBWgb9JUD * filer: self-heal manifest reads on the filer and s3 paths too Every caller that already holds the location cache backing its lookup function can hand it over: the filer's read, copy and deletion paths and the log cache have the MasterClient right there, and s3api has the FilerClient. MinusChunks takes one for the same reason, since the deletion path resolves manifests through it. Only the shell tools and the replication sinks, whose lookup functions cache privately with nothing to invalidate, keep passing nil. Claude-Session: https://claude.ai/code/session_01FK3oGC5ZVeJYvNBWgb9JUD --------- Co-authored-by: bruce-zzz <bruce.zou@hhy-data.com>
638 lines
22 KiB
Go
638 lines
22 KiB
Go
package filer
|
|
|
|
import (
|
|
"container/heap"
|
|
"context"
|
|
"fmt"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"google.golang.org/grpc"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/storage"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/wdclient"
|
|
)
|
|
|
|
const (
|
|
// Maximum number of retry attempts for failed deletions
|
|
MaxRetryAttempts = 10
|
|
// Initial retry delay (will be doubled with each attempt)
|
|
InitialRetryDelay = 5 * time.Minute
|
|
// Maximum retry delay
|
|
MaxRetryDelay = 6 * time.Hour
|
|
// Interval for checking retry queue for ready items
|
|
DeletionRetryPollInterval = 1 * time.Minute
|
|
// Maximum number of items to process per retry iteration
|
|
DeletionRetryBatchSize = 1000
|
|
// Maximum number of error details to include in log messages
|
|
MaxLoggedErrorDetails = 10
|
|
// Interval for polling the deletion queue for new items
|
|
// Using a prime number to de-synchronize with other periodic tasks
|
|
DeletionPollInterval = 1123 * time.Millisecond
|
|
// Maximum number of file IDs to delete per batch (roughly 20 bytes per file ID)
|
|
DeletionBatchSize = 100000
|
|
)
|
|
|
|
// retryablePatterns contains transient conditions specific to the deletion
|
|
// pipeline, on top of the network and service failures util.IsTransientErrorMessage
|
|
// already covers.
|
|
//
|
|
// Context cancellation counts as retryable here but not in util: this decides
|
|
// whether to requeue the deletion, not whether to retry a call, and a cancelled
|
|
// batch leaves the file still needing deletion.
|
|
var retryablePatterns = []string{
|
|
"is read only", // Volume temporarily read-only (tiering, maintenance)
|
|
"error reading from server", // Network I/O errors
|
|
"closed network connection", // Network connection closed unexpectedly
|
|
"timeout", // Operation timeout (network or server)
|
|
"deadline exceeded", // Context deadline exceeded
|
|
"context canceled", // Context cancellation
|
|
"lookup error", // Volume lookup failures
|
|
"lookup failed", // Volume server discovery issues
|
|
"too many requests", // Rate limiting / backpressure
|
|
"try again", // Explicit retry suggestion
|
|
}
|
|
|
|
// DeletionRetryItem represents a file deletion that failed and needs to be retried
|
|
type DeletionRetryItem struct {
|
|
FileId string
|
|
RetryCount int
|
|
NextRetryAt time.Time
|
|
LastError string
|
|
heapIndex int // index in the heap (for heap.Interface)
|
|
inFlight bool // true when item is being processed, prevents duplicate additions
|
|
}
|
|
|
|
// retryHeap implements heap.Interface for DeletionRetryItem
|
|
// Items are ordered by NextRetryAt (earliest first)
|
|
type retryHeap []*DeletionRetryItem
|
|
|
|
// Compile-time assertion that retryHeap implements heap.Interface
|
|
var _ heap.Interface = (*retryHeap)(nil)
|
|
|
|
func (h retryHeap) Len() int { return len(h) }
|
|
|
|
func (h retryHeap) Less(i, j int) bool {
|
|
return h[i].NextRetryAt.Before(h[j].NextRetryAt)
|
|
}
|
|
|
|
func (h retryHeap) Swap(i, j int) {
|
|
h[i], h[j] = h[j], h[i]
|
|
h[i].heapIndex = i
|
|
h[j].heapIndex = j
|
|
}
|
|
|
|
func (h *retryHeap) Push(x any) {
|
|
item := x.(*DeletionRetryItem)
|
|
item.heapIndex = len(*h)
|
|
*h = append(*h, item)
|
|
}
|
|
|
|
func (h *retryHeap) Pop() any {
|
|
old := *h
|
|
n := len(old)
|
|
item := old[n-1]
|
|
old[n-1] = nil // avoid memory leak
|
|
item.heapIndex = -1 // mark as removed
|
|
*h = old[0 : n-1]
|
|
return item
|
|
}
|
|
|
|
// DeletionRetryQueue manages the queue of failed deletions that need to be retried.
|
|
// Uses a min-heap ordered by NextRetryAt for efficient retrieval of ready items.
|
|
//
|
|
// LIMITATION: Current implementation stores retry queue in memory only.
|
|
// On filer restart, all pending retries are lost. With MaxRetryDelay up to 6 hours,
|
|
// process restarts during this window will cause retry state loss.
|
|
//
|
|
// TODO: Consider persisting retry queue to durable storage for production resilience:
|
|
// - Option 1: Leverage existing Filer store (KV operations)
|
|
// - Option 2: Periodic snapshots to disk with recovery on startup
|
|
// - Option 3: Write-ahead log for retry queue mutations
|
|
// - Trade-offs: Performance vs durability, complexity vs reliability
|
|
//
|
|
// For now, accepting in-memory storage as pragmatic initial implementation.
|
|
// Lost retries will be eventually consistent as files remain in deletion queue.
|
|
type DeletionRetryQueue struct {
|
|
heap retryHeap
|
|
itemIndex map[string]*DeletionRetryItem // for O(1) lookup by FileId
|
|
lock sync.Mutex
|
|
}
|
|
|
|
// NewDeletionRetryQueue creates a new retry queue
|
|
func NewDeletionRetryQueue() *DeletionRetryQueue {
|
|
q := &DeletionRetryQueue{
|
|
heap: make(retryHeap, 0),
|
|
itemIndex: make(map[string]*DeletionRetryItem),
|
|
}
|
|
heap.Init(&q.heap)
|
|
return q
|
|
}
|
|
|
|
// calculateBackoff calculates the exponential backoff delay for a given retry count.
|
|
// Uses exponential backoff formula: InitialRetryDelay * 2^(retryCount-1)
|
|
// The first retry (retryCount=1) uses InitialRetryDelay, second uses 2x, third uses 4x, etc.
|
|
// Includes overflow protection and caps at MaxRetryDelay.
|
|
func calculateBackoff(retryCount int) time.Duration {
|
|
// The first retry is attempt 1, but shift should start at 0
|
|
if retryCount <= 1 {
|
|
return InitialRetryDelay
|
|
}
|
|
|
|
shiftAmount := uint(retryCount - 1)
|
|
|
|
// time.Duration is an int64. A left shift of 63 or more will result in a
|
|
// negative number or zero. The multiplication can also overflow much earlier
|
|
// (around a shift of 25 for a 5-minute initial delay).
|
|
// The `delay <= 0` check below correctly catches all these overflow cases.
|
|
delay := InitialRetryDelay << shiftAmount
|
|
|
|
if delay <= 0 || delay > MaxRetryDelay {
|
|
return MaxRetryDelay
|
|
}
|
|
|
|
return delay
|
|
}
|
|
|
|
// AddOrUpdate adds a new failed deletion or updates an existing one
|
|
// Time complexity: O(log N) for insertion/update
|
|
func (q *DeletionRetryQueue) AddOrUpdate(fileId string, errorMsg string) {
|
|
q.lock.Lock()
|
|
defer q.lock.Unlock()
|
|
|
|
// Check if item already exists (including in-flight items)
|
|
if item, exists := q.itemIndex[fileId]; exists {
|
|
// Item is already in the queue or being processed. Just update the error.
|
|
// The existing retry schedule should proceed.
|
|
// RetryCount is only incremented in RequeueForRetry when an actual retry is performed.
|
|
item.LastError = errorMsg
|
|
if item.inFlight {
|
|
glog.V(2).Infof("retry for %s in-flight: attempt %d, will preserve retry state", fileId, item.RetryCount)
|
|
} else {
|
|
glog.V(2).Infof("retry for %s already scheduled: attempt %d, next retry in %v", fileId, item.RetryCount, time.Until(item.NextRetryAt))
|
|
}
|
|
return
|
|
}
|
|
|
|
// Add new item
|
|
delay := InitialRetryDelay
|
|
item := &DeletionRetryItem{
|
|
FileId: fileId,
|
|
RetryCount: 1,
|
|
NextRetryAt: time.Now().Add(delay),
|
|
LastError: errorMsg,
|
|
inFlight: false,
|
|
}
|
|
heap.Push(&q.heap, item)
|
|
q.itemIndex[fileId] = item
|
|
glog.V(2).Infof("added retry for %s: next retry in %v", fileId, delay)
|
|
}
|
|
|
|
// RequeueForRetry re-adds a previously failed item back to the queue with incremented retry count.
|
|
// This method MUST be used when re-queuing items from processRetryBatch to preserve retry state.
|
|
// Time complexity: O(log N) for insertion
|
|
func (q *DeletionRetryQueue) RequeueForRetry(item *DeletionRetryItem, errorMsg string) {
|
|
q.lock.Lock()
|
|
defer q.lock.Unlock()
|
|
|
|
// Increment retry count
|
|
item.RetryCount++
|
|
item.LastError = errorMsg
|
|
|
|
// Calculate next retry time with exponential backoff
|
|
delay := calculateBackoff(item.RetryCount)
|
|
item.NextRetryAt = time.Now().Add(delay)
|
|
item.inFlight = false // Clear in-flight flag
|
|
glog.V(2).Infof("requeued retry for %s: attempt %d, next retry in %v", item.FileId, item.RetryCount, delay)
|
|
|
|
// Re-add to heap (item still in itemIndex)
|
|
heap.Push(&q.heap, item)
|
|
}
|
|
|
|
// GetReadyItems returns items that are ready to be retried and marks them as in-flight
|
|
// Time complexity: O(K log N) where K is the number of ready items
|
|
// Items are processed in order of NextRetryAt (earliest first)
|
|
func (q *DeletionRetryQueue) GetReadyItems(maxItems int) []*DeletionRetryItem {
|
|
q.lock.Lock()
|
|
defer q.lock.Unlock()
|
|
|
|
now := time.Now()
|
|
var readyItems []*DeletionRetryItem
|
|
|
|
// Peek at items from the top of the heap (earliest NextRetryAt)
|
|
for len(q.heap) > 0 && len(readyItems) < maxItems {
|
|
item := q.heap[0]
|
|
|
|
// If the earliest item is not ready yet, no other items are ready either
|
|
if item.NextRetryAt.After(now) {
|
|
break
|
|
}
|
|
|
|
// Remove from heap but keep in itemIndex with inFlight flag
|
|
heap.Pop(&q.heap)
|
|
|
|
if item.RetryCount <= MaxRetryAttempts {
|
|
item.inFlight = true // Mark as being processed
|
|
readyItems = append(readyItems, item)
|
|
} else {
|
|
// Max attempts reached, log and discard completely
|
|
delete(q.itemIndex, item.FileId)
|
|
glog.Warningf("max retry attempts (%d) reached for %s, last error: %s", MaxRetryAttempts, item.FileId, item.LastError)
|
|
}
|
|
}
|
|
|
|
return readyItems
|
|
}
|
|
|
|
// Remove removes an item from the queue (called when deletion succeeds or fails permanently)
|
|
// Time complexity: O(1)
|
|
func (q *DeletionRetryQueue) Remove(item *DeletionRetryItem) {
|
|
q.lock.Lock()
|
|
defer q.lock.Unlock()
|
|
|
|
// Item was already removed from heap by GetReadyItems, just remove from index
|
|
delete(q.itemIndex, item.FileId)
|
|
}
|
|
|
|
// Size returns the current size of the retry queue
|
|
func (q *DeletionRetryQueue) Size() int {
|
|
q.lock.Lock()
|
|
defer q.lock.Unlock()
|
|
return len(q.heap)
|
|
}
|
|
|
|
func LookupByMasterClientFn(masterClient *wdclient.MasterClient) func(vids []string) (map[string]*operation.LookupResult, error) {
|
|
return func(vids []string) (map[string]*operation.LookupResult, error) {
|
|
m := make(map[string]*operation.LookupResult)
|
|
for _, vid := range vids {
|
|
locs, _ := masterClient.GetVidLocations(vid)
|
|
var locations []operation.Location
|
|
for _, loc := range locs {
|
|
locations = append(locations, operation.Location{
|
|
Url: loc.Url,
|
|
PublicUrl: loc.PublicUrl,
|
|
GrpcPort: loc.GrpcPort,
|
|
})
|
|
}
|
|
m[vid] = &operation.LookupResult{
|
|
VolumeOrFileId: vid,
|
|
Locations: locations,
|
|
}
|
|
}
|
|
return m, nil
|
|
}
|
|
}
|
|
|
|
func (f *Filer) loopProcessingDeletion() {
|
|
|
|
lookupFunc := LookupByMasterClientFn(f.MasterClient)
|
|
|
|
// Start retry processor in a separate goroutine
|
|
go f.loopProcessingDeletionRetry(lookupFunc)
|
|
|
|
ticker := time.NewTicker(DeletionPollInterval)
|
|
defer ticker.Stop()
|
|
|
|
for {
|
|
select {
|
|
case <-f.deletionQuit:
|
|
glog.V(0).Infof("deletion processor shutting down")
|
|
return
|
|
case <-ticker.C:
|
|
f.FileIdDeletionQueue.Consume(func(fileIds []string) {
|
|
for i := 0; i < len(fileIds); i += DeletionBatchSize {
|
|
end := i + DeletionBatchSize
|
|
if end > len(fileIds) {
|
|
end = len(fileIds)
|
|
}
|
|
toDeleteFileIds := fileIds[i:end]
|
|
f.processDeletionBatch(toDeleteFileIds, lookupFunc)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
}
|
|
|
|
// processDeletionBatch handles deletion of a batch of file IDs and processes results.
|
|
// It classifies errors into retryable and permanent categories, adds retryable failures
|
|
// to the retry queue, and logs appropriate messages.
|
|
func (f *Filer) processDeletionBatch(toDeleteFileIds []string, lookupFunc func([]string) (map[string]*operation.LookupResult, error)) {
|
|
// Deduplicate file IDs to prevent incorrect retry count increments for the same file ID within a single batch.
|
|
uniqueFileIdsSlice := make([]string, 0, len(toDeleteFileIds))
|
|
processed := make(map[string]struct{}, len(toDeleteFileIds))
|
|
for _, fileId := range toDeleteFileIds {
|
|
if _, found := processed[fileId]; !found {
|
|
processed[fileId] = struct{}{}
|
|
uniqueFileIdsSlice = append(uniqueFileIdsSlice, fileId)
|
|
}
|
|
}
|
|
|
|
if len(uniqueFileIdsSlice) == 0 {
|
|
return
|
|
}
|
|
|
|
// Delete files and classify outcomes
|
|
outcomes := deleteFilesAndClassify(f.GrpcDialOption, uniqueFileIdsSlice, lookupFunc)
|
|
|
|
// Process outcomes
|
|
var successCount, notFoundCount, retryableErrorCount, permanentErrorCount int
|
|
var errorDetails []string
|
|
|
|
for _, fileId := range uniqueFileIdsSlice {
|
|
outcome := outcomes[fileId]
|
|
|
|
switch outcome.status {
|
|
case deletionOutcomeSuccess:
|
|
successCount++
|
|
case deletionOutcomeNotFound:
|
|
notFoundCount++
|
|
case deletionOutcomeRetryable, deletionOutcomeNoResult:
|
|
retryableErrorCount++
|
|
f.DeletionRetryQueue.AddOrUpdate(fileId, outcome.errorMsg)
|
|
if len(errorDetails) < MaxLoggedErrorDetails {
|
|
errorDetails = append(errorDetails, fileId+": "+outcome.errorMsg+" (will retry)")
|
|
}
|
|
case deletionOutcomePermanent:
|
|
permanentErrorCount++
|
|
if len(errorDetails) < MaxLoggedErrorDetails {
|
|
errorDetails = append(errorDetails, fileId+": "+outcome.errorMsg+" (permanent)")
|
|
}
|
|
}
|
|
}
|
|
|
|
if successCount > 0 || notFoundCount > 0 {
|
|
glog.V(2).Infof("deleted %d files successfully, %d already deleted (not found)", successCount, notFoundCount)
|
|
}
|
|
|
|
totalErrors := retryableErrorCount + permanentErrorCount
|
|
if totalErrors > 0 {
|
|
logMessage := fmt.Sprintf("failed to delete %d/%d files (%d retryable, %d permanent)",
|
|
totalErrors, len(uniqueFileIdsSlice), retryableErrorCount, permanentErrorCount)
|
|
if len(errorDetails) > 0 {
|
|
if totalErrors > MaxLoggedErrorDetails {
|
|
logMessage += fmt.Sprintf(" (showing first %d)", len(errorDetails))
|
|
}
|
|
glog.V(0).Infof("%s: %v", logMessage, strings.Join(errorDetails, "; "))
|
|
} else {
|
|
glog.V(0).Info(logMessage)
|
|
}
|
|
}
|
|
|
|
if f.DeletionRetryQueue.Size() > 0 {
|
|
glog.V(2).Infof("retry queue size: %d", f.DeletionRetryQueue.Size())
|
|
}
|
|
}
|
|
|
|
const (
|
|
deletionOutcomeSuccess = "success"
|
|
deletionOutcomeNotFound = "not_found"
|
|
deletionOutcomeRetryable = "retryable"
|
|
deletionOutcomePermanent = "permanent"
|
|
deletionOutcomeNoResult = "no_result"
|
|
)
|
|
|
|
// deletionOutcome represents the result of classifying deletion results for a file
|
|
type deletionOutcome struct {
|
|
status string // One of the deletionOutcome* constants
|
|
errorMsg string
|
|
}
|
|
|
|
// deleteFilesAndClassify performs deletion and classifies outcomes for a list of file IDs
|
|
func deleteFilesAndClassify(grpcDialOption grpc.DialOption, fileIds []string, lookupFunc func([]string) (map[string]*operation.LookupResult, error)) map[string]deletionOutcome {
|
|
// Perform deletion
|
|
results := operation.DeleteFileIdsWithLookupVolumeId(grpcDialOption, fileIds, lookupFunc)
|
|
|
|
// Group results by file ID to handle multiple results for replicated volumes
|
|
resultsByFileId := make(map[string][]*volume_server_pb.DeleteResult)
|
|
for _, result := range results {
|
|
resultsByFileId[result.FileId] = append(resultsByFileId[result.FileId], result)
|
|
}
|
|
|
|
// Classify outcome for each file
|
|
outcomes := make(map[string]deletionOutcome, len(fileIds))
|
|
for _, fileId := range fileIds {
|
|
outcomes[fileId] = classifyDeletionOutcome(fileId, resultsByFileId)
|
|
}
|
|
|
|
return outcomes
|
|
}
|
|
|
|
// classifyDeletionOutcome examines all deletion results for a file ID and determines the overall outcome
|
|
// Uses a single pass through results with early return for permanent errors (highest priority)
|
|
// Priority: Permanent > Retryable > Success > Not Found
|
|
func classifyDeletionOutcome(fileId string, resultsByFileId map[string][]*volume_server_pb.DeleteResult) deletionOutcome {
|
|
fileIdResults, found := resultsByFileId[fileId]
|
|
if !found || len(fileIdResults) == 0 {
|
|
return deletionOutcome{
|
|
status: deletionOutcomeNoResult,
|
|
errorMsg: "no deletion result from volume server",
|
|
}
|
|
}
|
|
|
|
var firstRetryableError string
|
|
hasSuccess := false
|
|
|
|
for _, res := range fileIdResults {
|
|
if res.Error == "" {
|
|
hasSuccess = true
|
|
continue
|
|
}
|
|
if strings.Contains(res.Error, storage.ErrorDeleted.Error()) || res.Error == "not found" {
|
|
continue
|
|
}
|
|
|
|
if isRetryableError(res.Error) {
|
|
if firstRetryableError == "" {
|
|
firstRetryableError = res.Error
|
|
}
|
|
} else {
|
|
// Permanent error takes highest precedence - return immediately
|
|
return deletionOutcome{status: deletionOutcomePermanent, errorMsg: res.Error}
|
|
}
|
|
}
|
|
|
|
if firstRetryableError != "" {
|
|
return deletionOutcome{status: deletionOutcomeRetryable, errorMsg: firstRetryableError}
|
|
}
|
|
|
|
if hasSuccess {
|
|
return deletionOutcome{status: deletionOutcomeSuccess, errorMsg: ""}
|
|
}
|
|
|
|
// If we are here, all results were "not found"
|
|
return deletionOutcome{status: deletionOutcomeNotFound, errorMsg: ""}
|
|
}
|
|
|
|
// isRetryableError determines if an error is retryable based on its message.
|
|
//
|
|
// Current implementation uses string matching which is brittle and may break
|
|
// if error messages change in dependencies. This is acceptable for the initial
|
|
// implementation but should be improved in the future.
|
|
//
|
|
// TODO: Consider these improvements for more robust error handling:
|
|
// - Pass DeleteResult instead of just error string to access Status codes
|
|
// - Use HTTP status codes (503 Service Unavailable, 429 Too Many Requests, etc.)
|
|
// - Implement structured error types that can be checked with errors.Is/errors.As
|
|
// - Extract and check gRPC status codes for better classification
|
|
// - Add error wrapping in the deletion pipeline to preserve error context
|
|
//
|
|
// For now, we use conservative string matching for known transient error patterns.
|
|
func isRetryableError(errorMsg string) bool {
|
|
// Empty errors are not retryable
|
|
if errorMsg == "" {
|
|
return false
|
|
}
|
|
|
|
if util.IsTransientErrorMessage(errorMsg) {
|
|
return true
|
|
}
|
|
|
|
errorLower := strings.ToLower(errorMsg)
|
|
for _, pattern := range retryablePatterns {
|
|
if strings.Contains(errorLower, pattern) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// loopProcessingDeletionRetry processes the retry queue for failed deletions
|
|
func (f *Filer) loopProcessingDeletionRetry(lookupFunc func([]string) (map[string]*operation.LookupResult, error)) {
|
|
|
|
ticker := time.NewTicker(DeletionRetryPollInterval)
|
|
defer ticker.Stop()
|
|
|
|
for {
|
|
select {
|
|
case <-f.deletionQuit:
|
|
glog.V(0).Infof("retry processor shutting down, %d items remaining in queue", f.DeletionRetryQueue.Size())
|
|
return
|
|
case <-ticker.C:
|
|
// Process all ready items in batches until queue is empty
|
|
totalProcessed := 0
|
|
for {
|
|
readyItems := f.DeletionRetryQueue.GetReadyItems(DeletionRetryBatchSize)
|
|
if len(readyItems) == 0 {
|
|
break
|
|
}
|
|
|
|
f.processRetryBatch(readyItems, lookupFunc)
|
|
totalProcessed += len(readyItems)
|
|
}
|
|
|
|
if totalProcessed > 0 {
|
|
glog.V(1).Infof("retried deletion of %d files", totalProcessed)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// processRetryBatch attempts to retry deletion of files and processes results.
|
|
// Successfully deleted items are removed from tracking, retryable failures are
|
|
// re-queued with updated retry counts, and permanent errors are logged and discarded.
|
|
func (f *Filer) processRetryBatch(readyItems []*DeletionRetryItem, lookupFunc func([]string) (map[string]*operation.LookupResult, error)) {
|
|
// Extract file IDs from retry items
|
|
fileIds := make([]string, 0, len(readyItems))
|
|
for _, item := range readyItems {
|
|
fileIds = append(fileIds, item.FileId)
|
|
}
|
|
|
|
// Delete files and classify outcomes
|
|
outcomes := deleteFilesAndClassify(f.GrpcDialOption, fileIds, lookupFunc)
|
|
|
|
// Process outcomes - iterate over readyItems to ensure all items are accounted for
|
|
var successCount, notFoundCount, retryCount, permanentErrorCount int
|
|
for _, item := range readyItems {
|
|
outcome := outcomes[item.FileId]
|
|
|
|
switch outcome.status {
|
|
case deletionOutcomeSuccess:
|
|
successCount++
|
|
f.DeletionRetryQueue.Remove(item) // Remove from queue (success)
|
|
glog.V(2).Infof("retry successful for %s after %d attempts", item.FileId, item.RetryCount)
|
|
case deletionOutcomeNotFound:
|
|
notFoundCount++
|
|
f.DeletionRetryQueue.Remove(item) // Remove from queue (already deleted)
|
|
case deletionOutcomeRetryable, deletionOutcomeNoResult:
|
|
retryCount++
|
|
if outcome.status == deletionOutcomeNoResult {
|
|
glog.Warningf("no deletion result for retried file %s, re-queuing to avoid loss", item.FileId)
|
|
}
|
|
f.DeletionRetryQueue.RequeueForRetry(item, outcome.errorMsg)
|
|
case deletionOutcomePermanent:
|
|
permanentErrorCount++
|
|
f.DeletionRetryQueue.Remove(item) // Remove from queue (permanent failure)
|
|
glog.Warningf("permanent error on retry for %s after %d attempts: %s", item.FileId, item.RetryCount, outcome.errorMsg)
|
|
}
|
|
}
|
|
|
|
if successCount > 0 || notFoundCount > 0 {
|
|
glog.V(1).Infof("retry: deleted %d files successfully, %d already deleted", successCount, notFoundCount)
|
|
}
|
|
if retryCount > 0 {
|
|
glog.V(1).Infof("retry: %d files still failing, will retry again later", retryCount)
|
|
}
|
|
if permanentErrorCount > 0 {
|
|
glog.Warningf("retry: %d files failed with permanent errors", permanentErrorCount)
|
|
}
|
|
}
|
|
|
|
func (f *Filer) DeleteUncommittedChunks(ctx context.Context, chunks []*filer_pb.FileChunk) {
|
|
f.doDeleteChunks(ctx, chunks)
|
|
}
|
|
|
|
func (f *Filer) DeleteChunks(ctx context.Context, fullpath util.FullPath, chunks []*filer_pb.FileChunk) {
|
|
rule := f.FilerConf.MatchStorageRule(string(fullpath))
|
|
if rule.DisableChunkDeletion {
|
|
return
|
|
}
|
|
f.doDeleteChunks(ctx, chunks)
|
|
}
|
|
|
|
func (f *Filer) doDeleteChunks(ctx context.Context, chunks []*filer_pb.FileChunk) {
|
|
for _, chunk := range chunks {
|
|
if !chunk.IsChunkManifest {
|
|
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
|
continue
|
|
}
|
|
dataChunks, manifestResolveErr := ResolveOneChunkManifest(ctx, f.MasterClient.LookupFileId, chunk, f.MasterClient)
|
|
if manifestResolveErr != nil {
|
|
glog.V(0).InfofCtx(ctx, "failed to resolve manifest %s: %v", chunk.FileId, manifestResolveErr)
|
|
}
|
|
for _, dChunk := range dataChunks {
|
|
f.FileIdDeletionQueue.EnQueue(dChunk.GetFileIdString())
|
|
}
|
|
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
|
}
|
|
}
|
|
|
|
func (f *Filer) DeleteChunksNotRecursive(chunks []*filer_pb.FileChunk) {
|
|
for _, chunk := range chunks {
|
|
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
|
}
|
|
}
|
|
|
|
func (f *Filer) deleteChunksIfNotNew(ctx context.Context, oldEntry, newEntry *Entry) {
|
|
var oldChunks, newChunks []*filer_pb.FileChunk
|
|
if oldEntry != nil {
|
|
oldChunks = oldEntry.GetChunks()
|
|
}
|
|
if newEntry != nil {
|
|
newChunks = newEntry.GetChunks()
|
|
}
|
|
|
|
toDelete, err := MinusChunks(ctx, f.MasterClient.GetLookupFileIdFunction(), oldChunks, newChunks, f.MasterClient)
|
|
if err != nil {
|
|
glog.ErrorfCtx(ctx, "Failed to resolve old entry chunks when delete old entry chunks. new: %s, old: %s", newChunks, oldChunks)
|
|
return
|
|
}
|
|
f.DeleteChunksNotRecursive(toDelete)
|
|
}
|