mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
admin/maintenance: reload in-flight tasks on startup instead of discarding them (#9857)
* admin/maintenance: reload in-flight tasks on startup instead of discarding LoadTasksFromPersistence deleted all persisted task files on startup and relied on the scanner to re-detect, so saved task state was never consumed — the persistence was effectively write-only. Reload non-terminal tasks (pending/assigned/in_progress) into the queue, resetting in-flight ones to pending since their worker is gone after a restart (maintenance tasks are idempotent). Terminal task files are dropped; the scanner still backfills anything not persisted. * address review: nil-guard reloaded tasks and SyncTask to ActiveTopology - skip nil entries from LoadAllTaskStates (corrupted state) - re-sync restored tasks with MaintenanceIntegration so ActiveTopology (in-memory, empty on startup) knows about them; otherwise GetNextTask's AssignTask rejects them as unknown and they never get assigned
This commit is contained in:
@@ -697,8 +697,7 @@ func (m *MockPersistence) DeleteAllTaskStates() error
|
||||
func (m *MockPersistence) CleanupCompletedTasks() error { return nil }
|
||||
func (m *MockPersistence) SaveTaskPolicy(taskType string, policy *TaskPolicy) error { return nil }
|
||||
|
||||
func TestMaintenanceQueue_LoadTasksStartsEmpty(t *testing.T) {
|
||||
// Setup
|
||||
func TestMaintenanceQueue_LoadRequeuesInFlightTasks(t *testing.T) {
|
||||
policy := &MaintenancePolicy{
|
||||
TaskPolicies: map[string]*worker_pb.TaskPolicy{
|
||||
"balance": {MaxConcurrent: 1},
|
||||
@@ -706,24 +705,30 @@ func TestMaintenanceQueue_LoadTasksStartsEmpty(t *testing.T) {
|
||||
}
|
||||
mq := NewMaintenanceQueue(policy)
|
||||
|
||||
// Setup mock persistence with tasks — these should NOT be loaded
|
||||
mockTask := &MaintenanceTask{
|
||||
ID: "old_task_123",
|
||||
Type: "balance",
|
||||
Status: TaskStatusPending,
|
||||
}
|
||||
mq.SetPersistence(&MockPersistence{tasks: []*MaintenanceTask{mockTask}})
|
||||
// Non-terminal tasks must be re-queued across a restart; an in-progress
|
||||
// task is reset to pending (its worker is gone). Terminal tasks are dropped.
|
||||
mq.SetPersistence(&MockPersistence{tasks: []*MaintenanceTask{
|
||||
{ID: "pending_1", Type: "balance", Status: TaskStatusPending},
|
||||
{ID: "inprogress_1", Type: "balance", Status: TaskStatusInProgress, WorkerID: "gone", Progress: 42},
|
||||
{ID: "done_1", Type: "balance", Status: TaskStatusCompleted},
|
||||
}})
|
||||
|
||||
// LoadTasksFromPersistence should be a no-op — scanner will re-detect
|
||||
err := mq.LoadTasksFromPersistence()
|
||||
if err != nil {
|
||||
if err := mq.LoadTasksFromPersistence(); err != nil {
|
||||
t.Fatalf("LoadTasksFromPersistence failed: %v", err)
|
||||
}
|
||||
|
||||
// Queue should be empty — tasks will be re-detected by scanner
|
||||
stats := mq.GetStats()
|
||||
if stats.TotalTasks != 0 {
|
||||
t.Errorf("Expected 0 tasks after startup, got %d", stats.TotalTasks)
|
||||
if stats := mq.GetStats(); stats.TotalTasks != 2 {
|
||||
t.Errorf("expected 2 re-queued tasks, got %d", stats.TotalTasks)
|
||||
}
|
||||
ip := mq.tasks["inprogress_1"]
|
||||
if ip == nil {
|
||||
t.Fatal("in-progress task was not re-queued")
|
||||
}
|
||||
if ip.Status != TaskStatusPending || ip.WorkerID != "" || ip.Progress != 0 {
|
||||
t.Errorf("in-progress task not reset to pending: status=%q worker=%q progress=%v", ip.Status, ip.WorkerID, ip.Progress)
|
||||
}
|
||||
if mq.tasks["done_1"] != nil {
|
||||
t.Error("completed task should not be re-queued")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user