mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 14:57:48 +02:00
* admin: delete persisted state when scan cancels pending tasks Each detection cycle cancels every pending task of a type before re-detecting it, and the cancel path saved the cancelled task back to disk. Nothing ever removed those files, so -dataDir/tasks gained one orphaned .pb per candidate volume per scan cycle. Cancelled is terminal, so drop the file the same way CompleteTask does for completed/failed tasks. The cancelled entry stays in memory for the UI until the next purge. Refs #11595 * admin: delete persisted state when CancelTask cancels a pending task The manual cancel path only updated memory, leaving the pending .pb on disk where a restart would resurrect the cancelled task as pending and the file would linger until then. Delete it like the scan-cycle cancel path now does. * admin: count cancelled tasks toward task retention cleanup CleanupOldTasks and ConfigPersistence.CleanupCompletedTasks only filtered completed/failed tasks, so cancelled entries were exempt from retention in both memory and on disk. Treat all terminal states alike; nil CompletedAt entries also count and sort last, so they are pruned first. * admin: run task file retention in the periodic cleanup loop cleanupCompletedTasks had no callers, so the on-disk retention bound never ran during uptime. Invoke it from performCleanup alongside the in-memory CleanupOldTasks sweep. * admin: guard task state writes against stale saves and failed deletes saveTaskState runs after mq.mutex is released, so the task may have gone terminal in between; a delayed pending save could then recreate the file a cancel just deleted and resurrect the task on restart. Skip saving non-terminal snapshots once the live task is terminal or gone. If a cancel file removal fails, fall back to writing the cancelled snapshot so the file is terminal rather than pending. deleteTaskState now returns its error, and CancelTask captures task.Status while still holding the queue lock. * admin: serialize task file check+write against cancel deletes The saveTaskState guard still had a check-then-write window: a pending snapshot could pass the terminal check before a cancel deleted the file, then write it back after. A persistMu on the queue now covers the check+save and the cancel paths' delete (with its terminal-state fallback), so the two cannot interleave for the same task.
688 lines
23 KiB
Go
688 lines
23 KiB
Go
package maintenance
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
|
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
|
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
|
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/ec_balance"
|
|
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
|
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
|
"github.com/seaweedfs/seaweedfs/weed/worker/types"
|
|
)
|
|
|
|
// BuildPolicyFromTaskConfigs loads each registered task's configuration and builds the
|
|
// MaintenancePolicy the maintenance system runs on.
|
|
//
|
|
// Every entry is produced by the task's own ToTaskPolicy(), so a policy entry always carries
|
|
// exactly what that task's config holds. Hand-copying the fields here instead made this a
|
|
// second, silently diverging definition of every task's policy: it dropped the erasure
|
|
// coding preferred tags and replica placement and the balance IO rate limit outright.
|
|
//
|
|
// Every task registered through base.RegisterTask needs an entry, because a task type the
|
|
// policy does not list has no enabled flag and no concurrency limit of its own -
|
|
// IsTaskEnabled reports false for a missing entry.
|
|
//
|
|
// configPersistence is duck-typed as interface{} because weed/admin/dash already imports this
|
|
// package, so importing *dash.ConfigPersistence back here would create an import cycle. It must be
|
|
// a value implementing the LoadXTaskPolicy() accessors the task loaders assert on; passing nil (or
|
|
// anything else) makes every task fall back to its compiled-in defaults.
|
|
func BuildPolicyFromTaskConfigs(configPersistence interface{}) *worker_pb.MaintenancePolicy {
|
|
policy := &worker_pb.MaintenancePolicy{
|
|
GlobalMaxConcurrent: 4,
|
|
DefaultRepeatIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
|
DefaultCheckIntervalSeconds: 12 * 3600, // 12 hours in seconds
|
|
TaskPolicies: make(map[string]*worker_pb.TaskPolicy),
|
|
}
|
|
|
|
if vacuumConfig := vacuum.LoadConfigFromPersistence(configPersistence); vacuumConfig != nil {
|
|
policy.TaskPolicies[string(types.TaskTypeVacuum)] = vacuumConfig.ToTaskPolicy()
|
|
}
|
|
|
|
if ecConfig := erasure_coding.LoadConfigFromPersistence(configPersistence); ecConfig != nil {
|
|
policy.TaskPolicies[string(types.TaskTypeErasureCoding)] = ecConfig.ToTaskPolicy()
|
|
}
|
|
|
|
if balanceConfig := balance.LoadConfigFromPersistence(configPersistence); balanceConfig != nil {
|
|
policy.TaskPolicies[string(types.TaskTypeBalance)] = balanceConfig.ToTaskPolicy()
|
|
}
|
|
|
|
if ecBalanceConfig := ec_balance.LoadConfigFromPersistence(configPersistence); ecBalanceConfig != nil {
|
|
policy.TaskPolicies[string(types.TaskTypeECBalance)] = ecBalanceConfig.ToTaskPolicy()
|
|
}
|
|
|
|
glog.V(1).Infof("Built maintenance policy from separate task configs - %d task policies loaded", len(policy.TaskPolicies))
|
|
return policy
|
|
}
|
|
|
|
// MaintenanceManager coordinates the maintenance system
|
|
type MaintenanceManager struct {
|
|
config *MaintenanceConfig
|
|
scanner *MaintenanceScanner
|
|
queue *MaintenanceQueue
|
|
adminClient AdminClient
|
|
running bool
|
|
stopChan chan struct{}
|
|
// Error handling and backoff
|
|
errorCount int
|
|
lastError error
|
|
lastErrorTime time.Time
|
|
backoffDelay time.Duration
|
|
mutex sync.RWMutex
|
|
scanInProgress bool
|
|
}
|
|
|
|
// NewMaintenanceManager creates a new maintenance manager.
|
|
//
|
|
// configPersistence is the config store to read persisted task configs from when the policy has to
|
|
// be built here. See BuildPolicyFromTaskConfigs for why it is duck-typed; pass nil when no config
|
|
// store is available.
|
|
func NewMaintenanceManager(adminClient AdminClient, config *MaintenanceConfig, configPersistence interface{}) *MaintenanceManager {
|
|
if config == nil {
|
|
config = DefaultMaintenanceConfig()
|
|
}
|
|
|
|
// Use the policy from the config (which is populated from separate task files in LoadMaintenanceConfig)
|
|
policy := config.Policy
|
|
if policy == nil {
|
|
// Fallback: build policy from separate task configuration files if not already populated
|
|
policy = BuildPolicyFromTaskConfigs(configPersistence)
|
|
}
|
|
|
|
queue := NewMaintenanceQueue(policy)
|
|
scanner := NewMaintenanceScanner(adminClient, policy, queue)
|
|
|
|
return &MaintenanceManager{
|
|
config: config,
|
|
scanner: scanner,
|
|
queue: queue,
|
|
adminClient: adminClient,
|
|
stopChan: make(chan struct{}),
|
|
backoffDelay: time.Second, // Start with 1 second backoff
|
|
}
|
|
}
|
|
|
|
// Start begins the maintenance manager
|
|
func (mm *MaintenanceManager) Start() error {
|
|
if !mm.config.GetEnabled() {
|
|
glog.V(1).Infof("Maintenance system is disabled")
|
|
return nil
|
|
}
|
|
|
|
// Validate configuration durations to prevent ticker panics
|
|
if err := mm.validateConfig(); err != nil {
|
|
return fmt.Errorf("invalid maintenance configuration: %w", err)
|
|
}
|
|
|
|
mm.mutex.Lock()
|
|
mm.running = true
|
|
mm.mutex.Unlock()
|
|
|
|
// Start background processes
|
|
go mm.scanLoop()
|
|
go mm.cleanupLoop()
|
|
go mm.topologyStatusLoop() // Periodic diagnostic logging
|
|
|
|
glog.Infof("Maintenance manager started with scan interval %ds", mm.config.ScanIntervalSeconds)
|
|
return nil
|
|
}
|
|
|
|
// validateConfig validates the maintenance configuration durations
|
|
func (mm *MaintenanceManager) validateConfig() error {
|
|
if mm.config.ScanIntervalSeconds <= 0 {
|
|
glog.Warningf("Invalid scan interval %ds, using default 30m", mm.config.ScanIntervalSeconds)
|
|
mm.config.ScanIntervalSeconds = 30 * 60 // 30 minutes in seconds
|
|
}
|
|
|
|
if mm.config.CleanupIntervalSeconds <= 0 {
|
|
glog.Warningf("Invalid cleanup interval %ds, using default 24h", mm.config.CleanupIntervalSeconds)
|
|
mm.config.CleanupIntervalSeconds = 24 * 60 * 60 // 24 hours in seconds
|
|
}
|
|
|
|
if mm.config.WorkerTimeoutSeconds <= 0 {
|
|
glog.Warningf("Invalid worker timeout %ds, using default 5m", mm.config.WorkerTimeoutSeconds)
|
|
mm.config.WorkerTimeoutSeconds = 5 * 60 // 5 minutes in seconds
|
|
}
|
|
|
|
if mm.config.TaskTimeoutSeconds <= 0 {
|
|
glog.Warningf("Invalid task timeout %ds, using default 2h", mm.config.TaskTimeoutSeconds)
|
|
mm.config.TaskTimeoutSeconds = 2 * 60 * 60 // 2 hours in seconds
|
|
}
|
|
|
|
if mm.config.RetryDelaySeconds <= 0 {
|
|
glog.Warningf("Invalid retry delay %ds, using default 15m", mm.config.RetryDelaySeconds)
|
|
mm.config.RetryDelaySeconds = 15 * 60 // 15 minutes in seconds
|
|
}
|
|
|
|
if mm.config.TaskRetentionSeconds <= 0 {
|
|
glog.Warningf("Invalid task retention %ds, using default 168h", mm.config.TaskRetentionSeconds)
|
|
mm.config.TaskRetentionSeconds = 7 * 24 * 60 * 60 // 7 days in seconds
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// IsRunning returns whether the maintenance manager is currently running.
|
|
// running is guarded by mm.mutex because the background loops read it on every
|
|
// iteration while Start and Stop are called from the admin server's goroutines.
|
|
func (mm *MaintenanceManager) IsRunning() bool {
|
|
mm.mutex.RLock()
|
|
defer mm.mutex.RUnlock()
|
|
return mm.running
|
|
}
|
|
|
|
// Stop terminates the maintenance manager. It is a no-op when the manager is not
|
|
// running, so a second call cannot close the already closed stop channel.
|
|
func (mm *MaintenanceManager) Stop() {
|
|
mm.mutex.Lock()
|
|
if !mm.running {
|
|
mm.mutex.Unlock()
|
|
return
|
|
}
|
|
mm.running = false
|
|
close(mm.stopChan)
|
|
mm.mutex.Unlock()
|
|
|
|
glog.Infof("Maintenance manager stopped")
|
|
}
|
|
|
|
// scanLoop periodically scans for maintenance tasks with adaptive timing
|
|
func (mm *MaintenanceManager) scanLoop() {
|
|
scanInterval := time.Duration(mm.config.ScanIntervalSeconds) * time.Second
|
|
|
|
// activeInterval is the interval the ticker is actually running at right now. It has
|
|
// to be tracked separately from the configured scanInterval, because the error backoff
|
|
// replaces the ticker with a much shorter one. Comparing the target against the
|
|
// configured interval instead never restores the normal cadence: once the errors stop,
|
|
// getScanInterval returns scanInterval again, the comparison comes out false, and the
|
|
// ticker is left at the backoff delay. A single transient scan failure therefore pinned
|
|
// the scanner to one scan per second forever.
|
|
activeInterval := scanInterval
|
|
ticker := time.NewTicker(activeInterval)
|
|
// Wrapped in a closure so the replacement ticker is stopped, not the one that happened
|
|
// to be current when the defer was registered.
|
|
defer func() { ticker.Stop() }()
|
|
|
|
for mm.IsRunning() {
|
|
select {
|
|
case <-mm.stopChan:
|
|
return
|
|
case <-ticker.C:
|
|
glog.V(1).Infof("Performing maintenance scan every %v", activeInterval)
|
|
|
|
// Use the same synchronization as TriggerScan to prevent concurrent scans
|
|
if err := mm.triggerScanInternal(false); err != nil {
|
|
glog.V(1).Infof("Scheduled scan skipped: %v", err)
|
|
}
|
|
|
|
// Adjust ticker interval based on error state (read error state safely)
|
|
currentInterval := mm.getScanInterval(scanInterval)
|
|
|
|
// Reset ticker whenever the target differs from what the ticker is running at,
|
|
// which covers both entering the backoff and returning to the normal cadence.
|
|
if currentInterval != activeInterval {
|
|
ticker.Stop()
|
|
ticker = time.NewTicker(currentInterval)
|
|
glog.V(1).Infof("Maintenance scan cadence changed from %v to %v", activeInterval, currentInterval)
|
|
activeInterval = currentInterval
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// getScanInterval safely reads the current scan interval with error backoff
|
|
func (mm *MaintenanceManager) getScanInterval(baseInterval time.Duration) time.Duration {
|
|
mm.mutex.RLock()
|
|
defer mm.mutex.RUnlock()
|
|
|
|
if mm.errorCount > 0 {
|
|
// Use backoff delay when there are errors
|
|
currentInterval := mm.backoffDelay
|
|
if currentInterval > baseInterval {
|
|
// Don't make it longer than the configured interval * 10
|
|
maxInterval := baseInterval * 10
|
|
if currentInterval > maxInterval {
|
|
currentInterval = maxInterval
|
|
}
|
|
}
|
|
return currentInterval
|
|
}
|
|
return baseInterval
|
|
}
|
|
|
|
// cleanupLoop periodically cleans up old tasks and stale workers
|
|
func (mm *MaintenanceManager) cleanupLoop() {
|
|
cleanupInterval := time.Duration(mm.config.CleanupIntervalSeconds) * time.Second
|
|
ticker := time.NewTicker(cleanupInterval)
|
|
defer ticker.Stop()
|
|
|
|
for mm.IsRunning() {
|
|
select {
|
|
case <-mm.stopChan:
|
|
return
|
|
case <-ticker.C:
|
|
mm.performCleanup()
|
|
}
|
|
}
|
|
}
|
|
|
|
// topologyStatusLoop periodically logs topology status for diagnostics
|
|
func (mm *MaintenanceManager) topologyStatusLoop() {
|
|
// Log topology status every 5 minutes for diagnostic purposes
|
|
statusInterval := 5 * time.Minute
|
|
ticker := time.NewTicker(statusInterval)
|
|
defer ticker.Stop()
|
|
|
|
for mm.IsRunning() {
|
|
select {
|
|
case <-mm.stopChan:
|
|
return
|
|
case <-ticker.C:
|
|
mm.logTopologyStatus()
|
|
}
|
|
}
|
|
}
|
|
|
|
// logTopologyStatus logs current topology and worker status for diagnostics
|
|
func (mm *MaintenanceManager) logTopologyStatus() {
|
|
if mm.scanner == nil || mm.scanner.integration == nil {
|
|
glog.V(2).Infof("Topology status: scanner/integration not available")
|
|
return
|
|
}
|
|
|
|
activeTopology := mm.scanner.integration.GetActiveTopology()
|
|
if activeTopology == nil {
|
|
glog.V(1).Infof("Topology status: ActiveTopology is nil")
|
|
return
|
|
}
|
|
|
|
diskCount := activeTopology.GetDiskCount()
|
|
nodeCount := len(activeTopology.GetAllNodes())
|
|
|
|
// Get queue stats
|
|
stats := mm.queue.GetStats()
|
|
workerCount := len(mm.queue.GetWorkers())
|
|
|
|
mm.mutex.RLock()
|
|
errorCount := mm.errorCount
|
|
mm.mutex.RUnlock()
|
|
|
|
glog.V(1).Infof("Topology status: %d nodes, %d disks, %d workers, %d pending tasks, %d running tasks, errors: %d",
|
|
nodeCount, diskCount, workerCount,
|
|
stats.TasksByStatus[TaskStatusPending],
|
|
stats.TasksByStatus[TaskStatusInProgress]+stats.TasksByStatus[TaskStatusAssigned],
|
|
errorCount)
|
|
}
|
|
|
|
// performScan executes a maintenance scan with error handling and backoff
|
|
func (mm *MaintenanceManager) performScan() {
|
|
defer func() {
|
|
// Always reset scan in progress flag when done
|
|
mm.mutex.Lock()
|
|
mm.scanInProgress = false
|
|
mm.mutex.Unlock()
|
|
}()
|
|
|
|
glog.Infof("Starting maintenance scan...")
|
|
|
|
results, err := mm.scanner.ScanForMaintenanceTasks()
|
|
stats_collect.AdminMaintenanceLastScanTimestampSeconds.SetToCurrentTime()
|
|
if err != nil {
|
|
// Handle scan error
|
|
mm.mutex.Lock()
|
|
mm.handleScanError(err)
|
|
mm.mutex.Unlock()
|
|
glog.Warningf("Maintenance scan failed: %v", err)
|
|
return
|
|
}
|
|
|
|
// Scan succeeded - update state and process results
|
|
mm.handleScanSuccess(results)
|
|
}
|
|
|
|
// handleScanSuccess processes successful scan results with proper lock management
|
|
func (mm *MaintenanceManager) handleScanSuccess(results []*TaskDetectionResult) {
|
|
// Update manager state first
|
|
mm.mutex.Lock()
|
|
mm.resetErrorTracking()
|
|
taskCount := len(results)
|
|
mm.mutex.Unlock()
|
|
|
|
if taskCount > 0 {
|
|
// Count tasks by type for logging (outside of lock)
|
|
taskCounts := make(map[MaintenanceTaskType]int)
|
|
for _, result := range results {
|
|
taskCounts[result.TaskType]++
|
|
}
|
|
|
|
// Add tasks to queue (no manager lock held)
|
|
mm.queue.AddTasksFromResults(results)
|
|
|
|
// Log detailed scan results
|
|
glog.Infof("Maintenance scan completed: found %d tasks", taskCount)
|
|
for taskType, count := range taskCounts {
|
|
glog.Infof(" - %s: %d tasks", taskType, count)
|
|
}
|
|
} else {
|
|
glog.Infof("Maintenance scan completed: no maintenance tasks needed")
|
|
}
|
|
}
|
|
|
|
// handleScanError handles scan errors with exponential backoff and reduced logging
|
|
func (mm *MaintenanceManager) handleScanError(err error) {
|
|
now := time.Now()
|
|
mm.errorCount++
|
|
mm.lastError = err
|
|
mm.lastErrorTime = now
|
|
|
|
// Use exponential backoff with jitter
|
|
if mm.errorCount > 1 {
|
|
mm.backoffDelay = mm.backoffDelay * 2
|
|
if mm.backoffDelay > 5*time.Minute {
|
|
mm.backoffDelay = 5 * time.Minute // Cap at 5 minutes
|
|
}
|
|
}
|
|
|
|
// Reduce log frequency based on error count and time
|
|
shouldLog := false
|
|
if mm.errorCount <= 3 {
|
|
// Log first 3 errors immediately
|
|
shouldLog = true
|
|
} else if mm.errorCount <= 10 && mm.errorCount%3 == 0 {
|
|
// Log every 3rd error for errors 4-10
|
|
shouldLog = true
|
|
} else if mm.errorCount%10 == 0 {
|
|
// Log every 10th error after that
|
|
shouldLog = true
|
|
}
|
|
|
|
if shouldLog {
|
|
// Check if it's a connection error to provide better messaging
|
|
if isConnectionError(err) {
|
|
if mm.errorCount == 1 {
|
|
glog.Errorf("Maintenance scan failed: %v (will retry with backoff)", err)
|
|
} else {
|
|
glog.Errorf("Maintenance scan still failing after %d attempts: %v (backoff: %v)",
|
|
mm.errorCount, err, mm.backoffDelay)
|
|
}
|
|
} else {
|
|
glog.Errorf("Maintenance scan failed: %v", err)
|
|
}
|
|
} else {
|
|
// Use debug level for suppressed errors
|
|
glog.V(3).Infof("Maintenance scan failed (error #%d, suppressed): %v", mm.errorCount, err)
|
|
}
|
|
}
|
|
|
|
// resetErrorTracking resets error tracking when scan succeeds
|
|
func (mm *MaintenanceManager) resetErrorTracking() {
|
|
if mm.errorCount > 0 {
|
|
glog.V(1).Infof("Maintenance scan recovered after %d failed attempts", mm.errorCount)
|
|
mm.errorCount = 0
|
|
mm.lastError = nil
|
|
mm.backoffDelay = time.Second // Reset to initial delay
|
|
}
|
|
}
|
|
|
|
// isConnectionError checks if the error is a connection-related error
|
|
func isConnectionError(err error) bool {
|
|
if err == nil {
|
|
return false
|
|
}
|
|
errStr := err.Error()
|
|
return strings.Contains(errStr, "connection refused") ||
|
|
strings.Contains(errStr, "connection error") ||
|
|
strings.Contains(errStr, "dial tcp") ||
|
|
strings.Contains(errStr, "connection timeout") ||
|
|
strings.Contains(errStr, "no route to host") ||
|
|
strings.Contains(errStr, "network unreachable")
|
|
}
|
|
|
|
// performCleanup cleans up old tasks and stale workers
|
|
func (mm *MaintenanceManager) performCleanup() {
|
|
glog.V(2).Infof("Starting maintenance cleanup")
|
|
|
|
taskRetention := time.Duration(mm.config.TaskRetentionSeconds) * time.Second
|
|
workerTimeout := time.Duration(mm.config.WorkerTimeoutSeconds) * time.Second
|
|
|
|
removedTasks := mm.queue.CleanupOldTasks(taskRetention)
|
|
removedWorkers := mm.queue.RemoveStaleWorkers(workerTimeout)
|
|
mm.queue.cleanupCompletedTasks()
|
|
|
|
// Clean up stale pending operations (operations running for more than 4 hours)
|
|
staleOperationTimeout := 4 * time.Hour
|
|
removedOperations := 0
|
|
if mm.scanner != nil && mm.scanner.integration != nil {
|
|
pendingOps := mm.scanner.integration.GetPendingOperations()
|
|
if pendingOps != nil {
|
|
removedOperations = pendingOps.CleanupStaleOperations(staleOperationTimeout)
|
|
}
|
|
}
|
|
|
|
if removedTasks > 0 || removedWorkers > 0 || removedOperations > 0 {
|
|
glog.V(1).Infof("Cleanup completed: removed %d old tasks, %d stale workers, and %d stale operations",
|
|
removedTasks, removedWorkers, removedOperations)
|
|
}
|
|
}
|
|
|
|
// GetQueue returns the maintenance queue
|
|
func (mm *MaintenanceManager) GetQueue() *MaintenanceQueue {
|
|
return mm.queue
|
|
}
|
|
|
|
// GetConfig returns the maintenance configuration
|
|
func (mm *MaintenanceManager) GetConfig() *MaintenanceConfig {
|
|
return mm.config
|
|
}
|
|
|
|
// GetStats returns maintenance statistics
|
|
func (mm *MaintenanceManager) GetStats() *MaintenanceStats {
|
|
stats := mm.queue.GetStats()
|
|
|
|
mm.mutex.RLock()
|
|
defer mm.mutex.RUnlock()
|
|
|
|
stats.LastScanTime = time.Now() // Would need to track this properly
|
|
|
|
// Calculate next scan time based on current error state
|
|
scanInterval := time.Duration(mm.config.ScanIntervalSeconds) * time.Second
|
|
nextScanInterval := scanInterval
|
|
if mm.errorCount > 0 {
|
|
nextScanInterval = mm.backoffDelay
|
|
maxInterval := scanInterval * 10
|
|
if nextScanInterval > maxInterval {
|
|
nextScanInterval = maxInterval
|
|
}
|
|
}
|
|
stats.NextScanTime = time.Now().Add(nextScanInterval)
|
|
|
|
return stats
|
|
}
|
|
|
|
// ReloadTaskConfigurations reloads task configurations from the current policy
|
|
func (mm *MaintenanceManager) ReloadTaskConfigurations() error {
|
|
mm.mutex.Lock()
|
|
defer mm.mutex.Unlock()
|
|
|
|
// Trigger configuration reload in the integration layer
|
|
if mm.scanner != nil && mm.scanner.integration != nil {
|
|
mm.scanner.integration.ConfigureTasksFromPolicy()
|
|
glog.V(1).Infof("Task configurations reloaded from policy")
|
|
return nil
|
|
}
|
|
|
|
return fmt.Errorf("integration not available for configuration reload")
|
|
}
|
|
|
|
// GetErrorState returns the current error state for monitoring
|
|
func (mm *MaintenanceManager) GetErrorState() (errorCount int, lastError error, backoffDelay time.Duration) {
|
|
mm.mutex.RLock()
|
|
defer mm.mutex.RUnlock()
|
|
return mm.errorCount, mm.lastError, mm.backoffDelay
|
|
}
|
|
|
|
// GetTasks returns tasks with filtering
|
|
func (mm *MaintenanceManager) GetTasks(status MaintenanceTaskStatus, taskType MaintenanceTaskType, limit int) []*MaintenanceTask {
|
|
return mm.queue.GetTasks(status, taskType, limit)
|
|
}
|
|
|
|
// GetWorkers returns all registered workers
|
|
func (mm *MaintenanceManager) GetWorkers() []*MaintenanceWorker {
|
|
return mm.queue.GetWorkers()
|
|
}
|
|
|
|
// GetWorkerSlots returns used/max task slots per worker ID.
|
|
func (mm *MaintenanceManager) GetWorkerSlots() map[string]WorkerSlots {
|
|
return mm.queue.GetWorkerSlots()
|
|
}
|
|
|
|
// TriggerScan manually triggers a maintenance scan
|
|
func (mm *MaintenanceManager) TriggerScan() error {
|
|
return mm.triggerScanInternal(true)
|
|
}
|
|
|
|
// triggerScanInternal handles both manual and automatic scan triggers
|
|
func (mm *MaintenanceManager) triggerScanInternal(isManual bool) error {
|
|
// running and scanInProgress are checked under one lock so a Stop that lands
|
|
// between the two checks cannot leave a scan running after shutdown.
|
|
mm.mutex.Lock()
|
|
if !mm.running {
|
|
mm.mutex.Unlock()
|
|
return fmt.Errorf("maintenance manager is not running")
|
|
}
|
|
|
|
// Prevent multiple concurrent scans
|
|
if mm.scanInProgress {
|
|
mm.mutex.Unlock()
|
|
if isManual {
|
|
glog.V(1).Infof("Manual scan already in progress, ignoring trigger request")
|
|
} else {
|
|
glog.V(2).Infof("Automatic scan already in progress, ignoring scheduled scan")
|
|
}
|
|
return fmt.Errorf("scan already in progress")
|
|
}
|
|
mm.scanInProgress = true
|
|
mm.mutex.Unlock()
|
|
|
|
go mm.performScan()
|
|
return nil
|
|
}
|
|
|
|
// UpdateConfig updates the maintenance configuration
|
|
func (mm *MaintenanceManager) UpdateConfig(config *MaintenanceConfig) error {
|
|
if config == nil {
|
|
return fmt.Errorf("config cannot be nil")
|
|
}
|
|
|
|
mm.config = config
|
|
mm.queue.policy = config.Policy
|
|
mm.scanner.policy = config.Policy
|
|
|
|
// Propagate global policy changes to individual task configuration files
|
|
if config.Policy != nil {
|
|
mm.saveTaskConfigsFromPolicy(config.Policy)
|
|
}
|
|
|
|
// The integration holds its own reference to the policy and is what pushes the
|
|
// enabled flag and the concurrency limit into the registered detectors and
|
|
// schedulers. Without this the queue and the scanner saw the new policy while the
|
|
// detectors kept scanning under the old one until the next restart.
|
|
if mm.scanner != nil && mm.scanner.integration != nil {
|
|
mm.scanner.integration.SetPolicy(config.Policy)
|
|
}
|
|
|
|
glog.V(1).Infof("Maintenance configuration updated")
|
|
return nil
|
|
}
|
|
|
|
// saveTaskConfigsFromPolicy propagates global policy settings to separate task configuration files
|
|
func (mm *MaintenanceManager) saveTaskConfigsFromPolicy(policy *worker_pb.MaintenancePolicy) {
|
|
if mm.queue.persistence == nil || policy == nil {
|
|
return
|
|
}
|
|
|
|
glog.V(1).Infof("Propagating maintenance policy changes to separate task configs")
|
|
for taskType, taskPolicy := range policy.TaskPolicies {
|
|
if err := mm.queue.persistence.SaveTaskPolicy(taskType, taskPolicy); err != nil {
|
|
glog.Errorf("Failed to save task policy for %s: %v", taskType, err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// CancelTask cancels a pending task
|
|
func (mm *MaintenanceManager) CancelTask(taskID string) error {
|
|
mm.queue.mutex.Lock()
|
|
|
|
task, exists := mm.queue.tasks[taskID]
|
|
if !exists {
|
|
mm.queue.mutex.Unlock()
|
|
return fmt.Errorf("task %s not found", taskID)
|
|
}
|
|
if task.Status != TaskStatusPending {
|
|
status := task.Status
|
|
mm.queue.mutex.Unlock()
|
|
return fmt.Errorf("task %s cannot be cancelled (status: %s)", taskID, status)
|
|
}
|
|
|
|
task.Status = TaskStatusCancelled
|
|
completedTime := time.Now()
|
|
task.CompletedAt = &completedTime
|
|
cancelledSnapshot := snapshotTask(task)
|
|
|
|
// Remove from pending tasks
|
|
for i, pendingTask := range mm.queue.pendingTasks {
|
|
if pendingTask.ID == taskID {
|
|
mm.queue.pendingTasks = append(mm.queue.pendingTasks[:i], mm.queue.pendingTasks[i+1:]...)
|
|
break
|
|
}
|
|
}
|
|
|
|
// Notify ActiveTopology to release capacity
|
|
if mm.scanner != nil && mm.scanner.integration != nil {
|
|
if at := mm.scanner.integration.GetActiveTopology(); at != nil {
|
|
_ = at.CompleteTask(taskID)
|
|
}
|
|
}
|
|
mm.queue.mutex.Unlock()
|
|
|
|
if mm.queue.persistence != nil {
|
|
mm.queue.persistMu.Lock()
|
|
if mm.queue.deleteTaskStateLocked(taskID) != nil {
|
|
mm.queue.saveTaskStateLocked(cancelledSnapshot)
|
|
}
|
|
mm.queue.persistMu.Unlock()
|
|
}
|
|
glog.V(2).Infof("Cancelled task %s", taskID)
|
|
return nil
|
|
}
|
|
|
|
// RegisterWorker registers a new worker
|
|
func (mm *MaintenanceManager) RegisterWorker(worker *MaintenanceWorker) {
|
|
mm.queue.RegisterWorker(worker)
|
|
}
|
|
|
|
// GetNextTask returns the next task for a worker
|
|
func (mm *MaintenanceManager) GetNextTask(workerID string, capabilities []MaintenanceTaskType) *MaintenanceTask {
|
|
return mm.queue.GetNextTask(workerID, capabilities)
|
|
}
|
|
|
|
// CompleteTask marks a task as completed
|
|
func (mm *MaintenanceManager) CompleteTask(taskID string, error string) {
|
|
mm.queue.CompleteTask(taskID, error)
|
|
}
|
|
|
|
// UpdateTaskProgress updates task progress
|
|
func (mm *MaintenanceManager) UpdateTaskProgress(taskID string, progress float64) {
|
|
mm.queue.UpdateTaskProgress(taskID, progress)
|
|
}
|
|
|
|
// UpdateWorkerHeartbeat updates worker heartbeat
|
|
func (mm *MaintenanceManager) UpdateWorkerHeartbeat(workerID string) {
|
|
mm.queue.UpdateWorkerHeartbeat(workerID)
|
|
}
|