package orchestrator import ( "context" "encoding/json" "fmt" "time" "cloudipvalidator/internal/db" ) // This file implements the automatic check cycle: an optional, repeating // "clear queue -> scan floating IPs -> wait until every queued address has // reached a terminal state -> wait interval" scenario. Steps 1-2 run as ONE // background scan job (StartScan{ClearFirst:true}) so the control loop and // autoCycleMu are never held across OpenStack/DB-heavy work: the cycle sits in // phase `scanning` while the job runs and each step merely polls it. Step 3 // needs no code at all because Tick already picks up `queued` addresses. All // state lives in the database (db.AutoCycle), so the cycle survives a // control-api restart (a restart in phase `scanning` simply starts the scan // again) and the interval/limits can be changed at runtime. // GetAutoCycle returns the current auto-cycle configuration and state. func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) { return o.DB.GetAutoCycle(ctx) } // StartAutoCycle enables the auto-cycle; the first cycle begins on the next // AutoCycleStep. It is idempotent: if the cycle is already enabled nothing // changes (in particular a running cycle is not restarted). func (o *Orchestrator) StartAutoCycle(ctx context.Context) error { o.autoCycleMu.Lock() defer o.autoCycleMu.Unlock() ac, err := o.DB.GetAutoCycle(ctx) if err != nil { return fmt.Errorf("get auto cycle: %w", err) } if ac.Enabled { return nil } now := db.Now() if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil { return fmt.Errorf("enable auto cycle: %w", err) } o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{ "reason": "enabled", "interval_seconds": ac.IntervalSeconds, "max_run_seconds": ac.MaxRunSeconds, })) return nil } // StopAutoCycle disables the auto-cycle and returns it to idle. Checks that // are already in flight are NOT cancelled — they finish normally and land in // the registry; only the repetition stops. func (o *Orchestrator) StopAutoCycle(ctx context.Context) error { o.autoCycleMu.Lock() defer o.autoCycleMu.Unlock() ac, err := o.DB.GetAutoCycle(ctx) if err != nil { return fmt.Errorf("get auto cycle: %w", err) } if !ac.Enabled { return nil } // "stopped" describes an interrupted run. Stopping during the pause // between cycles must not overwrite the result of the last finished one. outcome := "" if ac.Phase == db.AutoCyclePhaseRunning || ac.Phase == db.AutoCyclePhaseScanning { outcome = db.AutoCycleOutcomeStopped } if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil { return fmt.Errorf("disable auto cycle: %w", err) } if ac.Phase == db.AutoCyclePhaseScanning { // Persisted first, so a step racing in right after cannot restart the // scan; then abort the background job (queue chunks already enqueued // stay, like in-flight checks). o.CancelScan() } o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{ "phase": ac.Phase, })) return nil } // AutoCycleStep advances the auto-cycle state machine by one step. It is // called by the control-api loop right after Tick. func (o *Orchestrator) AutoCycleStep(ctx context.Context) { o.autoCycleStep(ctx, db.Now()) } // autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive // the state machine deterministically without sleeping. func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) { o.autoCycleMu.Lock() defer o.autoCycleMu.Unlock() // Read inside the lock: Start/Stop may have changed the row since the // caller last looked. ac, err := o.DB.GetAutoCycle(ctx) if err != nil { o.Log.Error("auto cycle: read state", "err", err) return } if !ac.Enabled { if ac.Phase != db.AutoCyclePhaseIdle { if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil { o.Log.Error("auto cycle: reset phase to idle", "err", err) } } return } switch ac.Phase { case db.AutoCyclePhaseScanning: o.autoCycleCheckScan(ctx, ac, now) return case db.AutoCyclePhaseRunning: o.autoCycleCheckRun(ctx, ac, now) return } // idle or waiting: start a new cycle once next_run_at has come. if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) { return } o.autoCycleStartRun(ctx, ac, now) } // autoCycleStartRun begins a cycle: it starts the background scan job (clear // the queue, then discover and enqueue the free floating IPs) and moves to // phase `scanning` right away. Nothing slow happens here, so autoCycleMu is // released immediately and the control loop keeps ticking. func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) { if _, started := o.StartScan(ScanOptions{ClearFirst: true}); !started { // A scan started by someone else (an operator's manual or dry-run scan, // the periodic scan) is in flight. It is not this cycle's scan: it may // not clear the queue, or may not enqueue anything at all (dry run), so // adopting it would end in a "completed" cycle over an untouched or // empty queue. Change nothing and try again on the next step, once it // has finished (it is bounded by fip_scan_timeout_seconds). o.Log.Info("auto cycle: another floating ip scan is running, waiting for it to finish") return } st := autoCycleStateOf(ac) st.LastRunStartedAt = &now st.RunStartedAt = nil st.NextRunAt = nil st.Phase = db.AutoCyclePhaseScanning st.LastError = "" if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) } } // autoCycleCheckScan handles the scanning phase by polling the scan job. func (o *Orchestrator) autoCycleCheckScan(ctx context.Context, ac db.AutoCycle, now time.Time) { scan := o.ScanStatus() switch { case scan.Running: return case scan.State == ScanIdle: // Phase `scanning` but no job in this process: control-api restarted // mid-scan. Starting again is idempotent (clear + scan). o.Log.Info("auto cycle: restarting floating ip scan after restart") o.StartScan(ScanOptions{ClearFirst: true}) return } interval := time.Duration(ac.IntervalSeconds) * time.Second next := now.Add(interval) st := autoCycleStateOf(ac) switch scan.State { case ScanError: o.Log.Error("auto cycle: step failed", "step", "scan floating ips", "err", scan.Error) st.Phase = db.AutoCyclePhaseWaiting st.RunStartedAt = nil st.NextRunAt = &next st.LastRunFinishedAt = &now st.LastOutcome = db.AutoCycleOutcomeError st.LastError = "scan floating ips: " + scan.Error if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) return } o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{ "step": "scan floating ips", "error": scan.Error, })) case ScanCancelled: if life := o.lifetimeErr(); life != nil { // The process is shutting down: leave the phase as is so the next // start resumes the cycle (scanning without a job restarts it). return } // Cancelled by an operator (CancelScan): treat as an interrupted run. st.Phase = db.AutoCyclePhaseWaiting st.RunStartedAt = nil st.NextRunAt = &next st.LastRunFinishedAt = &now st.LastOutcome = db.AutoCycleOutcomeStopped if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) } default: // ScanDone st.LastScannedFree = scan.Free st.LastError = "" if scan.Free == 0 { // Nothing was queued; waiting for completion would never end. o.Log.Info("auto cycle: no free floating ips, waiting for next interval") st.Phase = db.AutoCyclePhaseWaiting st.RunStartedAt = nil st.NextRunAt = &next st.LastRunFinishedAt = &now st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) } return } // max_run_seconds counts from the end of the scan. st.Phase = db.AutoCyclePhaseRunning st.RunStartedAt = &now st.NextRunAt = nil if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) return } o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{ "reason": "cycle", "scanned_free": scan.Free, })) } } func (o *Orchestrator) lifetimeErr() error { o.scan.mu.Lock() defer o.scan.mu.Unlock() return o.scan.lifetime().Err() } // autoCycleCheckRun handles the running phase: finish the cycle once every // queued address is terminal, or give up after max_run_seconds. func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) { // An empty queue counts as finished: right after the scan it cannot be // empty (scanned > 0), so it only happens when an operator cleared or // deleted every address mid-cycle — and then there is nothing to wait for // (with max_run_seconds=0 the cycle would otherwise hang forever). // EXISTS is cheap enough to run on every tick, however long the queue. pending, err := o.DB.AnyNonTerminalIP(ctx) if err != nil { o.Log.Error("auto cycle: check pending ips", "err", err) return } interval := time.Duration(ac.IntervalSeconds) * time.Second next := now.Add(interval) if !pending { _, addresses, err := o.DB.CountIPsByState(ctx) if err != nil { o.Log.Error("auto cycle: count ips", "err", err) return } st := autoCycleStateOf(ac) st.Phase = db.AutoCyclePhaseWaiting st.RunStartedAt = nil st.NextRunAt = &next st.LastRunFinishedAt = &now st.LastOutcome = db.AutoCycleOutcomeCompleted st.LastError = "" st.RunsTotal++ if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) return } o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{ "addresses": addresses, "runs": st.RunsTotal, })) return } if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil && now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second { _, addresses, err := o.DB.CountIPsByState(ctx) if err != nil { o.Log.Error("auto cycle: count ips", "err", err) return } // The queue is left untouched: the next cycle clears it anyway, and // an operator can still inspect what got stuck. st := autoCycleStateOf(ac) st.Phase = db.AutoCyclePhaseWaiting st.RunStartedAt = nil st.NextRunAt = &next st.LastRunFinishedAt = &now st.LastOutcome = db.AutoCycleOutcomeTimeout st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds) if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil { o.Log.Error("auto cycle: save state", "err", err) return } o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{ "max_run_seconds": ac.MaxRunSeconds, "addresses": addresses, })) } } // isTerminalIPState reports whether an address has finished its check cycle // for good (its result, if any, is already written to the registry). // db.IPOccupied counts: such an address never enters the check cycle. func isTerminalIPState(state string) bool { switch state { case db.IPDone, db.IPFailed, db.IPOccupied: return true } return false } func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState { return db.AutoCycleState{ Phase: ac.Phase, RunStartedAt: ac.RunStartedAt, NextRunAt: ac.NextRunAt, LastRunStartedAt: ac.LastRunStartedAt, LastRunFinishedAt: ac.LastRunFinishedAt, LastOutcome: ac.LastOutcome, LastError: ac.LastError, LastScannedFree: ac.LastScannedFree, RunsTotal: ac.RunsTotal, } } func autoCyclePayload(m map[string]any) string { b, err := json.Marshal(m) if err != nil { return "{}" } return string(b) }