Files
cloud-ip-validator/internal/orchestrator/autocycle.go
T

350 lines
12 KiB
Go
Raw Normal View History

package orchestrator
import (
"context"
"encoding/json"
"fmt"
"time"
"cloudipvalidator/internal/db"
)
// This file implements the automatic check cycle: an optional, repeating
// "clear queue -> scan floating IPs -> wait until every queued address has
// reached a terminal state -> wait interval" scenario. Steps 1-2 run as ONE
// background scan job (StartScan{ClearFirst:true}) so the control loop and
// autoCycleMu are never held across OpenStack/DB-heavy work: the cycle sits in
// phase `scanning` while the job runs and each step merely polls it. Step 3
// needs no code at all because Tick already picks up `queued` addresses. All
// state lives in the database (db.AutoCycle), so the cycle survives a
// control-api restart (a restart in phase `scanning` simply starts the scan
// again) and the interval/limits can be changed at runtime.
// GetAutoCycle returns the current auto-cycle configuration and state.
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
return o.DB.GetAutoCycle(ctx)
}
// StartAutoCycle enables the auto-cycle; the first cycle begins on the next
// AutoCycleStep. It is idempotent: if the cycle is already enabled nothing
// changes (in particular a running cycle is not restarted).
func (o *Orchestrator) StartAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if ac.Enabled {
return nil
}
now := db.Now()
if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil {
return fmt.Errorf("enable auto cycle: %w", err)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "enabled",
"interval_seconds": ac.IntervalSeconds,
"max_run_seconds": ac.MaxRunSeconds,
}))
return nil
}
// StopAutoCycle disables the auto-cycle and returns it to idle. Checks that
// are already in flight are NOT cancelled — they finish normally and land in
// the registry; only the repetition stops.
func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if !ac.Enabled {
return nil
}
// "stopped" describes an interrupted run. Stopping during the pause
// between cycles must not overwrite the result of the last finished one.
outcome := ""
if ac.Phase == db.AutoCyclePhaseRunning || ac.Phase == db.AutoCyclePhaseScanning {
outcome = db.AutoCycleOutcomeStopped
}
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
return fmt.Errorf("disable auto cycle: %w", err)
}
if ac.Phase == db.AutoCyclePhaseScanning {
// Persisted first, so a step racing in right after cannot restart the
// scan; then abort the background job (queue chunks already enqueued
// stay, like in-flight checks).
o.CancelScan()
}
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
"phase": ac.Phase,
}))
return nil
}
// AutoCycleStep advances the auto-cycle state machine by one step. It is
// called by the control-api loop right after Tick.
func (o *Orchestrator) AutoCycleStep(ctx context.Context) {
o.autoCycleStep(ctx, db.Now())
}
// autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive
// the state machine deterministically without sleeping.
func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
// Read inside the lock: Start/Stop may have changed the row since the
// caller last looked.
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
o.Log.Error("auto cycle: read state", "err", err)
return
}
if !ac.Enabled {
if ac.Phase != db.AutoCyclePhaseIdle {
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil {
o.Log.Error("auto cycle: reset phase to idle", "err", err)
}
}
return
}
switch ac.Phase {
case db.AutoCyclePhaseScanning:
o.autoCycleCheckScan(ctx, ac, now)
return
case db.AutoCyclePhaseRunning:
o.autoCycleCheckRun(ctx, ac, now)
return
}
// idle or waiting: start a new cycle once next_run_at has come.
if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) {
return
}
o.autoCycleStartRun(ctx, ac, now)
}
// autoCycleStartRun begins a cycle: it starts the background scan job (clear
// the queue, then discover and enqueue the free floating IPs) and moves to
// phase `scanning` right away. Nothing slow happens here, so autoCycleMu is
// released immediately and the control loop keeps ticking.
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
if _, started := o.StartScan(ScanOptions{ClearFirst: true}); !started {
// A scan started by someone else (an operator's manual or dry-run scan,
// the periodic scan) is in flight. It is not this cycle's scan: it may
// not clear the queue, or may not enqueue anything at all (dry run), so
// adopting it would end in a "completed" cycle over an untouched or
// empty queue. Change nothing and try again on the next step, once it
// has finished (it is bounded by fip_scan_timeout_seconds).
o.Log.Info("auto cycle: another floating ip scan is running, waiting for it to finish")
return
}
st := autoCycleStateOf(ac)
st.LastRunStartedAt = &now
st.RunStartedAt = nil
st.NextRunAt = nil
st.Phase = db.AutoCyclePhaseScanning
st.LastError = ""
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
}
// autoCycleCheckScan handles the scanning phase by polling the scan job.
func (o *Orchestrator) autoCycleCheckScan(ctx context.Context, ac db.AutoCycle, now time.Time) {
scan := o.ScanStatus()
switch {
case scan.Running:
return
case scan.State == ScanIdle:
// Phase `scanning` but no job in this process: control-api restarted
// mid-scan. Starting again is idempotent (clear + scan).
o.Log.Info("auto cycle: restarting floating ip scan after restart")
o.StartScan(ScanOptions{ClearFirst: true})
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
st := autoCycleStateOf(ac)
switch scan.State {
case ScanError:
o.Log.Error("auto cycle: step failed", "step", "scan floating ips", "err", scan.Error)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeError
st.LastError = "scan floating ips: " + scan.Error
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
"step": "scan floating ips",
"error": scan.Error,
}))
case ScanCancelled:
if life := o.lifetimeErr(); life != nil {
// The process is shutting down: leave the phase as is so the next
// start resumes the cycle (scanning without a job restarts it).
return
}
// Cancelled by an operator (CancelScan): treat as an interrupted run.
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeStopped
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
default: // ScanDone
st.LastScannedFree = scan.Free
st.LastError = ""
if scan.Free == 0 {
// Nothing was queued; waiting for completion would never end.
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
return
}
// max_run_seconds counts from the end of the scan.
st.Phase = db.AutoCyclePhaseRunning
st.RunStartedAt = &now
st.NextRunAt = nil
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "cycle",
"scanned_free": scan.Free,
}))
}
}
func (o *Orchestrator) lifetimeErr() error {
o.scan.mu.Lock()
defer o.scan.mu.Unlock()
return o.scan.lifetime().Err()
}
// autoCycleCheckRun handles the running phase: finish the cycle once every
// queued address is terminal, or give up after max_run_seconds.
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
// An empty queue counts as finished: right after the scan it cannot be
// empty (scanned > 0), so it only happens when an operator cleared or
// deleted every address mid-cycle — and then there is nothing to wait for
// (with max_run_seconds=0 the cycle would otherwise hang forever).
// EXISTS is cheap enough to run on every tick, however long the queue.
pending, err := o.DB.AnyNonTerminalIP(ctx)
if err != nil {
o.Log.Error("auto cycle: check pending ips", "err", err)
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
if !pending {
_, addresses, err := o.DB.CountIPsByState(ctx)
if err != nil {
o.Log.Error("auto cycle: count ips", "err", err)
return
}
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeCompleted
st.LastError = ""
st.RunsTotal++
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
"addresses": addresses,
"runs": st.RunsTotal,
}))
return
}
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
_, addresses, err := o.DB.CountIPsByState(ctx)
if err != nil {
o.Log.Error("auto cycle: count ips", "err", err)
return
}
// The queue is left untouched: the next cycle clears it anyway, and
// an operator can still inspect what got stuck.
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeTimeout
st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds)
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
"max_run_seconds": ac.MaxRunSeconds,
"addresses": addresses,
}))
}
}
// isTerminalIPState reports whether an address has finished its check cycle
// for good (its result, if any, is already written to the registry).
// db.IPOccupied counts: such an address never enters the check cycle.
func isTerminalIPState(state string) bool {
switch state {
case db.IPDone, db.IPFailed, db.IPOccupied:
return true
}
return false
}
func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState {
return db.AutoCycleState{
Phase: ac.Phase,
RunStartedAt: ac.RunStartedAt,
NextRunAt: ac.NextRunAt,
LastRunStartedAt: ac.LastRunStartedAt,
LastRunFinishedAt: ac.LastRunFinishedAt,
LastOutcome: ac.LastOutcome,
LastError: ac.LastError,
LastScannedFree: ac.LastScannedFree,
RunsTotal: ac.RunsTotal,
}
}
func autoCyclePayload(m map[string]any) string {
b, err := json.Marshal(m)
if err != nil {
return "{}"
}
return string(b)
}