The "Scan Floating IP" button failed with a client timeout: the project now holds ~6.4k floating IPs and the scan listed them all in one unpaginated, timeout-less Neutron request on the HTTP request context. openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them small) with per-page retry/backoff on transport errors, 5xx and 429, and every request now has a timeout (also ends hangs inside the orchestrator tick). orchestrator: the scan is a single-flight background job on the process context with progress (clearing/listing/enqueuing/done/error), dry_run, full discovery before anything is enqueued, then SubmitIPs in chunks of 500 in ascending IP order; a failed read leaves the queue untouched. The auto-cycle gets a "scanning" phase that polls the job, so the control loop and autoCycleMu are never held across OpenStack/DB work; it recovers after a restart and waits for (instead of adopting) a scan started by someone else. db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY counters, EXISTS completion check, set-based ClearAllIPs. API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging and filters on /admin/ips and /admin/registry (bare arrays without limit), results_by_overall in /admin/status. dashboard: scan progress panel and dry-run button, paginated /ips and /registry with server-side filters, Overview on counters and capped lists with progress/ETA, "select all N by filter", hx-params fix for per-row buttons, real counts in confirmations. Also: docs (API, USAGE, DASHBOARD, README), plan and review under docs/changes/, bin/ rebuilt with new SHA256SUMS. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
351 lines
12 KiB
Go
351 lines
12 KiB
Go
package orchestrator
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"time"
|
|
|
|
"cloudipvalidator/internal/db"
|
|
)
|
|
|
|
// This file implements the automatic check cycle: an optional, repeating
|
|
// "clear queue -> scan floating IPs -> wait until every queued address has
|
|
// reached a terminal state -> wait interval" scenario. Steps 1-2 run as ONE
|
|
// background scan job (StartScan{ClearFirst:true}) so the control loop and
|
|
// autoCycleMu are never held across OpenStack/DB-heavy work: the cycle sits in
|
|
// phase `scanning` while the job runs and each step merely polls it. Step 3
|
|
// needs no code at all because Tick already picks up `queued` addresses. All
|
|
// state lives in the database (db.AutoCycle), so the cycle survives a
|
|
// control-api restart (a restart in phase `scanning` simply starts the scan
|
|
// again) and the interval/limits can be changed at runtime.
|
|
|
|
// GetAutoCycle returns the current auto-cycle configuration and state.
|
|
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
|
|
return o.DB.GetAutoCycle(ctx)
|
|
}
|
|
|
|
// StartAutoCycle enables the auto-cycle; the first cycle begins on the next
|
|
// AutoCycleStep. It is idempotent: if the cycle is already enabled nothing
|
|
// changes (in particular a running cycle is not restarted).
|
|
func (o *Orchestrator) StartAutoCycle(ctx context.Context) error {
|
|
o.autoCycleMu.Lock()
|
|
defer o.autoCycleMu.Unlock()
|
|
|
|
ac, err := o.DB.GetAutoCycle(ctx)
|
|
if err != nil {
|
|
return fmt.Errorf("get auto cycle: %w", err)
|
|
}
|
|
if ac.Enabled {
|
|
return nil
|
|
}
|
|
now := db.Now()
|
|
if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil {
|
|
return fmt.Errorf("enable auto cycle: %w", err)
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
|
"reason": "enabled",
|
|
"interval_seconds": ac.IntervalSeconds,
|
|
"max_run_seconds": ac.MaxRunSeconds,
|
|
}))
|
|
return nil
|
|
}
|
|
|
|
// StopAutoCycle disables the auto-cycle and returns it to idle. Checks that
|
|
// are already in flight are NOT cancelled — they finish normally and land in
|
|
// the registry; only the repetition stops.
|
|
func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
|
|
o.autoCycleMu.Lock()
|
|
defer o.autoCycleMu.Unlock()
|
|
|
|
ac, err := o.DB.GetAutoCycle(ctx)
|
|
if err != nil {
|
|
return fmt.Errorf("get auto cycle: %w", err)
|
|
}
|
|
if !ac.Enabled {
|
|
return nil
|
|
}
|
|
// "stopped" describes an interrupted run. Stopping during the pause
|
|
// between cycles must not overwrite the result of the last finished one.
|
|
outcome := ""
|
|
if ac.Phase == db.AutoCyclePhaseRunning || ac.Phase == db.AutoCyclePhaseScanning {
|
|
outcome = db.AutoCycleOutcomeStopped
|
|
}
|
|
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
|
|
return fmt.Errorf("disable auto cycle: %w", err)
|
|
}
|
|
if ac.Phase == db.AutoCyclePhaseScanning {
|
|
// Persisted first, so a step racing in right after cannot restart the
|
|
// scan; then abort the background job (queue chunks already enqueued
|
|
// stay, like in-flight checks).
|
|
o.CancelScan()
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
|
|
"phase": ac.Phase,
|
|
}))
|
|
return nil
|
|
}
|
|
|
|
// AutoCycleStep advances the auto-cycle state machine by one step. It is
|
|
// called by the control-api loop right after Tick.
|
|
func (o *Orchestrator) AutoCycleStep(ctx context.Context) {
|
|
o.autoCycleStep(ctx, db.Now())
|
|
}
|
|
|
|
// autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive
|
|
// the state machine deterministically without sleeping.
|
|
func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
|
|
o.autoCycleMu.Lock()
|
|
defer o.autoCycleMu.Unlock()
|
|
|
|
// Read inside the lock: Start/Stop may have changed the row since the
|
|
// caller last looked.
|
|
ac, err := o.DB.GetAutoCycle(ctx)
|
|
if err != nil {
|
|
o.Log.Error("auto cycle: read state", "err", err)
|
|
return
|
|
}
|
|
|
|
if !ac.Enabled {
|
|
if ac.Phase != db.AutoCyclePhaseIdle {
|
|
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil {
|
|
o.Log.Error("auto cycle: reset phase to idle", "err", err)
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
switch ac.Phase {
|
|
case db.AutoCyclePhaseScanning:
|
|
o.autoCycleCheckScan(ctx, ac, now)
|
|
return
|
|
case db.AutoCyclePhaseRunning:
|
|
o.autoCycleCheckRun(ctx, ac, now)
|
|
return
|
|
}
|
|
|
|
// idle or waiting: start a new cycle once next_run_at has come.
|
|
if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) {
|
|
return
|
|
}
|
|
o.autoCycleStartRun(ctx, ac, now)
|
|
}
|
|
|
|
// autoCycleStartRun begins a cycle: it starts the background scan job (clear
|
|
// the queue, then discover and enqueue the free floating IPs) and moves to
|
|
// phase `scanning` right away. Nothing slow happens here, so autoCycleMu is
|
|
// released immediately and the control loop keeps ticking.
|
|
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
|
if _, started := o.StartScan(ScanOptions{ClearFirst: true}); !started {
|
|
// A scan started by someone else (an operator's manual or dry-run scan,
|
|
// the periodic scan) is in flight. It is not this cycle's scan: it may
|
|
// not clear the queue, or may not enqueue anything at all (dry run), so
|
|
// adopting it would end in a "completed" cycle over an untouched or
|
|
// empty queue. Change nothing and try again on the next step, once it
|
|
// has finished (it is bounded by fip_scan_timeout_seconds).
|
|
o.Log.Info("auto cycle: another floating ip scan is running, waiting for it to finish")
|
|
return
|
|
}
|
|
|
|
st := autoCycleStateOf(ac)
|
|
st.LastRunStartedAt = &now
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = nil
|
|
st.Phase = db.AutoCyclePhaseScanning
|
|
st.LastError = ""
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
}
|
|
}
|
|
|
|
// autoCycleCheckScan handles the scanning phase by polling the scan job.
|
|
func (o *Orchestrator) autoCycleCheckScan(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
|
scan := o.ScanStatus()
|
|
switch {
|
|
case scan.Running:
|
|
return
|
|
case scan.State == ScanIdle:
|
|
// Phase `scanning` but no job in this process: control-api restarted
|
|
// mid-scan. Starting again is idempotent (clear + scan).
|
|
o.Log.Info("auto cycle: restarting floating ip scan after restart")
|
|
o.StartScan(ScanOptions{ClearFirst: true})
|
|
return
|
|
}
|
|
|
|
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
|
next := now.Add(interval)
|
|
st := autoCycleStateOf(ac)
|
|
|
|
switch scan.State {
|
|
case ScanError:
|
|
o.Log.Error("auto cycle: step failed", "step", "scan floating ips", "err", scan.Error)
|
|
st.Phase = db.AutoCyclePhaseWaiting
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = &next
|
|
st.LastRunFinishedAt = &now
|
|
st.LastOutcome = db.AutoCycleOutcomeError
|
|
st.LastError = "scan floating ips: " + scan.Error
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
return
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
|
|
"step": "scan floating ips",
|
|
"error": scan.Error,
|
|
}))
|
|
|
|
case ScanCancelled:
|
|
if life := o.lifetimeErr(); life != nil {
|
|
// The process is shutting down: leave the phase as is so the next
|
|
// start resumes the cycle (scanning without a job restarts it).
|
|
return
|
|
}
|
|
// Cancelled by an operator (CancelScan): treat as an interrupted run.
|
|
st.Phase = db.AutoCyclePhaseWaiting
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = &next
|
|
st.LastRunFinishedAt = &now
|
|
st.LastOutcome = db.AutoCycleOutcomeStopped
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
}
|
|
|
|
default: // ScanDone
|
|
st.LastScannedFree = scan.Free
|
|
st.LastError = ""
|
|
if scan.Free == 0 {
|
|
// Nothing was queued; waiting for completion would never end.
|
|
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
|
|
st.Phase = db.AutoCyclePhaseWaiting
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = &next
|
|
st.LastRunFinishedAt = &now
|
|
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
}
|
|
return
|
|
}
|
|
// max_run_seconds counts from the end of the scan.
|
|
st.Phase = db.AutoCyclePhaseRunning
|
|
st.RunStartedAt = &now
|
|
st.NextRunAt = nil
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
return
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
|
"reason": "cycle",
|
|
"scanned_free": scan.Free,
|
|
}))
|
|
}
|
|
}
|
|
|
|
func (o *Orchestrator) lifetimeErr() error {
|
|
o.scan.mu.Lock()
|
|
defer o.scan.mu.Unlock()
|
|
return o.scan.lifetime().Err()
|
|
}
|
|
|
|
// autoCycleCheckRun handles the running phase: finish the cycle once every
|
|
// queued address is terminal, or give up after max_run_seconds.
|
|
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
|
// An empty queue counts as finished: right after the scan it cannot be
|
|
// empty (scanned > 0), so it only happens when an operator cleared or
|
|
// deleted every address mid-cycle — and then there is nothing to wait for
|
|
// (with max_run_seconds=0 the cycle would otherwise hang forever).
|
|
// EXISTS is cheap enough to run on every tick, however long the queue.
|
|
pending, err := o.DB.AnyNonTerminalIP(ctx)
|
|
if err != nil {
|
|
o.Log.Error("auto cycle: check pending ips", "err", err)
|
|
return
|
|
}
|
|
|
|
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
|
next := now.Add(interval)
|
|
|
|
if !pending {
|
|
_, addresses, err := o.DB.CountIPsByState(ctx)
|
|
if err != nil {
|
|
o.Log.Error("auto cycle: count ips", "err", err)
|
|
return
|
|
}
|
|
st := autoCycleStateOf(ac)
|
|
st.Phase = db.AutoCyclePhaseWaiting
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = &next
|
|
st.LastRunFinishedAt = &now
|
|
st.LastOutcome = db.AutoCycleOutcomeCompleted
|
|
st.LastError = ""
|
|
st.RunsTotal++
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
return
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
|
|
"addresses": addresses,
|
|
"runs": st.RunsTotal,
|
|
}))
|
|
return
|
|
}
|
|
|
|
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
|
|
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
|
|
_, addresses, err := o.DB.CountIPsByState(ctx)
|
|
if err != nil {
|
|
o.Log.Error("auto cycle: count ips", "err", err)
|
|
return
|
|
}
|
|
// The queue is left untouched: the next cycle clears it anyway, and
|
|
// an operator can still inspect what got stuck.
|
|
st := autoCycleStateOf(ac)
|
|
st.Phase = db.AutoCyclePhaseWaiting
|
|
st.RunStartedAt = nil
|
|
st.NextRunAt = &next
|
|
st.LastRunFinishedAt = &now
|
|
st.LastOutcome = db.AutoCycleOutcomeTimeout
|
|
st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds)
|
|
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
|
o.Log.Error("auto cycle: save state", "err", err)
|
|
return
|
|
}
|
|
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
|
|
"max_run_seconds": ac.MaxRunSeconds,
|
|
"addresses": addresses,
|
|
}))
|
|
}
|
|
}
|
|
|
|
// isTerminalIPState reports whether an address has finished its check cycle
|
|
// for good (its result, if any, is already written to the registry).
|
|
// db.IPOccupied counts: such an address never enters the check cycle.
|
|
func isTerminalIPState(state string) bool {
|
|
switch state {
|
|
case db.IPDone, db.IPFailed, db.IPOccupied:
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState {
|
|
return db.AutoCycleState{
|
|
Phase: ac.Phase,
|
|
RunStartedAt: ac.RunStartedAt,
|
|
NextRunAt: ac.NextRunAt,
|
|
LastRunStartedAt: ac.LastRunStartedAt,
|
|
LastRunFinishedAt: ac.LastRunFinishedAt,
|
|
LastOutcome: ac.LastOutcome,
|
|
LastError: ac.LastError,
|
|
LastScannedFree: ac.LastScannedFree,
|
|
RunsTotal: ac.RunsTotal,
|
|
}
|
|
}
|
|
|
|
func autoCyclePayload(m map[string]any) string {
|
|
b, err := json.Marshal(m)
|
|
if err != nil {
|
|
return "{}"
|
|
}
|
|
return string(b)
|
|
}
|