Scan floating IPs in the background, page by page, so thousands of addresses work

The "Scan Floating IP" button failed with a client timeout: the project now
holds ~6.4k floating IPs and the scan listed them all in one unpaginated,
timeout-less Neutron request on the HTTP request context.

openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them
small) with per-page retry/backoff on transport errors, 5xx and 429, and every
request now has a timeout (also ends hangs inside the orchestrator tick).

orchestrator: the scan is a single-flight background job on the process
context with progress (clearing/listing/enqueuing/done/error), dry_run, full
discovery before anything is enqueued, then SubmitIPs in chunks of 500 in
ascending IP order; a failed read leaves the queue untouched. The auto-cycle
gets a "scanning" phase that polls the job, so the control loop and
autoCycleMu are never held across OpenStack/DB work; it recovers after a
restart and waits for (instead of adopting) a scan started by someone else.

db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY
counters, EXISTS completion check, set-based ClearAllIPs.

API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging
and filters on /admin/ips and /admin/registry (bare arrays without limit),
results_by_overall in /admin/status.

dashboard: scan progress panel and dry-run button, paginated /ips and
/registry with server-side filters, Overview on counters and capped lists
with progress/ETA, "select all N by filter", hx-params fix for per-row
buttons, real counts in confirmations.

Also: docs (API, USAGE, DASHBOARD, README), plan and review under
docs/changes/, bin/ rebuilt with new SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-01 19:31:11 +03:00
1 parent debf2afed2
commit aff8fe38b5
61 files changed
+5833 -536

No files matched your search

+131 -63
View File
@@ -11,11 +11,14 @@ import (
// This file implements the automatic check cycle: an optional, repeating
// "clear queue -> scan floating IPs -> wait until every queued address has
// reached a terminal state -> wait interval" scenario. Steps 1-2 reuse
// ClearQueue/ScanFloatingIPs verbatim; step 3 needs no code at all because
// Tick already picks up `queued` addresses. All state lives in the database
// (db.AutoCycle), so the cycle survives a control-api restart and the
// interval/limits can be changed at runtime.
// reached a terminal state -> wait interval" scenario. Steps 1-2 run as ONE
// background scan job (StartScan{ClearFirst:true}) so the control loop and
// autoCycleMu are never held across OpenStack/DB-heavy work: the cycle sits in
// phase `scanning` while the job runs and each step merely polls it. Step 3
// needs no code at all because Tick already picks up `queued` addresses. All
// state lives in the database (db.AutoCycle), so the cycle survives a
// control-api restart (a restart in phase `scanning` simply starts the scan
// again) and the interval/limits can be changed at runtime.
// GetAutoCycle returns the current auto-cycle configuration and state.
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
@@ -65,12 +68,18 @@ func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
// "stopped" describes an interrupted run. Stopping during the pause
// between cycles must not overwrite the result of the last finished one.
outcome := ""
if ac.Phase == db.AutoCyclePhaseRunning {
if ac.Phase == db.AutoCyclePhaseRunning || ac.Phase == db.AutoCyclePhaseScanning {
outcome = db.AutoCycleOutcomeStopped
}
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
return fmt.Errorf("disable auto cycle: %w", err)
}
if ac.Phase == db.AutoCyclePhaseScanning {
// Persisted first, so a step racing in right after cannot restart the
// scan; then abort the background job (queue chunks already enqueued
// stay, like in-flight checks).
o.CancelScan()
}
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
"phase": ac.Phase,
}))
@@ -106,7 +115,11 @@ func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
return
}
if ac.Phase == db.AutoCyclePhaseRunning {
switch ac.Phase {
case db.AutoCyclePhaseScanning:
o.autoCycleCheckScan(ctx, ac, now)
return
case db.AutoCyclePhaseRunning:
o.autoCycleCheckRun(ctx, ac, now)
return
}
@@ -118,95 +131,145 @@ func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
o.autoCycleStartRun(ctx, ac, now)
}
// autoCycleStartRun performs steps 1-2 of the scenario (clear the queue,
// scan floating IPs) and moves the state to running, or straight to waiting
// if there is nothing to wait for.
// autoCycleStartRun begins a cycle: it starts the background scan job (clear
// the queue, then discover and enqueue the free floating IPs) and moves to
// phase `scanning` right away. Nothing slow happens here, so autoCycleMu is
// released immediately and the control loop keeps ticking.
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
if _, started := o.StartScan(ScanOptions{ClearFirst: true}); !started {
// A scan started by someone else (an operator's manual or dry-run scan,
// the periodic scan) is in flight. It is not this cycle's scan: it may
// not clear the queue, or may not enqueue anything at all (dry run), so
// adopting it would end in a "completed" cycle over an untouched or
// empty queue. Change nothing and try again on the next step, once it
// has finished (it is bounded by fip_scan_timeout_seconds).
o.Log.Info("auto cycle: another floating ip scan is running, waiting for it to finish")
return
}
st := autoCycleStateOf(ac)
st.LastRunStartedAt = &now
st.RunStartedAt = nil
st.NextRunAt = nil
st.Phase = db.AutoCyclePhaseScanning
st.LastError = ""
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
}
fail := func(step string, err error) {
o.Log.Error("auto cycle: step failed", "step", step, "err", err)
// autoCycleCheckScan handles the scanning phase by polling the scan job.
func (o *Orchestrator) autoCycleCheckScan(ctx context.Context, ac db.AutoCycle, now time.Time) {
scan := o.ScanStatus()
switch {
case scan.Running:
return
case scan.State == ScanIdle:
// Phase `scanning` but no job in this process: control-api restarted
// mid-scan. Starting again is idempotent (clear + scan).
o.Log.Info("auto cycle: restarting floating ip scan after restart")
o.StartScan(ScanOptions{ClearFirst: true})
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
st := autoCycleStateOf(ac)
switch scan.State {
case ScanError:
o.Log.Error("auto cycle: step failed", "step", "scan floating ips", "err", scan.Error)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeError
st.LastError = fmt.Sprintf("%s: %v", step, err)
if uerr := o.DB.UpdateAutoCycleState(ctx, st); uerr != nil {
o.Log.Error("auto cycle: save state", "err", uerr)
st.LastError = "scan floating ips: " + scan.Error
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
"step": step,
"error": err.Error(),
"step": "scan floating ips",
"error": scan.Error,
}))
}
if _, err := o.ClearQueue(ctx); err != nil {
fail("clear queue", err)
return
}
_, scanned, err := o.ScanFloatingIPs(ctx)
if err != nil {
fail("scan floating ips", err)
return
}
st.LastScannedFree = scanned
st.LastError = ""
if scanned == 0 {
// Nothing was queued; waiting for completion would never end.
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
case ScanCancelled:
if life := o.lifetimeErr(); life != nil {
// The process is shutting down: leave the phase as is so the next
// start resumes the cycle (scanning without a job restarts it).
return
}
// Cancelled by an operator (CancelScan): treat as an interrupted run.
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
st.LastOutcome = db.AutoCycleOutcomeStopped
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
return
}
st.Phase = db.AutoCyclePhaseRunning
st.RunStartedAt = &now
st.NextRunAt = nil
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
default: // ScanDone
st.LastScannedFree = scan.Free
st.LastError = ""
if scan.Free == 0 {
// Nothing was queued; waiting for completion would never end.
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
return
}
// max_run_seconds counts from the end of the scan.
st.Phase = db.AutoCyclePhaseRunning
st.RunStartedAt = &now
st.NextRunAt = nil
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "cycle",
"scanned_free": scan.Free,
}))
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "cycle",
"scanned_free": scanned,
}))
}
func (o *Orchestrator) lifetimeErr() error {
o.scan.mu.Lock()
defer o.scan.mu.Unlock()
return o.scan.lifetime().Err()
}
// autoCycleCheckRun handles the running phase: finish the cycle once every
// queued address is terminal, or give up after max_run_seconds.
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
items, err := o.DB.ListIPs(ctx)
// An empty queue counts as finished: right after the scan it cannot be
// empty (scanned > 0), so it only happens when an operator cleared or
// deleted every address mid-cycle — and then there is nothing to wait for
// (with max_run_seconds=0 the cycle would otherwise hang forever).
// EXISTS is cheap enough to run on every tick, however long the queue.
pending, err := o.DB.AnyNonTerminalIP(ctx)
if err != nil {
o.Log.Error("auto cycle: list ips", "err", err)
o.Log.Error("auto cycle: check pending ips", "err", err)
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
// An empty queue counts as finished: right after the scan it cannot be
// empty (scanned > 0), so it only happens when an operator cleared or
// deleted every address mid-cycle — and then there is nothing to wait for
// (with max_run_seconds=0 the cycle would otherwise hang forever).
allTerminal := true
for _, it := range items {
if !isTerminalIPState(it.State) {
allTerminal = false
break
if !pending {
_, addresses, err := o.DB.CountIPsByState(ctx)
if err != nil {
o.Log.Error("auto cycle: count ips", "err", err)
return
}
}
if allTerminal {
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
@@ -220,7 +283,7 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
"addresses": len(items),
"addresses": addresses,
"runs": st.RunsTotal,
}))
return
@@ -228,6 +291,11 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
_, addresses, err := o.DB.CountIPsByState(ctx)
if err != nil {
o.Log.Error("auto cycle: count ips", "err", err)
return
}
// The queue is left untouched: the next cycle clears it anyway, and
// an operator can still inspect what got stuck.
st := autoCycleStateOf(ac)
@@ -243,7 +311,7 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
}
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
"max_run_seconds": ac.MaxRunSeconds,
"addresses": len(items),
"addresses": addresses,
}))
}
}