Scan floating IPs in the background, page by page, so thousands of addresses work
The "Scan Floating IP" button failed with a client timeout: the project now holds ~6.4k floating IPs and the scan listed them all in one unpaginated, timeout-less Neutron request on the HTTP request context. openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them small) with per-page retry/backoff on transport errors, 5xx and 429, and every request now has a timeout (also ends hangs inside the orchestrator tick). orchestrator: the scan is a single-flight background job on the process context with progress (clearing/listing/enqueuing/done/error), dry_run, full discovery before anything is enqueued, then SubmitIPs in chunks of 500 in ascending IP order; a failed read leaves the queue untouched. The auto-cycle gets a "scanning" phase that polls the job, so the control loop and autoCycleMu are never held across OpenStack/DB work; it recovers after a restart and waits for (instead of adopting) a scan started by someone else. db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY counters, EXISTS completion check, set-based ClearAllIPs. API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging and filters on /admin/ips and /admin/registry (bare arrays without limit), results_by_overall in /admin/status. dashboard: scan progress panel and dry-run button, paginated /ips and /registry with server-side filters, Overview on counters and capped lists with progress/ETA, "select all N by filter", hx-params fix for per-row buttons, real counts in confirmations. Also: docs (API, USAGE, DASHBOARD, README), plan and review under docs/changes/, bin/ rebuilt with new SHA256SUMS. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
debf2afed2
commit
aff8fe38b5
61 files changed
+5833
-536
No files matched your search
@@ -11,11 +11,14 @@ import (
|
||||
|
||||
// This file implements the automatic check cycle: an optional, repeating
|
||||
// "clear queue -> scan floating IPs -> wait until every queued address has
|
||||
// reached a terminal state -> wait interval" scenario. Steps 1-2 reuse
|
||||
// ClearQueue/ScanFloatingIPs verbatim; step 3 needs no code at all because
|
||||
// Tick already picks up `queued` addresses. All state lives in the database
|
||||
// (db.AutoCycle), so the cycle survives a control-api restart and the
|
||||
// interval/limits can be changed at runtime.
|
||||
// reached a terminal state -> wait interval" scenario. Steps 1-2 run as ONE
|
||||
// background scan job (StartScan{ClearFirst:true}) so the control loop and
|
||||
// autoCycleMu are never held across OpenStack/DB-heavy work: the cycle sits in
|
||||
// phase `scanning` while the job runs and each step merely polls it. Step 3
|
||||
// needs no code at all because Tick already picks up `queued` addresses. All
|
||||
// state lives in the database (db.AutoCycle), so the cycle survives a
|
||||
// control-api restart (a restart in phase `scanning` simply starts the scan
|
||||
// again) and the interval/limits can be changed at runtime.
|
||||
|
||||
// GetAutoCycle returns the current auto-cycle configuration and state.
|
||||
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
|
||||
@@ -65,12 +68,18 @@ func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
|
||||
// "stopped" describes an interrupted run. Stopping during the pause
|
||||
// between cycles must not overwrite the result of the last finished one.
|
||||
outcome := ""
|
||||
if ac.Phase == db.AutoCyclePhaseRunning {
|
||||
if ac.Phase == db.AutoCyclePhaseRunning || ac.Phase == db.AutoCyclePhaseScanning {
|
||||
outcome = db.AutoCycleOutcomeStopped
|
||||
}
|
||||
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
|
||||
return fmt.Errorf("disable auto cycle: %w", err)
|
||||
}
|
||||
if ac.Phase == db.AutoCyclePhaseScanning {
|
||||
// Persisted first, so a step racing in right after cannot restart the
|
||||
// scan; then abort the background job (queue chunks already enqueued
|
||||
// stay, like in-flight checks).
|
||||
o.CancelScan()
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
|
||||
"phase": ac.Phase,
|
||||
}))
|
||||
@@ -106,7 +115,11 @@ func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
|
||||
return
|
||||
}
|
||||
|
||||
if ac.Phase == db.AutoCyclePhaseRunning {
|
||||
switch ac.Phase {
|
||||
case db.AutoCyclePhaseScanning:
|
||||
o.autoCycleCheckScan(ctx, ac, now)
|
||||
return
|
||||
case db.AutoCyclePhaseRunning:
|
||||
o.autoCycleCheckRun(ctx, ac, now)
|
||||
return
|
||||
}
|
||||
@@ -118,95 +131,145 @@ func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
|
||||
o.autoCycleStartRun(ctx, ac, now)
|
||||
}
|
||||
|
||||
// autoCycleStartRun performs steps 1-2 of the scenario (clear the queue,
|
||||
// scan floating IPs) and moves the state to running, or straight to waiting
|
||||
// if there is nothing to wait for.
|
||||
// autoCycleStartRun begins a cycle: it starts the background scan job (clear
|
||||
// the queue, then discover and enqueue the free floating IPs) and moves to
|
||||
// phase `scanning` right away. Nothing slow happens here, so autoCycleMu is
|
||||
// released immediately and the control loop keeps ticking.
|
||||
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
||||
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
||||
next := now.Add(interval)
|
||||
if _, started := o.StartScan(ScanOptions{ClearFirst: true}); !started {
|
||||
// A scan started by someone else (an operator's manual or dry-run scan,
|
||||
// the periodic scan) is in flight. It is not this cycle's scan: it may
|
||||
// not clear the queue, or may not enqueue anything at all (dry run), so
|
||||
// adopting it would end in a "completed" cycle over an untouched or
|
||||
// empty queue. Change nothing and try again on the next step, once it
|
||||
// has finished (it is bounded by fip_scan_timeout_seconds).
|
||||
o.Log.Info("auto cycle: another floating ip scan is running, waiting for it to finish")
|
||||
return
|
||||
}
|
||||
|
||||
st := autoCycleStateOf(ac)
|
||||
st.LastRunStartedAt = &now
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = nil
|
||||
st.Phase = db.AutoCyclePhaseScanning
|
||||
st.LastError = ""
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
fail := func(step string, err error) {
|
||||
o.Log.Error("auto cycle: step failed", "step", step, "err", err)
|
||||
// autoCycleCheckScan handles the scanning phase by polling the scan job.
|
||||
func (o *Orchestrator) autoCycleCheckScan(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
||||
scan := o.ScanStatus()
|
||||
switch {
|
||||
case scan.Running:
|
||||
return
|
||||
case scan.State == ScanIdle:
|
||||
// Phase `scanning` but no job in this process: control-api restarted
|
||||
// mid-scan. Starting again is idempotent (clear + scan).
|
||||
o.Log.Info("auto cycle: restarting floating ip scan after restart")
|
||||
o.StartScan(ScanOptions{ClearFirst: true})
|
||||
return
|
||||
}
|
||||
|
||||
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
||||
next := now.Add(interval)
|
||||
st := autoCycleStateOf(ac)
|
||||
|
||||
switch scan.State {
|
||||
case ScanError:
|
||||
o.Log.Error("auto cycle: step failed", "step", "scan floating ips", "err", scan.Error)
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeError
|
||||
st.LastError = fmt.Sprintf("%s: %v", step, err)
|
||||
if uerr := o.DB.UpdateAutoCycleState(ctx, st); uerr != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", uerr)
|
||||
st.LastError = "scan floating ips: " + scan.Error
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
|
||||
"step": step,
|
||||
"error": err.Error(),
|
||||
"step": "scan floating ips",
|
||||
"error": scan.Error,
|
||||
}))
|
||||
}
|
||||
|
||||
if _, err := o.ClearQueue(ctx); err != nil {
|
||||
fail("clear queue", err)
|
||||
return
|
||||
}
|
||||
_, scanned, err := o.ScanFloatingIPs(ctx)
|
||||
if err != nil {
|
||||
fail("scan floating ips", err)
|
||||
return
|
||||
}
|
||||
st.LastScannedFree = scanned
|
||||
st.LastError = ""
|
||||
|
||||
if scanned == 0 {
|
||||
// Nothing was queued; waiting for completion would never end.
|
||||
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
|
||||
case ScanCancelled:
|
||||
if life := o.lifetimeErr(); life != nil {
|
||||
// The process is shutting down: leave the phase as is so the next
|
||||
// start resumes the cycle (scanning without a job restarts it).
|
||||
return
|
||||
}
|
||||
// Cancelled by an operator (CancelScan): treat as an interrupted run.
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
|
||||
st.LastOutcome = db.AutoCycleOutcomeStopped
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
st.Phase = db.AutoCyclePhaseRunning
|
||||
st.RunStartedAt = &now
|
||||
st.NextRunAt = nil
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
default: // ScanDone
|
||||
st.LastScannedFree = scan.Free
|
||||
st.LastError = ""
|
||||
if scan.Free == 0 {
|
||||
// Nothing was queued; waiting for completion would never end.
|
||||
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
// max_run_seconds counts from the end of the scan.
|
||||
st.Phase = db.AutoCyclePhaseRunning
|
||||
st.RunStartedAt = &now
|
||||
st.NextRunAt = nil
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
||||
"reason": "cycle",
|
||||
"scanned_free": scan.Free,
|
||||
}))
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
||||
"reason": "cycle",
|
||||
"scanned_free": scanned,
|
||||
}))
|
||||
}
|
||||
|
||||
func (o *Orchestrator) lifetimeErr() error {
|
||||
o.scan.mu.Lock()
|
||||
defer o.scan.mu.Unlock()
|
||||
return o.scan.lifetime().Err()
|
||||
}
|
||||
|
||||
// autoCycleCheckRun handles the running phase: finish the cycle once every
|
||||
// queued address is terminal, or give up after max_run_seconds.
|
||||
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
||||
items, err := o.DB.ListIPs(ctx)
|
||||
// An empty queue counts as finished: right after the scan it cannot be
|
||||
// empty (scanned > 0), so it only happens when an operator cleared or
|
||||
// deleted every address mid-cycle — and then there is nothing to wait for
|
||||
// (with max_run_seconds=0 the cycle would otherwise hang forever).
|
||||
// EXISTS is cheap enough to run on every tick, however long the queue.
|
||||
pending, err := o.DB.AnyNonTerminalIP(ctx)
|
||||
if err != nil {
|
||||
o.Log.Error("auto cycle: list ips", "err", err)
|
||||
o.Log.Error("auto cycle: check pending ips", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
||||
next := now.Add(interval)
|
||||
|
||||
// An empty queue counts as finished: right after the scan it cannot be
|
||||
// empty (scanned > 0), so it only happens when an operator cleared or
|
||||
// deleted every address mid-cycle — and then there is nothing to wait for
|
||||
// (with max_run_seconds=0 the cycle would otherwise hang forever).
|
||||
allTerminal := true
|
||||
for _, it := range items {
|
||||
if !isTerminalIPState(it.State) {
|
||||
allTerminal = false
|
||||
break
|
||||
if !pending {
|
||||
_, addresses, err := o.DB.CountIPsByState(ctx)
|
||||
if err != nil {
|
||||
o.Log.Error("auto cycle: count ips", "err", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
if allTerminal {
|
||||
st := autoCycleStateOf(ac)
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
@@ -220,7 +283,7 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
|
||||
"addresses": len(items),
|
||||
"addresses": addresses,
|
||||
"runs": st.RunsTotal,
|
||||
}))
|
||||
return
|
||||
@@ -228,6 +291,11 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
|
||||
|
||||
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
|
||||
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
|
||||
_, addresses, err := o.DB.CountIPsByState(ctx)
|
||||
if err != nil {
|
||||
o.Log.Error("auto cycle: count ips", "err", err)
|
||||
return
|
||||
}
|
||||
// The queue is left untouched: the next cycle clears it anyway, and
|
||||
// an operator can still inspect what got stuck.
|
||||
st := autoCycleStateOf(ac)
|
||||
@@ -243,7 +311,7 @@ func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, n
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
|
||||
"max_run_seconds": ac.MaxRunSeconds,
|
||||
"addresses": len(items),
|
||||
"addresses": addresses,
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user