Scan floating IPs in the background, page by page, so thousands of addresses work

The "Scan Floating IP" button failed with a client timeout: the project now
holds ~6.4k floating IPs and the scan listed them all in one unpaginated,
timeout-less Neutron request on the HTTP request context.

openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them
small) with per-page retry/backoff on transport errors, 5xx and 429, and every
request now has a timeout (also ends hangs inside the orchestrator tick).

orchestrator: the scan is a single-flight background job on the process
context with progress (clearing/listing/enqueuing/done/error), dry_run, full
discovery before anything is enqueued, then SubmitIPs in chunks of 500 in
ascending IP order; a failed read leaves the queue untouched. The auto-cycle
gets a "scanning" phase that polls the job, so the control loop and
autoCycleMu are never held across OpenStack/DB work; it recovers after a
restart and waits for (instead of adopting) a scan started by someone else.

db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY
counters, EXISTS completion check, set-based ClearAllIPs.

API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging
and filters on /admin/ips and /admin/registry (bare arrays without limit),
results_by_overall in /admin/status.

dashboard: scan progress panel and dry-run button, paginated /ips and
/registry with server-side filters, Overview on counters and capped lists
with progress/ETA, "select all N by filter", hx-params fix for per-row
buttons, real counts in confirmations.

Also: docs (API, USAGE, DASHBOARD, README), plan and review under
docs/changes/, bin/ rebuilt with new SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-01 19:31:11 +03:00
1 parent debf2afed2
commit aff8fe38b5
61 files changed
+5833 -536

No files matched your search

+294 -15
View File
@@ -58,6 +58,37 @@ func countEvents(t *testing.T, d *db.DB, eventType string) int {
return n
}
// waitScan blocks until the background scan job is no longer running.
func waitScan(t *testing.T, o *Orchestrator) ScanStatus {
t.Helper()
return waitScanFor(t, o, 30*time.Second)
}
func waitScanFor(t *testing.T, o *Orchestrator, limit time.Duration) ScanStatus {
t.Helper()
deadline := time.Now().Add(limit)
for {
if st := o.ScanStatus(); !st.Running {
return st
}
if time.Now().After(deadline) {
t.Fatalf("scan job did not finish in time: %+v", o.ScanStatus())
}
time.Sleep(2 * time.Millisecond)
}
}
// stepStartCycle drives one cycle start: the first step launches the
// background scan (phase scanning), then we wait for the job and the second
// step, at the same virtual time, consumes its result.
func stepStartCycle(t *testing.T, o *Orchestrator, now time.Time) {
t.Helper()
ctx := context.Background()
o.autoCycleStep(ctx, now)
waitScan(t, o)
o.autoCycleStep(ctx, now)
}
func TestAutoCycleDisabledIsNoOp(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
@@ -111,7 +142,7 @@ func TestAutoCycleStartClearsQueueAndScans(t *testing.T) {
}
now := db.Now()
o.autoCycleStep(ctx, now)
stepStartCycle(t, o, now)
got := queuedAddresses(t, d)
if _, stale := got["9.9.9.9"]; stale {
@@ -154,7 +185,7 @@ func TestAutoCycleWaitsWhileChecksInProgressAndTickPicksUpQueue(t *testing.T) {
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
// Existing Tick logic starts the checks on its own.
o.Tick(ctx)
@@ -198,7 +229,7 @@ func TestAutoCycleCompletesAndRepeatsAfterInterval(t *testing.T) {
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
// done, failed and occupied are all terminal.
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPDone, "1.1.1.1"); err != nil {
@@ -241,7 +272,7 @@ func TestAutoCycleCompletesAndRepeatsAfterInterval(t *testing.T) {
}
// At next_run_at: new cycle starts, queue is rebuilt from scratch.
o.autoCycleStep(ctx, wantNext)
stepStartCycle(t, o, wantNext)
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected running again, got %+v", ac)
@@ -267,7 +298,7 @@ func TestAutoCycleTimeout(t *testing.T) {
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
o.autoCycleStep(ctx, t0.Add(300*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
@@ -302,7 +333,7 @@ func TestAutoCycleNoLimitNeverTimesOut(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
o.autoCycleStep(ctx, t0.Add(1000*time.Hour))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("max_run_seconds=0 means no limit, got %+v", ac)
@@ -320,7 +351,7 @@ func TestAutoCycleManualClearMidCycleCompletes(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected running after start, got %+v", ac)
}
@@ -350,7 +381,7 @@ func TestAutoCycleNoFreeIPs(t *testing.T) {
}
now := db.Now()
o.autoCycleStep(ctx, now)
stepStartCycle(t, o, now)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeNoFreeIPs {
@@ -378,7 +409,7 @@ func TestAutoCycleOpenStackErrorRetriesNextInterval(t *testing.T) {
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeError {
@@ -400,7 +431,7 @@ func TestAutoCycleOpenStackErrorRetriesNextInterval(t *testing.T) {
// OpenStack recovers: the next interval starts a normal cycle and the
// stale error is cleared.
mock.ListFailure = nil
o.autoCycleStep(ctx, t0.Add(60*time.Second))
stepStartCycle(t, o, t0.Add(60*time.Second))
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastError != "" {
t.Fatalf("expected running with cleared error, got %+v", ac)
@@ -415,7 +446,7 @@ func TestAutoCycleStartIsIdempotent(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("second start: %v", err)
@@ -440,7 +471,7 @@ func TestAutoCycleStopMidCycle(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
o.Tick(ctx) // check in flight
if err := o.StopAutoCycle(ctx); err != nil {
@@ -484,7 +515,7 @@ func TestAutoCycleStopWhileWaitingKeepsLastOutcome(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
finishAllIPs(t, d, db.IPDone)
o.autoCycleStep(ctx, t0.Add(5*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
@@ -512,7 +543,7 @@ func TestAutoCycleSurvivesRestart(t *testing.T) {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
stepStartCycle(t, o, t0)
// A fresh Orchestrator on the same database (a control-api restart)
// continues the running phase instead of starting over.
@@ -537,8 +568,256 @@ func TestAutoCycleSurvivesRestart(t *testing.T) {
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting {
t.Fatalf("expected waiting until next_run_at, got %+v", ac)
}
o3.autoCycleStep(ctx, t1.Add(300*time.Second))
stepStartCycle(t, o3, t1.Add(300*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected a new cycle at next_run_at, got %+v", ac)
}
}
func TestAutoCycleScanningPhaseThenRunning(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedMany("fip", 30)
mock.PageSize = 10
mock.PageDelay = 100 * time.Millisecond
setAutoCycleParams(t, d, 60, 300)
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
t.Fatal(err)
}
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatal(err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseScanning || ac.RunStartedAt != nil || ac.NextRunAt != nil {
t.Fatalf("expected scanning without run_started_at, got %+v", ac)
}
if ac.LastRunStartedAt == nil || !ac.LastRunStartedAt.Equal(t0) {
t.Fatalf("expected last_run_started_at=%v, got %v", t0, ac.LastRunStartedAt)
}
if !o.ScanStatus().Running {
t.Fatalf("expected the scan job to be running")
}
// While the job runs, further steps leave the phase alone, even far past
// max_run_seconds (it only counts from the end of the scan).
o.autoCycleStep(ctx, t0.Add(time.Hour))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseScanning || ac.LastOutcome != "" {
t.Fatalf("expected still scanning, got %+v", ac)
}
waitScan(t, o)
t1 := t0.Add(2 * time.Hour)
o.autoCycleStep(ctx, t1)
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastScannedFree != 30 {
t.Fatalf("expected running with last_scanned_free=30, got %+v", ac)
}
if ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t1) {
t.Fatalf("run_started_at must be the scan end %v, got %v", t1, ac.RunStartedAt)
}
if got := queuedAddresses(t, d); len(got) != 30 {
t.Fatalf("expected the 30 scanned addresses only, got %d", len(got))
}
// max_run_seconds is measured from t1.
o.autoCycleStep(ctx, t1.Add(300*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected still running at the limit, got %+v", ac)
}
o.autoCycleStep(ctx, t1.Add(301*time.Second))
if ac := getAutoCycle(t, d); ac.LastOutcome != db.AutoCycleOutcomeTimeout {
t.Fatalf("expected timeout, got %+v", ac)
}
}
func TestAutoCycleScanErrorSurfacesAndRetries(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedMany("fip", 5)
mock.ListFailure = errors.New("neutron is down")
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatal(err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseScanning {
t.Fatalf("expected scanning first, got %+v", ac)
}
waitScan(t, o)
t1 := t0.Add(time.Second)
o.autoCycleStep(ctx, t1)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeError ||
!strings.Contains(ac.LastError, "neutron is down") || ac.NextRunAt == nil || !ac.NextRunAt.Equal(t1.Add(time.Minute)) {
t.Fatalf("expected waiting/error with next_run_at=t1+interval, got %+v", ac)
}
if countEvents(t, d, "auto_cycle_error") != 1 {
t.Fatalf("expected one auto_cycle_error event")
}
}
func TestAutoCycleRestartInScanningPhaseStartsScanAgain(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedMany("fip", 12)
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
t.Fatal(err)
}
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatal(err)
}
t0 := db.Now()
// Persisted state of a process that died mid-scan.
if err := d.UpdateAutoCycleState(ctx, db.AutoCycleState{Phase: db.AutoCyclePhaseScanning, LastRunStartedAt: &t0}); err != nil {
t.Fatal(err)
}
o2 := &Orchestrator{DB: d, OS: mock, Cfg: o.Cfg, Agg: o.Agg, Log: o.Log}
t.Cleanup(func() { o2.CancelScan() })
if st := o2.ScanStatus(); st.State != ScanIdle {
t.Fatalf("fresh process must have no scan job, got %+v", st)
}
o2.autoCycleStep(ctx, t0.Add(time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseScanning {
t.Fatalf("expected to stay in scanning while the new job runs, got %+v", ac)
}
if st := o2.ScanStatus(); st.State == ScanIdle {
t.Fatalf("expected a new scan job to be started")
}
waitScan(t, o2)
t1 := t0.Add(2 * time.Second)
o2.autoCycleStep(ctx, t1)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastScannedFree != 12 || ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t1) {
t.Fatalf("expected running after the restarted scan, got %+v", ac)
}
got := queuedAddresses(t, d)
if _, stale := got["9.9.9.9"]; stale || len(got) != 12 {
t.Fatalf("restarted scan must clear and re-scan, got %d rows (stale=%v)", len(got), stale)
}
}
func TestAutoCycleStopCancelsScan(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedMany("fip", 30)
mock.PageSize = 10
mock.PageDelay = 5 * time.Second
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatal(err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseScanning || !o.ScanStatus().Running {
t.Fatalf("precondition: scanning with a live job, got %+v", ac)
}
if err := o.StopAutoCycle(ctx); err != nil {
t.Fatalf("stop: %v", err)
}
ac := getAutoCycle(t, d)
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.LastOutcome != db.AutoCycleOutcomeStopped {
t.Fatalf("expected disabled/idle/stopped, got %+v", ac)
}
if st := o.ScanStatus(); st.State != ScanCancelled || st.Running {
t.Fatalf("expected the scan to be cancelled, got %+v", st)
}
// Later steps do nothing.
o.autoCycleStep(ctx, t0.Add(time.Hour))
if ac := getAutoCycle(t, d); ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle {
t.Fatalf("expected no activity after stop, got %+v", ac)
}
}
// A slow scan must neither block the control loop (Tick, auto-cycle steps)
// nor Start/Stop/Get on the auto-cycle: autoCycleMu is never held across it.
func TestSlowScanDoesNotBlockTickOrAutoCycleCalls(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedMany("fip", 20)
mock.PageSize = 10
mock.PageDelay = 1500 * time.Millisecond
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
t.Fatal(err)
}
within := func(what string, limit time.Duration, f func()) {
t.Helper()
start := time.Now()
f()
if el := time.Since(start); el > limit {
t.Fatalf("%s took %v while a scan was running (limit %v)", what, el, limit)
}
}
within("StartAutoCycle", 500*time.Millisecond, func() {
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatal(err)
}
})
t0 := db.Now()
within("first auto-cycle step", 500*time.Millisecond, func() { o.autoCycleStep(ctx, t0) })
if !o.ScanStatus().Running {
t.Fatalf("scan should be in flight")
}
within("Tick", 500*time.Millisecond, func() { o.Tick(ctx) })
within("polling auto-cycle step", 500*time.Millisecond, func() { o.autoCycleStep(ctx, t0.Add(time.Second)) })
within("GetAutoCycle", 500*time.Millisecond, func() {
if _, err := o.GetAutoCycle(ctx); err != nil {
t.Fatal(err)
}
})
within("StopAutoCycle", 3*time.Second, func() {
if err := o.StopAutoCycle(ctx); err != nil {
t.Fatal(err)
}
})
}
// A scan that is already running (an operator's dry run, a manual or periodic
// scan) is NOT a cycle's scan: it may not clear the queue or enqueue anything.
// The cycle must wait for it and then run its own clear+scan; following the
// foreign job would end in "completed" over an untouched/empty queue.
func TestAutoCycleWaitsForForeignScanInsteadOfFollowingIt(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
mock.PageSize = 1
mock.PageDelay = 80 * time.Millisecond // keeps the dry run in flight for a while
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
t.Fatalf("seed queue: %v", err)
}
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
if _, started := o.StartScan(ScanOptions{DryRun: true}); !started {
t.Fatalf("precondition: the dry run must start")
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
ac := getAutoCycle(t, d)
if ac.Phase == db.AutoCyclePhaseScanning || ac.Phase == db.AutoCyclePhaseRunning {
t.Fatalf("cycle must not adopt the foreign scan, got phase %q", ac.Phase)
}
if got := queuedAddresses(t, d); got["9.9.9.9"] != db.IPQueued {
t.Fatalf("queue must be untouched while the foreign scan runs, got %v", got)
}
waitScan(t, o) // the dry run ends
mock.PageDelay = 0
stepStartCycle(t, o, t0.Add(time.Second))
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastScannedFree != 1 {
t.Fatalf("expected the cycle's own scan to complete (running, 1 free), got %+v", ac)
}
got := queuedAddresses(t, d)
if _, stale := got["9.9.9.9"]; stale || got["1.1.1.1"] != db.IPQueued {
t.Fatalf("the cycle's own scan must clear the old queue and enqueue 1.1.1.1, got %v", got)
}
}