Scan floating IPs in the background, page by page, so thousands of addresses work
The "Scan Floating IP" button failed with a client timeout: the project now holds ~6.4k floating IPs and the scan listed them all in one unpaginated, timeout-less Neutron request on the HTTP request context. openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them small) with per-page retry/backoff on transport errors, 5xx and 429, and every request now has a timeout (also ends hangs inside the orchestrator tick). orchestrator: the scan is a single-flight background job on the process context with progress (clearing/listing/enqueuing/done/error), dry_run, full discovery before anything is enqueued, then SubmitIPs in chunks of 500 in ascending IP order; a failed read leaves the queue untouched. The auto-cycle gets a "scanning" phase that polls the job, so the control loop and autoCycleMu are never held across OpenStack/DB work; it recovers after a restart and waits for (instead of adopting) a scan started by someone else. db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY counters, EXISTS completion check, set-based ClearAllIPs. API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging and filters on /admin/ips and /admin/registry (bare arrays without limit), results_by_overall in /admin/status. dashboard: scan progress panel and dry-run button, paginated /ips and /registry with server-side filters, Overview on counters and capped lists with progress/ETA, "select all N by filter", hx-params fix for per-row buttons, real counts in confirmations. Also: docs (API, USAGE, DASHBOARD, README), plan and review under docs/changes/, bin/ rebuilt with new SHA256SUMS. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
debf2afed2
commit
aff8fe38b5
61 files changed
+5833
-536
No files matched your search
+14
-2
@@ -59,6 +59,9 @@ func run(configPath string, log *slog.Logger) error {
|
||||
}
|
||||
|
||||
orch := orchestrator.New(database, osClient, cfg, log)
|
||||
// Background jobs (the floating-IP scan) live as long as the process, not
|
||||
// as long as the HTTP request or loop iteration that started them.
|
||||
orch.SetContext(ctx)
|
||||
|
||||
adminToken := os.Getenv(cfg.Auth.AdminTokenEnv)
|
||||
agentToken := os.Getenv(cfg.Auth.AgentTokenEnv)
|
||||
@@ -135,8 +138,10 @@ func runOrchestratorLoop(ctx context.Context, orch *orchestrator.Orchestrator, c
|
||||
} else if ac.Enabled {
|
||||
continue
|
||||
}
|
||||
if _, _, err := orch.ScanFloatingIPs(ctx); err != nil {
|
||||
log.Error("scan floating ips", "err", err)
|
||||
// Non-blocking: the scan runs in the background (single-flight, so
|
||||
// a still-running scan is simply joined) and must not stall Tick.
|
||||
if st, started := orch.StartScan(orchestrator.ScanOptions{}); !started {
|
||||
log.Info("periodic floating ip scan skipped: a scan is already running", "state", st.State)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -158,11 +163,18 @@ func newOpenStackClient(ctx context.Context, cfg *config.ControlAPI) (openstack.
|
||||
}
|
||||
|
||||
func newRealOpenStackClient(ctx context.Context, cfg *config.ControlAPI) (openstack.FloatingIPClient, error) {
|
||||
retries := cfg.OpenStack.ListPageRetries
|
||||
if retries < 0 {
|
||||
retries = 0 // negative in the config disables retries
|
||||
}
|
||||
clientCfg := openstack.ClientConfig{
|
||||
AuthURL: os.Getenv(cfg.OpenStack.AuthURLEnv),
|
||||
ProjectID: os.Getenv(cfg.OpenStack.ProjectIDEnv),
|
||||
Region: os.Getenv(cfg.OpenStack.RegionEnv),
|
||||
Interface: os.Getenv(cfg.OpenStack.InterfaceEnv),
|
||||
|
||||
RequestTimeout: time.Duration(cfg.OpenStack.RequestTimeoutSeconds) * time.Second,
|
||||
ListPageRetries: retries,
|
||||
}
|
||||
|
||||
switch cfg.OpenStack.AuthMethod {
|
||||
|
||||
Reference in new issue
Block a user