The "Scan Floating IP" button failed with a client timeout: the project now holds ~6.4k floating IPs and the scan listed them all in one unpaginated, timeout-less Neutron request on the HTTP request context. openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them small) with per-page retry/backoff on transport errors, 5xx and 429, and every request now has a timeout (also ends hangs inside the orchestrator tick). orchestrator: the scan is a single-flight background job on the process context with progress (clearing/listing/enqueuing/done/error), dry_run, full discovery before anything is enqueued, then SubmitIPs in chunks of 500 in ascending IP order; a failed read leaves the queue untouched. The auto-cycle gets a "scanning" phase that polls the job, so the control loop and autoCycleMu are never held across OpenStack/DB work; it recovers after a restart and waits for (instead of adopting) a scan started by someone else. db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY counters, EXISTS completion check, set-based ClearAllIPs. API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging and filters on /admin/ips and /admin/registry (bare arrays without limit), results_by_overall in /admin/status. dashboard: scan progress panel and dry-run button, paginated /ips and /registry with server-side filters, Overview on counters and capped lists with progress/ETA, "select all N by filter", hx-params fix for per-row buttons, real counts in confirmations. Also: docs (API, USAGE, DASHBOARD, README), plan and review under docs/changes/, bin/ rebuilt with new SHA256SUMS. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
399 lines
13 KiB
Go
399 lines
13 KiB
Go
package orchestrator
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"net/netip"
|
|
"strconv"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"cloudipvalidator/internal/db"
|
|
)
|
|
|
|
func eventPayload(t *testing.T, d *db.DB, eventType string) string {
|
|
t.Helper()
|
|
var p string
|
|
if err := d.QueryRowContext(context.Background(),
|
|
`SELECT payload FROM events WHERE event_type=? ORDER BY id DESC LIMIT 1`, eventType).Scan(&p); err != nil {
|
|
t.Fatalf("read %s event: %v", eventType, err)
|
|
}
|
|
return p
|
|
}
|
|
|
|
func TestScanStatusIdleBeforeAnyScan(t *testing.T) {
|
|
// Zero-value Orchestrator literal (as several tests build it) must work.
|
|
o := &Orchestrator{}
|
|
st := o.ScanStatus()
|
|
if st.State != ScanIdle || st.Running || st.StartedAt != nil {
|
|
t.Fatalf("expected idle, got %+v", st)
|
|
}
|
|
if o.CancelScan() {
|
|
t.Fatalf("nothing to cancel")
|
|
}
|
|
}
|
|
|
|
func TestScanJobSingleFlightAndProgress(t *testing.T) {
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.SeedMany("fip", 600)
|
|
mock.SeedWithPort("busy", "203.0.113.9", "svc", "port-x")
|
|
mock.PageSize = 100
|
|
mock.PageDelay = 40 * time.Millisecond
|
|
|
|
st, started := o.StartScan(ScanOptions{})
|
|
if !started || !st.Running || st.State != ScanListing || st.StartedAt == nil {
|
|
t.Fatalf("expected a started listing job, got started=%v %+v", started, st)
|
|
}
|
|
st2, started2 := o.StartScan(ScanOptions{DryRun: true})
|
|
if started2 || !st2.Running || st2.DryRun {
|
|
t.Fatalf("second start must join the running job (not dry-run), got started=%v %+v", started2, st2)
|
|
}
|
|
|
|
// The synchronous wrapper joins the same job instead of scanning twice.
|
|
type out struct {
|
|
res db.SubmitIPsResult
|
|
scanned int
|
|
err error
|
|
}
|
|
ch := make(chan out, 1)
|
|
go func() {
|
|
r, n, err := o.ScanFloatingIPs(context.Background())
|
|
ch <- out{r, n, err}
|
|
}()
|
|
|
|
// Progress is visible while the job runs.
|
|
sawProgress := false
|
|
for i := 0; i < 200 && o.ScanStatus().Running; i++ {
|
|
if s := o.ScanStatus(); s.Pages > 0 && s.Pages < 7 {
|
|
sawProgress = true
|
|
}
|
|
time.Sleep(5 * time.Millisecond)
|
|
}
|
|
got := <-ch
|
|
if got.err != nil || got.scanned != 600 || len(got.res.Added) != 600 {
|
|
t.Fatalf("wrapper result: %+v", got)
|
|
}
|
|
if !sawProgress {
|
|
t.Fatalf("expected to observe intermediate progress")
|
|
}
|
|
|
|
fin := waitScan(t, o)
|
|
if fin.State != ScanDone || fin.Running || fin.FinishedAt == nil || fin.Error != "" {
|
|
t.Fatalf("expected done, got %+v", fin)
|
|
}
|
|
if fin.Pages != 7 || fin.Discovered != 601 || fin.Free != 600 || fin.Added != 600 {
|
|
t.Fatalf("unexpected counters: %+v", fin)
|
|
}
|
|
if countEvents(t, d, "fip_scan") != 1 {
|
|
t.Fatalf("expected exactly one fip_scan event, got %d", countEvents(t, d, "fip_scan"))
|
|
}
|
|
var p struct {
|
|
ScannedFree int `json:"scanned_free"`
|
|
Pages int `json:"pages"`
|
|
Added int `json:"added"`
|
|
}
|
|
if err := json.Unmarshal([]byte(eventPayload(t, d, "fip_scan")), &p); err != nil || p.ScannedFree != 600 || p.Added != 600 || p.Pages != 7 {
|
|
t.Fatalf("fip_scan payload: %+v err=%v", p, err)
|
|
}
|
|
|
|
// A new scan after completion starts a fresh job; everything is requeued
|
|
// or reordered (idempotent), nothing added.
|
|
mock.PageDelay = 0
|
|
st3, started3 := o.StartScan(ScanOptions{})
|
|
if !started3 {
|
|
t.Fatalf("a finished job must not block a new one: %+v", st3)
|
|
}
|
|
fin = waitScan(t, o)
|
|
if fin.State != ScanDone || fin.Added != 0 || fin.Reordered != 600 {
|
|
t.Fatalf("rescan: %+v", fin)
|
|
}
|
|
}
|
|
|
|
func TestScanJobDryRunLeavesQueueUntouched(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.SeedMany("fip", 450)
|
|
mock.PageSize = 200
|
|
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
o.StartScan(ScanOptions{DryRun: true, ClearFirst: true}) // ClearFirst is ignored for dry runs
|
|
st := waitScan(t, o)
|
|
if st.State != ScanDone || !st.DryRun || st.Free != 450 || st.Pages != 3 || st.Added != 0 {
|
|
t.Fatalf("dry run status: %+v", st)
|
|
}
|
|
if got := queuedAddresses(t, d); len(got) != 1 || got["9.9.9.9"] != db.IPQueued {
|
|
t.Fatalf("dry run must not touch the queue, got %d rows", len(got))
|
|
}
|
|
if countEvents(t, d, "fip_scan") != 0 || countEvents(t, d, "queue_cleared") != 0 {
|
|
t.Fatalf("dry run must not emit scan/clear events")
|
|
}
|
|
}
|
|
|
|
func TestScanJobReadErrorLeavesQueueUntouched(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.SeedMany("fip", 450)
|
|
mock.PageSize = 200
|
|
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// Page 1 succeeds, page 2 fails: the 200 already read must NOT be queued.
|
|
mock.ListFailures = []error{nil, errors.New("neutron exploded")}
|
|
|
|
o.StartScan(ScanOptions{})
|
|
st := waitScan(t, o)
|
|
if st.State != ScanError || !strings.Contains(st.Error, "neutron exploded") || st.FinishedAt == nil {
|
|
t.Fatalf("expected error state, got %+v", st)
|
|
}
|
|
if st.Pages != 1 || st.Added != 0 {
|
|
t.Fatalf("expected 1 page read and nothing added, got %+v", st)
|
|
}
|
|
if got := queuedAddresses(t, d); len(got) != 1 || got["9.9.9.9"] != db.IPQueued {
|
|
t.Fatalf("queue must be untouched after a read error, got %d rows", len(got))
|
|
}
|
|
if countEvents(t, d, "fip_scan") != 0 {
|
|
t.Fatalf("no fip_scan event on failure")
|
|
}
|
|
|
|
// The wrapper reports the same error.
|
|
mock.ListFailure = errors.New("still down")
|
|
if _, _, err := o.ScanFloatingIPs(ctx); err == nil || !strings.Contains(err.Error(), "still down") {
|
|
t.Fatalf("expected wrapped list error, got %v", err)
|
|
}
|
|
}
|
|
|
|
func TestScanJobClearFirst(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-1", "1.1.1.1", "svc")
|
|
if err := d.SeedQueue(ctx, []string{"9.9.9.9", "8.8.8.8"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
st, _ := o.StartScan(ScanOptions{ClearFirst: true})
|
|
if st.State != ScanClearing {
|
|
t.Fatalf("expected clearing as first state, got %s", st.State)
|
|
}
|
|
fin := waitScan(t, o)
|
|
if fin.State != ScanDone || fin.Added != 1 {
|
|
t.Fatalf("clear-first scan: %+v", fin)
|
|
}
|
|
if got := queuedAddresses(t, d); len(got) != 1 || got["1.1.1.1"] != db.IPQueued {
|
|
t.Fatalf("expected only the scanned address, got %v", got)
|
|
}
|
|
var payload struct {
|
|
Count int `json:"count"`
|
|
Addresses []string `json:"addresses"`
|
|
}
|
|
if err := json.Unmarshal([]byte(eventPayload(t, d, "queue_cleared")), &payload); err != nil || payload.Count != 2 || len(payload.Addresses) != 2 {
|
|
t.Fatalf("queue_cleared payload: %+v err=%v", payload, err)
|
|
}
|
|
}
|
|
|
|
// 6440 free addresses (the real stand's size) plus a few out-of-order ones:
|
|
// everything is queued in ascending numeric IP order within a sane time.
|
|
func TestScanJobEnqueuesAllInAscendingOrderAtScale(t *testing.T) {
|
|
if testing.Short() {
|
|
t.Skip("scale test skipped in -short mode")
|
|
}
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
o.ScanPageSize = 200
|
|
mock.SeedMany("fip", 6440)
|
|
// IDs sort before "fip-*", addresses sort numerically before 198.18.*;
|
|
// 10.0.0.9 < 10.0.0.200 numerically but not as strings.
|
|
mock.Seed("aaa-2", "10.0.0.200", "svc")
|
|
mock.Seed("aaa-1", "10.0.0.9", "svc")
|
|
mock.SeedWithPort("occupied", "203.0.113.1", "svc", "port-x")
|
|
|
|
start := time.Now()
|
|
o.StartScan(ScanOptions{})
|
|
st := waitScanFor(t, o, 5*time.Minute) // -race is an order of magnitude slower
|
|
elapsed := time.Since(start)
|
|
|
|
if st.State != ScanDone || st.Free != 6442 || st.Added != 6442 || st.Discovered != 6443 || st.Pages != 33 {
|
|
t.Fatalf("scan status: %+v", st)
|
|
}
|
|
if elapsed > 3*time.Minute {
|
|
t.Fatalf("scanning 6442 addresses took %v", elapsed)
|
|
}
|
|
t.Logf("scan+enqueue of %d addresses took %v", st.Added, elapsed)
|
|
|
|
items, err := d.ListIPs(ctx)
|
|
if err != nil || len(items) != 6442 {
|
|
t.Fatalf("queue: n=%d err=%v", len(items), err)
|
|
}
|
|
var prev netip.Addr
|
|
for i, it := range items {
|
|
ip := netip.MustParseAddr(it.IPAddress)
|
|
if i > 0 && ip.Compare(prev) <= 0 {
|
|
t.Fatalf("queue not ascending at %d: %s after %s", i, it.IPAddress, prev)
|
|
}
|
|
if it.Sequence != i {
|
|
t.Fatalf("expected contiguous sequences, row %d has %d", i, it.Sequence)
|
|
}
|
|
prev = ip
|
|
}
|
|
if items[0].IPAddress != "10.0.0.9" || items[1].IPAddress != "10.0.0.200" {
|
|
t.Fatalf("numeric ordering broken: %s, %s", items[0].IPAddress, items[1].IPAddress)
|
|
}
|
|
if _, stale := queuedAddresses(t, d)["203.0.113.1"]; stale {
|
|
t.Fatalf("occupied address must not be queued")
|
|
}
|
|
}
|
|
|
|
func TestScanJobCancelAndLifetimeContext(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.SeedMany("fip", 100)
|
|
mock.PageSize = 10
|
|
mock.PageDelay = 5 * time.Second
|
|
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
o.StartScan(ScanOptions{})
|
|
time.Sleep(20 * time.Millisecond)
|
|
start := time.Now()
|
|
if !o.CancelScan() {
|
|
t.Fatalf("expected a running scan to be cancelled")
|
|
}
|
|
if time.Since(start) > 3*time.Second {
|
|
t.Fatalf("cancel took too long")
|
|
}
|
|
st := o.ScanStatus()
|
|
if st.State != ScanCancelled || st.Running || st.FinishedAt == nil {
|
|
t.Fatalf("expected cancelled, got %+v", st)
|
|
}
|
|
if got := queuedAddresses(t, d); len(got) != 1 {
|
|
t.Fatalf("cancelled scan must not enqueue, got %d rows", len(got))
|
|
}
|
|
|
|
// Cancelling the process-lifetime context also stops a running scan.
|
|
life, cancelLife := context.WithCancel(context.Background())
|
|
o.SetContext(life)
|
|
o.StartScan(ScanOptions{})
|
|
time.Sleep(20 * time.Millisecond)
|
|
cancelLife()
|
|
if st := waitScan(t, o); st.State != ScanCancelled {
|
|
t.Fatalf("expected cancelled by lifetime ctx, got %+v", st)
|
|
}
|
|
}
|
|
|
|
func TestScanJobDeadline(t *testing.T) {
|
|
o, _, mock := newTestOrchestrator(t, 180)
|
|
o.Cfg.FIPScanTimeoutSeconds = 1
|
|
mock.SeedMany("fip", 20)
|
|
mock.PageSize = 10
|
|
mock.PageDelay = 10 * time.Second
|
|
|
|
o.StartScan(ScanOptions{})
|
|
st := waitScan(t, o)
|
|
if st.State != ScanError || !strings.Contains(st.Error, "timed out") {
|
|
t.Fatalf("expected timeout error, got %+v", st)
|
|
}
|
|
}
|
|
|
|
func TestScanFloatingIPsWrapperHonoursCallerContext(t *testing.T) {
|
|
o, _, mock := newTestOrchestrator(t, 180)
|
|
mock.SeedMany("fip", 20)
|
|
mock.PageSize = 10
|
|
mock.PageDelay = 5 * time.Second
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Millisecond)
|
|
defer cancel()
|
|
if _, _, err := o.ScanFloatingIPs(ctx); !errors.Is(err, context.DeadlineExceeded) {
|
|
t.Fatalf("expected the caller's ctx error, got %v", err)
|
|
}
|
|
// The job itself keeps going until cancelled (cleanup cancels it).
|
|
if !o.ScanStatus().Running {
|
|
t.Fatalf("background job should still be running")
|
|
}
|
|
}
|
|
|
|
func TestSortAddressesAscending(t *testing.T) {
|
|
in := []string{"10.0.0.10", "zzz", "10.0.0.9", "2001:db8::1", "9.255.255.255", "aaa", "10.0.0.2"}
|
|
sortAddressesAscending(in)
|
|
want := []string{"9.255.255.255", "10.0.0.2", "10.0.0.9", "10.0.0.10", "2001:db8::1", "aaa", "zzz"}
|
|
for i := range want {
|
|
if in[i] != want[i] {
|
|
t.Fatalf("got %v want %v", in, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestClearQueueEventPayloadIsTruncated(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, _ := newTestOrchestrator(t, 180)
|
|
var addrs []string
|
|
for i := 0; i < 120; i++ {
|
|
addrs = append(addrs, "10.1.0."+strconv.Itoa(i))
|
|
}
|
|
if err := d.SeedQueue(ctx, addrs); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
res, err := o.ClearQueue(ctx)
|
|
if err != nil || len(res.Deleted) != 120 {
|
|
t.Fatalf("clear: %+v err=%v", res, err)
|
|
}
|
|
var p struct {
|
|
Count int `json:"count"`
|
|
Addresses []string `json:"addresses"`
|
|
Truncated bool `json:"truncated"`
|
|
}
|
|
if err := json.Unmarshal([]byte(eventPayload(t, d, "queue_cleared")), &p); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if p.Count != 120 || len(p.Addresses) != 50 || !p.Truncated || p.Addresses[0] != "10.1.0.0" {
|
|
t.Fatalf("payload: count=%d addrs=%d truncated=%v", p.Count, len(p.Addresses), p.Truncated)
|
|
}
|
|
}
|
|
|
|
func TestDeleteIPsDisassociatesOnlyAttachedFIPs(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-a", "1.1.1.1", "svc")
|
|
mock.Seed("fip-b", "2.2.2.2", "svc")
|
|
if _, err := d.SubmitIPs(ctx, []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
claimed, err := d.ClaimNextQueued(ctx, "validator-1", time.Minute)
|
|
if err != nil || claimed == nil || claimed.IPAddress != "1.1.1.1" {
|
|
t.Fatalf("claim: %+v err=%v", claimed, err)
|
|
}
|
|
if err := mock.AssociateFloatingIP(ctx, "fip-a", "port-1"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := d.SetFIPAssociated(ctx, claimed.ID, "fip-a", time.Minute); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// An unrelated, externally associated FIP must stay as it is.
|
|
if err := mock.AssociateFloatingIP(ctx, "fip-b", "other-port"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
res, err := o.DeleteIPs(ctx, []string{"1.1.1.1", "2.2.2.2", "nope"})
|
|
if err != nil || len(res.Deleted) != 2 || len(res.NotFound) != 1 {
|
|
t.Fatalf("delete: %+v err=%v", res, err)
|
|
}
|
|
a, _ := mock.GetFloatingIPByAddress(ctx, "1.1.1.1")
|
|
b, _ := mock.GetFloatingIPByAddress(ctx, "2.2.2.2")
|
|
if a.PortID != "" {
|
|
t.Fatalf("attached fip must be disassociated, still on %q", a.PortID)
|
|
}
|
|
if b.PortID != "other-port" {
|
|
t.Fatalf("row without recorded fip_id must not trigger a disassociate, got %q", b.PortID)
|
|
}
|
|
v, _ := d.GetValidator(ctx, "validator-1")
|
|
if v.State != db.ValidatorIdle {
|
|
t.Fatalf("validator must be freed, got %s", v.State)
|
|
}
|
|
}
|