Files
ayurishchevandClaude Sonnet 5.5 aff8fe38b5 Scan floating IPs in the background, page by page, so thousands of addresses work
The "Scan Floating IP" button failed with a client timeout: the project now
holds ~6.4k floating IPs and the scan listed them all in one unpaginated,
timeout-less Neutron request on the HTTP request context.

openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them
small) with per-page retry/backoff on transport errors, 5xx and 429, and every
request now has a timeout (also ends hangs inside the orchestrator tick).

orchestrator: the scan is a single-flight background job on the process
context with progress (clearing/listing/enqueuing/done/error), dry_run, full
discovery before anything is enqueued, then SubmitIPs in chunks of 500 in
ascending IP order; a failed read leaves the queue untouched. The auto-cycle
gets a "scanning" phase that polls the job, so the control loop and
autoCycleMu are never held across OpenStack/DB work; it recovers after a
restart and waits for (instead of adopting) a scan started by someone else.

db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY
counters, EXISTS completion check, set-based ClearAllIPs.

API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging
and filters on /admin/ips and /admin/registry (bare arrays without limit),
results_by_overall in /admin/status.

dashboard: scan progress panel and dry-run button, paginated /ips and
/registry with server-side filters, Overview on counters and capped lists
with progress/ETA, "select all N by filter", hx-params fix for per-row
buttons, real counts in confirmations.

Also: docs (API, USAGE, DASHBOARD, README), plan and review under
docs/changes/, bin/ rebuilt with new SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-01 19:31:11 +03:00

304 lines
8.2 KiB
Go

package db
import "time"
// Validator and IP lifecycle states. Kept as typed string constants rather
// than a Go enum type so they round-trip through SQLite TEXT columns and
// JSON without conversion.
const (
ValidatorUnregistered = "unregistered"
ValidatorIdle = "idle"
ValidatorAssigned = "assigned"
ValidatorChecking = "checking"
ValidatorUnreachable = "unreachable"
// Site (prober) availability states — no assigned/checking equivalent,
// since a prober isn't bound to one IP at a time.
SiteUnregistered = "unregistered"
SiteIdle = "idle"
SiteUnreachable = "unreachable"
IPQueued = "queued"
IPAssigningFIP = "assigning_fip"
IPAwaitingSelfCheck = "awaiting_self_check"
IPChecking = "checking"
IPAggregating = "aggregating"
IPDone = "done"
IPFailed = "failed"
// IPOccupied is a terminal state distinct from IPFailed: the floating IP
// was found already associated to a different port at claim time, so the
// check cycle never started for this attempt. See
// Orchestrator.associateFIP and db.MarkFIPOccupied.
IPOccupied = "occupied"
ResultPass = "pass"
ResultPartial = "partial"
ResultFail = "fail"
ResultCancelled = "cancelled"
SourceEgress = "egress"
)
// IPStates lists every valid ip_queue state, in lifecycle order.
var IPStates = []string{
IPQueued, IPAssigningFIP, IPAwaitingSelfCheck, IPChecking, IPAggregating,
IPDone, IPFailed, IPOccupied,
}
// IsValidIPState reports whether s is one of IPStates.
func IsValidIPState(s string) bool {
for _, v := range IPStates {
if v == s {
return true
}
}
return false
}
// IsValidResult reports whether s is a valid overall result value
// (pass | partial | fail | cancelled).
func IsValidResult(s string) bool {
switch s {
case ResultPass, ResultPartial, ResultFail, ResultCancelled:
return true
}
return false
}
// InboundSource returns the checks.source value for the given prober site
// index (1-based), e.g. InboundSource(1) == "inbound-site-1".
func InboundSource(siteIndex int) string {
return "inbound-site-" + itoa(siteIndex)
}
func itoa(n int) string {
if n == 0 {
return "0"
}
neg := n < 0
if neg {
n = -n
}
var buf [20]byte
i := len(buf)
for n > 0 {
i--
buf[i] = byte('0' + n%10)
n /= 10
}
if neg {
i--
buf[i] = '-'
}
return string(buf[i:])
}
type Validator struct {
ValidatorID string
Hostname string
OSPortID string
State string
CurrentIPID *int64
AgentVersion string
LastHeartbeatAt *time.Time
CreatedAt time.Time
UpdatedAt time.Time
}
type IPQueueItem struct {
ID int64
IPAddress string
Sequence int
State string
OwnerValidatorID *string
FIPID string
AttemptNumber int
RetryCount int
LeaseExpiresAt *time.Time
EgressComplete bool
OverallResult string
AssignedAt *time.Time
FIPAssociatedAt *time.Time
AggregatedAt *time.Time
FIPReleasedAt *time.Time
RegistryID int64
CycleID int
CreatedAt time.Time
UpdatedAt time.Time
}
type Check struct {
ID int64
RegistryID int64
CycleID int
// IPID is the ip_queue row this check was originally recorded against.
// It's cleared to 0 (SQL NULL) if that row was later deleted — history
// stays reachable via RegistryID/CycleID regardless (see
// migrations/0007_ip_registry.sql). 0 is never a valid ip_queue id
// (AUTOINCREMENT starts at 1), so it unambiguously means "orphaned."
IPID int64
IPAddress string
AttemptNumber int
ValidatorID string
Source string
CheckType string
Target string
Success bool
LatencyMS int64
Detail string
CheckedAt time.Time
CreatedAt time.Time
}
// RegistryItem is a durable per-address record that survives an address
// being removed from ip_queue and later re-added — see
// migrations/0007_ip_registry.sql. NextCycle is the cycle_id that will be
// assigned the next time this address is (re)submitted; it only ever
// increases, so cycle_id stays unique for this address even across
// ip_queue row deletion/recreation.
type RegistryItem struct {
ID int64
IPAddress string
FirstSeenAt time.Time
LastSeenAt time.Time
NextCycle int
CreatedAt time.Time
UpdatedAt time.Time
}
type Event struct {
ID int64
SourceType string
SourceID string
IPID *int64
// RegistryID/CycleID are resolved from IPID at insert time (see
// InsertEvent) and stay set even after the ip_queue row IPID pointed to
// is later deleted, so retention pruning can cut events at the same
// cycle boundary as checks — see migrations/0007_ip_registry.sql.
RegistryID int64
CycleID int
EventType string
Payload string
OccurredAt time.Time
CreatedAt time.Time
}
type Site struct {
Index int
SiteID string
Hostname string
State string
LastHeartbeatAt *time.Time
CreatedAt time.Time
UpdatedAt time.Time
}
type TargetGroup struct {
Name string
Targets []string
CreatedAt time.Time
UpdatedAt time.Time
}
type CheckType struct {
Name string
Enabled bool
TargetGroups []string
CreatedAt time.Time
UpdatedAt time.Time
}
// ResolvedCheckType is a check type with its target groups already expanded
// into a flat target list — what AssignmentForValidator hands to an agent.
type ResolvedCheckType struct {
Type string
Targets []string
}
// SubmitIPsResult categorizes how each address in a SubmitIPs call was
// handled.
type SubmitIPsResult struct {
Added []string
Requeued []string
Reordered []string
SkippedInProgress []string
}
// DeleteIPsResult categorizes how each address in a DeleteIPs call (or a
// ClearQueue call, which is DeleteIPs given every currently queued address)
// was handled.
type DeleteIPsResult struct {
Deleted []string
NotFound []string
}
// Settings is the singleton row of orchestrator-wide scalar knobs that are
// admin-configurable at runtime (see queries_settings.go).
type Settings struct {
FIPSettleSeconds int
// HistoryRetentionCycles caps how many recent check cycles are kept per
// registry address (see PruneRegistryHistory); 0 means unlimited.
HistoryRetentionCycles int
CreatedAt time.Time
UpdatedAt time.Time
}
// InboundChecksSettings is the singleton row describing what the prober
// checks on every site for every in-flight IP (TCP ports + optional ICMP).
// Admin-configurable at runtime (see queries_inbound.go).
type InboundChecksSettings struct {
Ports []int
ICMP bool
CreatedAt time.Time
UpdatedAt time.Time
}
// Auto-cycle phases and outcomes (see AutoCycle).
const (
AutoCyclePhaseIdle = "idle"
AutoCyclePhaseRunning = "running"
AutoCyclePhaseWaiting = "waiting"
// AutoCyclePhaseScanning: the queue is being cleared and the floating IPs
// are being (re)discovered and enqueued by the background scan job.
AutoCyclePhaseScanning = "scanning"
AutoCycleOutcomeCompleted = "completed"
AutoCycleOutcomeNoFreeIPs = "no_free_ips"
AutoCycleOutcomeTimeout = "timeout"
AutoCycleOutcomeError = "error"
AutoCycleOutcomeStopped = "stopped"
)
// AutoCycle is the singleton row describing the automatic check cycle:
// its configuration (Enabled, IntervalSeconds, MaxRunSeconds) and its
// persisted runtime state (Phase and timestamps). MaxRunSeconds == 0 means
// no limit on how long a run may wait for checks to finish.
type AutoCycle struct {
Enabled bool
IntervalSeconds int
MaxRunSeconds int
Phase string
RunStartedAt *time.Time
NextRunAt *time.Time
LastRunStartedAt *time.Time
LastRunFinishedAt *time.Time
LastOutcome string
LastError string
LastScannedFree int
RunsTotal int
}
// AutoCycleState is a full replacement of the runtime-state columns of the
// auto_cycle row (everything except configuration and enabled flag).
type AutoCycleState struct {
Phase string
RunStartedAt *time.Time
NextRunAt *time.Time
LastRunStartedAt *time.Time
LastRunFinishedAt *time.Time
LastOutcome string
LastError string
LastScannedFree int
RunsTotal int
}