Add optional automatic check cycle (clear queue -> scan FIPs -> wait -> repeat)

An admin-controlled scenario that repeats what the operator does by hand:
clear the IP queue, scan and enqueue all free Floating IPs, wait until every
queued address reaches a terminal state (so results are in the Registry),
then wait a configurable interval and start over.

- control-api: new auto_cycle singleton table (migration 0008) holding
  enabled/interval/max-run settings and persisted phase state, so the cycle
  survives restarts; engine in orchestrator/autocycle.go driven from the
  existing loop tick with an injectable "now" for deterministic tests.
- Interval (default 1h, min 60s) and max wait (default unlimited, timeout
  outcome) are runtime settings, never hardcoded.
- The periodic fip_scan_interval_seconds scan is skipped while the cycle is
  enabled. An emptied queue mid-cycle counts as finished; stopping during
  the pause keeps the last cycle's outcome.
- API: GET/PUT /api/v1/admin/auto-cycle, POST .../start, POST .../stop.
- admin-dashboard: "Автоматический цикл" panel on /settings and an
  "Автоцикл активен" indicator on /overview.
- Tests for db, orchestrator, httpapi and dashboard; run-local-e2e.sh now
  exercises a full auto cycle; docs updated; bin/ rebuilt with refreshed
  SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-01 10:28:53 +03:00
1 parent 008ae1b0db
commit cd37b10f3b
35 files changed
+2205 -16

No files matched your search

+282
View File
@@ -0,0 +1,282 @@
package orchestrator
import (
"context"
"encoding/json"
"fmt"
"time"
"cloudipvalidator/internal/db"
)
// This file implements the automatic check cycle: an optional, repeating
// "clear queue -> scan floating IPs -> wait until every queued address has
// reached a terminal state -> wait interval" scenario. Steps 1-2 reuse
// ClearQueue/ScanFloatingIPs verbatim; step 3 needs no code at all because
// Tick already picks up `queued` addresses. All state lives in the database
// (db.AutoCycle), so the cycle survives a control-api restart and the
// interval/limits can be changed at runtime.
// GetAutoCycle returns the current auto-cycle configuration and state.
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
return o.DB.GetAutoCycle(ctx)
}
// StartAutoCycle enables the auto-cycle; the first cycle begins on the next
// AutoCycleStep. It is idempotent: if the cycle is already enabled nothing
// changes (in particular a running cycle is not restarted).
func (o *Orchestrator) StartAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if ac.Enabled {
return nil
}
now := db.Now()
if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil {
return fmt.Errorf("enable auto cycle: %w", err)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "enabled",
"interval_seconds": ac.IntervalSeconds,
"max_run_seconds": ac.MaxRunSeconds,
}))
return nil
}
// StopAutoCycle disables the auto-cycle and returns it to idle. Checks that
// are already in flight are NOT cancelled — they finish normally and land in
// the registry; only the repetition stops.
func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if !ac.Enabled {
return nil
}
// "stopped" describes an interrupted run. Stopping during the pause
// between cycles must not overwrite the result of the last finished one.
outcome := ""
if ac.Phase == db.AutoCyclePhaseRunning {
outcome = db.AutoCycleOutcomeStopped
}
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
return fmt.Errorf("disable auto cycle: %w", err)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
"phase": ac.Phase,
}))
return nil
}
// AutoCycleStep advances the auto-cycle state machine by one step. It is
// called by the control-api loop right after Tick.
func (o *Orchestrator) AutoCycleStep(ctx context.Context) {
o.autoCycleStep(ctx, db.Now())
}
// autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive
// the state machine deterministically without sleeping.
func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
// Read inside the lock: Start/Stop may have changed the row since the
// caller last looked.
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
o.Log.Error("auto cycle: read state", "err", err)
return
}
if !ac.Enabled {
if ac.Phase != db.AutoCyclePhaseIdle {
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil {
o.Log.Error("auto cycle: reset phase to idle", "err", err)
}
}
return
}
if ac.Phase == db.AutoCyclePhaseRunning {
o.autoCycleCheckRun(ctx, ac, now)
return
}
// idle or waiting: start a new cycle once next_run_at has come.
if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) {
return
}
o.autoCycleStartRun(ctx, ac, now)
}
// autoCycleStartRun performs steps 1-2 of the scenario (clear the queue,
// scan floating IPs) and moves the state to running, or straight to waiting
// if there is nothing to wait for.
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
st := autoCycleStateOf(ac)
st.LastRunStartedAt = &now
st.RunStartedAt = nil
fail := func(step string, err error) {
o.Log.Error("auto cycle: step failed", "step", step, "err", err)
st.Phase = db.AutoCyclePhaseWaiting
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeError
st.LastError = fmt.Sprintf("%s: %v", step, err)
if uerr := o.DB.UpdateAutoCycleState(ctx, st); uerr != nil {
o.Log.Error("auto cycle: save state", "err", uerr)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
"step": step,
"error": err.Error(),
}))
}
if _, err := o.ClearQueue(ctx); err != nil {
fail("clear queue", err)
return
}
_, scanned, err := o.ScanFloatingIPs(ctx)
if err != nil {
fail("scan floating ips", err)
return
}
st.LastScannedFree = scanned
st.LastError = ""
if scanned == 0 {
// Nothing was queued; waiting for completion would never end.
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
st.Phase = db.AutoCyclePhaseWaiting
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
return
}
st.Phase = db.AutoCyclePhaseRunning
st.RunStartedAt = &now
st.NextRunAt = nil
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "cycle",
"scanned_free": scanned,
}))
}
// autoCycleCheckRun handles the running phase: finish the cycle once every
// queued address is terminal, or give up after max_run_seconds.
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
items, err := o.DB.ListIPs(ctx)
if err != nil {
o.Log.Error("auto cycle: list ips", "err", err)
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
// An empty queue counts as finished: right after the scan it cannot be
// empty (scanned > 0), so it only happens when an operator cleared or
// deleted every address mid-cycle — and then there is nothing to wait for
// (with max_run_seconds=0 the cycle would otherwise hang forever).
allTerminal := true
for _, it := range items {
if !isTerminalIPState(it.State) {
allTerminal = false
break
}
}
if allTerminal {
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeCompleted
st.LastError = ""
st.RunsTotal++
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
"addresses": len(items),
"runs": st.RunsTotal,
}))
return
}
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
// The queue is left untouched: the next cycle clears it anyway, and
// an operator can still inspect what got stuck.
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeTimeout
st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds)
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
"max_run_seconds": ac.MaxRunSeconds,
"addresses": len(items),
}))
}
}
// isTerminalIPState reports whether an address has finished its check cycle
// for good (its result, if any, is already written to the registry).
// db.IPOccupied counts: such an address never enters the check cycle.
func isTerminalIPState(state string) bool {
switch state {
case db.IPDone, db.IPFailed, db.IPOccupied:
return true
}
return false
}
func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState {
return db.AutoCycleState{
Phase: ac.Phase,
RunStartedAt: ac.RunStartedAt,
NextRunAt: ac.NextRunAt,
LastRunStartedAt: ac.LastRunStartedAt,
LastRunFinishedAt: ac.LastRunFinishedAt,
LastOutcome: ac.LastOutcome,
LastError: ac.LastError,
LastScannedFree: ac.LastScannedFree,
RunsTotal: ac.RunsTotal,
}
}
func autoCyclePayload(m map[string]any) string {
b, err := json.Marshal(m)
if err != nil {
return "{}"
}
return string(b)
}
+544
View File
@@ -0,0 +1,544 @@
package orchestrator
import (
"context"
"errors"
"strings"
"testing"
"time"
"cloudipvalidator/internal/db"
)
func getAutoCycle(t *testing.T, d *db.DB) db.AutoCycle {
t.Helper()
ac, err := d.GetAutoCycle(context.Background())
if err != nil {
t.Fatalf("get auto cycle: %v", err)
}
return ac
}
func setAutoCycleParams(t *testing.T, d *db.DB, interval, maxRun int) {
t.Helper()
if err := d.SetAutoCycleParams(context.Background(), &interval, &maxRun); err != nil {
t.Fatalf("set auto cycle params: %v", err)
}
}
// finishAllIPs simulates completed checks by moving every queued address to
// the given terminal state directly (the full check pipeline is covered by
// TestHappyPath).
func finishAllIPs(t *testing.T, d *db.DB, state string) {
t.Helper()
if _, err := d.ExecContext(context.Background(), `UPDATE ip_queue SET state=?`, state); err != nil {
t.Fatalf("finish ips: %v", err)
}
}
func queuedAddresses(t *testing.T, d *db.DB) map[string]string {
t.Helper()
items, err := d.ListIPs(context.Background())
if err != nil {
t.Fatalf("list ips: %v", err)
}
out := make(map[string]string, len(items))
for _, it := range items {
out[it.IPAddress] = it.State
}
return out
}
func countEvents(t *testing.T, d *db.DB, eventType string) int {
t.Helper()
var n int
if err := d.QueryRowContext(context.Background(), `SELECT COUNT(*) FROM events WHERE event_type=?`, eventType).Scan(&n); err != nil {
t.Fatalf("count events: %v", err)
}
return n
}
func TestAutoCycleDisabledIsNoOp(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
t.Fatalf("seed queue: %v", err)
}
o.autoCycleStep(ctx, db.Now())
ac := getAutoCycle(t, d)
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle {
t.Fatalf("expected disabled+idle, got %+v", ac)
}
if got := queuedAddresses(t, d); len(got) != 1 || got["9.9.9.9"] != db.IPQueued {
t.Fatalf("queue must be untouched while disabled, got %v", got)
}
}
func TestAutoCycleDisabledResetsStalePhase(t *testing.T) {
ctx := context.Background()
o, d, _ := newTestOrchestrator(t, 180)
now := db.Now()
if err := d.UpdateAutoCycleState(ctx, db.AutoCycleState{Phase: db.AutoCyclePhaseRunning, RunStartedAt: &now}); err != nil {
t.Fatalf("update state: %v", err)
}
o.autoCycleStep(ctx, now)
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseIdle || ac.RunStartedAt != nil {
t.Fatalf("expected phase reset to idle, got %+v", ac)
}
}
func TestAutoCycleStartClearsQueueAndScans(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
mock.Seed("fip-2", "2.2.2.2", "svc")
mock.SeedWithPort("fip-3", "3.3.3.3", "svc", "someone-elses-port")
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
t.Fatalf("seed queue: %v", err)
}
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
ac := getAutoCycle(t, d)
if !ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.NextRunAt == nil {
t.Fatalf("expected enabled+idle with next_run_at set after start, got %+v", ac)
}
now := db.Now()
o.autoCycleStep(ctx, now)
got := queuedAddresses(t, d)
if _, stale := got["9.9.9.9"]; stale {
t.Fatalf("old queue entry must be cleared, got %v", got)
}
if len(got) != 2 || got["1.1.1.1"] != db.IPQueued || got["2.2.2.2"] != db.IPQueued {
t.Fatalf("expected the two free FIPs queued, got %v", got)
}
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected running, got %s", ac.Phase)
}
if ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(now) {
t.Fatalf("expected run_started_at=%v, got %v", now, ac.RunStartedAt)
}
if ac.LastRunStartedAt == nil || !ac.LastRunStartedAt.Equal(now) {
t.Fatalf("expected last_run_started_at=%v, got %v", now, ac.LastRunStartedAt)
}
if ac.LastScannedFree != 2 {
t.Fatalf("expected last_scanned_free=2, got %d", ac.LastScannedFree)
}
if countEvents(t, d, "queue_cleared") != 1 || countEvents(t, d, "fip_scan") != 1 {
t.Fatalf("expected queue_cleared and fip_scan events")
}
if countEvents(t, d, "auto_cycle_started") != 2 { // enable + cycle start
t.Fatalf("expected two auto_cycle_started events, got %d", countEvents(t, d, "auto_cycle_started"))
}
}
func TestAutoCycleWaitsWhileChecksInProgressAndTickPicksUpQueue(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
mock.Seed("fip-2", "2.2.2.2", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
t.Fatalf("register validator: %v", err)
}
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
// Existing Tick logic starts the checks on its own.
o.Tick(ctx)
states := queuedAddresses(t, d)
inProgress := 0
for _, s := range states {
if s == db.IPAwaitingSelfCheck {
inProgress++
}
}
if inProgress != 1 {
t.Fatalf("expected exactly one address claimed by Tick, got %v", states)
}
o.autoCycleStep(ctx, t0.Add(time.Minute))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning || ac.RunsTotal != 0 {
t.Fatalf("expected still running, got %+v", ac)
}
// One address done, the other still queued: still not finished.
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPDone, "1.1.1.1"); err != nil {
t.Fatalf("update: %v", err)
}
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPQueued, "2.2.2.2"); err != nil {
t.Fatalf("update: %v", err)
}
o.autoCycleStep(ctx, t0.Add(2*time.Minute))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected still running with a queued address left, got %+v", ac)
}
}
func TestAutoCycleCompletesAndRepeatsAfterInterval(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
mock.Seed("fip-2", "2.2.2.2", "svc")
setAutoCycleParams(t, d, 600, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
// done, failed and occupied are all terminal.
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPDone, "1.1.1.1"); err != nil {
t.Fatalf("update: %v", err)
}
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPOccupied, "2.2.2.2"); err != nil {
t.Fatalf("update: %v", err)
}
t1 := t0.Add(5 * time.Minute)
o.autoCycleStep(ctx, t1)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
t.Fatalf("expected waiting/completed, got %+v", ac)
}
if ac.RunsTotal != 1 {
t.Fatalf("expected runs_total=1, got %d", ac.RunsTotal)
}
wantNext := t1.Add(600 * time.Second)
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(wantNext) {
t.Fatalf("expected next_run_at=%v (completion + interval), got %v", wantNext, ac.NextRunAt)
}
if ac.LastRunFinishedAt == nil || !ac.LastRunFinishedAt.Equal(t1) {
t.Fatalf("expected last_run_finished_at=%v, got %v", t1, ac.LastRunFinishedAt)
}
if ac.RunStartedAt != nil {
t.Fatalf("expected run_started_at cleared, got %v", ac.RunStartedAt)
}
if countEvents(t, d, "auto_cycle_completed") != 1 {
t.Fatalf("expected one auto_cycle_completed event")
}
// Before next_run_at: nothing happens, the finished queue is kept.
o.autoCycleStep(ctx, wantNext.Add(-time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting || ac.RunsTotal != 1 {
t.Fatalf("expected still waiting, got %+v", ac)
}
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPDone {
t.Fatalf("queue must not be touched while waiting, got %v", got)
}
// At next_run_at: new cycle starts, queue is rebuilt from scratch.
o.autoCycleStep(ctx, wantNext)
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected running again, got %+v", ac)
}
if ac.LastRunStartedAt == nil || !ac.LastRunStartedAt.Equal(wantNext) {
t.Fatalf("expected last_run_started_at=%v, got %v", wantNext, ac.LastRunStartedAt)
}
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPQueued || got["2.2.2.2"] != db.IPQueued {
t.Fatalf("expected both addresses re-queued, got %v", got)
}
if ac.RunsTotal != 1 {
t.Fatalf("runs_total only counts completed cycles, got %d", ac.RunsTotal)
}
}
func TestAutoCycleTimeout(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
setAutoCycleParams(t, d, 60, 300)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
o.autoCycleStep(ctx, t0.Add(300*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected still running exactly at the limit, got %+v", ac)
}
t1 := t0.Add(301 * time.Second)
o.autoCycleStep(ctx, t1)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeTimeout {
t.Fatalf("expected waiting/timeout, got %+v", ac)
}
if ac.RunsTotal != 0 {
t.Fatalf("timeout must not count as completed, got runs_total=%d", ac.RunsTotal)
}
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t1.Add(60*time.Second)) {
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
}
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPQueued {
t.Fatalf("queue must be left untouched on timeout, got %v", got)
}
if countEvents(t, d, "auto_cycle_timeout") != 1 {
t.Fatalf("expected one auto_cycle_timeout event")
}
}
func TestAutoCycleNoLimitNeverTimesOut(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
o.autoCycleStep(ctx, t0.Add(1000*time.Hour))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("max_run_seconds=0 means no limit, got %+v", ac)
}
}
// An operator pressing «Очистить всё» in the middle of a cycle leaves nothing
// to wait for; with max_run_seconds=0 the cycle would otherwise hang forever.
func TestAutoCycleManualClearMidCycleCompletes(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected running after start, got %+v", ac)
}
if _, err := o.ClearQueue(ctx); err != nil {
t.Fatalf("manual clear: %v", err)
}
t1 := t0.Add(5 * time.Second)
o.autoCycleStep(ctx, t1)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
t.Fatalf("expected waiting/completed after the queue was emptied, got %+v", ac)
}
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t1.Add(60*time.Second)) {
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
}
}
func TestAutoCycleNoFreeIPs(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.SeedWithPort("fip-1", "1.1.1.1", "svc", "someone-elses-port")
setAutoCycleParams(t, d, 120, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
now := db.Now()
o.autoCycleStep(ctx, now)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeNoFreeIPs {
t.Fatalf("expected waiting/no_free_ips, got %+v", ac)
}
if ac.LastScannedFree != 0 {
t.Fatalf("expected last_scanned_free=0, got %d", ac.LastScannedFree)
}
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(now.Add(120*time.Second)) {
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
}
if ac.RunStartedAt != nil {
t.Fatalf("run_started_at must stay unset, got %v", ac.RunStartedAt)
}
}
func TestAutoCycleOpenStackErrorRetriesNextInterval(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
mock.ListFailure = errors.New("neutron is down")
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeError {
t.Fatalf("expected waiting/error, got %+v", ac)
}
if !strings.Contains(ac.LastError, "neutron is down") {
t.Fatalf("expected last_error to mention the cause, got %q", ac.LastError)
}
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t0.Add(60*time.Second)) {
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
}
if countEvents(t, d, "auto_cycle_error") != 1 {
t.Fatalf("expected one auto_cycle_error event")
}
if !ac.Enabled {
t.Fatalf("an error must not disable the auto-cycle")
}
// OpenStack recovers: the next interval starts a normal cycle and the
// stale error is cleared.
mock.ListFailure = nil
o.autoCycleStep(ctx, t0.Add(60*time.Second))
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastError != "" {
t.Fatalf("expected running with cleared error, got %+v", ac)
}
}
func TestAutoCycleStartIsIdempotent(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("second start: %v", err)
}
ac := getAutoCycle(t, d)
if !ac.Enabled || ac.Phase != db.AutoCyclePhaseRunning || ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t0) {
t.Fatalf("second start must not restart a running cycle, got %+v", ac)
}
if countEvents(t, d, "auto_cycle_started") != 2 { // enable + cycle start, not a third
t.Fatalf("expected no extra started event, got %d", countEvents(t, d, "auto_cycle_started"))
}
}
func TestAutoCycleStopMidCycle(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
t.Fatalf("register validator: %v", err)
}
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
o.Tick(ctx) // check in flight
if err := o.StopAutoCycle(ctx); err != nil {
t.Fatalf("stop: %v", err)
}
ac := getAutoCycle(t, d)
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.LastOutcome != db.AutoCycleOutcomeStopped {
t.Fatalf("expected disabled/idle/stopped, got %+v", ac)
}
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPAwaitingSelfCheck {
t.Fatalf("in-flight check must not be cancelled, got %v", got)
}
if countEvents(t, d, "auto_cycle_stopped") != 1 {
t.Fatalf("expected one auto_cycle_stopped event")
}
// Further steps do nothing, even far in the future.
o.autoCycleStep(ctx, t0.Add(100*time.Hour))
ac = getAutoCycle(t, d)
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.RunsTotal != 0 {
t.Fatalf("expected no activity after stop, got %+v", ac)
}
// Stopping again is a harmless no-op.
if err := o.StopAutoCycle(ctx); err != nil {
t.Fatalf("second stop: %v", err)
}
if countEvents(t, d, "auto_cycle_stopped") != 1 {
t.Fatalf("second stop must not emit another event")
}
}
// Stopping during the pause between cycles must keep the result of the last
// finished cycle visible instead of replacing it with "stopped".
func TestAutoCycleStopWhileWaitingKeepsLastOutcome(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
setAutoCycleParams(t, d, 60, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
finishAllIPs(t, d, db.IPDone)
o.autoCycleStep(ctx, t0.Add(5*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
t.Fatalf("precondition: expected waiting/completed, got %+v", ac)
}
if err := o.StopAutoCycle(ctx); err != nil {
t.Fatalf("stop: %v", err)
}
ac := getAutoCycle(t, d)
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle {
t.Fatalf("expected disabled/idle, got %+v", ac)
}
if ac.LastOutcome != db.AutoCycleOutcomeCompleted || ac.RunsTotal != 1 {
t.Fatalf("stop in the pause must keep last_outcome=completed, got %+v", ac)
}
}
func TestAutoCycleSurvivesRestart(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.1.1.1", "svc")
setAutoCycleParams(t, d, 300, 0)
if err := o.StartAutoCycle(ctx); err != nil {
t.Fatalf("start: %v", err)
}
t0 := db.Now()
o.autoCycleStep(ctx, t0)
// A fresh Orchestrator on the same database (a control-api restart)
// continues the running phase instead of starting over.
o2 := &Orchestrator{DB: d, OS: mock, Cfg: o.Cfg, Agg: o.Agg, Log: o.Log}
o2.autoCycleStep(ctx, t0.Add(time.Minute))
ac := getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseRunning || ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t0) {
t.Fatalf("expected the running phase to continue, got %+v", ac)
}
finishAllIPs(t, d, db.IPDone)
t1 := t0.Add(2 * time.Minute)
o2.autoCycleStep(ctx, t1)
ac = getAutoCycle(t, d)
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
t.Fatalf("expected waiting/completed, got %+v", ac)
}
// A third instance still honours the persisted next_run_at.
o3 := &Orchestrator{DB: d, OS: mock, Cfg: o.Cfg, Agg: o.Agg, Log: o.Log}
o3.autoCycleStep(ctx, t1.Add(299*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting {
t.Fatalf("expected waiting until next_run_at, got %+v", ac)
}
o3.autoCycleStep(ctx, t1.Add(300*time.Second))
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
t.Fatalf("expected a new cycle at next_run_at, got %+v", ac)
}
}
+5
View File
@@ -14,6 +14,7 @@ import (
"errors"
"fmt"
"log/slog"
"sync"
"time"
"cloudipvalidator/internal/config"
@@ -36,6 +37,10 @@ type Orchestrator struct {
Cfg config.OrchestratorConfig
Agg config.AggregationConfig
Log *slog.Logger
// autoCycleMu serializes AutoCycleStep with StartAutoCycle/StopAutoCycle
// so an API call can never interleave with a half-finished step.
autoCycleMu sync.Mutex
}
// New constructs an Orchestrator. Egress check types/targets, prober sites,