Files
cloud-ip-validator/internal/orchestrator/autocycle.go
T
ayurishchevandClaude Sonnet 5.5 cd37b10f3b Add optional automatic check cycle (clear queue -> scan FIPs -> wait -> repeat)
An admin-controlled scenario that repeats what the operator does by hand:
clear the IP queue, scan and enqueue all free Floating IPs, wait until every
queued address reaches a terminal state (so results are in the Registry),
then wait a configurable interval and start over.

- control-api: new auto_cycle singleton table (migration 0008) holding
  enabled/interval/max-run settings and persisted phase state, so the cycle
  survives restarts; engine in orchestrator/autocycle.go driven from the
  existing loop tick with an injectable "now" for deterministic tests.
- Interval (default 1h, min 60s) and max wait (default unlimited, timeout
  outcome) are runtime settings, never hardcoded.
- The periodic fip_scan_interval_seconds scan is skipped while the cycle is
  enabled. An emptied queue mid-cycle counts as finished; stopping during
  the pause keeps the last cycle's outcome.
- API: GET/PUT /api/v1/admin/auto-cycle, POST .../start, POST .../stop.
- admin-dashboard: "Автоматический цикл" panel on /settings and an
  "Автоцикл активен" indicator on /overview.
- Tests for db, orchestrator, httpapi and dashboard; run-local-e2e.sh now
  exercises a full auto cycle; docs updated; bin/ rebuilt with refreshed
  SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-01 10:28:53 +03:00

283 lines
8.8 KiB
Go

package orchestrator
import (
"context"
"encoding/json"
"fmt"
"time"
"cloudipvalidator/internal/db"
)
// This file implements the automatic check cycle: an optional, repeating
// "clear queue -> scan floating IPs -> wait until every queued address has
// reached a terminal state -> wait interval" scenario. Steps 1-2 reuse
// ClearQueue/ScanFloatingIPs verbatim; step 3 needs no code at all because
// Tick already picks up `queued` addresses. All state lives in the database
// (db.AutoCycle), so the cycle survives a control-api restart and the
// interval/limits can be changed at runtime.
// GetAutoCycle returns the current auto-cycle configuration and state.
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
return o.DB.GetAutoCycle(ctx)
}
// StartAutoCycle enables the auto-cycle; the first cycle begins on the next
// AutoCycleStep. It is idempotent: if the cycle is already enabled nothing
// changes (in particular a running cycle is not restarted).
func (o *Orchestrator) StartAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if ac.Enabled {
return nil
}
now := db.Now()
if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil {
return fmt.Errorf("enable auto cycle: %w", err)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "enabled",
"interval_seconds": ac.IntervalSeconds,
"max_run_seconds": ac.MaxRunSeconds,
}))
return nil
}
// StopAutoCycle disables the auto-cycle and returns it to idle. Checks that
// are already in flight are NOT cancelled — they finish normally and land in
// the registry; only the repetition stops.
func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
return fmt.Errorf("get auto cycle: %w", err)
}
if !ac.Enabled {
return nil
}
// "stopped" describes an interrupted run. Stopping during the pause
// between cycles must not overwrite the result of the last finished one.
outcome := ""
if ac.Phase == db.AutoCyclePhaseRunning {
outcome = db.AutoCycleOutcomeStopped
}
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
return fmt.Errorf("disable auto cycle: %w", err)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
"phase": ac.Phase,
}))
return nil
}
// AutoCycleStep advances the auto-cycle state machine by one step. It is
// called by the control-api loop right after Tick.
func (o *Orchestrator) AutoCycleStep(ctx context.Context) {
o.autoCycleStep(ctx, db.Now())
}
// autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive
// the state machine deterministically without sleeping.
func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
o.autoCycleMu.Lock()
defer o.autoCycleMu.Unlock()
// Read inside the lock: Start/Stop may have changed the row since the
// caller last looked.
ac, err := o.DB.GetAutoCycle(ctx)
if err != nil {
o.Log.Error("auto cycle: read state", "err", err)
return
}
if !ac.Enabled {
if ac.Phase != db.AutoCyclePhaseIdle {
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil {
o.Log.Error("auto cycle: reset phase to idle", "err", err)
}
}
return
}
if ac.Phase == db.AutoCyclePhaseRunning {
o.autoCycleCheckRun(ctx, ac, now)
return
}
// idle or waiting: start a new cycle once next_run_at has come.
if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) {
return
}
o.autoCycleStartRun(ctx, ac, now)
}
// autoCycleStartRun performs steps 1-2 of the scenario (clear the queue,
// scan floating IPs) and moves the state to running, or straight to waiting
// if there is nothing to wait for.
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
st := autoCycleStateOf(ac)
st.LastRunStartedAt = &now
st.RunStartedAt = nil
fail := func(step string, err error) {
o.Log.Error("auto cycle: step failed", "step", step, "err", err)
st.Phase = db.AutoCyclePhaseWaiting
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeError
st.LastError = fmt.Sprintf("%s: %v", step, err)
if uerr := o.DB.UpdateAutoCycleState(ctx, st); uerr != nil {
o.Log.Error("auto cycle: save state", "err", uerr)
}
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
"step": step,
"error": err.Error(),
}))
}
if _, err := o.ClearQueue(ctx); err != nil {
fail("clear queue", err)
return
}
_, scanned, err := o.ScanFloatingIPs(ctx)
if err != nil {
fail("scan floating ips", err)
return
}
st.LastScannedFree = scanned
st.LastError = ""
if scanned == 0 {
// Nothing was queued; waiting for completion would never end.
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
st.Phase = db.AutoCyclePhaseWaiting
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
}
return
}
st.Phase = db.AutoCyclePhaseRunning
st.RunStartedAt = &now
st.NextRunAt = nil
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
"reason": "cycle",
"scanned_free": scanned,
}))
}
// autoCycleCheckRun handles the running phase: finish the cycle once every
// queued address is terminal, or give up after max_run_seconds.
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
items, err := o.DB.ListIPs(ctx)
if err != nil {
o.Log.Error("auto cycle: list ips", "err", err)
return
}
interval := time.Duration(ac.IntervalSeconds) * time.Second
next := now.Add(interval)
// An empty queue counts as finished: right after the scan it cannot be
// empty (scanned > 0), so it only happens when an operator cleared or
// deleted every address mid-cycle — and then there is nothing to wait for
// (with max_run_seconds=0 the cycle would otherwise hang forever).
allTerminal := true
for _, it := range items {
if !isTerminalIPState(it.State) {
allTerminal = false
break
}
}
if allTerminal {
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeCompleted
st.LastError = ""
st.RunsTotal++
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
"addresses": len(items),
"runs": st.RunsTotal,
}))
return
}
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
// The queue is left untouched: the next cycle clears it anyway, and
// an operator can still inspect what got stuck.
st := autoCycleStateOf(ac)
st.Phase = db.AutoCyclePhaseWaiting
st.RunStartedAt = nil
st.NextRunAt = &next
st.LastRunFinishedAt = &now
st.LastOutcome = db.AutoCycleOutcomeTimeout
st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds)
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
o.Log.Error("auto cycle: save state", "err", err)
return
}
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
"max_run_seconds": ac.MaxRunSeconds,
"addresses": len(items),
}))
}
}
// isTerminalIPState reports whether an address has finished its check cycle
// for good (its result, if any, is already written to the registry).
// db.IPOccupied counts: such an address never enters the check cycle.
func isTerminalIPState(state string) bool {
switch state {
case db.IPDone, db.IPFailed, db.IPOccupied:
return true
}
return false
}
func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState {
return db.AutoCycleState{
Phase: ac.Phase,
RunStartedAt: ac.RunStartedAt,
NextRunAt: ac.NextRunAt,
LastRunStartedAt: ac.LastRunStartedAt,
LastRunFinishedAt: ac.LastRunFinishedAt,
LastOutcome: ac.LastOutcome,
LastError: ac.LastError,
LastScannedFree: ac.LastScannedFree,
RunsTotal: ac.RunsTotal,
}
}
func autoCyclePayload(m map[string]any) string {
b, err := json.Marshal(m)
if err != nil {
return "{}"
}
return string(b)
}