Keep one address per validator; fix heartbeat handling and queue clear

A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42
addresses to fail without a single check. A validator busy with slow
checks went silent, was marked unreachable, and its next heartbeat put it
back to idle while it still held the address; it was handed a second one,
whose association never ran (the in-flight guard was keyed by validator),
and both waited for their leases to expire.

- Heartbeat/re-register return an unreachable validator to assigned when
  it still holds an address, else idle.
- A validator is released only from the address it currently holds
  (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an
  unreachable validator stays unreachable until its next heartbeat, so a
  dead validator is no longer handed a new address every lease period.
- ClaimNextQueued refuses a validator that still has an address; a
  ReconcileValidators pass on every tick repairs rows that disagree with
  the queue.
- Association guard is keyed by address, not validator.
- The agent sends heartbeats from their own goroutine.
- Clear queue / delete: detach only floating IPs of unfinished rows (done,
  failed and occupied rows kept their fip_id and made a clear issue >1000
  sequential cloud calls: 256 s), at most 8 in parallel; the operation no
  longer dies with the client connection (10 minute limit).

Includes the incident analysis and the plan under analysis/ and
docs/changes/, and rebuilt bin/control-api and bin/validator-agent.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-02 14:42:12 +03:00
1 parent cf4a883363
commit 0532baff09
17 files changed
+1293 -62

No files matched your search

+40 -4
View File
@@ -19,6 +19,7 @@ import (
neturl "net/url"
"os"
"strings"
"sync/atomic"
"time"
"cloudipvalidator/internal/apiclient"
@@ -33,6 +34,10 @@ type Agent struct {
lastHandledIPID int64
// busy is true while an assignment is being worked on; it is reported in
// the heartbeat body (informational on the control-api side).
busy atomic.Bool
// registerRetryInitial/Max govern the backoff used while waiting for a
// successful registration (see registerWithRetry): control-api may not
// be up yet at agent boot, or may come and go across a redeploy, and the
@@ -70,6 +75,15 @@ func (a *Agent) Run(ctx context.Context) error {
}
interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second
// Heartbeats run on their own schedule. Sent from the poll loop they
// stopped for as long as a slow assignment took (an address whose
// outbound targets all time out keeps the loop busy for ~40 s), which
// control-api reads as a lost validator after heartbeat_timeout_seconds.
hbCtx, stopHeartbeat := context.WithCancel(ctx)
defer stopHeartbeat()
go a.heartbeatLoop(hbCtx, interval)
ticker := time.NewTicker(interval)
defer ticker.Stop()
@@ -148,12 +162,32 @@ type checkConfigDTO struct {
Targets []string `json:"targets"`
}
func (a *Agent) pollOnce(ctx context.Context) {
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: "idle"}, nil); err != nil {
a.log.Error("heartbeat", "err", err)
return
// heartbeatLoop sends a heartbeat now and then every interval until ctx is
// cancelled.
func (a *Agent) heartbeatLoop(ctx context.Context, interval time.Duration) {
ticker := time.NewTicker(interval)
defer ticker.Stop()
for {
a.sendHeartbeat(ctx)
select {
case <-ctx.Done():
return
case <-ticker.C:
}
}
}
func (a *Agent) sendHeartbeat(ctx context.Context) {
state := "idle"
if a.busy.Load() {
state = "checking"
}
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: state}, nil); err != nil && ctx.Err() == nil {
a.log.Error("heartbeat", "err", err)
}
}
func (a *Agent) pollOnce(ctx context.Context) {
var assignment assignmentResp
ok, err := a.client.Do(ctx, "GET", "/api/v1/agents/"+a.cfg.ValidatorID+"/assignment", nil, &assignment)
if err != nil {
@@ -169,6 +203,8 @@ func (a *Agent) pollOnce(ctx context.Context) {
return // already handled this IP's work this attempt
}
a.busy.Store(true)
defer a.busy.Store(false)
switch assignment.Phase {
case "awaiting_self_check":
a.handleSelfCheckAndRun(ctx, assignment)
+72
View File
@@ -0,0 +1,72 @@
package agentcore
import (
"context"
"fmt"
"net/http"
"net/http/httptest"
"sync/atomic"
"testing"
"time"
"cloudipvalidator/internal/config"
)
// While the agent is busy with a slow assignment (an address whose outbound
// targets time out keeps it occupied for tens of seconds) it must keep sending
// heartbeats; control-api marks a validator that stays silent for
// heartbeat_timeout_seconds as unreachable.
func TestHeartbeatContinuesDuringSlowChecks(t *testing.T) {
slowTarget := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
time.Sleep(3500 * time.Millisecond)
}))
defer slowTarget.Close()
var heartbeats, assignments int32
mux := http.NewServeMux()
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
fmt.Fprint(w, `{"ok":true}`)
})
mux.HandleFunc("POST /api/v1/agents/val-1/heartbeat", func(w http.ResponseWriter, r *http.Request) {
atomic.AddInt32(&heartbeats, 1)
fmt.Fprint(w, `{"ok":true}`)
})
mux.HandleFunc("GET /api/v1/agents/val-1/assignment", func(w http.ResponseWriter, r *http.Request) {
if atomic.AddInt32(&assignments, 1) > 1 {
w.WriteHeader(http.StatusNoContent)
return
}
fmt.Fprintf(w, `{"ip_id":1,"ip_address":"1.1.1.1","phase":"checking","check_config":[{"type":"https","targets":[%q]}]}`, slowTarget.URL)
})
mux.HandleFunc("POST /api/v1/agents/val-1/results", func(w http.ResponseWriter, r *http.Request) { fmt.Fprint(w, `{"ok":true}`) })
mux.HandleFunc("POST /api/v1/agents/val-1/complete", func(w http.ResponseWriter, r *http.Request) { fmt.Fprint(w, `{"ok":true}`) })
capi := httptest.NewServer(mux)
defer capi.Close()
a := New(&config.ValidatorAgent{
ValidatorID: "val-1", ControlAPIURL: capi.URL, PollIntervalSeconds: 1,
Checks: config.AgentChecks{HTTPSTimeoutSeconds: 10, ICMPTimeoutSeconds: 1, ICMPCount: 1},
}, testLogger())
ctx, cancel := context.WithCancel(context.Background())
done := make(chan struct{})
go func() { _ = a.Run(ctx); close(done) }()
time.Sleep(3 * time.Second) // the slow check (3.5 s) is still running
during := atomic.LoadInt32(&heartbeats)
cancel()
select {
case <-done:
case <-time.After(10 * time.Second):
t.Fatal("Run did not stop after the context was cancelled")
}
// One per second plus the first: 3-4 in 3 s. With heartbeats in the poll
// loop there is exactly one, sent before the slow assignment started.
if during < 3 {
t.Fatalf("%d heartbeats in 3 s while a check was running, want at least 3", during)
}
if atomic.LoadInt32(&assignments) < 1 {
t.Fatal("the assignment was never fetched, the test did not exercise a busy agent")
}
}
+27 -28
View File
@@ -101,7 +101,7 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
res, err = tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=?, updated_at=?
WHERE validator_id=? AND state=?
WHERE validator_id=? AND state=? AND current_ip_id IS NULL
`, ValidatorAssigned, item.ID, timeToDB(now), validatorID, ValidatorIdle)
if err != nil {
return nil, err
@@ -207,7 +207,8 @@ func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error {
}
// ReleaseFIP records that the floating IP has been disassociated and frees
// the owning validator back to idle, in one transaction.
// the owning validator (if this address is still its current one, see
// freeValidatorSQL), in one transaction.
func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) error {
tx, err := d.BeginTx(ctx, nil)
if err != nil {
@@ -219,10 +220,7 @@ func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) err
if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET fip_released_at=?, updated_at=? WHERE id=?`, now, now, ipID); err != nil {
return err
}
if _, err := tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
WHERE validator_id=?
`, ValidatorIdle, now, validatorID); err != nil {
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
return err
}
return tx.Commit()
@@ -252,10 +250,7 @@ func (d *DB) MarkFIPOccupied(ctx context.Context, ipID int64, validatorID string
return err
}
if validatorID != "" {
if _, err := tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
WHERE validator_id=?
`, ValidatorIdle, now, validatorID); err != nil {
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
return err
}
}
@@ -315,10 +310,7 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
}
if validatorID != "" {
if _, err := tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
WHERE validator_id=?
`, ValidatorIdle, now, validatorID); err != nil {
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
return err
}
}
@@ -520,9 +512,11 @@ func (d *DB) DeleteIPs(ctx context.Context, addresses []string) (DeleteIPsResult
func deleteIPTx(ctx context.Context, tx *sql.Tx, ipID int64) error {
now := timeToDB(Now())
if _, err := tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
UPDATE validators SET current_ip_id=NULL,
state = CASE WHEN state=? THEN state ELSE ? END,
updated_at=?
WHERE current_ip_id=?
`, ValidatorIdle, now, ipID); err != nil {
`, ValidatorUnreachable, ValidatorIdle, now, ipID); err != nil {
return fmt.Errorf("free owning validator: %w", err)
}
if _, err := tx.ExecContext(ctx, `UPDATE checks SET ip_id=NULL WHERE ip_id=?`, ipID); err != nil {
@@ -579,9 +573,11 @@ func (d *DB) ClearAllIPs(ctx context.Context) ([]string, error) {
now := timeToDB(Now())
if _, err := tx.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
UPDATE validators SET current_ip_id=NULL,
state = CASE WHEN state=? THEN state ELSE ? END,
updated_at=?
WHERE current_ip_id IS NOT NULL
`, ValidatorIdle, now); err != nil {
`, ValidatorUnreachable, ValidatorIdle, now); err != nil {
return nil, fmt.Errorf("free owning validators: %w", err)
}
if _, err := tx.ExecContext(ctx, `UPDATE checks SET ip_id=NULL WHERE ip_id IS NOT NULL`); err != nil {
@@ -610,11 +606,15 @@ type FIPRef struct {
FIPID string
}
// ListFIPRefs returns every queue row with an attached floating IP (fip_id
// set) — typically at most one per validator — so a bulk clear can
// disassociate them without loading the whole queue.
// ListFIPRefs returns the queue rows that may still hold a floating IP: an
// fip_id is set and the row is not finished. A finished row (done, failed,
// occupied) keeps its fip_id for display, but its floating IP was already
// disassociated before the final state was written, so listing it would only
// make a bulk clear issue thousands of pointless cloud calls. At most one row
// per validator qualifies, so a clear does not need to load the whole queue.
func (d *DB) ListFIPRefs(ctx context.Context) ([]FIPRef, error) {
rows, err := d.QueryContext(ctx, `SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' ORDER BY id`)
rows, err := d.QueryContext(ctx, `SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND state NOT IN (?, ?, ?) ORDER BY id`,
IPDone, IPFailed, IPOccupied)
if err != nil {
return nil, err
}
@@ -631,8 +631,7 @@ func (d *DB) ListFIPRefs(ctx context.Context) ([]FIPRef, error) {
}
// ListFIPRefsByAddresses is ListFIPRefs restricted to the given addresses
// (unknown addresses and rows without an attached floating IP are simply
// absent), using a handful of IN (...) queries instead of one lookup per
// (unknown addresses and rows that hold no floating IP are simply absent), using a handful of IN (...) queries instead of one lookup per
// address.
func (d *DB) ListFIPRefsByAddresses(ctx context.Context, addresses []string) ([]FIPRef, error) {
const chunk = 500
@@ -643,12 +642,12 @@ func (d *DB) ListFIPRefsByAddresses(ctx context.Context, addresses []string) ([]
end = len(addresses)
}
part := addresses[start:end]
args := make([]any, len(part))
for i, a := range part {
args[i] = a
args := []any{IPDone, IPFailed, IPOccupied}
for _, a := range part {
args = append(args, a)
}
rows, err := d.QueryContext(ctx,
`SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND ip_address IN (`+placeholders(len(part))+`)`, args...)
`SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND state NOT IN (?, ?, ?) AND ip_address IN (`+placeholders(len(part))+`)`, args...)
if err != nil {
return nil, err
}
+186
View File
@@ -0,0 +1,186 @@
package db
import (
"context"
"fmt"
"testing"
"time"
)
func validatorState(t *testing.T, d *DB, id string) *Validator {
t.Helper()
v, err := d.GetValidator(testCtx(t), id)
if err != nil {
t.Fatalf("get validator %s: %v", id, err)
}
return v
}
func testCtx(t *testing.T) context.Context {
t.Helper()
return context.Background()
}
// claimFor2 seeds an address and returns its id without claiming it.
func claimFor2(t *testing.T, d *DB, addr string) int64 {
t.Helper()
ctx := testCtx(t)
if err := d.SeedQueue(ctx, []string{addr}); err != nil {
t.Fatalf("seed %s: %v", addr, err)
}
ip, err := d.GetIPByAddress(ctx, addr)
if err != nil {
t.Fatalf("get %s: %v", addr, err)
}
return ip.ID
}
// claimFor seeds an address and claims it for the validator.
func claimFor(t *testing.T, d *DB, addr, validatorID string) *IPQueueItem {
t.Helper()
ctx := testCtx(t)
if err := d.SeedQueue(ctx, []string{addr}); err != nil {
t.Fatalf("seed %s: %v", addr, err)
}
item, err := d.ClaimNextQueued(ctx, validatorID, time.Minute)
if err != nil || item == nil {
t.Fatalf("claim %s for %s: item=%v err=%v", addr, validatorID, item, err)
}
return item
}
func TestHeartbeatKeepsAssignedWhenValidatorStillHoldsAnAddress(t *testing.T) {
d, ctx := newTestDB(t)
_ = d.AdminCreateValidator(ctx, "v1", "p1")
_ = d.AdminCreateValidator(ctx, "v2", "p2")
item := claimFor(t, d, "1.1.1.1", "v1")
_ = d.MarkValidatorUnreachable(ctx, "v1")
_ = d.MarkValidatorUnreachable(ctx, "v2")
if err := d.Heartbeat(ctx, "v1"); err != nil {
t.Fatal(err)
}
if v := validatorState(t, d, "v1"); v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != item.ID {
t.Fatalf("v1 after heartbeat: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, item.ID)
}
if err := d.Heartbeat(ctx, "v2"); err != nil {
t.Fatal(err)
}
if v := validatorState(t, d, "v2"); v.State != ValidatorIdle {
t.Fatalf("v2 after heartbeat: state=%s, want idle", v.State)
}
}
func TestRegisterValidatorReactivationKeepsAssignedAddress(t *testing.T) {
d, ctx := newTestDB(t)
_ = d.AdminCreateValidator(ctx, "v1", "p1")
item := claimFor(t, d, "1.1.1.1", "v1")
_ = d.MarkValidatorUnreachable(ctx, "v1")
if err := d.RegisterValidator(ctx, "v1", "host", "p1", "v"); err != nil {
t.Fatal(err)
}
if v := validatorState(t, d, "v1"); v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != item.ID {
t.Fatalf("after re-register: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, item.ID)
}
}
func TestStaleReleasesLeaveTheCurrentAddressAlone(t *testing.T) {
cases := map[string]func(d *DB, ipID int64) error{
"ReleaseFIP": func(d *DB, id int64) error { return d.ReleaseFIP(context.Background(), id, "v1") },
"RequeueOrFail": func(d *DB, id int64) error { return d.RequeueOrFail(context.Background(), id, "v1", 3) },
"MarkFIPOccupied": func(d *DB, id int64) error { return d.MarkFIPOccupied(context.Background(), id, "v1") },
"FreeValidator": func(d *DB, id int64) error { return d.FreeValidator(context.Background(), "v1", id) },
}
for name, release := range cases {
t.Run(name, func(t *testing.T) {
d, ctx := newTestDB(t)
_ = d.AdminCreateValidator(ctx, "v1", "p1")
stale := claimFor(t, d, "1.1.1.1", "v1")
// v1 has moved on to another address (set directly: the claim path
// itself refuses a validator that is still busy).
cur := claimFor2(t, d, "2.2.2.2")
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state='checking', owner_validator_id='v1' WHERE id=?`, cur); err != nil {
t.Fatal(err)
}
if _, err := d.ExecContext(ctx, `UPDATE validators SET state='assigned', current_ip_id=? WHERE validator_id='v1'`, cur); err != nil {
t.Fatal(err)
}
if err := release(d, stale.ID); err != nil {
t.Fatalf("%s: %v", name, err)
}
v := validatorState(t, d, "v1")
if v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != cur {
t.Fatalf("%s freed a validator that holds another address: state=%s current_ip=%v", name, v.State, v.CurrentIPID)
}
})
}
}
// The right address frees the validator, but an unreachable one stays so.
func TestReleaseKeepsUnreachableValidatorUnreachable(t *testing.T) {
d, ctx := newTestDB(t)
_ = d.AdminCreateValidator(ctx, "v1", "p1")
_ = d.AdminCreateValidator(ctx, "v2", "p2")
a := claimFor(t, d, "1.1.1.1", "v1")
b := claimFor(t, d, "2.2.2.2", "v2")
_ = d.MarkValidatorUnreachable(ctx, "v1")
if err := d.ReleaseFIP(ctx, a.ID, "v1"); err != nil {
t.Fatal(err)
}
if v := validatorState(t, d, "v1"); v.State != ValidatorUnreachable || v.CurrentIPID != nil {
t.Fatalf("v1: state=%s current_ip=%v, want unreachable and empty", v.State, v.CurrentIPID)
}
if err := d.RequeueOrFail(ctx, b.ID, "v2", 3); err != nil {
t.Fatal(err)
}
if v := validatorState(t, d, "v2"); v.State != ValidatorIdle || v.CurrentIPID != nil {
t.Fatalf("v2: state=%s current_ip=%v, want idle and empty", v.State, v.CurrentIPID)
}
}
func TestClaimRefusesValidatorThatStillHoldsAnAddress(t *testing.T) {
d, ctx := newTestDB(t)
_ = d.AdminCreateValidator(ctx, "v1", "p1")
first := claimFor(t, d, "1.1.1.1", "v1")
// An old version could leave a validator idle while it still pointed at an address.
if _, err := d.ExecContext(ctx, `UPDATE validators SET state='idle' WHERE validator_id='v1'`); err != nil {
t.Fatal(err)
}
_ = d.SeedQueue(ctx, []string{"2.2.2.2"})
got, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
if err != nil || got != nil {
t.Fatalf("claim for a validator that holds %d: item=%v err=%v, want nothing", first.ID, got, err)
}
if ip, _ := d.GetIPByAddress(ctx, "2.2.2.2"); ip.State != IPQueued {
t.Fatalf("2.2.2.2 is %s, want queued", ip.State)
}
}
func TestListFIPRefsSkipsFinishedAddresses(t *testing.T) {
d, ctx := newTestDB(t)
var addrs []string
for i := 0; i < 6; i++ {
addrs = append(addrs, fmt.Sprintf("10.0.0.%d", i+1))
}
_ = d.SeedQueue(ctx, addrs)
states := []string{"done", "failed", "occupied", "awaiting_self_check", "checking", "aggregating"}
for i, a := range addrs {
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, fip_id=? WHERE ip_address=?`, states[i], fmt.Sprintf("fip-%d", i), a); err != nil {
t.Fatal(err)
}
}
refs, err := d.ListFIPRefs(ctx)
if err != nil {
t.Fatal(err)
}
if len(refs) != 3 {
t.Fatalf("ListFIPRefs returned %d rows, want the 3 unfinished ones: %+v", len(refs), refs)
}
byAddr, err := d.ListFIPRefsByAddresses(ctx, addrs)
if err != nil || len(byAddr) != 3 {
t.Fatalf("ListFIPRefsByAddresses returned %d rows (err %v), want 3", len(byAddr), err)
}
}
+63 -12
View File
@@ -27,24 +27,32 @@ func (d *DB) RegisterValidator(ctx context.Context, validatorID, hostname, osPor
}
// A brand-new row already lands in ValidatorIdle via the INSERT branch;
// a re-registering validator that was 'unregistered' or 'unreachable'
// (but not mid-assignment) should also come back to idle.
// comes back: to idle when it holds no address, to assigned when it still
// does (its current_ip_id is kept, so it must not be handed another).
_, err = d.ExecContext(ctx, `
UPDATE validators SET state=?, updated_at=?
UPDATE validators SET
state = CASE WHEN current_ip_id IS NULL THEN ? ELSE ? END,
updated_at=?
WHERE validator_id=? AND state IN (?, ?)
`, ValidatorIdle, now, validatorID, ValidatorUnregistered, ValidatorUnreachable)
`, ValidatorIdle, ValidatorAssigned, now, validatorID, ValidatorUnregistered, ValidatorUnreachable)
if err != nil {
return fmt.Errorf("register validator (reactivate): %w", err)
}
return nil
}
// Heartbeat records a sign of life. A validator that was marked unreachable
// returns to idle if it holds no address, but to assigned if it still does:
// it was only silent (for example busy with slow checks), and handing it a
// second address while it works on the first would leave the second one
// without an owner that can ever pick it up.
func (d *DB) Heartbeat(ctx context.Context, validatorID string) error {
now := timeToDB(Now())
res, err := d.ExecContext(ctx, `
UPDATE validators SET last_heartbeat_at=?, updated_at=?,
state = CASE WHEN state=? THEN ? ELSE state END
state = CASE WHEN state=? THEN (CASE WHEN current_ip_id IS NULL THEN ? ELSE ? END) ELSE state END
WHERE validator_id=?
`, now, now, ValidatorUnreachable, ValidatorIdle, validatorID)
`, now, now, ValidatorUnreachable, ValidatorIdle, ValidatorAssigned, validatorID)
if err != nil {
return fmt.Errorf("heartbeat: %w", err)
}
@@ -138,16 +146,59 @@ func (d *DB) MarkValidatorUnreachable(ctx context.Context, validatorID string) e
return err
}
// FreeValidator returns a validator to idle with no assigned IP. Used after
// an IP finishes (success or failure) or is reclaimed by the lease sweep.
func (d *DB) FreeValidator(ctx context.Context, validatorID string) error {
_, err := d.ExecContext(ctx, `
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
WHERE validator_id=?
`, ValidatorIdle, timeToDB(Now()), validatorID)
// freeValidatorSQL releases a validator from the address it holds. It only
// applies when the validator's current address is the one being released
// (args: now, validator id, ip id): a late release of an old address must not
// free a validator that has already moved on to another one. An unreachable
// validator stays unreachable until its next heartbeat, so a dead validator
// is not handed new addresses just because its lease was reclaimed.
var freeValidatorSQL = fmt.Sprintf(`
UPDATE validators SET current_ip_id=NULL,
state = CASE WHEN state='%s' THEN state ELSE '%s' END,
updated_at=?
WHERE validator_id=? AND current_ip_id=?`, ValidatorUnreachable, ValidatorIdle)
// FreeValidator releases a validator from the given address (see
// freeValidatorSQL). Used after an IP finishes (success or failure) or is
// reclaimed by the lease sweep.
func (d *DB) FreeValidator(ctx context.Context, validatorID string, ipID int64) error {
_, err := d.ExecContext(ctx, freeValidatorSQL, timeToDB(Now()), validatorID, ipID)
return err
}
// ReconcileValidators repairs validators whose state disagrees with the
// queue: a validator pointing at an address that no longer exists, is
// finished, or belongs to another validator is released; an "assigned"
// validator that holds nothing goes back to idle. The invariants normally
// hold by construction (every change is one transaction); this heals what a
// crash or an older version left behind. It returns the number of validators
// repaired.
func (d *DB) ReconcileValidators(ctx context.Context) (int64, error) {
now := timeToDB(Now())
res, err := d.ExecContext(ctx, fmt.Sprintf(`
UPDATE validators SET current_ip_id=NULL,
state = CASE WHEN state='%s' THEN state ELSE '%s' END,
updated_at=?
WHERE current_ip_id IS NOT NULL AND NOT EXISTS (
SELECT 1 FROM ip_queue q
WHERE q.id = validators.current_ip_id
AND q.owner_validator_id = validators.validator_id
AND q.state NOT IN ('%s','%s','%s'))`,
ValidatorUnreachable, ValidatorIdle, IPDone, IPFailed, IPOccupied), now)
if err != nil {
return 0, fmt.Errorf("reconcile validators: %w", err)
}
n, _ := res.RowsAffected()
res, err = d.ExecContext(ctx, `
UPDATE validators SET state=?, updated_at=?
WHERE state=? AND current_ip_id IS NULL`, ValidatorIdle, now, ValidatorAssigned)
if err != nil {
return n, fmt.Errorf("reconcile validators: %w", err)
}
m, _ := res.RowsAffected()
return n + m, nil
}
// AdminCreateValidator registers a brand-new validator via the admin API.
// Unlike RegisterValidator (used by the agent's self-registration call),
// this refuses to upsert over an existing row.
+27 -4
View File
@@ -1,17 +1,32 @@
package httpapi
import (
"context"
"errors"
"fmt"
"net/http"
"net/url"
"strconv"
"strings"
"time"
"cloudipvalidator/internal/db"
"cloudipvalidator/internal/orchestrator"
)
// destructiveOpTimeout bounds a cancel, delete or clear. Such an operation
// talks to the cloud as well as the database, and abandoning it halfway
// leaves floating IPs detached from rows that still exist (or the reverse),
// so it must not die with the client connection: a client that gives up
// (a dashboard or curl timeout) only stops waiting for the answer.
const destructiveOpTimeout = 10 * time.Minute
// detachedContext returns a context that ignores cancellation of the request
// but keeps its values, with destructiveOpTimeout as the upper bound.
func detachedContext(r *http.Request) (context.Context, context.CancelFunc) {
return context.WithTimeout(context.WithoutCancel(r.Context()), destructiveOpTimeout)
}
func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusOK, okResponse{OK: true})
}
@@ -273,7 +288,9 @@ func toScanStatusDTO(st orchestrator.ScanStatus) scanStatusDTO {
// need to be disassociated in OpenStack.
func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
address := r.PathValue("ip")
if err := s.Orch.ForceCancel(r.Context(), address); err != nil {
ctx, cancel := detachedContext(r)
defer cancel()
if err := s.Orch.ForceCancel(ctx, address); err != nil {
writeDBError(w, err)
return
}
@@ -286,7 +303,9 @@ func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
// associated floating IP needs disassociating first.
func (s *Server) handleAdminDeleteIP(w http.ResponseWriter, r *http.Request) {
address := r.PathValue("ip")
if err := s.Orch.DeleteIP(r.Context(), address); err != nil {
ctx, cancel := detachedContext(r)
defer cancel()
if err := s.Orch.DeleteIP(ctx, address); err != nil {
writeDBError(w, err)
return
}
@@ -306,7 +325,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusBadRequest, "addresses must not be empty")
return
}
result, err := s.Orch.DeleteIPs(r.Context(), req.Addresses)
ctx, cancel := detachedContext(r)
defer cancel()
result, err := s.Orch.DeleteIPs(ctx, req.Addresses)
if err != nil {
writeDBError(w, err)
return
@@ -320,7 +341,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
// handleAdminClearQueue permanently removes every address currently in the
// queue, including those actively being checked.
func (s *Server) handleAdminClearQueue(w http.ResponseWriter, r *http.Request) {
result, err := s.Orch.ClearQueue(r.Context())
ctx, cancel := detachedContext(r)
defer cancel()
result, err := s.Orch.ClearQueue(ctx)
if err != nil {
writeDBError(w, err)
return
@@ -0,0 +1,65 @@
package httpapi
import (
"context"
"net/http"
"testing"
"time"
"cloudipvalidator/internal/openstack"
)
// slowDetachOS delays every Disassociate so the client can give up first.
type slowDetachOS struct {
*openstack.MockClient
delay time.Duration
}
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
time.Sleep(s.delay)
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
}
// A client that gives up (a dashboard or curl timeout) must not abort
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
// part of the floating IPs were detached, the database was left untouched and
// the checks kept running.
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
fc, d, orch, mock := newConfigTestHarness(t)
ctx := context.Background()
mock.Seed("fip-1", "1.2.3.4", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
t.Fatal(err)
}
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
t.Fatal(err)
}
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
}
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
defer cancel()
// No body: the server only notices that the client has left (and cancels
// the request context) when it is not waiting for a request body.
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
if resp, err := fc.client.Do(req); err == nil {
resp.Body.Close()
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
}
deadline := time.Now().Add(5 * time.Second)
for {
ips, _ := d.ListIPs(ctx)
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
if len(ips) == 0 && fip.PortID == "" {
return
}
if time.Now().After(deadline) {
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
}
time.Sleep(50 * time.Millisecond)
}
}
+37 -12
View File
@@ -114,6 +114,11 @@ func (o *Orchestrator) leaseTTL() time.Duration {
// own goroutine, so validators never wait for each other. In Async mode
// Tick does not wait for those goroutines; otherwise it waits for them.
func (o *Orchestrator) Tick(ctx context.Context) {
if n, err := o.DB.ReconcileValidators(ctx); err != nil {
o.Log.Error("reconcile validators", "err", err)
} else if n > 0 {
o.Log.Warn("repaired validators that disagreed with the queue", "count", n)
}
if err := o.assignIdleValidators(ctx); err != nil {
o.Log.Error("assign idle validators", "err", err)
}
@@ -146,7 +151,11 @@ func (o *Orchestrator) assignIdleValidators(ctx context.Context) error {
}
o.Log.Info("claimed ip", "validator", v.ValidatorID, "ip", item.IPAddress, "ip_id", item.ID)
v, item := v, item
o.spawn("assign:"+v.ValidatorID, func() {
// Keyed by address, not validator: the key only guards against
// starting the same association twice. A validator-wide key made a
// second address claimed while the first was still associating skip
// its association and wait for the lease to expire.
o.spawn(fmt.Sprintf("assign:%d", item.ID), func() {
if err := o.associateFIP(ctx, v.ValidatorID, v.OSPortID, item); err != nil {
o.Log.Error("associate fip", "validator", v.ValidatorID, "ip", item.IPAddress, "err", err)
}
@@ -466,7 +475,7 @@ func (o *Orchestrator) ForceCancel(ctx context.Context, ipAddress string) error
}
if item.OwnerValidatorID != nil {
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID); err != nil {
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID, item.ID); err != nil {
return fmt.Errorf("free validator: %w", err)
}
o.releaseValidatorPorts(ctx, o.validatorsByID(ctx, *item.OwnerValidatorID))
@@ -522,11 +531,7 @@ func (o *Orchestrator) DeleteIPs(ctx context.Context, addresses []string) (db.De
if err != nil {
o.Log.Error("list attached fips before delete", "err", err)
}
for _, ref := range refs {
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
o.Log.Error("disassociate fip on delete", "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
}
}
o.disassociateAll(ctx, refs, "delete")
owners := o.busyValidatorsFor(ctx, addresses)
result, err := o.DB.DeleteIPs(ctx, addresses)
@@ -538,6 +543,30 @@ func (o *Orchestrator) DeleteIPs(ctx context.Context, addresses []string) (db.De
return result, nil
}
// maxParallelDetach bounds how many floating IPs a bulk delete or clear
// detaches at the same time (each is one Neutron call).
const maxParallelDetach = 8
// disassociateAll detaches the given floating IPs best-effort, at most
// maxParallelDetach at a time; a failure is logged and does not stop the rest.
func (o *Orchestrator) disassociateAll(ctx context.Context, refs []db.FIPRef, what string) {
var wg sync.WaitGroup
sem := make(chan struct{}, maxParallelDetach)
for _, ref := range refs {
ref := ref
sem <- struct{}{}
wg.Add(1)
go func() {
defer wg.Done()
defer func() { <-sem }()
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
o.Log.Error("disassociate fip on "+what, "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
}
}()
}
wg.Wait()
}
// ClearQueue deletes every address currently in the queue, regardless of
// state — the "delete everything" operation. It is set-based (see
// db.ClearAllIPs): O(1) statements however many rows there are. Floating IPs
@@ -548,11 +577,7 @@ func (o *Orchestrator) ClearQueue(ctx context.Context) (db.DeleteIPsResult, erro
if err != nil {
return db.DeleteIPsResult{}, fmt.Errorf("list attached fips: %w", err)
}
for _, ref := range refs {
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
o.Log.Error("disassociate fip on clear queue", "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
}
}
o.disassociateAll(ctx, refs, "clear queue")
deleted, err := o.DB.ClearAllIPs(ctx)
if err != nil {
@@ -0,0 +1,401 @@
package orchestrator
import (
"context"
"fmt"
"math/rand"
"sync/atomic"
"testing"
"time"
"cloudipvalidator/internal/db"
"cloudipvalidator/internal/openstack"
)
// finishChecks reports a successful egress run and all three inbound sites
// for the address, so the next Tick aggregates it and releases the validator.
func finishChecks(t *testing.T, o *Orchestrator, ip *db.IPQueueItem, validatorID string) {
t.Helper()
ctx := context.Background()
if err := o.RecordCheck(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
ValidatorID: validatorID, Source: db.SourceEgress, CheckType: "https",
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
}); err != nil {
t.Fatalf("record egress check: %v", err)
}
if err := o.MarkEgressComplete(ctx, ip.ID); err != nil {
t.Fatalf("mark egress complete: %v", err)
}
for site := 1; site <= 3; site++ {
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
if err := o.RecordCheck(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress,
Success: true, CheckedAt: db.Now(),
}); err != nil {
t.Fatalf("record inbound check: %v", err)
}
}
if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil {
t.Fatalf("mark site complete: %v", err)
}
}
}
// The incident of 2026-10-02: a validator busy with slow checks goes silent,
// is marked unreachable, and its next heartbeat used to return it to idle
// although it still held the address. It was then handed a second address,
// whose association never ran, and both stalled until their leases expired.
func TestSilentValidatorKeepsItsAddressAfterHeartbeat(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-a", "1.1.1.1", "svc")
mock.Seed("fip-b", "2.2.2.2", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
t.Fatal(err)
}
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
t.Fatal(err)
}
o.Tick(ctx) // validator-1 claims 1.1.1.1
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
if err := o.SelfCheckResult(ctx, "validator-1", a.ID, true, "ok"); err != nil {
t.Fatal(err)
}
// The agent goes silent for longer than heartbeat_timeout_seconds ...
if err := d.MarkValidatorUnreachable(ctx, "validator-1"); err != nil {
t.Fatal(err)
}
// ... and then speaks again while the checks are still running.
if err := d.Heartbeat(ctx, "validator-1"); err != nil {
t.Fatal(err)
}
v, _ := d.GetValidator(ctx, "validator-1")
if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != a.ID {
t.Fatalf("after heartbeat: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, a.ID)
}
o.Tick(ctx)
if b, _ := d.GetIPByAddress(ctx, "2.2.2.2"); b.State != db.IPQueued {
t.Fatalf("the busy validator was given a second address: 2.2.2.2 is %s", b.State)
}
// The first address finishes: only now the validator may take the next one.
a, _ = d.GetIP(ctx, a.ID)
finishChecks(t, o, a, "validator-1")
o.Tick(ctx) // aggregates and releases 1.1.1.1
o.Tick(ctx) // validator-1 is idle again and claims 2.2.2.2
if b, _ := d.GetIPByAddress(ctx, "2.2.2.2"); b.State != db.IPAwaitingSelfCheck {
t.Fatalf("2.2.2.2 is %s, want awaiting_self_check once the validator is free", b.State)
}
}
// A dead validator whose lease was reclaimed must stay out of rotation until
// it speaks again; before, reclaiming set it idle and it was handed a fresh
// address every lease period, burning each address's retries.
func TestUnreachableValidatorGetsNoAddressesAfterLeaseReclaim(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 1)
mock.Seed("fip-a", "1.1.1.1", "svc")
mock.Seed("fip-b", "2.2.2.2", "svc")
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
_ = d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"})
o.Tick(ctx) // claims 1.1.1.1, the agent never answers
if err := d.MarkValidatorUnreachable(ctx, "validator-1"); err != nil {
t.Fatal(err)
}
time.Sleep(1100 * time.Millisecond)
o.Cfg.LeaseTTLSeconds = 180
o.Tick(ctx) // lease sweep reclaims 1.1.1.1
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
if a.RetryCount != 1 {
t.Fatalf("retry_count = %d, want the lease to be reclaimed once", a.RetryCount)
}
v, _ := d.GetValidator(ctx, "validator-1")
if v.State != db.ValidatorUnreachable || v.CurrentIPID != nil {
t.Fatalf("validator state=%s current_ip=%v, want unreachable and empty", v.State, v.CurrentIPID)
}
o.Tick(ctx)
if a, _ := d.GetIPByAddress(ctx, "1.1.1.1"); a.State != db.IPQueued {
t.Fatalf("an unreachable validator was handed an address: 1.1.1.1 is %s", a.State)
}
if err := d.Heartbeat(ctx, "validator-1"); err != nil { // it is back
t.Fatal(err)
}
if v, _ := d.GetValidator(ctx, "validator-1"); v.State != db.ValidatorIdle {
t.Fatalf("after heartbeat state=%s, want idle (it holds nothing)", v.State)
}
o.Tick(ctx)
if a, _ := d.GetIPByAddress(ctx, "1.1.1.1"); a.State != db.IPAwaitingSelfCheck {
t.Fatalf("1.1.1.1 is %s, want it picked up again", a.State)
}
}
// countingOS counts Disassociate calls.
type countingOS struct {
*openstack.MockClient
disassociations int32
}
func (c *countingOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
atomic.AddInt32(&c.disassociations, 1)
return c.MockClient.DisassociateFloatingIP(ctx, fipID)
}
// Finished addresses keep their fip_id for display but their floating IP is
// already free; clearing the queue must not make a cloud call for each of
// them (on 2026-10-02 that was >1000 sequential calls: 256 s).
func TestClearQueueDoesNotTouchFinishedAddresses(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
cos := &countingOS{MockClient: mock}
o.OS = cos
const finished = 300
var addrs []string
for i := 0; i < finished; i++ {
addrs = append(addrs, fmt.Sprintf("10.9.%d.%d", i/200, i%200+1))
}
if err := d.SeedQueue(ctx, addrs); err != nil {
t.Fatal(err)
}
for i, a := range addrs {
state := []string{db.IPDone, db.IPFailed, db.IPOccupied}[i%3]
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, fip_id=? WHERE ip_address=?`, state, fmt.Sprintf("fip-old-%d", i), a); err != nil {
t.Fatal(err)
}
}
// One address really holds a floating IP.
mock.Seed("fip-live", "1.2.3.4", "svc")
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
live, _ := d.GetIPByAddress(ctx, "1.2.3.4")
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET sequence=-1 WHERE id=?`, live.ID); err != nil {
t.Fatal(err)
}
o.Tick(ctx) // validator-1 takes the live address (lowest sequence) and attaches it
if got := portFIPs(t, mock, "port-1"); len(got) != 1 {
t.Fatalf("setup: the live address is not attached (%d floating ips on port-1)", len(got))
}
atomic.StoreInt32(&cos.disassociations, 0)
start := time.Now()
if _, err := o.ClearQueue(ctx); err != nil {
t.Fatalf("clear queue: %v", err)
}
// One call for the live address; the port sweep finds nothing left.
if n := atomic.LoadInt32(&cos.disassociations); n != 1 {
t.Fatalf("clear queue made %d disassociate calls, want 1 (finished addresses must be skipped)", n)
}
if got := portFIPs(t, mock, "port-1"); len(got) != 0 {
t.Fatalf("the live floating ip is still attached: %+v", got)
}
if left, _ := d.ListIPs(ctx); len(left) != 0 {
t.Fatalf("%d rows left after clear", len(left))
}
t.Logf("clear of %d rows took %s", finished+1, time.Since(start))
}
// A bulk detach runs in parallel but never above maxParallelDetach at once.
func TestDisassociateAllIsBoundedAndParallel(t *testing.T) {
o, _, mock := newTestOrchestrator(t, 180)
g := &gaugeOS{MockClient: mock, delay: 30 * time.Millisecond}
o.OS = g
var refs []db.FIPRef
for i := 0; i < 40; i++ {
id := fmt.Sprintf("fip-%d", i)
mock.Seed(id, fmt.Sprintf("10.8.0.%d", i+1), "svc")
refs = append(refs, db.FIPRef{IPID: int64(i), IPAddress: id, FIPID: id})
}
start := time.Now()
o.disassociateAll(context.Background(), refs, "test")
elapsed := time.Since(start)
if peak := atomic.LoadInt32(&g.peak); peak < 2 || peak > maxParallelDetach {
t.Fatalf("peak concurrency %d, want between 2 and %d", peak, maxParallelDetach)
}
if elapsed > 600*time.Millisecond { // 40 x 30 ms sequentially is 1.2 s
t.Fatalf("took %s: the detach is not parallel", elapsed)
}
}
type gaugeOS struct {
*openstack.MockClient
delay time.Duration
cur, peak int32
}
func (g *gaugeOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
n := atomic.AddInt32(&g.cur, 1)
for {
p := atomic.LoadInt32(&g.peak)
if n <= p || atomic.CompareAndSwapInt32(&g.peak, p, n) {
break
}
}
time.Sleep(g.delay)
atomic.AddInt32(&g.cur, -1)
return g.MockClient.DisassociateFloatingIP(ctx, fipID)
}
// Rows that disagree with the queue are repaired by ReconcileValidators (run on every Tick).
func TestReconcileRepairsInconsistentValidators(t *testing.T) {
ctx := context.Background()
_, d, _ := newTestOrchestrator(t, 180)
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
_ = d.RegisterValidator(ctx, "validator-2", "host", "port-2", "v")
_ = d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"})
finished, _ := d.GetIPByAddress(ctx, "1.1.1.1")
foreign, _ := d.GetIPByAddress(ctx, "2.2.2.2")
// validator-1 still points at a finished address; validator-2 at an
// address owned by somebody else (what an older version could leave behind).
for _, q := range []string{
fmt.Sprintf(`UPDATE ip_queue SET state='done' WHERE id=%d`, finished.ID),
fmt.Sprintf(`UPDATE ip_queue SET state='checking', owner_validator_id='validator-1' WHERE id=%d`, foreign.ID),
fmt.Sprintf(`UPDATE validators SET state='assigned', current_ip_id=%d WHERE validator_id='validator-1'`, finished.ID),
fmt.Sprintf(`UPDATE validators SET state='assigned', current_ip_id=%d WHERE validator_id='validator-2'`, foreign.ID),
} {
if _, err := d.ExecContext(ctx, q); err != nil {
t.Fatal(err)
}
}
n, err := d.ReconcileValidators(ctx)
if err != nil || n != 2 {
t.Fatalf("reconcile repaired %d validators (err %v), want 2", n, err)
}
for _, id := range []string{"validator-1", "validator-2"} {
v, _ := d.GetValidator(ctx, id)
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
t.Fatalf("%s: state=%s current_ip=%v, want idle", id, v.State, v.CurrentIPID)
}
}
// A consistent validator is left alone.
if n, _ := d.ReconcileValidators(ctx); n != 0 {
t.Fatalf("second reconcile repaired %d, want 0", n)
}
}
// Randomised run: many validators, random silences and association failures.
// After every tick each validator holds at most one address, the validator
// and the address agree on who holds what, and no lease is ever reclaimed.
func TestRandomFlowKeepsOneAddressPerValidator(t *testing.T) {
ctx := context.Background()
rng := rand.New(rand.NewSource(42))
o, d, mock := newTestOrchestrator(t, 180)
const validators, addresses = 12, 60
var addrs []string
for i := 0; i < validators; i++ {
_ = d.RegisterValidator(ctx, fmt.Sprintf("validator-%d", i+1), "host", fmt.Sprintf("port-%d", i+1), "v")
}
for i := 0; i < addresses; i++ {
a := fmt.Sprintf("10.7.%d.%d", i/200, i%200+1)
addrs = append(addrs, a)
mock.Seed(fmt.Sprintf("fip-%d", i), a, "svc")
}
if err := d.SeedQueue(ctx, addrs); err != nil {
t.Fatal(err)
}
checkInvariants := func(tick int) {
t.Helper()
vs, _ := d.ListValidators(ctx)
holders := map[int64]string{}
for _, v := range vs {
if v.CurrentIPID == nil {
if v.State == db.ValidatorAssigned {
t.Fatalf("tick %d: %s is assigned but holds nothing", tick, v.ValidatorID)
}
continue
}
ip, err := d.GetIP(ctx, *v.CurrentIPID)
if err != nil {
t.Fatalf("tick %d: %s points at a missing address %d", tick, v.ValidatorID, *v.CurrentIPID)
}
if ip.OwnerValidatorID == nil || *ip.OwnerValidatorID != v.ValidatorID {
t.Fatalf("tick %d: %s holds %s but its owner is %v", tick, v.ValidatorID, ip.IPAddress, ip.OwnerValidatorID)
}
if ip.State == db.IPDone || ip.State == db.IPFailed || ip.State == db.IPOccupied {
t.Fatalf("tick %d: %s still holds the finished address %s", tick, v.ValidatorID, ip.IPAddress)
}
if prev, ok := holders[ip.ID]; ok {
t.Fatalf("tick %d: %s is held by both %s and %s", tick, ip.IPAddress, prev, v.ValidatorID)
}
holders[ip.ID] = v.ValidatorID
}
// Every owned, unfinished address is the current one of its owner.
ips, _ := d.ListIPs(ctx)
perOwner := map[string]int{}
for _, ip := range ips {
if ip.OwnerValidatorID != nil && ip.State != db.IPDone && ip.State != db.IPFailed && ip.State != db.IPOccupied {
perOwner[*ip.OwnerValidatorID]++
}
}
for owner, n := range perOwner {
if n > 1 {
t.Fatalf("tick %d: %s owns %d unfinished addresses", tick, owner, n)
}
}
}
checkTicks := map[int64]int{} // address id -> ticks spent in checking
done := false
for tick := 1; tick <= 600 && !done; tick++ {
// Occasionally an association fails (409 on a port).
if rng.Intn(15) == 0 {
mock.AssociateFailures = map[string]error{fmt.Sprintf("fip-%d", rng.Intn(addresses)): fmt.Errorf("409 conflict")}
}
o.Tick(ctx)
vs, _ := d.ListValidators(ctx)
for _, v := range vs {
// A busy agent sometimes goes silent, then speaks again.
if v.CurrentIPID != nil && v.State == db.ValidatorAssigned && rng.Intn(10) == 0 {
_ = d.MarkValidatorUnreachable(ctx, v.ValidatorID)
}
if v.State == db.ValidatorUnreachable && rng.Intn(3) == 0 {
_ = d.Heartbeat(ctx, v.ValidatorID)
}
item, _, err := o.AssignmentForValidator(ctx, v.ValidatorID)
if err != nil || item == nil {
continue
}
if item.State == db.IPAwaitingSelfCheck {
_ = o.SelfCheckResult(ctx, v.ValidatorID, item.ID, true, "ok")
continue
}
checkTicks[item.ID]++
if checkTicks[item.ID] >= 1+rng.Intn(4) {
finishChecks(t, o, item, v.ValidatorID)
delete(checkTicks, item.ID)
}
}
checkInvariants(tick)
counts, _, _ := d.CountIPsByState(ctx)
unfinished := 0
for st, n := range counts {
if st != db.IPDone && st != db.IPFailed && st != db.IPOccupied {
unfinished += n
}
}
done = unfinished == 0
}
if !done {
counts, _, _ := d.CountIPsByState(ctx)
t.Fatalf("not all addresses finished: %v", counts)
}
var expired int
if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM events WHERE event_type='lease_expired'`).Scan(&expired); err != nil {
t.Fatal(err)
}
if expired != 0 {
t.Fatalf("%d leases were reclaimed, want 0", expired)
}
}