Keep one address per validator; fix heartbeat handling and queue clear
A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42 addresses to fail without a single check. A validator busy with slow checks went silent, was marked unreachable, and its next heartbeat put it back to idle while it still held the address; it was handed a second one, whose association never ran (the in-flight guard was keyed by validator), and both waited for their leases to expire. - Heartbeat/re-register return an unreachable validator to assigned when it still holds an address, else idle. - A validator is released only from the address it currently holds (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an unreachable validator stays unreachable until its next heartbeat, so a dead validator is no longer handed a new address every lease period. - ClaimNextQueued refuses a validator that still has an address; a ReconcileValidators pass on every tick repairs rows that disagree with the queue. - Association guard is keyed by address, not validator. - The agent sends heartbeats from their own goroutine. - Clear queue / delete: detach only floating IPs of unfinished rows (done, failed and occupied rows kept their fip_id and made a clear issue >1000 sequential cloud calls: 256 s), at most 8 in parallel; the operation no longer dies with the client connection (10 minute limit). Includes the incident analysis and the plan under analysis/ and docs/changes/, and rebuilt bin/control-api and bin/validator-agent. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
cf4a883363
commit
0532baff09
17 files changed
+1293
-62
No files matched your search
@@ -19,6 +19,7 @@ import (
|
||||
neturl "net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/apiclient"
|
||||
@@ -33,6 +34,10 @@ type Agent struct {
|
||||
|
||||
lastHandledIPID int64
|
||||
|
||||
// busy is true while an assignment is being worked on; it is reported in
|
||||
// the heartbeat body (informational on the control-api side).
|
||||
busy atomic.Bool
|
||||
|
||||
// registerRetryInitial/Max govern the backoff used while waiting for a
|
||||
// successful registration (see registerWithRetry): control-api may not
|
||||
// be up yet at agent boot, or may come and go across a redeploy, and the
|
||||
@@ -70,6 +75,15 @@ func (a *Agent) Run(ctx context.Context) error {
|
||||
}
|
||||
|
||||
interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second
|
||||
|
||||
// Heartbeats run on their own schedule. Sent from the poll loop they
|
||||
// stopped for as long as a slow assignment took (an address whose
|
||||
// outbound targets all time out keeps the loop busy for ~40 s), which
|
||||
// control-api reads as a lost validator after heartbeat_timeout_seconds.
|
||||
hbCtx, stopHeartbeat := context.WithCancel(ctx)
|
||||
defer stopHeartbeat()
|
||||
go a.heartbeatLoop(hbCtx, interval)
|
||||
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
@@ -148,12 +162,32 @@ type checkConfigDTO struct {
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
func (a *Agent) pollOnce(ctx context.Context) {
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: "idle"}, nil); err != nil {
|
||||
a.log.Error("heartbeat", "err", err)
|
||||
return
|
||||
// heartbeatLoop sends a heartbeat now and then every interval until ctx is
|
||||
// cancelled.
|
||||
func (a *Agent) heartbeatLoop(ctx context.Context, interval time.Duration) {
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
a.sendHeartbeat(ctx)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) sendHeartbeat(ctx context.Context) {
|
||||
state := "idle"
|
||||
if a.busy.Load() {
|
||||
state = "checking"
|
||||
}
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: state}, nil); err != nil && ctx.Err() == nil {
|
||||
a.log.Error("heartbeat", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) pollOnce(ctx context.Context) {
|
||||
var assignment assignmentResp
|
||||
ok, err := a.client.Do(ctx, "GET", "/api/v1/agents/"+a.cfg.ValidatorID+"/assignment", nil, &assignment)
|
||||
if err != nil {
|
||||
@@ -169,6 +203,8 @@ func (a *Agent) pollOnce(ctx context.Context) {
|
||||
return // already handled this IP's work this attempt
|
||||
}
|
||||
|
||||
a.busy.Store(true)
|
||||
defer a.busy.Store(false)
|
||||
switch assignment.Phase {
|
||||
case "awaiting_self_check":
|
||||
a.handleSelfCheckAndRun(ctx, assignment)
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
package agentcore
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
// While the agent is busy with a slow assignment (an address whose outbound
|
||||
// targets time out keeps it occupied for tens of seconds) it must keep sending
|
||||
// heartbeats; control-api marks a validator that stays silent for
|
||||
// heartbeat_timeout_seconds as unreachable.
|
||||
func TestHeartbeatContinuesDuringSlowChecks(t *testing.T) {
|
||||
slowTarget := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
time.Sleep(3500 * time.Millisecond)
|
||||
}))
|
||||
defer slowTarget.Close()
|
||||
|
||||
var heartbeats, assignments int32
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
|
||||
fmt.Fprint(w, `{"ok":true}`)
|
||||
})
|
||||
mux.HandleFunc("POST /api/v1/agents/val-1/heartbeat", func(w http.ResponseWriter, r *http.Request) {
|
||||
atomic.AddInt32(&heartbeats, 1)
|
||||
fmt.Fprint(w, `{"ok":true}`)
|
||||
})
|
||||
mux.HandleFunc("GET /api/v1/agents/val-1/assignment", func(w http.ResponseWriter, r *http.Request) {
|
||||
if atomic.AddInt32(&assignments, 1) > 1 {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
return
|
||||
}
|
||||
fmt.Fprintf(w, `{"ip_id":1,"ip_address":"1.1.1.1","phase":"checking","check_config":[{"type":"https","targets":[%q]}]}`, slowTarget.URL)
|
||||
})
|
||||
mux.HandleFunc("POST /api/v1/agents/val-1/results", func(w http.ResponseWriter, r *http.Request) { fmt.Fprint(w, `{"ok":true}`) })
|
||||
mux.HandleFunc("POST /api/v1/agents/val-1/complete", func(w http.ResponseWriter, r *http.Request) { fmt.Fprint(w, `{"ok":true}`) })
|
||||
capi := httptest.NewServer(mux)
|
||||
defer capi.Close()
|
||||
|
||||
a := New(&config.ValidatorAgent{
|
||||
ValidatorID: "val-1", ControlAPIURL: capi.URL, PollIntervalSeconds: 1,
|
||||
Checks: config.AgentChecks{HTTPSTimeoutSeconds: 10, ICMPTimeoutSeconds: 1, ICMPCount: 1},
|
||||
}, testLogger())
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan struct{})
|
||||
go func() { _ = a.Run(ctx); close(done) }()
|
||||
|
||||
time.Sleep(3 * time.Second) // the slow check (3.5 s) is still running
|
||||
during := atomic.LoadInt32(&heartbeats)
|
||||
cancel()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("Run did not stop after the context was cancelled")
|
||||
}
|
||||
|
||||
// One per second plus the first: 3-4 in 3 s. With heartbeats in the poll
|
||||
// loop there is exactly one, sent before the slow assignment started.
|
||||
if during < 3 {
|
||||
t.Fatalf("%d heartbeats in 3 s while a check was running, want at least 3", during)
|
||||
}
|
||||
if atomic.LoadInt32(&assignments) < 1 {
|
||||
t.Fatal("the assignment was never fetched, the test did not exercise a busy agent")
|
||||
}
|
||||
}
|
||||
@@ -101,7 +101,7 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
|
||||
|
||||
res, err = tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=?, updated_at=?
|
||||
WHERE validator_id=? AND state=?
|
||||
WHERE validator_id=? AND state=? AND current_ip_id IS NULL
|
||||
`, ValidatorAssigned, item.ID, timeToDB(now), validatorID, ValidatorIdle)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -207,7 +207,8 @@ func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error {
|
||||
}
|
||||
|
||||
// ReleaseFIP records that the floating IP has been disassociated and frees
|
||||
// the owning validator back to idle, in one transaction.
|
||||
// the owning validator (if this address is still its current one, see
|
||||
// freeValidatorSQL), in one transaction.
|
||||
func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
@@ -219,10 +220,7 @@ func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) err
|
||||
if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET fip_released_at=?, updated_at=? WHERE id=?`, now, now, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, now, validatorID); err != nil {
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
@@ -252,10 +250,7 @@ func (d *DB) MarkFIPOccupied(ctx context.Context, ipID int64, validatorID string
|
||||
return err
|
||||
}
|
||||
if validatorID != "" {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, now, validatorID); err != nil {
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
@@ -315,10 +310,7 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
}
|
||||
|
||||
if validatorID != "" {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, now, validatorID); err != nil {
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
@@ -520,9 +512,11 @@ func (d *DB) DeleteIPs(ctx context.Context, addresses []string) (DeleteIPsResult
|
||||
func deleteIPTx(ctx context.Context, tx *sql.Tx, ipID int64) error {
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
UPDATE validators SET current_ip_id=NULL,
|
||||
state = CASE WHEN state=? THEN state ELSE ? END,
|
||||
updated_at=?
|
||||
WHERE current_ip_id=?
|
||||
`, ValidatorIdle, now, ipID); err != nil {
|
||||
`, ValidatorUnreachable, ValidatorIdle, now, ipID); err != nil {
|
||||
return fmt.Errorf("free owning validator: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `UPDATE checks SET ip_id=NULL WHERE ip_id=?`, ipID); err != nil {
|
||||
@@ -579,9 +573,11 @@ func (d *DB) ClearAllIPs(ctx context.Context) ([]string, error) {
|
||||
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
UPDATE validators SET current_ip_id=NULL,
|
||||
state = CASE WHEN state=? THEN state ELSE ? END,
|
||||
updated_at=?
|
||||
WHERE current_ip_id IS NOT NULL
|
||||
`, ValidatorIdle, now); err != nil {
|
||||
`, ValidatorUnreachable, ValidatorIdle, now); err != nil {
|
||||
return nil, fmt.Errorf("free owning validators: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `UPDATE checks SET ip_id=NULL WHERE ip_id IS NOT NULL`); err != nil {
|
||||
@@ -610,11 +606,15 @@ type FIPRef struct {
|
||||
FIPID string
|
||||
}
|
||||
|
||||
// ListFIPRefs returns every queue row with an attached floating IP (fip_id
|
||||
// set) — typically at most one per validator — so a bulk clear can
|
||||
// disassociate them without loading the whole queue.
|
||||
// ListFIPRefs returns the queue rows that may still hold a floating IP: an
|
||||
// fip_id is set and the row is not finished. A finished row (done, failed,
|
||||
// occupied) keeps its fip_id for display, but its floating IP was already
|
||||
// disassociated before the final state was written, so listing it would only
|
||||
// make a bulk clear issue thousands of pointless cloud calls. At most one row
|
||||
// per validator qualifies, so a clear does not need to load the whole queue.
|
||||
func (d *DB) ListFIPRefs(ctx context.Context) ([]FIPRef, error) {
|
||||
rows, err := d.QueryContext(ctx, `SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' ORDER BY id`)
|
||||
rows, err := d.QueryContext(ctx, `SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND state NOT IN (?, ?, ?) ORDER BY id`,
|
||||
IPDone, IPFailed, IPOccupied)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -631,8 +631,7 @@ func (d *DB) ListFIPRefs(ctx context.Context) ([]FIPRef, error) {
|
||||
}
|
||||
|
||||
// ListFIPRefsByAddresses is ListFIPRefs restricted to the given addresses
|
||||
// (unknown addresses and rows without an attached floating IP are simply
|
||||
// absent), using a handful of IN (...) queries instead of one lookup per
|
||||
// (unknown addresses and rows that hold no floating IP are simply absent), using a handful of IN (...) queries instead of one lookup per
|
||||
// address.
|
||||
func (d *DB) ListFIPRefsByAddresses(ctx context.Context, addresses []string) ([]FIPRef, error) {
|
||||
const chunk = 500
|
||||
@@ -643,12 +642,12 @@ func (d *DB) ListFIPRefsByAddresses(ctx context.Context, addresses []string) ([]
|
||||
end = len(addresses)
|
||||
}
|
||||
part := addresses[start:end]
|
||||
args := make([]any, len(part))
|
||||
for i, a := range part {
|
||||
args[i] = a
|
||||
args := []any{IPDone, IPFailed, IPOccupied}
|
||||
for _, a := range part {
|
||||
args = append(args, a)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx,
|
||||
`SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND ip_address IN (`+placeholders(len(part))+`)`, args...)
|
||||
`SELECT id, ip_address, fip_id FROM ip_queue WHERE fip_id<>'' AND state NOT IN (?, ?, ?) AND ip_address IN (`+placeholders(len(part))+`)`, args...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func validatorState(t *testing.T, d *DB, id string) *Validator {
|
||||
t.Helper()
|
||||
v, err := d.GetValidator(testCtx(t), id)
|
||||
if err != nil {
|
||||
t.Fatalf("get validator %s: %v", id, err)
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
func testCtx(t *testing.T) context.Context {
|
||||
t.Helper()
|
||||
return context.Background()
|
||||
}
|
||||
|
||||
// claimFor2 seeds an address and returns its id without claiming it.
|
||||
func claimFor2(t *testing.T, d *DB, addr string) int64 {
|
||||
t.Helper()
|
||||
ctx := testCtx(t)
|
||||
if err := d.SeedQueue(ctx, []string{addr}); err != nil {
|
||||
t.Fatalf("seed %s: %v", addr, err)
|
||||
}
|
||||
ip, err := d.GetIPByAddress(ctx, addr)
|
||||
if err != nil {
|
||||
t.Fatalf("get %s: %v", addr, err)
|
||||
}
|
||||
return ip.ID
|
||||
}
|
||||
|
||||
// claimFor seeds an address and claims it for the validator.
|
||||
func claimFor(t *testing.T, d *DB, addr, validatorID string) *IPQueueItem {
|
||||
t.Helper()
|
||||
ctx := testCtx(t)
|
||||
if err := d.SeedQueue(ctx, []string{addr}); err != nil {
|
||||
t.Fatalf("seed %s: %v", addr, err)
|
||||
}
|
||||
item, err := d.ClaimNextQueued(ctx, validatorID, time.Minute)
|
||||
if err != nil || item == nil {
|
||||
t.Fatalf("claim %s for %s: item=%v err=%v", addr, validatorID, item, err)
|
||||
}
|
||||
return item
|
||||
}
|
||||
|
||||
func TestHeartbeatKeepsAssignedWhenValidatorStillHoldsAnAddress(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
_ = d.AdminCreateValidator(ctx, "v1", "p1")
|
||||
_ = d.AdminCreateValidator(ctx, "v2", "p2")
|
||||
item := claimFor(t, d, "1.1.1.1", "v1")
|
||||
_ = d.MarkValidatorUnreachable(ctx, "v1")
|
||||
_ = d.MarkValidatorUnreachable(ctx, "v2")
|
||||
|
||||
if err := d.Heartbeat(ctx, "v1"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v := validatorState(t, d, "v1"); v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != item.ID {
|
||||
t.Fatalf("v1 after heartbeat: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, item.ID)
|
||||
}
|
||||
if err := d.Heartbeat(ctx, "v2"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v := validatorState(t, d, "v2"); v.State != ValidatorIdle {
|
||||
t.Fatalf("v2 after heartbeat: state=%s, want idle", v.State)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterValidatorReactivationKeepsAssignedAddress(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
_ = d.AdminCreateValidator(ctx, "v1", "p1")
|
||||
item := claimFor(t, d, "1.1.1.1", "v1")
|
||||
_ = d.MarkValidatorUnreachable(ctx, "v1")
|
||||
|
||||
if err := d.RegisterValidator(ctx, "v1", "host", "p1", "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v := validatorState(t, d, "v1"); v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != item.ID {
|
||||
t.Fatalf("after re-register: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, item.ID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStaleReleasesLeaveTheCurrentAddressAlone(t *testing.T) {
|
||||
cases := map[string]func(d *DB, ipID int64) error{
|
||||
"ReleaseFIP": func(d *DB, id int64) error { return d.ReleaseFIP(context.Background(), id, "v1") },
|
||||
"RequeueOrFail": func(d *DB, id int64) error { return d.RequeueOrFail(context.Background(), id, "v1", 3) },
|
||||
"MarkFIPOccupied": func(d *DB, id int64) error { return d.MarkFIPOccupied(context.Background(), id, "v1") },
|
||||
"FreeValidator": func(d *DB, id int64) error { return d.FreeValidator(context.Background(), "v1", id) },
|
||||
}
|
||||
for name, release := range cases {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
_ = d.AdminCreateValidator(ctx, "v1", "p1")
|
||||
stale := claimFor(t, d, "1.1.1.1", "v1")
|
||||
// v1 has moved on to another address (set directly: the claim path
|
||||
// itself refuses a validator that is still busy).
|
||||
cur := claimFor2(t, d, "2.2.2.2")
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state='checking', owner_validator_id='v1' WHERE id=?`, cur); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := d.ExecContext(ctx, `UPDATE validators SET state='assigned', current_ip_id=? WHERE validator_id='v1'`, cur); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if err := release(d, stale.ID); err != nil {
|
||||
t.Fatalf("%s: %v", name, err)
|
||||
}
|
||||
v := validatorState(t, d, "v1")
|
||||
if v.State != ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != cur {
|
||||
t.Fatalf("%s freed a validator that holds another address: state=%s current_ip=%v", name, v.State, v.CurrentIPID)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The right address frees the validator, but an unreachable one stays so.
|
||||
func TestReleaseKeepsUnreachableValidatorUnreachable(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
_ = d.AdminCreateValidator(ctx, "v1", "p1")
|
||||
_ = d.AdminCreateValidator(ctx, "v2", "p2")
|
||||
a := claimFor(t, d, "1.1.1.1", "v1")
|
||||
b := claimFor(t, d, "2.2.2.2", "v2")
|
||||
_ = d.MarkValidatorUnreachable(ctx, "v1")
|
||||
|
||||
if err := d.ReleaseFIP(ctx, a.ID, "v1"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v := validatorState(t, d, "v1"); v.State != ValidatorUnreachable || v.CurrentIPID != nil {
|
||||
t.Fatalf("v1: state=%s current_ip=%v, want unreachable and empty", v.State, v.CurrentIPID)
|
||||
}
|
||||
if err := d.RequeueOrFail(ctx, b.ID, "v2", 3); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v := validatorState(t, d, "v2"); v.State != ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("v2: state=%s current_ip=%v, want idle and empty", v.State, v.CurrentIPID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestClaimRefusesValidatorThatStillHoldsAnAddress(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
_ = d.AdminCreateValidator(ctx, "v1", "p1")
|
||||
first := claimFor(t, d, "1.1.1.1", "v1")
|
||||
// An old version could leave a validator idle while it still pointed at an address.
|
||||
if _, err := d.ExecContext(ctx, `UPDATE validators SET state='idle' WHERE validator_id='v1'`); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_ = d.SeedQueue(ctx, []string{"2.2.2.2"})
|
||||
got, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
|
||||
if err != nil || got != nil {
|
||||
t.Fatalf("claim for a validator that holds %d: item=%v err=%v, want nothing", first.ID, got, err)
|
||||
}
|
||||
if ip, _ := d.GetIPByAddress(ctx, "2.2.2.2"); ip.State != IPQueued {
|
||||
t.Fatalf("2.2.2.2 is %s, want queued", ip.State)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListFIPRefsSkipsFinishedAddresses(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
var addrs []string
|
||||
for i := 0; i < 6; i++ {
|
||||
addrs = append(addrs, fmt.Sprintf("10.0.0.%d", i+1))
|
||||
}
|
||||
_ = d.SeedQueue(ctx, addrs)
|
||||
states := []string{"done", "failed", "occupied", "awaiting_self_check", "checking", "aggregating"}
|
||||
for i, a := range addrs {
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, fip_id=? WHERE ip_address=?`, states[i], fmt.Sprintf("fip-%d", i), a); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
refs, err := d.ListFIPRefs(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(refs) != 3 {
|
||||
t.Fatalf("ListFIPRefs returned %d rows, want the 3 unfinished ones: %+v", len(refs), refs)
|
||||
}
|
||||
byAddr, err := d.ListFIPRefsByAddresses(ctx, addrs)
|
||||
if err != nil || len(byAddr) != 3 {
|
||||
t.Fatalf("ListFIPRefsByAddresses returned %d rows (err %v), want 3", len(byAddr), err)
|
||||
}
|
||||
}
|
||||
@@ -27,24 +27,32 @@ func (d *DB) RegisterValidator(ctx context.Context, validatorID, hostname, osPor
|
||||
}
|
||||
// A brand-new row already lands in ValidatorIdle via the INSERT branch;
|
||||
// a re-registering validator that was 'unregistered' or 'unreachable'
|
||||
// (but not mid-assignment) should also come back to idle.
|
||||
// comes back: to idle when it holds no address, to assigned when it still
|
||||
// does (its current_ip_id is kept, so it must not be handed another).
|
||||
_, err = d.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, updated_at=?
|
||||
UPDATE validators SET
|
||||
state = CASE WHEN current_ip_id IS NULL THEN ? ELSE ? END,
|
||||
updated_at=?
|
||||
WHERE validator_id=? AND state IN (?, ?)
|
||||
`, ValidatorIdle, now, validatorID, ValidatorUnregistered, ValidatorUnreachable)
|
||||
`, ValidatorIdle, ValidatorAssigned, now, validatorID, ValidatorUnregistered, ValidatorUnreachable)
|
||||
if err != nil {
|
||||
return fmt.Errorf("register validator (reactivate): %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Heartbeat records a sign of life. A validator that was marked unreachable
|
||||
// returns to idle if it holds no address, but to assigned if it still does:
|
||||
// it was only silent (for example busy with slow checks), and handing it a
|
||||
// second address while it works on the first would leave the second one
|
||||
// without an owner that can ever pick it up.
|
||||
func (d *DB) Heartbeat(ctx context.Context, validatorID string) error {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, `
|
||||
UPDATE validators SET last_heartbeat_at=?, updated_at=?,
|
||||
state = CASE WHEN state=? THEN ? ELSE state END
|
||||
state = CASE WHEN state=? THEN (CASE WHEN current_ip_id IS NULL THEN ? ELSE ? END) ELSE state END
|
||||
WHERE validator_id=?
|
||||
`, now, now, ValidatorUnreachable, ValidatorIdle, validatorID)
|
||||
`, now, now, ValidatorUnreachable, ValidatorIdle, ValidatorAssigned, validatorID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("heartbeat: %w", err)
|
||||
}
|
||||
@@ -138,16 +146,59 @@ func (d *DB) MarkValidatorUnreachable(ctx context.Context, validatorID string) e
|
||||
return err
|
||||
}
|
||||
|
||||
// FreeValidator returns a validator to idle with no assigned IP. Used after
|
||||
// an IP finishes (success or failure) or is reclaimed by the lease sweep.
|
||||
func (d *DB) FreeValidator(ctx context.Context, validatorID string) error {
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, timeToDB(Now()), validatorID)
|
||||
// freeValidatorSQL releases a validator from the address it holds. It only
|
||||
// applies when the validator's current address is the one being released
|
||||
// (args: now, validator id, ip id): a late release of an old address must not
|
||||
// free a validator that has already moved on to another one. An unreachable
|
||||
// validator stays unreachable until its next heartbeat, so a dead validator
|
||||
// is not handed new addresses just because its lease was reclaimed.
|
||||
var freeValidatorSQL = fmt.Sprintf(`
|
||||
UPDATE validators SET current_ip_id=NULL,
|
||||
state = CASE WHEN state='%s' THEN state ELSE '%s' END,
|
||||
updated_at=?
|
||||
WHERE validator_id=? AND current_ip_id=?`, ValidatorUnreachable, ValidatorIdle)
|
||||
|
||||
// FreeValidator releases a validator from the given address (see
|
||||
// freeValidatorSQL). Used after an IP finishes (success or failure) or is
|
||||
// reclaimed by the lease sweep.
|
||||
func (d *DB) FreeValidator(ctx context.Context, validatorID string, ipID int64) error {
|
||||
_, err := d.ExecContext(ctx, freeValidatorSQL, timeToDB(Now()), validatorID, ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
// ReconcileValidators repairs validators whose state disagrees with the
|
||||
// queue: a validator pointing at an address that no longer exists, is
|
||||
// finished, or belongs to another validator is released; an "assigned"
|
||||
// validator that holds nothing goes back to idle. The invariants normally
|
||||
// hold by construction (every change is one transaction); this heals what a
|
||||
// crash or an older version left behind. It returns the number of validators
|
||||
// repaired.
|
||||
func (d *DB) ReconcileValidators(ctx context.Context) (int64, error) {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, fmt.Sprintf(`
|
||||
UPDATE validators SET current_ip_id=NULL,
|
||||
state = CASE WHEN state='%s' THEN state ELSE '%s' END,
|
||||
updated_at=?
|
||||
WHERE current_ip_id IS NOT NULL AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_queue q
|
||||
WHERE q.id = validators.current_ip_id
|
||||
AND q.owner_validator_id = validators.validator_id
|
||||
AND q.state NOT IN ('%s','%s','%s'))`,
|
||||
ValidatorUnreachable, ValidatorIdle, IPDone, IPFailed, IPOccupied), now)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("reconcile validators: %w", err)
|
||||
}
|
||||
n, _ := res.RowsAffected()
|
||||
res, err = d.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, updated_at=?
|
||||
WHERE state=? AND current_ip_id IS NULL`, ValidatorIdle, now, ValidatorAssigned)
|
||||
if err != nil {
|
||||
return n, fmt.Errorf("reconcile validators: %w", err)
|
||||
}
|
||||
m, _ := res.RowsAffected()
|
||||
return n + m, nil
|
||||
}
|
||||
|
||||
// AdminCreateValidator registers a brand-new validator via the admin API.
|
||||
// Unlike RegisterValidator (used by the agent's self-registration call),
|
||||
// this refuses to upsert over an existing row.
|
||||
|
||||
@@ -1,17 +1,32 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/orchestrator"
|
||||
)
|
||||
|
||||
// destructiveOpTimeout bounds a cancel, delete or clear. Such an operation
|
||||
// talks to the cloud as well as the database, and abandoning it halfway
|
||||
// leaves floating IPs detached from rows that still exist (or the reverse),
|
||||
// so it must not die with the client connection: a client that gives up
|
||||
// (a dashboard or curl timeout) only stops waiting for the answer.
|
||||
const destructiveOpTimeout = 10 * time.Minute
|
||||
|
||||
// detachedContext returns a context that ignores cancellation of the request
|
||||
// but keeps its values, with destructiveOpTimeout as the upper bound.
|
||||
func detachedContext(r *http.Request) (context.Context, context.CancelFunc) {
|
||||
return context.WithTimeout(context.WithoutCancel(r.Context()), destructiveOpTimeout)
|
||||
}
|
||||
|
||||
func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
@@ -273,7 +288,9 @@ func toScanStatusDTO(st orchestrator.ScanStatus) scanStatusDTO {
|
||||
// need to be disassociated in OpenStack.
|
||||
func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
|
||||
address := r.PathValue("ip")
|
||||
if err := s.Orch.ForceCancel(r.Context(), address); err != nil {
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
if err := s.Orch.ForceCancel(ctx, address); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
@@ -286,7 +303,9 @@ func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
|
||||
// associated floating IP needs disassociating first.
|
||||
func (s *Server) handleAdminDeleteIP(w http.ResponseWriter, r *http.Request) {
|
||||
address := r.PathValue("ip")
|
||||
if err := s.Orch.DeleteIP(r.Context(), address); err != nil {
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
if err := s.Orch.DeleteIP(ctx, address); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
@@ -306,7 +325,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusBadRequest, "addresses must not be empty")
|
||||
return
|
||||
}
|
||||
result, err := s.Orch.DeleteIPs(r.Context(), req.Addresses)
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
result, err := s.Orch.DeleteIPs(ctx, req.Addresses)
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
@@ -320,7 +341,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
|
||||
// handleAdminClearQueue permanently removes every address currently in the
|
||||
// queue, including those actively being checked.
|
||||
func (s *Server) handleAdminClearQueue(w http.ResponseWriter, r *http.Request) {
|
||||
result, err := s.Orch.ClearQueue(r.Context())
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
result, err := s.Orch.ClearQueue(ctx)
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
// slowDetachOS delays every Disassociate so the client can give up first.
|
||||
type slowDetachOS struct {
|
||||
*openstack.MockClient
|
||||
delay time.Duration
|
||||
}
|
||||
|
||||
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
time.Sleep(s.delay)
|
||||
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
|
||||
}
|
||||
|
||||
// A client that gives up (a dashboard or curl timeout) must not abort
|
||||
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
|
||||
// part of the floating IPs were detached, the database was left untouched and
|
||||
// the checks kept running.
|
||||
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
|
||||
fc, d, orch, mock := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
||||
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
|
||||
}
|
||||
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
|
||||
|
||||
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
|
||||
defer cancel()
|
||||
// No body: the server only notices that the client has left (and cancels
|
||||
// the request context) when it is not waiting for a request body.
|
||||
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
|
||||
if resp, err := fc.client.Do(req); err == nil {
|
||||
resp.Body.Close()
|
||||
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for {
|
||||
ips, _ := d.ListIPs(ctx)
|
||||
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
|
||||
if len(ips) == 0 && fip.PortID == "" {
|
||||
return
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
@@ -114,6 +114,11 @@ func (o *Orchestrator) leaseTTL() time.Duration {
|
||||
// own goroutine, so validators never wait for each other. In Async mode
|
||||
// Tick does not wait for those goroutines; otherwise it waits for them.
|
||||
func (o *Orchestrator) Tick(ctx context.Context) {
|
||||
if n, err := o.DB.ReconcileValidators(ctx); err != nil {
|
||||
o.Log.Error("reconcile validators", "err", err)
|
||||
} else if n > 0 {
|
||||
o.Log.Warn("repaired validators that disagreed with the queue", "count", n)
|
||||
}
|
||||
if err := o.assignIdleValidators(ctx); err != nil {
|
||||
o.Log.Error("assign idle validators", "err", err)
|
||||
}
|
||||
@@ -146,7 +151,11 @@ func (o *Orchestrator) assignIdleValidators(ctx context.Context) error {
|
||||
}
|
||||
o.Log.Info("claimed ip", "validator", v.ValidatorID, "ip", item.IPAddress, "ip_id", item.ID)
|
||||
v, item := v, item
|
||||
o.spawn("assign:"+v.ValidatorID, func() {
|
||||
// Keyed by address, not validator: the key only guards against
|
||||
// starting the same association twice. A validator-wide key made a
|
||||
// second address claimed while the first was still associating skip
|
||||
// its association and wait for the lease to expire.
|
||||
o.spawn(fmt.Sprintf("assign:%d", item.ID), func() {
|
||||
if err := o.associateFIP(ctx, v.ValidatorID, v.OSPortID, item); err != nil {
|
||||
o.Log.Error("associate fip", "validator", v.ValidatorID, "ip", item.IPAddress, "err", err)
|
||||
}
|
||||
@@ -466,7 +475,7 @@ func (o *Orchestrator) ForceCancel(ctx context.Context, ipAddress string) error
|
||||
}
|
||||
|
||||
if item.OwnerValidatorID != nil {
|
||||
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID); err != nil {
|
||||
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID, item.ID); err != nil {
|
||||
return fmt.Errorf("free validator: %w", err)
|
||||
}
|
||||
o.releaseValidatorPorts(ctx, o.validatorsByID(ctx, *item.OwnerValidatorID))
|
||||
@@ -522,11 +531,7 @@ func (o *Orchestrator) DeleteIPs(ctx context.Context, addresses []string) (db.De
|
||||
if err != nil {
|
||||
o.Log.Error("list attached fips before delete", "err", err)
|
||||
}
|
||||
for _, ref := range refs {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on delete", "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
|
||||
}
|
||||
}
|
||||
o.disassociateAll(ctx, refs, "delete")
|
||||
|
||||
owners := o.busyValidatorsFor(ctx, addresses)
|
||||
result, err := o.DB.DeleteIPs(ctx, addresses)
|
||||
@@ -538,6 +543,30 @@ func (o *Orchestrator) DeleteIPs(ctx context.Context, addresses []string) (db.De
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// maxParallelDetach bounds how many floating IPs a bulk delete or clear
|
||||
// detaches at the same time (each is one Neutron call).
|
||||
const maxParallelDetach = 8
|
||||
|
||||
// disassociateAll detaches the given floating IPs best-effort, at most
|
||||
// maxParallelDetach at a time; a failure is logged and does not stop the rest.
|
||||
func (o *Orchestrator) disassociateAll(ctx context.Context, refs []db.FIPRef, what string) {
|
||||
var wg sync.WaitGroup
|
||||
sem := make(chan struct{}, maxParallelDetach)
|
||||
for _, ref := range refs {
|
||||
ref := ref
|
||||
sem <- struct{}{}
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on "+what, "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// ClearQueue deletes every address currently in the queue, regardless of
|
||||
// state — the "delete everything" operation. It is set-based (see
|
||||
// db.ClearAllIPs): O(1) statements however many rows there are. Floating IPs
|
||||
@@ -548,11 +577,7 @@ func (o *Orchestrator) ClearQueue(ctx context.Context) (db.DeleteIPsResult, erro
|
||||
if err != nil {
|
||||
return db.DeleteIPsResult{}, fmt.Errorf("list attached fips: %w", err)
|
||||
}
|
||||
for _, ref := range refs {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, ref.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on clear queue", "ip_id", ref.IPID, "fip_id", ref.FIPID, "err", err)
|
||||
}
|
||||
}
|
||||
o.disassociateAll(ctx, refs, "clear queue")
|
||||
|
||||
deleted, err := o.DB.ClearAllIPs(ctx)
|
||||
if err != nil {
|
||||
|
||||
@@ -0,0 +1,401 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
// finishChecks reports a successful egress run and all three inbound sites
|
||||
// for the address, so the next Tick aggregates it and releases the validator.
|
||||
func finishChecks(t *testing.T, o *Orchestrator, ip *db.IPQueueItem, validatorID string) {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: validatorID, Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record egress check: %v", err)
|
||||
}
|
||||
if err := o.MarkEgressComplete(ctx, ip.ID); err != nil {
|
||||
t.Fatalf("mark egress complete: %v", err)
|
||||
}
|
||||
for site := 1; site <= 3; site++ {
|
||||
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress,
|
||||
Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record inbound check: %v", err)
|
||||
}
|
||||
}
|
||||
if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil {
|
||||
t.Fatalf("mark site complete: %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The incident of 2026-10-02: a validator busy with slow checks goes silent,
|
||||
// is marked unreachable, and its next heartbeat used to return it to idle
|
||||
// although it still held the address. It was then handed a second address,
|
||||
// whose association never ran, and both stalled until their leases expired.
|
||||
func TestSilentValidatorKeepsItsAddressAfterHeartbeat(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-a", "1.1.1.1", "svc")
|
||||
mock.Seed("fip-b", "2.2.2.2", "svc")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
o.Tick(ctx) // validator-1 claims 1.1.1.1
|
||||
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if err := o.SelfCheckResult(ctx, "validator-1", a.ID, true, "ok"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// The agent goes silent for longer than heartbeat_timeout_seconds ...
|
||||
if err := d.MarkValidatorUnreachable(ctx, "validator-1"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// ... and then speaks again while the checks are still running.
|
||||
if err := d.Heartbeat(ctx, "validator-1"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
v, _ := d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != a.ID {
|
||||
t.Fatalf("after heartbeat: state=%s current_ip=%v, want assigned to %d", v.State, v.CurrentIPID, a.ID)
|
||||
}
|
||||
|
||||
o.Tick(ctx)
|
||||
if b, _ := d.GetIPByAddress(ctx, "2.2.2.2"); b.State != db.IPQueued {
|
||||
t.Fatalf("the busy validator was given a second address: 2.2.2.2 is %s", b.State)
|
||||
}
|
||||
|
||||
// The first address finishes: only now the validator may take the next one.
|
||||
a, _ = d.GetIP(ctx, a.ID)
|
||||
finishChecks(t, o, a, "validator-1")
|
||||
o.Tick(ctx) // aggregates and releases 1.1.1.1
|
||||
o.Tick(ctx) // validator-1 is idle again and claims 2.2.2.2
|
||||
if b, _ := d.GetIPByAddress(ctx, "2.2.2.2"); b.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("2.2.2.2 is %s, want awaiting_self_check once the validator is free", b.State)
|
||||
}
|
||||
}
|
||||
|
||||
// A dead validator whose lease was reclaimed must stay out of rotation until
|
||||
// it speaks again; before, reclaiming set it idle and it was handed a fresh
|
||||
// address every lease period, burning each address's retries.
|
||||
func TestUnreachableValidatorGetsNoAddressesAfterLeaseReclaim(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 1)
|
||||
mock.Seed("fip-a", "1.1.1.1", "svc")
|
||||
mock.Seed("fip-b", "2.2.2.2", "svc")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
|
||||
_ = d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"})
|
||||
|
||||
o.Tick(ctx) // claims 1.1.1.1, the agent never answers
|
||||
if err := d.MarkValidatorUnreachable(ctx, "validator-1"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
time.Sleep(1100 * time.Millisecond)
|
||||
o.Cfg.LeaseTTLSeconds = 180
|
||||
o.Tick(ctx) // lease sweep reclaims 1.1.1.1
|
||||
|
||||
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if a.RetryCount != 1 {
|
||||
t.Fatalf("retry_count = %d, want the lease to be reclaimed once", a.RetryCount)
|
||||
}
|
||||
v, _ := d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorUnreachable || v.CurrentIPID != nil {
|
||||
t.Fatalf("validator state=%s current_ip=%v, want unreachable and empty", v.State, v.CurrentIPID)
|
||||
}
|
||||
o.Tick(ctx)
|
||||
if a, _ := d.GetIPByAddress(ctx, "1.1.1.1"); a.State != db.IPQueued {
|
||||
t.Fatalf("an unreachable validator was handed an address: 1.1.1.1 is %s", a.State)
|
||||
}
|
||||
|
||||
if err := d.Heartbeat(ctx, "validator-1"); err != nil { // it is back
|
||||
t.Fatal(err)
|
||||
}
|
||||
if v, _ := d.GetValidator(ctx, "validator-1"); v.State != db.ValidatorIdle {
|
||||
t.Fatalf("after heartbeat state=%s, want idle (it holds nothing)", v.State)
|
||||
}
|
||||
o.Tick(ctx)
|
||||
if a, _ := d.GetIPByAddress(ctx, "1.1.1.1"); a.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("1.1.1.1 is %s, want it picked up again", a.State)
|
||||
}
|
||||
}
|
||||
|
||||
// countingOS counts Disassociate calls.
|
||||
type countingOS struct {
|
||||
*openstack.MockClient
|
||||
disassociations int32
|
||||
}
|
||||
|
||||
func (c *countingOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
atomic.AddInt32(&c.disassociations, 1)
|
||||
return c.MockClient.DisassociateFloatingIP(ctx, fipID)
|
||||
}
|
||||
|
||||
// Finished addresses keep their fip_id for display but their floating IP is
|
||||
// already free; clearing the queue must not make a cloud call for each of
|
||||
// them (on 2026-10-02 that was >1000 sequential calls: 256 s).
|
||||
func TestClearQueueDoesNotTouchFinishedAddresses(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
cos := &countingOS{MockClient: mock}
|
||||
o.OS = cos
|
||||
|
||||
const finished = 300
|
||||
var addrs []string
|
||||
for i := 0; i < finished; i++ {
|
||||
addrs = append(addrs, fmt.Sprintf("10.9.%d.%d", i/200, i%200+1))
|
||||
}
|
||||
if err := d.SeedQueue(ctx, addrs); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for i, a := range addrs {
|
||||
state := []string{db.IPDone, db.IPFailed, db.IPOccupied}[i%3]
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, fip_id=? WHERE ip_address=?`, state, fmt.Sprintf("fip-old-%d", i), a); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
// One address really holds a floating IP.
|
||||
mock.Seed("fip-live", "1.2.3.4", "svc")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
live, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET sequence=-1 WHERE id=?`, live.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
o.Tick(ctx) // validator-1 takes the live address (lowest sequence) and attaches it
|
||||
if got := portFIPs(t, mock, "port-1"); len(got) != 1 {
|
||||
t.Fatalf("setup: the live address is not attached (%d floating ips on port-1)", len(got))
|
||||
}
|
||||
atomic.StoreInt32(&cos.disassociations, 0)
|
||||
|
||||
start := time.Now()
|
||||
if _, err := o.ClearQueue(ctx); err != nil {
|
||||
t.Fatalf("clear queue: %v", err)
|
||||
}
|
||||
// One call for the live address; the port sweep finds nothing left.
|
||||
if n := atomic.LoadInt32(&cos.disassociations); n != 1 {
|
||||
t.Fatalf("clear queue made %d disassociate calls, want 1 (finished addresses must be skipped)", n)
|
||||
}
|
||||
if got := portFIPs(t, mock, "port-1"); len(got) != 0 {
|
||||
t.Fatalf("the live floating ip is still attached: %+v", got)
|
||||
}
|
||||
if left, _ := d.ListIPs(ctx); len(left) != 0 {
|
||||
t.Fatalf("%d rows left after clear", len(left))
|
||||
}
|
||||
t.Logf("clear of %d rows took %s", finished+1, time.Since(start))
|
||||
}
|
||||
|
||||
// A bulk detach runs in parallel but never above maxParallelDetach at once.
|
||||
func TestDisassociateAllIsBoundedAndParallel(t *testing.T) {
|
||||
o, _, mock := newTestOrchestrator(t, 180)
|
||||
g := &gaugeOS{MockClient: mock, delay: 30 * time.Millisecond}
|
||||
o.OS = g
|
||||
var refs []db.FIPRef
|
||||
for i := 0; i < 40; i++ {
|
||||
id := fmt.Sprintf("fip-%d", i)
|
||||
mock.Seed(id, fmt.Sprintf("10.8.0.%d", i+1), "svc")
|
||||
refs = append(refs, db.FIPRef{IPID: int64(i), IPAddress: id, FIPID: id})
|
||||
}
|
||||
start := time.Now()
|
||||
o.disassociateAll(context.Background(), refs, "test")
|
||||
elapsed := time.Since(start)
|
||||
if peak := atomic.LoadInt32(&g.peak); peak < 2 || peak > maxParallelDetach {
|
||||
t.Fatalf("peak concurrency %d, want between 2 and %d", peak, maxParallelDetach)
|
||||
}
|
||||
if elapsed > 600*time.Millisecond { // 40 x 30 ms sequentially is 1.2 s
|
||||
t.Fatalf("took %s: the detach is not parallel", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
type gaugeOS struct {
|
||||
*openstack.MockClient
|
||||
delay time.Duration
|
||||
cur, peak int32
|
||||
}
|
||||
|
||||
func (g *gaugeOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
n := atomic.AddInt32(&g.cur, 1)
|
||||
for {
|
||||
p := atomic.LoadInt32(&g.peak)
|
||||
if n <= p || atomic.CompareAndSwapInt32(&g.peak, p, n) {
|
||||
break
|
||||
}
|
||||
}
|
||||
time.Sleep(g.delay)
|
||||
atomic.AddInt32(&g.cur, -1)
|
||||
return g.MockClient.DisassociateFloatingIP(ctx, fipID)
|
||||
}
|
||||
|
||||
// Rows that disagree with the queue are repaired by ReconcileValidators (run on every Tick).
|
||||
func TestReconcileRepairsInconsistentValidators(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
_, d, _ := newTestOrchestrator(t, 180)
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v")
|
||||
_ = d.RegisterValidator(ctx, "validator-2", "host", "port-2", "v")
|
||||
_ = d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"})
|
||||
|
||||
finished, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
foreign, _ := d.GetIPByAddress(ctx, "2.2.2.2")
|
||||
// validator-1 still points at a finished address; validator-2 at an
|
||||
// address owned by somebody else (what an older version could leave behind).
|
||||
for _, q := range []string{
|
||||
fmt.Sprintf(`UPDATE ip_queue SET state='done' WHERE id=%d`, finished.ID),
|
||||
fmt.Sprintf(`UPDATE ip_queue SET state='checking', owner_validator_id='validator-1' WHERE id=%d`, foreign.ID),
|
||||
fmt.Sprintf(`UPDATE validators SET state='assigned', current_ip_id=%d WHERE validator_id='validator-1'`, finished.ID),
|
||||
fmt.Sprintf(`UPDATE validators SET state='assigned', current_ip_id=%d WHERE validator_id='validator-2'`, foreign.ID),
|
||||
} {
|
||||
if _, err := d.ExecContext(ctx, q); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
n, err := d.ReconcileValidators(ctx)
|
||||
if err != nil || n != 2 {
|
||||
t.Fatalf("reconcile repaired %d validators (err %v), want 2", n, err)
|
||||
}
|
||||
for _, id := range []string{"validator-1", "validator-2"} {
|
||||
v, _ := d.GetValidator(ctx, id)
|
||||
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("%s: state=%s current_ip=%v, want idle", id, v.State, v.CurrentIPID)
|
||||
}
|
||||
}
|
||||
// A consistent validator is left alone.
|
||||
if n, _ := d.ReconcileValidators(ctx); n != 0 {
|
||||
t.Fatalf("second reconcile repaired %d, want 0", n)
|
||||
}
|
||||
}
|
||||
|
||||
// Randomised run: many validators, random silences and association failures.
|
||||
// After every tick each validator holds at most one address, the validator
|
||||
// and the address agree on who holds what, and no lease is ever reclaimed.
|
||||
func TestRandomFlowKeepsOneAddressPerValidator(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
rng := rand.New(rand.NewSource(42))
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
|
||||
const validators, addresses = 12, 60
|
||||
var addrs []string
|
||||
for i := 0; i < validators; i++ {
|
||||
_ = d.RegisterValidator(ctx, fmt.Sprintf("validator-%d", i+1), "host", fmt.Sprintf("port-%d", i+1), "v")
|
||||
}
|
||||
for i := 0; i < addresses; i++ {
|
||||
a := fmt.Sprintf("10.7.%d.%d", i/200, i%200+1)
|
||||
addrs = append(addrs, a)
|
||||
mock.Seed(fmt.Sprintf("fip-%d", i), a, "svc")
|
||||
}
|
||||
if err := d.SeedQueue(ctx, addrs); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
checkInvariants := func(tick int) {
|
||||
t.Helper()
|
||||
vs, _ := d.ListValidators(ctx)
|
||||
holders := map[int64]string{}
|
||||
for _, v := range vs {
|
||||
if v.CurrentIPID == nil {
|
||||
if v.State == db.ValidatorAssigned {
|
||||
t.Fatalf("tick %d: %s is assigned but holds nothing", tick, v.ValidatorID)
|
||||
}
|
||||
continue
|
||||
}
|
||||
ip, err := d.GetIP(ctx, *v.CurrentIPID)
|
||||
if err != nil {
|
||||
t.Fatalf("tick %d: %s points at a missing address %d", tick, v.ValidatorID, *v.CurrentIPID)
|
||||
}
|
||||
if ip.OwnerValidatorID == nil || *ip.OwnerValidatorID != v.ValidatorID {
|
||||
t.Fatalf("tick %d: %s holds %s but its owner is %v", tick, v.ValidatorID, ip.IPAddress, ip.OwnerValidatorID)
|
||||
}
|
||||
if ip.State == db.IPDone || ip.State == db.IPFailed || ip.State == db.IPOccupied {
|
||||
t.Fatalf("tick %d: %s still holds the finished address %s", tick, v.ValidatorID, ip.IPAddress)
|
||||
}
|
||||
if prev, ok := holders[ip.ID]; ok {
|
||||
t.Fatalf("tick %d: %s is held by both %s and %s", tick, ip.IPAddress, prev, v.ValidatorID)
|
||||
}
|
||||
holders[ip.ID] = v.ValidatorID
|
||||
}
|
||||
// Every owned, unfinished address is the current one of its owner.
|
||||
ips, _ := d.ListIPs(ctx)
|
||||
perOwner := map[string]int{}
|
||||
for _, ip := range ips {
|
||||
if ip.OwnerValidatorID != nil && ip.State != db.IPDone && ip.State != db.IPFailed && ip.State != db.IPOccupied {
|
||||
perOwner[*ip.OwnerValidatorID]++
|
||||
}
|
||||
}
|
||||
for owner, n := range perOwner {
|
||||
if n > 1 {
|
||||
t.Fatalf("tick %d: %s owns %d unfinished addresses", tick, owner, n)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
checkTicks := map[int64]int{} // address id -> ticks spent in checking
|
||||
done := false
|
||||
for tick := 1; tick <= 600 && !done; tick++ {
|
||||
// Occasionally an association fails (409 on a port).
|
||||
if rng.Intn(15) == 0 {
|
||||
mock.AssociateFailures = map[string]error{fmt.Sprintf("fip-%d", rng.Intn(addresses)): fmt.Errorf("409 conflict")}
|
||||
}
|
||||
o.Tick(ctx)
|
||||
|
||||
vs, _ := d.ListValidators(ctx)
|
||||
for _, v := range vs {
|
||||
// A busy agent sometimes goes silent, then speaks again.
|
||||
if v.CurrentIPID != nil && v.State == db.ValidatorAssigned && rng.Intn(10) == 0 {
|
||||
_ = d.MarkValidatorUnreachable(ctx, v.ValidatorID)
|
||||
}
|
||||
if v.State == db.ValidatorUnreachable && rng.Intn(3) == 0 {
|
||||
_ = d.Heartbeat(ctx, v.ValidatorID)
|
||||
}
|
||||
item, _, err := o.AssignmentForValidator(ctx, v.ValidatorID)
|
||||
if err != nil || item == nil {
|
||||
continue
|
||||
}
|
||||
if item.State == db.IPAwaitingSelfCheck {
|
||||
_ = o.SelfCheckResult(ctx, v.ValidatorID, item.ID, true, "ok")
|
||||
continue
|
||||
}
|
||||
checkTicks[item.ID]++
|
||||
if checkTicks[item.ID] >= 1+rng.Intn(4) {
|
||||
finishChecks(t, o, item, v.ValidatorID)
|
||||
delete(checkTicks, item.ID)
|
||||
}
|
||||
}
|
||||
checkInvariants(tick)
|
||||
|
||||
counts, _, _ := d.CountIPsByState(ctx)
|
||||
unfinished := 0
|
||||
for st, n := range counts {
|
||||
if st != db.IPDone && st != db.IPFailed && st != db.IPOccupied {
|
||||
unfinished += n
|
||||
}
|
||||
}
|
||||
done = unfinished == 0
|
||||
}
|
||||
if !done {
|
||||
counts, _, _ := d.CountIPsByState(ctx)
|
||||
t.Fatalf("not all addresses finished: %v", counts)
|
||||
}
|
||||
var expired int
|
||||
if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM events WHERE event_type='lease_expired'`).Scan(&expired); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if expired != 0 {
|
||||
t.Fatalf("%d leases were reclaimed, want 0", expired)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user