A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42 addresses to fail without a single check. A validator busy with slow checks went silent, was marked unreachable, and its next heartbeat put it back to idle while it still held the address; it was handed a second one, whose association never ran (the in-flight guard was keyed by validator), and both waited for their leases to expire. - Heartbeat/re-register return an unreachable validator to assigned when it still holds an address, else idle. - A validator is released only from the address it currently holds (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an unreachable validator stays unreachable until its next heartbeat, so a dead validator is no longer handed a new address every lease period. - ClaimNextQueued refuses a validator that still has an address; a ReconcileValidators pass on every tick repairs rows that disagree with the queue. - Association guard is keyed by address, not validator. - The agent sends heartbeats from their own goroutine. - Clear queue / delete: detach only floating IPs of unfinished rows (done, failed and occupied rows kept their fip_id and made a clear issue >1000 sequential cloud calls: 256 s), at most 8 in parallel; the operation no longer dies with the client connection (10 minute limit). Includes the incident analysis and the plan under analysis/ and docs/changes/, and rebuilt bin/control-api and bin/validator-agent. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
66 lines
2.2 KiB
Go
66 lines
2.2 KiB
Go
package httpapi
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"testing"
|
|
"time"
|
|
|
|
"cloudipvalidator/internal/openstack"
|
|
)
|
|
|
|
// slowDetachOS delays every Disassociate so the client can give up first.
|
|
type slowDetachOS struct {
|
|
*openstack.MockClient
|
|
delay time.Duration
|
|
}
|
|
|
|
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
|
time.Sleep(s.delay)
|
|
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
|
|
}
|
|
|
|
// A client that gives up (a dashboard or curl timeout) must not abort
|
|
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
|
|
// part of the floating IPs were detached, the database was left untouched and
|
|
// the checks kept running.
|
|
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
|
|
fc, d, orch, mock := newConfigTestHarness(t)
|
|
ctx := context.Background()
|
|
mock.Seed("fip-1", "1.2.3.4", "svc")
|
|
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
|
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
|
|
}
|
|
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
|
|
|
|
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
|
|
defer cancel()
|
|
// No body: the server only notices that the client has left (and cancels
|
|
// the request context) when it is not waiting for a request body.
|
|
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
|
|
if resp, err := fc.client.Do(req); err == nil {
|
|
resp.Body.Close()
|
|
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
|
|
}
|
|
|
|
deadline := time.Now().Add(5 * time.Second)
|
|
for {
|
|
ips, _ := d.ListIPs(ctx)
|
|
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
|
|
if len(ips) == 0 && fip.PortID == "" {
|
|
return
|
|
}
|
|
if time.Now().After(deadline) {
|
|
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
|
|
}
|
|
time.Sleep(50 * time.Millisecond)
|
|
}
|
|
}
|