Files
cloud-ip-validator/internal/httpapi/handlers_detached_test.go
T
ayurishchevandClaude Sonnet 5.5 0532baff09 Keep one address per validator; fix heartbeat handling and queue clear
A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42
addresses to fail without a single check. A validator busy with slow
checks went silent, was marked unreachable, and its next heartbeat put it
back to idle while it still held the address; it was handed a second one,
whose association never ran (the in-flight guard was keyed by validator),
and both waited for their leases to expire.

- Heartbeat/re-register return an unreachable validator to assigned when
  it still holds an address, else idle.
- A validator is released only from the address it currently holds
  (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an
  unreachable validator stays unreachable until its next heartbeat, so a
  dead validator is no longer handed a new address every lease period.
- ClaimNextQueued refuses a validator that still has an address; a
  ReconcileValidators pass on every tick repairs rows that disagree with
  the queue.
- Association guard is keyed by address, not validator.
- The agent sends heartbeats from their own goroutine.
- Clear queue / delete: detach only floating IPs of unfinished rows (done,
  failed and occupied rows kept their fip_id and made a clear issue >1000
  sequential cloud calls: 256 s), at most 8 in parallel; the operation no
  longer dies with the client connection (10 minute limit).

Includes the incident analysis and the plan under analysis/ and
docs/changes/, and rebuilt bin/control-api and bin/validator-agent.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-02 14:42:12 +03:00

66 lines
2.2 KiB
Go

package httpapi
import (
"context"
"net/http"
"testing"
"time"
"cloudipvalidator/internal/openstack"
)
// slowDetachOS delays every Disassociate so the client can give up first.
type slowDetachOS struct {
*openstack.MockClient
delay time.Duration
}
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
time.Sleep(s.delay)
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
}
// A client that gives up (a dashboard or curl timeout) must not abort
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
// part of the floating IPs were detached, the database was left untouched and
// the checks kept running.
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
fc, d, orch, mock := newConfigTestHarness(t)
ctx := context.Background()
mock.Seed("fip-1", "1.2.3.4", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
t.Fatal(err)
}
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
t.Fatal(err)
}
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
}
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
defer cancel()
// No body: the server only notices that the client has left (and cancels
// the request context) when it is not waiting for a request body.
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
if resp, err := fc.client.Do(req); err == nil {
resp.Body.Close()
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
}
deadline := time.Now().Add(5 * time.Second)
for {
ips, _ := d.ListIPs(ctx)
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
if len(ips) == 0 && fip.PortID == "" {
return
}
if time.Now().After(deadline) {
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
}
time.Sleep(50 * time.Millisecond)
}
}