Keep one address per validator; fix heartbeat handling and queue clear

A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42
addresses to fail without a single check. A validator busy with slow
checks went silent, was marked unreachable, and its next heartbeat put it
back to idle while it still held the address; it was handed a second one,
whose association never ran (the in-flight guard was keyed by validator),
and both waited for their leases to expire.

- Heartbeat/re-register return an unreachable validator to assigned when
  it still holds an address, else idle.
- A validator is released only from the address it currently holds
  (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an
  unreachable validator stays unreachable until its next heartbeat, so a
  dead validator is no longer handed a new address every lease period.
- ClaimNextQueued refuses a validator that still has an address; a
  ReconcileValidators pass on every tick repairs rows that disagree with
  the queue.
- Association guard is keyed by address, not validator.
- The agent sends heartbeats from their own goroutine.
- Clear queue / delete: detach only floating IPs of unfinished rows (done,
  failed and occupied rows kept their fip_id and made a clear issue >1000
  sequential cloud calls: 256 s), at most 8 in parallel; the operation no
  longer dies with the client connection (10 minute limit).

Includes the incident analysis and the plan under analysis/ and
docs/changes/, and rebuilt bin/control-api and bin/validator-agent.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-02 14:42:12 +03:00
1 parent cf4a883363
commit 0532baff09
17 files changed
+1293 -62

No files matched your search

+27 -4
View File
@@ -1,17 +1,32 @@
package httpapi
import (
"context"
"errors"
"fmt"
"net/http"
"net/url"
"strconv"
"strings"
"time"
"cloudipvalidator/internal/db"
"cloudipvalidator/internal/orchestrator"
)
// destructiveOpTimeout bounds a cancel, delete or clear. Such an operation
// talks to the cloud as well as the database, and abandoning it halfway
// leaves floating IPs detached from rows that still exist (or the reverse),
// so it must not die with the client connection: a client that gives up
// (a dashboard or curl timeout) only stops waiting for the answer.
const destructiveOpTimeout = 10 * time.Minute
// detachedContext returns a context that ignores cancellation of the request
// but keeps its values, with destructiveOpTimeout as the upper bound.
func detachedContext(r *http.Request) (context.Context, context.CancelFunc) {
return context.WithTimeout(context.WithoutCancel(r.Context()), destructiveOpTimeout)
}
func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusOK, okResponse{OK: true})
}
@@ -273,7 +288,9 @@ func toScanStatusDTO(st orchestrator.ScanStatus) scanStatusDTO {
// need to be disassociated in OpenStack.
func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
address := r.PathValue("ip")
if err := s.Orch.ForceCancel(r.Context(), address); err != nil {
ctx, cancel := detachedContext(r)
defer cancel()
if err := s.Orch.ForceCancel(ctx, address); err != nil {
writeDBError(w, err)
return
}
@@ -286,7 +303,9 @@ func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
// associated floating IP needs disassociating first.
func (s *Server) handleAdminDeleteIP(w http.ResponseWriter, r *http.Request) {
address := r.PathValue("ip")
if err := s.Orch.DeleteIP(r.Context(), address); err != nil {
ctx, cancel := detachedContext(r)
defer cancel()
if err := s.Orch.DeleteIP(ctx, address); err != nil {
writeDBError(w, err)
return
}
@@ -306,7 +325,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusBadRequest, "addresses must not be empty")
return
}
result, err := s.Orch.DeleteIPs(r.Context(), req.Addresses)
ctx, cancel := detachedContext(r)
defer cancel()
result, err := s.Orch.DeleteIPs(ctx, req.Addresses)
if err != nil {
writeDBError(w, err)
return
@@ -320,7 +341,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
// handleAdminClearQueue permanently removes every address currently in the
// queue, including those actively being checked.
func (s *Server) handleAdminClearQueue(w http.ResponseWriter, r *http.Request) {
result, err := s.Orch.ClearQueue(r.Context())
ctx, cancel := detachedContext(r)
defer cancel()
result, err := s.Orch.ClearQueue(ctx)
if err != nil {
writeDBError(w, err)
return
@@ -0,0 +1,65 @@
package httpapi
import (
"context"
"net/http"
"testing"
"time"
"cloudipvalidator/internal/openstack"
)
// slowDetachOS delays every Disassociate so the client can give up first.
type slowDetachOS struct {
*openstack.MockClient
delay time.Duration
}
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
time.Sleep(s.delay)
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
}
// A client that gives up (a dashboard or curl timeout) must not abort
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
// part of the floating IPs were detached, the database was left untouched and
// the checks kept running.
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
fc, d, orch, mock := newConfigTestHarness(t)
ctx := context.Background()
mock.Seed("fip-1", "1.2.3.4", "svc")
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
t.Fatal(err)
}
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
t.Fatal(err)
}
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
}
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
defer cancel()
// No body: the server only notices that the client has left (and cancels
// the request context) when it is not waiting for a request body.
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
if resp, err := fc.client.Do(req); err == nil {
resp.Body.Close()
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
}
deadline := time.Now().Add(5 * time.Second)
for {
ips, _ := d.ListIPs(ctx)
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
if len(ips) == 0 && fip.PortID == "" {
return
}
if time.Now().After(deadline) {
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
}
time.Sleep(50 * time.Millisecond)
}
}