Keep one address per validator; fix heartbeat handling and queue clear
A mass check on 2026-10-02 stalled 7 of 20 validators and sent 42 addresses to fail without a single check. A validator busy with slow checks went silent, was marked unreachable, and its next heartbeat put it back to idle while it still held the address; it was handed a second one, whose association never ran (the in-flight guard was keyed by validator), and both waited for their leases to expire. - Heartbeat/re-register return an unreachable validator to assigned when it still holds an address, else idle. - A validator is released only from the address it currently holds (ReleaseFIP, RequeueOrFail, MarkFIPOccupied, FreeValidator); an unreachable validator stays unreachable until its next heartbeat, so a dead validator is no longer handed a new address every lease period. - ClaimNextQueued refuses a validator that still has an address; a ReconcileValidators pass on every tick repairs rows that disagree with the queue. - Association guard is keyed by address, not validator. - The agent sends heartbeats from their own goroutine. - Clear queue / delete: detach only floating IPs of unfinished rows (done, failed and occupied rows kept their fip_id and made a clear issue >1000 sequential cloud calls: 256 s), at most 8 in parallel; the operation no longer dies with the client connection (10 minute limit). Includes the incident analysis and the plan under analysis/ and docs/changes/, and rebuilt bin/control-api and bin/validator-agent. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
cf4a883363
commit
0532baff09
17 files changed
+1293
-62
No files matched your search
@@ -1,17 +1,32 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/orchestrator"
|
||||
)
|
||||
|
||||
// destructiveOpTimeout bounds a cancel, delete or clear. Such an operation
|
||||
// talks to the cloud as well as the database, and abandoning it halfway
|
||||
// leaves floating IPs detached from rows that still exist (or the reverse),
|
||||
// so it must not die with the client connection: a client that gives up
|
||||
// (a dashboard or curl timeout) only stops waiting for the answer.
|
||||
const destructiveOpTimeout = 10 * time.Minute
|
||||
|
||||
// detachedContext returns a context that ignores cancellation of the request
|
||||
// but keeps its values, with destructiveOpTimeout as the upper bound.
|
||||
func detachedContext(r *http.Request) (context.Context, context.CancelFunc) {
|
||||
return context.WithTimeout(context.WithoutCancel(r.Context()), destructiveOpTimeout)
|
||||
}
|
||||
|
||||
func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
@@ -273,7 +288,9 @@ func toScanStatusDTO(st orchestrator.ScanStatus) scanStatusDTO {
|
||||
// need to be disassociated in OpenStack.
|
||||
func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
|
||||
address := r.PathValue("ip")
|
||||
if err := s.Orch.ForceCancel(r.Context(), address); err != nil {
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
if err := s.Orch.ForceCancel(ctx, address); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
@@ -286,7 +303,9 @@ func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) {
|
||||
// associated floating IP needs disassociating first.
|
||||
func (s *Server) handleAdminDeleteIP(w http.ResponseWriter, r *http.Request) {
|
||||
address := r.PathValue("ip")
|
||||
if err := s.Orch.DeleteIP(r.Context(), address); err != nil {
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
if err := s.Orch.DeleteIP(ctx, address); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
@@ -306,7 +325,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusBadRequest, "addresses must not be empty")
|
||||
return
|
||||
}
|
||||
result, err := s.Orch.DeleteIPs(r.Context(), req.Addresses)
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
result, err := s.Orch.DeleteIPs(ctx, req.Addresses)
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
@@ -320,7 +341,9 @@ func (s *Server) handleAdminDeleteIPs(w http.ResponseWriter, r *http.Request) {
|
||||
// handleAdminClearQueue permanently removes every address currently in the
|
||||
// queue, including those actively being checked.
|
||||
func (s *Server) handleAdminClearQueue(w http.ResponseWriter, r *http.Request) {
|
||||
result, err := s.Orch.ClearQueue(r.Context())
|
||||
ctx, cancel := detachedContext(r)
|
||||
defer cancel()
|
||||
result, err := s.Orch.ClearQueue(ctx)
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
// slowDetachOS delays every Disassociate so the client can give up first.
|
||||
type slowDetachOS struct {
|
||||
*openstack.MockClient
|
||||
delay time.Duration
|
||||
}
|
||||
|
||||
func (s slowDetachOS) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
time.Sleep(s.delay)
|
||||
return s.MockClient.DisassociateFloatingIP(ctx, fipID)
|
||||
}
|
||||
|
||||
// A client that gives up (a dashboard or curl timeout) must not abort
|
||||
// "clear queue" halfway: on 2026-10-02 the request context was cancelled after
|
||||
// part of the floating IPs were detached, the database was left untouched and
|
||||
// the checks kept running.
|
||||
func TestClearQueueSurvivesClientDisconnect(t *testing.T) {
|
||||
fc, d, orch, mock := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host", "port-1", "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
orch.Tick(ctx) // validator-1 takes 1.2.3.4 and attaches its floating IP
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
||||
t.Fatalf("setup: floating ip not attached (port %q)", fip.PortID)
|
||||
}
|
||||
orch.OS = slowDetachOS{MockClient: mock, delay: 400 * time.Millisecond}
|
||||
|
||||
reqCtx, cancel := context.WithTimeout(ctx, 100*time.Millisecond)
|
||||
defer cancel()
|
||||
// No body: the server only notices that the client has left (and cancels
|
||||
// the request context) when it is not waiting for a request body.
|
||||
req, _ := http.NewRequestWithContext(reqCtx, http.MethodPost, fc.base+"/api/v1/admin/ips/clear", nil)
|
||||
if resp, err := fc.client.Do(req); err == nil {
|
||||
resp.Body.Close()
|
||||
t.Fatalf("the client was expected to time out, got status %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for {
|
||||
ips, _ := d.ListIPs(ctx)
|
||||
fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4")
|
||||
if len(ips) == 0 && fip.PortID == "" {
|
||||
return
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("clear did not finish after the client left: %d rows, floating ip on port %q", len(ips), fip.PortID)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user