Retry a failed self-check on another validator; add the self-check failure ceiling

A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-04 09:50:12 +03:00
1 parent b7669c9e41
commit e95b5eb7d5
34 files changed
+1092 -82

No files matched your search

+12 -3
View File
@@ -162,12 +162,21 @@ type putCheckTypeRequest struct {
Targets []string `json:"targets"`
}
// orchestratorSettingsDTO doubles as both the GET response and the PUT
// request body for /api/v1/admin/config/orchestrator — a single-field DTO,
// same shape both ways, like putSiteRequest/siteDTO.
// orchestratorSettingsDTO is the GET response for
// /api/v1/admin/config/orchestrator.
type orchestratorSettingsDTO struct {
FIPSettleSeconds int `json:"fip_settle_seconds"`
HistoryRetentionCycles int `json:"history_retention_cycles"`
SelfCheckMaxAttempts int `json:"self_check_max_attempts"`
}
// putOrchestratorSettingsRequest is the PUT body. SelfCheckMaxAttempts is a
// pointer so that a client written before the field existed (it sends only
// the first two) leaves the ceiling unchanged instead of failing validation.
type putOrchestratorSettingsRequest struct {
FIPSettleSeconds int `json:"fip_settle_seconds"`
HistoryRetentionCycles int `json:"history_retention_cycles"`
SelfCheckMaxAttempts *int `json:"self_check_max_attempts"`
}
// inboundChecksDTO doubles as both the GET response and the PUT request
+16 -4
View File
@@ -178,11 +178,23 @@ func (s *Server) handleAdminIPDetail(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusInternalServerError, err.Error())
return
}
// Self-check failures of this address in its current run.
runID, err := s.DB.GetIPRunID(r.Context(), item.ID)
if err != nil {
writeError(w, http.StatusInternalServerError, err.Error())
return
}
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), item.RegistryID, runID)
if err != nil {
writeError(w, http.StatusInternalServerError, err.Error())
return
}
writeJSON(w, http.StatusOK, struct {
IP *db.IPQueueItem `json:"ip"`
Checks []db.Check `json:"checks"`
Events []db.Event `json:"events"`
}{item, checks, events})
IP *db.IPQueueItem `json:"ip"`
Checks []db.Check `json:"checks"`
Events []db.Event `json:"events"`
SelfCheckFailedOn []string `json:"self_check_failed_on"`
}{item, checks, events, failedOn})
}
func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) {
+24 -3
View File
@@ -1,6 +1,7 @@
package httpapi
import (
"fmt"
"net/http"
"strconv"
@@ -203,6 +204,7 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: settings.FIPSettleSeconds,
HistoryRetentionCycles: settings.HistoryRetentionCycles,
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
})
}
@@ -211,11 +213,17 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
// needs cross-field validation against the static lease_ttl_seconds/
// self_check_timeout_seconds config, which only the orchestrator has.
func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *http.Request) {
var req orchestratorSettingsDTO
var req putOrchestratorSettingsRequest
if err := readJSON(r, &req); err != nil {
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
return
}
// Checked before anything is saved, so a bad ceiling does not leave the
// other two settings half-applied. An omitted field keeps the current value.
if a := req.SelfCheckMaxAttempts; a != nil && (*a < db.MinSelfCheckMaxAttempts || *a > db.MaxSelfCheckMaxAttempts) {
writeError(w, http.StatusBadRequest, fmt.Sprintf("self_check_max_attempts must be in %d..%d", db.MinSelfCheckMaxAttempts, db.MaxSelfCheckMaxAttempts))
return
}
if err := s.Orch.SetFIPSettleSeconds(r.Context(), req.FIPSettleSeconds); err != nil {
writeDBError(w, err)
return
@@ -224,9 +232,22 @@ func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *h
writeDBError(w, err)
return
}
if req.SelfCheckMaxAttempts != nil {
if err := s.DB.SetSelfCheckMaxAttempts(r.Context(), *req.SelfCheckMaxAttempts); err != nil {
writeDBError(w, err)
return
}
}
// Answer with what is stored now, so an omitted ceiling shows its current value.
settings, err := s.DB.GetSettings(r.Context())
if err != nil {
writeDBError(w, err)
return
}
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: req.FIPSettleSeconds,
HistoryRetentionCycles: req.HistoryRetentionCycles,
FIPSettleSeconds: settings.FIPSettleSeconds,
HistoryRetentionCycles: settings.HistoryRetentionCycles,
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
})
}
+23 -10
View File
@@ -426,19 +426,32 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
if err := json.Unmarshal(body, &got); err != nil {
t.Fatalf("unmarshal get response: %v", err)
}
if got.FIPSettleSeconds != 0 {
t.Fatalf("expected default 0, got %+v", got)
if got.FIPSettleSeconds != 0 || got.SelfCheckMaxAttempts != 5 {
t.Fatalf("expected defaults 0 and 5, got %+v", got)
}
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 20})
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20})
if resp.StatusCode != http.StatusOK {
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
}
if err := json.Unmarshal(body, &got); err != nil {
t.Fatalf("unmarshal put response: %v", err)
}
if got.FIPSettleSeconds != 20 {
t.Fatalf("expected 20, got %+v", got)
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 5 {
t.Fatalf("expected 20 with the omitted ceiling kept at 5, got %+v", got)
}
eight := 8
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &eight})
if resp.StatusCode != http.StatusOK {
t.Fatalf("put settings with ceiling: status=%d body=%s", resp.StatusCode, body)
}
for _, bad := range []int{0, 51} {
bad := bad
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &bad})
if resp.StatusCode != http.StatusBadRequest {
t.Fatalf("expected 400 for self_check_max_attempts=%d, status=%d body=%s", bad, resp.StatusCode, body)
}
}
resp, body = fc.do(http.MethodGet, "/api/v1/admin/config/orchestrator", nil)
@@ -448,8 +461,8 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
if err := json.Unmarshal(body, &got); err != nil {
t.Fatalf("unmarshal get-after-put response: %v", err)
}
if got.FIPSettleSeconds != 20 {
t.Fatalf("expected 20 to persist, got %+v", got)
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 8 {
t.Fatalf("expected 20 and 8 to persist, got %+v", got)
}
}
@@ -459,12 +472,12 @@ func TestOrchestratorSettingsPutValidation(t *testing.T) {
fc, _, _, _ := newConfigTestHarness(t)
// newConfigTestHarness: LeaseTTLSeconds=180, SelfCheckTimeoutSeconds=10.
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 175})
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 175})
if resp.StatusCode != http.StatusBadRequest {
t.Fatalf("expected 400 for settle seconds too close to lease ttl, status=%d body=%s", resp.StatusCode, body)
}
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: -1})
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: -1})
if resp.StatusCode != http.StatusBadRequest {
t.Fatalf("expected 400 for negative settle seconds, status=%d body=%s", resp.StatusCode, body)
}
@@ -479,7 +492,7 @@ func TestFIPSettleDelayGatesAssignmentEndpoint(t *testing.T) {
ctx := context.Background()
mock.Seed("fip-1", "9.9.9.9", "svc-project")
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 1})
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 1})
if resp.StatusCode != http.StatusOK {
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
}
+10 -3
View File
@@ -81,10 +81,17 @@ func (s *Server) handleAdminRegistryHistory(w http.ResponseWriter, r *http.Reque
writeError(w, http.StatusInternalServerError, err.Error())
return
}
// Self-check failures over every run of the address.
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), summary.ID, 0)
if err != nil {
writeError(w, http.StatusInternalServerError, err.Error())
return
}
writeJSON(w, http.StatusOK, struct {
Registry registryDTO `json:"registry"`
Checks []db.Check `json:"checks"`
}{registrySummaryToDTO(*summary), checks})
Registry registryDTO `json:"registry"`
Checks []db.Check `json:"checks"`
SelfCheckFailedOn []string `json:"self_check_failed_on"`
}{registrySummaryToDTO(*summary), checks, failedOn})
}
func registrySummaryToDTO(s db.RegistrySummary) registryDTO {
@@ -219,3 +219,51 @@ func TestAdminRegistryLevels(t *testing.T) {
t.Errorf("empty level must serialise by_type as [], got %s", body)
}
}
// TestSelfCheckFailedOnInAddressEndpoints proves GET /admin/ips/{ip} and GET
// /admin/registry/{ip} list the validators whose self-check of the address
// failed ([] when none).
func TestSelfCheckFailedOnInAddressEndpoints(t *testing.T) {
fc, d, orch, mock := newConfigTestHarness(t)
ctx := context.Background()
mock.Seed("fip-1", "9.9.9.9", "svc-project")
fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"})
fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"})
fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}})
failedOn := func(path string) []string {
t.Helper()
resp, body := fc.do(http.MethodGet, path, nil)
if resp.StatusCode != http.StatusOK {
t.Fatalf("get %s: status=%d body=%s", path, resp.StatusCode, body)
}
var got struct {
SelfCheckFailedOn []string `json:"self_check_failed_on"`
}
if err := json.Unmarshal(body, &got); err != nil {
t.Fatalf("unmarshal %s: %v", path, err)
}
if got.SelfCheckFailedOn == nil {
t.Fatalf("%s: self_check_failed_on must be [] rather than null, body=%s", path, body)
}
return got.SelfCheckFailedOn
}
if got := failedOn("/api/v1/admin/ips/9.9.9.9"); len(got) != 0 {
t.Fatalf("expected no failures yet, got %v", got)
}
orch.Tick(ctx) // claim + associate -> awaiting_self_check
ip, err := d.GetIPByAddress(ctx, "9.9.9.9")
if err != nil {
t.Fatal(err)
}
if err := orch.SelfCheckResult(ctx, "validator-1", ip.ID, false, "ip echo timeout"); err != nil {
t.Fatalf("self-check result: %v", err)
}
for _, path := range []string{"/api/v1/admin/ips/9.9.9.9", "/api/v1/admin/registry/9.9.9.9"} {
if got := failedOn(path); !reflect.DeepEqual(got, []string{"validator-1"}) {
t.Fatalf("%s: expected [validator-1], got %v", path, got)
}
}
}