Retry a failed self-check on another validator; add the self-check failure ceiling

A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-04 09:50:12 +03:00
1 parent b7669c9e41
commit e95b5eb7d5
34 files changed
+1092 -82

No files matched your search

+24 -3
View File
@@ -1,6 +1,7 @@
package httpapi
import (
"fmt"
"net/http"
"strconv"
@@ -203,6 +204,7 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: settings.FIPSettleSeconds,
HistoryRetentionCycles: settings.HistoryRetentionCycles,
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
})
}
@@ -211,11 +213,17 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
// needs cross-field validation against the static lease_ttl_seconds/
// self_check_timeout_seconds config, which only the orchestrator has.
func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *http.Request) {
var req orchestratorSettingsDTO
var req putOrchestratorSettingsRequest
if err := readJSON(r, &req); err != nil {
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
return
}
// Checked before anything is saved, so a bad ceiling does not leave the
// other two settings half-applied. An omitted field keeps the current value.
if a := req.SelfCheckMaxAttempts; a != nil && (*a < db.MinSelfCheckMaxAttempts || *a > db.MaxSelfCheckMaxAttempts) {
writeError(w, http.StatusBadRequest, fmt.Sprintf("self_check_max_attempts must be in %d..%d", db.MinSelfCheckMaxAttempts, db.MaxSelfCheckMaxAttempts))
return
}
if err := s.Orch.SetFIPSettleSeconds(r.Context(), req.FIPSettleSeconds); err != nil {
writeDBError(w, err)
return
@@ -224,9 +232,22 @@ func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *h
writeDBError(w, err)
return
}
if req.SelfCheckMaxAttempts != nil {
if err := s.DB.SetSelfCheckMaxAttempts(r.Context(), *req.SelfCheckMaxAttempts); err != nil {
writeDBError(w, err)
return
}
}
// Answer with what is stored now, so an omitted ceiling shows its current value.
settings, err := s.DB.GetSettings(r.Context())
if err != nil {
writeDBError(w, err)
return
}
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: req.FIPSettleSeconds,
HistoryRetentionCycles: req.HistoryRetentionCycles,
FIPSettleSeconds: settings.FIPSettleSeconds,
HistoryRetentionCycles: settings.HistoryRetentionCycles,
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
})
}