Retry a failed self-check on another validator; add the self-check failure ceiling

A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-04 09:50:12 +03:00
1 parent b7669c9e41
commit e95b5eb7d5
34 files changed
+1092 -82

No files matched your search

+15 -2
View File
@@ -37,6 +37,10 @@ type fakeControlAPI struct {
inboundICMP bool
historyRetentionCycles int
selfCheckMaxAttempts int
// selfCheckFailedOn is served as self_check_failed_on of the address
// detail and registry history endpoints.
selfCheckFailedOn []string
// Analytics: the run selector, the report JSON per run, the lists per
// "run/kind[/class]" and the subnet list of /settings.
@@ -94,6 +98,8 @@ func newFakeControlAPI(t *testing.T) (*fakeControlAPI, string) {
registry: map[string]registryItem{},
registryChecks: map[string][]check{},
autoCycle: autoCycleDTO{IntervalSeconds: 3600, Phase: "idle"},
selfCheckMaxAttempts: 5,
}
ts := httptest.NewServer(f.handler())
t.Cleanup(ts.Close)
@@ -191,7 +197,7 @@ func (f *fakeControlAPI) handler() http.Handler {
addr := r.PathValue("ip")
for _, ip := range f.ips {
if ip.IPAddress == addr {
writeJSON(w, http.StatusOK, ipDetailResponse{IP: ip, Checks: []check{}, Events: []event{}})
writeJSON(w, http.StatusOK, ipDetailResponse{IP: ip, Checks: []check{}, Events: []event{}, SelfCheckFailedOn: f.selfCheckFailedOn})
return
}
}
@@ -310,6 +316,7 @@ func (f *fakeControlAPI) handler() http.Handler {
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: f.fipSettleSeconds,
HistoryRetentionCycles: f.historyRetentionCycles,
SelfCheckMaxAttempts: f.selfCheckMaxAttempts,
})
})
mux.HandleFunc("PUT /api/v1/admin/config/orchestrator", func(w http.ResponseWriter, r *http.Request) {
@@ -325,11 +332,17 @@ func (f *fakeControlAPI) handler() http.Handler {
writeAPIErr(w, http.StatusBadRequest, "history_retention_cycles must be >= 0")
return
}
if req.SelfCheckMaxAttempts < 1 || req.SelfCheckMaxAttempts > 50 {
writeAPIErr(w, http.StatusBadRequest, "self_check_max_attempts must be in 1..50")
return
}
f.fipSettleSeconds = req.FIPSettleSeconds
f.historyRetentionCycles = req.HistoryRetentionCycles
f.selfCheckMaxAttempts = req.SelfCheckMaxAttempts
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
FIPSettleSeconds: f.fipSettleSeconds,
HistoryRetentionCycles: f.historyRetentionCycles,
SelfCheckMaxAttempts: f.selfCheckMaxAttempts,
})
})
@@ -474,7 +487,7 @@ func (f *fakeControlAPI) handler() http.Handler {
writeAPIErr(w, http.StatusNotFound, "unknown ip: "+addr)
return
}
writeJSON(w, http.StatusOK, registryHistoryResponse{Registry: item, Checks: f.registryChecks[addr]})
writeJSON(w, http.StatusOK, registryHistoryResponse{Registry: item, Checks: f.registryChecks[addr], SelfCheckFailedOn: f.selfCheckFailedOn})
})
mux.HandleFunc("GET /api/v1/admin/config/subnets", func(w http.ResponseWriter, r *http.Request) {