Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.
Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b7669c9e41
commit
e95b5eb7d5
34 files changed
+1092
-82
No files matched your search
@@ -162,12 +162,21 @@ type putCheckTypeRequest struct {
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
// orchestratorSettingsDTO doubles as both the GET response and the PUT
|
||||
// request body for /api/v1/admin/config/orchestrator — a single-field DTO,
|
||||
// same shape both ways, like putSiteRequest/siteDTO.
|
||||
// orchestratorSettingsDTO is the GET response for
|
||||
// /api/v1/admin/config/orchestrator.
|
||||
type orchestratorSettingsDTO struct {
|
||||
FIPSettleSeconds int `json:"fip_settle_seconds"`
|
||||
HistoryRetentionCycles int `json:"history_retention_cycles"`
|
||||
SelfCheckMaxAttempts int `json:"self_check_max_attempts"`
|
||||
}
|
||||
|
||||
// putOrchestratorSettingsRequest is the PUT body. SelfCheckMaxAttempts is a
|
||||
// pointer so that a client written before the field existed (it sends only
|
||||
// the first two) leaves the ceiling unchanged instead of failing validation.
|
||||
type putOrchestratorSettingsRequest struct {
|
||||
FIPSettleSeconds int `json:"fip_settle_seconds"`
|
||||
HistoryRetentionCycles int `json:"history_retention_cycles"`
|
||||
SelfCheckMaxAttempts *int `json:"self_check_max_attempts"`
|
||||
}
|
||||
|
||||
// inboundChecksDTO doubles as both the GET response and the PUT request
|
||||
|
||||
@@ -178,11 +178,23 @@ func (s *Server) handleAdminIPDetail(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
// Self-check failures of this address in its current run.
|
||||
runID, err := s.DB.GetIPRunID(r.Context(), item.ID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), item.RegistryID, runID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, struct {
|
||||
IP *db.IPQueueItem `json:"ip"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
Events []db.Event `json:"events"`
|
||||
}{item, checks, events})
|
||||
IP *db.IPQueueItem `json:"ip"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
Events []db.Event `json:"events"`
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}{item, checks, events, failedOn})
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strconv"
|
||||
|
||||
@@ -203,6 +204,7 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: settings.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: settings.HistoryRetentionCycles,
|
||||
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -211,11 +213,17 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
// needs cross-field validation against the static lease_ttl_seconds/
|
||||
// self_check_timeout_seconds config, which only the orchestrator has.
|
||||
func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *http.Request) {
|
||||
var req orchestratorSettingsDTO
|
||||
var req putOrchestratorSettingsRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
// Checked before anything is saved, so a bad ceiling does not leave the
|
||||
// other two settings half-applied. An omitted field keeps the current value.
|
||||
if a := req.SelfCheckMaxAttempts; a != nil && (*a < db.MinSelfCheckMaxAttempts || *a > db.MaxSelfCheckMaxAttempts) {
|
||||
writeError(w, http.StatusBadRequest, fmt.Sprintf("self_check_max_attempts must be in %d..%d", db.MinSelfCheckMaxAttempts, db.MaxSelfCheckMaxAttempts))
|
||||
return
|
||||
}
|
||||
if err := s.Orch.SetFIPSettleSeconds(r.Context(), req.FIPSettleSeconds); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
@@ -224,9 +232,22 @@ func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
if req.SelfCheckMaxAttempts != nil {
|
||||
if err := s.DB.SetSelfCheckMaxAttempts(r.Context(), *req.SelfCheckMaxAttempts); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
// Answer with what is stored now, so an omitted ceiling shows its current value.
|
||||
settings, err := s.DB.GetSettings(r.Context())
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: req.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: req.HistoryRetentionCycles,
|
||||
FIPSettleSeconds: settings.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: settings.HistoryRetentionCycles,
|
||||
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -426,19 +426,32 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal get response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 0 {
|
||||
t.Fatalf("expected default 0, got %+v", got)
|
||||
if got.FIPSettleSeconds != 0 || got.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected defaults 0 and 5, got %+v", got)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 20})
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal put response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 20 {
|
||||
t.Fatalf("expected 20, got %+v", got)
|
||||
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected 20 with the omitted ceiling kept at 5, got %+v", got)
|
||||
}
|
||||
|
||||
eight := 8
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &eight})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings with ceiling: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
for _, bad := range []int{0, 51} {
|
||||
bad := bad
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &bad})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for self_check_max_attempts=%d, status=%d body=%s", bad, resp.StatusCode, body)
|
||||
}
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/config/orchestrator", nil)
|
||||
@@ -448,8 +461,8 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal get-after-put response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 20 {
|
||||
t.Fatalf("expected 20 to persist, got %+v", got)
|
||||
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 8 {
|
||||
t.Fatalf("expected 20 and 8 to persist, got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -459,12 +472,12 @@ func TestOrchestratorSettingsPutValidation(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
// newConfigTestHarness: LeaseTTLSeconds=180, SelfCheckTimeoutSeconds=10.
|
||||
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 175})
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 175})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for settle seconds too close to lease ttl, status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: -1})
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: -1})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for negative settle seconds, status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
@@ -479,7 +492,7 @@ func TestFIPSettleDelayGatesAssignmentEndpoint(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "9.9.9.9", "svc-project")
|
||||
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 1})
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 1})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
@@ -81,10 +81,17 @@ func (s *Server) handleAdminRegistryHistory(w http.ResponseWriter, r *http.Reque
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
// Self-check failures over every run of the address.
|
||||
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), summary.ID, 0)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, struct {
|
||||
Registry registryDTO `json:"registry"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
}{registrySummaryToDTO(*summary), checks})
|
||||
Registry registryDTO `json:"registry"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}{registrySummaryToDTO(*summary), checks, failedOn})
|
||||
}
|
||||
|
||||
func registrySummaryToDTO(s db.RegistrySummary) registryDTO {
|
||||
|
||||
@@ -219,3 +219,51 @@ func TestAdminRegistryLevels(t *testing.T) {
|
||||
t.Errorf("empty level must serialise by_type as [], got %s", body)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSelfCheckFailedOnInAddressEndpoints proves GET /admin/ips/{ip} and GET
|
||||
// /admin/registry/{ip} list the validators whose self-check of the address
|
||||
// failed ([] when none).
|
||||
func TestSelfCheckFailedOnInAddressEndpoints(t *testing.T) {
|
||||
fc, d, orch, mock := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "9.9.9.9", "svc-project")
|
||||
fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}})
|
||||
|
||||
failedOn := func(path string) []string {
|
||||
t.Helper()
|
||||
resp, body := fc.do(http.MethodGet, path, nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get %s: status=%d body=%s", path, resp.StatusCode, body)
|
||||
}
|
||||
var got struct {
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal %s: %v", path, err)
|
||||
}
|
||||
if got.SelfCheckFailedOn == nil {
|
||||
t.Fatalf("%s: self_check_failed_on must be [] rather than null, body=%s", path, body)
|
||||
}
|
||||
return got.SelfCheckFailedOn
|
||||
}
|
||||
if got := failedOn("/api/v1/admin/ips/9.9.9.9"); len(got) != 0 {
|
||||
t.Fatalf("expected no failures yet, got %v", got)
|
||||
}
|
||||
|
||||
orch.Tick(ctx) // claim + associate -> awaiting_self_check
|
||||
ip, err := d.GetIPByAddress(ctx, "9.9.9.9")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := orch.SelfCheckResult(ctx, "validator-1", ip.ID, false, "ip echo timeout"); err != nil {
|
||||
t.Fatalf("self-check result: %v", err)
|
||||
}
|
||||
|
||||
for _, path := range []string{"/api/v1/admin/ips/9.9.9.9", "/api/v1/admin/registry/9.9.9.9"} {
|
||||
if got := failedOn(path); !reflect.DeepEqual(got, []string{"validator-1"}) {
|
||||
t.Fatalf("%s: expected [validator-1], got %v", path, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user