Retry a failed self-check on another validator; add the self-check failure ceiling

A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-04 09:50:12 +03:00
1 parent b7669c9e41
commit e95b5eb7d5
34 files changed
+1092 -82

No files matched your search

+41 -12
View File
@@ -14,6 +14,7 @@ import (
"errors"
"fmt"
"log/slog"
"strings"
"sync"
"time"
@@ -243,9 +244,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
return nil
}
// Detach the floating IP before the address goes back to the queue.
// requeueOrFail frees the validator in the database but knows nothing
// about the cloud: a floating IP left on the validator's port makes
// every later association on that port fail with 409 ("fixed IP
// The database side (db.FailSelfCheck) frees the validator but knows
// nothing about the cloud: a floating IP left on the validator's port
// makes every later association on that port fail with 409 ("fixed IP
// already has a floating IP"). Best-effort, like the other release
// paths — the database state must be freed even if Neutron hiccups.
if item.FIPID != "" {
@@ -253,20 +254,48 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
o.Log.Error("disassociate fip after failed self-check", "ip_id", ipID, "fip_id", item.FIPID, "err", err)
}
}
if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries {
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail)
return nil
}
// Retry association without fully requeuing: re-drive the same
// claim by cycling back through requeue/claim keeps the logic in
// one place at the cost of the IP briefly returning to `queued`.
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail)
return nil
return o.failSelfCheck(ctx, item, validatorID, detail)
}
return o.DB.SetChecking(ctx, ipID, o.leaseTTL())
}
// failSelfCheck records a failed self-check and decides what happens to the
// address (see db.FailSelfCheck): the ceiling self_check_max_attempts is read
// from the settings on every failure, so a change applies to the next one. The
// retry_count / max_retries pair is not involved: a failed self-check has its
// own ceiling, and the validator that failed it is not handed this address
// again in the current round (db.ClaimNextQueued). The validator itself stays
// in service.
func (o *Orchestrator) failSelfCheck(ctx context.Context, item *db.IPQueueItem, validatorID, detail string) error {
settings, err := o.DB.GetSettings(ctx)
if err != nil {
return err
}
res, err := o.DB.FailSelfCheck(ctx, item.ID, validatorID, detail, settings.SelfCheckMaxAttempts)
if errors.Is(err, db.ErrInvalidState) {
o.Log.Warn("ignoring failed self-check for an address the validator does not hold",
"validator", validatorID, "ip_id", item.ID)
return nil
}
if err != nil {
return err
}
if res.Failed {
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
fmt.Sprintf(`{"reason":%q}`, fmt.Sprintf("self-check failed %d times (on %s), giving up: %s",
res.Failures, strings.Join(res.Validators, ", "), detail)))
return nil
}
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
fmt.Sprintf(`{"reason":%q}`, "self-check failed, retrying: "+detail))
if !res.NewRound {
o.event(ctx, "control-api", "", &item.ID, "validator_excluded",
fmt.Sprintf(`{"validator_id":%q,"failures":%d}`, validatorID, res.Failures))
}
return nil
}
// AssignmentForValidator returns the check config for a validator's current
// IP if it's ready to be worked on (awaiting_self_check or checking),
// or nil if the validator has nothing to do right now. The check config is