Retry a failed self-check on another validator; add the self-check failure ceiling

A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-04 09:50:12 +03:00
1 parent b7669c9e41
commit e95b5eb7d5
34 files changed
+1092 -82

No files matched your search

+194 -14
View File
@@ -44,10 +44,10 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
}
}
if _, err := tx.ExecContext(ctx, `
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(ip_address) DO NOTHING
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
`, addr, i, IPQueued, registryID, cycle, runID, cycle, now, now); err != nil {
return fmt.Errorf("seed %s: %w", addr, err)
}
}
@@ -55,8 +55,11 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
}
// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to
// the given idle validator. It returns (nil, nil) if the validator isn't
// idle or no IP is queued. The DB connection pool is capped at one physical
// the given idle validator, skipping addresses this validator failed the
// self-check of in the current round (see FailSelfCheck): such an address
// stays queued for the other validators and does not block the ones behind
// it. It returns (nil, nil) if the validator isn't idle or no IP is
// claimable for it. The DB connection pool is capped at one physical
// connection (see Open), so this transaction already has exclusive access
// to the database for its duration — no other claim, requeue, or update can
// interleave — which combined with the conditional UPDATEs (checked via
@@ -82,9 +85,12 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
var item IPQueueItem
err = tx.QueryRowContext(ctx, `
SELECT id, ip_address, sequence, attempt_number, retry_count
FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1
`, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
SELECT q.id, q.ip_address, q.sequence, q.attempt_number, q.retry_count
FROM ip_queue q WHERE q.state=? AND NOT EXISTS (
SELECT 1 FROM ip_self_check_failures f
WHERE f.registry_id=q.registry_id AND f.validator_id=? AND f.cycle_id>=q.sc_round_start_cycle)
ORDER BY q.sequence LIMIT 1
`, IPQueued, validatorID).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
if err == sql.ErrNoRows {
return nil, nil
}
@@ -352,6 +358,178 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
return tx.Commit()
}
// SelfCheckFailure is the outcome of FailSelfCheck.
type SelfCheckFailure struct {
// Failures is the number of failed self-checks in the address's current
// series, this one included.
Failures int
// Failed: the ceiling was reached and the address got the verdict fail.
Failed bool
// Validators lists the validators that failed the self-check in this
// series, oldest first (with repeats if one failed it more than once).
Validators []string
// NewRound: the address was queued again, and every working validator had
// already failed it in the round, so the round was reset — the exclusions
// no longer apply (always false when Failed).
NewRound bool
}
// FailSelfCheck handles a failed self-check of the address ipID held by
// validatorID, in one transaction: it records the failure in
// ip_self_check_failures and in ip_queue.sc_failures, then
//
// - if sc_failures reached maxAttempts: marks the address failed with the
// verdict fail;
// - otherwise sends it back to the queue without touching retry_count (a
// failed self-check has its own ceiling, unlike a failed association or
// an expired lease, see RequeueOrFail). The validator is excluded from
// this address for the rest of the round (ClaimNextQueued). If no
// working validator (idle, assigned, checking) is left without a failure
// in the round, a new round starts: the exclusions lapse and the retry
// follows the usual rules, so with fewer validators than maxAttempts the
// address never gets stuck in the queue.
//
// The validator is freed in the same transaction. Returns ErrInvalidState if
// the address is not awaiting_self_check on this validator (a late report).
func (d *DB) FailSelfCheck(ctx context.Context, ipID int64, validatorID, detail string, maxAttempts int) (SelfCheckFailure, error) {
var out SelfCheckFailure
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return out, err
}
defer tx.Rollback()
var registryID, cycle, roundStart int64
var runID sql.NullInt64
var attempt, failures int
err = tx.QueryRowContext(ctx, `
SELECT registry_id, run_id, cycle_id, attempt_number, sc_failures, sc_round_start_cycle
FROM ip_queue WHERE id=? AND state=? AND owner_validator_id=?
`, ipID, IPAwaitingSelfCheck, validatorID).Scan(&registryID, &runID, &cycle, &attempt, &failures, &roundStart)
if err == sql.ErrNoRows {
return out, fmt.Errorf("ip_id %d is not awaiting self-check on %s: %w", ipID, validatorID, ErrInvalidState)
}
if err != nil {
return out, err
}
now := timeToDB(Now())
if _, err := tx.ExecContext(ctx, `
INSERT INTO ip_self_check_failures (registry_id, run_id, cycle_id, attempt_number, validator_id, failed_at, detail)
VALUES (?, ?, ?, ?, ?, ?, ?)
`, registryID, runID, cycle, attempt, validatorID, now, detail); err != nil {
return out, fmt.Errorf("record self-check failure: %w", err)
}
failures++
out.Failures = failures
rows, err := tx.QueryContext(ctx, `
SELECT validator_id FROM (
SELECT id, validator_id FROM ip_self_check_failures WHERE registry_id=? ORDER BY id DESC LIMIT ?
) ORDER BY id
`, registryID, failures)
if err != nil {
return out, err
}
for rows.Next() {
var v string
if err := rows.Scan(&v); err != nil {
rows.Close()
return out, err
}
out.Validators = append(out.Validators, v)
}
if err := rows.Err(); err != nil {
rows.Close()
return out, err
}
rows.Close()
if failures >= maxAttempts {
out.Failed = true
if _, err := tx.ExecContext(ctx, `
UPDATE ip_queue SET
state=?, sc_failures=?, overall_result=?, aggregated_at=?, updated_at=?
WHERE id=?
`, IPFailed, failures, ResultFail, now, now, ipID); err != nil {
return out, err
}
if err := upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now); err != nil {
return out, err
}
if err := finalizeRunsTx(ctx, tx, now); err != nil {
return out, err
}
} else {
newCycle, err := nextRegistryCycleTx(ctx, tx, registryID, now)
if err != nil {
return out, err
}
var left int
if err := tx.QueryRowContext(ctx, `
SELECT COUNT(*) FROM validators v
WHERE v.state IN (?, ?, ?) AND NOT EXISTS (
SELECT 1 FROM ip_self_check_failures f
WHERE f.registry_id=? AND f.validator_id=v.validator_id AND f.cycle_id>=?)
`, ValidatorIdle, ValidatorAssigned, ValidatorChecking, registryID, roundStart).Scan(&left); err != nil {
return out, err
}
if left == 0 {
out.NewRound = true
roundStart = int64(newCycle)
}
if _, err := tx.ExecContext(ctx, `
UPDATE ip_queue SET
state=?, owner_validator_id=NULL, fip_id='', attempt_number=attempt_number+1,
cycle_id=?, sc_failures=?, sc_round_start_cycle=?, lease_expires_at=NULL, egress_complete=0,
overall_result='', assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, updated_at=?
WHERE id=?
`, IPQueued, newCycle, failures, roundStart, now, ipID); err != nil {
return out, err
}
}
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
return out, err
}
return out, tx.Commit()
}
// GetIPRunID returns the check run the queue row belongs to (0 if none).
func (d *DB) GetIPRunID(ctx context.Context, ipID int64) (int64, error) {
var runID sql.NullInt64
if err := d.QueryRowContext(ctx, `SELECT run_id FROM ip_queue WHERE id=?`, ipID).Scan(&runID); err != nil {
return 0, err
}
return runID.Int64, nil
}
// ListSelfCheckFailedOn returns the distinct validators whose self-check of
// the address failed, in alphabetical order: over the whole history of the
// address, or, if runID > 0, only the failures that happened in that run.
func (d *DB) ListSelfCheckFailedOn(ctx context.Context, registryID, runID int64) ([]string, error) {
q := `SELECT DISTINCT validator_id FROM ip_self_check_failures WHERE registry_id=?`
args := []any{registryID}
if runID > 0 {
q += ` AND run_id=?`
args = append(args, runID)
}
rows, err := d.QueryContext(ctx, q+` ORDER BY validator_id`, args...)
if err != nil {
return nil, err
}
defer rows.Close()
out := []string{}
for rows.Next() {
var v string
if err := rows.Scan(&v); err != nil {
return nil, err
}
out = append(out, v)
}
return out, rows.Err()
}
// SubmitIPs is the single admin entry point for both "add new addresses to
// the queue" and "force a re-check of an already-finished address" — the
// same list can freely mix both. Addresses are processed in one
@@ -360,7 +538,9 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
// - unknown address: inserted as a new queued row.
// - address currently done/failed/occupied: reset to queued (new attempt,
// retry_count cleared — this is a deliberate admin-triggered restart,
// not a system retry).
// not a system retry). It also starts a new series of self-check
// failures (sc_failures=0, validators that failed it before are no
// longer excluded); the history in ip_self_check_failures stays.
// - address currently queued (not yet claimed): left in state=queued,
// only its sequence is updated.
// - address currently mid-check (assigning_fip / awaiting_self_check /
@@ -426,9 +606,9 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
return result, rErr
}
if _, err := tx.ExecContext(ctx, `
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
`, addr, seq, IPQueued, registryID, cycle, rid, cycle, now, now); err != nil {
return result, fmt.Errorf("insert %s: %w", addr, err)
}
result.Added = append(result.Added, addr)
@@ -453,10 +633,10 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
UPDATE ip_queue SET
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
overall_result='', run_id=?,
overall_result='', run_id=?, sc_failures=0, sc_round_start_cycle=?,
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
WHERE ip_address=?
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
`, IPQueued, seq, cycle, rid, cycle, now, addr); err != nil {
return result, fmt.Errorf("requeue %s: %w", addr, err)
}
result.Requeued = append(result.Requeued, addr)