Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.
Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b7669c9e41
commit
e95b5eb7d5
34 files changed
+1092
-82
No files matched your search
+194
-14
@@ -44,10 +44,10 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_address) DO NOTHING
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, cycle, now, now); err != nil {
|
||||
return fmt.Errorf("seed %s: %w", addr, err)
|
||||
}
|
||||
}
|
||||
@@ -55,8 +55,11 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
|
||||
// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to
|
||||
// the given idle validator. It returns (nil, nil) if the validator isn't
|
||||
// idle or no IP is queued. The DB connection pool is capped at one physical
|
||||
// the given idle validator, skipping addresses this validator failed the
|
||||
// self-check of in the current round (see FailSelfCheck): such an address
|
||||
// stays queued for the other validators and does not block the ones behind
|
||||
// it. It returns (nil, nil) if the validator isn't idle or no IP is
|
||||
// claimable for it. The DB connection pool is capped at one physical
|
||||
// connection (see Open), so this transaction already has exclusive access
|
||||
// to the database for its duration — no other claim, requeue, or update can
|
||||
// interleave — which combined with the conditional UPDATEs (checked via
|
||||
@@ -82,9 +85,12 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
|
||||
|
||||
var item IPQueueItem
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT id, ip_address, sequence, attempt_number, retry_count
|
||||
FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1
|
||||
`, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
SELECT q.id, q.ip_address, q.sequence, q.attempt_number, q.retry_count
|
||||
FROM ip_queue q WHERE q.state=? AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=q.registry_id AND f.validator_id=? AND f.cycle_id>=q.sc_round_start_cycle)
|
||||
ORDER BY q.sequence LIMIT 1
|
||||
`, IPQueued, validatorID).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
@@ -352,6 +358,178 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// SelfCheckFailure is the outcome of FailSelfCheck.
|
||||
type SelfCheckFailure struct {
|
||||
// Failures is the number of failed self-checks in the address's current
|
||||
// series, this one included.
|
||||
Failures int
|
||||
// Failed: the ceiling was reached and the address got the verdict fail.
|
||||
Failed bool
|
||||
// Validators lists the validators that failed the self-check in this
|
||||
// series, oldest first (with repeats if one failed it more than once).
|
||||
Validators []string
|
||||
// NewRound: the address was queued again, and every working validator had
|
||||
// already failed it in the round, so the round was reset — the exclusions
|
||||
// no longer apply (always false when Failed).
|
||||
NewRound bool
|
||||
}
|
||||
|
||||
// FailSelfCheck handles a failed self-check of the address ipID held by
|
||||
// validatorID, in one transaction: it records the failure in
|
||||
// ip_self_check_failures and in ip_queue.sc_failures, then
|
||||
//
|
||||
// - if sc_failures reached maxAttempts: marks the address failed with the
|
||||
// verdict fail;
|
||||
// - otherwise sends it back to the queue without touching retry_count (a
|
||||
// failed self-check has its own ceiling, unlike a failed association or
|
||||
// an expired lease, see RequeueOrFail). The validator is excluded from
|
||||
// this address for the rest of the round (ClaimNextQueued). If no
|
||||
// working validator (idle, assigned, checking) is left without a failure
|
||||
// in the round, a new round starts: the exclusions lapse and the retry
|
||||
// follows the usual rules, so with fewer validators than maxAttempts the
|
||||
// address never gets stuck in the queue.
|
||||
//
|
||||
// The validator is freed in the same transaction. Returns ErrInvalidState if
|
||||
// the address is not awaiting_self_check on this validator (a late report).
|
||||
func (d *DB) FailSelfCheck(ctx context.Context, ipID int64, validatorID, detail string, maxAttempts int) (SelfCheckFailure, error) {
|
||||
var out SelfCheckFailure
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
var registryID, cycle, roundStart int64
|
||||
var runID sql.NullInt64
|
||||
var attempt, failures int
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT registry_id, run_id, cycle_id, attempt_number, sc_failures, sc_round_start_cycle
|
||||
FROM ip_queue WHERE id=? AND state=? AND owner_validator_id=?
|
||||
`, ipID, IPAwaitingSelfCheck, validatorID).Scan(®istryID, &runID, &cycle, &attempt, &failures, &roundStart)
|
||||
if err == sql.ErrNoRows {
|
||||
return out, fmt.Errorf("ip_id %d is not awaiting self-check on %s: %w", ipID, validatorID, ErrInvalidState)
|
||||
}
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_self_check_failures (registry_id, run_id, cycle_id, attempt_number, validator_id, failed_at, detail)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`, registryID, runID, cycle, attempt, validatorID, now, detail); err != nil {
|
||||
return out, fmt.Errorf("record self-check failure: %w", err)
|
||||
}
|
||||
failures++
|
||||
out.Failures = failures
|
||||
|
||||
rows, err := tx.QueryContext(ctx, `
|
||||
SELECT validator_id FROM (
|
||||
SELECT id, validator_id FROM ip_self_check_failures WHERE registry_id=? ORDER BY id DESC LIMIT ?
|
||||
) ORDER BY id
|
||||
`, registryID, failures)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
out.Validators = append(out.Validators, v)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
rows.Close()
|
||||
|
||||
if failures >= maxAttempts {
|
||||
out.Failed = true
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, sc_failures=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPFailed, failures, ResultFail, now, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
} else {
|
||||
newCycle, err := nextRegistryCycleTx(ctx, tx, registryID, now)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
var left int
|
||||
if err := tx.QueryRowContext(ctx, `
|
||||
SELECT COUNT(*) FROM validators v
|
||||
WHERE v.state IN (?, ?, ?) AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=? AND f.validator_id=v.validator_id AND f.cycle_id>=?)
|
||||
`, ValidatorIdle, ValidatorAssigned, ValidatorChecking, registryID, roundStart).Scan(&left); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if left == 0 {
|
||||
out.NewRound = true
|
||||
roundStart = int64(newCycle)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, owner_validator_id=NULL, fip_id='', attempt_number=attempt_number+1,
|
||||
cycle_id=?, sc_failures=?, sc_round_start_cycle=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPQueued, newCycle, failures, roundStart, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
return out, tx.Commit()
|
||||
}
|
||||
|
||||
// GetIPRunID returns the check run the queue row belongs to (0 if none).
|
||||
func (d *DB) GetIPRunID(ctx context.Context, ipID int64) (int64, error) {
|
||||
var runID sql.NullInt64
|
||||
if err := d.QueryRowContext(ctx, `SELECT run_id FROM ip_queue WHERE id=?`, ipID).Scan(&runID); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return runID.Int64, nil
|
||||
}
|
||||
|
||||
// ListSelfCheckFailedOn returns the distinct validators whose self-check of
|
||||
// the address failed, in alphabetical order: over the whole history of the
|
||||
// address, or, if runID > 0, only the failures that happened in that run.
|
||||
func (d *DB) ListSelfCheckFailedOn(ctx context.Context, registryID, runID int64) ([]string, error) {
|
||||
q := `SELECT DISTINCT validator_id FROM ip_self_check_failures WHERE registry_id=?`
|
||||
args := []any{registryID}
|
||||
if runID > 0 {
|
||||
q += ` AND run_id=?`
|
||||
args = append(args, runID)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx, q+` ORDER BY validator_id`, args...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
out := []string{}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// SubmitIPs is the single admin entry point for both "add new addresses to
|
||||
// the queue" and "force a re-check of an already-finished address" — the
|
||||
// same list can freely mix both. Addresses are processed in one
|
||||
@@ -360,7 +538,9 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
// - unknown address: inserted as a new queued row.
|
||||
// - address currently done/failed/occupied: reset to queued (new attempt,
|
||||
// retry_count cleared — this is a deliberate admin-triggered restart,
|
||||
// not a system retry).
|
||||
// not a system retry). It also starts a new series of self-check
|
||||
// failures (sc_failures=0, validators that failed it before are no
|
||||
// longer excluded); the history in ip_self_check_failures stays.
|
||||
// - address currently queued (not yet claimed): left in state=queued,
|
||||
// only its sequence is updated.
|
||||
// - address currently mid-check (assigning_fip / awaiting_self_check /
|
||||
@@ -426,9 +606,9 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
return result, rErr
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, cycle, now, now); err != nil {
|
||||
return result, fmt.Errorf("insert %s: %w", addr, err)
|
||||
}
|
||||
result.Added = append(result.Added, addr)
|
||||
@@ -453,10 +633,10 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
UPDATE ip_queue SET
|
||||
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
|
||||
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', run_id=?,
|
||||
overall_result='', run_id=?, sc_failures=0, sc_round_start_cycle=?,
|
||||
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
WHERE ip_address=?
|
||||
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
|
||||
`, IPQueued, seq, cycle, rid, cycle, now, addr); err != nil {
|
||||
return result, fmt.Errorf("requeue %s: %w", addr, err)
|
||||
}
|
||||
result.Requeued = append(result.Requeued, addr)
|
||||
|
||||
Reference in new issue
Block a user