Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.
Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b7669c9e41
commit
e95b5eb7d5
34 files changed
+1092
-82
No files matched your search
@@ -46,6 +46,9 @@ var verdictIntegritySchema string
|
||||
//go:embed migrations/0011_check_runs.sql
|
||||
var checkRunsSchema string
|
||||
|
||||
//go:embed migrations/0012_self_check_failures.sql
|
||||
var selfCheckFailuresSchema string
|
||||
|
||||
// migrations is the ordered list of schema versions. Each entry's SQL is
|
||||
// applied, in order, for any version greater than the database's current
|
||||
// PRAGMA user_version — so a fresh database walks the whole list and an
|
||||
@@ -65,6 +68,7 @@ var migrations = []struct {
|
||||
{9, scaleIndexesSchema},
|
||||
{10, verdictIntegritySchema},
|
||||
{11, checkRunsSchema},
|
||||
{12, selfCheckFailuresSchema},
|
||||
}
|
||||
|
||||
type DB struct {
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
-- Self-check failures per address (see
|
||||
-- docs/changes/2026-10-04_08-01_self-check-exclude-validator-plan.md).
|
||||
--
|
||||
-- A failed self-check no longer sends the address back to the validator that
|
||||
-- failed it: ClaimNextQueued skips an address for every validator that failed
|
||||
-- it in the current round. ip_self_check_failures is the permanent history of
|
||||
-- those failures; it is keyed by registry_id (like checks and events), so it
|
||||
-- outlives the ip_queue row and is not touched by re-checks. No foreign keys
|
||||
-- on purpose: the manual cleanup in docs/ADMIN_CLEANUP.md deletes freely.
|
||||
--
|
||||
-- A round is the part of a series of failures in which validators that failed
|
||||
-- stay excluded. ip_queue.sc_round_start_cycle is the first cycle_id of the
|
||||
-- current round: a failure excludes its validator only if its cycle_id is not
|
||||
-- below it. cycle_id (not attempt_number) is used because it never repeats for
|
||||
-- an address, even when the ip_queue row is deleted and created again.
|
||||
--
|
||||
-- ip_queue.sc_failures counts self-check failures of the current series; a
|
||||
-- manual re-check or re-submission starts a new series.
|
||||
--
|
||||
-- settings.self_check_max_attempts is the ceiling of self-check failures per
|
||||
-- address, after which the address gets the verdict fail (1..50, default 5).
|
||||
|
||||
CREATE TABLE ip_self_check_failures (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
registry_id INTEGER NOT NULL,
|
||||
run_id INTEGER,
|
||||
cycle_id INTEGER NOT NULL,
|
||||
attempt_number INTEGER NOT NULL,
|
||||
validator_id TEXT NOT NULL,
|
||||
failed_at TIMESTAMP NOT NULL,
|
||||
detail TEXT NOT NULL DEFAULT ''
|
||||
);
|
||||
CREATE INDEX idx_sc_failures_registry ON ip_self_check_failures(registry_id, validator_id);
|
||||
|
||||
ALTER TABLE ip_queue ADD COLUMN sc_failures INTEGER NOT NULL DEFAULT 0;
|
||||
ALTER TABLE ip_queue ADD COLUMN sc_round_start_cycle INTEGER NOT NULL DEFAULT 0;
|
||||
|
||||
ALTER TABLE settings ADD COLUMN self_check_max_attempts INTEGER NOT NULL DEFAULT 5;
|
||||
+12
-2
@@ -276,10 +276,20 @@ type Settings struct {
|
||||
// HistoryRetentionCycles caps how many recent check cycles are kept per
|
||||
// registry address (see PruneRegistryHistory); 0 means unlimited.
|
||||
HistoryRetentionCycles int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
// SelfCheckMaxAttempts is the ceiling of failed self-checks per address
|
||||
// (1..MaxSelfCheckMaxAttempts); reaching it gives the address the verdict
|
||||
// fail (see FailSelfCheck).
|
||||
SelfCheckMaxAttempts int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
// Bounds of Settings.SelfCheckMaxAttempts.
|
||||
const (
|
||||
MinSelfCheckMaxAttempts = 1
|
||||
MaxSelfCheckMaxAttempts = 50
|
||||
)
|
||||
|
||||
// InboundChecksSettings is the singleton row describing what the prober
|
||||
// checks on every site for every in-flight IP (TCP ports + optional ICMP).
|
||||
// Admin-configurable at runtime (see queries_inbound.go).
|
||||
|
||||
+194
-14
@@ -44,10 +44,10 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_address) DO NOTHING
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, cycle, now, now); err != nil {
|
||||
return fmt.Errorf("seed %s: %w", addr, err)
|
||||
}
|
||||
}
|
||||
@@ -55,8 +55,11 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
|
||||
// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to
|
||||
// the given idle validator. It returns (nil, nil) if the validator isn't
|
||||
// idle or no IP is queued. The DB connection pool is capped at one physical
|
||||
// the given idle validator, skipping addresses this validator failed the
|
||||
// self-check of in the current round (see FailSelfCheck): such an address
|
||||
// stays queued for the other validators and does not block the ones behind
|
||||
// it. It returns (nil, nil) if the validator isn't idle or no IP is
|
||||
// claimable for it. The DB connection pool is capped at one physical
|
||||
// connection (see Open), so this transaction already has exclusive access
|
||||
// to the database for its duration — no other claim, requeue, or update can
|
||||
// interleave — which combined with the conditional UPDATEs (checked via
|
||||
@@ -82,9 +85,12 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
|
||||
|
||||
var item IPQueueItem
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT id, ip_address, sequence, attempt_number, retry_count
|
||||
FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1
|
||||
`, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
SELECT q.id, q.ip_address, q.sequence, q.attempt_number, q.retry_count
|
||||
FROM ip_queue q WHERE q.state=? AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=q.registry_id AND f.validator_id=? AND f.cycle_id>=q.sc_round_start_cycle)
|
||||
ORDER BY q.sequence LIMIT 1
|
||||
`, IPQueued, validatorID).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
@@ -352,6 +358,178 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// SelfCheckFailure is the outcome of FailSelfCheck.
|
||||
type SelfCheckFailure struct {
|
||||
// Failures is the number of failed self-checks in the address's current
|
||||
// series, this one included.
|
||||
Failures int
|
||||
// Failed: the ceiling was reached and the address got the verdict fail.
|
||||
Failed bool
|
||||
// Validators lists the validators that failed the self-check in this
|
||||
// series, oldest first (with repeats if one failed it more than once).
|
||||
Validators []string
|
||||
// NewRound: the address was queued again, and every working validator had
|
||||
// already failed it in the round, so the round was reset — the exclusions
|
||||
// no longer apply (always false when Failed).
|
||||
NewRound bool
|
||||
}
|
||||
|
||||
// FailSelfCheck handles a failed self-check of the address ipID held by
|
||||
// validatorID, in one transaction: it records the failure in
|
||||
// ip_self_check_failures and in ip_queue.sc_failures, then
|
||||
//
|
||||
// - if sc_failures reached maxAttempts: marks the address failed with the
|
||||
// verdict fail;
|
||||
// - otherwise sends it back to the queue without touching retry_count (a
|
||||
// failed self-check has its own ceiling, unlike a failed association or
|
||||
// an expired lease, see RequeueOrFail). The validator is excluded from
|
||||
// this address for the rest of the round (ClaimNextQueued). If no
|
||||
// working validator (idle, assigned, checking) is left without a failure
|
||||
// in the round, a new round starts: the exclusions lapse and the retry
|
||||
// follows the usual rules, so with fewer validators than maxAttempts the
|
||||
// address never gets stuck in the queue.
|
||||
//
|
||||
// The validator is freed in the same transaction. Returns ErrInvalidState if
|
||||
// the address is not awaiting_self_check on this validator (a late report).
|
||||
func (d *DB) FailSelfCheck(ctx context.Context, ipID int64, validatorID, detail string, maxAttempts int) (SelfCheckFailure, error) {
|
||||
var out SelfCheckFailure
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
var registryID, cycle, roundStart int64
|
||||
var runID sql.NullInt64
|
||||
var attempt, failures int
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT registry_id, run_id, cycle_id, attempt_number, sc_failures, sc_round_start_cycle
|
||||
FROM ip_queue WHERE id=? AND state=? AND owner_validator_id=?
|
||||
`, ipID, IPAwaitingSelfCheck, validatorID).Scan(®istryID, &runID, &cycle, &attempt, &failures, &roundStart)
|
||||
if err == sql.ErrNoRows {
|
||||
return out, fmt.Errorf("ip_id %d is not awaiting self-check on %s: %w", ipID, validatorID, ErrInvalidState)
|
||||
}
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_self_check_failures (registry_id, run_id, cycle_id, attempt_number, validator_id, failed_at, detail)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`, registryID, runID, cycle, attempt, validatorID, now, detail); err != nil {
|
||||
return out, fmt.Errorf("record self-check failure: %w", err)
|
||||
}
|
||||
failures++
|
||||
out.Failures = failures
|
||||
|
||||
rows, err := tx.QueryContext(ctx, `
|
||||
SELECT validator_id FROM (
|
||||
SELECT id, validator_id FROM ip_self_check_failures WHERE registry_id=? ORDER BY id DESC LIMIT ?
|
||||
) ORDER BY id
|
||||
`, registryID, failures)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
out.Validators = append(out.Validators, v)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
rows.Close()
|
||||
|
||||
if failures >= maxAttempts {
|
||||
out.Failed = true
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, sc_failures=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPFailed, failures, ResultFail, now, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
} else {
|
||||
newCycle, err := nextRegistryCycleTx(ctx, tx, registryID, now)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
var left int
|
||||
if err := tx.QueryRowContext(ctx, `
|
||||
SELECT COUNT(*) FROM validators v
|
||||
WHERE v.state IN (?, ?, ?) AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=? AND f.validator_id=v.validator_id AND f.cycle_id>=?)
|
||||
`, ValidatorIdle, ValidatorAssigned, ValidatorChecking, registryID, roundStart).Scan(&left); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if left == 0 {
|
||||
out.NewRound = true
|
||||
roundStart = int64(newCycle)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, owner_validator_id=NULL, fip_id='', attempt_number=attempt_number+1,
|
||||
cycle_id=?, sc_failures=?, sc_round_start_cycle=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPQueued, newCycle, failures, roundStart, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
return out, tx.Commit()
|
||||
}
|
||||
|
||||
// GetIPRunID returns the check run the queue row belongs to (0 if none).
|
||||
func (d *DB) GetIPRunID(ctx context.Context, ipID int64) (int64, error) {
|
||||
var runID sql.NullInt64
|
||||
if err := d.QueryRowContext(ctx, `SELECT run_id FROM ip_queue WHERE id=?`, ipID).Scan(&runID); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return runID.Int64, nil
|
||||
}
|
||||
|
||||
// ListSelfCheckFailedOn returns the distinct validators whose self-check of
|
||||
// the address failed, in alphabetical order: over the whole history of the
|
||||
// address, or, if runID > 0, only the failures that happened in that run.
|
||||
func (d *DB) ListSelfCheckFailedOn(ctx context.Context, registryID, runID int64) ([]string, error) {
|
||||
q := `SELECT DISTINCT validator_id FROM ip_self_check_failures WHERE registry_id=?`
|
||||
args := []any{registryID}
|
||||
if runID > 0 {
|
||||
q += ` AND run_id=?`
|
||||
args = append(args, runID)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx, q+` ORDER BY validator_id`, args...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
out := []string{}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// SubmitIPs is the single admin entry point for both "add new addresses to
|
||||
// the queue" and "force a re-check of an already-finished address" — the
|
||||
// same list can freely mix both. Addresses are processed in one
|
||||
@@ -360,7 +538,9 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
// - unknown address: inserted as a new queued row.
|
||||
// - address currently done/failed/occupied: reset to queued (new attempt,
|
||||
// retry_count cleared — this is a deliberate admin-triggered restart,
|
||||
// not a system retry).
|
||||
// not a system retry). It also starts a new series of self-check
|
||||
// failures (sc_failures=0, validators that failed it before are no
|
||||
// longer excluded); the history in ip_self_check_failures stays.
|
||||
// - address currently queued (not yet claimed): left in state=queued,
|
||||
// only its sequence is updated.
|
||||
// - address currently mid-check (assigning_fip / awaiting_self_check /
|
||||
@@ -426,9 +606,9 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
return result, rErr
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, cycle, now, now); err != nil {
|
||||
return result, fmt.Errorf("insert %s: %w", addr, err)
|
||||
}
|
||||
result.Added = append(result.Added, addr)
|
||||
@@ -453,10 +633,10 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
UPDATE ip_queue SET
|
||||
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
|
||||
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', run_id=?,
|
||||
overall_result='', run_id=?, sc_failures=0, sc_round_start_cycle=?,
|
||||
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
WHERE ip_address=?
|
||||
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
|
||||
`, IPQueued, seq, cycle, rid, cycle, now, addr); err != nil {
|
||||
return result, fmt.Errorf("requeue %s: %w", addr, err)
|
||||
}
|
||||
result.Requeued = append(result.Requeued, addr)
|
||||
|
||||
@@ -383,7 +383,7 @@ func TestMigration0011BuildsRunsFromExistingData(t *testing.T) {
|
||||
}
|
||||
var ver int
|
||||
d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver)
|
||||
if ver != 11 {
|
||||
if ver != 12 {
|
||||
t.Errorf("user_version = %d", ver)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
// claimAwaitingSelfCheck claims the next address for the validator and brings
|
||||
// it to awaiting_self_check, as the orchestrator does after the association.
|
||||
func claimAwaitingSelfCheck(t *testing.T, ctx context.Context, d *DB, validatorID string) *IPQueueItem {
|
||||
t.Helper()
|
||||
item, err := d.ClaimNextQueued(ctx, validatorID, time.Minute)
|
||||
if err != nil || item == nil {
|
||||
t.Fatalf("claim for %s: item=%+v err=%v", validatorID, item, err)
|
||||
}
|
||||
if err := d.SetFIPAssociated(ctx, item.ID, "fip-"+item.IPAddress, time.Minute); err != nil {
|
||||
t.Fatalf("set fip associated: %v", err)
|
||||
}
|
||||
return item
|
||||
}
|
||||
|
||||
// A validator that failed the self-check of an address is not given that
|
||||
// address again, takes the next one instead, and another validator takes the
|
||||
// excluded one; the exclusion covers that address only.
|
||||
func TestClaimSkipsAddressExcludedForValidator(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
first := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if first.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected 1.1.1.1 first, got %s", first.IPAddress)
|
||||
}
|
||||
res, err := d.FailSelfCheck(ctx, first.ID, "v1", "ip echo timeout", 5)
|
||||
if err != nil {
|
||||
t.Fatalf("fail self-check: %v", err)
|
||||
}
|
||||
if res.Failed || res.NewRound || res.Failures != 1 {
|
||||
t.Fatalf("expected a plain retry after the first failure, got %+v", res)
|
||||
}
|
||||
back, _ := d.GetIP(ctx, first.ID)
|
||||
if back.State != IPQueued || back.RetryCount != 0 {
|
||||
t.Fatalf("expected queued with retry_count untouched, got state=%s retry_count=%d", back.State, back.RetryCount)
|
||||
}
|
||||
if v, _ := d.GetValidator(ctx, "v1"); v.State != ValidatorIdle {
|
||||
t.Fatalf("expected v1 freed to idle, got %s", v.State)
|
||||
}
|
||||
|
||||
// v1 skips 1.1.1.1 (still ahead in the queue) and takes 2.2.2.2.
|
||||
second, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
|
||||
if err != nil || second == nil || second.IPAddress != "2.2.2.2" {
|
||||
t.Fatalf("expected v1 to take 2.2.2.2, got %+v err=%v", second, err)
|
||||
}
|
||||
// v2 takes the excluded address.
|
||||
other, err := d.ClaimNextQueued(ctx, "v2", time.Minute)
|
||||
if err != nil || other == nil || other.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected v2 to take 1.1.1.1, got %+v err=%v", other, err)
|
||||
}
|
||||
}
|
||||
|
||||
// When every working validator has failed the address in the round, a new
|
||||
// round starts and the exclusions lapse; an unreachable validator does not
|
||||
// count as working.
|
||||
func TestSelfCheckNewRoundLiftsExclusions(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if _, err := d.Exec(`UPDATE validators SET state=? WHERE validator_id='v2'`, ValidatorUnreachable); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
res, err := d.FailSelfCheck(ctx, item.ID, "v1", "x", 5)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.NewRound {
|
||||
t.Fatalf("expected a new round (the only working validator failed it), got %+v", res)
|
||||
}
|
||||
again, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
|
||||
if err != nil || again == nil || again.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected v1 to get the address again in the new round, got %+v err=%v", again, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Reaching the ceiling gives the verdict fail; a re-submission starts a new
|
||||
// series (counter and exclusions) but keeps the failure history.
|
||||
func TestSelfCheckCeilingAndResubmit(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if res, err := d.FailSelfCheck(ctx, item.ID, "v1", "x", 2); err != nil || res.Failed {
|
||||
t.Fatalf("below the ceiling: res=%+v err=%v", res, err)
|
||||
}
|
||||
item = claimAwaitingSelfCheck(t, ctx, d, "v2")
|
||||
res, err := d.FailSelfCheck(ctx, item.ID, "v2", "y", 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.Failed || res.Failures != 2 || len(res.Validators) != 2 || res.Validators[0] != "v1" || res.Validators[1] != "v2" {
|
||||
t.Fatalf("expected fail at the ceiling on v1, v2, got %+v", res)
|
||||
}
|
||||
ip, _ := d.GetIP(ctx, item.ID)
|
||||
if ip.State != IPFailed || ip.OverallResult != ResultFail {
|
||||
t.Fatalf("expected failed/fail, got %s/%s", ip.State, ip.OverallResult)
|
||||
}
|
||||
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var failures int
|
||||
if err := d.QueryRow(`SELECT sc_failures FROM ip_queue WHERE id=?`, item.ID).Scan(&failures); err != nil || failures != 0 {
|
||||
t.Fatalf("expected the series reset to 0, got %d err=%v", failures, err)
|
||||
}
|
||||
if got, err := d.ListSelfCheckFailedOn(ctx, ip.RegistryID, 0); err != nil || len(got) != 2 || got[0] != "v1" || got[1] != "v2" {
|
||||
t.Fatalf("expected the history kept (v1, v2), got %v err=%v", got, err)
|
||||
}
|
||||
// v1 failed it before the re-submission, but is no longer excluded.
|
||||
if again, err := d.ClaimNextQueued(ctx, "v1", time.Minute); err != nil || again == nil {
|
||||
t.Fatalf("expected v1 to claim the re-submitted address, got %+v err=%v", again, err)
|
||||
}
|
||||
}
|
||||
|
||||
// A late report for an address the validator does not hold is refused.
|
||||
func TestFailSelfCheckRequiresAwaitingSelfCheckOnValidator(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if _, err := d.FailSelfCheck(ctx, item.ID, "v2", "late", 5); !errors.Is(err, ErrInvalidState) {
|
||||
t.Fatalf("expected ErrInvalidState for another validator, got %v", err)
|
||||
}
|
||||
cur, _ := d.GetIP(ctx, item.ID)
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, cur.RegistryID, 0); len(got) != 0 {
|
||||
t.Fatalf("a refused report must leave no history, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelfCheckMaxAttemptsSetting(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.BootstrapFromConfig(ctx, &config.ControlAPI{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s, err := d.GetSettings(ctx); err != nil || s.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected default 5, got %+v err=%v", s, err)
|
||||
}
|
||||
for _, bad := range []int{0, -1, 51} {
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, bad); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation for %d, got %v", bad, err)
|
||||
}
|
||||
}
|
||||
for _, ok := range []int{1, 50} {
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, ok); err != nil {
|
||||
t.Fatalf("set %d: %v", ok, err)
|
||||
}
|
||||
if s, _ := d.GetSettings(ctx); s.SelfCheckMaxAttempts != ok {
|
||||
t.Fatalf("expected %d, got %d", ok, s.SelfCheckMaxAttempts)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -13,8 +13,8 @@ func (d *DB) GetSettings(ctx context.Context) (Settings, error) {
|
||||
var s Settings
|
||||
var createdAt, updatedAt string
|
||||
err := d.QueryRowContext(ctx, `
|
||||
SELECT fip_settle_seconds, history_retention_cycles, created_at, updated_at FROM settings WHERE id=1
|
||||
`).Scan(&s.FIPSettleSeconds, &s.HistoryRetentionCycles, &createdAt, &updatedAt)
|
||||
SELECT fip_settle_seconds, history_retention_cycles, self_check_max_attempts, created_at, updated_at FROM settings WHERE id=1
|
||||
`).Scan(&s.FIPSettleSeconds, &s.HistoryRetentionCycles, &s.SelfCheckMaxAttempts, &createdAt, &updatedAt)
|
||||
if err != nil {
|
||||
return Settings{}, err
|
||||
}
|
||||
@@ -58,3 +58,17 @@ func (d *DB) SetHistoryRetentionCycles(ctx context.Context, cycles int) error {
|
||||
`, cycles, now)
|
||||
return err
|
||||
}
|
||||
|
||||
// SetSelfCheckMaxAttempts persists the ceiling of failed self-checks per
|
||||
// address (1..MaxSelfCheckMaxAttempts). It applies from the next failed
|
||||
// self-check, without a restart.
|
||||
func (d *DB) SetSelfCheckMaxAttempts(ctx context.Context, attempts int) error {
|
||||
if attempts < MinSelfCheckMaxAttempts || attempts > MaxSelfCheckMaxAttempts {
|
||||
return fmt.Errorf("self_check_max_attempts must be in %d..%d: %w", MinSelfCheckMaxAttempts, MaxSelfCheckMaxAttempts, ErrValidation)
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE settings SET self_check_max_attempts=?, updated_at=? WHERE id=1
|
||||
`, attempts, now)
|
||||
return err
|
||||
}
|
||||
@@ -240,7 +240,7 @@ func TestMigration0010MarksRowsAfterVerdict(t *testing.T) {
|
||||
t.Errorf("ssh: after_verdict=%d recorded=%s created=%s", a, rec, cr)
|
||||
}
|
||||
var ver int
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 11 {
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 12 {
|
||||
t.Errorf("user_version=%d err=%v", ver, err)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user