Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.
Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b7669c9e41
commit
e95b5eb7d5
34 files changed
+1092
-82
No files matched your search
@@ -356,10 +356,10 @@ func (c *client) GetOrchestratorSettings(ctx context.Context) (orchestratorSetti
|
||||
return out, err
|
||||
}
|
||||
|
||||
func (c *client) PutOrchestratorSettings(ctx context.Context, fipSettleSeconds, historyRetentionCycles int) (orchestratorSettingsDTO, error) {
|
||||
func (c *client) PutOrchestratorSettings(ctx context.Context, fipSettleSeconds, historyRetentionCycles, selfCheckMaxAttempts int) (orchestratorSettingsDTO, error) {
|
||||
var out orchestratorSettingsDTO
|
||||
err := c.do(ctx, http.MethodPut, "/api/v1/admin/config/orchestrator",
|
||||
orchestratorSettingsDTO{FIPSettleSeconds: fipSettleSeconds, HistoryRetentionCycles: historyRetentionCycles}, &out)
|
||||
orchestratorSettingsDTO{FIPSettleSeconds: fipSettleSeconds, HistoryRetentionCycles: historyRetentionCycles, SelfCheckMaxAttempts: selfCheckMaxAttempts}, &out)
|
||||
return out, err
|
||||
}
|
||||
|
||||
|
||||
@@ -37,6 +37,10 @@ type fakeControlAPI struct {
|
||||
inboundICMP bool
|
||||
|
||||
historyRetentionCycles int
|
||||
selfCheckMaxAttempts int
|
||||
// selfCheckFailedOn is served as self_check_failed_on of the address
|
||||
// detail and registry history endpoints.
|
||||
selfCheckFailedOn []string
|
||||
|
||||
// Analytics: the run selector, the report JSON per run, the lists per
|
||||
// "run/kind[/class]" and the subnet list of /settings.
|
||||
@@ -94,6 +98,8 @@ func newFakeControlAPI(t *testing.T) (*fakeControlAPI, string) {
|
||||
registry: map[string]registryItem{},
|
||||
registryChecks: map[string][]check{},
|
||||
autoCycle: autoCycleDTO{IntervalSeconds: 3600, Phase: "idle"},
|
||||
|
||||
selfCheckMaxAttempts: 5,
|
||||
}
|
||||
ts := httptest.NewServer(f.handler())
|
||||
t.Cleanup(ts.Close)
|
||||
@@ -191,7 +197,7 @@ func (f *fakeControlAPI) handler() http.Handler {
|
||||
addr := r.PathValue("ip")
|
||||
for _, ip := range f.ips {
|
||||
if ip.IPAddress == addr {
|
||||
writeJSON(w, http.StatusOK, ipDetailResponse{IP: ip, Checks: []check{}, Events: []event{}})
|
||||
writeJSON(w, http.StatusOK, ipDetailResponse{IP: ip, Checks: []check{}, Events: []event{}, SelfCheckFailedOn: f.selfCheckFailedOn})
|
||||
return
|
||||
}
|
||||
}
|
||||
@@ -310,6 +316,7 @@ func (f *fakeControlAPI) handler() http.Handler {
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: f.fipSettleSeconds,
|
||||
HistoryRetentionCycles: f.historyRetentionCycles,
|
||||
SelfCheckMaxAttempts: f.selfCheckMaxAttempts,
|
||||
})
|
||||
})
|
||||
mux.HandleFunc("PUT /api/v1/admin/config/orchestrator", func(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -325,11 +332,17 @@ func (f *fakeControlAPI) handler() http.Handler {
|
||||
writeAPIErr(w, http.StatusBadRequest, "history_retention_cycles must be >= 0")
|
||||
return
|
||||
}
|
||||
if req.SelfCheckMaxAttempts < 1 || req.SelfCheckMaxAttempts > 50 {
|
||||
writeAPIErr(w, http.StatusBadRequest, "self_check_max_attempts must be in 1..50")
|
||||
return
|
||||
}
|
||||
f.fipSettleSeconds = req.FIPSettleSeconds
|
||||
f.historyRetentionCycles = req.HistoryRetentionCycles
|
||||
f.selfCheckMaxAttempts = req.SelfCheckMaxAttempts
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: f.fipSettleSeconds,
|
||||
HistoryRetentionCycles: f.historyRetentionCycles,
|
||||
SelfCheckMaxAttempts: f.selfCheckMaxAttempts,
|
||||
})
|
||||
})
|
||||
|
||||
@@ -474,7 +487,7 @@ func (f *fakeControlAPI) handler() http.Handler {
|
||||
writeAPIErr(w, http.StatusNotFound, "unknown ip: "+addr)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, registryHistoryResponse{Registry: item, Checks: f.registryChecks[addr]})
|
||||
writeJSON(w, http.StatusOK, registryHistoryResponse{Registry: item, Checks: f.registryChecks[addr], SelfCheckFailedOn: f.selfCheckFailedOn})
|
||||
})
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/config/subnets", func(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
@@ -95,6 +95,9 @@ type ipDetailResponse struct {
|
||||
IP ipQueueItem `json:"ip"`
|
||||
Checks []check `json:"checks"`
|
||||
Events []event `json:"events"`
|
||||
// SelfCheckFailedOn lists the validators whose self-check of this address
|
||||
// failed (in its current run).
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}
|
||||
|
||||
type validator struct {
|
||||
@@ -308,6 +311,9 @@ func (t typeStat) Class() string { return statClass(t.OK, t.Total) }
|
||||
type registryHistoryResponse struct {
|
||||
Registry registryItem `json:"registry"`
|
||||
Checks []check `json:"checks"`
|
||||
// SelfCheckFailedOn lists the validators whose self-check of this address
|
||||
// failed, over every run.
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}
|
||||
|
||||
type validatorDTO struct {
|
||||
@@ -344,6 +350,7 @@ type errorResponse struct {
|
||||
type orchestratorSettingsDTO struct {
|
||||
FIPSettleSeconds int `json:"fip_settle_seconds"`
|
||||
HistoryRetentionCycles int `json:"history_retention_cycles"`
|
||||
SelfCheckMaxAttempts int `json:"self_check_max_attempts"`
|
||||
}
|
||||
|
||||
type inboundChecksDTO struct {
|
||||
|
||||
@@ -77,7 +77,12 @@ func (s *Server) handleSettingsPut(w http.ResponseWriter, r *http.Request) {
|
||||
s.renderSettingsForm(w, r, &apiErr{Status: http.StatusBadRequest, Message: "глубина истории должна быть целым числом циклов"})
|
||||
return
|
||||
}
|
||||
_, err = s.CA.PutOrchestratorSettings(r.Context(), seconds, retentionCycles)
|
||||
selfCheckMax, err := strconv.Atoi(r.PostFormValue("self_check_max_attempts"))
|
||||
if err != nil {
|
||||
s.renderSettingsForm(w, r, &apiErr{Status: http.StatusBadRequest, Message: "потолок провалов self-check должен быть целым числом"})
|
||||
return
|
||||
}
|
||||
_, err = s.CA.PutOrchestratorSettings(r.Context(), seconds, retentionCycles, selfCheckMax)
|
||||
s.renderSettingsForm(w, r, err)
|
||||
}
|
||||
|
||||
|
||||
@@ -307,9 +307,12 @@ func TestSettingsGetAndPut(t *testing.T) {
|
||||
if !strings.Contains(page, `value="0"`) {
|
||||
t.Fatalf("expected default 0 in the form, got:\n%s", page)
|
||||
}
|
||||
if !strings.Contains(page, "Потолок провалов self-check на адрес") || !strings.Contains(page, `id="self_check_max_attempts" name="self_check_max_attempts" min="1" max="50" step="1" value="5"`) {
|
||||
t.Fatalf("expected the self-check ceiling field with the default 5, got:\n%s", page)
|
||||
}
|
||||
|
||||
body := postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"10"},
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"10"}, "self_check_max_attempts": {"8"},
|
||||
})
|
||||
if !strings.Contains(body, `value="15"`) || !strings.Contains(body, `value="10"`) {
|
||||
t.Fatalf("expected updated values 15/10 in re-rendered form, got:\n%s", body)
|
||||
@@ -320,11 +323,14 @@ func TestSettingsGetAndPut(t *testing.T) {
|
||||
if fake.historyRetentionCycles != 10 {
|
||||
t.Fatalf("expected fake control-api history_retention_cycles updated, got %d", fake.historyRetentionCycles)
|
||||
}
|
||||
if fake.selfCheckMaxAttempts != 8 || !strings.Contains(body, `name="self_check_max_attempts" min="1" max="50" step="1" value="8"`) {
|
||||
t.Fatalf("expected the self-check ceiling 8 saved and shown, got %d:\n%s", fake.selfCheckMaxAttempts, body)
|
||||
}
|
||||
|
||||
// A control-api validation error (negative value here) surfaces via
|
||||
// the banner, not a crash.
|
||||
body = postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"-1"}, "history_retention_cycles": {"10"},
|
||||
"fip_settle_seconds": {"-1"}, "history_retention_cycles": {"10"}, "self_check_max_attempts": {"8"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for invalid value, got:\n%s", body)
|
||||
@@ -333,7 +339,7 @@ func TestSettingsGetAndPut(t *testing.T) {
|
||||
// A non-numeric value is caught by the dashboard itself before it ever
|
||||
// reaches control-api.
|
||||
body = postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"not-a-number"}, "history_retention_cycles": {"10"},
|
||||
"fip_settle_seconds": {"not-a-number"}, "history_retention_cycles": {"10"}, "self_check_max_attempts": {"8"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for non-numeric value, got:\n%s", body)
|
||||
@@ -341,11 +347,49 @@ func TestSettingsGetAndPut(t *testing.T) {
|
||||
|
||||
// Same for a non-numeric retention value.
|
||||
body = postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"not-a-number"},
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"not-a-number"}, "self_check_max_attempts": {"8"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for non-numeric retention value, got:\n%s", body)
|
||||
}
|
||||
|
||||
// A ceiling outside 1..50 is refused by control-api and shown as a
|
||||
// warning; the form keeps the stored value.
|
||||
body = postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"10"}, "self_check_max_attempts": {"51"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") || fake.selfCheckMaxAttempts != 8 {
|
||||
t.Fatalf("expected a warning and the stored ceiling 8 kept, got %d:\n%s", fake.selfCheckMaxAttempts, body)
|
||||
}
|
||||
|
||||
// A non-numeric ceiling is caught by the dashboard itself.
|
||||
body = postForm(t, ts, "PUT", "/settings", map[string][]string{
|
||||
"fip_settle_seconds": {"15"}, "history_retention_cycles": {"10"}, "self_check_max_attempts": {"many"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for non-numeric ceiling, got:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
// The address pages say on which validators the self-check failed.
|
||||
func TestAddressPagesShowSelfCheckFailedOn(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
now := time.Now()
|
||||
fake.ips = []ipQueueItem{{IPAddress: "5.5.5.5", State: "checking", UpdatedAt: now, CreatedAt: now}}
|
||||
fake.registry["5.5.5.5"] = registryItem{IPAddress: "5.5.5.5", FirstSeenAt: now, LastSeenAt: now}
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
for _, path := range []string{"/ips/5.5.5.5", "/registry/5.5.5.5"} {
|
||||
if page := get(t, ts, path); strings.Contains(page, "Self-check не прошёл на") {
|
||||
t.Fatalf("%s: no failures, but the line is shown:\n%s", path, page)
|
||||
}
|
||||
}
|
||||
fake.selfCheckFailedOn = []string{"vkiplab-v17", "vkiplab-v13"}
|
||||
for _, path := range []string{"/ips/5.5.5.5", "/registry/5.5.5.5"} {
|
||||
if page := get(t, ts, path); !strings.Contains(page, "Self-check не прошёл на: vkiplab-v17, vkiplab-v13") {
|
||||
t.Fatalf("%s: expected the failed validators line, got:\n%s", path, page)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSettingsPageShowsInboundChecks(t *testing.T) {
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
создан: {{fmtTime .Detail.IP.CreatedAt}} · назначен: {{fmtTime .Detail.IP.AssignedAt}} ·
|
||||
агрегирован: {{fmtTime .Detail.IP.AggregatedAt}} · FIP освобождён: {{fmtTime .Detail.IP.FIPReleasedAt}}
|
||||
</p>
|
||||
{{if .Detail.SelfCheckFailedOn}}<p class="muted">Self-check не прошёл на: {{join .Detail.SelfCheckFailedOn ", "}}</p>{{end}}
|
||||
|
||||
<h2 class="section-title">Проверки (попытка {{.Detail.IP.AttemptNumber}})</h2>
|
||||
{{if .Detail.Checks}}
|
||||
|
||||
@@ -34,6 +34,7 @@
|
||||
впервые замечен: {{fmtTime .History.Registry.FirstSeenAt}} · последний раз замечен: {{fmtTime .History.Registry.LastSeenAt}}
|
||||
· всего циклов проверки: {{.History.Registry.TotalCycles}}
|
||||
</p>
|
||||
{{if .History.SelfCheckFailedOn}}<p class="muted">Self-check не прошёл на: {{join .History.SelfCheckFailedOn ", "}}</p>{{end}}
|
||||
|
||||
<h2 class="section-title">История проверок (все сохранённые циклы)</h2>
|
||||
{{if .History.Checks}}
|
||||
|
||||
@@ -80,6 +80,10 @@
|
||||
<a href="/registry">реестре</a> для каждого адреса. Адрес и его накопленная статистика остаются в реестре даже
|
||||
после удаления из очереди — эта настройка ограничивает только глубину истории конкретных проверок, не сам реестр.
|
||||
<code>0</code> — хранить без ограничения.</p>
|
||||
<p class="muted" style="margin-bottom:16px">Сколько раз self-check может не пройти у одного адреса, прежде чем адрес
|
||||
получит итог <code>fail</code> (допустимо 1–50, по умолчанию 5). Повторы идут на других валидаторах: валидатор, на
|
||||
котором адрес не прошёл self-check, этому адресу больше не выдаётся (на остальные адреса это не влияет).
|
||||
Значение действует на следующих повторах, без перезапуска.</p>
|
||||
<form hx-put="/settings" hx-target="#settings-form-wrap" hx-swap="innerHTML">
|
||||
<div class="field-row">
|
||||
<div class="field">
|
||||
@@ -90,6 +94,10 @@
|
||||
<label for="history_retention_cycles">Глубина истории проверок (циклов на адрес, 0 = не ограничено)</label>
|
||||
<input type="number" id="history_retention_cycles" name="history_retention_cycles" min="0" step="1" value="{{.Settings.HistoryRetentionCycles}}" required>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="self_check_max_attempts">Потолок провалов self-check на адрес</label>
|
||||
<input type="number" id="self_check_max_attempts" name="self_check_max_attempts" min="1" max="50" step="1" value="{{.Settings.SelfCheckMaxAttempts}}" required>
|
||||
</div>
|
||||
<button type="submit" class="btn btn-primary">Сохранить</button>
|
||||
</div>
|
||||
</form>
|
||||
|
||||
@@ -46,6 +46,9 @@ var verdictIntegritySchema string
|
||||
//go:embed migrations/0011_check_runs.sql
|
||||
var checkRunsSchema string
|
||||
|
||||
//go:embed migrations/0012_self_check_failures.sql
|
||||
var selfCheckFailuresSchema string
|
||||
|
||||
// migrations is the ordered list of schema versions. Each entry's SQL is
|
||||
// applied, in order, for any version greater than the database's current
|
||||
// PRAGMA user_version — so a fresh database walks the whole list and an
|
||||
@@ -65,6 +68,7 @@ var migrations = []struct {
|
||||
{9, scaleIndexesSchema},
|
||||
{10, verdictIntegritySchema},
|
||||
{11, checkRunsSchema},
|
||||
{12, selfCheckFailuresSchema},
|
||||
}
|
||||
|
||||
type DB struct {
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
-- Self-check failures per address (see
|
||||
-- docs/changes/2026-10-04_08-01_self-check-exclude-validator-plan.md).
|
||||
--
|
||||
-- A failed self-check no longer sends the address back to the validator that
|
||||
-- failed it: ClaimNextQueued skips an address for every validator that failed
|
||||
-- it in the current round. ip_self_check_failures is the permanent history of
|
||||
-- those failures; it is keyed by registry_id (like checks and events), so it
|
||||
-- outlives the ip_queue row and is not touched by re-checks. No foreign keys
|
||||
-- on purpose: the manual cleanup in docs/ADMIN_CLEANUP.md deletes freely.
|
||||
--
|
||||
-- A round is the part of a series of failures in which validators that failed
|
||||
-- stay excluded. ip_queue.sc_round_start_cycle is the first cycle_id of the
|
||||
-- current round: a failure excludes its validator only if its cycle_id is not
|
||||
-- below it. cycle_id (not attempt_number) is used because it never repeats for
|
||||
-- an address, even when the ip_queue row is deleted and created again.
|
||||
--
|
||||
-- ip_queue.sc_failures counts self-check failures of the current series; a
|
||||
-- manual re-check or re-submission starts a new series.
|
||||
--
|
||||
-- settings.self_check_max_attempts is the ceiling of self-check failures per
|
||||
-- address, after which the address gets the verdict fail (1..50, default 5).
|
||||
|
||||
CREATE TABLE ip_self_check_failures (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
registry_id INTEGER NOT NULL,
|
||||
run_id INTEGER,
|
||||
cycle_id INTEGER NOT NULL,
|
||||
attempt_number INTEGER NOT NULL,
|
||||
validator_id TEXT NOT NULL,
|
||||
failed_at TIMESTAMP NOT NULL,
|
||||
detail TEXT NOT NULL DEFAULT ''
|
||||
);
|
||||
CREATE INDEX idx_sc_failures_registry ON ip_self_check_failures(registry_id, validator_id);
|
||||
|
||||
ALTER TABLE ip_queue ADD COLUMN sc_failures INTEGER NOT NULL DEFAULT 0;
|
||||
ALTER TABLE ip_queue ADD COLUMN sc_round_start_cycle INTEGER NOT NULL DEFAULT 0;
|
||||
|
||||
ALTER TABLE settings ADD COLUMN self_check_max_attempts INTEGER NOT NULL DEFAULT 5;
|
||||
+12
-2
@@ -276,10 +276,20 @@ type Settings struct {
|
||||
// HistoryRetentionCycles caps how many recent check cycles are kept per
|
||||
// registry address (see PruneRegistryHistory); 0 means unlimited.
|
||||
HistoryRetentionCycles int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
// SelfCheckMaxAttempts is the ceiling of failed self-checks per address
|
||||
// (1..MaxSelfCheckMaxAttempts); reaching it gives the address the verdict
|
||||
// fail (see FailSelfCheck).
|
||||
SelfCheckMaxAttempts int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
// Bounds of Settings.SelfCheckMaxAttempts.
|
||||
const (
|
||||
MinSelfCheckMaxAttempts = 1
|
||||
MaxSelfCheckMaxAttempts = 50
|
||||
)
|
||||
|
||||
// InboundChecksSettings is the singleton row describing what the prober
|
||||
// checks on every site for every in-flight IP (TCP ports + optional ICMP).
|
||||
// Admin-configurable at runtime (see queries_inbound.go).
|
||||
|
||||
+194
-14
@@ -44,10 +44,10 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_address) DO NOTHING
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, cycle, now, now); err != nil {
|
||||
return fmt.Errorf("seed %s: %w", addr, err)
|
||||
}
|
||||
}
|
||||
@@ -55,8 +55,11 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
}
|
||||
|
||||
// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to
|
||||
// the given idle validator. It returns (nil, nil) if the validator isn't
|
||||
// idle or no IP is queued. The DB connection pool is capped at one physical
|
||||
// the given idle validator, skipping addresses this validator failed the
|
||||
// self-check of in the current round (see FailSelfCheck): such an address
|
||||
// stays queued for the other validators and does not block the ones behind
|
||||
// it. It returns (nil, nil) if the validator isn't idle or no IP is
|
||||
// claimable for it. The DB connection pool is capped at one physical
|
||||
// connection (see Open), so this transaction already has exclusive access
|
||||
// to the database for its duration — no other claim, requeue, or update can
|
||||
// interleave — which combined with the conditional UPDATEs (checked via
|
||||
@@ -82,9 +85,12 @@ func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL t
|
||||
|
||||
var item IPQueueItem
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT id, ip_address, sequence, attempt_number, retry_count
|
||||
FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1
|
||||
`, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
SELECT q.id, q.ip_address, q.sequence, q.attempt_number, q.retry_count
|
||||
FROM ip_queue q WHERE q.state=? AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=q.registry_id AND f.validator_id=? AND f.cycle_id>=q.sc_round_start_cycle)
|
||||
ORDER BY q.sequence LIMIT 1
|
||||
`, IPQueued, validatorID).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
@@ -352,6 +358,178 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// SelfCheckFailure is the outcome of FailSelfCheck.
|
||||
type SelfCheckFailure struct {
|
||||
// Failures is the number of failed self-checks in the address's current
|
||||
// series, this one included.
|
||||
Failures int
|
||||
// Failed: the ceiling was reached and the address got the verdict fail.
|
||||
Failed bool
|
||||
// Validators lists the validators that failed the self-check in this
|
||||
// series, oldest first (with repeats if one failed it more than once).
|
||||
Validators []string
|
||||
// NewRound: the address was queued again, and every working validator had
|
||||
// already failed it in the round, so the round was reset — the exclusions
|
||||
// no longer apply (always false when Failed).
|
||||
NewRound bool
|
||||
}
|
||||
|
||||
// FailSelfCheck handles a failed self-check of the address ipID held by
|
||||
// validatorID, in one transaction: it records the failure in
|
||||
// ip_self_check_failures and in ip_queue.sc_failures, then
|
||||
//
|
||||
// - if sc_failures reached maxAttempts: marks the address failed with the
|
||||
// verdict fail;
|
||||
// - otherwise sends it back to the queue without touching retry_count (a
|
||||
// failed self-check has its own ceiling, unlike a failed association or
|
||||
// an expired lease, see RequeueOrFail). The validator is excluded from
|
||||
// this address for the rest of the round (ClaimNextQueued). If no
|
||||
// working validator (idle, assigned, checking) is left without a failure
|
||||
// in the round, a new round starts: the exclusions lapse and the retry
|
||||
// follows the usual rules, so with fewer validators than maxAttempts the
|
||||
// address never gets stuck in the queue.
|
||||
//
|
||||
// The validator is freed in the same transaction. Returns ErrInvalidState if
|
||||
// the address is not awaiting_self_check on this validator (a late report).
|
||||
func (d *DB) FailSelfCheck(ctx context.Context, ipID int64, validatorID, detail string, maxAttempts int) (SelfCheckFailure, error) {
|
||||
var out SelfCheckFailure
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
var registryID, cycle, roundStart int64
|
||||
var runID sql.NullInt64
|
||||
var attempt, failures int
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT registry_id, run_id, cycle_id, attempt_number, sc_failures, sc_round_start_cycle
|
||||
FROM ip_queue WHERE id=? AND state=? AND owner_validator_id=?
|
||||
`, ipID, IPAwaitingSelfCheck, validatorID).Scan(®istryID, &runID, &cycle, &attempt, &failures, &roundStart)
|
||||
if err == sql.ErrNoRows {
|
||||
return out, fmt.Errorf("ip_id %d is not awaiting self-check on %s: %w", ipID, validatorID, ErrInvalidState)
|
||||
}
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_self_check_failures (registry_id, run_id, cycle_id, attempt_number, validator_id, failed_at, detail)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`, registryID, runID, cycle, attempt, validatorID, now, detail); err != nil {
|
||||
return out, fmt.Errorf("record self-check failure: %w", err)
|
||||
}
|
||||
failures++
|
||||
out.Failures = failures
|
||||
|
||||
rows, err := tx.QueryContext(ctx, `
|
||||
SELECT validator_id FROM (
|
||||
SELECT id, validator_id FROM ip_self_check_failures WHERE registry_id=? ORDER BY id DESC LIMIT ?
|
||||
) ORDER BY id
|
||||
`, registryID, failures)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
out.Validators = append(out.Validators, v)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
rows.Close()
|
||||
|
||||
if failures >= maxAttempts {
|
||||
out.Failed = true
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, sc_failures=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPFailed, failures, ResultFail, now, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return out, err
|
||||
}
|
||||
} else {
|
||||
newCycle, err := nextRegistryCycleTx(ctx, tx, registryID, now)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
var left int
|
||||
if err := tx.QueryRowContext(ctx, `
|
||||
SELECT COUNT(*) FROM validators v
|
||||
WHERE v.state IN (?, ?, ?) AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_self_check_failures f
|
||||
WHERE f.registry_id=? AND f.validator_id=v.validator_id AND f.cycle_id>=?)
|
||||
`, ValidatorIdle, ValidatorAssigned, ValidatorChecking, registryID, roundStart).Scan(&left); err != nil {
|
||||
return out, err
|
||||
}
|
||||
if left == 0 {
|
||||
out.NewRound = true
|
||||
roundStart = int64(newCycle)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, owner_validator_id=NULL, fip_id='', attempt_number=attempt_number+1,
|
||||
cycle_id=?, sc_failures=?, sc_round_start_cycle=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPQueued, newCycle, failures, roundStart, now, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx, freeValidatorSQL, now, validatorID, ipID); err != nil {
|
||||
return out, err
|
||||
}
|
||||
return out, tx.Commit()
|
||||
}
|
||||
|
||||
// GetIPRunID returns the check run the queue row belongs to (0 if none).
|
||||
func (d *DB) GetIPRunID(ctx context.Context, ipID int64) (int64, error) {
|
||||
var runID sql.NullInt64
|
||||
if err := d.QueryRowContext(ctx, `SELECT run_id FROM ip_queue WHERE id=?`, ipID).Scan(&runID); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return runID.Int64, nil
|
||||
}
|
||||
|
||||
// ListSelfCheckFailedOn returns the distinct validators whose self-check of
|
||||
// the address failed, in alphabetical order: over the whole history of the
|
||||
// address, or, if runID > 0, only the failures that happened in that run.
|
||||
func (d *DB) ListSelfCheckFailedOn(ctx context.Context, registryID, runID int64) ([]string, error) {
|
||||
q := `SELECT DISTINCT validator_id FROM ip_self_check_failures WHERE registry_id=?`
|
||||
args := []any{registryID}
|
||||
if runID > 0 {
|
||||
q += ` AND run_id=?`
|
||||
args = append(args, runID)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx, q+` ORDER BY validator_id`, args...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
out := []string{}
|
||||
for rows.Next() {
|
||||
var v string
|
||||
if err := rows.Scan(&v); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// SubmitIPs is the single admin entry point for both "add new addresses to
|
||||
// the queue" and "force a re-check of an already-finished address" — the
|
||||
// same list can freely mix both. Addresses are processed in one
|
||||
@@ -360,7 +538,9 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
// - unknown address: inserted as a new queued row.
|
||||
// - address currently done/failed/occupied: reset to queued (new attempt,
|
||||
// retry_count cleared — this is a deliberate admin-triggered restart,
|
||||
// not a system retry).
|
||||
// not a system retry). It also starts a new series of self-check
|
||||
// failures (sc_failures=0, validators that failed it before are no
|
||||
// longer excluded); the history in ip_self_check_failures stays.
|
||||
// - address currently queued (not yet claimed): left in state=queued,
|
||||
// only its sequence is updated.
|
||||
// - address currently mid-check (assigning_fip / awaiting_self_check /
|
||||
@@ -426,9 +606,9 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
return result, rErr
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, sc_round_start_cycle, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, cycle, now, now); err != nil {
|
||||
return result, fmt.Errorf("insert %s: %w", addr, err)
|
||||
}
|
||||
result.Added = append(result.Added, addr)
|
||||
@@ -453,10 +633,10 @@ func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (
|
||||
UPDATE ip_queue SET
|
||||
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
|
||||
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', run_id=?,
|
||||
overall_result='', run_id=?, sc_failures=0, sc_round_start_cycle=?,
|
||||
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
WHERE ip_address=?
|
||||
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
|
||||
`, IPQueued, seq, cycle, rid, cycle, now, addr); err != nil {
|
||||
return result, fmt.Errorf("requeue %s: %w", addr, err)
|
||||
}
|
||||
result.Requeued = append(result.Requeued, addr)
|
||||
|
||||
@@ -383,7 +383,7 @@ func TestMigration0011BuildsRunsFromExistingData(t *testing.T) {
|
||||
}
|
||||
var ver int
|
||||
d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver)
|
||||
if ver != 11 {
|
||||
if ver != 12 {
|
||||
t.Errorf("user_version = %d", ver)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
// claimAwaitingSelfCheck claims the next address for the validator and brings
|
||||
// it to awaiting_self_check, as the orchestrator does after the association.
|
||||
func claimAwaitingSelfCheck(t *testing.T, ctx context.Context, d *DB, validatorID string) *IPQueueItem {
|
||||
t.Helper()
|
||||
item, err := d.ClaimNextQueued(ctx, validatorID, time.Minute)
|
||||
if err != nil || item == nil {
|
||||
t.Fatalf("claim for %s: item=%+v err=%v", validatorID, item, err)
|
||||
}
|
||||
if err := d.SetFIPAssociated(ctx, item.ID, "fip-"+item.IPAddress, time.Minute); err != nil {
|
||||
t.Fatalf("set fip associated: %v", err)
|
||||
}
|
||||
return item
|
||||
}
|
||||
|
||||
// A validator that failed the self-check of an address is not given that
|
||||
// address again, takes the next one instead, and another validator takes the
|
||||
// excluded one; the exclusion covers that address only.
|
||||
func TestClaimSkipsAddressExcludedForValidator(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
first := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if first.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected 1.1.1.1 first, got %s", first.IPAddress)
|
||||
}
|
||||
res, err := d.FailSelfCheck(ctx, first.ID, "v1", "ip echo timeout", 5)
|
||||
if err != nil {
|
||||
t.Fatalf("fail self-check: %v", err)
|
||||
}
|
||||
if res.Failed || res.NewRound || res.Failures != 1 {
|
||||
t.Fatalf("expected a plain retry after the first failure, got %+v", res)
|
||||
}
|
||||
back, _ := d.GetIP(ctx, first.ID)
|
||||
if back.State != IPQueued || back.RetryCount != 0 {
|
||||
t.Fatalf("expected queued with retry_count untouched, got state=%s retry_count=%d", back.State, back.RetryCount)
|
||||
}
|
||||
if v, _ := d.GetValidator(ctx, "v1"); v.State != ValidatorIdle {
|
||||
t.Fatalf("expected v1 freed to idle, got %s", v.State)
|
||||
}
|
||||
|
||||
// v1 skips 1.1.1.1 (still ahead in the queue) and takes 2.2.2.2.
|
||||
second, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
|
||||
if err != nil || second == nil || second.IPAddress != "2.2.2.2" {
|
||||
t.Fatalf("expected v1 to take 2.2.2.2, got %+v err=%v", second, err)
|
||||
}
|
||||
// v2 takes the excluded address.
|
||||
other, err := d.ClaimNextQueued(ctx, "v2", time.Minute)
|
||||
if err != nil || other == nil || other.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected v2 to take 1.1.1.1, got %+v err=%v", other, err)
|
||||
}
|
||||
}
|
||||
|
||||
// When every working validator has failed the address in the round, a new
|
||||
// round starts and the exclusions lapse; an unreachable validator does not
|
||||
// count as working.
|
||||
func TestSelfCheckNewRoundLiftsExclusions(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if _, err := d.Exec(`UPDATE validators SET state=? WHERE validator_id='v2'`, ValidatorUnreachable); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
res, err := d.FailSelfCheck(ctx, item.ID, "v1", "x", 5)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.NewRound {
|
||||
t.Fatalf("expected a new round (the only working validator failed it), got %+v", res)
|
||||
}
|
||||
again, err := d.ClaimNextQueued(ctx, "v1", time.Minute)
|
||||
if err != nil || again == nil || again.IPAddress != "1.1.1.1" {
|
||||
t.Fatalf("expected v1 to get the address again in the new round, got %+v err=%v", again, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Reaching the ceiling gives the verdict fail; a re-submission starts a new
|
||||
// series (counter and exclusions) but keeps the failure history.
|
||||
func TestSelfCheckCeilingAndResubmit(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if res, err := d.FailSelfCheck(ctx, item.ID, "v1", "x", 2); err != nil || res.Failed {
|
||||
t.Fatalf("below the ceiling: res=%+v err=%v", res, err)
|
||||
}
|
||||
item = claimAwaitingSelfCheck(t, ctx, d, "v2")
|
||||
res, err := d.FailSelfCheck(ctx, item.ID, "v2", "y", 2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !res.Failed || res.Failures != 2 || len(res.Validators) != 2 || res.Validators[0] != "v1" || res.Validators[1] != "v2" {
|
||||
t.Fatalf("expected fail at the ceiling on v1, v2, got %+v", res)
|
||||
}
|
||||
ip, _ := d.GetIP(ctx, item.ID)
|
||||
if ip.State != IPFailed || ip.OverallResult != ResultFail {
|
||||
t.Fatalf("expected failed/fail, got %s/%s", ip.State, ip.OverallResult)
|
||||
}
|
||||
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var failures int
|
||||
if err := d.QueryRow(`SELECT sc_failures FROM ip_queue WHERE id=?`, item.ID).Scan(&failures); err != nil || failures != 0 {
|
||||
t.Fatalf("expected the series reset to 0, got %d err=%v", failures, err)
|
||||
}
|
||||
if got, err := d.ListSelfCheckFailedOn(ctx, ip.RegistryID, 0); err != nil || len(got) != 2 || got[0] != "v1" || got[1] != "v2" {
|
||||
t.Fatalf("expected the history kept (v1, v2), got %v err=%v", got, err)
|
||||
}
|
||||
// v1 failed it before the re-submission, but is no longer excluded.
|
||||
if again, err := d.ClaimNextQueued(ctx, "v1", time.Minute); err != nil || again == nil {
|
||||
t.Fatalf("expected v1 to claim the re-submitted address, got %+v err=%v", again, err)
|
||||
}
|
||||
}
|
||||
|
||||
// A late report for an address the validator does not hold is refused.
|
||||
func TestFailSelfCheckRequiresAwaitingSelfCheckOnValidator(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range []string{"v1", "v2"} {
|
||||
if err := d.AdminCreateValidator(ctx, v, "port-"+v); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
item := claimAwaitingSelfCheck(t, ctx, d, "v1")
|
||||
if _, err := d.FailSelfCheck(ctx, item.ID, "v2", "late", 5); !errors.Is(err, ErrInvalidState) {
|
||||
t.Fatalf("expected ErrInvalidState for another validator, got %v", err)
|
||||
}
|
||||
cur, _ := d.GetIP(ctx, item.ID)
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, cur.RegistryID, 0); len(got) != 0 {
|
||||
t.Fatalf("a refused report must leave no history, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelfCheckMaxAttemptsSetting(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.BootstrapFromConfig(ctx, &config.ControlAPI{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s, err := d.GetSettings(ctx); err != nil || s.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected default 5, got %+v err=%v", s, err)
|
||||
}
|
||||
for _, bad := range []int{0, -1, 51} {
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, bad); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation for %d, got %v", bad, err)
|
||||
}
|
||||
}
|
||||
for _, ok := range []int{1, 50} {
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, ok); err != nil {
|
||||
t.Fatalf("set %d: %v", ok, err)
|
||||
}
|
||||
if s, _ := d.GetSettings(ctx); s.SelfCheckMaxAttempts != ok {
|
||||
t.Fatalf("expected %d, got %d", ok, s.SelfCheckMaxAttempts)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -13,8 +13,8 @@ func (d *DB) GetSettings(ctx context.Context) (Settings, error) {
|
||||
var s Settings
|
||||
var createdAt, updatedAt string
|
||||
err := d.QueryRowContext(ctx, `
|
||||
SELECT fip_settle_seconds, history_retention_cycles, created_at, updated_at FROM settings WHERE id=1
|
||||
`).Scan(&s.FIPSettleSeconds, &s.HistoryRetentionCycles, &createdAt, &updatedAt)
|
||||
SELECT fip_settle_seconds, history_retention_cycles, self_check_max_attempts, created_at, updated_at FROM settings WHERE id=1
|
||||
`).Scan(&s.FIPSettleSeconds, &s.HistoryRetentionCycles, &s.SelfCheckMaxAttempts, &createdAt, &updatedAt)
|
||||
if err != nil {
|
||||
return Settings{}, err
|
||||
}
|
||||
@@ -58,3 +58,17 @@ func (d *DB) SetHistoryRetentionCycles(ctx context.Context, cycles int) error {
|
||||
`, cycles, now)
|
||||
return err
|
||||
}
|
||||
|
||||
// SetSelfCheckMaxAttempts persists the ceiling of failed self-checks per
|
||||
// address (1..MaxSelfCheckMaxAttempts). It applies from the next failed
|
||||
// self-check, without a restart.
|
||||
func (d *DB) SetSelfCheckMaxAttempts(ctx context.Context, attempts int) error {
|
||||
if attempts < MinSelfCheckMaxAttempts || attempts > MaxSelfCheckMaxAttempts {
|
||||
return fmt.Errorf("self_check_max_attempts must be in %d..%d: %w", MinSelfCheckMaxAttempts, MaxSelfCheckMaxAttempts, ErrValidation)
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE settings SET self_check_max_attempts=?, updated_at=? WHERE id=1
|
||||
`, attempts, now)
|
||||
return err
|
||||
}
|
||||
@@ -240,7 +240,7 @@ func TestMigration0010MarksRowsAfterVerdict(t *testing.T) {
|
||||
t.Errorf("ssh: after_verdict=%d recorded=%s created=%s", a, rec, cr)
|
||||
}
|
||||
var ver int
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 11 {
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 12 {
|
||||
t.Errorf("user_version=%d err=%v", ver, err)
|
||||
}
|
||||
}
|
||||
@@ -162,12 +162,21 @@ type putCheckTypeRequest struct {
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
// orchestratorSettingsDTO doubles as both the GET response and the PUT
|
||||
// request body for /api/v1/admin/config/orchestrator — a single-field DTO,
|
||||
// same shape both ways, like putSiteRequest/siteDTO.
|
||||
// orchestratorSettingsDTO is the GET response for
|
||||
// /api/v1/admin/config/orchestrator.
|
||||
type orchestratorSettingsDTO struct {
|
||||
FIPSettleSeconds int `json:"fip_settle_seconds"`
|
||||
HistoryRetentionCycles int `json:"history_retention_cycles"`
|
||||
SelfCheckMaxAttempts int `json:"self_check_max_attempts"`
|
||||
}
|
||||
|
||||
// putOrchestratorSettingsRequest is the PUT body. SelfCheckMaxAttempts is a
|
||||
// pointer so that a client written before the field existed (it sends only
|
||||
// the first two) leaves the ceiling unchanged instead of failing validation.
|
||||
type putOrchestratorSettingsRequest struct {
|
||||
FIPSettleSeconds int `json:"fip_settle_seconds"`
|
||||
HistoryRetentionCycles int `json:"history_retention_cycles"`
|
||||
SelfCheckMaxAttempts *int `json:"self_check_max_attempts"`
|
||||
}
|
||||
|
||||
// inboundChecksDTO doubles as both the GET response and the PUT request
|
||||
|
||||
@@ -178,11 +178,23 @@ func (s *Server) handleAdminIPDetail(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
// Self-check failures of this address in its current run.
|
||||
runID, err := s.DB.GetIPRunID(r.Context(), item.ID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), item.RegistryID, runID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, struct {
|
||||
IP *db.IPQueueItem `json:"ip"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
Events []db.Event `json:"events"`
|
||||
}{item, checks, events})
|
||||
IP *db.IPQueueItem `json:"ip"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
Events []db.Event `json:"events"`
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}{item, checks, events, failedOn})
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strconv"
|
||||
|
||||
@@ -203,6 +204,7 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: settings.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: settings.HistoryRetentionCycles,
|
||||
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -211,11 +213,17 @@ func (s *Server) handleConfigGetOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
// needs cross-field validation against the static lease_ttl_seconds/
|
||||
// self_check_timeout_seconds config, which only the orchestrator has.
|
||||
func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *http.Request) {
|
||||
var req orchestratorSettingsDTO
|
||||
var req putOrchestratorSettingsRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
// Checked before anything is saved, so a bad ceiling does not leave the
|
||||
// other two settings half-applied. An omitted field keeps the current value.
|
||||
if a := req.SelfCheckMaxAttempts; a != nil && (*a < db.MinSelfCheckMaxAttempts || *a > db.MaxSelfCheckMaxAttempts) {
|
||||
writeError(w, http.StatusBadRequest, fmt.Sprintf("self_check_max_attempts must be in %d..%d", db.MinSelfCheckMaxAttempts, db.MaxSelfCheckMaxAttempts))
|
||||
return
|
||||
}
|
||||
if err := s.Orch.SetFIPSettleSeconds(r.Context(), req.FIPSettleSeconds); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
@@ -224,9 +232,22 @@ func (s *Server) handleConfigPutOrchestratorSettings(w http.ResponseWriter, r *h
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
if req.SelfCheckMaxAttempts != nil {
|
||||
if err := s.DB.SetSelfCheckMaxAttempts(r.Context(), *req.SelfCheckMaxAttempts); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
// Answer with what is stored now, so an omitted ceiling shows its current value.
|
||||
settings, err := s.DB.GetSettings(r.Context())
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, orchestratorSettingsDTO{
|
||||
FIPSettleSeconds: req.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: req.HistoryRetentionCycles,
|
||||
FIPSettleSeconds: settings.FIPSettleSeconds,
|
||||
HistoryRetentionCycles: settings.HistoryRetentionCycles,
|
||||
SelfCheckMaxAttempts: settings.SelfCheckMaxAttempts,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -426,19 +426,32 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal get response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 0 {
|
||||
t.Fatalf("expected default 0, got %+v", got)
|
||||
if got.FIPSettleSeconds != 0 || got.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected defaults 0 and 5, got %+v", got)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 20})
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal put response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 20 {
|
||||
t.Fatalf("expected 20, got %+v", got)
|
||||
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 5 {
|
||||
t.Fatalf("expected 20 with the omitted ceiling kept at 5, got %+v", got)
|
||||
}
|
||||
|
||||
eight := 8
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &eight})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings with ceiling: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
for _, bad := range []int{0, 51} {
|
||||
bad := bad
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 20, SelfCheckMaxAttempts: &bad})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for self_check_max_attempts=%d, status=%d body=%s", bad, resp.StatusCode, body)
|
||||
}
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/config/orchestrator", nil)
|
||||
@@ -448,8 +461,8 @@ func TestOrchestratorSettingsGetPut(t *testing.T) {
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal get-after-put response: %v", err)
|
||||
}
|
||||
if got.FIPSettleSeconds != 20 {
|
||||
t.Fatalf("expected 20 to persist, got %+v", got)
|
||||
if got.FIPSettleSeconds != 20 || got.SelfCheckMaxAttempts != 8 {
|
||||
t.Fatalf("expected 20 and 8 to persist, got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -459,12 +472,12 @@ func TestOrchestratorSettingsPutValidation(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
// newConfigTestHarness: LeaseTTLSeconds=180, SelfCheckTimeoutSeconds=10.
|
||||
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 175})
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 175})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for settle seconds too close to lease ttl, status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: -1})
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: -1})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("expected 400 for negative settle seconds, status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
@@ -479,7 +492,7 @@ func TestFIPSettleDelayGatesAssignmentEndpoint(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "9.9.9.9", "svc-project")
|
||||
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", orchestratorSettingsDTO{FIPSettleSeconds: 1})
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/config/orchestrator", putOrchestratorSettingsRequest{FIPSettleSeconds: 1})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put settings: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
@@ -81,10 +81,17 @@ func (s *Server) handleAdminRegistryHistory(w http.ResponseWriter, r *http.Reque
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
// Self-check failures over every run of the address.
|
||||
failedOn, err := s.DB.ListSelfCheckFailedOn(r.Context(), summary.ID, 0)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, struct {
|
||||
Registry registryDTO `json:"registry"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
}{registrySummaryToDTO(*summary), checks})
|
||||
Registry registryDTO `json:"registry"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}{registrySummaryToDTO(*summary), checks, failedOn})
|
||||
}
|
||||
|
||||
func registrySummaryToDTO(s db.RegistrySummary) registryDTO {
|
||||
|
||||
@@ -219,3 +219,51 @@ func TestAdminRegistryLevels(t *testing.T) {
|
||||
t.Errorf("empty level must serialise by_type as [], got %s", body)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSelfCheckFailedOnInAddressEndpoints proves GET /admin/ips/{ip} and GET
|
||||
// /admin/registry/{ip} list the validators whose self-check of the address
|
||||
// failed ([] when none).
|
||||
func TestSelfCheckFailedOnInAddressEndpoints(t *testing.T) {
|
||||
fc, d, orch, mock := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "9.9.9.9", "svc-project")
|
||||
fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}})
|
||||
|
||||
failedOn := func(path string) []string {
|
||||
t.Helper()
|
||||
resp, body := fc.do(http.MethodGet, path, nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get %s: status=%d body=%s", path, resp.StatusCode, body)
|
||||
}
|
||||
var got struct {
|
||||
SelfCheckFailedOn []string `json:"self_check_failed_on"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &got); err != nil {
|
||||
t.Fatalf("unmarshal %s: %v", path, err)
|
||||
}
|
||||
if got.SelfCheckFailedOn == nil {
|
||||
t.Fatalf("%s: self_check_failed_on must be [] rather than null, body=%s", path, body)
|
||||
}
|
||||
return got.SelfCheckFailedOn
|
||||
}
|
||||
if got := failedOn("/api/v1/admin/ips/9.9.9.9"); len(got) != 0 {
|
||||
t.Fatalf("expected no failures yet, got %v", got)
|
||||
}
|
||||
|
||||
orch.Tick(ctx) // claim + associate -> awaiting_self_check
|
||||
ip, err := d.GetIPByAddress(ctx, "9.9.9.9")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := orch.SelfCheckResult(ctx, "validator-1", ip.ID, false, "ip echo timeout"); err != nil {
|
||||
t.Fatalf("self-check result: %v", err)
|
||||
}
|
||||
|
||||
for _, path := range []string{"/api/v1/admin/ips/9.9.9.9", "/api/v1/admin/registry/9.9.9.9"} {
|
||||
if got := failedOn(path); !reflect.DeepEqual(got, []string{"validator-1"}) {
|
||||
t.Fatalf("%s: expected [validator-1], got %v", path, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -243,9 +244,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
|
||||
return nil
|
||||
}
|
||||
// Detach the floating IP before the address goes back to the queue.
|
||||
// requeueOrFail frees the validator in the database but knows nothing
|
||||
// about the cloud: a floating IP left on the validator's port makes
|
||||
// every later association on that port fail with 409 ("fixed IP
|
||||
// The database side (db.FailSelfCheck) frees the validator but knows
|
||||
// nothing about the cloud: a floating IP left on the validator's port
|
||||
// makes every later association on that port fail with 409 ("fixed IP
|
||||
// already has a floating IP"). Best-effort, like the other release
|
||||
// paths — the database state must be freed even if Neutron hiccups.
|
||||
if item.FIPID != "" {
|
||||
@@ -253,20 +254,48 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
|
||||
o.Log.Error("disassociate fip after failed self-check", "ip_id", ipID, "fip_id", item.FIPID, "err", err)
|
||||
}
|
||||
}
|
||||
if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries {
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail)
|
||||
return nil
|
||||
}
|
||||
// Retry association without fully requeuing: re-drive the same
|
||||
// claim by cycling back through requeue/claim keeps the logic in
|
||||
// one place at the cost of the IP briefly returning to `queued`.
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail)
|
||||
return nil
|
||||
return o.failSelfCheck(ctx, item, validatorID, detail)
|
||||
}
|
||||
|
||||
return o.DB.SetChecking(ctx, ipID, o.leaseTTL())
|
||||
}
|
||||
|
||||
// failSelfCheck records a failed self-check and decides what happens to the
|
||||
// address (see db.FailSelfCheck): the ceiling self_check_max_attempts is read
|
||||
// from the settings on every failure, so a change applies to the next one. The
|
||||
// retry_count / max_retries pair is not involved: a failed self-check has its
|
||||
// own ceiling, and the validator that failed it is not handed this address
|
||||
// again in the current round (db.ClaimNextQueued). The validator itself stays
|
||||
// in service.
|
||||
func (o *Orchestrator) failSelfCheck(ctx context.Context, item *db.IPQueueItem, validatorID, detail string) error {
|
||||
settings, err := o.DB.GetSettings(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res, err := o.DB.FailSelfCheck(ctx, item.ID, validatorID, detail, settings.SelfCheckMaxAttempts)
|
||||
if errors.Is(err, db.ErrInvalidState) {
|
||||
o.Log.Warn("ignoring failed self-check for an address the validator does not hold",
|
||||
"validator", validatorID, "ip_id", item.ID)
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if res.Failed {
|
||||
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
|
||||
fmt.Sprintf(`{"reason":%q}`, fmt.Sprintf("self-check failed %d times (on %s), giving up: %s",
|
||||
res.Failures, strings.Join(res.Validators, ", "), detail)))
|
||||
return nil
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
|
||||
fmt.Sprintf(`{"reason":%q}`, "self-check failed, retrying: "+detail))
|
||||
if !res.NewRound {
|
||||
o.event(ctx, "control-api", "", &item.ID, "validator_excluded",
|
||||
fmt.Sprintf(`{"validator_id":%q,"failures":%d}`, validatorID, res.Failures))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// AssignmentForValidator returns the check config for a validator's current
|
||||
// IP if it's ready to be worked on (awaiting_self_check or checking),
|
||||
// or nil if the validator has nothing to do right now. The check config is
|
||||
|
||||
@@ -1162,3 +1162,169 @@ func TestLateFailedSelfCheckIsIgnored(t *testing.T) {
|
||||
t.Fatalf("a late report changed the address state to %s", cur.State)
|
||||
}
|
||||
}
|
||||
|
||||
// reportSelfChecks answers the self-check of every address that is waiting for
|
||||
// one: a validator in failing reports a failure, any other a success.
|
||||
func reportSelfChecks(t *testing.T, ctx context.Context, o *Orchestrator, d *db.DB, failing ...string) {
|
||||
t.Helper()
|
||||
fails := map[string]bool{}
|
||||
for _, v := range failing {
|
||||
fails[v] = true
|
||||
}
|
||||
validators, err := d.ListValidators(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range validators {
|
||||
if v.CurrentIPID == nil {
|
||||
continue
|
||||
}
|
||||
ip, err := d.GetIP(ctx, *v.CurrentIPID)
|
||||
if err != nil || ip.State != db.IPAwaitingSelfCheck {
|
||||
continue
|
||||
}
|
||||
if err := o.SelfCheckResult(ctx, v.ValidatorID, ip.ID, !fails[v.ValidatorID], "ip echo timeout"); err != nil {
|
||||
t.Fatalf("self-check result of %s: %v", v.ValidatorID, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func registerValidators(t *testing.T, ctx context.Context, d *db.DB, ids ...string) {
|
||||
t.Helper()
|
||||
for i, id := range ids {
|
||||
if err := d.RegisterValidator(ctx, id, "h", fmt.Sprintf("port-%d", i+1), "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func eventTypes(t *testing.T, ctx context.Context, d *db.DB, ipID int64) map[string]int {
|
||||
t.Helper()
|
||||
events, err := d.ListEventsForIP(ctx, ipID)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
out := map[string]int{}
|
||||
for _, e := range events {
|
||||
out[e.EventType]++
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// The incident: v1 fails the self-check of 1.1.1.1. The address must not go
|
||||
// back to v1; another validator takes it and passes, while v1 keeps taking
|
||||
// other addresses.
|
||||
func TestSelfCheckFailureExcludesValidatorForThatAddress(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
for i, a := range []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"} {
|
||||
mock.Seed(fmt.Sprintf("fip-%d", i+1), a, "svc-project")
|
||||
}
|
||||
registerValidators(t, ctx, d, "v1", "v2", "v3")
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
o.Tick(ctx) // v1 -> 1.1.1.1, v2 -> 2.2.2.2, v3 stays idle
|
||||
reportSelfChecks(t, ctx, o, d, "v1")
|
||||
a1, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if a1.State != db.IPQueued || a1.RetryCount != 0 {
|
||||
t.Fatalf("expected 1.1.1.1 back in the queue with retry_count 0, got %s/%d", a1.State, a1.RetryCount)
|
||||
}
|
||||
if v1, _ := d.GetValidator(ctx, "v1"); v1.State != db.ValidatorIdle {
|
||||
t.Fatalf("v1 must stay in service, got %s", v1.State)
|
||||
}
|
||||
|
||||
if _, err := d.SubmitIPs(ctx, []string{"3.3.3.3"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
o.Tick(ctx) // v1 skips 1.1.1.1 and takes 3.3.3.3; v3 takes 1.1.1.1
|
||||
a1, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
a3, _ := d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if a1.OwnerValidatorID == nil || *a1.OwnerValidatorID != "v3" {
|
||||
t.Fatalf("expected 1.1.1.1 on v3, got %v", a1.OwnerValidatorID)
|
||||
}
|
||||
if a3.OwnerValidatorID == nil || *a3.OwnerValidatorID != "v1" {
|
||||
t.Fatalf("expected 3.3.3.3 on v1, got %v", a3.OwnerValidatorID)
|
||||
}
|
||||
|
||||
reportSelfChecks(t, ctx, o, d) // v1 passes this time: the others are fine
|
||||
a1, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
a3, _ = d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if a1.State != db.IPChecking || a3.State != db.IPChecking {
|
||||
t.Fatalf("expected both addresses to pass the self-check, got %s and %s", a1.State, a3.State)
|
||||
}
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, a1.RegistryID, 0); len(got) != 1 || got[0] != "v1" {
|
||||
t.Fatalf("expected self-check failures on v1 only, got %v", got)
|
||||
}
|
||||
if ev := eventTypes(t, ctx, d, a1.ID); ev["validator_excluded"] != 1 {
|
||||
t.Fatalf("expected one validator_excluded event, got %v", ev)
|
||||
}
|
||||
}
|
||||
|
||||
// The ceiling does not depend on the number of validators and is read at each
|
||||
// failure: the address fails once it is reached, not earlier, even though the
|
||||
// third validator never tried it.
|
||||
func TestSelfCheckCeilingGivesFailAndFollowsSetting(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
registerValidators(t, ctx, d, "v1", "v2", "v3")
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
o.Tick(ctx)
|
||||
reportSelfChecks(t, ctx, o, d, "v1", "v2", "v3")
|
||||
if ip, _ := d.GetIPByAddress(ctx, "1.1.1.1"); ip.State != db.IPQueued {
|
||||
t.Fatalf("expected queued after the first failure (ceiling 5), got %s", ip.State)
|
||||
}
|
||||
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, 2); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
o.Tick(ctx)
|
||||
reportSelfChecks(t, ctx, o, d, "v1", "v2", "v3")
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if ip.State != db.IPFailed || ip.OverallResult != db.ResultFail || ip.RetryCount != 0 {
|
||||
t.Fatalf("expected failed/fail at the lowered ceiling 2, got %s/%s retry_count=%d", ip.State, ip.OverallResult, ip.RetryCount)
|
||||
}
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, ip.RegistryID, 0); len(got) != 2 {
|
||||
t.Fatalf("expected failures on two validators, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// With fewer working validators than the ceiling (down to a single one) the
|
||||
// retries continue on the validators in a new round, up to the ceiling; the
|
||||
// address never gets stuck in the queue.
|
||||
func TestSelfCheckFewerValidatorsThanCeilingRetriesUntilCeiling(t *testing.T) {
|
||||
for _, validators := range [][]string{{"v1"}, {"v1", "v2"}} {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
registerValidators(t, ctx, d, validators...)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, 5); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for attempt := 1; attempt <= 5; attempt++ {
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("%d validators, attempt %d: expected the address to be handed out, got %s", len(validators), attempt, ip.State)
|
||||
}
|
||||
reportSelfChecks(t, ctx, o, d, validators...)
|
||||
ip, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
want := db.IPQueued
|
||||
if attempt == 5 {
|
||||
want = db.IPFailed
|
||||
}
|
||||
if ip.State != want {
|
||||
t.Fatalf("%d validators, after failure %d: expected %s, got %s", len(validators), attempt, want, ip.State)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user