Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.
Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
b7669c9e41
commit
e95b5eb7d5
34 files changed
+1092
-82
No files matched your search
@@ -14,6 +14,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -243,9 +244,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
|
||||
return nil
|
||||
}
|
||||
// Detach the floating IP before the address goes back to the queue.
|
||||
// requeueOrFail frees the validator in the database but knows nothing
|
||||
// about the cloud: a floating IP left on the validator's port makes
|
||||
// every later association on that port fail with 409 ("fixed IP
|
||||
// The database side (db.FailSelfCheck) frees the validator but knows
|
||||
// nothing about the cloud: a floating IP left on the validator's port
|
||||
// makes every later association on that port fail with 409 ("fixed IP
|
||||
// already has a floating IP"). Best-effort, like the other release
|
||||
// paths — the database state must be freed even if Neutron hiccups.
|
||||
if item.FIPID != "" {
|
||||
@@ -253,20 +254,48 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
|
||||
o.Log.Error("disassociate fip after failed self-check", "ip_id", ipID, "fip_id", item.FIPID, "err", err)
|
||||
}
|
||||
}
|
||||
if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries {
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail)
|
||||
return nil
|
||||
}
|
||||
// Retry association without fully requeuing: re-drive the same
|
||||
// claim by cycling back through requeue/claim keeps the logic in
|
||||
// one place at the cost of the IP briefly returning to `queued`.
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail)
|
||||
return nil
|
||||
return o.failSelfCheck(ctx, item, validatorID, detail)
|
||||
}
|
||||
|
||||
return o.DB.SetChecking(ctx, ipID, o.leaseTTL())
|
||||
}
|
||||
|
||||
// failSelfCheck records a failed self-check and decides what happens to the
|
||||
// address (see db.FailSelfCheck): the ceiling self_check_max_attempts is read
|
||||
// from the settings on every failure, so a change applies to the next one. The
|
||||
// retry_count / max_retries pair is not involved: a failed self-check has its
|
||||
// own ceiling, and the validator that failed it is not handed this address
|
||||
// again in the current round (db.ClaimNextQueued). The validator itself stays
|
||||
// in service.
|
||||
func (o *Orchestrator) failSelfCheck(ctx context.Context, item *db.IPQueueItem, validatorID, detail string) error {
|
||||
settings, err := o.DB.GetSettings(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
res, err := o.DB.FailSelfCheck(ctx, item.ID, validatorID, detail, settings.SelfCheckMaxAttempts)
|
||||
if errors.Is(err, db.ErrInvalidState) {
|
||||
o.Log.Warn("ignoring failed self-check for an address the validator does not hold",
|
||||
"validator", validatorID, "ip_id", item.ID)
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if res.Failed {
|
||||
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
|
||||
fmt.Sprintf(`{"reason":%q}`, fmt.Sprintf("self-check failed %d times (on %s), giving up: %s",
|
||||
res.Failures, strings.Join(res.Validators, ", "), detail)))
|
||||
return nil
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "retry_or_fail",
|
||||
fmt.Sprintf(`{"reason":%q}`, "self-check failed, retrying: "+detail))
|
||||
if !res.NewRound {
|
||||
o.event(ctx, "control-api", "", &item.ID, "validator_excluded",
|
||||
fmt.Sprintf(`{"validator_id":%q,"failures":%d}`, validatorID, res.Failures))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// AssignmentForValidator returns the check config for a validator's current
|
||||
// IP if it's ready to be worked on (awaiting_self_check or checking),
|
||||
// or nil if the validator has nothing to do right now. The check config is
|
||||
|
||||
@@ -1162,3 +1162,169 @@ func TestLateFailedSelfCheckIsIgnored(t *testing.T) {
|
||||
t.Fatalf("a late report changed the address state to %s", cur.State)
|
||||
}
|
||||
}
|
||||
|
||||
// reportSelfChecks answers the self-check of every address that is waiting for
|
||||
// one: a validator in failing reports a failure, any other a success.
|
||||
func reportSelfChecks(t *testing.T, ctx context.Context, o *Orchestrator, d *db.DB, failing ...string) {
|
||||
t.Helper()
|
||||
fails := map[string]bool{}
|
||||
for _, v := range failing {
|
||||
fails[v] = true
|
||||
}
|
||||
validators, err := d.ListValidators(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, v := range validators {
|
||||
if v.CurrentIPID == nil {
|
||||
continue
|
||||
}
|
||||
ip, err := d.GetIP(ctx, *v.CurrentIPID)
|
||||
if err != nil || ip.State != db.IPAwaitingSelfCheck {
|
||||
continue
|
||||
}
|
||||
if err := o.SelfCheckResult(ctx, v.ValidatorID, ip.ID, !fails[v.ValidatorID], "ip echo timeout"); err != nil {
|
||||
t.Fatalf("self-check result of %s: %v", v.ValidatorID, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func registerValidators(t *testing.T, ctx context.Context, d *db.DB, ids ...string) {
|
||||
t.Helper()
|
||||
for i, id := range ids {
|
||||
if err := d.RegisterValidator(ctx, id, "h", fmt.Sprintf("port-%d", i+1), "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func eventTypes(t *testing.T, ctx context.Context, d *db.DB, ipID int64) map[string]int {
|
||||
t.Helper()
|
||||
events, err := d.ListEventsForIP(ctx, ipID)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
out := map[string]int{}
|
||||
for _, e := range events {
|
||||
out[e.EventType]++
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// The incident: v1 fails the self-check of 1.1.1.1. The address must not go
|
||||
// back to v1; another validator takes it and passes, while v1 keeps taking
|
||||
// other addresses.
|
||||
func TestSelfCheckFailureExcludesValidatorForThatAddress(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
for i, a := range []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"} {
|
||||
mock.Seed(fmt.Sprintf("fip-%d", i+1), a, "svc-project")
|
||||
}
|
||||
registerValidators(t, ctx, d, "v1", "v2", "v3")
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
o.Tick(ctx) // v1 -> 1.1.1.1, v2 -> 2.2.2.2, v3 stays idle
|
||||
reportSelfChecks(t, ctx, o, d, "v1")
|
||||
a1, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if a1.State != db.IPQueued || a1.RetryCount != 0 {
|
||||
t.Fatalf("expected 1.1.1.1 back in the queue with retry_count 0, got %s/%d", a1.State, a1.RetryCount)
|
||||
}
|
||||
if v1, _ := d.GetValidator(ctx, "v1"); v1.State != db.ValidatorIdle {
|
||||
t.Fatalf("v1 must stay in service, got %s", v1.State)
|
||||
}
|
||||
|
||||
if _, err := d.SubmitIPs(ctx, []string{"3.3.3.3"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
o.Tick(ctx) // v1 skips 1.1.1.1 and takes 3.3.3.3; v3 takes 1.1.1.1
|
||||
a1, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
a3, _ := d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if a1.OwnerValidatorID == nil || *a1.OwnerValidatorID != "v3" {
|
||||
t.Fatalf("expected 1.1.1.1 on v3, got %v", a1.OwnerValidatorID)
|
||||
}
|
||||
if a3.OwnerValidatorID == nil || *a3.OwnerValidatorID != "v1" {
|
||||
t.Fatalf("expected 3.3.3.3 on v1, got %v", a3.OwnerValidatorID)
|
||||
}
|
||||
|
||||
reportSelfChecks(t, ctx, o, d) // v1 passes this time: the others are fine
|
||||
a1, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
a3, _ = d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if a1.State != db.IPChecking || a3.State != db.IPChecking {
|
||||
t.Fatalf("expected both addresses to pass the self-check, got %s and %s", a1.State, a3.State)
|
||||
}
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, a1.RegistryID, 0); len(got) != 1 || got[0] != "v1" {
|
||||
t.Fatalf("expected self-check failures on v1 only, got %v", got)
|
||||
}
|
||||
if ev := eventTypes(t, ctx, d, a1.ID); ev["validator_excluded"] != 1 {
|
||||
t.Fatalf("expected one validator_excluded event, got %v", ev)
|
||||
}
|
||||
}
|
||||
|
||||
// The ceiling does not depend on the number of validators and is read at each
|
||||
// failure: the address fails once it is reached, not earlier, even though the
|
||||
// third validator never tried it.
|
||||
func TestSelfCheckCeilingGivesFailAndFollowsSetting(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
registerValidators(t, ctx, d, "v1", "v2", "v3")
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
o.Tick(ctx)
|
||||
reportSelfChecks(t, ctx, o, d, "v1", "v2", "v3")
|
||||
if ip, _ := d.GetIPByAddress(ctx, "1.1.1.1"); ip.State != db.IPQueued {
|
||||
t.Fatalf("expected queued after the first failure (ceiling 5), got %s", ip.State)
|
||||
}
|
||||
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, 2); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
o.Tick(ctx)
|
||||
reportSelfChecks(t, ctx, o, d, "v1", "v2", "v3")
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if ip.State != db.IPFailed || ip.OverallResult != db.ResultFail || ip.RetryCount != 0 {
|
||||
t.Fatalf("expected failed/fail at the lowered ceiling 2, got %s/%s retry_count=%d", ip.State, ip.OverallResult, ip.RetryCount)
|
||||
}
|
||||
if got, _ := d.ListSelfCheckFailedOn(ctx, ip.RegistryID, 0); len(got) != 2 {
|
||||
t.Fatalf("expected failures on two validators, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// With fewer working validators than the ceiling (down to a single one) the
|
||||
// retries continue on the validators in a new round, up to the ceiling; the
|
||||
// address never gets stuck in the queue.
|
||||
func TestSelfCheckFewerValidatorsThanCeilingRetriesUntilCeiling(t *testing.T) {
|
||||
for _, validators := range [][]string{{"v1"}, {"v1", "v2"}} {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
registerValidators(t, ctx, d, validators...)
|
||||
if err := d.SeedQueue(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SetSelfCheckMaxAttempts(ctx, 5); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for attempt := 1; attempt <= 5; attempt++ {
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("%d validators, attempt %d: expected the address to be handed out, got %s", len(validators), attempt, ip.State)
|
||||
}
|
||||
reportSelfChecks(t, ctx, o, d, validators...)
|
||||
ip, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
want := db.IPQueued
|
||||
if attempt == 5 {
|
||||
want = db.IPFailed
|
||||
}
|
||||
if ip.State != want {
|
||||
t.Fatalf("%d validators, after failure %d: expected %s, got %s", len(validators), attempt, want, ip.State)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user