Retry validator-agent/prober registration; show validator hostname
Both binaries registered with control-api exactly once at startup and
exited (os.Exit(1)) on any failure — including control-api simply not
being up yet (no ordering guarantee between the two at boot/redeploy) or
the admin not having added this validator_id/site_id to the config yet.
Run() now retries registration with capped exponential backoff (3s->30s)
until it succeeds or the process is asked to shut down, instead of
crashing; registerWithRetry is identical in agentcore and probercore
since their Run/register shape already was.
Separately, the admin dashboard's Validators page had no hostname column
even though the agent already reports one on register (mirroring the
prober) and control-api already persists it — only the admin-config read
DTO (validatorDTO in httpapi and dashboard) dropped it before it reached
the template. Added hostname + last_heartbeat_at to that DTO end-to-end
and a Хост/Heartbeat column to validators.html, matching sites.html.
Rebuilt bin/{control-api,admin-dashboard,prober,validator-agent} and
bin/SHA256SUMS per docs/SETUP.md's documented build recipe.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
399c64e801
commit
95f8066eed
15 files changed
+320
-23
No files matched your search
@@ -30,6 +30,13 @@ type Agent struct {
|
||||
log *slog.Logger
|
||||
|
||||
lastHandledIPID int64
|
||||
|
||||
// registerRetryInitial/Max govern the backoff used while waiting for a
|
||||
// successful registration (see registerWithRetry): control-api may not
|
||||
// be up yet at agent boot, or may come and go across a redeploy, and the
|
||||
// agent should keep waiting rather than exit.
|
||||
registerRetryInitial time.Duration
|
||||
registerRetryMax time.Duration
|
||||
}
|
||||
|
||||
func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent {
|
||||
@@ -38,17 +45,19 @@ func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent {
|
||||
timeout = 10 * time.Second
|
||||
}
|
||||
return &Agent{
|
||||
cfg: cfg,
|
||||
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
|
||||
log: log,
|
||||
cfg: cfg,
|
||||
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
|
||||
log: log,
|
||||
registerRetryInitial: 3 * time.Second,
|
||||
registerRetryMax: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
// Run registers with the Control API and polls forever until ctx is
|
||||
// cancelled.
|
||||
func (a *Agent) Run(ctx context.Context) error {
|
||||
if err := a.register(ctx); err != nil {
|
||||
return fmt.Errorf("register: %w", err)
|
||||
if err := a.registerWithRetry(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second
|
||||
@@ -88,6 +97,32 @@ func (a *Agent) register(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// registerWithRetry retries register with capped exponential backoff until
|
||||
// it succeeds or ctx is cancelled. Control-api may not be reachable yet at
|
||||
// agent boot (started before control-api, or a network blip), or may reject
|
||||
// the request until an admin adds this validator_id to its config — either
|
||||
// way the agent should keep waiting rather than exit, since both conditions
|
||||
// can resolve on their own after the agent has already started.
|
||||
func (a *Agent) registerWithRetry(ctx context.Context) error {
|
||||
delay := a.registerRetryInitial
|
||||
for {
|
||||
err := a.register(ctx)
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
a.log.Warn("registration failed, will retry", "err", err, "retry_in", delay)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-time.After(delay):
|
||||
}
|
||||
delay *= 2
|
||||
if delay > a.registerRetryMax {
|
||||
delay = a.registerRetryMax
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type heartbeatReq struct {
|
||||
LocalState string `json:"local_state"`
|
||||
}
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
package agentcore
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/apiclient"
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
func testLogger() *slog.Logger {
|
||||
return slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
||||
}
|
||||
|
||||
// TestRegisterWithRetrySucceedsAfterControlAPIComesUp mirrors starting the
|
||||
// validator-agent before control-api is up (or before an admin has
|
||||
// configured this validator_id): the first attempts fail, and the agent
|
||||
// must keep retrying — not give up — until registration succeeds.
|
||||
func TestRegisterWithRetrySucceedsAfterControlAPIComesUp(t *testing.T) {
|
||||
var attempts int32
|
||||
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
|
||||
n := atomic.AddInt32(&attempts, 1)
|
||||
if n < 3 {
|
||||
w.WriteHeader(http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusOK)
|
||||
w.Write([]byte(`{"ok":true,"poll_interval_seconds":5}`))
|
||||
})
|
||||
ts := httptest.NewServer(mux)
|
||||
defer ts.Close()
|
||||
|
||||
a := &Agent{
|
||||
cfg: &config.ValidatorAgent{ValidatorID: "val-1", ControlAPIURL: ts.URL},
|
||||
client: apiclient.New(ts.URL, 5*time.Second),
|
||||
log: testLogger(),
|
||||
registerRetryInitial: time.Millisecond,
|
||||
registerRetryMax: 5 * time.Millisecond,
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
||||
defer cancel()
|
||||
|
||||
if err := a.registerWithRetry(ctx); err != nil {
|
||||
t.Fatalf("expected eventual success, got err: %v", err)
|
||||
}
|
||||
if got := atomic.LoadInt32(&attempts); got != 3 {
|
||||
t.Fatalf("expected exactly 3 attempts, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRegisterWithRetryStopsOnCancel confirms a validator-agent waiting on
|
||||
// an unreachable/unconfigured control-api can still be shut down promptly
|
||||
// (e.g. via SIGTERM) instead of retrying forever with no way out.
|
||||
func TestRegisterWithRetryStopsOnCancel(t *testing.T) {
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusBadRequest)
|
||||
})
|
||||
ts := httptest.NewServer(mux)
|
||||
defer ts.Close()
|
||||
|
||||
a := &Agent{
|
||||
cfg: &config.ValidatorAgent{ValidatorID: "val-1", ControlAPIURL: ts.URL},
|
||||
client: apiclient.New(ts.URL, 5*time.Second),
|
||||
log: testLogger(),
|
||||
registerRetryInitial: 10 * time.Millisecond,
|
||||
registerRetryMax: 10 * time.Millisecond,
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
go func() {
|
||||
time.Sleep(30 * time.Millisecond)
|
||||
cancel()
|
||||
}()
|
||||
|
||||
var err error
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
defer close(done)
|
||||
err = a.registerWithRetry(ctx)
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("registerWithRetry did not return after ctx cancellation")
|
||||
}
|
||||
if err != context.Canceled {
|
||||
t.Fatalf("expected context.Canceled, got %v", err)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user