Retry validator-agent/prober registration; show validator hostname

Both binaries registered with control-api exactly once at startup and
exited (os.Exit(1)) on any failure — including control-api simply not
being up yet (no ordering guarantee between the two at boot/redeploy) or
the admin not having added this validator_id/site_id to the config yet.
Run() now retries registration with capped exponential backoff (3s->30s)
until it succeeds or the process is asked to shut down, instead of
crashing; registerWithRetry is identical in agentcore and probercore
since their Run/register shape already was.

Separately, the admin dashboard's Validators page had no hostname column
even though the agent already reports one on register (mirroring the
prober) and control-api already persists it — only the admin-config read
DTO (validatorDTO in httpapi and dashboard) dropped it before it reached
the template. Added hostname + last_heartbeat_at to that DTO end-to-end
and a Хост/Heartbeat column to validators.html, matching sites.html.

Rebuilt bin/{control-api,admin-dashboard,prober,validator-agent} and
bin/SHA256SUMS per docs/SETUP.md's documented build recipe.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5 committed 2026-09-18 10:43:00 +03:00
1 parent 399c64e801
commit 95f8066eed
15 files changed
+320 -23

No files matched your search

+40 -5
View File
@@ -30,6 +30,13 @@ type Agent struct {
log *slog.Logger
lastHandledIPID int64
// registerRetryInitial/Max govern the backoff used while waiting for a
// successful registration (see registerWithRetry): control-api may not
// be up yet at agent boot, or may come and go across a redeploy, and the
// agent should keep waiting rather than exit.
registerRetryInitial time.Duration
registerRetryMax time.Duration
}
func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent {
@@ -38,17 +45,19 @@ func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent {
timeout = 10 * time.Second
}
return &Agent{
cfg: cfg,
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
log: log,
cfg: cfg,
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
log: log,
registerRetryInitial: 3 * time.Second,
registerRetryMax: 30 * time.Second,
}
}
// Run registers with the Control API and polls forever until ctx is
// cancelled.
func (a *Agent) Run(ctx context.Context) error {
if err := a.register(ctx); err != nil {
return fmt.Errorf("register: %w", err)
if err := a.registerWithRetry(ctx); err != nil {
return err
}
interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second
@@ -88,6 +97,32 @@ func (a *Agent) register(ctx context.Context) error {
return nil
}
// registerWithRetry retries register with capped exponential backoff until
// it succeeds or ctx is cancelled. Control-api may not be reachable yet at
// agent boot (started before control-api, or a network blip), or may reject
// the request until an admin adds this validator_id to its config — either
// way the agent should keep waiting rather than exit, since both conditions
// can resolve on their own after the agent has already started.
func (a *Agent) registerWithRetry(ctx context.Context) error {
delay := a.registerRetryInitial
for {
err := a.register(ctx)
if err == nil {
return nil
}
a.log.Warn("registration failed, will retry", "err", err, "retry_in", delay)
select {
case <-ctx.Done():
return ctx.Err()
case <-time.After(delay):
}
delay *= 2
if delay > a.registerRetryMax {
delay = a.registerRetryMax
}
}
}
type heartbeatReq struct {
LocalState string `json:"local_state"`
}
+100
View File
@@ -0,0 +1,100 @@
package agentcore
import (
"context"
"log/slog"
"net/http"
"net/http/httptest"
"os"
"sync/atomic"
"testing"
"time"
"cloudipvalidator/internal/apiclient"
"cloudipvalidator/internal/config"
)
func testLogger() *slog.Logger {
return slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
}
// TestRegisterWithRetrySucceedsAfterControlAPIComesUp mirrors starting the
// validator-agent before control-api is up (or before an admin has
// configured this validator_id): the first attempts fail, and the agent
// must keep retrying — not give up — until registration succeeds.
func TestRegisterWithRetrySucceedsAfterControlAPIComesUp(t *testing.T) {
var attempts int32
mux := http.NewServeMux()
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
n := atomic.AddInt32(&attempts, 1)
if n < 3 {
w.WriteHeader(http.StatusServiceUnavailable)
return
}
w.WriteHeader(http.StatusOK)
w.Write([]byte(`{"ok":true,"poll_interval_seconds":5}`))
})
ts := httptest.NewServer(mux)
defer ts.Close()
a := &Agent{
cfg: &config.ValidatorAgent{ValidatorID: "val-1", ControlAPIURL: ts.URL},
client: apiclient.New(ts.URL, 5*time.Second),
log: testLogger(),
registerRetryInitial: time.Millisecond,
registerRetryMax: 5 * time.Millisecond,
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
if err := a.registerWithRetry(ctx); err != nil {
t.Fatalf("expected eventual success, got err: %v", err)
}
if got := atomic.LoadInt32(&attempts); got != 3 {
t.Fatalf("expected exactly 3 attempts, got %d", got)
}
}
// TestRegisterWithRetryStopsOnCancel confirms a validator-agent waiting on
// an unreachable/unconfigured control-api can still be shut down promptly
// (e.g. via SIGTERM) instead of retrying forever with no way out.
func TestRegisterWithRetryStopsOnCancel(t *testing.T) {
mux := http.NewServeMux()
mux.HandleFunc("POST /api/v1/agents/register", func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusBadRequest)
})
ts := httptest.NewServer(mux)
defer ts.Close()
a := &Agent{
cfg: &config.ValidatorAgent{ValidatorID: "val-1", ControlAPIURL: ts.URL},
client: apiclient.New(ts.URL, 5*time.Second),
log: testLogger(),
registerRetryInitial: 10 * time.Millisecond,
registerRetryMax: 10 * time.Millisecond,
}
ctx, cancel := context.WithCancel(context.Background())
go func() {
time.Sleep(30 * time.Millisecond)
cancel()
}()
var err error
done := make(chan struct{})
go func() {
defer close(done)
err = a.registerWithRetry(ctx)
}()
select {
case <-done:
case <-time.After(2 * time.Second):
t.Fatal("registerWithRetry did not return after ctx cancellation")
}
if err != context.Canceled {
t.Fatalf("expected context.Canceled, got %v", err)
}
}