New external site management. New external prober heartbeat feature.

This commit is contained in:
ayurishchev committed 2026-08-26 20:47:54 +03:00
1 parent 42f584dd3c
commit ef24cc9858
38 files changed
+1058 -158

No files matched your search

+38 -22
View File
@@ -401,7 +401,12 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
return fmt.Errorf("list sites: %w", err)
}
for _, item := range checking {
if !o.isReadyToAggregate(item, deadline, sites) {
ready, err := o.isReadyToAggregate(ctx, item, deadline, sites)
if err != nil {
o.Log.Error("check ready to aggregate", "ip_id", item.ID, "err", err)
continue
}
if !ready {
continue
}
if err := o.aggregateAndRelease(ctx, item); err != nil {
@@ -416,34 +421,26 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
// checking-window deadline. Which inbound sources it's expecting is driven
// entirely by the currently configured sites — inbound checks are optional:
// an empty (or partial) sites configuration means this IP is ready as soon
// as egress completes (or after the corresponding subset of siteN_complete
// flags), with no need to wait on a prober that will never exist. This is
// what makes inbound checks genuinely opt-in rather than a hardcoded
// expectation of exactly three sites.
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time, sites []db.Site) bool {
// as egress completes (or after the corresponding subset of sites has
// reported in ip_site_checks), with no need to wait on a prober that will
// never exist, and no cap on how many sites can be configured.
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
return true
return true, nil
}
if !item.EgressComplete {
return false
return false, nil
}
completed, err := o.DB.ListCompletedSiteIndices(ctx, item.ID, item.AttemptNumber)
if err != nil {
return false, err
}
for _, s := range sites {
switch s.Index {
case 1:
if !item.Site1Complete {
return false
}
case 2:
if !item.Site2Complete {
return false
}
case 3:
if !item.Site3Complete {
return false
}
if !completed[s.Index] {
return false, nil
}
}
return true
return true, nil
}
func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueItem) error {
@@ -585,6 +582,25 @@ func (o *Orchestrator) SweepStaleHeartbeats(ctx context.Context) error {
return nil
}
// SweepStaleSiteHeartbeats marks prober sites unreachable if they haven't
// heartbeated within HeartbeatTimeoutSeconds — mirrors SweepStaleHeartbeats
// exactly, for sites instead of validators.
func (o *Orchestrator) SweepStaleSiteHeartbeats(ctx context.Context) error {
cutoff := db.Now().Add(-time.Duration(o.Cfg.HeartbeatTimeoutSeconds) * time.Second)
stale, err := o.DB.ListStaleSiteHeartbeats(ctx, cutoff)
if err != nil {
return err
}
for _, s := range stale {
if err := o.DB.MarkSiteUnreachable(ctx, s.SiteID); err != nil {
o.Log.Error("mark site unreachable", "site_id", s.SiteID, "err", err)
continue
}
o.event(ctx, "control-api", "", nil, "site_unreachable", fmt.Sprintf(`{"site_id":%q}`, s.SiteID))
}
return nil
}
// RecordEvent is the exported entry point httpapi uses to log
// agent/prober-reported audit events (config_received, fip_changed,
// error, etc.) through the same path as internally generated events.
@@ -624,3 +624,65 @@ func TestExpectedCheckCountReflectsInboundChecksConfigChange(t *testing.T) {
t.Fatalf("expected pass, got %s", ip.OverallResult)
}
}
// TestFourSitesAllMustReportBeforeAggregation confirms there's no hardcoded
// cap of three sites: with 4 sites configured, aggregation must wait on
// all 4, not silently treat the 4th as always-complete.
func TestFourSitesAllMustReportBeforeAggregation(t *testing.T) {
ctx := context.Background()
sites := []config.SiteConfig{
{SiteID: "site-1", Index: 1}, {SiteID: "site-2", Index: 2},
{SiteID: "site-3", Index: 3}, {SiteID: "site-4", Index: 4},
}
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
mock.Seed("fip-1", "1.2.3.4", "svc-project")
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
o.Tick(ctx)
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
ip, _ = d.GetIP(ctx, ip.ID)
_ = o.RecordCheck(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
})
_ = o.MarkEgressComplete(ctx, ip.ID)
for _, site := range []int{1, 2, 3} {
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
_ = o.RecordCheck(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
})
}
_ = o.MarkSiteComplete(ctx, ip.ID, site)
}
// Sites 1-3 reported, site 4 (the case beyond the old fixed-3 cap)
// hasn't — must not aggregate yet.
o.Tick(ctx)
ip, _ = d.GetIP(ctx, ip.ID)
if ip.State != db.IPChecking {
t.Fatalf("expected still checking (site-4 pending), got %s", ip.State)
}
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
_ = o.RecordCheck(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
Source: db.InboundSource(4), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
})
}
_ = o.MarkSiteComplete(ctx, ip.ID, 4)
o.Tick(ctx)
ip, _ = d.GetIP(ctx, ip.ID)
if ip.State != db.IPDone {
t.Fatalf("expected done once all 4 sites reported, got %s", ip.State)
}
if ip.OverallResult != db.ResultPass {
t.Fatalf("expected pass, got %s", ip.OverallResult)
}
}