New external site management. New external prober heartbeat feature.
This commit is contained in:
1 parent
42f584dd3c
commit
ef24cc9858
38 files changed
+1058
-158
No files matched your search
@@ -401,7 +401,12 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
return fmt.Errorf("list sites: %w", err)
|
||||
}
|
||||
for _, item := range checking {
|
||||
if !o.isReadyToAggregate(item, deadline, sites) {
|
||||
ready, err := o.isReadyToAggregate(ctx, item, deadline, sites)
|
||||
if err != nil {
|
||||
o.Log.Error("check ready to aggregate", "ip_id", item.ID, "err", err)
|
||||
continue
|
||||
}
|
||||
if !ready {
|
||||
continue
|
||||
}
|
||||
if err := o.aggregateAndRelease(ctx, item); err != nil {
|
||||
@@ -416,34 +421,26 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
// checking-window deadline. Which inbound sources it's expecting is driven
|
||||
// entirely by the currently configured sites — inbound checks are optional:
|
||||
// an empty (or partial) sites configuration means this IP is ready as soon
|
||||
// as egress completes (or after the corresponding subset of siteN_complete
|
||||
// flags), with no need to wait on a prober that will never exist. This is
|
||||
// what makes inbound checks genuinely opt-in rather than a hardcoded
|
||||
// expectation of exactly three sites.
|
||||
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time, sites []db.Site) bool {
|
||||
// as egress completes (or after the corresponding subset of sites has
|
||||
// reported in ip_site_checks), with no need to wait on a prober that will
|
||||
// never exist, and no cap on how many sites can be configured.
|
||||
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
|
||||
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
|
||||
return true
|
||||
return true, nil
|
||||
}
|
||||
if !item.EgressComplete {
|
||||
return false
|
||||
return false, nil
|
||||
}
|
||||
completed, err := o.DB.ListCompletedSiteIndices(ctx, item.ID, item.AttemptNumber)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
for _, s := range sites {
|
||||
switch s.Index {
|
||||
case 1:
|
||||
if !item.Site1Complete {
|
||||
return false
|
||||
}
|
||||
case 2:
|
||||
if !item.Site2Complete {
|
||||
return false
|
||||
}
|
||||
case 3:
|
||||
if !item.Site3Complete {
|
||||
return false
|
||||
}
|
||||
if !completed[s.Index] {
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
return true
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueItem) error {
|
||||
@@ -585,6 +582,25 @@ func (o *Orchestrator) SweepStaleHeartbeats(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// SweepStaleSiteHeartbeats marks prober sites unreachable if they haven't
|
||||
// heartbeated within HeartbeatTimeoutSeconds — mirrors SweepStaleHeartbeats
|
||||
// exactly, for sites instead of validators.
|
||||
func (o *Orchestrator) SweepStaleSiteHeartbeats(ctx context.Context) error {
|
||||
cutoff := db.Now().Add(-time.Duration(o.Cfg.HeartbeatTimeoutSeconds) * time.Second)
|
||||
stale, err := o.DB.ListStaleSiteHeartbeats(ctx, cutoff)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, s := range stale {
|
||||
if err := o.DB.MarkSiteUnreachable(ctx, s.SiteID); err != nil {
|
||||
o.Log.Error("mark site unreachable", "site_id", s.SiteID, "err", err)
|
||||
continue
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "site_unreachable", fmt.Sprintf(`{"site_id":%q}`, s.SiteID))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// RecordEvent is the exported entry point httpapi uses to log
|
||||
// agent/prober-reported audit events (config_received, fip_changed,
|
||||
// error, etc.) through the same path as internally generated events.
|
||||
|
||||
@@ -624,3 +624,65 @@ func TestExpectedCheckCountReflectsInboundChecksConfigChange(t *testing.T) {
|
||||
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFourSitesAllMustReportBeforeAggregation confirms there's no hardcoded
|
||||
// cap of three sites: with 4 sites configured, aggregation must wait on
|
||||
// all 4, not silently treat the 4th as always-complete.
|
||||
func TestFourSitesAllMustReportBeforeAggregation(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
sites := []config.SiteConfig{
|
||||
{SiteID: "site-1", Index: 1}, {SiteID: "site-2", Index: 2},
|
||||
{SiteID: "site-3", Index: 3}, {SiteID: "site-4", Index: 4},
|
||||
}
|
||||
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.MarkEgressComplete(ctx, ip.ID)
|
||||
|
||||
for _, site := range []int{1, 2, 3} {
|
||||
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
}
|
||||
_ = o.MarkSiteComplete(ctx, ip.ID, site)
|
||||
}
|
||||
|
||||
// Sites 1-3 reported, site 4 (the case beyond the old fixed-3 cap)
|
||||
// hasn't — must not aggregate yet.
|
||||
o.Tick(ctx)
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPChecking {
|
||||
t.Fatalf("expected still checking (site-4 pending), got %s", ip.State)
|
||||
}
|
||||
|
||||
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(4), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
}
|
||||
_ = o.MarkSiteComplete(ctx, ip.ID, 4)
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done once all 4 sites reported, got %s", ip.State)
|
||||
}
|
||||
if ip.OverallResult != db.ResultPass {
|
||||
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user