Show egress/ingress levels in the registry; freeze checks at the verdict
Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.
Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
event carries the egress/ingress check counts.
Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
db73409e8f
commit
864208238f
34 files changed
+1570
-72
No files matched your search
@@ -354,8 +354,20 @@ func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int,
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
// completion signal (egress or a given site's full port+icmp sweep),
|
||||
// updates the corresponding *_complete flag.
|
||||
// RecordCheckIfOpen stores one result for an address that is still being
|
||||
// checked. It returns false when the result was dropped because the address
|
||||
// already has (or is computing) its verdict, or the result belongs to an
|
||||
// earlier attempt: checks are frozen at the verdict so that the verdict and
|
||||
// the stored checks always agree.
|
||||
func (o *Orchestrator) RecordCheckIfOpen(ctx context.Context, c db.Check) (bool, error) {
|
||||
return o.DB.UpsertCheckIfOpen(ctx, c)
|
||||
}
|
||||
|
||||
// RecordCheck is RecordCheckIfOpen for callers that do not need to know
|
||||
// whether the result was dropped.
|
||||
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
|
||||
return o.DB.UpsertCheck(ctx, c)
|
||||
_, err := o.RecordCheckIfOpen(ctx, c)
|
||||
return err
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
|
||||
@@ -664,7 +676,15 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
// reported in ip_site_checks), with no need to wait on a prober that will
|
||||
// never exist, and no cap on how many sites can be configured.
|
||||
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
|
||||
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
|
||||
// The window counts from the start of checking, not from the assignment:
|
||||
// assigned_at also covers floating-IP association, the settle pause and the
|
||||
// self-check, which would leave only a few seconds for the checks. Rows from
|
||||
// before checking_started_at existed fall back to assigned_at.
|
||||
started := item.CheckingStartedAt
|
||||
if started == nil {
|
||||
started = item.AssignedAt
|
||||
}
|
||||
if started != nil && started.Before(deadline) {
|
||||
return true, nil
|
||||
}
|
||||
if !item.EgressComplete {
|
||||
@@ -695,36 +715,15 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
if err != nil {
|
||||
return fmt.Errorf("expected check count: %w", err)
|
||||
}
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing := expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
var result string
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
result, passCount, missing := computeVerdict(checks, expected, o.Agg.MissingCountsAsFail)
|
||||
egress, ingress := countByLevel(checks)
|
||||
|
||||
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
|
||||
return err
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "aggregated",
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d,"egress":%d,"ingress":%d}`,
|
||||
result, len(checks), passCount, missing, egress, ingress))
|
||||
|
||||
if settings, err := o.DB.GetSettings(ctx); err != nil {
|
||||
o.Log.Error("get settings for history retention", "ip_id", item.ID, "err", err)
|
||||
@@ -751,6 +750,52 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
return nil
|
||||
}
|
||||
|
||||
// computeVerdict is the single rule that turns an address's stored checks into
|
||||
// its overall result. Egress and ingress checks count alike, one check one
|
||||
// vote. A check that was expected but has no stored row counts as a failure,
|
||||
// so an incomplete set can never be "pass". The verdict is a pure function of
|
||||
// the stored checks and the expected count: recomputing it later from the
|
||||
// same rows gives the same result.
|
||||
func computeVerdict(checks []db.Check, expected int, missingCountsAsFail bool) (result string, passCount, missing int) {
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing = expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && missingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
return result, passCount, missing
|
||||
}
|
||||
|
||||
// countByLevel counts checks per level (egress, ingress), for the audit
|
||||
// payload of the aggregated event.
|
||||
func countByLevel(checks []db.Check) (egress, ingress int) {
|
||||
for _, c := range checks {
|
||||
switch db.CheckLevel(c.Source) {
|
||||
case db.LevelEgress:
|
||||
egress++
|
||||
case db.LevelIngress:
|
||||
ingress++
|
||||
}
|
||||
}
|
||||
return egress, ingress
|
||||
}
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe). Reads the current check_types/targets/sites/inbound
|
||||
|
||||
Reference in new issue
Block a user