Show egress/ingress levels in the registry; freeze checks at the verdict

Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.

Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
  poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
  verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
  result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
  written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
  event carries the egress/ingress check counts.

Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-03 17:59:52 +03:00
1 parent db73409e8f
commit 864208238f
34 files changed
+1570 -72

No files matched your search

+72 -27
View File
@@ -354,8 +354,20 @@ func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int,
// RecordCheck upserts a single check result and, if it represents a
// completion signal (egress or a given site's full port+icmp sweep),
// updates the corresponding *_complete flag.
// RecordCheckIfOpen stores one result for an address that is still being
// checked. It returns false when the result was dropped because the address
// already has (or is computing) its verdict, or the result belongs to an
// earlier attempt: checks are frozen at the verdict so that the verdict and
// the stored checks always agree.
func (o *Orchestrator) RecordCheckIfOpen(ctx context.Context, c db.Check) (bool, error) {
return o.DB.UpsertCheckIfOpen(ctx, c)
}
// RecordCheck is RecordCheckIfOpen for callers that do not need to know
// whether the result was dropped.
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
return o.DB.UpsertCheck(ctx, c)
_, err := o.RecordCheckIfOpen(ctx, c)
return err
}
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
@@ -664,7 +676,15 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
// reported in ip_site_checks), with no need to wait on a prober that will
// never exist, and no cap on how many sites can be configured.
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
// The window counts from the start of checking, not from the assignment:
// assigned_at also covers floating-IP association, the settle pause and the
// self-check, which would leave only a few seconds for the checks. Rows from
// before checking_started_at existed fall back to assigned_at.
started := item.CheckingStartedAt
if started == nil {
started = item.AssignedAt
}
if started != nil && started.Before(deadline) {
return true, nil
}
if !item.EgressComplete {
@@ -695,36 +715,15 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
if err != nil {
return fmt.Errorf("expected check count: %w", err)
}
passCount := 0
for _, c := range checks {
if c.Success {
passCount++
}
}
missing := expected - len(checks)
if missing < 0 {
missing = 0
}
failCount := (len(checks) - passCount) + missing
var result string
switch {
case passCount > 0 && failCount == 0:
result = db.ResultPass
case passCount == 0:
result = db.ResultFail
default:
result = db.ResultPartial
}
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
result = db.ResultPartial
}
result, passCount, missing := computeVerdict(checks, expected, o.Agg.MissingCountsAsFail)
egress, ingress := countByLevel(checks)
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
return err
}
o.event(ctx, "control-api", "", &item.ID, "aggregated",
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d,"egress":%d,"ingress":%d}`,
result, len(checks), passCount, missing, egress, ingress))
if settings, err := o.DB.GetSettings(ctx); err != nil {
o.Log.Error("get settings for history retention", "ip_id", item.ID, "err", err)
@@ -751,6 +750,52 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
return nil
}
// computeVerdict is the single rule that turns an address's stored checks into
// its overall result. Egress and ingress checks count alike, one check one
// vote. A check that was expected but has no stored row counts as a failure,
// so an incomplete set can never be "pass". The verdict is a pure function of
// the stored checks and the expected count: recomputing it later from the
// same rows gives the same result.
func computeVerdict(checks []db.Check, expected int, missingCountsAsFail bool) (result string, passCount, missing int) {
for _, c := range checks {
if c.Success {
passCount++
}
}
missing = expected - len(checks)
if missing < 0 {
missing = 0
}
failCount := (len(checks) - passCount) + missing
switch {
case passCount > 0 && failCount == 0:
result = db.ResultPass
case passCount == 0:
result = db.ResultFail
default:
result = db.ResultPartial
}
if missing > 0 && missingCountsAsFail && result == db.ResultPass {
result = db.ResultPartial
}
return result, passCount, missing
}
// countByLevel counts checks per level (egress, ingress), for the audit
// payload of the aggregated event.
func countByLevel(checks []db.Check) (egress, ingress int) {
for _, c := range checks {
switch db.CheckLevel(c.Source) {
case db.LevelEgress:
egress++
case db.LevelIngress:
ingress++
}
}
return egress, ingress
}
// expectedCheckCount is the number of check rows a fully-reported IP should
// have: one per (egress check-type x target) plus one per (site x inbound
// port/icmp probe). Reads the current check_types/targets/sites/inbound