Show egress/ingress levels in the registry; freeze checks at the verdict

Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.

Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
  poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
  verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
  result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
  written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
  event carries the egress/ingress check counts.

Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-03 17:59:52 +03:00
1 parent db73409e8f
commit 864208238f
34 files changed
+1570 -72

No files matched your search

+72 -27
View File
@@ -354,8 +354,20 @@ func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int,
// RecordCheck upserts a single check result and, if it represents a
// completion signal (egress or a given site's full port+icmp sweep),
// updates the corresponding *_complete flag.
// RecordCheckIfOpen stores one result for an address that is still being
// checked. It returns false when the result was dropped because the address
// already has (or is computing) its verdict, or the result belongs to an
// earlier attempt: checks are frozen at the verdict so that the verdict and
// the stored checks always agree.
func (o *Orchestrator) RecordCheckIfOpen(ctx context.Context, c db.Check) (bool, error) {
return o.DB.UpsertCheckIfOpen(ctx, c)
}
// RecordCheck is RecordCheckIfOpen for callers that do not need to know
// whether the result was dropped.
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
return o.DB.UpsertCheck(ctx, c)
_, err := o.RecordCheckIfOpen(ctx, c)
return err
}
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
@@ -664,7 +676,15 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
// reported in ip_site_checks), with no need to wait on a prober that will
// never exist, and no cap on how many sites can be configured.
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
// The window counts from the start of checking, not from the assignment:
// assigned_at also covers floating-IP association, the settle pause and the
// self-check, which would leave only a few seconds for the checks. Rows from
// before checking_started_at existed fall back to assigned_at.
started := item.CheckingStartedAt
if started == nil {
started = item.AssignedAt
}
if started != nil && started.Before(deadline) {
return true, nil
}
if !item.EgressComplete {
@@ -695,36 +715,15 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
if err != nil {
return fmt.Errorf("expected check count: %w", err)
}
passCount := 0
for _, c := range checks {
if c.Success {
passCount++
}
}
missing := expected - len(checks)
if missing < 0 {
missing = 0
}
failCount := (len(checks) - passCount) + missing
var result string
switch {
case passCount > 0 && failCount == 0:
result = db.ResultPass
case passCount == 0:
result = db.ResultFail
default:
result = db.ResultPartial
}
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
result = db.ResultPartial
}
result, passCount, missing := computeVerdict(checks, expected, o.Agg.MissingCountsAsFail)
egress, ingress := countByLevel(checks)
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
return err
}
o.event(ctx, "control-api", "", &item.ID, "aggregated",
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d,"egress":%d,"ingress":%d}`,
result, len(checks), passCount, missing, egress, ingress))
if settings, err := o.DB.GetSettings(ctx); err != nil {
o.Log.Error("get settings for history retention", "ip_id", item.ID, "err", err)
@@ -751,6 +750,52 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
return nil
}
// computeVerdict is the single rule that turns an address's stored checks into
// its overall result. Egress and ingress checks count alike, one check one
// vote. A check that was expected but has no stored row counts as a failure,
// so an incomplete set can never be "pass". The verdict is a pure function of
// the stored checks and the expected count: recomputing it later from the
// same rows gives the same result.
func computeVerdict(checks []db.Check, expected int, missingCountsAsFail bool) (result string, passCount, missing int) {
for _, c := range checks {
if c.Success {
passCount++
}
}
missing = expected - len(checks)
if missing < 0 {
missing = 0
}
failCount := (len(checks) - passCount) + missing
switch {
case passCount > 0 && failCount == 0:
result = db.ResultPass
case passCount == 0:
result = db.ResultFail
default:
result = db.ResultPartial
}
if missing > 0 && missingCountsAsFail && result == db.ResultPass {
result = db.ResultPartial
}
return result, passCount, missing
}
// countByLevel counts checks per level (egress, ingress), for the audit
// payload of the aggregated event.
func countByLevel(checks []db.Check) (egress, ingress int) {
for _, c := range checks {
switch db.CheckLevel(c.Source) {
case db.LevelEgress:
egress++
case db.LevelIngress:
ingress++
}
}
return egress, ingress
}
// expectedCheckCount is the number of check rows a fully-reported IP should
// have: one per (egress check-type x target) plus one per (site x inbound
// port/icmp probe). Reads the current check_types/targets/sites/inbound
@@ -0,0 +1,127 @@
package orchestrator
import (
"context"
"testing"
"time"
"cloudipvalidator/internal/db"
)
func chk(src, ct string, ok bool) db.Check {
return db.Check{Source: src, CheckType: ct, Success: ok}
}
func TestComputeVerdict(t *testing.T) {
in1 := db.InboundSource(1)
cases := []struct {
name string
checks []db.Check
expected int
want string
missing int
}{
{"all pass, complete", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 2, db.ResultPass, 0},
{"egress failure", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", true)}, 2, db.ResultPartial, 0},
{"ingress failure", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "ssh", false)}, 2, db.ResultPartial, 0},
{"all recorded pass but one missing", []db.Check{chk(db.SourceEgress, "https", true)}, 2, db.ResultPartial, 1},
{"all fail", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", false)}, 2, db.ResultFail, 0},
{"nothing recorded", nil, 2, db.ResultFail, 2},
{"more recorded than expected", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 1, db.ResultPass, 0},
}
for _, c := range cases {
got, _, missing := computeVerdict(c.checks, c.expected, true)
if got != c.want || missing != c.missing {
t.Errorf("%s: got %s missing=%d, want %s missing=%d", c.name, got, missing, c.want, c.missing)
}
}
}
// The window starts when checking starts: an address handed out long ago but
// that only just began checking must not be cut off.
func TestAggregationWindowCountsFromCheckingStart(t *testing.T) {
ctx := context.Background()
o, _, _ := newTestOrchestrator(t, 180)
deadline := time.Now().Add(-120 * time.Second)
long, recent := time.Now().Add(-10*time.Minute), time.Now()
item := db.IPQueueItem{AssignedAt: &long, CheckingStartedAt: &recent}
ready, err := o.isReadyToAggregate(ctx, item, deadline, nil)
if err != nil || ready {
t.Fatalf("recent checking start must not be cut off by an old assignment: ready=%v err=%v", ready, err)
}
started := time.Now().Add(-3 * time.Minute)
item.CheckingStartedAt = &started
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
t.Fatalf("window elapsed since checking start: ready=%v err=%v", ready, err)
}
// Rows from before checking_started_at existed fall back to assigned_at.
item.CheckingStartedAt = nil
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
t.Fatalf("fallback to assigned_at: ready=%v err=%v", ready, err)
}
}
// After the verdict nothing can change the stored checks, and the verdict is
// exactly what the stored checks give when computed again.
func TestVerdictMatchesStoredChecksAndLateResultIsDropped(t *testing.T) {
ctx := context.Background()
o, d, mock := newTestOrchestrator(t, 180)
mock.Seed("fip-1", "1.2.3.4", "svc-project")
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
o.Tick(ctx)
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
ip, _ = d.GetIP(ctx, ip.ID)
rec := func(src, ct string, ok bool) bool {
t.Helper()
written, err := o.RecordCheckIfOpen(ctx, db.Check{
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, ValidatorID: "validator-1",
Source: src, CheckType: ct, Target: "t", Success: ok, CheckedAt: db.Now(),
})
if err != nil {
t.Fatal(err)
}
return written
}
if !rec(db.SourceEgress, "https", true) || !rec(db.InboundSource(1), "tcp-22", false) {
t.Fatal("results while checking must be stored")
}
_ = o.MarkEgressComplete(ctx, ip.ID)
o.Cfg.CheckingWindowSeconds = 0
time.Sleep(5 * time.Millisecond)
o.Tick(ctx)
ip, _ = d.GetIP(ctx, ip.ID)
if ip.State != db.IPDone {
t.Fatalf("expected done, got %s", ip.State)
}
// A late result: a new check, and an attempt to flip a stored one.
if rec(db.InboundSource(2), "icmp", true) {
t.Fatal("new check after the verdict was stored")
}
if rec(db.InboundSource(1), "tcp-22", true) {
t.Fatal("overwrite after the verdict was accepted")
}
stored, err := d.ListChecksForAttempt(ctx, ip.ID, ip.AttemptNumber)
if err != nil {
t.Fatal(err)
}
if len(stored) != 2 {
t.Fatalf("stored checks changed after the verdict: %d rows", len(stored))
}
expected, err := o.expectedCheckCount(ctx)
if err != nil {
t.Fatal(err)
}
if got, _, _ := computeVerdict(stored, expected, o.Agg.MissingCountsAsFail); got != ip.OverallResult {
t.Fatalf("recomputed verdict %s differs from stored %s", got, ip.OverallResult)
}
}