Show egress/ingress levels in the registry; freeze checks at the verdict
Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.
Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
event carries the egress/ingress check counts.
Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
db73409e8f
commit
864208238f
34 files changed
+1570
-72
No files matched your search
@@ -354,8 +354,20 @@ func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int,
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
// completion signal (egress or a given site's full port+icmp sweep),
|
||||
// updates the corresponding *_complete flag.
|
||||
// RecordCheckIfOpen stores one result for an address that is still being
|
||||
// checked. It returns false when the result was dropped because the address
|
||||
// already has (or is computing) its verdict, or the result belongs to an
|
||||
// earlier attempt: checks are frozen at the verdict so that the verdict and
|
||||
// the stored checks always agree.
|
||||
func (o *Orchestrator) RecordCheckIfOpen(ctx context.Context, c db.Check) (bool, error) {
|
||||
return o.DB.UpsertCheckIfOpen(ctx, c)
|
||||
}
|
||||
|
||||
// RecordCheck is RecordCheckIfOpen for callers that do not need to know
|
||||
// whether the result was dropped.
|
||||
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
|
||||
return o.DB.UpsertCheck(ctx, c)
|
||||
_, err := o.RecordCheckIfOpen(ctx, c)
|
||||
return err
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
|
||||
@@ -664,7 +676,15 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
// reported in ip_site_checks), with no need to wait on a prober that will
|
||||
// never exist, and no cap on how many sites can be configured.
|
||||
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
|
||||
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
|
||||
// The window counts from the start of checking, not from the assignment:
|
||||
// assigned_at also covers floating-IP association, the settle pause and the
|
||||
// self-check, which would leave only a few seconds for the checks. Rows from
|
||||
// before checking_started_at existed fall back to assigned_at.
|
||||
started := item.CheckingStartedAt
|
||||
if started == nil {
|
||||
started = item.AssignedAt
|
||||
}
|
||||
if started != nil && started.Before(deadline) {
|
||||
return true, nil
|
||||
}
|
||||
if !item.EgressComplete {
|
||||
@@ -695,36 +715,15 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
if err != nil {
|
||||
return fmt.Errorf("expected check count: %w", err)
|
||||
}
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing := expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
var result string
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
result, passCount, missing := computeVerdict(checks, expected, o.Agg.MissingCountsAsFail)
|
||||
egress, ingress := countByLevel(checks)
|
||||
|
||||
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
|
||||
return err
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "aggregated",
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d,"egress":%d,"ingress":%d}`,
|
||||
result, len(checks), passCount, missing, egress, ingress))
|
||||
|
||||
if settings, err := o.DB.GetSettings(ctx); err != nil {
|
||||
o.Log.Error("get settings for history retention", "ip_id", item.ID, "err", err)
|
||||
@@ -751,6 +750,52 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
return nil
|
||||
}
|
||||
|
||||
// computeVerdict is the single rule that turns an address's stored checks into
|
||||
// its overall result. Egress and ingress checks count alike, one check one
|
||||
// vote. A check that was expected but has no stored row counts as a failure,
|
||||
// so an incomplete set can never be "pass". The verdict is a pure function of
|
||||
// the stored checks and the expected count: recomputing it later from the
|
||||
// same rows gives the same result.
|
||||
func computeVerdict(checks []db.Check, expected int, missingCountsAsFail bool) (result string, passCount, missing int) {
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing = expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && missingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
return result, passCount, missing
|
||||
}
|
||||
|
||||
// countByLevel counts checks per level (egress, ingress), for the audit
|
||||
// payload of the aggregated event.
|
||||
func countByLevel(checks []db.Check) (egress, ingress int) {
|
||||
for _, c := range checks {
|
||||
switch db.CheckLevel(c.Source) {
|
||||
case db.LevelEgress:
|
||||
egress++
|
||||
case db.LevelIngress:
|
||||
ingress++
|
||||
}
|
||||
}
|
||||
return egress, ingress
|
||||
}
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe). Reads the current check_types/targets/sites/inbound
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func chk(src, ct string, ok bool) db.Check {
|
||||
return db.Check{Source: src, CheckType: ct, Success: ok}
|
||||
}
|
||||
|
||||
func TestComputeVerdict(t *testing.T) {
|
||||
in1 := db.InboundSource(1)
|
||||
cases := []struct {
|
||||
name string
|
||||
checks []db.Check
|
||||
expected int
|
||||
want string
|
||||
missing int
|
||||
}{
|
||||
{"all pass, complete", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 2, db.ResultPass, 0},
|
||||
{"egress failure", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", true)}, 2, db.ResultPartial, 0},
|
||||
{"ingress failure", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "ssh", false)}, 2, db.ResultPartial, 0},
|
||||
{"all recorded pass but one missing", []db.Check{chk(db.SourceEgress, "https", true)}, 2, db.ResultPartial, 1},
|
||||
{"all fail", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", false)}, 2, db.ResultFail, 0},
|
||||
{"nothing recorded", nil, 2, db.ResultFail, 2},
|
||||
{"more recorded than expected", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 1, db.ResultPass, 0},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, _, missing := computeVerdict(c.checks, c.expected, true)
|
||||
if got != c.want || missing != c.missing {
|
||||
t.Errorf("%s: got %s missing=%d, want %s missing=%d", c.name, got, missing, c.want, c.missing)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The window starts when checking starts: an address handed out long ago but
|
||||
// that only just began checking must not be cut off.
|
||||
func TestAggregationWindowCountsFromCheckingStart(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, _, _ := newTestOrchestrator(t, 180)
|
||||
deadline := time.Now().Add(-120 * time.Second)
|
||||
long, recent := time.Now().Add(-10*time.Minute), time.Now()
|
||||
|
||||
item := db.IPQueueItem{AssignedAt: &long, CheckingStartedAt: &recent}
|
||||
ready, err := o.isReadyToAggregate(ctx, item, deadline, nil)
|
||||
if err != nil || ready {
|
||||
t.Fatalf("recent checking start must not be cut off by an old assignment: ready=%v err=%v", ready, err)
|
||||
}
|
||||
|
||||
started := time.Now().Add(-3 * time.Minute)
|
||||
item.CheckingStartedAt = &started
|
||||
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
|
||||
t.Fatalf("window elapsed since checking start: ready=%v err=%v", ready, err)
|
||||
}
|
||||
|
||||
// Rows from before checking_started_at existed fall back to assigned_at.
|
||||
item.CheckingStartedAt = nil
|
||||
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
|
||||
t.Fatalf("fallback to assigned_at: ready=%v err=%v", ready, err)
|
||||
}
|
||||
}
|
||||
|
||||
// After the verdict nothing can change the stored checks, and the verdict is
|
||||
// exactly what the stored checks give when computed again.
|
||||
func TestVerdictMatchesStoredChecksAndLateResultIsDropped(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
|
||||
rec := func(src, ct string, ok bool) bool {
|
||||
t.Helper()
|
||||
written, err := o.RecordCheckIfOpen(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, ValidatorID: "validator-1",
|
||||
Source: src, CheckType: ct, Target: "t", Success: ok, CheckedAt: db.Now(),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return written
|
||||
}
|
||||
if !rec(db.SourceEgress, "https", true) || !rec(db.InboundSource(1), "tcp-22", false) {
|
||||
t.Fatal("results while checking must be stored")
|
||||
}
|
||||
_ = o.MarkEgressComplete(ctx, ip.ID)
|
||||
o.Cfg.CheckingWindowSeconds = 0
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
|
||||
// A late result: a new check, and an attempt to flip a stored one.
|
||||
if rec(db.InboundSource(2), "icmp", true) {
|
||||
t.Fatal("new check after the verdict was stored")
|
||||
}
|
||||
if rec(db.InboundSource(1), "tcp-22", true) {
|
||||
t.Fatal("overwrite after the verdict was accepted")
|
||||
}
|
||||
|
||||
stored, err := d.ListChecksForAttempt(ctx, ip.ID, ip.AttemptNumber)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(stored) != 2 {
|
||||
t.Fatalf("stored checks changed after the verdict: %d rows", len(stored))
|
||||
}
|
||||
expected, err := o.expectedCheckCount(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, _, _ := computeVerdict(stored, expected, o.Agg.MissingCountsAsFail); got != ip.OverallResult {
|
||||
t.Fatalf("recomputed verdict %s differs from stored %s", got, ip.OverallResult)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user