Show egress/ingress levels in the registry; freeze checks at the verdict

Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.

Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
  poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
  verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
  result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
  written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
  event carries the egress/ingress check counts.

Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-03 17:59:52 +03:00
1 parent db73409e8f
commit 864208238f
34 files changed
+1570 -72

No files matched your search

+31 -6
View File
@@ -1,6 +1,8 @@
package httpapi
import (
"context"
"fmt"
"net/http"
"time"
@@ -41,9 +43,12 @@ func (s *Server) handleProberHeartbeat(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusOK, okResponse{OK: true})
}
// handleProberAssignments returns every IP currently in the checking
// state — probers work the whole active set each poll, not one IP at a
// time, since multiple validators run in parallel.
// handleProberAssignments returns every IP currently in the checking state
// that this site has not yet finished probing in the current attempt —
// probers work the whole active set each poll, not one IP at a time, since
// multiple validators run in parallel. An address is handed out until the
// site reports it complete, then no more: one probe round per site per
// attempt, so results are written once and never overwritten.
func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) {
siteID := r.PathValue("site_id")
idx, err := s.Orch.SiteIndexForID(r.Context(), siteID)
@@ -55,7 +60,7 @@ func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request)
writeError(w, http.StatusNotFound, "unknown site_id: "+siteID)
return
}
items, err := s.DB.ListChecking(r.Context())
items, err := s.DB.ListCheckingForSite(r.Context(), idx)
if err != nil {
writeError(w, http.StatusInternalServerError, err.Error())
return
@@ -93,6 +98,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
}
completed := map[int64]bool{}
dropped := map[int64]int{}
for _, res := range req.Results {
item, err := s.DB.GetIP(r.Context(), res.IPID)
if err != nil {
@@ -103,7 +109,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
if err != nil {
checkedAt = db.Now()
}
err = s.Orch.RecordCheck(r.Context(), db.Check{
written, err := s.Orch.RecordCheckIfOpen(r.Context(), db.Check{
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
Source: db.InboundSource(siteIndex), CheckType: res.CheckType, Target: res.IPAddress,
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
@@ -112,6 +118,10 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusInternalServerError, err.Error())
return
}
if !written {
dropped[item.ID]++
continue
}
if res.Complete {
completed[res.IPID] = true
}
@@ -122,5 +132,20 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
return
}
}
writeJSON(w, http.StatusOK, okResponse{OK: true})
writeJSON(w, http.StatusOK, s.reportDropped(r.Context(), "prober", siteID, db.InboundSource(siteIndex), dropped))
}
// reportDropped records one result_dropped event per address whose submitted
// checks were refused (the verdict already exists, or the attempt is old) and
// builds the response. The events are the audit trail of how often senders
// are late; in a healthy run there are none.
func (s *Server) reportDropped(ctx context.Context, sourceType, sourceID, checkSource string, dropped map[int64]int) resultsResponse {
total := 0
for ipID, n := range dropped {
total += n
id := ipID
s.Orch.RecordEvent(ctx, sourceType, sourceID, &id, "result_dropped",
fmt.Sprintf(`{"source":%q,"dropped":%d}`, checkSource, n))
}
return resultsResponse{OK: true, Ignored: total}
}