Show egress/ingress levels in the registry; freeze checks at the verdict
Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.
Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
event carries the egress/ingress check counts.
Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
db73409e8f
commit
864208238f
34 files changed
+1570
-72
No files matched your search
@@ -1,6 +1,8 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
@@ -41,9 +43,12 @@ func (s *Server) handleProberHeartbeat(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
// handleProberAssignments returns every IP currently in the checking
|
||||
// state — probers work the whole active set each poll, not one IP at a
|
||||
// time, since multiple validators run in parallel.
|
||||
// handleProberAssignments returns every IP currently in the checking state
|
||||
// that this site has not yet finished probing in the current attempt —
|
||||
// probers work the whole active set each poll, not one IP at a time, since
|
||||
// multiple validators run in parallel. An address is handed out until the
|
||||
// site reports it complete, then no more: one probe round per site per
|
||||
// attempt, so results are written once and never overwritten.
|
||||
func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) {
|
||||
siteID := r.PathValue("site_id")
|
||||
idx, err := s.Orch.SiteIndexForID(r.Context(), siteID)
|
||||
@@ -55,7 +60,7 @@ func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request)
|
||||
writeError(w, http.StatusNotFound, "unknown site_id: "+siteID)
|
||||
return
|
||||
}
|
||||
items, err := s.DB.ListChecking(r.Context())
|
||||
items, err := s.DB.ListCheckingForSite(r.Context(), idx)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
@@ -93,6 +98,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
completed := map[int64]bool{}
|
||||
dropped := map[int64]int{}
|
||||
for _, res := range req.Results {
|
||||
item, err := s.DB.GetIP(r.Context(), res.IPID)
|
||||
if err != nil {
|
||||
@@ -103,7 +109,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
if err != nil {
|
||||
checkedAt = db.Now()
|
||||
}
|
||||
err = s.Orch.RecordCheck(r.Context(), db.Check{
|
||||
written, err := s.Orch.RecordCheckIfOpen(r.Context(), db.Check{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
|
||||
Source: db.InboundSource(siteIndex), CheckType: res.CheckType, Target: res.IPAddress,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
|
||||
@@ -112,6 +118,10 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
if !written {
|
||||
dropped[item.ID]++
|
||||
continue
|
||||
}
|
||||
if res.Complete {
|
||||
completed[res.IPID] = true
|
||||
}
|
||||
@@ -122,5 +132,20 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
writeJSON(w, http.StatusOK, s.reportDropped(r.Context(), "prober", siteID, db.InboundSource(siteIndex), dropped))
|
||||
}
|
||||
|
||||
// reportDropped records one result_dropped event per address whose submitted
|
||||
// checks were refused (the verdict already exists, or the attempt is old) and
|
||||
// builds the response. The events are the audit trail of how often senders
|
||||
// are late; in a healthy run there are none.
|
||||
func (s *Server) reportDropped(ctx context.Context, sourceType, sourceID, checkSource string, dropped map[int64]int) resultsResponse {
|
||||
total := 0
|
||||
for ipID, n := range dropped {
|
||||
total += n
|
||||
id := ipID
|
||||
s.Orch.RecordEvent(ctx, sourceType, sourceID, &id, "result_dropped",
|
||||
fmt.Sprintf(`{"source":%q,"dropped":%d}`, checkSource, n))
|
||||
}
|
||||
return resultsResponse{OK: true, Ignored: total}
|
||||
}
|
||||
Reference in new issue
Block a user