Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.
Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
event carries the egress/ingress check counts.
Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
118 lines
4.4 KiB
Go
118 lines
4.4 KiB
Go
package db
|
|
|
|
import (
|
|
"context"
|
|
"database/sql"
|
|
)
|
|
|
|
// UpsertCheck records (or, on retry, overwrites) a single check result.
|
|
// registry_id/cycle_id are resolved from c.IPID's current ip_queue row at
|
|
// write time, so callers (agentcore/probercore) never need to know about
|
|
// the registry — see migrations/0007_ip_registry.sql. The
|
|
// UNIQUE(registry_id, cycle_id, source, check_type, target) constraint plus
|
|
// this upsert is what makes agent/prober result submission safely
|
|
// retryable without producing duplicate rows, now scoped to the durable
|
|
// per-address cycle rather than the ip_queue row's attempt_number, so it
|
|
// survives that row being deleted and the address later resubmitted.
|
|
func (d *DB) UpsertCheck(ctx context.Context, c Check) error {
|
|
_, err := d.UpsertCheckIfOpen(ctx, c)
|
|
return err
|
|
}
|
|
|
|
// UpsertCheckIfOpen is UpsertCheck for the live write path. It stores the
|
|
// check only while the address can still take results: the queue row exists,
|
|
// c.AttemptNumber is its current attempt, and it has not reached the
|
|
// aggregating state or a terminal one. Once the verdict is being computed the
|
|
// checks are frozen, so the verdict always matches the stored rows and a late
|
|
// probe of an already released floating IP cannot change them. It returns
|
|
// false (and writes nothing) when the result was dropped. The state test and
|
|
// the write are one statement, so they cannot interleave with SetAggregating.
|
|
func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
|
|
now := timeToDB(Now())
|
|
res, err := d.ExecContext(ctx, `
|
|
INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id,
|
|
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at)
|
|
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?
|
|
FROM ip_queue q
|
|
WHERE q.id=? AND q.attempt_number=? AND q.state NOT IN (?, ?, ?, ?)
|
|
ON CONFLICT(registry_id, cycle_id, source, check_type, target) DO UPDATE SET
|
|
validator_id=excluded.validator_id,
|
|
success=excluded.success,
|
|
latency_ms=excluded.latency_ms,
|
|
detail=excluded.detail,
|
|
checked_at=excluded.checked_at,
|
|
recorded_at=excluded.recorded_at
|
|
`, c.IPAddress, c.ValidatorID, c.Source, c.CheckType, c.Target,
|
|
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), now, now,
|
|
c.IPID, c.AttemptNumber, IPAggregating, IPDone, IPFailed, IPOccupied)
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
n, err := res.RowsAffected()
|
|
return n > 0, err
|
|
}
|
|
|
|
const checksSelect = `
|
|
SELECT id, registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target,
|
|
success, latency_ms, detail, checked_at, created_at
|
|
FROM checks
|
|
`
|
|
|
|
// ListChecksForAttempt returns every check recorded for an IP's current
|
|
// attempt — the input to overall-result aggregation.
|
|
func (d *DB) ListChecksForAttempt(ctx context.Context, ipID int64, attemptNumber int) ([]Check, error) {
|
|
rows, err := d.QueryContext(ctx, checksSelect+`
|
|
WHERE ip_id=? AND attempt_number=?
|
|
ORDER BY source, check_type, target
|
|
`, ipID, attemptNumber)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rows.Close()
|
|
return scanChecks(rows)
|
|
}
|
|
|
|
// ListChecksForRegistry returns an address's check history across every
|
|
// cycle still retained (see PruneRegistryHistory), newest cycle first. A
|
|
// nil limit returns everything currently retained.
|
|
func (d *DB) ListChecksForRegistry(ctx context.Context, registryID int64, limit *int) ([]Check, error) {
|
|
query := checksSelect + `WHERE registry_id=? ORDER BY cycle_id DESC, source, check_type, target`
|
|
args := []interface{}{registryID}
|
|
if limit != nil {
|
|
query += ` LIMIT ?`
|
|
args = append(args, *limit)
|
|
}
|
|
rows, err := d.QueryContext(ctx, query, args...)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
defer rows.Close()
|
|
return scanChecks(rows)
|
|
}
|
|
|
|
func scanChecks(rows *sql.Rows) ([]Check, error) {
|
|
var out []Check
|
|
for rows.Next() {
|
|
var c Check
|
|
var ipID sql.NullInt64
|
|
var checkedAt, createdAt string
|
|
if err := rows.Scan(&c.ID, &c.RegistryID, &c.CycleID, &ipID, &c.IPAddress, &c.AttemptNumber,
|
|
&c.ValidatorID, &c.Source, &c.CheckType, &c.Target, &c.Success, &c.LatencyMS, &c.Detail,
|
|
&checkedAt, &createdAt); err != nil {
|
|
return nil, err
|
|
}
|
|
if ipID.Valid {
|
|
c.IPID = ipID.Int64
|
|
}
|
|
var err error
|
|
if c.CheckedAt, err = dbToTime(checkedAt); err != nil {
|
|
return nil, err
|
|
}
|
|
if c.CreatedAt, err = dbToTime(createdAt); err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, c)
|
|
}
|
|
return out, rows.Err()
|
|
}
|