Add the Analytics section: check runs, analytics API and page

Runs (migration 0011): a run groups the cycles of one launch. It opens when an
address enters an idle queue, takes everything submitted or re-checked while it
is open and is finalized when all its addresses are done; a re-check after that
opens a new run, so results of different runs never mix. check_runs,
run_results (one result per address and run, with the verdict and the expected
and stored check counts), subnets, run_id on ip_queue and checks. Existing data
is split into runs at pauses of more than an hour; ingress checks get the
validator that held the address (also at write time from now on).

Analytics (internal/analytics): figures computed from the stored checks of the
latest cycle of each address in the run, as facts next to the verdict: summary,
reasons of partial, data quality, subnets, targets and the subnet x target
matrix by check type, ingress by site, error classes, validators, and the
address lists behind the indicators and error classes. API: analytics runs,
report, lists (JSON or CSV), subnet list; run and subnet filters for the
registry.

Dashboard: /analytics matching the approved mockup (run selector, indicators
with address lists and CSV, error-class dialogs, drill-down to the registry),
subnet list on /settings. Sidebar: the control-api link state, theme toggle and
logout moved to the top, the three dots next to the logo removed, sections
grouped.

Rebuilt bin/control-api and bin/admin-dashboard to match. Plan, summary and the
updated README, API, USAGE, DASHBOARD and ADMIN_CLEANUP docs are in docs/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5.5 committed 2026-10-03 18:36:03 +03:00
1 parent 864208238f
commit b7669c9e41
44 files changed
+4123 -62

No files matched your search

+8
View File
@@ -43,6 +43,9 @@ var scaleIndexesSchema string
//go:embed migrations/0010_verdict_integrity.sql
var verdictIntegritySchema string
//go:embed migrations/0011_check_runs.sql
var checkRunsSchema string
// migrations is the ordered list of schema versions. Each entry's SQL is
// applied, in order, for any version greater than the database's current
// PRAGMA user_version — so a fresh database walks the whole list and an
@@ -61,6 +64,7 @@ var migrations = []struct {
{8, autoCycleSchema},
{9, scaleIndexesSchema},
{10, verdictIntegritySchema},
{11, checkRunsSchema},
}
type DB struct {
@@ -96,6 +100,10 @@ func Open(ctx context.Context, path string) (*DB, error) {
sqlDB.Close()
return nil, fmt.Errorf("migrate: %w", err)
}
if err := d.adoptOrphanQueueRows(ctx); err != nil {
sqlDB.Close()
return nil, fmt.Errorf("adopt queue rows into a run: %w", err)
}
return d, nil
}
@@ -0,0 +1,98 @@
-- Check runs (see docs/changes/2026-10-03_16-39_analytics-section-plan.md).
--
-- cycle_id counts per address, so it cannot tell one launch from another. A
-- run groups the cycles of one launch: it opens when an address enters an idle
-- queue, takes every address submitted or re-checked while it is open, and is
-- finalized when all its queue rows are terminal. checks.run_id and
-- ip_queue.run_id carry the membership; run_results keeps one row per address
-- and run (its latest cycle) with the verdict, which ip_queue overwrites on a
-- re-check.
--
-- No foreign keys on purpose: the manual cleanup in docs/ADMIN_CLEANUP.md
-- deletes from these tables freely.
CREATE TABLE check_runs (
id INTEGER PRIMARY KEY AUTOINCREMENT,
kind TEXT NOT NULL DEFAULT 'manual', -- manual | auto
state TEXT NOT NULL DEFAULT 'open', -- open | finalized
started_at TIMESTAMP NOT NULL,
finalized_at TIMESTAMP
);
CREATE TABLE run_results (
run_id INTEGER NOT NULL,
registry_id INTEGER NOT NULL,
ip_address TEXT NOT NULL,
cycle_id INTEGER NOT NULL,
verdict TEXT NOT NULL, -- pass | partial | fail | cancelled
verdict_derived INTEGER NOT NULL DEFAULT 0, -- 1: computed from checks, not stored by the orchestrator
aggregated_at TIMESTAMP NOT NULL,
expected_checks INTEGER, -- NULL when unknown
recorded_checks INTEGER NOT NULL DEFAULT 0, -- checks stored when the verdict was made
PRIMARY KEY (run_id, registry_id)
);
CREATE INDEX idx_run_results_registry ON run_results(registry_id);
CREATE TABLE subnets (
cidr TEXT PRIMARY KEY,
label TEXT NOT NULL DEFAULT ''
);
ALTER TABLE ip_queue ADD COLUMN run_id INTEGER;
ALTER TABLE checks ADD COLUMN run_id INTEGER;
CREATE INDEX idx_checks_run ON checks(run_id, registry_id, cycle_id);
-- ---- existing data -------------------------------------------------------
-- Ingress checks never stored the validator. The one holding the address in a
-- cycle is named by its fip_associated event.
UPDATE checks SET validator_id = COALESCE((
SELECT json_extract(e.payload, '$.validator_id') FROM events e
WHERE e.event_type = 'fip_associated' AND e.registry_id = checks.registry_id AND e.cycle_id = checks.cycle_id
ORDER BY e.id DESC LIMIT 1), '')
WHERE source LIKE 'inbound-site-%' AND validator_id = '';
-- Runs of existing data: cycles whose last checks end less than an hour apart
-- belong to one run.
CREATE TEMP TABLE _bf_cyc AS
WITH c AS (
SELECT registry_id, cycle_id, MAX(julianday(checked_at)) AS t, MIN(checked_at) AS first_at, MAX(checked_at) AS last_at
FROM checks GROUP BY registry_id, cycle_id
), o AS (
SELECT *, CASE WHEN t - LAG(t) OVER (ORDER BY t) > 60.0 / 1440.0 THEN 1 ELSE 0 END AS brk FROM c
)
SELECT *, 1 + SUM(brk) OVER (ORDER BY t ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW) AS rid FROM o;
CREATE INDEX _bf_cyc_key ON _bf_cyc(registry_id, cycle_id);
INSERT INTO check_runs (id, kind, state, started_at, finalized_at)
SELECT rid, 'manual', 'finalized', MIN(first_at), MAX(last_at) FROM _bf_cyc GROUP BY rid;
UPDATE checks SET run_id = (
SELECT b.rid FROM _bf_cyc b WHERE b.registry_id = checks.registry_id AND b.cycle_id = checks.cycle_id);
UPDATE ip_queue SET run_id = (
SELECT b.rid FROM _bf_cyc b WHERE b.registry_id = ip_queue.registry_id AND b.cycle_id = ip_queue.cycle_id);
-- One result per address and run: the latest cycle of the address in the run.
-- The verdict is the orchestrator's when the queue row still holds that cycle,
-- otherwise it is derived from the stored checks.
INSERT INTO run_results (run_id, registry_id, ip_address, cycle_id, verdict, verdict_derived, aggregated_at, expected_checks, recorded_checks)
SELECT l.rid, l.registry_id, r.ip_address, l.cycle_id,
CASE WHEN q.id IS NOT NULL AND q.overall_result <> '' THEN q.overall_result
WHEN s.ok = 0 THEN 'fail' WHEN s.ok = s.n THEN 'pass' ELSE 'partial' END,
CASE WHEN q.id IS NOT NULL AND q.overall_result <> '' THEN 0 ELSE 1 END,
COALESCE(CASE WHEN q.overall_result <> '' THEN q.aggregated_at END, s.last_at),
(SELECT json_extract(e.payload, '$.checks') + json_extract(e.payload, '$.missing') FROM events e
WHERE e.event_type = 'aggregated' AND e.registry_id = l.registry_id AND e.cycle_id = l.cycle_id
ORDER BY e.id DESC LIMIT 1),
COALESCE((SELECT json_extract(e.payload, '$.checks') FROM events e
WHERE e.event_type = 'aggregated' AND e.registry_id = l.registry_id AND e.cycle_id = l.cycle_id
ORDER BY e.id DESC LIMIT 1), s.n_before)
FROM (SELECT rid, registry_id, MAX(cycle_id) AS cycle_id FROM _bf_cyc GROUP BY rid, registry_id) l
JOIN ip_registry r ON r.id = l.registry_id
JOIN (SELECT registry_id, cycle_id, COUNT(*) AS n, SUM(success) AS ok, MAX(checked_at) AS last_at,
SUM(CASE WHEN after_verdict = 0 THEN 1 ELSE 0 END) AS n_before
FROM checks GROUP BY registry_id, cycle_id) s ON s.registry_id = l.registry_id AND s.cycle_id = l.cycle_id
LEFT JOIN ip_queue q ON q.registry_id = l.registry_id AND q.cycle_id = l.cycle_id;
DROP TABLE _bf_cyc;
+5 -3
View File
@@ -31,8 +31,9 @@ func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
now := timeToDB(Now())
res, err := d.ExecContext(ctx, `
INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id,
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at)
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at, run_id)
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, COALESCE(NULLIF(?, ''), q.owner_validator_id, ''),
?, ?, ?, ?, ?, ?, ?, ?, ?, q.run_id
FROM ip_queue q
WHERE q.id=? AND q.attempt_number=? AND q.state NOT IN (?, ?, ?, ?)
ON CONFLICT(registry_id, cycle_id, source, check_type, target) DO UPDATE SET
@@ -41,7 +42,8 @@ func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
latency_ms=excluded.latency_ms,
detail=excluded.detail,
checked_at=excluded.checked_at,
recorded_at=excluded.recorded_at
recorded_at=excluded.recorded_at,
run_id=excluded.run_id
`, c.IPAddress, c.ValidatorID, c.Source, c.CheckType, c.Target,
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), now, now,
c.IPID, c.AttemptNumber, IPAggregating, IPDone, IPFailed, IPOccupied)
+93 -13
View File
@@ -21,6 +21,7 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
defer tx.Rollback()
now := timeToDB(Now())
runID := int64(0)
for i, addr := range addresses {
var exists bool
if err := tx.QueryRowContext(ctx, `SELECT EXISTS(SELECT 1 FROM ip_queue WHERE ip_address=?)`, addr).Scan(&exists); err != nil {
@@ -37,11 +38,16 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
if err != nil {
return err
}
if runID == 0 {
if runID, err = openRunTx(ctx, tx, RunManual, now); err != nil {
return err
}
}
if _, err := tx.ExecContext(ctx, `
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?)
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(ip_address) DO NOTHING
`, addr, i, IPQueued, registryID, cycle, now, now); err != nil {
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
return fmt.Errorf("seed %s: %w", addr, err)
}
}
@@ -194,16 +200,36 @@ func (d *DB) SetAggregating(ctx context.Context, ipID int64) error {
// FinishIP records the aggregated result and marks the IP done or failed.
func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error {
return d.FinishIPExpected(ctx, ipID, result, -1)
}
// FinishIPExpected is FinishIP that also records, in the address's run, how
// many checks were expected at the verdict (expected < 0: unknown) and
// finalizes the run if this was its last open address.
func (d *DB) FinishIPExpected(ctx context.Context, ipID int64, result string, expected int) error {
state := IPDone
if result == ResultFail {
state = IPFailed
}
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
defer tx.Rollback()
now := timeToDB(Now())
_, err := d.ExecContext(ctx, `
if _, err := tx.ExecContext(ctx, `
UPDATE ip_queue SET state=?, overall_result=?, aggregated_at=?, updated_at=?
WHERE id=?
`, state, result, now, now, ipID)
return err
`, state, result, now, now, ipID); err != nil {
return err
}
if err := upsertRunResultTx(ctx, tx, ipID, result, expected, now); err != nil {
return err
}
if err := finalizeRunsTx(ctx, tx, now); err != nil {
return err
}
return tx.Commit()
}
// ReleaseFIP records that the floating IP has been disassociated and frees
@@ -254,6 +280,9 @@ func (d *DB) MarkFIPOccupied(ctx context.Context, ipID int64, validatorID string
return err
}
}
if err := finalizeRunsTx(ctx, tx, now); err != nil {
return err
}
return tx.Commit()
}
@@ -304,6 +333,12 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
state=?, retry_count=?, overall_result=?, aggregated_at=?, updated_at=?
WHERE id=?
`, nextState, retryCount, ResultFail, now, now, ipID)
if err == nil {
err = upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now)
}
if err == nil {
err = finalizeRunsTx(ctx, tx, now)
}
}
if err != nil {
return err
@@ -337,6 +372,12 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
// sequence, so a batch's relative order is preserved and, critically,
// resubmitting the same list later reproduces the same relative order.
func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult, error) {
return d.SubmitIPsAs(ctx, addresses, RunManual)
}
// SubmitIPsAs is SubmitIPs that names the kind of run it opens when no run is
// open (RunManual or RunAuto); an already open run is joined whatever the kind.
func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (SubmitIPsResult, error) {
var result SubmitIPsResult
if len(addresses) == 0 {
return result, fmt.Errorf("addresses must not be empty: %w", ErrValidation)
@@ -354,6 +395,17 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
}
now := timeToDB(Now())
runID := int64(0)
ensureRun := func() (int64, error) {
if runID == 0 {
id, err := openRunTx(ctx, tx, kind, now)
if err != nil {
return 0, err
}
runID = id
}
return runID, nil
}
for i, addr := range addresses {
seq := base + i
@@ -369,10 +421,14 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
if cErr != nil {
return result, cErr
}
rid, rErr := ensureRun()
if rErr != nil {
return result, rErr
}
if _, err := tx.ExecContext(ctx, `
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?)
`, addr, seq, IPQueued, registryID, cycle, now, now); err != nil {
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
return result, fmt.Errorf("insert %s: %w", addr, err)
}
result.Added = append(result.Added, addr)
@@ -389,14 +445,18 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
if cErr != nil {
return result, cErr
}
rid, rErr := ensureRun()
if rErr != nil {
return result, rErr
}
if _, err := tx.ExecContext(ctx, `
UPDATE ip_queue SET
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
overall_result='',
overall_result='', run_id=?,
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
WHERE ip_address=?
`, IPQueued, seq, cycle, now, addr); err != nil {
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
return result, fmt.Errorf("requeue %s: %w", addr, err)
}
result.Requeued = append(result.Requeued, addr)
@@ -431,7 +491,12 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
// closed by the single-connection transactional UPDATE below).
func (d *DB) CancelIP(ctx context.Context, ipID int64) error {
now := timeToDB(Now())
res, err := d.ExecContext(ctx, `
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
defer tx.Rollback()
res, err := tx.ExecContext(ctx, `
UPDATE ip_queue SET
state=?, overall_result=?, aggregated_at=?, owner_validator_id=NULL, fip_id='',
lease_expires_at=NULL, updated_at=?
@@ -443,7 +508,13 @@ func (d *DB) CancelIP(ctx context.Context, ipID int64) error {
if n, _ := res.RowsAffected(); n == 0 {
return fmt.Errorf("ip_id %d already finished: %w", ipID, ErrInvalidState)
}
return nil
if err := upsertRunResultTx(ctx, tx, ipID, ResultCancelled, -1, now); err != nil {
return err
}
if err := finalizeRunsTx(ctx, tx, now); err != nil {
return err
}
return tx.Commit()
}
// DeleteIP permanently removes an ip_queue row, along with its full check
@@ -463,6 +534,9 @@ func (d *DB) DeleteIP(ctx context.Context, ipID int64) error {
if err := deleteIPTx(ctx, tx, ipID); err != nil {
return err
}
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
return err
}
return tx.Commit()
}
@@ -497,6 +571,9 @@ func (d *DB) DeleteIPs(ctx context.Context, addresses []string) (DeleteIPsResult
result.Deleted = append(result.Deleted, addr)
}
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
return result, err
}
if err := tx.Commit(); err != nil {
return result, err
}
@@ -592,6 +669,9 @@ func (d *DB) ClearAllIPs(ctx context.Context) ([]string, error) {
if _, err := tx.ExecContext(ctx, `DELETE FROM ip_queue`); err != nil {
return nil, fmt.Errorf("delete ip_queue rows: %w", err)
}
if err := finalizeRunsTx(ctx, tx, now); err != nil {
return nil, err
}
if err := tx.Commit(); err != nil {
return nil, err
}
+45
View File
@@ -4,6 +4,7 @@ import (
"context"
"database/sql"
"fmt"
"net/netip"
"sort"
"strings"
"time"
@@ -120,6 +121,34 @@ func (d *DB) ListRegistry(ctx context.Context) ([]RegistrySummary, error) {
type RegistryFilter struct {
Query string // substring of ip_address
LastResult string // pass|partial|fail|cancelled — same meaning as RegistrySummary.LastResult
RunID int64 // only addresses that have a result in this run
Subnet string // only addresses inside this CIDR
}
// subnetIDs returns the registry ids of the addresses inside prefix. SQLite
// has no CIDR operators, so the registry's addresses are filtered here.
func (d *DB) subnetIDs(ctx context.Context, cidr string) ([]any, error) {
p, err := netip.ParsePrefix(cidr)
if err != nil {
return nil, fmt.Errorf("subnet %q: %v: %w", cidr, err, ErrValidation)
}
rows, err := d.QueryContext(ctx, `SELECT id, ip_address FROM ip_registry`)
if err != nil {
return nil, err
}
defer rows.Close()
var ids []any
for rows.Next() {
var id int64
var ip string
if err := rows.Scan(&id, &ip); err != nil {
return nil, err
}
if a, err := netip.ParseAddr(ip); err == nil && p.Contains(a) {
ids = append(ids, id)
}
}
return ids, rows.Err()
}
// lastResultCond is the SQL form of fillRegistrySummary's LastResult rule,
@@ -155,6 +184,22 @@ func (d *DB) ListRegistryPage(ctx context.Context, f RegistryFilter, limit, offs
conds = append(conds, lastResultCond)
args = append(args, f.LastResult, f.LastResult)
}
if f.RunID > 0 {
conds = append(conds, "r.id IN (SELECT registry_id FROM run_results WHERE run_id = ?)")
args = append(args, f.RunID)
}
if f.Subnet != "" {
ids, err := d.subnetIDs(ctx, f.Subnet)
if err != nil {
return nil, 0, err
}
if len(ids) == 0 {
conds = append(conds, "0 = 1")
} else {
conds = append(conds, "r.id IN ("+strings.TrimSuffix(strings.Repeat("?,", len(ids)), ",")+")")
args = append(args, ids...)
}
}
from := ` FROM ip_registry r LEFT JOIN ip_queue q ON q.registry_id = r.id `
where := ""
if len(conds) > 0 {
+387
View File
@@ -0,0 +1,387 @@
package db
import (
"context"
"database/sql"
"fmt"
"net/netip"
"sort"
"time"
)
// Run kinds and states. A run groups the cycles of one launch of checks — see
// migrations/0011_check_runs.sql.
const (
RunManual = "manual"
RunAuto = "auto"
RunOpen = "open"
RunFinalized = "finalized"
)
// CheckRun is one launch of checks. FinalizedAt is nil while it is open.
type CheckRun struct {
ID int64
Kind string
State string
StartedAt time.Time
FinalizedAt *time.Time
}
// RunCounts is the verdict tally of a run.
type RunCounts struct {
Addresses int
Pass int
Partial int
Fail int
Cancelled int
}
// RunSummary is a run with its verdict tally, for the run selector. Pending
// is the number of its queue rows still being processed (open runs only).
type RunSummary struct {
CheckRun
RunCounts
Pending int
Total int
}
// openRunTx returns the id of the open run, creating one of the given kind if
// none is open. Called when addresses enter the queue: while a run is open
// everything submitted or re-checked joins it.
func openRunTx(ctx context.Context, tx *sql.Tx, kind, now string) (int64, error) {
var id int64
err := tx.QueryRowContext(ctx, `SELECT id FROM check_runs WHERE state=? ORDER BY id DESC LIMIT 1`, RunOpen).Scan(&id)
if err == nil {
return id, nil
}
if err != sql.ErrNoRows {
return 0, err
}
if kind == "" {
kind = RunManual
}
res, err := tx.ExecContext(ctx, `INSERT INTO check_runs (kind, state, started_at) VALUES (?, ?, ?)`, kind, RunOpen, now)
if err != nil {
return 0, fmt.Errorf("open run: %w", err)
}
return res.LastInsertId()
}
// finalizeRunsTx finalizes every open run that has no queue row left in a
// non-terminal state, and drops finalized runs that ended up with no results
// at all (everything was deleted before any verdict). Called wherever a queue
// row reaches a terminal state or disappears.
func finalizeRunsTx(ctx context.Context, tx *sql.Tx, now string) error {
if _, err := tx.ExecContext(ctx, `
UPDATE check_runs SET state=?, finalized_at=COALESCE(
(SELECT MAX(aggregated_at) FROM run_results WHERE run_id=check_runs.id), ?)
WHERE state=? AND NOT EXISTS (
SELECT 1 FROM ip_queue q WHERE q.run_id=check_runs.id AND q.state NOT IN (?, ?, ?))
`, RunFinalized, now, RunOpen, IPDone, IPFailed, IPOccupied); err != nil {
return fmt.Errorf("finalize runs: %w", err)
}
if _, err := tx.ExecContext(ctx, `
DELETE FROM check_runs WHERE state=? AND NOT EXISTS (SELECT 1 FROM run_results r WHERE r.run_id=check_runs.id)
`, RunFinalized); err != nil {
return fmt.Errorf("drop empty runs: %w", err)
}
return nil
}
// FinalizeRuns finalizes runs whose addresses are all done. The orchestrator
// calls it every tick as a safety net; the terminal transitions already do it.
func (d *DB) FinalizeRuns(ctx context.Context) error {
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
defer tx.Rollback()
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
return err
}
return tx.Commit()
}
// upsertRunResultTx stores the verdict of the address's current cycle in its
// run. A re-check inside an open run replaces the earlier result. expected < 0
// means unknown. Rows without a run (queue rows that predate runs and never
// got one) are skipped.
func upsertRunResultTx(ctx context.Context, tx *sql.Tx, ipID int64, verdict string, expected int, now string) error {
var exp sql.NullInt64
if expected >= 0 {
exp = sql.NullInt64{Int64: int64(expected), Valid: true}
}
_, err := tx.ExecContext(ctx, `
INSERT INTO run_results (run_id, registry_id, ip_address, cycle_id, verdict, aggregated_at, expected_checks, recorded_checks)
SELECT q.run_id, q.registry_id, q.ip_address, q.cycle_id, ?, ?, ?,
(SELECT COUNT(*) FROM checks c WHERE c.registry_id=q.registry_id AND c.cycle_id=q.cycle_id)
FROM ip_queue q WHERE q.id=? AND q.run_id IS NOT NULL
ON CONFLICT(run_id, registry_id) DO UPDATE SET
cycle_id=excluded.cycle_id, verdict=excluded.verdict, verdict_derived=0,
aggregated_at=excluded.aggregated_at, expected_checks=excluded.expected_checks,
recorded_checks=excluded.recorded_checks
`, verdict, now, exp, ipID)
if err != nil {
return fmt.Errorf("record run result: %w", err)
}
return nil
}
// adoptOrphanQueueRows gives every live queue row without a run (rows that
// predate runs) the open run, creating one. Runs once after migrating.
func (d *DB) adoptOrphanQueueRows(ctx context.Context) error {
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
defer tx.Rollback()
var n int
if err := tx.QueryRowContext(ctx, `SELECT COUNT(*) FROM ip_queue WHERE run_id IS NULL AND state NOT IN (?, ?, ?)`,
IPDone, IPFailed, IPOccupied).Scan(&n); err != nil {
return err
}
if n == 0 {
return nil
}
now := timeToDB(Now())
id, err := openRunTx(ctx, tx, RunManual, now)
if err != nil {
return err
}
if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET run_id=? WHERE run_id IS NULL AND state NOT IN (?, ?, ?)`,
id, IPDone, IPFailed, IPOccupied); err != nil {
return err
}
return tx.Commit()
}
// ListRuns returns every run, newest first, with its verdict tally.
func (d *DB) ListRuns(ctx context.Context) ([]RunSummary, error) {
rows, err := d.QueryContext(ctx, `
SELECT r.id, r.kind, r.state, r.started_at, r.finalized_at,
COALESCE(SUM(CASE WHEN x.verdict IS NOT NULL THEN 1 ELSE 0 END), 0),
COALESCE(SUM(CASE WHEN x.verdict='pass' THEN 1 ELSE 0 END), 0),
COALESCE(SUM(CASE WHEN x.verdict='partial' THEN 1 ELSE 0 END), 0),
COALESCE(SUM(CASE WHEN x.verdict='fail' THEN 1 ELSE 0 END), 0),
COALESCE(SUM(CASE WHEN x.verdict='cancelled' THEN 1 ELSE 0 END), 0),
(SELECT COUNT(*) FROM ip_queue q WHERE q.run_id=r.id),
(SELECT COUNT(*) FROM ip_queue q WHERE q.run_id=r.id AND q.state NOT IN (?, ?, ?))
FROM check_runs r LEFT JOIN run_results x ON x.run_id=r.id
GROUP BY r.id ORDER BY r.id DESC
`, IPDone, IPFailed, IPOccupied)
if err != nil {
return nil, err
}
defer rows.Close()
var out []RunSummary
for rows.Next() {
var s RunSummary
var started string
var finalized sql.NullString
if err := rows.Scan(&s.ID, &s.Kind, &s.State, &started, &finalized,
&s.Addresses, &s.Pass, &s.Partial, &s.Fail, &s.Cancelled, &s.Total, &s.Pending); err != nil {
return nil, err
}
if s.StartedAt, err = dbToTime(started); err != nil {
return nil, err
}
if s.FinalizedAt, err = nullStringToTimePtr(finalized); err != nil {
return nil, err
}
out = append(out, s)
}
return out, rows.Err()
}
// GetRun returns one run, or ErrNotFound.
func (d *DB) GetRun(ctx context.Context, id int64) (*CheckRun, error) {
var r CheckRun
var started string
var finalized sql.NullString
err := d.QueryRowContext(ctx, `SELECT id, kind, state, started_at, finalized_at FROM check_runs WHERE id=?`, id).
Scan(&r.ID, &r.Kind, &r.State, &started, &finalized)
if err == sql.ErrNoRows {
return nil, fmt.Errorf("run %d: %w", id, ErrNotFound)
}
if err != nil {
return nil, err
}
if r.StartedAt, err = dbToTime(started); err != nil {
return nil, err
}
if r.FinalizedAt, err = nullStringToTimePtr(finalized); err != nil {
return nil, err
}
return &r, nil
}
// RunResult is one address's result in a run.
type RunResult struct {
RegistryID int64
IPAddress string
CycleID int
Verdict string
Derived bool
AggregatedAt time.Time
ExpectedChecks int // -1 when unknown
RecordedChecks int
}
// ListRunResults returns the results of a run in address order of insertion.
func (d *DB) ListRunResults(ctx context.Context, runID int64) ([]RunResult, error) {
rows, err := d.QueryContext(ctx, `
SELECT registry_id, ip_address, cycle_id, verdict, verdict_derived, aggregated_at, expected_checks, recorded_checks
FROM run_results WHERE run_id=? ORDER BY registry_id`, runID)
if err != nil {
return nil, err
}
defer rows.Close()
var out []RunResult
for rows.Next() {
var r RunResult
var agg string
var exp sql.NullInt64
if err := rows.Scan(&r.RegistryID, &r.IPAddress, &r.CycleID, &r.Verdict, &r.Derived, &agg, &exp, &r.RecordedChecks); err != nil {
return nil, err
}
if r.AggregatedAt, err = dbToTime(agg); err != nil {
return nil, err
}
r.ExpectedChecks = -1
if exp.Valid {
r.ExpectedChecks = int(exp.Int64)
}
out = append(out, r)
}
return out, rows.Err()
}
// RunCheck is one stored check of a run's result cycle, with what the
// analytics needs to place it.
type RunCheck struct {
RegistryID int64
Source string
CheckType string
Target string
Success bool
ValidatorID string
Detail string
RecordedAt time.Time
AfterVerdict bool
}
// EachRunCheck calls fn for every check of the cycles that make up the run's
// results (the latest cycle of each address in the run), in one pass over the
// run_id index. fn must not call back into the DB (one connection).
func (d *DB) EachRunCheck(ctx context.Context, runID int64, fn func(RunCheck)) error {
rows, err := d.QueryContext(ctx, `
SELECT c.registry_id, c.source, c.check_type, c.target, c.success, c.validator_id, c.detail,
COALESCE(c.recorded_at, c.created_at), c.after_verdict
FROM checks c JOIN run_results r ON r.run_id=c.run_id AND r.registry_id=c.registry_id AND r.cycle_id=c.cycle_id
WHERE c.run_id=?`, runID)
if err != nil {
return err
}
defer rows.Close()
for rows.Next() {
var c RunCheck
var rec string
if err := rows.Scan(&c.RegistryID, &c.Source, &c.CheckType, &c.Target, &c.Success, &c.ValidatorID, &c.Detail, &rec, &c.AfterVerdict); err != nil {
return err
}
if c.RecordedAt, err = dbToTime(rec); err != nil {
return err
}
fn(c)
}
return rows.Err()
}
// RunDataVersion changes whenever a check of the run is written, so a cache of
// computed analytics can tell when it is stale.
func (d *DB) RunDataVersion(ctx context.Context, runID int64) (string, error) {
var maxID, n sql.NullInt64
var rec sql.NullString
if err := d.QueryRowContext(ctx, `SELECT MAX(id), COUNT(*), MAX(recorded_at) FROM checks WHERE run_id=?`, runID).Scan(&maxID, &n, &rec); err != nil {
return "", err
}
var res sql.NullString
if err := d.QueryRowContext(ctx, `SELECT MAX(aggregated_at) FROM run_results WHERE run_id=?`, runID).Scan(&res); err != nil {
return "", err
}
return fmt.Sprintf("%d/%d/%s/%s", maxID.Int64, n.Int64, rec.String, res.String), nil
}
// Subnet is one entry of the administrator's subnet list.
type Subnet struct {
CIDR string
Label string
}
// ListSubnets returns the configured subnets, sorted by prefix.
func (d *DB) ListSubnets(ctx context.Context) ([]Subnet, error) {
rows, err := d.QueryContext(ctx, `SELECT cidr, label FROM subnets`)
if err != nil {
return nil, err
}
defer rows.Close()
var out []Subnet
for rows.Next() {
var s Subnet
if err := rows.Scan(&s.CIDR, &s.Label); err != nil {
return nil, err
}
out = append(out, s)
}
if err := rows.Err(); err != nil {
return nil, err
}
sort.Slice(out, func(i, j int) bool {
a, _ := netip.ParsePrefix(out[i].CIDR)
b, _ := netip.ParsePrefix(out[j].CIDR)
if a.Addr() != b.Addr() {
return a.Addr().Less(b.Addr())
}
return a.Bits() < b.Bits()
})
return out, nil
}
// ReplaceSubnets replaces the whole subnet list. Every entry must be a valid
// CIDR; entries are stored in canonical form (host bits cleared) and
// duplicates collapse.
func (d *DB) ReplaceSubnets(ctx context.Context, subnets []Subnet) error {
canon := map[string]string{}
for _, s := range subnets {
p, err := netip.ParsePrefix(s.CIDR)
if err != nil {
return fmt.Errorf("subnet %q: %v: %w", s.CIDR, err, ErrValidation)
}
canon[p.Masked().String()] = s.Label
}
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
defer tx.Rollback()
if _, err := tx.ExecContext(ctx, `DELETE FROM subnets`); err != nil {
return err
}
for cidr, label := range canon {
if _, err := tx.ExecContext(ctx, `INSERT INTO subnets (cidr, label) VALUES (?, ?)`, cidr, label); err != nil {
return err
}
}
return tx.Commit()
}
// CountRecheckedInRun returns how many addresses have more than one cycle of
// checks inside the run (re-checked while the run was open).
func (d *DB) CountRecheckedInRun(ctx context.Context, runID int64) (int, error) {
var n int
err := d.QueryRowContext(ctx, `
SELECT COUNT(*) FROM (SELECT registry_id FROM checks WHERE run_id=? GROUP BY registry_id HAVING COUNT(DISTINCT cycle_id) > 1)
`, runID).Scan(&n)
return n, err
}
+389
View File
@@ -0,0 +1,389 @@
package db
import (
"context"
"database/sql"
"errors"
"path/filepath"
"testing"
"time"
)
func submit(t *testing.T, d *DB, kind string, addrs ...string) {
t.Helper()
if _, err := d.SubmitIPsAs(context.Background(), addrs, kind); err != nil {
t.Fatal(err)
}
}
func finish(t *testing.T, d *DB, addr, verdict string, expected int) {
t.Helper()
ctx := context.Background()
ip, err := d.GetIPByAddress(ctx, addr)
if err != nil {
t.Fatal(err)
}
if err := d.FinishIPExpected(ctx, ip.ID, verdict, expected); err != nil {
t.Fatal(err)
}
}
func runs(t *testing.T, d *DB) []RunSummary {
t.Helper()
r, err := d.ListRuns(context.Background())
if err != nil {
t.Fatal(err)
}
return r
}
// An address entering an idle queue opens a run; everything submitted while it
// is open joins it; it is finalized when the last address is done.
func TestRunOpensJoinsAndFinalizes(t *testing.T) {
d, ctx := newTestDB(t)
submit(t, d, RunAuto, "1.1.1.1", "2.2.2.2")
submit(t, d, RunManual, "3.3.3.3") // joins the open run, kind stays auto
rs := runs(t, d)
if len(rs) != 1 || rs[0].State != RunOpen || rs[0].Kind != RunAuto || rs[0].Total != 3 || rs[0].Pending != 3 {
t.Fatalf("expected one open auto run with 3 pending rows: %+v", rs)
}
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
c, _ := d.GetIPByAddress(ctx, "3.3.3.3")
if a.ID == 0 || c.ID == 0 {
t.Fatal("rows missing")
}
finish(t, d, "1.1.1.1", ResultPass, 22)
finish(t, d, "2.2.2.2", ResultPartial, 22)
if rs = runs(t, d); rs[0].State != RunOpen || rs[0].Addresses != 2 || rs[0].Pending != 1 {
t.Fatalf("one address still pending, the run stays open: %+v", rs[0])
}
finish(t, d, "3.3.3.3", ResultFail, -1)
rs = runs(t, d)
if rs[0].State != RunFinalized || rs[0].FinalizedAt == nil || rs[0].Pass != 1 || rs[0].Partial != 1 || rs[0].Fail != 1 || rs[0].Addresses != 3 {
t.Fatalf("expected a finalized run with the three verdicts: %+v", rs[0])
}
res, err := d.ListRunResults(ctx, rs[0].ID)
if err != nil || len(res) != 3 {
t.Fatalf("results: %v %v", res, err)
}
byIP := map[string]RunResult{}
for _, r := range res {
byIP[r.IPAddress] = r
}
if byIP["1.1.1.1"].ExpectedChecks != 22 || byIP["3.3.3.3"].ExpectedChecks != -1 || byIP["2.2.2.2"].Verdict != ResultPartial {
t.Fatalf("results: %+v", byIP)
}
}
// A re-check after the run is finalized opens a new run and leaves the old one
// as it was; the old cycle's checks keep their run.
func TestRecheckAfterFinalizeOpensNewRun(t *testing.T) {
d, ctx := newTestDB(t)
ip := checkingIP(t, d, "1.1.1.1")
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", true)); err != nil {
t.Fatal(err)
}
finish(t, d, "1.1.1.1", ResultPass, 1)
first := runs(t, d)[0]
if first.State != RunFinalized {
t.Fatalf("expected finalized: %+v", first)
}
submit(t, d, RunManual, "1.1.1.1") // re-check of a finished address
rs := runs(t, d)
if len(rs) != 2 || rs[0].State != RunOpen || rs[0].ID == first.ID {
t.Fatalf("a re-check after the run ended must open a new run: %+v", rs)
}
ip, _ = d.GetIPByAddress(ctx, "1.1.1.1")
if err := d.SetChecking(ctx, ip.ID, time.Minute); err != nil {
t.Fatal(err)
}
ip, _ = d.GetIP(ctx, ip.ID)
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", false)); err != nil {
t.Fatal(err)
}
finish(t, d, "1.1.1.1", ResultFail, 1)
var n1, n2 int
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id=?`, first.ID).Scan(&n1)
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id=?`, rs[0].ID).Scan(&n2)
if n1 != 1 || n2 != 1 {
t.Fatalf("each run keeps its own cycle's check: %d %d", n1, n2)
}
old, _ := d.ListRunResults(ctx, first.ID)
cur, _ := d.ListRunResults(ctx, rs[0].ID)
if len(old) != 1 || old[0].Verdict != ResultPass || old[0].CycleID != 1 || len(cur) != 1 || cur[0].Verdict != ResultFail || cur[0].CycleID != 2 {
t.Fatalf("results must not cross: old=%+v cur=%+v", old, cur)
}
// The checks of a run are the ones of its result cycle, nothing else.
var got []RunCheck
if err := d.EachRunCheck(ctx, first.ID, func(c RunCheck) { got = append(got, c) }); err != nil || len(got) != 1 || !got[0].Success {
t.Fatalf("run 1 checks: %+v %v", got, err)
}
}
// A re-check while the run is still open joins it and replaces the address's
// result, so a run holds one result per address.
func TestRecheckInsideOpenRunReplacesResult(t *testing.T) {
d, ctx := newTestDB(t)
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
finish(t, d, "1.1.1.1", ResultFail, 1)
submit(t, d, RunManual, "1.1.1.1") // while 2.2.2.2 is still pending
finish(t, d, "1.1.1.1", ResultPass, 1)
finish(t, d, "2.2.2.2", ResultPass, 1)
rs := runs(t, d)
if len(rs) != 1 || rs[0].Addresses != 2 || rs[0].Pass != 2 || rs[0].Fail != 0 {
t.Fatalf("expected one run with the latest verdicts: %+v", rs)
}
res, _ := d.ListRunResults(ctx, rs[0].ID)
for _, r := range res {
if r.IPAddress == "1.1.1.1" && r.CycleID != 2 {
t.Fatalf("the re-checked address must show its latest cycle: %+v", r)
}
}
if n, err := d.CountRecheckedInRun(ctx, rs[0].ID); err != nil || n != 0 {
t.Fatalf("no checks stored, so no re-check counted: %d %v", n, err)
}
}
func TestRunEndsWhenQueueIsClearedOrDeleted(t *testing.T) {
d, ctx := newTestDB(t)
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
finish(t, d, "1.1.1.1", ResultPass, 1)
if _, err := d.ClearAllIPs(ctx); err != nil {
t.Fatal(err)
}
rs := runs(t, d)
if len(rs) != 1 || rs[0].State != RunFinalized || rs[0].Addresses != 1 {
t.Fatalf("clearing ends the run with what it has: %+v", rs)
}
// The next submission is a new run, not a join of the ended one.
submit(t, d, RunManual, "3.3.3.3")
if rs = runs(t, d); len(rs) != 2 || rs[0].State != RunOpen {
t.Fatalf("expected a new open run: %+v", rs)
}
// A run with no result at all disappears when its rows are deleted.
ip, _ := d.GetIPByAddress(ctx, "3.3.3.3")
if err := d.DeleteIP(ctx, ip.ID); err != nil {
t.Fatal(err)
}
if rs = runs(t, d); len(rs) != 1 {
t.Fatalf("an empty run must be dropped: %+v", rs)
}
}
func TestCancelAndRetryFailureRecordResults(t *testing.T) {
d, ctx := newTestDB(t)
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
if err := d.CancelIP(ctx, a.ID); err != nil {
t.Fatal(err)
}
b, _ := d.GetIPByAddress(ctx, "2.2.2.2")
if err := d.RequeueOrFail(ctx, b.ID, "", 0); err != nil { // retries exhausted at once
t.Fatal(err)
}
rs := runs(t, d)
if rs[0].State != RunFinalized || rs[0].Cancelled != 1 || rs[0].Fail != 1 {
t.Fatalf("cancelled and failed addresses are results of the run: %+v", rs[0])
}
}
// Ingress checks name the validator that held the address.
func TestIngressCheckTakesValidatorOfTheAddress(t *testing.T) {
d, ctx := newTestDB(t)
if err := d.RegisterValidator(ctx, "validator-7", "host", "port", "v"); err != nil {
t.Fatal(err)
}
ip := checkingIP(t, d, "1.1.1.1")
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET owner_validator_id='validator-7' WHERE id=?`, ip.ID); err != nil {
t.Fatal(err)
}
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", true)); err != nil {
t.Fatal(err)
}
var v string
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE ip_id=?`, ip.ID).Scan(&v); err != nil || v != "validator-7" {
t.Fatalf("validator of an ingress check = %q err=%v", v, err)
}
// An explicit validator (egress checks) is kept.
c := checkOf(ip, "https", true)
c.Source, c.ValidatorID, c.Target = SourceEgress, "validator-9", "https://x"
if _, err := d.UpsertCheckIfOpen(ctx, c); err != nil {
t.Fatal(err)
}
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE ip_id=? AND source='egress'`, ip.ID).Scan(&v); err != nil || v != "validator-9" {
t.Fatalf("egress validator = %q err=%v", v, err)
}
}
func TestSubnetsReplaceAndValidate(t *testing.T) {
d, ctx := newTestDB(t)
if err := d.ReplaceSubnets(ctx, []Subnet{{CIDR: "10.1.2.3/24", Label: "a"}, {CIDR: "10.0.0.0/8"}, {CIDR: "10.1.2.0/24", Label: "dup"}}); err != nil {
t.Fatal(err)
}
got, err := d.ListSubnets(ctx)
if err != nil || len(got) != 2 || got[0].CIDR != "10.0.0.0/8" || got[1].CIDR != "10.1.2.0/24" {
t.Fatalf("subnets must be canonical, de-duplicated and sorted by prefix: %+v %v", got, err)
}
if err := d.ReplaceSubnets(ctx, []Subnet{{CIDR: "nonsense"}}); !errors.Is(err, ErrValidation) {
t.Fatalf("expected ErrValidation, got %v", err)
}
if got, _ := d.ListSubnets(ctx); len(got) != 2 {
t.Fatalf("a rejected list must leave the old one: %+v", got)
}
if err := d.ReplaceSubnets(ctx, nil); err != nil {
t.Fatal(err)
}
if got, _ := d.ListSubnets(ctx); len(got) != 0 {
t.Fatalf("an empty list clears: %+v", got)
}
}
func TestRegistryFilterByRunAndSubnet(t *testing.T) {
d, ctx := newTestDB(t)
submit(t, d, RunManual, "10.0.0.1", "10.0.0.2", "10.0.1.1")
for _, a := range []string{"10.0.0.1", "10.0.0.2", "10.0.1.1"} {
finish(t, d, a, ResultPass, -1)
}
runID := runs(t, d)[0].ID
submit(t, d, RunManual, "10.0.0.2") // second run holds only this address
finish(t, d, "10.0.0.2", ResultFail, -1)
secondID := runs(t, d)[0].ID
count := func(f RegistryFilter) int {
t.Helper()
_, total, err := d.ListRegistryPage(ctx, f, 50, 0)
if err != nil {
t.Fatal(err)
}
return total
}
if n := count(RegistryFilter{RunID: runID}); n != 3 {
t.Errorf("run 1: %d", n)
}
if n := count(RegistryFilter{RunID: secondID}); n != 1 {
t.Errorf("run 2: %d", n)
}
if n := count(RegistryFilter{Subnet: "10.0.0.0/24"}); n != 2 {
t.Errorf("subnet /24: %d", n)
}
if n := count(RegistryFilter{RunID: secondID, Subnet: "10.0.1.0/24"}); n != 0 {
t.Errorf("run 2 and the other subnet: %d", n)
}
if n := count(RegistryFilter{Subnet: "192.168.0.0/16"}); n != 0 {
t.Errorf("subnet with no address: %d", n)
}
if _, _, err := d.ListRegistryPage(ctx, RegistryFilter{Subnet: "x"}, 10, 0); !errors.Is(err, ErrValidation) {
t.Errorf("bad subnet: %v", err)
}
}
// Migration 0011 on a database of version 10: runs are cut at pauses of more
// than an hour, results come from the queue row or the checks, ingress checks
// get their validator from the fip_associated event, and live queue rows
// without a run are adopted into an open run when the database is opened.
func TestMigration0011BuildsRunsFromExistingData(t *testing.T) {
ctx := context.Background()
path := filepath.Join(t.TempDir(), "old.db")
raw, err := sql.Open("sqlite", path)
if err != nil {
t.Fatal(err)
}
raw.SetMaxOpenConns(1)
for _, m := range migrations {
if m.version > 10 {
break
}
if _, err := raw.ExecContext(ctx, m.sql); err != nil {
t.Fatalf("migration %d: %v", m.version, err)
}
}
raw.ExecContext(ctx, `PRAGMA user_version=10`)
exec := func(q string, args ...any) {
t.Helper()
if _, err := raw.ExecContext(ctx, q, args...); err != nil {
t.Fatalf("seed: %v\n%s", err, q)
}
}
const ts = "2026-10-02T13:00:00Z"
for i, ip := range []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"} {
exec(`INSERT INTO ip_registry (id, ip_address, first_seen_at, last_seen_at, next_cycle, created_at, updated_at) VALUES (?, ?, ?, ?, 3, ?, ?)`, i+1, ip, ts, ts, ts, ts)
}
// 1.1.1.1 and 2.2.2.2 finish minutes apart (run 1); 1.1.1.1 is checked
// again three hours later (run 2). 3.3.3.3 is still in the queue.
exec(`INSERT INTO ip_queue (id, ip_address, sequence, state, overall_result, aggregated_at, registry_id, cycle_id, created_at, updated_at)
VALUES (1, '1.1.1.1', 1, 'done', 'fail', '2026-10-02T16:00:30Z', 1, 2, ?, ?), (2, '2.2.2.2', 2, 'done', 'pass', '2026-10-02T13:05:30Z', 2, 1, ?, ?),
(3, '3.3.3.3', 3, 'checking', '', NULL, 3, 1, ?, ?)`, ts, ts, ts, ts, ts, ts)
chk := func(reg, cyc int, src, typ string, ok int, at string) {
exec(`INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target, success, checked_at, created_at)
VALUES (?, ?, ?, 'x', 1, '', ?, ?, 't', ?, ?, ?)`, reg, cyc, reg, src, typ, ok, at, at)
}
chk(1, 1, "egress", "https", 1, "2026-10-02T13:00:10Z")
chk(1, 1, "inbound-site-1", "icmp", 1, "2026-10-02T13:00:20Z")
chk(2, 1, "egress", "https", 1, "2026-10-02T13:05:00Z")
chk(1, 2, "egress", "https", 0, "2026-10-02T16:00:10Z")
exec(`INSERT INTO events (source_type, source_id, ip_id, event_type, payload, occurred_at, registry_id, cycle_id) VALUES
('control-api', '', 1, 'fip_associated', '{"fip_id":"f","validator_id":"vkiplab-v5"}', ?, 1, 1),
('control-api', '', 2, 'aggregated', '{"result":"pass","checks":1,"passed":1,"missing":1}', ?, 2, 1)`, ts, ts)
raw.Close()
d, err := Open(ctx, path)
if err != nil {
t.Fatalf("open (runs migration 11): %v", err)
}
defer d.Close()
rs := runs(t, d)
// run 1 and run 2 from the checks, plus the open run that adopted 3.3.3.3
if len(rs) != 3 {
t.Fatalf("expected 3 runs, got %+v", rs)
}
var first, second, open RunSummary
for _, r := range rs {
switch {
case r.State == RunOpen:
open = r
case r.Addresses == 2:
first = r
default:
second = r
}
}
if first.Pass != 2 || first.Fail != 0 || second.Fail != 1 || second.Addresses != 1 || open.Total != 1 {
t.Fatalf("runs: first=%+v second=%+v open=%+v", first, second, open)
}
res, _ := d.ListRunResults(ctx, first.ID)
for _, r := range res {
if r.IPAddress == "1.1.1.1" && (r.CycleID != 1 || r.Verdict != ResultPass || !r.Derived) {
// the queue row holds the later cycle, so this one is derived from its checks
t.Errorf("1.1.1.1 in the first run: %+v", r)
}
if r.IPAddress == "2.2.2.2" && (r.ExpectedChecks != 2 || r.RecordedChecks != 1 || r.Verdict != ResultPass || r.Derived) {
t.Errorf("result from the aggregated event: %+v", r)
}
}
res2, _ := d.ListRunResults(ctx, second.ID)
if len(res2) != 1 || res2[0].CycleID != 2 || res2[0].Verdict != ResultFail || res2[0].Derived {
t.Errorf("second run result: %+v", res2)
}
var v string
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE source='inbound-site-1'`).Scan(&v); err != nil || v != "vkiplab-v5" {
t.Errorf("ingress validator = %q err=%v", v, err)
}
var unset int
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id IS NULL`).Scan(&unset)
if unset != 0 {
t.Errorf("%d checks without a run", unset)
}
var ver int
d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver)
if ver != 11 {
t.Errorf("user_version = %d", ver)
}
}
@@ -240,7 +240,7 @@ func TestMigration0010MarksRowsAfterVerdict(t *testing.T) {
t.Errorf("ssh: after_verdict=%d recorded=%s created=%s", a, rec, cr)
}
var ver int
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 10 {
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 11 {
t.Errorf("user_version=%d err=%v", ver, err)
}
}