Add the Analytics section: check runs, analytics API and page
Runs (migration 0011): a run groups the cycles of one launch. It opens when an address enters an idle queue, takes everything submitted or re-checked while it is open and is finalized when all its addresses are done; a re-check after that opens a new run, so results of different runs never mix. check_runs, run_results (one result per address and run, with the verdict and the expected and stored check counts), subnets, run_id on ip_queue and checks. Existing data is split into runs at pauses of more than an hour; ingress checks get the validator that held the address (also at write time from now on). Analytics (internal/analytics): figures computed from the stored checks of the latest cycle of each address in the run, as facts next to the verdict: summary, reasons of partial, data quality, subnets, targets and the subnet x target matrix by check type, ingress by site, error classes, validators, and the address lists behind the indicators and error classes. API: analytics runs, report, lists (JSON or CSV), subnet list; run and subnet filters for the registry. Dashboard: /analytics matching the approved mockup (run selector, indicators with address lists and CSV, error-class dialogs, drill-down to the registry), subnet list on /settings. Sidebar: the control-api link state, theme toggle and logout moved to the top, the three dots next to the logo removed, sections grouped. Rebuilt bin/control-api and bin/admin-dashboard to match. Plan, summary and the updated README, API, USAGE, DASHBOARD and ADMIN_CLEANUP docs are in docs/. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
864208238f
commit
b7669c9e41
44 files changed
+4123
-62
No files matched your search
@@ -43,6 +43,9 @@ var scaleIndexesSchema string
|
||||
//go:embed migrations/0010_verdict_integrity.sql
|
||||
var verdictIntegritySchema string
|
||||
|
||||
//go:embed migrations/0011_check_runs.sql
|
||||
var checkRunsSchema string
|
||||
|
||||
// migrations is the ordered list of schema versions. Each entry's SQL is
|
||||
// applied, in order, for any version greater than the database's current
|
||||
// PRAGMA user_version — so a fresh database walks the whole list and an
|
||||
@@ -61,6 +64,7 @@ var migrations = []struct {
|
||||
{8, autoCycleSchema},
|
||||
{9, scaleIndexesSchema},
|
||||
{10, verdictIntegritySchema},
|
||||
{11, checkRunsSchema},
|
||||
}
|
||||
|
||||
type DB struct {
|
||||
@@ -96,6 +100,10 @@ func Open(ctx context.Context, path string) (*DB, error) {
|
||||
sqlDB.Close()
|
||||
return nil, fmt.Errorf("migrate: %w", err)
|
||||
}
|
||||
if err := d.adoptOrphanQueueRows(ctx); err != nil {
|
||||
sqlDB.Close()
|
||||
return nil, fmt.Errorf("adopt queue rows into a run: %w", err)
|
||||
}
|
||||
return d, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
-- Check runs (see docs/changes/2026-10-03_16-39_analytics-section-plan.md).
|
||||
--
|
||||
-- cycle_id counts per address, so it cannot tell one launch from another. A
|
||||
-- run groups the cycles of one launch: it opens when an address enters an idle
|
||||
-- queue, takes every address submitted or re-checked while it is open, and is
|
||||
-- finalized when all its queue rows are terminal. checks.run_id and
|
||||
-- ip_queue.run_id carry the membership; run_results keeps one row per address
|
||||
-- and run (its latest cycle) with the verdict, which ip_queue overwrites on a
|
||||
-- re-check.
|
||||
--
|
||||
-- No foreign keys on purpose: the manual cleanup in docs/ADMIN_CLEANUP.md
|
||||
-- deletes from these tables freely.
|
||||
|
||||
CREATE TABLE check_runs (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
kind TEXT NOT NULL DEFAULT 'manual', -- manual | auto
|
||||
state TEXT NOT NULL DEFAULT 'open', -- open | finalized
|
||||
started_at TIMESTAMP NOT NULL,
|
||||
finalized_at TIMESTAMP
|
||||
);
|
||||
|
||||
CREATE TABLE run_results (
|
||||
run_id INTEGER NOT NULL,
|
||||
registry_id INTEGER NOT NULL,
|
||||
ip_address TEXT NOT NULL,
|
||||
cycle_id INTEGER NOT NULL,
|
||||
verdict TEXT NOT NULL, -- pass | partial | fail | cancelled
|
||||
verdict_derived INTEGER NOT NULL DEFAULT 0, -- 1: computed from checks, not stored by the orchestrator
|
||||
aggregated_at TIMESTAMP NOT NULL,
|
||||
expected_checks INTEGER, -- NULL when unknown
|
||||
recorded_checks INTEGER NOT NULL DEFAULT 0, -- checks stored when the verdict was made
|
||||
PRIMARY KEY (run_id, registry_id)
|
||||
);
|
||||
CREATE INDEX idx_run_results_registry ON run_results(registry_id);
|
||||
|
||||
CREATE TABLE subnets (
|
||||
cidr TEXT PRIMARY KEY,
|
||||
label TEXT NOT NULL DEFAULT ''
|
||||
);
|
||||
|
||||
ALTER TABLE ip_queue ADD COLUMN run_id INTEGER;
|
||||
ALTER TABLE checks ADD COLUMN run_id INTEGER;
|
||||
CREATE INDEX idx_checks_run ON checks(run_id, registry_id, cycle_id);
|
||||
|
||||
-- ---- existing data -------------------------------------------------------
|
||||
-- Ingress checks never stored the validator. The one holding the address in a
|
||||
-- cycle is named by its fip_associated event.
|
||||
UPDATE checks SET validator_id = COALESCE((
|
||||
SELECT json_extract(e.payload, '$.validator_id') FROM events e
|
||||
WHERE e.event_type = 'fip_associated' AND e.registry_id = checks.registry_id AND e.cycle_id = checks.cycle_id
|
||||
ORDER BY e.id DESC LIMIT 1), '')
|
||||
WHERE source LIKE 'inbound-site-%' AND validator_id = '';
|
||||
|
||||
-- Runs of existing data: cycles whose last checks end less than an hour apart
|
||||
-- belong to one run.
|
||||
CREATE TEMP TABLE _bf_cyc AS
|
||||
WITH c AS (
|
||||
SELECT registry_id, cycle_id, MAX(julianday(checked_at)) AS t, MIN(checked_at) AS first_at, MAX(checked_at) AS last_at
|
||||
FROM checks GROUP BY registry_id, cycle_id
|
||||
), o AS (
|
||||
SELECT *, CASE WHEN t - LAG(t) OVER (ORDER BY t) > 60.0 / 1440.0 THEN 1 ELSE 0 END AS brk FROM c
|
||||
)
|
||||
SELECT *, 1 + SUM(brk) OVER (ORDER BY t ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW) AS rid FROM o;
|
||||
|
||||
CREATE INDEX _bf_cyc_key ON _bf_cyc(registry_id, cycle_id);
|
||||
|
||||
INSERT INTO check_runs (id, kind, state, started_at, finalized_at)
|
||||
SELECT rid, 'manual', 'finalized', MIN(first_at), MAX(last_at) FROM _bf_cyc GROUP BY rid;
|
||||
|
||||
UPDATE checks SET run_id = (
|
||||
SELECT b.rid FROM _bf_cyc b WHERE b.registry_id = checks.registry_id AND b.cycle_id = checks.cycle_id);
|
||||
|
||||
UPDATE ip_queue SET run_id = (
|
||||
SELECT b.rid FROM _bf_cyc b WHERE b.registry_id = ip_queue.registry_id AND b.cycle_id = ip_queue.cycle_id);
|
||||
|
||||
-- One result per address and run: the latest cycle of the address in the run.
|
||||
-- The verdict is the orchestrator's when the queue row still holds that cycle,
|
||||
-- otherwise it is derived from the stored checks.
|
||||
INSERT INTO run_results (run_id, registry_id, ip_address, cycle_id, verdict, verdict_derived, aggregated_at, expected_checks, recorded_checks)
|
||||
SELECT l.rid, l.registry_id, r.ip_address, l.cycle_id,
|
||||
CASE WHEN q.id IS NOT NULL AND q.overall_result <> '' THEN q.overall_result
|
||||
WHEN s.ok = 0 THEN 'fail' WHEN s.ok = s.n THEN 'pass' ELSE 'partial' END,
|
||||
CASE WHEN q.id IS NOT NULL AND q.overall_result <> '' THEN 0 ELSE 1 END,
|
||||
COALESCE(CASE WHEN q.overall_result <> '' THEN q.aggregated_at END, s.last_at),
|
||||
(SELECT json_extract(e.payload, '$.checks') + json_extract(e.payload, '$.missing') FROM events e
|
||||
WHERE e.event_type = 'aggregated' AND e.registry_id = l.registry_id AND e.cycle_id = l.cycle_id
|
||||
ORDER BY e.id DESC LIMIT 1),
|
||||
COALESCE((SELECT json_extract(e.payload, '$.checks') FROM events e
|
||||
WHERE e.event_type = 'aggregated' AND e.registry_id = l.registry_id AND e.cycle_id = l.cycle_id
|
||||
ORDER BY e.id DESC LIMIT 1), s.n_before)
|
||||
FROM (SELECT rid, registry_id, MAX(cycle_id) AS cycle_id FROM _bf_cyc GROUP BY rid, registry_id) l
|
||||
JOIN ip_registry r ON r.id = l.registry_id
|
||||
JOIN (SELECT registry_id, cycle_id, COUNT(*) AS n, SUM(success) AS ok, MAX(checked_at) AS last_at,
|
||||
SUM(CASE WHEN after_verdict = 0 THEN 1 ELSE 0 END) AS n_before
|
||||
FROM checks GROUP BY registry_id, cycle_id) s ON s.registry_id = l.registry_id AND s.cycle_id = l.cycle_id
|
||||
LEFT JOIN ip_queue q ON q.registry_id = l.registry_id AND q.cycle_id = l.cycle_id;
|
||||
|
||||
DROP TABLE _bf_cyc;
|
||||
@@ -31,8 +31,9 @@ func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, `
|
||||
INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id,
|
||||
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at)
|
||||
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?
|
||||
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at, run_id)
|
||||
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, COALESCE(NULLIF(?, ''), q.owner_validator_id, ''),
|
||||
?, ?, ?, ?, ?, ?, ?, ?, ?, q.run_id
|
||||
FROM ip_queue q
|
||||
WHERE q.id=? AND q.attempt_number=? AND q.state NOT IN (?, ?, ?, ?)
|
||||
ON CONFLICT(registry_id, cycle_id, source, check_type, target) DO UPDATE SET
|
||||
@@ -41,7 +42,8 @@ func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
|
||||
latency_ms=excluded.latency_ms,
|
||||
detail=excluded.detail,
|
||||
checked_at=excluded.checked_at,
|
||||
recorded_at=excluded.recorded_at
|
||||
recorded_at=excluded.recorded_at,
|
||||
run_id=excluded.run_id
|
||||
`, c.IPAddress, c.ValidatorID, c.Source, c.CheckType, c.Target,
|
||||
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), now, now,
|
||||
c.IPID, c.AttemptNumber, IPAggregating, IPDone, IPFailed, IPOccupied)
|
||||
|
||||
@@ -21,6 +21,7 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
defer tx.Rollback()
|
||||
|
||||
now := timeToDB(Now())
|
||||
runID := int64(0)
|
||||
for i, addr := range addresses {
|
||||
var exists bool
|
||||
if err := tx.QueryRowContext(ctx, `SELECT EXISTS(SELECT 1 FROM ip_queue WHERE ip_address=?)`, addr).Scan(&exists); err != nil {
|
||||
@@ -37,11 +38,16 @@ func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if runID == 0 {
|
||||
if runID, err = openRunTx(ctx, tx, RunManual, now); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_address) DO NOTHING
|
||||
`, addr, i, IPQueued, registryID, cycle, now, now); err != nil {
|
||||
`, addr, i, IPQueued, registryID, cycle, runID, now, now); err != nil {
|
||||
return fmt.Errorf("seed %s: %w", addr, err)
|
||||
}
|
||||
}
|
||||
@@ -194,16 +200,36 @@ func (d *DB) SetAggregating(ctx context.Context, ipID int64) error {
|
||||
|
||||
// FinishIP records the aggregated result and marks the IP done or failed.
|
||||
func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error {
|
||||
return d.FinishIPExpected(ctx, ipID, result, -1)
|
||||
}
|
||||
|
||||
// FinishIPExpected is FinishIP that also records, in the address's run, how
|
||||
// many checks were expected at the verdict (expected < 0: unknown) and
|
||||
// finalizes the run if this was its last open address.
|
||||
func (d *DB) FinishIPExpected(ctx context.Context, ipID int64, result string, expected int) error {
|
||||
state := IPDone
|
||||
if result == ResultFail {
|
||||
state = IPFailed
|
||||
}
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, state, result, now, now, ipID)
|
||||
return err
|
||||
`, state, result, now, now, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := upsertRunResultTx(ctx, tx, ipID, result, expected, now); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// ReleaseFIP records that the floating IP has been disassociated and frees
|
||||
@@ -254,6 +280,9 @@ func (d *DB) MarkFIPOccupied(ctx context.Context, ipID int64, validatorID string
|
||||
return err
|
||||
}
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
@@ -304,6 +333,12 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
state=?, retry_count=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, nextState, retryCount, ResultFail, now, now, ipID)
|
||||
if err == nil {
|
||||
err = upsertRunResultTx(ctx, tx, ipID, ResultFail, -1, now)
|
||||
}
|
||||
if err == nil {
|
||||
err = finalizeRunsTx(ctx, tx, now)
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -337,6 +372,12 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
// sequence, so a batch's relative order is preserved and, critically,
|
||||
// resubmitting the same list later reproduces the same relative order.
|
||||
func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult, error) {
|
||||
return d.SubmitIPsAs(ctx, addresses, RunManual)
|
||||
}
|
||||
|
||||
// SubmitIPsAs is SubmitIPs that names the kind of run it opens when no run is
|
||||
// open (RunManual or RunAuto); an already open run is joined whatever the kind.
|
||||
func (d *DB) SubmitIPsAs(ctx context.Context, addresses []string, kind string) (SubmitIPsResult, error) {
|
||||
var result SubmitIPsResult
|
||||
if len(addresses) == 0 {
|
||||
return result, fmt.Errorf("addresses must not be empty: %w", ErrValidation)
|
||||
@@ -354,6 +395,17 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
|
||||
}
|
||||
|
||||
now := timeToDB(Now())
|
||||
runID := int64(0)
|
||||
ensureRun := func() (int64, error) {
|
||||
if runID == 0 {
|
||||
id, err := openRunTx(ctx, tx, kind, now)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
runID = id
|
||||
}
|
||||
return runID, nil
|
||||
}
|
||||
for i, addr := range addresses {
|
||||
seq := base + i
|
||||
|
||||
@@ -369,10 +421,14 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
|
||||
if cErr != nil {
|
||||
return result, cErr
|
||||
}
|
||||
rid, rErr := ensureRun()
|
||||
if rErr != nil {
|
||||
return result, rErr
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, now, now); err != nil {
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, registry_id, cycle_id, run_id, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
`, addr, seq, IPQueued, registryID, cycle, rid, now, now); err != nil {
|
||||
return result, fmt.Errorf("insert %s: %w", addr, err)
|
||||
}
|
||||
result.Added = append(result.Added, addr)
|
||||
@@ -389,14 +445,18 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
|
||||
if cErr != nil {
|
||||
return result, cErr
|
||||
}
|
||||
rid, rErr := ensureRun()
|
||||
if rErr != nil {
|
||||
return result, rErr
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
|
||||
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='',
|
||||
overall_result='', run_id=?,
|
||||
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
WHERE ip_address=?
|
||||
`, IPQueued, seq, cycle, now, addr); err != nil {
|
||||
`, IPQueued, seq, cycle, rid, now, addr); err != nil {
|
||||
return result, fmt.Errorf("requeue %s: %w", addr, err)
|
||||
}
|
||||
result.Requeued = append(result.Requeued, addr)
|
||||
@@ -431,7 +491,12 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
|
||||
// closed by the single-connection transactional UPDATE below).
|
||||
func (d *DB) CancelIP(ctx context.Context, ipID int64) error {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, `
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
res, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, overall_result=?, aggregated_at=?, owner_validator_id=NULL, fip_id='',
|
||||
lease_expires_at=NULL, updated_at=?
|
||||
@@ -443,7 +508,13 @@ func (d *DB) CancelIP(ctx context.Context, ipID int64) error {
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return fmt.Errorf("ip_id %d already finished: %w", ipID, ErrInvalidState)
|
||||
}
|
||||
return nil
|
||||
if err := upsertRunResultTx(ctx, tx, ipID, ResultCancelled, -1, now); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// DeleteIP permanently removes an ip_queue row, along with its full check
|
||||
@@ -463,6 +534,9 @@ func (d *DB) DeleteIP(ctx context.Context, ipID int64) error {
|
||||
if err := deleteIPTx(ctx, tx, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
@@ -497,6 +571,9 @@ func (d *DB) DeleteIPs(ctx context.Context, addresses []string) (DeleteIPsResult
|
||||
result.Deleted = append(result.Deleted, addr)
|
||||
}
|
||||
|
||||
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
|
||||
return result, err
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
return result, err
|
||||
}
|
||||
@@ -592,6 +669,9 @@ func (d *DB) ClearAllIPs(ctx context.Context) ([]string, error) {
|
||||
if _, err := tx.ExecContext(ctx, `DELETE FROM ip_queue`); err != nil {
|
||||
return nil, fmt.Errorf("delete ip_queue rows: %w", err)
|
||||
}
|
||||
if err := finalizeRunsTx(ctx, tx, now); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"net/netip"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -120,6 +121,34 @@ func (d *DB) ListRegistry(ctx context.Context) ([]RegistrySummary, error) {
|
||||
type RegistryFilter struct {
|
||||
Query string // substring of ip_address
|
||||
LastResult string // pass|partial|fail|cancelled — same meaning as RegistrySummary.LastResult
|
||||
RunID int64 // only addresses that have a result in this run
|
||||
Subnet string // only addresses inside this CIDR
|
||||
}
|
||||
|
||||
// subnetIDs returns the registry ids of the addresses inside prefix. SQLite
|
||||
// has no CIDR operators, so the registry's addresses are filtered here.
|
||||
func (d *DB) subnetIDs(ctx context.Context, cidr string) ([]any, error) {
|
||||
p, err := netip.ParsePrefix(cidr)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("subnet %q: %v: %w", cidr, err, ErrValidation)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx, `SELECT id, ip_address FROM ip_registry`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var ids []any
|
||||
for rows.Next() {
|
||||
var id int64
|
||||
var ip string
|
||||
if err := rows.Scan(&id, &ip); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if a, err := netip.ParseAddr(ip); err == nil && p.Contains(a) {
|
||||
ids = append(ids, id)
|
||||
}
|
||||
}
|
||||
return ids, rows.Err()
|
||||
}
|
||||
|
||||
// lastResultCond is the SQL form of fillRegistrySummary's LastResult rule,
|
||||
@@ -155,6 +184,22 @@ func (d *DB) ListRegistryPage(ctx context.Context, f RegistryFilter, limit, offs
|
||||
conds = append(conds, lastResultCond)
|
||||
args = append(args, f.LastResult, f.LastResult)
|
||||
}
|
||||
if f.RunID > 0 {
|
||||
conds = append(conds, "r.id IN (SELECT registry_id FROM run_results WHERE run_id = ?)")
|
||||
args = append(args, f.RunID)
|
||||
}
|
||||
if f.Subnet != "" {
|
||||
ids, err := d.subnetIDs(ctx, f.Subnet)
|
||||
if err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
if len(ids) == 0 {
|
||||
conds = append(conds, "0 = 1")
|
||||
} else {
|
||||
conds = append(conds, "r.id IN ("+strings.TrimSuffix(strings.Repeat("?,", len(ids)), ",")+")")
|
||||
args = append(args, ids...)
|
||||
}
|
||||
}
|
||||
from := ` FROM ip_registry r LEFT JOIN ip_queue q ON q.registry_id = r.id `
|
||||
where := ""
|
||||
if len(conds) > 0 {
|
||||
|
||||
@@ -0,0 +1,387 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"net/netip"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Run kinds and states. A run groups the cycles of one launch of checks — see
|
||||
// migrations/0011_check_runs.sql.
|
||||
const (
|
||||
RunManual = "manual"
|
||||
RunAuto = "auto"
|
||||
|
||||
RunOpen = "open"
|
||||
RunFinalized = "finalized"
|
||||
)
|
||||
|
||||
// CheckRun is one launch of checks. FinalizedAt is nil while it is open.
|
||||
type CheckRun struct {
|
||||
ID int64
|
||||
Kind string
|
||||
State string
|
||||
StartedAt time.Time
|
||||
FinalizedAt *time.Time
|
||||
}
|
||||
|
||||
// RunCounts is the verdict tally of a run.
|
||||
type RunCounts struct {
|
||||
Addresses int
|
||||
Pass int
|
||||
Partial int
|
||||
Fail int
|
||||
Cancelled int
|
||||
}
|
||||
|
||||
// RunSummary is a run with its verdict tally, for the run selector. Pending
|
||||
// is the number of its queue rows still being processed (open runs only).
|
||||
type RunSummary struct {
|
||||
CheckRun
|
||||
RunCounts
|
||||
Pending int
|
||||
Total int
|
||||
}
|
||||
|
||||
// openRunTx returns the id of the open run, creating one of the given kind if
|
||||
// none is open. Called when addresses enter the queue: while a run is open
|
||||
// everything submitted or re-checked joins it.
|
||||
func openRunTx(ctx context.Context, tx *sql.Tx, kind, now string) (int64, error) {
|
||||
var id int64
|
||||
err := tx.QueryRowContext(ctx, `SELECT id FROM check_runs WHERE state=? ORDER BY id DESC LIMIT 1`, RunOpen).Scan(&id)
|
||||
if err == nil {
|
||||
return id, nil
|
||||
}
|
||||
if err != sql.ErrNoRows {
|
||||
return 0, err
|
||||
}
|
||||
if kind == "" {
|
||||
kind = RunManual
|
||||
}
|
||||
res, err := tx.ExecContext(ctx, `INSERT INTO check_runs (kind, state, started_at) VALUES (?, ?, ?)`, kind, RunOpen, now)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("open run: %w", err)
|
||||
}
|
||||
return res.LastInsertId()
|
||||
}
|
||||
|
||||
// finalizeRunsTx finalizes every open run that has no queue row left in a
|
||||
// non-terminal state, and drops finalized runs that ended up with no results
|
||||
// at all (everything was deleted before any verdict). Called wherever a queue
|
||||
// row reaches a terminal state or disappears.
|
||||
func finalizeRunsTx(ctx context.Context, tx *sql.Tx, now string) error {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE check_runs SET state=?, finalized_at=COALESCE(
|
||||
(SELECT MAX(aggregated_at) FROM run_results WHERE run_id=check_runs.id), ?)
|
||||
WHERE state=? AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_queue q WHERE q.run_id=check_runs.id AND q.state NOT IN (?, ?, ?))
|
||||
`, RunFinalized, now, RunOpen, IPDone, IPFailed, IPOccupied); err != nil {
|
||||
return fmt.Errorf("finalize runs: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
DELETE FROM check_runs WHERE state=? AND NOT EXISTS (SELECT 1 FROM run_results r WHERE r.run_id=check_runs.id)
|
||||
`, RunFinalized); err != nil {
|
||||
return fmt.Errorf("drop empty runs: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// FinalizeRuns finalizes runs whose addresses are all done. The orchestrator
|
||||
// calls it every tick as a safety net; the terminal transitions already do it.
|
||||
func (d *DB) FinalizeRuns(ctx context.Context) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
if err := finalizeRunsTx(ctx, tx, timeToDB(Now())); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// upsertRunResultTx stores the verdict of the address's current cycle in its
|
||||
// run. A re-check inside an open run replaces the earlier result. expected < 0
|
||||
// means unknown. Rows without a run (queue rows that predate runs and never
|
||||
// got one) are skipped.
|
||||
func upsertRunResultTx(ctx context.Context, tx *sql.Tx, ipID int64, verdict string, expected int, now string) error {
|
||||
var exp sql.NullInt64
|
||||
if expected >= 0 {
|
||||
exp = sql.NullInt64{Int64: int64(expected), Valid: true}
|
||||
}
|
||||
_, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO run_results (run_id, registry_id, ip_address, cycle_id, verdict, aggregated_at, expected_checks, recorded_checks)
|
||||
SELECT q.run_id, q.registry_id, q.ip_address, q.cycle_id, ?, ?, ?,
|
||||
(SELECT COUNT(*) FROM checks c WHERE c.registry_id=q.registry_id AND c.cycle_id=q.cycle_id)
|
||||
FROM ip_queue q WHERE q.id=? AND q.run_id IS NOT NULL
|
||||
ON CONFLICT(run_id, registry_id) DO UPDATE SET
|
||||
cycle_id=excluded.cycle_id, verdict=excluded.verdict, verdict_derived=0,
|
||||
aggregated_at=excluded.aggregated_at, expected_checks=excluded.expected_checks,
|
||||
recorded_checks=excluded.recorded_checks
|
||||
`, verdict, now, exp, ipID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("record run result: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// adoptOrphanQueueRows gives every live queue row without a run (rows that
|
||||
// predate runs) the open run, creating one. Runs once after migrating.
|
||||
func (d *DB) adoptOrphanQueueRows(ctx context.Context) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
var n int
|
||||
if err := tx.QueryRowContext(ctx, `SELECT COUNT(*) FROM ip_queue WHERE run_id IS NULL AND state NOT IN (?, ?, ?)`,
|
||||
IPDone, IPFailed, IPOccupied).Scan(&n); err != nil {
|
||||
return err
|
||||
}
|
||||
if n == 0 {
|
||||
return nil
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
id, err := openRunTx(ctx, tx, RunManual, now)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET run_id=? WHERE run_id IS NULL AND state NOT IN (?, ?, ?)`,
|
||||
id, IPDone, IPFailed, IPOccupied); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// ListRuns returns every run, newest first, with its verdict tally.
|
||||
func (d *DB) ListRuns(ctx context.Context) ([]RunSummary, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT r.id, r.kind, r.state, r.started_at, r.finalized_at,
|
||||
COALESCE(SUM(CASE WHEN x.verdict IS NOT NULL THEN 1 ELSE 0 END), 0),
|
||||
COALESCE(SUM(CASE WHEN x.verdict='pass' THEN 1 ELSE 0 END), 0),
|
||||
COALESCE(SUM(CASE WHEN x.verdict='partial' THEN 1 ELSE 0 END), 0),
|
||||
COALESCE(SUM(CASE WHEN x.verdict='fail' THEN 1 ELSE 0 END), 0),
|
||||
COALESCE(SUM(CASE WHEN x.verdict='cancelled' THEN 1 ELSE 0 END), 0),
|
||||
(SELECT COUNT(*) FROM ip_queue q WHERE q.run_id=r.id),
|
||||
(SELECT COUNT(*) FROM ip_queue q WHERE q.run_id=r.id AND q.state NOT IN (?, ?, ?))
|
||||
FROM check_runs r LEFT JOIN run_results x ON x.run_id=r.id
|
||||
GROUP BY r.id ORDER BY r.id DESC
|
||||
`, IPDone, IPFailed, IPOccupied)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []RunSummary
|
||||
for rows.Next() {
|
||||
var s RunSummary
|
||||
var started string
|
||||
var finalized sql.NullString
|
||||
if err := rows.Scan(&s.ID, &s.Kind, &s.State, &started, &finalized,
|
||||
&s.Addresses, &s.Pass, &s.Partial, &s.Fail, &s.Cancelled, &s.Total, &s.Pending); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if s.StartedAt, err = dbToTime(started); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if s.FinalizedAt, err = nullStringToTimePtr(finalized); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, s)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// GetRun returns one run, or ErrNotFound.
|
||||
func (d *DB) GetRun(ctx context.Context, id int64) (*CheckRun, error) {
|
||||
var r CheckRun
|
||||
var started string
|
||||
var finalized sql.NullString
|
||||
err := d.QueryRowContext(ctx, `SELECT id, kind, state, started_at, finalized_at FROM check_runs WHERE id=?`, id).
|
||||
Scan(&r.ID, &r.Kind, &r.State, &started, &finalized)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, fmt.Errorf("run %d: %w", id, ErrNotFound)
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if r.StartedAt, err = dbToTime(started); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if r.FinalizedAt, err = nullStringToTimePtr(finalized); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &r, nil
|
||||
}
|
||||
|
||||
// RunResult is one address's result in a run.
|
||||
type RunResult struct {
|
||||
RegistryID int64
|
||||
IPAddress string
|
||||
CycleID int
|
||||
Verdict string
|
||||
Derived bool
|
||||
AggregatedAt time.Time
|
||||
ExpectedChecks int // -1 when unknown
|
||||
RecordedChecks int
|
||||
}
|
||||
|
||||
// ListRunResults returns the results of a run in address order of insertion.
|
||||
func (d *DB) ListRunResults(ctx context.Context, runID int64) ([]RunResult, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT registry_id, ip_address, cycle_id, verdict, verdict_derived, aggregated_at, expected_checks, recorded_checks
|
||||
FROM run_results WHERE run_id=? ORDER BY registry_id`, runID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []RunResult
|
||||
for rows.Next() {
|
||||
var r RunResult
|
||||
var agg string
|
||||
var exp sql.NullInt64
|
||||
if err := rows.Scan(&r.RegistryID, &r.IPAddress, &r.CycleID, &r.Verdict, &r.Derived, &agg, &exp, &r.RecordedChecks); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if r.AggregatedAt, err = dbToTime(agg); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
r.ExpectedChecks = -1
|
||||
if exp.Valid {
|
||||
r.ExpectedChecks = int(exp.Int64)
|
||||
}
|
||||
out = append(out, r)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// RunCheck is one stored check of a run's result cycle, with what the
|
||||
// analytics needs to place it.
|
||||
type RunCheck struct {
|
||||
RegistryID int64
|
||||
Source string
|
||||
CheckType string
|
||||
Target string
|
||||
Success bool
|
||||
ValidatorID string
|
||||
Detail string
|
||||
RecordedAt time.Time
|
||||
AfterVerdict bool
|
||||
}
|
||||
|
||||
// EachRunCheck calls fn for every check of the cycles that make up the run's
|
||||
// results (the latest cycle of each address in the run), in one pass over the
|
||||
// run_id index. fn must not call back into the DB (one connection).
|
||||
func (d *DB) EachRunCheck(ctx context.Context, runID int64, fn func(RunCheck)) error {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT c.registry_id, c.source, c.check_type, c.target, c.success, c.validator_id, c.detail,
|
||||
COALESCE(c.recorded_at, c.created_at), c.after_verdict
|
||||
FROM checks c JOIN run_results r ON r.run_id=c.run_id AND r.registry_id=c.registry_id AND r.cycle_id=c.cycle_id
|
||||
WHERE c.run_id=?`, runID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer rows.Close()
|
||||
for rows.Next() {
|
||||
var c RunCheck
|
||||
var rec string
|
||||
if err := rows.Scan(&c.RegistryID, &c.Source, &c.CheckType, &c.Target, &c.Success, &c.ValidatorID, &c.Detail, &rec, &c.AfterVerdict); err != nil {
|
||||
return err
|
||||
}
|
||||
if c.RecordedAt, err = dbToTime(rec); err != nil {
|
||||
return err
|
||||
}
|
||||
fn(c)
|
||||
}
|
||||
return rows.Err()
|
||||
}
|
||||
|
||||
// RunDataVersion changes whenever a check of the run is written, so a cache of
|
||||
// computed analytics can tell when it is stale.
|
||||
func (d *DB) RunDataVersion(ctx context.Context, runID int64) (string, error) {
|
||||
var maxID, n sql.NullInt64
|
||||
var rec sql.NullString
|
||||
if err := d.QueryRowContext(ctx, `SELECT MAX(id), COUNT(*), MAX(recorded_at) FROM checks WHERE run_id=?`, runID).Scan(&maxID, &n, &rec); err != nil {
|
||||
return "", err
|
||||
}
|
||||
var res sql.NullString
|
||||
if err := d.QueryRowContext(ctx, `SELECT MAX(aggregated_at) FROM run_results WHERE run_id=?`, runID).Scan(&res); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return fmt.Sprintf("%d/%d/%s/%s", maxID.Int64, n.Int64, rec.String, res.String), nil
|
||||
}
|
||||
|
||||
// Subnet is one entry of the administrator's subnet list.
|
||||
type Subnet struct {
|
||||
CIDR string
|
||||
Label string
|
||||
}
|
||||
|
||||
// ListSubnets returns the configured subnets, sorted by prefix.
|
||||
func (d *DB) ListSubnets(ctx context.Context) ([]Subnet, error) {
|
||||
rows, err := d.QueryContext(ctx, `SELECT cidr, label FROM subnets`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Subnet
|
||||
for rows.Next() {
|
||||
var s Subnet
|
||||
if err := rows.Scan(&s.CIDR, &s.Label); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, s)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool {
|
||||
a, _ := netip.ParsePrefix(out[i].CIDR)
|
||||
b, _ := netip.ParsePrefix(out[j].CIDR)
|
||||
if a.Addr() != b.Addr() {
|
||||
return a.Addr().Less(b.Addr())
|
||||
}
|
||||
return a.Bits() < b.Bits()
|
||||
})
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ReplaceSubnets replaces the whole subnet list. Every entry must be a valid
|
||||
// CIDR; entries are stored in canonical form (host bits cleared) and
|
||||
// duplicates collapse.
|
||||
func (d *DB) ReplaceSubnets(ctx context.Context, subnets []Subnet) error {
|
||||
canon := map[string]string{}
|
||||
for _, s := range subnets {
|
||||
p, err := netip.ParsePrefix(s.CIDR)
|
||||
if err != nil {
|
||||
return fmt.Errorf("subnet %q: %v: %w", s.CIDR, err, ErrValidation)
|
||||
}
|
||||
canon[p.Masked().String()] = s.Label
|
||||
}
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
if _, err := tx.ExecContext(ctx, `DELETE FROM subnets`); err != nil {
|
||||
return err
|
||||
}
|
||||
for cidr, label := range canon {
|
||||
if _, err := tx.ExecContext(ctx, `INSERT INTO subnets (cidr, label) VALUES (?, ?)`, cidr, label); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// CountRecheckedInRun returns how many addresses have more than one cycle of
|
||||
// checks inside the run (re-checked while the run was open).
|
||||
func (d *DB) CountRecheckedInRun(ctx context.Context, runID int64) (int, error) {
|
||||
var n int
|
||||
err := d.QueryRowContext(ctx, `
|
||||
SELECT COUNT(*) FROM (SELECT registry_id FROM checks WHERE run_id=? GROUP BY registry_id HAVING COUNT(DISTINCT cycle_id) > 1)
|
||||
`, runID).Scan(&n)
|
||||
return n, err
|
||||
}
|
||||
@@ -0,0 +1,389 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func submit(t *testing.T, d *DB, kind string, addrs ...string) {
|
||||
t.Helper()
|
||||
if _, err := d.SubmitIPsAs(context.Background(), addrs, kind); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func finish(t *testing.T, d *DB, addr, verdict string, expected int) {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
ip, err := d.GetIPByAddress(ctx, addr)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.FinishIPExpected(ctx, ip.ID, verdict, expected); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
func runs(t *testing.T, d *DB) []RunSummary {
|
||||
t.Helper()
|
||||
r, err := d.ListRuns(context.Background())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// An address entering an idle queue opens a run; everything submitted while it
|
||||
// is open joins it; it is finalized when the last address is done.
|
||||
func TestRunOpensJoinsAndFinalizes(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
submit(t, d, RunAuto, "1.1.1.1", "2.2.2.2")
|
||||
submit(t, d, RunManual, "3.3.3.3") // joins the open run, kind stays auto
|
||||
|
||||
rs := runs(t, d)
|
||||
if len(rs) != 1 || rs[0].State != RunOpen || rs[0].Kind != RunAuto || rs[0].Total != 3 || rs[0].Pending != 3 {
|
||||
t.Fatalf("expected one open auto run with 3 pending rows: %+v", rs)
|
||||
}
|
||||
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
c, _ := d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if a.ID == 0 || c.ID == 0 {
|
||||
t.Fatal("rows missing")
|
||||
}
|
||||
|
||||
finish(t, d, "1.1.1.1", ResultPass, 22)
|
||||
finish(t, d, "2.2.2.2", ResultPartial, 22)
|
||||
if rs = runs(t, d); rs[0].State != RunOpen || rs[0].Addresses != 2 || rs[0].Pending != 1 {
|
||||
t.Fatalf("one address still pending, the run stays open: %+v", rs[0])
|
||||
}
|
||||
finish(t, d, "3.3.3.3", ResultFail, -1)
|
||||
|
||||
rs = runs(t, d)
|
||||
if rs[0].State != RunFinalized || rs[0].FinalizedAt == nil || rs[0].Pass != 1 || rs[0].Partial != 1 || rs[0].Fail != 1 || rs[0].Addresses != 3 {
|
||||
t.Fatalf("expected a finalized run with the three verdicts: %+v", rs[0])
|
||||
}
|
||||
res, err := d.ListRunResults(ctx, rs[0].ID)
|
||||
if err != nil || len(res) != 3 {
|
||||
t.Fatalf("results: %v %v", res, err)
|
||||
}
|
||||
byIP := map[string]RunResult{}
|
||||
for _, r := range res {
|
||||
byIP[r.IPAddress] = r
|
||||
}
|
||||
if byIP["1.1.1.1"].ExpectedChecks != 22 || byIP["3.3.3.3"].ExpectedChecks != -1 || byIP["2.2.2.2"].Verdict != ResultPartial {
|
||||
t.Fatalf("results: %+v", byIP)
|
||||
}
|
||||
}
|
||||
|
||||
// A re-check after the run is finalized opens a new run and leaves the old one
|
||||
// as it was; the old cycle's checks keep their run.
|
||||
func TestRecheckAfterFinalizeOpensNewRun(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
ip := checkingIP(t, d, "1.1.1.1")
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", true)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
finish(t, d, "1.1.1.1", ResultPass, 1)
|
||||
first := runs(t, d)[0]
|
||||
if first.State != RunFinalized {
|
||||
t.Fatalf("expected finalized: %+v", first)
|
||||
}
|
||||
|
||||
submit(t, d, RunManual, "1.1.1.1") // re-check of a finished address
|
||||
rs := runs(t, d)
|
||||
if len(rs) != 2 || rs[0].State != RunOpen || rs[0].ID == first.ID {
|
||||
t.Fatalf("a re-check after the run ended must open a new run: %+v", rs)
|
||||
}
|
||||
ip, _ = d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if err := d.SetChecking(ctx, ip.ID, time.Minute); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", false)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
finish(t, d, "1.1.1.1", ResultFail, 1)
|
||||
|
||||
var n1, n2 int
|
||||
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id=?`, first.ID).Scan(&n1)
|
||||
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id=?`, rs[0].ID).Scan(&n2)
|
||||
if n1 != 1 || n2 != 1 {
|
||||
t.Fatalf("each run keeps its own cycle's check: %d %d", n1, n2)
|
||||
}
|
||||
old, _ := d.ListRunResults(ctx, first.ID)
|
||||
cur, _ := d.ListRunResults(ctx, rs[0].ID)
|
||||
if len(old) != 1 || old[0].Verdict != ResultPass || old[0].CycleID != 1 || len(cur) != 1 || cur[0].Verdict != ResultFail || cur[0].CycleID != 2 {
|
||||
t.Fatalf("results must not cross: old=%+v cur=%+v", old, cur)
|
||||
}
|
||||
// The checks of a run are the ones of its result cycle, nothing else.
|
||||
var got []RunCheck
|
||||
if err := d.EachRunCheck(ctx, first.ID, func(c RunCheck) { got = append(got, c) }); err != nil || len(got) != 1 || !got[0].Success {
|
||||
t.Fatalf("run 1 checks: %+v %v", got, err)
|
||||
}
|
||||
}
|
||||
|
||||
// A re-check while the run is still open joins it and replaces the address's
|
||||
// result, so a run holds one result per address.
|
||||
func TestRecheckInsideOpenRunReplacesResult(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
|
||||
finish(t, d, "1.1.1.1", ResultFail, 1)
|
||||
submit(t, d, RunManual, "1.1.1.1") // while 2.2.2.2 is still pending
|
||||
finish(t, d, "1.1.1.1", ResultPass, 1)
|
||||
finish(t, d, "2.2.2.2", ResultPass, 1)
|
||||
|
||||
rs := runs(t, d)
|
||||
if len(rs) != 1 || rs[0].Addresses != 2 || rs[0].Pass != 2 || rs[0].Fail != 0 {
|
||||
t.Fatalf("expected one run with the latest verdicts: %+v", rs)
|
||||
}
|
||||
res, _ := d.ListRunResults(ctx, rs[0].ID)
|
||||
for _, r := range res {
|
||||
if r.IPAddress == "1.1.1.1" && r.CycleID != 2 {
|
||||
t.Fatalf("the re-checked address must show its latest cycle: %+v", r)
|
||||
}
|
||||
}
|
||||
if n, err := d.CountRecheckedInRun(ctx, rs[0].ID); err != nil || n != 0 {
|
||||
t.Fatalf("no checks stored, so no re-check counted: %d %v", n, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunEndsWhenQueueIsClearedOrDeleted(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
|
||||
finish(t, d, "1.1.1.1", ResultPass, 1)
|
||||
if _, err := d.ClearAllIPs(ctx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
rs := runs(t, d)
|
||||
if len(rs) != 1 || rs[0].State != RunFinalized || rs[0].Addresses != 1 {
|
||||
t.Fatalf("clearing ends the run with what it has: %+v", rs)
|
||||
}
|
||||
|
||||
// The next submission is a new run, not a join of the ended one.
|
||||
submit(t, d, RunManual, "3.3.3.3")
|
||||
if rs = runs(t, d); len(rs) != 2 || rs[0].State != RunOpen {
|
||||
t.Fatalf("expected a new open run: %+v", rs)
|
||||
}
|
||||
// A run with no result at all disappears when its rows are deleted.
|
||||
ip, _ := d.GetIPByAddress(ctx, "3.3.3.3")
|
||||
if err := d.DeleteIP(ctx, ip.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if rs = runs(t, d); len(rs) != 1 {
|
||||
t.Fatalf("an empty run must be dropped: %+v", rs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCancelAndRetryFailureRecordResults(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
submit(t, d, RunManual, "1.1.1.1", "2.2.2.2")
|
||||
a, _ := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if err := d.CancelIP(ctx, a.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b, _ := d.GetIPByAddress(ctx, "2.2.2.2")
|
||||
if err := d.RequeueOrFail(ctx, b.ID, "", 0); err != nil { // retries exhausted at once
|
||||
t.Fatal(err)
|
||||
}
|
||||
rs := runs(t, d)
|
||||
if rs[0].State != RunFinalized || rs[0].Cancelled != 1 || rs[0].Fail != 1 {
|
||||
t.Fatalf("cancelled and failed addresses are results of the run: %+v", rs[0])
|
||||
}
|
||||
}
|
||||
|
||||
// Ingress checks name the validator that held the address.
|
||||
func TestIngressCheckTakesValidatorOfTheAddress(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.RegisterValidator(ctx, "validator-7", "host", "port", "v"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip := checkingIP(t, d, "1.1.1.1")
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET owner_validator_id='validator-7' WHERE id=?`, ip.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", true)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var v string
|
||||
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE ip_id=?`, ip.ID).Scan(&v); err != nil || v != "validator-7" {
|
||||
t.Fatalf("validator of an ingress check = %q err=%v", v, err)
|
||||
}
|
||||
// An explicit validator (egress checks) is kept.
|
||||
c := checkOf(ip, "https", true)
|
||||
c.Source, c.ValidatorID, c.Target = SourceEgress, "validator-9", "https://x"
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, c); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE ip_id=? AND source='egress'`, ip.ID).Scan(&v); err != nil || v != "validator-9" {
|
||||
t.Fatalf("egress validator = %q err=%v", v, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubnetsReplaceAndValidate(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if err := d.ReplaceSubnets(ctx, []Subnet{{CIDR: "10.1.2.3/24", Label: "a"}, {CIDR: "10.0.0.0/8"}, {CIDR: "10.1.2.0/24", Label: "dup"}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := d.ListSubnets(ctx)
|
||||
if err != nil || len(got) != 2 || got[0].CIDR != "10.0.0.0/8" || got[1].CIDR != "10.1.2.0/24" {
|
||||
t.Fatalf("subnets must be canonical, de-duplicated and sorted by prefix: %+v %v", got, err)
|
||||
}
|
||||
if err := d.ReplaceSubnets(ctx, []Subnet{{CIDR: "nonsense"}}); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation, got %v", err)
|
||||
}
|
||||
if got, _ := d.ListSubnets(ctx); len(got) != 2 {
|
||||
t.Fatalf("a rejected list must leave the old one: %+v", got)
|
||||
}
|
||||
if err := d.ReplaceSubnets(ctx, nil); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, _ := d.ListSubnets(ctx); len(got) != 0 {
|
||||
t.Fatalf("an empty list clears: %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistryFilterByRunAndSubnet(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
submit(t, d, RunManual, "10.0.0.1", "10.0.0.2", "10.0.1.1")
|
||||
for _, a := range []string{"10.0.0.1", "10.0.0.2", "10.0.1.1"} {
|
||||
finish(t, d, a, ResultPass, -1)
|
||||
}
|
||||
runID := runs(t, d)[0].ID
|
||||
submit(t, d, RunManual, "10.0.0.2") // second run holds only this address
|
||||
finish(t, d, "10.0.0.2", ResultFail, -1)
|
||||
secondID := runs(t, d)[0].ID
|
||||
|
||||
count := func(f RegistryFilter) int {
|
||||
t.Helper()
|
||||
_, total, err := d.ListRegistryPage(ctx, f, 50, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return total
|
||||
}
|
||||
if n := count(RegistryFilter{RunID: runID}); n != 3 {
|
||||
t.Errorf("run 1: %d", n)
|
||||
}
|
||||
if n := count(RegistryFilter{RunID: secondID}); n != 1 {
|
||||
t.Errorf("run 2: %d", n)
|
||||
}
|
||||
if n := count(RegistryFilter{Subnet: "10.0.0.0/24"}); n != 2 {
|
||||
t.Errorf("subnet /24: %d", n)
|
||||
}
|
||||
if n := count(RegistryFilter{RunID: secondID, Subnet: "10.0.1.0/24"}); n != 0 {
|
||||
t.Errorf("run 2 and the other subnet: %d", n)
|
||||
}
|
||||
if n := count(RegistryFilter{Subnet: "192.168.0.0/16"}); n != 0 {
|
||||
t.Errorf("subnet with no address: %d", n)
|
||||
}
|
||||
if _, _, err := d.ListRegistryPage(ctx, RegistryFilter{Subnet: "x"}, 10, 0); !errors.Is(err, ErrValidation) {
|
||||
t.Errorf("bad subnet: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Migration 0011 on a database of version 10: runs are cut at pauses of more
|
||||
// than an hour, results come from the queue row or the checks, ingress checks
|
||||
// get their validator from the fip_associated event, and live queue rows
|
||||
// without a run are adopted into an open run when the database is opened.
|
||||
func TestMigration0011BuildsRunsFromExistingData(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
path := filepath.Join(t.TempDir(), "old.db")
|
||||
raw, err := sql.Open("sqlite", path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw.SetMaxOpenConns(1)
|
||||
for _, m := range migrations {
|
||||
if m.version > 10 {
|
||||
break
|
||||
}
|
||||
if _, err := raw.ExecContext(ctx, m.sql); err != nil {
|
||||
t.Fatalf("migration %d: %v", m.version, err)
|
||||
}
|
||||
}
|
||||
raw.ExecContext(ctx, `PRAGMA user_version=10`)
|
||||
exec := func(q string, args ...any) {
|
||||
t.Helper()
|
||||
if _, err := raw.ExecContext(ctx, q, args...); err != nil {
|
||||
t.Fatalf("seed: %v\n%s", err, q)
|
||||
}
|
||||
}
|
||||
const ts = "2026-10-02T13:00:00Z"
|
||||
for i, ip := range []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"} {
|
||||
exec(`INSERT INTO ip_registry (id, ip_address, first_seen_at, last_seen_at, next_cycle, created_at, updated_at) VALUES (?, ?, ?, ?, 3, ?, ?)`, i+1, ip, ts, ts, ts, ts)
|
||||
}
|
||||
// 1.1.1.1 and 2.2.2.2 finish minutes apart (run 1); 1.1.1.1 is checked
|
||||
// again three hours later (run 2). 3.3.3.3 is still in the queue.
|
||||
exec(`INSERT INTO ip_queue (id, ip_address, sequence, state, overall_result, aggregated_at, registry_id, cycle_id, created_at, updated_at)
|
||||
VALUES (1, '1.1.1.1', 1, 'done', 'fail', '2026-10-02T16:00:30Z', 1, 2, ?, ?), (2, '2.2.2.2', 2, 'done', 'pass', '2026-10-02T13:05:30Z', 2, 1, ?, ?),
|
||||
(3, '3.3.3.3', 3, 'checking', '', NULL, 3, 1, ?, ?)`, ts, ts, ts, ts, ts, ts)
|
||||
chk := func(reg, cyc int, src, typ string, ok int, at string) {
|
||||
exec(`INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target, success, checked_at, created_at)
|
||||
VALUES (?, ?, ?, 'x', 1, '', ?, ?, 't', ?, ?, ?)`, reg, cyc, reg, src, typ, ok, at, at)
|
||||
}
|
||||
chk(1, 1, "egress", "https", 1, "2026-10-02T13:00:10Z")
|
||||
chk(1, 1, "inbound-site-1", "icmp", 1, "2026-10-02T13:00:20Z")
|
||||
chk(2, 1, "egress", "https", 1, "2026-10-02T13:05:00Z")
|
||||
chk(1, 2, "egress", "https", 0, "2026-10-02T16:00:10Z")
|
||||
exec(`INSERT INTO events (source_type, source_id, ip_id, event_type, payload, occurred_at, registry_id, cycle_id) VALUES
|
||||
('control-api', '', 1, 'fip_associated', '{"fip_id":"f","validator_id":"vkiplab-v5"}', ?, 1, 1),
|
||||
('control-api', '', 2, 'aggregated', '{"result":"pass","checks":1,"passed":1,"missing":1}', ?, 2, 1)`, ts, ts)
|
||||
raw.Close()
|
||||
|
||||
d, err := Open(ctx, path)
|
||||
if err != nil {
|
||||
t.Fatalf("open (runs migration 11): %v", err)
|
||||
}
|
||||
defer d.Close()
|
||||
|
||||
rs := runs(t, d)
|
||||
// run 1 and run 2 from the checks, plus the open run that adopted 3.3.3.3
|
||||
if len(rs) != 3 {
|
||||
t.Fatalf("expected 3 runs, got %+v", rs)
|
||||
}
|
||||
var first, second, open RunSummary
|
||||
for _, r := range rs {
|
||||
switch {
|
||||
case r.State == RunOpen:
|
||||
open = r
|
||||
case r.Addresses == 2:
|
||||
first = r
|
||||
default:
|
||||
second = r
|
||||
}
|
||||
}
|
||||
if first.Pass != 2 || first.Fail != 0 || second.Fail != 1 || second.Addresses != 1 || open.Total != 1 {
|
||||
t.Fatalf("runs: first=%+v second=%+v open=%+v", first, second, open)
|
||||
}
|
||||
res, _ := d.ListRunResults(ctx, first.ID)
|
||||
for _, r := range res {
|
||||
if r.IPAddress == "1.1.1.1" && (r.CycleID != 1 || r.Verdict != ResultPass || !r.Derived) {
|
||||
// the queue row holds the later cycle, so this one is derived from its checks
|
||||
t.Errorf("1.1.1.1 in the first run: %+v", r)
|
||||
}
|
||||
if r.IPAddress == "2.2.2.2" && (r.ExpectedChecks != 2 || r.RecordedChecks != 1 || r.Verdict != ResultPass || r.Derived) {
|
||||
t.Errorf("result from the aggregated event: %+v", r)
|
||||
}
|
||||
}
|
||||
res2, _ := d.ListRunResults(ctx, second.ID)
|
||||
if len(res2) != 1 || res2[0].CycleID != 2 || res2[0].Verdict != ResultFail || res2[0].Derived {
|
||||
t.Errorf("second run result: %+v", res2)
|
||||
}
|
||||
var v string
|
||||
if err := d.QueryRowContext(ctx, `SELECT validator_id FROM checks WHERE source='inbound-site-1'`).Scan(&v); err != nil || v != "vkiplab-v5" {
|
||||
t.Errorf("ingress validator = %q err=%v", v, err)
|
||||
}
|
||||
var unset int
|
||||
d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE run_id IS NULL`).Scan(&unset)
|
||||
if unset != 0 {
|
||||
t.Errorf("%d checks without a run", unset)
|
||||
}
|
||||
var ver int
|
||||
d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver)
|
||||
if ver != 11 {
|
||||
t.Errorf("user_version = %d", ver)
|
||||
}
|
||||
}
|
||||
@@ -240,7 +240,7 @@ func TestMigration0010MarksRowsAfterVerdict(t *testing.T) {
|
||||
t.Errorf("ssh: after_verdict=%d recorded=%s created=%s", a, rec, cr)
|
||||
}
|
||||
var ver int
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 10 {
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 11 {
|
||||
t.Errorf("user_version=%d err=%v", ver, err)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user