Show egress/ingress levels in the registry; freeze checks at the verdict
Registry: the "last result" column now also shows, per level (egress,
ingress), how many of the recorded checks of the latest cycle succeeded, split
by check family (tcp-22 and tcp-443 are both "tcp"). One grouped query per
chunk of addresses; new fields last_cycle_id, egress, ingress in
GET /admin/registry; the dashboard renders them under the verdict.
Verdict integrity (migration 0010):
- the prober is handed an address once per site and attempt, not on every
poll, so results are no longer overwritten by later probe rounds;
- UpsertCheckIfOpen refuses writes once the address is aggregating or has its
verdict, or for an older attempt; senders get {"ok":true,"ignored":N} and a
result_dropped event is recorded;
- the checking window counts from checking_started_at, not from assigned_at;
- checks.recorded_at (server clock) and checks.after_verdict (flag for rows
written after the verdict in existing data);
- the verdict rule is a pure function (computeVerdict) and the aggregated
event carries the egress/ingress check counts.
Rebuilt bin/control-api and bin/admin-dashboard to match. Plans and summaries
are in docs/changes; README, API, USAGE, DASHBOARD and DIAGRAMS are updated.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
db73409e8f
commit
864208238f
34 files changed
+1570
-72
No files matched your search
@@ -264,8 +264,47 @@ type registryItem struct {
|
||||
LastCheckedAt *time.Time `json:"last_checked_at"`
|
||||
InQueue bool `json:"in_queue"`
|
||||
CurrentState string `json:"current_state"`
|
||||
// LastCycleID is the cycle counted by Egress and Ingress (0 = no checks).
|
||||
LastCycleID int `json:"last_cycle_id"`
|
||||
Egress levelResult `json:"egress"`
|
||||
Ingress levelResult `json:"ingress"`
|
||||
}
|
||||
|
||||
// levelResult is "ok of total" recorded checks of one level (egress or
|
||||
// ingress) in the last cycle, split by check family — see httpapi's
|
||||
// levelResultDTO.
|
||||
type levelResult struct {
|
||||
Total int `json:"total"`
|
||||
OK int `json:"ok"`
|
||||
ByType []typeStat `json:"by_type"`
|
||||
}
|
||||
|
||||
type typeStat struct {
|
||||
Type string `json:"type"`
|
||||
Total int `json:"total"`
|
||||
OK int `json:"ok"`
|
||||
}
|
||||
|
||||
// statClass picks the colour of an "ok of total" figure: all checks passed,
|
||||
// none passed, some passed, or nothing recorded.
|
||||
func statClass(ok, total int) string {
|
||||
switch {
|
||||
case total == 0:
|
||||
return "none"
|
||||
case ok == total:
|
||||
return "ok"
|
||||
case ok == 0:
|
||||
return "fail"
|
||||
}
|
||||
return "part"
|
||||
}
|
||||
|
||||
// Class is the CSS modifier for the level's total.
|
||||
func (l levelResult) Class() string { return statClass(l.OK, l.Total) }
|
||||
|
||||
// Class is the CSS modifier for one check family.
|
||||
func (t typeStat) Class() string { return statClass(t.OK, t.Total) }
|
||||
|
||||
type registryHistoryResponse struct {
|
||||
Registry registryItem `json:"registry"`
|
||||
Checks []check `json:"checks"`
|
||||
|
||||
@@ -590,6 +590,38 @@ func TestRegistryPageAndDetail(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRegistryPageShowsEgressIngressLevels proves the list shows, under the
|
||||
// verdict, "ok из total" per level and per check family, coloured by outcome,
|
||||
// and omits the block for an address with no recorded cycle.
|
||||
func TestRegistryPageShowsEgressIngressLevels(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
now := time.Now()
|
||||
fake.registry["9.9.9.9"] = registryItem{
|
||||
IPAddress: "9.9.9.9", FirstSeenAt: now, LastSeenAt: now, TotalCycles: 1, LastResult: "partial",
|
||||
LastCycleID: 1,
|
||||
Egress: levelResult{Total: 5, OK: 5, ByType: []typeStat{{"https", 3, 3}, {"icmp", 2, 2}}},
|
||||
Ingress: levelResult{Total: 4, OK: 3, ByType: []typeStat{{"icmp", 1, 1}, {"tcp", 3, 2}}},
|
||||
}
|
||||
fake.registry["8.8.8.8"] = registryItem{IPAddress: "8.8.8.8", FirstSeenAt: now, LastSeenAt: now}
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
body := get(t, ts, "/registry")
|
||||
for _, want := range []string{
|
||||
"Egress", "Ingress",
|
||||
`stat stat-ok">5 из 5<`, `stat stat-part">3 из 4<`,
|
||||
`chip chip-ok">https 3 из 3<`, `chip chip-ok">icmp 2 из 2<`,
|
||||
`chip chip-part">tcp 2 из 3<`,
|
||||
"цикла 1",
|
||||
} {
|
||||
if !strings.Contains(body, want) {
|
||||
t.Fatalf("expected %q in registry page, got:\n%s", want, body)
|
||||
}
|
||||
}
|
||||
if strings.Count(body, `class="levels"`) != 1 {
|
||||
t.Fatalf("expected the levels block only for the address with checks, got:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRegistryFilterByQueryAndStatus proves the ?q=&status= params on
|
||||
// /registry narrow the list by address substring and by LastResult, and
|
||||
// that the filter form echoes the applied values back into its inputs.
|
||||
|
||||
@@ -2,6 +2,7 @@ package dashboard
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"html/template"
|
||||
"net/http"
|
||||
"strconv"
|
||||
@@ -115,6 +116,24 @@ var funcMap = template.FuncMap{
|
||||
"join": strings.Join,
|
||||
"joinInts": joinInts,
|
||||
"pluralAddr": pluralAddr,
|
||||
"dict": dict,
|
||||
}
|
||||
|
||||
// dict builds a map from alternating key/value arguments, so a sub-template
|
||||
// can be given several values: {{template "x" dict "Name" .A "L" .B}}.
|
||||
func dict(kv ...any) (map[string]any, error) {
|
||||
if len(kv)%2 != 0 {
|
||||
return nil, fmt.Errorf("dict: odd number of arguments")
|
||||
}
|
||||
m := make(map[string]any, len(kv)/2)
|
||||
for i := 0; i < len(kv); i += 2 {
|
||||
k, ok := kv[i].(string)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("dict: key %v is not a string", kv[i])
|
||||
}
|
||||
m[k] = kv[i+1]
|
||||
}
|
||||
return m, nil
|
||||
}
|
||||
|
||||
// pluralAddr returns the correctly declined Russian word for "address"
|
||||
|
||||
@@ -422,6 +422,22 @@ td.num { font-family: var(--font-mono); font-variant-numeric: tabular-nums; colo
|
||||
.pill-cancel { background: var(--cancel-soft); color: var(--cancel); }
|
||||
.pill-occupied { background: var(--occupied-soft); color: var(--occupied); }
|
||||
|
||||
/* ---------- registry: egress / ingress levels ---------- */
|
||||
.levels { margin-top: 6px; display: grid; gap: 4px; }
|
||||
.level { display: flex; flex-wrap: wrap; align-items: baseline; gap: 4px 8px; font-size: 12px; }
|
||||
.level-name { min-width: 52px; color: var(--neutral); font-weight: 600; }
|
||||
.level-types { display: inline-flex; flex-wrap: wrap; gap: 4px; }
|
||||
.stat { font-weight: 700; white-space: nowrap; }
|
||||
.stat-ok { color: var(--success); }
|
||||
.stat-part { color: var(--warning); }
|
||||
.stat-fail { color: var(--danger); }
|
||||
.stat-none { color: var(--neutral); }
|
||||
.chip { padding: 1px 6px; border-radius: var(--radius-xs); font-size: 11px; white-space: nowrap; }
|
||||
.chip-ok { background: var(--success-soft); color: var(--success); }
|
||||
.chip-part { background: var(--warning-soft); color: var(--warning); }
|
||||
.chip-fail { background: var(--danger-soft); color: var(--danger); }
|
||||
.chip-none { background: var(--neutral-soft); color: var(--neutral); }
|
||||
|
||||
/* ---------- alerts ---------- */
|
||||
.alert {
|
||||
display: flex; gap: 10px; align-items: flex-start;
|
||||
|
||||
@@ -66,6 +66,15 @@
|
||||
</div>
|
||||
{{end}}
|
||||
|
||||
{{define "registry_level"}}
|
||||
<div class="level">
|
||||
<span class="level-name">{{.Name}}</span>
|
||||
{{if .L.Total}}<span class="stat stat-{{.L.Class}}">{{.L.OK}} из {{.L.Total}}</span>
|
||||
<span class="level-types">{{range .L.ByType}}<span class="chip chip-{{.Class}}">{{.Type}} {{.OK}} из {{.Total}}</span>{{end}}</span>
|
||||
{{else}}<span class="stat stat-none">—</span>{{end}}
|
||||
</div>
|
||||
{{end}}
|
||||
|
||||
{{define "registry_table"}}
|
||||
{{if .Items}}
|
||||
<div class="panel">
|
||||
@@ -85,6 +94,10 @@
|
||||
{{else if eq .LastResult "fail"}}<span class="pill pill-danger">fail</span>
|
||||
{{else if eq .LastResult "cancelled"}}<span class="pill pill-cancel">cancelled</span>
|
||||
{{else}}<span class="pill pill-neutral">—</span>{{end}}
|
||||
{{if .LastCycleID}}<div class="levels" title="Считаются записанные проверки цикла {{.LastCycleID}}; вердикт учитывает ещё и недостающие результаты.">
|
||||
{{template "registry_level" dict "Name" "Egress" "L" .Egress}}
|
||||
{{template "registry_level" dict "Name" "Ingress" "L" .Ingress}}
|
||||
</div>{{end}}
|
||||
</td>
|
||||
<td data-label="Сейчас в очереди">
|
||||
{{if .InQueue}}<span class="pill pill-info">{{.CurrentState}}</span>{{else}}<span class="pill pill-neutral">нет</span>{{end}}
|
||||
|
||||
@@ -40,6 +40,9 @@ var autoCycleSchema string
|
||||
//go:embed migrations/0009_scale_indexes.sql
|
||||
var scaleIndexesSchema string
|
||||
|
||||
//go:embed migrations/0010_verdict_integrity.sql
|
||||
var verdictIntegritySchema string
|
||||
|
||||
// migrations is the ordered list of schema versions. Each entry's SQL is
|
||||
// applied, in order, for any version greater than the database's current
|
||||
// PRAGMA user_version — so a fresh database walks the whole list and an
|
||||
@@ -57,6 +60,7 @@ var migrations = []struct {
|
||||
{7, ipRegistrySchema},
|
||||
{8, autoCycleSchema},
|
||||
{9, scaleIndexesSchema},
|
||||
{10, verdictIntegritySchema},
|
||||
}
|
||||
|
||||
type DB struct {
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
-- Verdict integrity (see docs/changes/2026-10-03_17-21_verdict-no-late-results-plan.md).
|
||||
--
|
||||
-- ip_queue.checking_started_at: when the address entered the "checking" state.
|
||||
-- The aggregation window (checking_window_seconds) is counted from here, not
|
||||
-- from assigned_at, which also covers floating-IP association, the settle
|
||||
-- pause and the self-check. NULL on rows that predate this migration; the
|
||||
-- orchestrator falls back to assigned_at for them.
|
||||
--
|
||||
-- checks.recorded_at: when control-api last wrote the row, by its own clock.
|
||||
-- checked_at comes from the probing machine's clock, so it cannot be compared
|
||||
-- reliably with aggregated_at. Existing rows get created_at (their first
|
||||
-- write); a later overwrite is not recoverable.
|
||||
--
|
||||
-- checks.after_verdict: 1 when the row was written after the address's
|
||||
-- verdict (checked_at later than aggregated_at). Set here for existing data
|
||||
-- only; new writes after the verdict are rejected, so it stays 0 afterwards.
|
||||
|
||||
ALTER TABLE ip_queue ADD COLUMN checking_started_at TIMESTAMP;
|
||||
|
||||
ALTER TABLE checks ADD COLUMN recorded_at TIMESTAMP;
|
||||
UPDATE checks SET recorded_at = created_at;
|
||||
|
||||
ALTER TABLE checks ADD COLUMN after_verdict INTEGER NOT NULL DEFAULT 0;
|
||||
UPDATE checks SET after_verdict = 1
|
||||
WHERE EXISTS (
|
||||
SELECT 1 FROM ip_queue q
|
||||
WHERE q.registry_id = checks.registry_id
|
||||
AND q.cycle_id = checks.cycle_id
|
||||
AND q.aggregated_at IS NOT NULL
|
||||
AND julianday(checks.checked_at) > julianday(q.aggregated_at)
|
||||
);
|
||||
+45
-8
@@ -1,6 +1,9 @@
|
||||
package db
|
||||
|
||||
import "time"
|
||||
import (
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Validator and IP lifecycle states. Kept as typed string constants rather
|
||||
// than a Go enum type so they round-trip through SQLite TEXT columns and
|
||||
@@ -71,6 +74,37 @@ func InboundSource(siteIndex int) string {
|
||||
return "inbound-site-" + itoa(siteIndex)
|
||||
}
|
||||
|
||||
// Check levels: the two directions a check can run in, derived from
|
||||
// checks.source (there is no separate direction column).
|
||||
const (
|
||||
LevelEgress = "egress"
|
||||
LevelIngress = "ingress"
|
||||
)
|
||||
|
||||
// CheckLevel maps a checks.source value to its level: "egress" for the
|
||||
// validator's own outbound checks, "ingress" for any prober site
|
||||
// ("inbound-site-N"). Any other source yields "".
|
||||
func CheckLevel(source string) string {
|
||||
switch {
|
||||
case source == SourceEgress:
|
||||
return LevelEgress
|
||||
case strings.HasPrefix(source, "inbound-site-"):
|
||||
return LevelIngress
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// CheckFamily maps a checks.check_type value to its family for grouping:
|
||||
// the part before the first "-", so tcp-22 and tcp-443 are both "tcp" while
|
||||
// https, icmp, ssh and tls-443 ("tls") stay distinct. A check type added in
|
||||
// the future is grouped by its own name without code changes.
|
||||
func CheckFamily(checkType string) string {
|
||||
if i := strings.IndexByte(checkType, '-'); i > 0 {
|
||||
return checkType[:i]
|
||||
}
|
||||
return checkType
|
||||
}
|
||||
|
||||
func itoa(n int) string {
|
||||
if n == 0 {
|
||||
return "0"
|
||||
@@ -118,13 +152,16 @@ type IPQueueItem struct {
|
||||
EgressComplete bool
|
||||
OverallResult string
|
||||
AssignedAt *time.Time
|
||||
FIPAssociatedAt *time.Time
|
||||
AggregatedAt *time.Time
|
||||
FIPReleasedAt *time.Time
|
||||
RegistryID int64
|
||||
CycleID int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
// CheckingStartedAt is when the address entered the checking state; the
|
||||
// aggregation window counts from it. nil on rows that predate it.
|
||||
CheckingStartedAt *time.Time
|
||||
FIPAssociatedAt *time.Time
|
||||
AggregatedAt *time.Time
|
||||
FIPReleasedAt *time.Time
|
||||
RegistryID int64
|
||||
CycleID int
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
type Check struct {
|
||||
|
||||
@@ -15,20 +15,41 @@ import (
|
||||
// per-address cycle rather than the ip_queue row's attempt_number, so it
|
||||
// survives that row being deleted and the address later resubmitted.
|
||||
func (d *DB) UpsertCheck(ctx context.Context, c Check) error {
|
||||
_, err := d.ExecContext(ctx, `
|
||||
_, err := d.UpsertCheckIfOpen(ctx, c)
|
||||
return err
|
||||
}
|
||||
|
||||
// UpsertCheckIfOpen is UpsertCheck for the live write path. It stores the
|
||||
// check only while the address can still take results: the queue row exists,
|
||||
// c.AttemptNumber is its current attempt, and it has not reached the
|
||||
// aggregating state or a terminal one. Once the verdict is being computed the
|
||||
// checks are frozen, so the verdict always matches the stored rows and a late
|
||||
// probe of an already released floating IP cannot change them. It returns
|
||||
// false (and writes nothing) when the result was dropped. The state test and
|
||||
// the write are one statement, so they cannot interleave with SetAggregating.
|
||||
func (d *DB) UpsertCheckIfOpen(ctx context.Context, c Check) (bool, error) {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, `
|
||||
INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id,
|
||||
source, check_type, target, success, latency_ms, detail, checked_at, created_at)
|
||||
VALUES ((SELECT registry_id FROM ip_queue WHERE id=?), (SELECT cycle_id FROM ip_queue WHERE id=?),
|
||||
?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
source, check_type, target, success, latency_ms, detail, checked_at, created_at, recorded_at)
|
||||
SELECT q.registry_id, q.cycle_id, q.id, ?, q.attempt_number, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?
|
||||
FROM ip_queue q
|
||||
WHERE q.id=? AND q.attempt_number=? AND q.state NOT IN (?, ?, ?, ?)
|
||||
ON CONFLICT(registry_id, cycle_id, source, check_type, target) DO UPDATE SET
|
||||
validator_id=excluded.validator_id,
|
||||
success=excluded.success,
|
||||
latency_ms=excluded.latency_ms,
|
||||
detail=excluded.detail,
|
||||
checked_at=excluded.checked_at
|
||||
`, c.IPID, c.IPID, c.IPID, c.IPAddress, c.AttemptNumber, c.ValidatorID, c.Source, c.CheckType, c.Target,
|
||||
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), timeToDB(Now()))
|
||||
return err
|
||||
checked_at=excluded.checked_at,
|
||||
recorded_at=excluded.recorded_at
|
||||
`, c.IPAddress, c.ValidatorID, c.Source, c.CheckType, c.Target,
|
||||
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), now, now,
|
||||
c.IPID, c.AttemptNumber, IPAggregating, IPDone, IPFailed, IPOccupied)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
n, err := res.RowsAffected()
|
||||
return n > 0, err
|
||||
}
|
||||
|
||||
const checksSelect = `
|
||||
|
||||
@@ -180,9 +180,9 @@ func (d *DB) KnownAddresses(ctx context.Context, addresses []string) (map[string
|
||||
func (d *DB) SetChecking(ctx context.Context, ipID int64, leaseTTL time.Duration) error {
|
||||
now := Now()
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, lease_expires_at=?, updated_at=?
|
||||
UPDATE ip_queue SET state=?, lease_expires_at=?, checking_started_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPChecking, timeToDB(now.Add(leaseTTL)), timeToDB(now), ipID)
|
||||
`, IPChecking, timeToDB(now.Add(leaseTTL)), timeToDB(now), timeToDB(now), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -295,7 +295,7 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string,
|
||||
UPDATE ip_queue SET
|
||||
state=?, owner_validator_id=NULL, fip_id='', retry_count=?, attempt_number=attempt_number+1,
|
||||
cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='', assigned_at=NULL, fip_associated_at=NULL, updated_at=?
|
||||
overall_result='', assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, updated_at=?
|
||||
WHERE id=?
|
||||
`, nextState, retryCount, cycle, now, ipID)
|
||||
} else {
|
||||
@@ -394,7 +394,7 @@ func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult
|
||||
state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0,
|
||||
attempt_number=attempt_number+1, cycle_id=?, lease_expires_at=NULL, egress_complete=0,
|
||||
overall_result='',
|
||||
assigned_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
assigned_at=NULL, checking_started_at=NULL, fip_associated_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=?
|
||||
WHERE ip_address=?
|
||||
`, IPQueued, seq, cycle, now, addr); err != nil {
|
||||
return result, fmt.Errorf("requeue %s: %w", addr, err)
|
||||
@@ -835,6 +835,24 @@ func (d *DB) ListChecking(ctx context.Context) ([]IPQueueItem, error) {
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
// ListCheckingForSite returns the addresses in the checking state that the
|
||||
// given prober site still has to probe: those for which it has not yet
|
||||
// reported completion in the current attempt. Without this filter a prober
|
||||
// would re-probe every checking address on every poll until the verdict.
|
||||
func (d *DB) ListCheckingForSite(ctx context.Context, siteIndex int) ([]IPQueueItem, error) {
|
||||
rows, err := d.QueryContext(ctx, ipQueueSelect+`
|
||||
WHERE state=? AND NOT EXISTS (
|
||||
SELECT 1 FROM ip_site_checks s
|
||||
WHERE s.ip_id=ip_queue.id AND s.attempt_number=ip_queue.attempt_number
|
||||
AND s.site_idx=? AND s.complete=1)
|
||||
ORDER BY sequence`, IPChecking, siteIndex)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
// ListExpiredLeases returns non-terminal IPs whose lease has expired —
|
||||
// candidates for the lease sweep (crash recovery + stuck-validator reclaim).
|
||||
func (d *DB) ListExpiredLeases(ctx context.Context, now time.Time) ([]IPQueueItem, error) {
|
||||
@@ -851,7 +869,7 @@ func (d *DB) ListExpiredLeases(ctx context.Context, now time.Time) ([]IPQueueIte
|
||||
const ipQueueSelect = `
|
||||
SELECT id, ip_address, sequence, state, owner_validator_id, fip_id, attempt_number, retry_count,
|
||||
lease_expires_at, egress_complete, overall_result,
|
||||
assigned_at, fip_associated_at, aggregated_at, fip_released_at,
|
||||
assigned_at, checking_started_at, fip_associated_at, aggregated_at, fip_released_at,
|
||||
registry_id, cycle_id, created_at, updated_at
|
||||
FROM ip_queue
|
||||
`
|
||||
@@ -871,14 +889,14 @@ func scanIPQueueItems(rows *sql.Rows) ([]IPQueueItem, error) {
|
||||
func scanIPQueueItem(row rowScanner) (*IPQueueItem, error) {
|
||||
var item IPQueueItem
|
||||
var owner sql.NullString
|
||||
var leaseExpires, assignedAt, fipAssociatedAt, aggregatedAt, fipReleasedAt sql.NullString
|
||||
var leaseExpires, assignedAt, checkingStartedAt, fipAssociatedAt, aggregatedAt, fipReleasedAt sql.NullString
|
||||
var registryID sql.NullInt64
|
||||
var createdAt, updatedAt string
|
||||
if err := row.Scan(
|
||||
&item.ID, &item.IPAddress, &item.Sequence, &item.State, &owner, &item.FIPID,
|
||||
&item.AttemptNumber, &item.RetryCount, &leaseExpires,
|
||||
&item.EgressComplete,
|
||||
&item.OverallResult, &assignedAt, &fipAssociatedAt, &aggregatedAt, &fipReleasedAt,
|
||||
&item.OverallResult, &assignedAt, &checkingStartedAt, &fipAssociatedAt, &aggregatedAt, &fipReleasedAt,
|
||||
®istryID, &item.CycleID, &createdAt, &updatedAt,
|
||||
); err != nil {
|
||||
return nil, err
|
||||
@@ -896,6 +914,9 @@ func scanIPQueueItem(row rowScanner) (*IPQueueItem, error) {
|
||||
if item.AssignedAt, err = nullStringToTimePtr(assignedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.CheckingStartedAt, err = nullStringToTimePtr(checkingStartedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.FIPAssociatedAt, err = nullStringToTimePtr(fipAssociatedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
@@ -56,6 +57,28 @@ type RegistrySummary struct {
|
||||
LastCheckedAt *time.Time
|
||||
InQueue bool
|
||||
CurrentState string
|
||||
|
||||
// LastCycleID is the newest cycle_id with recorded checks (0 if none).
|
||||
// Egress and Ingress count that cycle's recorded checks per level.
|
||||
LastCycleID int
|
||||
Egress LevelResult
|
||||
Ingress LevelResult
|
||||
}
|
||||
|
||||
// TypeStat counts the recorded checks of one check family (see CheckFamily)
|
||||
// within a level: Total checks, OK of them successful.
|
||||
type TypeStat struct {
|
||||
Type string
|
||||
Total int
|
||||
OK int
|
||||
}
|
||||
|
||||
// LevelResult is the "OK of Total" rollup of one level (egress or ingress) of
|
||||
// a cycle, with the same counts split by check family, sorted by Type.
|
||||
type LevelResult struct {
|
||||
Total int
|
||||
OK int
|
||||
ByType []TypeStat
|
||||
}
|
||||
|
||||
// ListRegistry returns every address ever submitted, newest first-seen
|
||||
@@ -87,6 +110,9 @@ func (d *DB) ListRegistry(ctx context.Context) ([]RegistrySummary, error) {
|
||||
}
|
||||
out[i] = s
|
||||
}
|
||||
if err := d.fillRegistryLevels(ctx, out); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
@@ -165,6 +191,9 @@ func (d *DB) ListRegistryPage(ctx context.Context, f RegistryFilter, limit, offs
|
||||
}
|
||||
out[i] = s
|
||||
}
|
||||
if err := d.fillRegistryLevels(ctx, out); err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
return out, total, nil
|
||||
}
|
||||
|
||||
@@ -186,6 +215,11 @@ func (d *DB) GetRegistryByAddress(ctx context.Context, address string) (*Registr
|
||||
if err := d.fillRegistrySummary(ctx, s); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
one := []RegistrySummary{*s}
|
||||
if err := d.fillRegistryLevels(ctx, one); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
*s = one[0]
|
||||
return s, nil
|
||||
}
|
||||
|
||||
@@ -253,6 +287,98 @@ func (d *DB) fillRegistrySummary(ctx context.Context, s *RegistrySummary) error
|
||||
return nil
|
||||
}
|
||||
|
||||
// registryLevelsChunk bounds the number of registry ids per query, well under
|
||||
// SQLite's bound-variable limit.
|
||||
const registryLevelsChunk = 500
|
||||
|
||||
// registryLevelsQuery is the grouped query of fillRegistryLevels for n
|
||||
// registry ids: per address, the counts of its newest cycle by source and
|
||||
// check type.
|
||||
func registryLevelsQuery(n int) string {
|
||||
return `
|
||||
SELECT c.registry_id, m.cid, c.source, c.check_type, COUNT(*), COALESCE(SUM(c.success), 0)
|
||||
FROM checks c
|
||||
JOIN (SELECT registry_id, MAX(cycle_id) AS cid FROM checks
|
||||
WHERE registry_id IN (` + strings.TrimSuffix(strings.Repeat("?,", n), ",") + `)
|
||||
GROUP BY registry_id) m
|
||||
ON m.registry_id = c.registry_id AND m.cid = c.cycle_id
|
||||
GROUP BY c.registry_id, m.cid, c.source, c.check_type
|
||||
`
|
||||
}
|
||||
|
||||
// fillRegistryLevels sets LastCycleID, Egress and Ingress on every summary in
|
||||
// sums: the recorded checks of each address's newest cycle (the same cycle
|
||||
// whose time is LastCheckedAt), counted per level and per check family. It
|
||||
// runs one grouped query per chunk of addresses, not one per address, over
|
||||
// idx_checks_registry_cycle. Checks whose source is neither egress nor an
|
||||
// inbound site are not counted. The counts follow the recorded rows only, so
|
||||
// they can differ from LastResult, which also treats missing results as
|
||||
// failures.
|
||||
func (d *DB) fillRegistryLevels(ctx context.Context, sums []RegistrySummary) error {
|
||||
pos := make(map[int64]int, len(sums))
|
||||
for i := range sums {
|
||||
pos[sums[i].ID] = i
|
||||
}
|
||||
type key struct {
|
||||
id int64
|
||||
level, family string
|
||||
}
|
||||
for start := 0; start < len(sums); start += registryLevelsChunk {
|
||||
end := min(start+registryLevelsChunk, len(sums))
|
||||
args := make([]any, 0, end-start)
|
||||
for _, s := range sums[start:end] {
|
||||
args = append(args, s.ID)
|
||||
}
|
||||
rows, err := d.QueryContext(ctx, registryLevelsQuery(len(args)), args...)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
stats := map[key]*TypeStat{}
|
||||
for rows.Next() {
|
||||
var id int64
|
||||
var cid, total, ok int
|
||||
var source, checkType string
|
||||
if err := rows.Scan(&id, &cid, &source, &checkType, &total, &ok); err != nil {
|
||||
rows.Close()
|
||||
return err
|
||||
}
|
||||
sums[pos[id]].LastCycleID = cid
|
||||
level := CheckLevel(source)
|
||||
if level == "" {
|
||||
continue
|
||||
}
|
||||
k := key{id, level, CheckFamily(checkType)}
|
||||
st := stats[k]
|
||||
if st == nil {
|
||||
st = &TypeStat{Type: k.family}
|
||||
stats[k] = st
|
||||
}
|
||||
st.Total += total
|
||||
st.OK += ok
|
||||
}
|
||||
err = rows.Err()
|
||||
rows.Close()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for k, st := range stats {
|
||||
lr := &sums[pos[k.id]].Egress
|
||||
if k.level == LevelIngress {
|
||||
lr = &sums[pos[k.id]].Ingress
|
||||
}
|
||||
lr.Total += st.Total
|
||||
lr.OK += st.OK
|
||||
lr.ByType = append(lr.ByType, *st)
|
||||
}
|
||||
}
|
||||
for i := range sums {
|
||||
for _, lr := range []*LevelResult{&sums[i].Egress, &sums[i].Ingress} {
|
||||
sort.Slice(lr.ByType, func(a, b int) bool { return lr.ByType[a].Type < lr.ByType[b].Type })
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// lastCycleResultFromChecks classifies the most recent cycle recorded for
|
||||
// registryID directly from its checks rows: pass if every recorded check
|
||||
// succeeded, fail if every one failed, partial on a mix. Returns "" if no
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCheckLevelAndFamily(t *testing.T) {
|
||||
for source, want := range map[string]string{
|
||||
"egress": LevelEgress, "inbound-site-1": LevelIngress, "inbound-site-12": LevelIngress,
|
||||
"": "", "other": "",
|
||||
} {
|
||||
if got := CheckLevel(source); got != want {
|
||||
t.Errorf("CheckLevel(%q) = %q, want %q", source, got, want)
|
||||
}
|
||||
}
|
||||
for ct, want := range map[string]string{
|
||||
"https": "https", "icmp": "icmp", "ssh": "ssh",
|
||||
"tcp-22": "tcp", "tcp-443": "tcp", "tls-443": "tls", "dns": "dns", "-x": "-x",
|
||||
} {
|
||||
if got := CheckFamily(ct); got != want {
|
||||
t.Errorf("CheckFamily(%q) = %q, want %q", ct, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// addCheck records one check for the address's current queue row.
|
||||
func addCheck(t *testing.T, d *DB, addr, source, checkType, target string, success bool) {
|
||||
t.Helper()
|
||||
ctx := t.Context()
|
||||
ip, err := d.GetIPByAddress(ctx, addr)
|
||||
if err != nil {
|
||||
t.Fatalf("get %s: %v", addr, err)
|
||||
}
|
||||
if err := d.UpsertCheck(ctx, Check{
|
||||
IPID: ip.ID, IPAddress: addr, AttemptNumber: ip.AttemptNumber,
|
||||
Source: source, CheckType: checkType, Target: target,
|
||||
Success: success, CheckedAt: Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("upsert check: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistryLevelsGroupByTypeAndLevel(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.2.3.4"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
a := "1.2.3.4"
|
||||
// Egress: https 2 of 3, icmp 1 of 1.
|
||||
addCheck(t, d, a, SourceEgress, "https", "https://a.test", true)
|
||||
addCheck(t, d, a, SourceEgress, "https", "https://b.test", true)
|
||||
addCheck(t, d, a, SourceEgress, "https", "https://c.test", false)
|
||||
addCheck(t, d, a, SourceEgress, "icmp", "a.test", true)
|
||||
// Ingress from two sites: tcp-22 and tcp-443 are one family; tls, ssh,
|
||||
// icmp and a type unknown today ("dns") are listed on their own.
|
||||
for site := 1; site <= 2; site++ {
|
||||
src := InboundSource(site)
|
||||
addCheck(t, d, a, src, "tcp-22", a, true)
|
||||
addCheck(t, d, a, src, "tcp-443", a, site == 1)
|
||||
addCheck(t, d, a, src, "tls-443", a, true)
|
||||
addCheck(t, d, a, src, "ssh", a, false)
|
||||
addCheck(t, d, a, src, "icmp", a, true)
|
||||
addCheck(t, d, a, src, "dns", a, true)
|
||||
}
|
||||
// A source that is neither egress nor an inbound site is not counted.
|
||||
addCheck(t, d, a, "manual", "https", "x", true)
|
||||
|
||||
s, err := d.GetRegistryByAddress(ctx, a)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
wantEgress := LevelResult{Total: 4, OK: 3, ByType: []TypeStat{
|
||||
{"https", 3, 2}, {"icmp", 1, 1},
|
||||
}}
|
||||
wantIngress := LevelResult{Total: 12, OK: 9, ByType: []TypeStat{
|
||||
{"dns", 2, 2}, {"icmp", 2, 2}, {"ssh", 2, 0}, {"tcp", 4, 3}, {"tls", 2, 2},
|
||||
}}
|
||||
if !reflect.DeepEqual(s.Egress, wantEgress) {
|
||||
t.Errorf("egress = %+v, want %+v", s.Egress, wantEgress)
|
||||
}
|
||||
if !reflect.DeepEqual(s.Ingress, wantIngress) {
|
||||
t.Errorf("ingress = %+v, want %+v", s.Ingress, wantIngress)
|
||||
}
|
||||
if s.LastCycleID != 1 {
|
||||
t.Errorf("LastCycleID = %d, want 1", s.LastCycleID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegistryLevelsNoChecksAreZero(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.2.3.4"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s, err := d.GetRegistryByAddress(ctx, "1.2.3.4")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s.LastCycleID != 0 || s.Egress.Total != 0 || s.Ingress.Total != 0 || len(s.Egress.ByType) != 0 {
|
||||
t.Fatalf("expected empty levels, got %+v", s)
|
||||
}
|
||||
}
|
||||
|
||||
// The counts follow the newest cycle, also after the queue row is deleted,
|
||||
// and one grouped query serves several addresses without mixing them up.
|
||||
func TestRegistryLevelsLatestCycleAndManyAddresses(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.1.1.1", "2.2.2.2"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
addCheck(t, d, "1.1.1.1", SourceEgress, "https", "t1", false)
|
||||
addCheck(t, d, "2.2.2.2", SourceEgress, "https", "t1", true)
|
||||
addCheck(t, d, "2.2.2.2", InboundSource(1), "tcp-22", "2.2.2.2", true)
|
||||
|
||||
// New cycle for 1.1.1.1: delete and submit again.
|
||||
ip, err := d.GetIPByAddress(ctx, "1.1.1.1")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.DeleteIP(ctx, ip.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := d.SubmitIPs(ctx, []string{"1.1.1.1"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
addCheck(t, d, "1.1.1.1", SourceEgress, "icmp", "t2", true)
|
||||
addCheck(t, d, "1.1.1.1", SourceEgress, "https", "t2", true)
|
||||
|
||||
all, err := d.ListRegistry(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got := map[string]RegistrySummary{}
|
||||
for _, s := range all {
|
||||
got[s.IPAddress] = s
|
||||
}
|
||||
a := got["1.1.1.1"]
|
||||
if a.LastCycleID != 2 || a.Egress.Total != 2 || a.Egress.OK != 2 || a.Ingress.Total != 0 {
|
||||
t.Errorf("1.1.1.1: %+v", a)
|
||||
}
|
||||
b := got["2.2.2.2"]
|
||||
if b.LastCycleID != 1 || b.Egress.Total != 1 || b.Ingress.Total != 1 || b.Ingress.OK != 1 {
|
||||
t.Errorf("2.2.2.2: %+v", b)
|
||||
}
|
||||
|
||||
// Page and single lookups agree with the full list.
|
||||
page, _, err := d.ListRegistryPage(ctx, RegistryFilter{}, 10, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, s := range page {
|
||||
if !reflect.DeepEqual(s.Egress, got[s.IPAddress].Egress) || !reflect.DeepEqual(s.Ingress, got[s.IPAddress].Ingress) {
|
||||
t.Errorf("page differs from list for %s", s.IPAddress)
|
||||
}
|
||||
}
|
||||
|
||||
// Delete the queue row of 2.2.2.2: the counts stay (history outlives it).
|
||||
ip2, _ := d.GetIPByAddress(ctx, "2.2.2.2")
|
||||
if err := d.DeleteIP(ctx, ip2.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s, err := d.GetRegistryByAddress(ctx, "2.2.2.2")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s.Egress.Total != 1 || s.Ingress.Total != 1 {
|
||||
t.Errorf("after delete: %+v", s)
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
@@ -350,6 +351,33 @@ func TestMigration0009Indexes(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRegistryLevelsQueryUsesIndex guards against a full scan of checks: both
|
||||
// the per-address MAX(cycle_id) and the join back must go through
|
||||
// idx_checks_registry_cycle.
|
||||
func TestRegistryLevelsQueryUsesIndex(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
rows, err := d.QueryContext(ctx, "EXPLAIN QUERY PLAN "+registryLevelsQuery(3), 1, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var plan string
|
||||
for rows.Next() {
|
||||
var id, parent, unused int
|
||||
var detail string
|
||||
if err := rows.Scan(&id, &parent, &unused, &detail); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
plan += detail + "\n"
|
||||
if strings.HasPrefix(detail, "SCAN") && strings.Contains(detail, "checks") {
|
||||
t.Errorf("full scan of checks in plan:\n%s", plan)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(plan, "idx_checks_registry_cycle") {
|
||||
t.Errorf("expected idx_checks_registry_cycle in plan:\n%s", plan)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScaleSmoke6440 pushes a realistic project size through the hot paths
|
||||
// with a loose time bound: the point is the absence of O(n^2) / N+1 work, not
|
||||
// a benchmark.
|
||||
@@ -369,11 +397,36 @@ func TestScaleSmoke6440(t *testing.T) {
|
||||
}
|
||||
submitDur := time.Since(start)
|
||||
|
||||
// A realistic check set for the addresses of the registry page below:
|
||||
// 12 egress and 18 ingress checks each.
|
||||
for _, addr := range addrs[3000:3100] {
|
||||
ip, err := d.GetIPByAddress(ctx, addr)
|
||||
if err != nil {
|
||||
t.Fatalf("get %s: %v", addr, err)
|
||||
}
|
||||
for i := 0; i < 30; i++ {
|
||||
source, checkType := SourceEgress, []string{"https", "icmp"}[i%2]
|
||||
if i >= 12 {
|
||||
source, checkType = InboundSource(i%3+1), fmt.Sprintf("tcp-%d", 20+i)
|
||||
}
|
||||
if err := d.UpsertCheck(ctx, Check{
|
||||
IPID: ip.ID, IPAddress: addr, AttemptNumber: ip.AttemptNumber,
|
||||
Source: source, CheckType: checkType, Target: fmt.Sprintf("t%d", i),
|
||||
Success: i%5 != 0, CheckedAt: Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("upsert check: %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
start = time.Now()
|
||||
page, total, err := d.ListRegistryPage(ctx, RegistryFilter{}, 100, 3000)
|
||||
if err != nil || total != 6440 || len(page) != 100 {
|
||||
t.Fatalf("registry page: total=%d len=%d err=%v", total, len(page), err)
|
||||
}
|
||||
if e, i := page[0].Egress.Total, page[0].Ingress.Total; e != 12 || i != 18 {
|
||||
t.Fatalf("levels of the first page row: egress=%d ingress=%d", e, i)
|
||||
}
|
||||
if _, total, err = d.ListRegistryPage(ctx, RegistryFilter{LastResult: ResultPass}, 100, 0); err != nil || total != 0 {
|
||||
t.Fatalf("registry last_result filter: total=%d err=%v", total, err)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// checkingIP submits one address and moves it to the checking state.
|
||||
func checkingIP(t *testing.T, d *DB, addr string) *IPQueueItem {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
if _, err := d.SubmitIPs(ctx, []string{addr}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip, err := d.GetIPByAddress(ctx, addr)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SetChecking(ctx, ip.ID, time.Minute); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip, err = d.GetIP(ctx, ip.ID)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return ip
|
||||
}
|
||||
|
||||
func checkOf(ip *IPQueueItem, ct string, ok bool) Check {
|
||||
return Check{IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: InboundSource(1), CheckType: ct, Target: ip.IPAddress, Success: ok, CheckedAt: Now()}
|
||||
}
|
||||
|
||||
func storedSuccess(t *testing.T, d *DB, ip *IPQueueItem, ct string) (success bool, found bool) {
|
||||
t.Helper()
|
||||
err := d.QueryRowContext(context.Background(),
|
||||
`SELECT success FROM checks WHERE ip_id=? AND check_type=?`, ip.ID, ct).Scan(&success)
|
||||
if err == sql.ErrNoRows {
|
||||
return false, false
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return success, true
|
||||
}
|
||||
|
||||
// Results are accepted while the address is being checked (also repeated ones,
|
||||
// idempotently) and refused from the moment the verdict is being computed.
|
||||
func TestUpsertCheckIfOpenFreezesAtVerdict(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
ip := checkingIP(t, d, "1.2.3.4")
|
||||
|
||||
for i := 0; i < 2; i++ {
|
||||
if ok, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "tcp-22", true)); err != nil || !ok {
|
||||
t.Fatalf("write %d while checking: ok=%v err=%v", i, ok, err)
|
||||
}
|
||||
}
|
||||
var n int
|
||||
if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM checks WHERE ip_id=?`, ip.ID).Scan(&n); err != nil || n != 1 {
|
||||
t.Fatalf("expected one row after repeated write, got %d err=%v", n, err)
|
||||
}
|
||||
|
||||
// An older attempt's result does not touch the current attempt.
|
||||
old := checkOf(ip, "icmp", true)
|
||||
old.AttemptNumber = ip.AttemptNumber - 1
|
||||
if ok, err := d.UpsertCheckIfOpen(ctx, old); err != nil || ok {
|
||||
t.Fatalf("stale attempt must be dropped: ok=%v err=%v", ok, err)
|
||||
}
|
||||
if _, found := storedSuccess(t, d, ip, "icmp"); found {
|
||||
t.Fatal("stale attempt wrote a row")
|
||||
}
|
||||
|
||||
if err := d.SetAggregating(ctx, ip.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Neither a new check nor a change of an existing one gets through.
|
||||
if ok, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "ssh", false)); err != nil || ok {
|
||||
t.Fatalf("new check while aggregating must be dropped: ok=%v err=%v", ok, err)
|
||||
}
|
||||
if ok, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "tcp-22", false)); err != nil || ok {
|
||||
t.Fatalf("overwrite while aggregating must be dropped: ok=%v err=%v", ok, err)
|
||||
}
|
||||
if err := d.FinishIP(ctx, ip.ID, ResultPass); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if ok, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "tcp-22", false)); err != nil || ok {
|
||||
t.Fatalf("overwrite after the verdict must be dropped: ok=%v err=%v", ok, err)
|
||||
}
|
||||
|
||||
if s, found := storedSuccess(t, d, ip, "tcp-22"); !found || !s {
|
||||
t.Fatalf("stored tcp-22 changed after the verdict: found=%v success=%v", found, s)
|
||||
}
|
||||
if _, found := storedSuccess(t, d, ip, "ssh"); found {
|
||||
t.Fatal("a check written after the verdict")
|
||||
}
|
||||
}
|
||||
|
||||
// recorded_at is the server's own write time and moves on every accepted write.
|
||||
func TestUpsertCheckIfOpenSetsRecordedAt(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
ip := checkingIP(t, d, "1.2.3.4")
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", true)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var first string
|
||||
if err := d.QueryRowContext(ctx, `SELECT recorded_at FROM checks WHERE ip_id=?`, ip.ID).Scan(&first); err != nil || first == "" {
|
||||
t.Fatalf("recorded_at not set: %q %v", first, err)
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
if _, err := d.UpsertCheckIfOpen(ctx, checkOf(ip, "icmp", false)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var second, created string
|
||||
if err := d.QueryRowContext(ctx, `SELECT recorded_at, created_at FROM checks WHERE ip_id=?`, ip.ID).Scan(&second, &created); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if second <= first || created != first {
|
||||
t.Fatalf("recorded_at must advance, created_at stay: first=%s second=%s created=%s", first, second, created)
|
||||
}
|
||||
}
|
||||
|
||||
// A prober site is handed an address until it reports it complete in the
|
||||
// current attempt; other sites are unaffected; a new attempt hands it out again.
|
||||
func TestListCheckingForSite(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
a := checkingIP(t, d, "1.1.1.1")
|
||||
b := checkingIP(t, d, "2.2.2.2")
|
||||
|
||||
names := func(items []IPQueueItem) string {
|
||||
s := ""
|
||||
for _, it := range items {
|
||||
s += it.IPAddress + " "
|
||||
}
|
||||
return s
|
||||
}
|
||||
for site := 1; site <= 2; site++ {
|
||||
items, err := d.ListCheckingForSite(ctx, site)
|
||||
if err != nil || len(items) != 2 {
|
||||
t.Fatalf("site %d before any report: %q err=%v", site, names(items), err)
|
||||
}
|
||||
}
|
||||
if err := d.SetSiteComplete(ctx, a.ID, 1); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if items, _ := d.ListCheckingForSite(ctx, 1); names(items) != "2.2.2.2 " {
|
||||
t.Fatalf("site 1 after completing 1.1.1.1: %q", names(items))
|
||||
}
|
||||
if items, _ := d.ListCheckingForSite(ctx, 2); len(items) != 2 {
|
||||
t.Fatalf("site 2 must still get both: %q", names(items))
|
||||
}
|
||||
|
||||
// A retry starts a new attempt: site 1 has to probe the address again.
|
||||
if err := d.RequeueOrFail(ctx, a.ID, "", 3); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.SetChecking(ctx, a.ID, time.Minute); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if items, _ := d.ListCheckingForSite(ctx, 1); len(items) != 2 {
|
||||
t.Fatalf("site 1 after a new attempt: %q", names(items))
|
||||
}
|
||||
_ = b
|
||||
}
|
||||
|
||||
// The aggregation window counts from the start of checking; a retry clears it.
|
||||
func TestCheckingStartedAt(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
ip := checkingIP(t, d, "1.2.3.4")
|
||||
if ip.CheckingStartedAt == nil || time.Since(*ip.CheckingStartedAt) > time.Minute {
|
||||
t.Fatalf("checking_started_at not set: %v", ip.CheckingStartedAt)
|
||||
}
|
||||
if err := d.RequeueOrFail(ctx, ip.ID, "", 3); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip, err := d.GetIP(ctx, ip.ID)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if ip.CheckingStartedAt != nil {
|
||||
t.Fatalf("checking_started_at must be cleared on requeue, got %v", ip.CheckingStartedAt)
|
||||
}
|
||||
}
|
||||
|
||||
// Migration 0010 flags the rows of an existing database that were written
|
||||
// after their address's verdict, and leaves the others alone.
|
||||
func TestMigration0010MarksRowsAfterVerdict(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
path := filepath.Join(t.TempDir(), "old.db")
|
||||
raw, err := sql.Open("sqlite", path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw.SetMaxOpenConns(1)
|
||||
for _, m := range migrations {
|
||||
if m.version > 9 {
|
||||
break
|
||||
}
|
||||
if _, err := raw.ExecContext(ctx, m.sql); err != nil {
|
||||
t.Fatalf("migration %d: %v", m.version, err)
|
||||
}
|
||||
}
|
||||
if _, err := raw.ExecContext(ctx, `PRAGMA user_version=9`); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, q := range []string{
|
||||
`INSERT INTO ip_registry (id, ip_address, first_seen_at, last_seen_at, next_cycle, created_at, updated_at)
|
||||
VALUES (1, '1.2.3.4', '2026-10-02T13:00:00Z', '2026-10-02T13:00:00Z', 2, '2026-10-02T13:00:00Z', '2026-10-02T13:00:00Z')`,
|
||||
`INSERT INTO ip_queue (id, ip_address, sequence, state, overall_result, aggregated_at, registry_id, cycle_id, created_at, updated_at)
|
||||
VALUES (1, '1.2.3.4', 1, 'done', 'pass', '2026-10-02T13:48:45.659Z', 1, 1, '2026-10-02T13:00:00Z', '2026-10-02T13:00:00Z')`,
|
||||
// before the verdict, and (with a longer fraction) after it
|
||||
`INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target, success, checked_at, created_at)
|
||||
VALUES (1, 1, 1, '1.2.3.4', 1, '', 'inbound-site-1', 'icmp', '1.2.3.4', 1, '2026-10-02T13:48:40.100000000Z', '2026-10-02T13:48:40.2Z')`,
|
||||
`INSERT INTO checks (registry_id, cycle_id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target, success, checked_at, created_at)
|
||||
VALUES (1, 1, 1, '1.2.3.4', 1, '', 'inbound-site-1', 'ssh', '1.2.3.4', 0, '2026-10-02T13:48:45.730314288Z', '2026-10-02T13:48:40.3Z')`,
|
||||
} {
|
||||
if _, err := raw.ExecContext(ctx, q); err != nil {
|
||||
t.Fatalf("seed: %v\n%s", err, q)
|
||||
}
|
||||
}
|
||||
raw.Close()
|
||||
|
||||
d, err := Open(ctx, path)
|
||||
if err != nil {
|
||||
t.Fatalf("open (runs migration 10): %v", err)
|
||||
}
|
||||
defer d.Close()
|
||||
flag := func(ct string) (after int, recorded, created string) {
|
||||
if err := d.QueryRowContext(ctx, `SELECT after_verdict, recorded_at, created_at FROM checks WHERE check_type=?`, ct).Scan(&after, &recorded, &created); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return
|
||||
}
|
||||
if a, rec, cr := flag("icmp"); a != 0 || rec != cr {
|
||||
t.Errorf("icmp: after_verdict=%d recorded=%s created=%s", a, rec, cr)
|
||||
}
|
||||
if a, rec, cr := flag("ssh"); a != 1 || rec != cr {
|
||||
t.Errorf("ssh: after_verdict=%d recorded=%s created=%s", a, rec, cr)
|
||||
}
|
||||
var ver int
|
||||
if err := d.QueryRowContext(ctx, `PRAGMA user_version`).Scan(&ver); err != nil || ver != 10 {
|
||||
t.Errorf("user_version=%d err=%v", ver, err)
|
||||
}
|
||||
}
|
||||
@@ -23,6 +23,15 @@ type okResponse struct {
|
||||
OK bool `json:"ok"`
|
||||
}
|
||||
|
||||
// resultsResponse answers the result-submission endpoints. Ignored counts the
|
||||
// submitted checks that were dropped because the address already has its
|
||||
// verdict (or the result belongs to an earlier attempt); the sender must not
|
||||
// retry them.
|
||||
type resultsResponse struct {
|
||||
OK bool `json:"ok"`
|
||||
Ignored int `json:"ignored,omitempty"`
|
||||
}
|
||||
|
||||
type observedIPResponse struct {
|
||||
IP string `json:"ip"`
|
||||
Source string `json:"source"`
|
||||
|
||||
@@ -93,6 +93,24 @@ type registryDTO struct {
|
||||
LastCheckedAt *time.Time `json:"last_checked_at"`
|
||||
InQueue bool `json:"in_queue"`
|
||||
CurrentState string `json:"current_state"`
|
||||
// LastCycleID is the cycle counted by Egress and Ingress (0 = no checks).
|
||||
LastCycleID int `json:"last_cycle_id"`
|
||||
Egress levelResultDTO `json:"egress"`
|
||||
Ingress levelResultDTO `json:"ingress"`
|
||||
}
|
||||
|
||||
// levelResultDTO is "ok of total" recorded checks of one level in the last
|
||||
// cycle, split by check family (tcp-22 and tcp-443 are both "tcp").
|
||||
type levelResultDTO struct {
|
||||
Total int `json:"total"`
|
||||
OK int `json:"ok"`
|
||||
ByType []typeStatDTO `json:"by_type"`
|
||||
}
|
||||
|
||||
type typeStatDTO struct {
|
||||
Type string `json:"type"`
|
||||
Total int `json:"total"`
|
||||
OK int `json:"ok"`
|
||||
}
|
||||
|
||||
type validatorDTO struct {
|
||||
|
||||
@@ -149,6 +149,7 @@ func (s *Server) handleAgentResults(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
dropped := map[int64]int{}
|
||||
for _, res := range req.Results {
|
||||
item, err := s.DB.GetIP(r.Context(), res.IPID)
|
||||
if err != nil {
|
||||
@@ -159,7 +160,7 @@ func (s *Server) handleAgentResults(w http.ResponseWriter, r *http.Request) {
|
||||
if err != nil {
|
||||
checkedAt = db.Now()
|
||||
}
|
||||
err = s.Orch.RecordCheck(r.Context(), db.Check{
|
||||
written, err := s.Orch.RecordCheckIfOpen(r.Context(), db.Check{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
|
||||
ValidatorID: id, Source: db.SourceEgress, CheckType: res.CheckType, Target: res.Target,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
|
||||
@@ -168,8 +169,11 @@ func (s *Server) handleAgentResults(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
if !written {
|
||||
dropped[item.ID]++
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
writeJSON(w, http.StatusOK, s.reportDropped(r.Context(), "validator-agent", id, db.SourceEgress, dropped))
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentComplete(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
@@ -41,9 +43,12 @@ func (s *Server) handleProberHeartbeat(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
// handleProberAssignments returns every IP currently in the checking
|
||||
// state — probers work the whole active set each poll, not one IP at a
|
||||
// time, since multiple validators run in parallel.
|
||||
// handleProberAssignments returns every IP currently in the checking state
|
||||
// that this site has not yet finished probing in the current attempt —
|
||||
// probers work the whole active set each poll, not one IP at a time, since
|
||||
// multiple validators run in parallel. An address is handed out until the
|
||||
// site reports it complete, then no more: one probe round per site per
|
||||
// attempt, so results are written once and never overwritten.
|
||||
func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) {
|
||||
siteID := r.PathValue("site_id")
|
||||
idx, err := s.Orch.SiteIndexForID(r.Context(), siteID)
|
||||
@@ -55,7 +60,7 @@ func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request)
|
||||
writeError(w, http.StatusNotFound, "unknown site_id: "+siteID)
|
||||
return
|
||||
}
|
||||
items, err := s.DB.ListChecking(r.Context())
|
||||
items, err := s.DB.ListCheckingForSite(r.Context(), idx)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
@@ -93,6 +98,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
completed := map[int64]bool{}
|
||||
dropped := map[int64]int{}
|
||||
for _, res := range req.Results {
|
||||
item, err := s.DB.GetIP(r.Context(), res.IPID)
|
||||
if err != nil {
|
||||
@@ -103,7 +109,7 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
if err != nil {
|
||||
checkedAt = db.Now()
|
||||
}
|
||||
err = s.Orch.RecordCheck(r.Context(), db.Check{
|
||||
written, err := s.Orch.RecordCheckIfOpen(r.Context(), db.Check{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
|
||||
Source: db.InboundSource(siteIndex), CheckType: res.CheckType, Target: res.IPAddress,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
|
||||
@@ -112,6 +118,10 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
if !written {
|
||||
dropped[item.ID]++
|
||||
continue
|
||||
}
|
||||
if res.Complete {
|
||||
completed[res.IPID] = true
|
||||
}
|
||||
@@ -122,5 +132,20 @@ func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
writeJSON(w, http.StatusOK, s.reportDropped(r.Context(), "prober", siteID, db.InboundSource(siteIndex), dropped))
|
||||
}
|
||||
|
||||
// reportDropped records one result_dropped event per address whose submitted
|
||||
// checks were refused (the verdict already exists, or the attempt is old) and
|
||||
// builds the response. The events are the audit trail of how often senders
|
||||
// are late; in a healthy run there are none.
|
||||
func (s *Server) reportDropped(ctx context.Context, sourceType, sourceID, checkSource string, dropped map[int64]int) resultsResponse {
|
||||
total := 0
|
||||
for ipID, n := range dropped {
|
||||
total += n
|
||||
id := ipID
|
||||
s.Orch.RecordEvent(ctx, sourceType, sourceID, &id, "result_dropped",
|
||||
fmt.Sprintf(`{"source":%q,"dropped":%d}`, checkSource, n))
|
||||
}
|
||||
return resultsResponse{OK: true, Ignored: total}
|
||||
}
|
||||
@@ -82,5 +82,16 @@ func registrySummaryToDTO(s db.RegistrySummary) registryDTO {
|
||||
LastCheckedAt: s.LastCheckedAt,
|
||||
InQueue: s.InQueue,
|
||||
CurrentState: s.CurrentState,
|
||||
LastCycleID: s.LastCycleID,
|
||||
Egress: levelResultToDTO(s.Egress),
|
||||
Ingress: levelResultToDTO(s.Ingress),
|
||||
}
|
||||
}
|
||||
|
||||
func levelResultToDTO(l db.LevelResult) levelResultDTO {
|
||||
out := levelResultDTO{Total: l.Total, OK: l.OK, ByType: make([]typeStatDTO, len(l.ByType))}
|
||||
for i, t := range l.ByType {
|
||||
out.ByType[i] = typeStatDTO{Type: t.Type, Total: t.Total, OK: t.OK}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -1,9 +1,11 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -149,3 +151,71 @@ func TestRegistryHistoryOutlivesIPDeletion(t *testing.T) {
|
||||
t.Fatalf("expected a single registry entry for the address, got %+v", all)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAdminRegistryLevels proves the registry endpoints report the last
|
||||
// cycle's recorded checks as "ok of total" per level (egress / ingress) and
|
||||
// per check family, and that by_type is an empty list, not null, when a
|
||||
// level has no checks.
|
||||
func TestAdminRegistryLevels(t *testing.T) {
|
||||
fc, d, _, _ := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
if _, err := d.SubmitIPs(ctx, []string{"7.7.7.7", "8.8.8.8"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ip, err := d.GetIPByAddress(ctx, "7.7.7.7")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, c := range []struct {
|
||||
source, typ, target string
|
||||
ok bool
|
||||
}{
|
||||
{db.SourceEgress, "https", "https://a.test", true},
|
||||
{db.SourceEgress, "https", "https://b.test", false},
|
||||
{db.SourceEgress, "icmp", "a.test", true},
|
||||
{db.InboundSource(1), "tcp-22", "7.7.7.7", true},
|
||||
{db.InboundSource(1), "tcp-443", "7.7.7.7", true},
|
||||
{db.InboundSource(2), "tcp-22", "7.7.7.7", false},
|
||||
} {
|
||||
if err := d.UpsertCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: "7.7.7.7", AttemptNumber: ip.AttemptNumber,
|
||||
Source: c.source, CheckType: c.typ, Target: c.target, Success: c.ok, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
resp, body := fc.do(http.MethodGet, "/api/v1/admin/registry/7.7.7.7", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("detail: %d %s", resp.StatusCode, body)
|
||||
}
|
||||
var detail struct {
|
||||
Registry registryDTO `json:"registry"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &detail); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r := detail.Registry
|
||||
if r.LastCycleID != 1 {
|
||||
t.Errorf("last_cycle_id = %d", r.LastCycleID)
|
||||
}
|
||||
wantEgress := levelResultDTO{Total: 3, OK: 2, ByType: []typeStatDTO{{"https", 2, 1}, {"icmp", 1, 1}}}
|
||||
wantIngress := levelResultDTO{Total: 3, OK: 2, ByType: []typeStatDTO{{"tcp", 3, 2}}}
|
||||
if !reflect.DeepEqual(r.Egress, wantEgress) || !reflect.DeepEqual(r.Ingress, wantIngress) {
|
||||
t.Errorf("egress=%+v ingress=%+v", r.Egress, r.Ingress)
|
||||
}
|
||||
|
||||
// The paginated list carries the same fields; an address without checks
|
||||
// serialises empty levels as {"total":0,"ok":0,"by_type":[]}.
|
||||
_, body = fc.do(http.MethodGet, "/api/v1/admin/registry?limit=10", nil)
|
||||
var page registryPageResponse
|
||||
if err := json.Unmarshal(body, &page); err != nil || len(page.Items) != 2 {
|
||||
t.Fatalf("list: %v %s", err, body)
|
||||
}
|
||||
if !reflect.DeepEqual(page.Items[0].Egress, wantEgress) {
|
||||
t.Errorf("list egress = %+v", page.Items[0].Egress)
|
||||
}
|
||||
if !bytes.Contains(body, []byte(`"egress":{"total":0,"ok":0,"by_type":[]}`)) {
|
||||
t.Errorf("empty level must serialise by_type as [], got %s", body)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
// setupCheckingIP stands up two sites and one validator and leaves 9.9.9.9 in
|
||||
// the checking state.
|
||||
func setupCheckingIP(t *testing.T) (*fakeClient, *db.DB, *db.IPQueueItem) {
|
||||
t.Helper()
|
||||
fc, d, orch, mock := newConfigTestHarness(t)
|
||||
ctx := context.Background()
|
||||
mock.Seed("fip-1", "9.9.9.9", "svc-project")
|
||||
fc.do(http.MethodPut, "/api/v1/admin/config/sites/1", putSiteRequest{SiteID: "site-1"})
|
||||
fc.do(http.MethodPut, "/api/v1/admin/config/sites/2", putSiteRequest{SiteID: "site-2"})
|
||||
fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}})
|
||||
orch.Tick(ctx)
|
||||
_, body := fc.do(http.MethodGet, "/api/v1/agents/validator-1/assignment", nil)
|
||||
var a assignmentResponse
|
||||
if err := json.Unmarshal(body, &a); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
fc.do(http.MethodPost, "/api/v1/agents/validator-1/self-check", selfCheckRequest{IPID: a.IPID, DetectedEgress: "9.9.9.9", Success: true, Detail: "matched"})
|
||||
fc.do(http.MethodPost, "/api/v1/probers/register", registerProberRequest{SiteID: "site-1"})
|
||||
fc.do(http.MethodPost, "/api/v1/probers/register", registerProberRequest{SiteID: "site-2"})
|
||||
ip, err := d.GetIP(ctx, a.IPID)
|
||||
if err != nil || ip.State != db.IPChecking {
|
||||
t.Fatalf("expected checking: %v %+v", err, ip)
|
||||
}
|
||||
return fc, d, ip
|
||||
}
|
||||
|
||||
func proberAssignmentsFor(t *testing.T, fc *fakeClient, site string) []proberAssignment {
|
||||
t.Helper()
|
||||
resp, body := fc.do(http.MethodGet, "/api/v1/probers/"+site+"/assignments", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("%s assignments: %d %s", site, resp.StatusCode, body)
|
||||
}
|
||||
var out []proberAssignment
|
||||
if err := json.Unmarshal(body, &out); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func proberResult(ip *db.IPQueueItem, ct string, ok, complete bool) proberResultDTO {
|
||||
return proberResultDTO{IPID: ip.ID, IPAddress: ip.IPAddress, CheckType: ct, Success: ok,
|
||||
CheckedAt: time.Now().UTC().Format(time.RFC3339Nano), Complete: complete}
|
||||
}
|
||||
|
||||
// A prober site gets an address until it has reported it complete, then not
|
||||
// again; the other site still gets it.
|
||||
func TestProberAssignmentsOncePerSite(t *testing.T) {
|
||||
fc, _, ip := setupCheckingIP(t)
|
||||
|
||||
if len(proberAssignmentsFor(t, fc, "site-1")) != 1 || len(proberAssignmentsFor(t, fc, "site-2")) != 1 {
|
||||
t.Fatal("both sites must get the address before reporting")
|
||||
}
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/probers/site-1/results", proberResultsRequest{
|
||||
Results: []proberResultDTO{proberResult(ip, "tcp-22", true, false), proberResult(ip, "icmp", true, true)},
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("results: %d %s", resp.StatusCode, body)
|
||||
}
|
||||
if got := proberAssignmentsFor(t, fc, "site-1"); len(got) != 0 {
|
||||
t.Fatalf("site-1 must not get the address again after completing it: %+v", got)
|
||||
}
|
||||
if got := proberAssignmentsFor(t, fc, "site-2"); len(got) != 1 {
|
||||
t.Fatalf("site-2 still has to probe the address: %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Results that arrive after the verdict are refused with ignored>0, leave the
|
||||
// stored checks untouched and leave a result_dropped event behind; the same
|
||||
// for the validator agent's results.
|
||||
func TestResultsAfterVerdictAreIgnored(t *testing.T) {
|
||||
fc, d, ip := setupCheckingIP(t)
|
||||
ctx := context.Background()
|
||||
|
||||
fc.do(http.MethodPost, "/api/v1/probers/site-1/results", proberResultsRequest{
|
||||
Results: []proberResultDTO{proberResult(ip, "tcp-22", true, true)},
|
||||
})
|
||||
fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{
|
||||
Results: []checkResultDTO{{IPID: ip.ID, CheckType: "https", Target: "https://example.test", Success: true, CheckedAt: time.Now().UTC().Format(time.RFC3339Nano)}},
|
||||
})
|
||||
if err := d.SetAggregating(ctx, ip.ID); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := d.FinishIP(ctx, ip.ID, db.ResultPartial); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// prober: one new check and one that would flip a stored success
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/probers/site-1/results", proberResultsRequest{
|
||||
Results: []proberResultDTO{proberResult(ip, "tcp-22", false, false), proberResult(ip, "ssh", false, true)},
|
||||
})
|
||||
var rr resultsResponse
|
||||
if resp.StatusCode != http.StatusOK || json.Unmarshal(body, &rr) != nil || !rr.OK || rr.Ignored != 2 {
|
||||
t.Fatalf("prober late results: %d %s", resp.StatusCode, body)
|
||||
}
|
||||
// agent
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{
|
||||
Results: []checkResultDTO{{IPID: ip.ID, CheckType: "https", Target: "https://example.test", Success: false, CheckedAt: time.Now().UTC().Format(time.RFC3339Nano)}},
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK || json.Unmarshal(body, &rr) != nil || rr.Ignored != 1 {
|
||||
t.Fatalf("agent late results: %d %s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
rows, err := d.ListChecksForAttempt(ctx, ip.ID, ip.AttemptNumber)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(rows) != 2 {
|
||||
t.Fatalf("expected the 2 checks written before the verdict, got %d", len(rows))
|
||||
}
|
||||
for _, c := range rows {
|
||||
if !c.Success {
|
||||
t.Errorf("%s/%s changed after the verdict", c.Source, c.CheckType)
|
||||
}
|
||||
}
|
||||
var dropped int
|
||||
if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM events WHERE event_type='result_dropped' AND ip_id=?`, ip.ID).Scan(&dropped); err != nil || dropped != 2 {
|
||||
t.Fatalf("expected 2 result_dropped events (prober, agent), got %d err=%v", dropped, err)
|
||||
}
|
||||
}
|
||||
@@ -354,8 +354,20 @@ func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int,
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
// completion signal (egress or a given site's full port+icmp sweep),
|
||||
// updates the corresponding *_complete flag.
|
||||
// RecordCheckIfOpen stores one result for an address that is still being
|
||||
// checked. It returns false when the result was dropped because the address
|
||||
// already has (or is computing) its verdict, or the result belongs to an
|
||||
// earlier attempt: checks are frozen at the verdict so that the verdict and
|
||||
// the stored checks always agree.
|
||||
func (o *Orchestrator) RecordCheckIfOpen(ctx context.Context, c db.Check) (bool, error) {
|
||||
return o.DB.UpsertCheckIfOpen(ctx, c)
|
||||
}
|
||||
|
||||
// RecordCheck is RecordCheckIfOpen for callers that do not need to know
|
||||
// whether the result was dropped.
|
||||
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
|
||||
return o.DB.UpsertCheck(ctx, c)
|
||||
_, err := o.RecordCheckIfOpen(ctx, c)
|
||||
return err
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
|
||||
@@ -664,7 +676,15 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
// reported in ip_site_checks), with no need to wait on a prober that will
|
||||
// never exist, and no cap on how many sites can be configured.
|
||||
func (o *Orchestrator) isReadyToAggregate(ctx context.Context, item db.IPQueueItem, deadline time.Time, sites []db.Site) (bool, error) {
|
||||
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
|
||||
// The window counts from the start of checking, not from the assignment:
|
||||
// assigned_at also covers floating-IP association, the settle pause and the
|
||||
// self-check, which would leave only a few seconds for the checks. Rows from
|
||||
// before checking_started_at existed fall back to assigned_at.
|
||||
started := item.CheckingStartedAt
|
||||
if started == nil {
|
||||
started = item.AssignedAt
|
||||
}
|
||||
if started != nil && started.Before(deadline) {
|
||||
return true, nil
|
||||
}
|
||||
if !item.EgressComplete {
|
||||
@@ -695,36 +715,15 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
if err != nil {
|
||||
return fmt.Errorf("expected check count: %w", err)
|
||||
}
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing := expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
var result string
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
result, passCount, missing := computeVerdict(checks, expected, o.Agg.MissingCountsAsFail)
|
||||
egress, ingress := countByLevel(checks)
|
||||
|
||||
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
|
||||
return err
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "aggregated",
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d,"egress":%d,"ingress":%d}`,
|
||||
result, len(checks), passCount, missing, egress, ingress))
|
||||
|
||||
if settings, err := o.DB.GetSettings(ctx); err != nil {
|
||||
o.Log.Error("get settings for history retention", "ip_id", item.ID, "err", err)
|
||||
@@ -751,6 +750,52 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
return nil
|
||||
}
|
||||
|
||||
// computeVerdict is the single rule that turns an address's stored checks into
|
||||
// its overall result. Egress and ingress checks count alike, one check one
|
||||
// vote. A check that was expected but has no stored row counts as a failure,
|
||||
// so an incomplete set can never be "pass". The verdict is a pure function of
|
||||
// the stored checks and the expected count: recomputing it later from the
|
||||
// same rows gives the same result.
|
||||
func computeVerdict(checks []db.Check, expected int, missingCountsAsFail bool) (result string, passCount, missing int) {
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing = expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && missingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
return result, passCount, missing
|
||||
}
|
||||
|
||||
// countByLevel counts checks per level (egress, ingress), for the audit
|
||||
// payload of the aggregated event.
|
||||
func countByLevel(checks []db.Check) (egress, ingress int) {
|
||||
for _, c := range checks {
|
||||
switch db.CheckLevel(c.Source) {
|
||||
case db.LevelEgress:
|
||||
egress++
|
||||
case db.LevelIngress:
|
||||
ingress++
|
||||
}
|
||||
}
|
||||
return egress, ingress
|
||||
}
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe). Reads the current check_types/targets/sites/inbound
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func chk(src, ct string, ok bool) db.Check {
|
||||
return db.Check{Source: src, CheckType: ct, Success: ok}
|
||||
}
|
||||
|
||||
func TestComputeVerdict(t *testing.T) {
|
||||
in1 := db.InboundSource(1)
|
||||
cases := []struct {
|
||||
name string
|
||||
checks []db.Check
|
||||
expected int
|
||||
want string
|
||||
missing int
|
||||
}{
|
||||
{"all pass, complete", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 2, db.ResultPass, 0},
|
||||
{"egress failure", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", true)}, 2, db.ResultPartial, 0},
|
||||
{"ingress failure", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "ssh", false)}, 2, db.ResultPartial, 0},
|
||||
{"all recorded pass but one missing", []db.Check{chk(db.SourceEgress, "https", true)}, 2, db.ResultPartial, 1},
|
||||
{"all fail", []db.Check{chk(db.SourceEgress, "https", false), chk(in1, "icmp", false)}, 2, db.ResultFail, 0},
|
||||
{"nothing recorded", nil, 2, db.ResultFail, 2},
|
||||
{"more recorded than expected", []db.Check{chk(db.SourceEgress, "https", true), chk(in1, "icmp", true)}, 1, db.ResultPass, 0},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, _, missing := computeVerdict(c.checks, c.expected, true)
|
||||
if got != c.want || missing != c.missing {
|
||||
t.Errorf("%s: got %s missing=%d, want %s missing=%d", c.name, got, missing, c.want, c.missing)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The window starts when checking starts: an address handed out long ago but
|
||||
// that only just began checking must not be cut off.
|
||||
func TestAggregationWindowCountsFromCheckingStart(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, _, _ := newTestOrchestrator(t, 180)
|
||||
deadline := time.Now().Add(-120 * time.Second)
|
||||
long, recent := time.Now().Add(-10*time.Minute), time.Now()
|
||||
|
||||
item := db.IPQueueItem{AssignedAt: &long, CheckingStartedAt: &recent}
|
||||
ready, err := o.isReadyToAggregate(ctx, item, deadline, nil)
|
||||
if err != nil || ready {
|
||||
t.Fatalf("recent checking start must not be cut off by an old assignment: ready=%v err=%v", ready, err)
|
||||
}
|
||||
|
||||
started := time.Now().Add(-3 * time.Minute)
|
||||
item.CheckingStartedAt = &started
|
||||
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
|
||||
t.Fatalf("window elapsed since checking start: ready=%v err=%v", ready, err)
|
||||
}
|
||||
|
||||
// Rows from before checking_started_at existed fall back to assigned_at.
|
||||
item.CheckingStartedAt = nil
|
||||
if ready, err := o.isReadyToAggregate(ctx, item, deadline, nil); err != nil || !ready {
|
||||
t.Fatalf("fallback to assigned_at: ready=%v err=%v", ready, err)
|
||||
}
|
||||
}
|
||||
|
||||
// After the verdict nothing can change the stored checks, and the verdict is
|
||||
// exactly what the stored checks give when computed again.
|
||||
func TestVerdictMatchesStoredChecksAndLateResultIsDropped(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
|
||||
rec := func(src, ct string, ok bool) bool {
|
||||
t.Helper()
|
||||
written, err := o.RecordCheckIfOpen(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, ValidatorID: "validator-1",
|
||||
Source: src, CheckType: ct, Target: "t", Success: ok, CheckedAt: db.Now(),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return written
|
||||
}
|
||||
if !rec(db.SourceEgress, "https", true) || !rec(db.InboundSource(1), "tcp-22", false) {
|
||||
t.Fatal("results while checking must be stored")
|
||||
}
|
||||
_ = o.MarkEgressComplete(ctx, ip.ID)
|
||||
o.Cfg.CheckingWindowSeconds = 0
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
|
||||
// A late result: a new check, and an attempt to flip a stored one.
|
||||
if rec(db.InboundSource(2), "icmp", true) {
|
||||
t.Fatal("new check after the verdict was stored")
|
||||
}
|
||||
if rec(db.InboundSource(1), "tcp-22", true) {
|
||||
t.Fatal("overwrite after the verdict was accepted")
|
||||
}
|
||||
|
||||
stored, err := d.ListChecksForAttempt(ctx, ip.ID, ip.AttemptNumber)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(stored) != 2 {
|
||||
t.Fatalf("stored checks changed after the verdict: %d rows", len(stored))
|
||||
}
|
||||
expected, err := o.expectedCheckCount(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, _, _ := computeVerdict(stored, expected, o.Agg.MissingCountsAsFail); got != ip.OverallResult {
|
||||
t.Fatalf("recomputed verdict %s differs from stored %s", got, ip.OverallResult)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user