Files
cloud-ip-validator/internal/db/db.go
T
ayurishchevandClaude Sonnet 5.5 e95b5eb7d5 Retry a failed self-check on another validator; add the self-check failure ceiling
A validator that failed the self-check of an address no longer gets that address
again in the current round (ClaimNextQueued skips it); the validator itself stays
in service and takes all other addresses. The verdict fail is set when the number
of failed self-checks of an address reaches settings.self_check_max_attempts
(1..50, default 5, independent of the number of validators); max_retries and
retry_count are no longer used for self-check. If every working validator has
already failed the address, a new round starts and the exclusions lapse.

Migration 0012: ip_self_check_failures (permanent history per registry address),
ip_queue.sc_failures and sc_round_start_cycle (cycle_id is used instead of
attempt_number, which restarts when a queue row is recreated), the setting.
db.FailSelfCheck does it in one transaction; re-submission starts a new series.
API: self_check_max_attempts in GET/PUT /admin/config/orchestrator,
self_check_failed_on in /admin/ips/{ip} and /admin/registry/{ip}. Dashboard: the
field on /settings and the line "Self-check не прошёл на: ..." on the address
pages. Docs, plan and summary in docs/changes/.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-04 09:50:12 +03:00

151 lines
4.1 KiB
Go

// Package db owns the SQLite connection, schema migrations, and all queries
// used by the Control API. It is the only package in the system that talks
// to the database directly — agents and probers never connect to it.
package db
import (
"context"
"database/sql"
_ "embed"
"fmt"
"time"
_ "modernc.org/sqlite"
)
//go:embed migrations/0001_init.sql
var initSchema string
//go:embed migrations/0002_dynamic_config.sql
var dynamicConfigSchema string
//go:embed migrations/0003_fip_settle_delay.sql
var fipSettleDelaySchema string
//go:embed migrations/0004_inbound_checks_admin.sql
var inboundChecksAdminSchema string
//go:embed migrations/0005_unbounded_sites.sql
var unboundedSitesSchema string
//go:embed migrations/0006_prober_heartbeat.sql
var proberHeartbeatSchema string
//go:embed migrations/0007_ip_registry.sql
var ipRegistrySchema string
//go:embed migrations/0008_auto_cycle.sql
var autoCycleSchema string
//go:embed migrations/0009_scale_indexes.sql
var scaleIndexesSchema string
//go:embed migrations/0010_verdict_integrity.sql
var verdictIntegritySchema string
//go:embed migrations/0011_check_runs.sql
var checkRunsSchema string
//go:embed migrations/0012_self_check_failures.sql
var selfCheckFailuresSchema string
// migrations is the ordered list of schema versions. Each entry's SQL is
// applied, in order, for any version greater than the database's current
// PRAGMA user_version — so a fresh database walks the whole list and an
// existing one only picks up what's new.
var migrations = []struct {
version int
sql string
}{
{1, initSchema},
{2, dynamicConfigSchema},
{3, fipSettleDelaySchema},
{4, inboundChecksAdminSchema},
{5, unboundedSitesSchema},
{6, proberHeartbeatSchema},
{7, ipRegistrySchema},
{8, autoCycleSchema},
{9, scaleIndexesSchema},
{10, verdictIntegritySchema},
{11, checkRunsSchema},
{12, selfCheckFailuresSchema},
}
type DB struct {
*sql.DB
}
// Open opens (creating if necessary) the SQLite database at path, applies
// pragmas suited to a single-writer WAL workload, and runs any pending
// schema migrations.
func Open(ctx context.Context, path string) (*DB, error) {
sqlDB, err := sql.Open("sqlite", path+"?_pragma=busy_timeout(5000)")
if err != nil {
return nil, fmt.Errorf("open sqlite: %w", err)
}
// Control API is the sole writer; one connection avoids SQLITE_BUSY
// entirely for writes while still allowing concurrent reads via WAL.
sqlDB.SetMaxOpenConns(1)
for _, pragma := range []string{
"PRAGMA journal_mode=WAL",
"PRAGMA synchronous=NORMAL",
"PRAGMA foreign_keys=ON",
"PRAGMA busy_timeout=5000",
} {
if _, err := sqlDB.ExecContext(ctx, pragma); err != nil {
sqlDB.Close()
return nil, fmt.Errorf("apply pragma %q: %w", pragma, err)
}
}
d := &DB{DB: sqlDB}
if err := d.migrate(ctx); err != nil {
sqlDB.Close()
return nil, fmt.Errorf("migrate: %w", err)
}
if err := d.adoptOrphanQueueRows(ctx); err != nil {
sqlDB.Close()
return nil, fmt.Errorf("adopt queue rows into a run: %w", err)
}
return d, nil
}
// migrate applies every pending migration in order, tracked via
// PRAGMA user_version so repeated startups only apply what's new (and a
// fresh database walks the whole list once).
func (d *DB) migrate(ctx context.Context) error {
var version int
if err := d.QueryRowContext(ctx, "PRAGMA user_version").Scan(&version); err != nil {
return fmt.Errorf("read user_version: %w", err)
}
for _, m := range migrations {
if m.version <= version {
continue
}
tx, err := d.BeginTx(ctx, nil)
if err != nil {
return err
}
if _, err := tx.ExecContext(ctx, m.sql); err != nil {
tx.Rollback()
return fmt.Errorf("apply migration %d: %w", m.version, err)
}
if _, err := tx.ExecContext(ctx, fmt.Sprintf("PRAGMA user_version=%d", m.version)); err != nil {
tx.Rollback()
return fmt.Errorf("set user_version=%d: %w", m.version, err)
}
if err := tx.Commit(); err != nil {
return fmt.Errorf("commit migration %d: %w", m.version, err)
}
}
return nil
}
// Now returns the current time truncated to millisecond precision, the
// granularity used consistently for all timestamp columns.
func Now() time.Time {
return time.Now().UTC().Truncate(time.Millisecond)
}