repo init
This commit is contained in:
commit
7e44db87b2
48 files changed
+5346
No files matched your search
@@ -0,0 +1,53 @@
|
||||
# Cloud IP Validator
|
||||
|
||||
Система проверки освобождённых публичных IPv4-адресов перед их повторной
|
||||
выдачей: каждый адрес привязывается как Floating IP к ВМ-валидатору в
|
||||
облаке (OpenStack), после чего проверяется одновременно в двух
|
||||
направлениях — исходящий трафик валидатора (egress: HTTPS/ICMP/опционально
|
||||
SSH до внешних целей) и входящая доступность самого адреса с трёх
|
||||
независимых внешних площадок (inbound: TCP 22/80/443/8080 + ICMP). Итог по
|
||||
каждому адресу — `pass`/`partial`/`fail`, с полной историей проверок в
|
||||
базе данных.
|
||||
|
||||
Три компонента: `control-api` (управляющий сервис, единственный со
|
||||
состоянием), `validator-agent` (работает на каждой ВМ-валидаторе, без
|
||||
состояния) и `prober` (работает на каждой из трёх внешних площадок, без
|
||||
состояния). Все три общаются между собой только через HTTP API
|
||||
control-api.
|
||||
|
||||
## Документация
|
||||
|
||||
| Документ | Для чего |
|
||||
|---|---|
|
||||
| [docs/SETUP.md](docs/SETUP.md) | Сборка, конфигурация, первый запуск стенда — с нуля |
|
||||
| [docs/USAGE.md](docs/USAGE.md) | Повседневная работа: постановка адресов в очередь, наблюдение за статусом, разбор результатов |
|
||||
| [docs/API.md](docs/API.md) | Спецификация HTTP API control-api и примеры запросов (curl) |
|
||||
| [docs/LOCAL_E2E.md](docs/LOCAL_E2E.md) | Полностью офлайн-прогон всей системы одним скриптом — без реального облака и интернета |
|
||||
|
||||
## Быстрый старт (60 секунд, без OpenStack)
|
||||
|
||||
Хотите просто увидеть систему в работе — без реального облака:
|
||||
|
||||
```bash
|
||||
go build ./... && go test ./...
|
||||
scripts/run-local-e2e.sh
|
||||
```
|
||||
|
||||
Скрипт сам поднимет все три компонента как локальные процессы (в режиме
|
||||
`openstack.mode: mock`) и прогонит один тестовый адрес через полный цикл
|
||||
проверки, включая демонстрацию восстановления после сбоя валидатора.
|
||||
Подробности — в [docs/LOCAL_E2E.md](docs/LOCAL_E2E.md).
|
||||
|
||||
## Быстрый старт (реальный стенд)
|
||||
|
||||
1. Соберите три бинарника и подготовьте конфиги —
|
||||
[docs/SETUP.md](docs/SETUP.md).
|
||||
2. Разверните `control-api` на управляющей машине, `validator-agent` —
|
||||
на каждой ВМ-валидаторе, `prober` — на каждой из трёх площадок
|
||||
([пошагово в docs/SETUP.md](docs/SETUP.md#развёртывание-control-api)).
|
||||
3. Добавьте адреса в очередь и наблюдайте за результатом —
|
||||
[docs/USAGE.md](docs/USAGE.md).
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/status | python3 -m json.tool
|
||||
```
|
||||
@@ -0,0 +1,4 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!-- Do not edit this file with editors other than draw.io -->
|
||||
<!DOCTYPE svg PUBLIC "-//W3C//DTD SVG 1.1//EN" "http://www.w3.org/Graphics/SVG/1.1/DTD/svg11.dtd">
|
||||
<svg xmlns="http://www.w3.org/2000/svg" style="background: #ffffff; background-color: var(--ge-adaptive-bg, #ffffff); color-scheme: light dark;" xmlns:xlink="http://www.w3.org/1999/xlink" version="1.1" width="852px" height="1052px" viewBox="0 0 852 1052" id="ge-svg-O02TjIDkx6OmIMSwEVdR" content="<mxfile host="embed.diagrams.net" pages="2" scale="1" border="0"> <diagram name="Page-1" id="RKSnAO8CdRtgprh_bYX6"> <mxGraphModel dx="2360" dy="2190" grid="1" gridSize="10" guides="1" tooltips="1" connect="1" arrows="1" fold="1" page="1" pageScale="1" pageWidth="1169" pageHeight="827" math="0" shadow="0"> <root> <mxCell id="0" /> <mxCell id="1" parent="0" /> <mxCell id="hEIzrYoLT3NPbS02OzTw-21" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;" value="" vertex="1"> <mxGeometry height="360" width="210" x="250" y="540" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-1" parent="1" style="rounded=0;whiteSpace=wrap;html=1;" value="Наш пул доступных публичных IP адресов" vertex="1"> <mxGeometry height="60" width="160" x="1040" y="-60" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-2" parent="1" style="rounded=0;whiteSpace=wrap;html=1;" value="Высвободившийся публичный IPv4 заводится в сервисный проект и подключается в качестве FIP-адреса к свободному валидатору" vertex="1"> <mxGeometry height="60" width="320" x="640" y="380" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-3" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;" value="" vertex="1"> <mxGeometry height="360" width="440" x="520" y="540" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-4" parent="1" style="text;html=1;whiteSpace=wrap;strokeColor=none;fillColor=none;align=center;verticalAlign=middle;rounded=0;fontStyle=1" value="Сервисный проект" vertex="1"> <mxGeometry height="30" width="130" x="675" y="550" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-8" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;dashed=1;dashPattern=8 8;" value="" vertex="1"> <mxGeometry height="260" width="155" x="550" y="610" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-5" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;" value="Валидатор-1" vertex="1"> <mxGeometry height="60" width="120" x="567.5" y="660" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-6" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;" value="Валидатор-2" vertex="1"> <mxGeometry height="60" width="120" x="567.5" y="730" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-7" parent="1" style="rounded=1;whiteSpace=wrap;html=1;absoluteArcSize=1;arcSize=15;" value="Валидатор-3" vertex="1"> <mxGeometry height="60" width="120" x="567.5" y="800" as="geometry" /> </mxCell> <mxCell id="hEIzrYoLT3NPbS02OzTw-10" parent="1" style="text;html=1;whiteSpace=wrap;strokeColor=none;fillColor=none;align=center;verticalAlign=middle;rounded=0;" value="FIP-ВалидаторыLine truncated
|
||||
|
After Width: | Height: | Size: 1.5 MiB |
@@ -0,0 +1,139 @@
|
||||
// Command control-api is the orchestrator/control-plane binary: it reads
|
||||
// the deployment config, opens the SQLite database, connects to OpenStack
|
||||
// (or a mock, per config), seeds the IP work queue, and serves the HTTP API
|
||||
// that validator-agents and probers poll against.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"flag"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/httpapi"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
"cloudipvalidator/internal/orchestrator"
|
||||
)
|
||||
|
||||
func main() {
|
||||
configPath := flag.String("config", "configs/control-api.yaml", "path to control-api config file")
|
||||
flag.Parse()
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo}))
|
||||
|
||||
if err := run(*configPath, log); err != nil {
|
||||
log.Error("fatal", "err", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(configPath string, log *slog.Logger) error {
|
||||
cfg, err := config.LoadControlAPI(configPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("load config: %w", err)
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
database, err := db.Open(ctx, cfg.Database.Path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("open database: %w", err)
|
||||
}
|
||||
defer database.Close()
|
||||
|
||||
for _, v := range cfg.Validators {
|
||||
if err := database.RegisterValidator(ctx, v.ValidatorID, "", v.OSPortID, ""); err != nil {
|
||||
return fmt.Errorf("seed validator %s: %w", v.ValidatorID, err)
|
||||
}
|
||||
}
|
||||
if err := database.SeedQueue(ctx, cfg.IPAddresses); err != nil {
|
||||
return fmt.Errorf("seed ip queue: %w", err)
|
||||
}
|
||||
|
||||
osClient, err := newOpenStackClient(ctx, cfg)
|
||||
if err != nil {
|
||||
return fmt.Errorf("init openstack client: %w", err)
|
||||
}
|
||||
|
||||
orch := orchestrator.New(database, osClient, cfg, log)
|
||||
|
||||
srv := httpapi.New(database, orch, log)
|
||||
httpServer := &http.Server{Addr: cfg.Server.ListenAddr, Handler: srv.Handler()}
|
||||
|
||||
go runOrchestratorLoop(ctx, orch, cfg, log)
|
||||
|
||||
errCh := make(chan error, 1)
|
||||
go func() {
|
||||
log.Info("listening", "addr", cfg.Server.ListenAddr)
|
||||
if err := httpServer.ListenAndServe(); err != nil && err != http.ErrServerClosed {
|
||||
errCh <- err
|
||||
}
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
log.Info("shutting down")
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
return httpServer.Shutdown(shutdownCtx)
|
||||
case err := <-errCh:
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
func runOrchestratorLoop(ctx context.Context, orch *orchestrator.Orchestrator, cfg *config.ControlAPI, log *slog.Logger) {
|
||||
interval := time.Duration(cfg.Orchestrator.PollIntervalSeconds) * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
heartbeatTicker := time.NewTicker(interval * 2)
|
||||
defer heartbeatTicker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
orch.Tick(ctx)
|
||||
case <-heartbeatTicker.C:
|
||||
if err := orch.SweepStaleHeartbeats(ctx); err != nil {
|
||||
log.Error("sweep stale heartbeats", "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func newOpenStackClient(ctx context.Context, cfg *config.ControlAPI) (openstack.FloatingIPClient, error) {
|
||||
if cfg.OpenStack.Mode == "real" {
|
||||
clientCfg := openstack.ClientConfig{
|
||||
AuthURL: os.Getenv(cfg.OpenStack.AuthURLEnv),
|
||||
Token: os.Getenv(cfg.OpenStack.TokenEnv),
|
||||
ProjectID: os.Getenv(cfg.OpenStack.ProjectIDEnv),
|
||||
ProjectName: os.Getenv(cfg.OpenStack.ProjectNameEnv),
|
||||
DomainName: os.Getenv(cfg.OpenStack.ProjectDomainEnv),
|
||||
Region: os.Getenv(cfg.OpenStack.RegionEnv),
|
||||
}
|
||||
if clientCfg.AuthURL == "" || clientCfg.Token == "" {
|
||||
return nil, fmt.Errorf("openstack.mode=real requires %s and %s to be set in the environment",
|
||||
cfg.OpenStack.AuthURLEnv, cfg.OpenStack.TokenEnv)
|
||||
}
|
||||
return openstack.NewClient(ctx, clientCfg)
|
||||
}
|
||||
|
||||
// Mock mode: pre-seed one synthetic floating-ip resource per
|
||||
// configured address, standing in for the pre-allocated Neutron
|
||||
// floating IPs a real deployment's service project would already have.
|
||||
mock := openstack.NewMockClient()
|
||||
for i, addr := range cfg.IPAddresses {
|
||||
mock.Seed(fmt.Sprintf("mock-fip-%d", i), addr, "mock-project")
|
||||
}
|
||||
return mock, nil
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
// Command prober runs on one of the external test sites. It polls the
|
||||
// Control API for the set of floating IPs currently under test and probes
|
||||
// each directly (TCP connect + ICMP echo) to measure inbound reachability
|
||||
// from this vantage point.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"flag"
|
||||
"log/slog"
|
||||
"os"
|
||||
"os/signal"
|
||||
"syscall"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/probercore"
|
||||
)
|
||||
|
||||
func main() {
|
||||
configPath := flag.String("config", "configs/prober.yaml", "path to prober config file")
|
||||
flag.Parse()
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo}))
|
||||
|
||||
cfg, err := config.LoadProber(*configPath)
|
||||
if err != nil {
|
||||
log.Error("load config", "err", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
prober := probercore.New(cfg, log)
|
||||
if err := prober.Run(ctx); err != nil && err != context.Canceled {
|
||||
log.Error("prober stopped", "err", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// Command validator-agent runs on a validator VM. It is stateless: every
|
||||
// decision it makes is driven by polling the Control API, per the
|
||||
// deployment constraint that validators must be able to restart freely
|
||||
// without any local state to reconcile.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"flag"
|
||||
"log/slog"
|
||||
"net"
|
||||
"os"
|
||||
"os/signal"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
|
||||
"cloudipvalidator/internal/agentcore"
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
func main() {
|
||||
configPath := flag.String("config", "configs/validator-agent.yaml", "path to validator-agent config file")
|
||||
stubPorts := flag.String("stub-ports", "", "TEST ONLY: comma-separated TCP ports to accept-and-close on, standing in for the validator's real listening services in the offline end-to-end harness (see docs/LOCAL_E2E.md)")
|
||||
flag.Parse()
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo}))
|
||||
|
||||
cfg, err := config.LoadValidatorAgent(*configPath)
|
||||
if err != nil {
|
||||
log.Error("load config", "err", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
if *stubPorts != "" {
|
||||
startStubListeners(ctx, log, *stubPorts)
|
||||
}
|
||||
|
||||
agent := agentcore.New(cfg, log)
|
||||
if err := agent.Run(ctx); err != nil && err != context.Canceled {
|
||||
log.Error("agent stopped", "err", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
// startStubListeners binds trivial accept-and-close TCP listeners on the
|
||||
// given ports, standing in for the base-minimum services (22/80/443/8080)
|
||||
// a real validator would already run, so an offline prober has something
|
||||
// to successfully connect to. ICMP needs no stub: the kernel answers echo
|
||||
// requests to any locally-bound address on its own.
|
||||
func startStubListeners(ctx context.Context, log *slog.Logger, portsCSV string) {
|
||||
for _, p := range strings.Split(portsCSV, ",") {
|
||||
p = strings.TrimSpace(p)
|
||||
if p == "" {
|
||||
continue
|
||||
}
|
||||
port, err := strconv.Atoi(p)
|
||||
if err != nil {
|
||||
log.Error("invalid stub port", "value", p, "err", err)
|
||||
continue
|
||||
}
|
||||
ln, err := net.Listen("tcp", ":"+strconv.Itoa(port))
|
||||
if err != nil {
|
||||
log.Error("stub listener", "port", port, "err", err)
|
||||
continue
|
||||
}
|
||||
log.Info("stub listener up", "port", port)
|
||||
go func() {
|
||||
<-ctx.Done()
|
||||
ln.Close()
|
||||
}()
|
||||
go func() {
|
||||
for {
|
||||
conn, err := ln.Accept()
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
conn.Close()
|
||||
}
|
||||
}()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
# Control API configuration.
|
||||
#
|
||||
# OpenStack credentials are never set here — only the *names* of the
|
||||
# environment variables to read them from. The actual values must be
|
||||
# supplied by the process environment (see deploy/systemd/control-api.service
|
||||
# and its EnvironmentFile=).
|
||||
|
||||
server:
|
||||
listen_addr: ":8080"
|
||||
|
||||
database:
|
||||
path: "/var/lib/cloud-ip-validator/control-api.db"
|
||||
|
||||
openstack:
|
||||
mode: "real" # "mock" | "real" — mock uses an in-memory
|
||||
# OpenStack stand-in for local dev/testing
|
||||
auth_url_env: "OS_AUTH_URL"
|
||||
token_env: "OS_TOKEN"
|
||||
project_id_env: "OS_PROJECT_ID"
|
||||
project_name_env: "OS_PROJECT_NAME"
|
||||
project_domain_env: "OS_PROJECT_DOMAIN_NAME"
|
||||
region_env: "OS_REGION_NAME"
|
||||
|
||||
orchestrator:
|
||||
poll_interval_seconds: 5
|
||||
self_check_timeout_seconds: 60
|
||||
max_self_check_retries: 3
|
||||
checking_window_seconds: 120
|
||||
max_retries: 3
|
||||
lease_ttl_seconds: 180
|
||||
heartbeat_timeout_seconds: 30
|
||||
|
||||
aggregation:
|
||||
missing_counts_as_fail: true
|
||||
|
||||
# Validators are VMs in the service project; os_port_id is the Neutron port
|
||||
# ID of each validator's primary NIC, used when associating a floating IP.
|
||||
validators:
|
||||
- validator_id: "validator_01"
|
||||
os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_1"
|
||||
- validator_id: "validator_02"
|
||||
os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_2"
|
||||
|
||||
# The three external prober sites, indexed 1-3 (matches ip_queue.siteN_complete).
|
||||
sites:
|
||||
- site_id: "site-1"
|
||||
index: 1
|
||||
- site_id: "site-2"
|
||||
index: 2
|
||||
- site_id: "site-3"
|
||||
index: 3
|
||||
|
||||
# Outbound/egress check types the validator-agent runs, and which target
|
||||
# group (below) each runs against.
|
||||
check_types:
|
||||
- name: "https"
|
||||
enabled: true
|
||||
targets: ["default-targets"]
|
||||
- name: "icmp"
|
||||
enabled: true
|
||||
targets: ["default-targets"]
|
||||
- name: "ssh"
|
||||
enabled: false
|
||||
targets: []
|
||||
|
||||
targets:
|
||||
default-targets:
|
||||
- "https://hub.docker.com"
|
||||
- "https://github.com"
|
||||
- "https://packages.ubuntu.com"
|
||||
|
||||
# Inbound checks the 3 external-site probers run directly against each
|
||||
# validator's currently-assigned floating IP.
|
||||
inbound_checks:
|
||||
ports: [22, 80, 443, 8080]
|
||||
icmp: true
|
||||
|
||||
# The pool of public IPv4 addresses to validate, in the order they'll be
|
||||
# processed (ip_queue.sequence). Every address here is checked through to
|
||||
# the end of the list.
|
||||
ip_addresses:
|
||||
- "203.0.113.10"
|
||||
- "203.0.113.11"
|
||||
- "203.0.113.12"
|
||||
@@ -0,0 +1,11 @@
|
||||
# Prober configuration. Runs on one of the 3 external test sites. site_id
|
||||
# must match one of the control-api config's `sites[].site_id`.
|
||||
|
||||
site_id: "site-1"
|
||||
control_api_url: "http://control-api.internal:8080"
|
||||
poll_interval_seconds: 5
|
||||
|
||||
checks:
|
||||
tcp_timeout_seconds: 5
|
||||
icmp_timeout_seconds: 5
|
||||
icmp_count: 3
|
||||
@@ -0,0 +1,17 @@
|
||||
# Validator-agent configuration. Runs on each validator VM. validator_id
|
||||
# must match one of the control-api config's `validators[].validator_id`.
|
||||
|
||||
validator_id: "validator_01"
|
||||
control_api_url: "http://control-api.internal:8080"
|
||||
poll_interval_seconds: 5
|
||||
|
||||
self_check:
|
||||
timeout_seconds: 10
|
||||
|
||||
checks:
|
||||
https_timeout_seconds: 10
|
||||
icmp_timeout_seconds: 5
|
||||
icmp_count: 3
|
||||
ssh:
|
||||
enabled: false
|
||||
timeout_seconds: 5
|
||||
@@ -0,0 +1,24 @@
|
||||
[Unit]
|
||||
Description=Cloud IP Validator - Control API
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=cloud-ip-validator
|
||||
Group=cloud-ip-validator
|
||||
ExecStart=/usr/local/bin/control-api -config /etc/cloud-ip-validator/control-api.yaml
|
||||
# Holds OS_AUTH_URL / OS_TOKEN / OS_PROJECT_ID / etc — the admin credential
|
||||
# for OpenStack Floating IP management. Keep this file mode 0600, owned by
|
||||
# the service user; never commit it or put credentials in the YAML config.
|
||||
EnvironmentFile=/etc/cloud-ip-validator/control-api.env
|
||||
WorkingDirectory=/var/lib/cloud-ip-validator
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/var/lib/cloud-ip-validator
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -0,0 +1,22 @@
|
||||
[Unit]
|
||||
Description=Cloud IP Validator - Prober
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=cloud-ip-validator
|
||||
Group=cloud-ip-validator
|
||||
ExecStart=/usr/local/bin/prober -config /etc/cloud-ip-validator/prober.yaml
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
NoNewPrivileges=true
|
||||
# Unprivileged ICMP echo requires either root or CAP_NET_RAW; grant only
|
||||
# the latter rather than running the prober as root.
|
||||
AmbientCapabilities=CAP_NET_RAW
|
||||
CapabilityBoundingSet=CAP_NET_RAW
|
||||
ProtectSystem=strict
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -0,0 +1,22 @@
|
||||
[Unit]
|
||||
Description=Cloud IP Validator - Validator Agent
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=cloud-ip-validator
|
||||
Group=cloud-ip-validator
|
||||
ExecStart=/usr/local/bin/validator-agent -config /etc/cloud-ip-validator/validator-agent.yaml
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
NoNewPrivileges=true
|
||||
# Unprivileged ICMP echo requires either root or CAP_NET_RAW; grant only
|
||||
# the latter rather than running the agent as root.
|
||||
AmbientCapabilities=CAP_NET_RAW
|
||||
CapabilityBoundingSet=CAP_NET_RAW
|
||||
ProtectSystem=strict
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
+376
@@ -0,0 +1,376 @@
|
||||
# API Control API
|
||||
|
||||
Control API — единственная точка входа в систему для `validator-agent`,
|
||||
`prober` и оператора (администратора). Все данные передаются в формате
|
||||
JSON, базовый префикс прикладных методов — `/api/v1`.
|
||||
|
||||
> **Важно.** На данный момент API не защищён аутентификацией/авторизацией
|
||||
> — эндпоинты доступны любому, кто может достучаться до порта control-api
|
||||
> по сети. Для эксплуатации за пределами доверенного сегмента сети
|
||||
> обязательно ограничьте доступ на уровне сети/файрвола (см.
|
||||
> [SETUP.md](SETUP.md#сетевые-доступы)). Добавление bearer-токена — известное
|
||||
> направление доработки, в текущей версии не реализовано.
|
||||
|
||||
Базовый URL в примерах — `http://control-api.internal:8080`, замените на
|
||||
адрес вашего стенда (см. `server.listen_addr` в конфиге control-api).
|
||||
|
||||
## Содержание
|
||||
|
||||
- [Общие соглашения](#общие-соглашения)
|
||||
- [Методы для validator-agent](#методы-для-validator-agent)
|
||||
- [Методы для prober](#методы-для-prober)
|
||||
- [Служебные и административные методы](#служебные-и-административные-методы)
|
||||
- [Модель состояний и связь методов с ней](#модель-состояний-и-связь-методов-с-ней)
|
||||
- [Сквозной пример работы (curl)](#сквозной-пример-работы-curl)
|
||||
|
||||
## Общие соглашения
|
||||
|
||||
- Тело запроса и ответа — JSON (`Content-Type: application/json`).
|
||||
- Успешные ответы возвращают `200 OK`, либо `204 No Content` (когда
|
||||
данных нет — например, у валидатора сейчас нет назначения).
|
||||
- Ошибки возвращают `4xx`/`5xx` и тело вида:
|
||||
```json
|
||||
{"error": "текст ошибки"}
|
||||
```
|
||||
- Временные метки (`checked_at` в запросах) передаются в формате
|
||||
RFC3339/RFC3339Nano, например `2026-08-21T09:15:00.123456789Z`. Если поле
|
||||
не удалось распарсить, сервер молча подставит текущее время сервера — не
|
||||
полагайтесь на это в продакшене, всегда передавайте валидную метку.
|
||||
- `validator_id` и `site_id` в пути запроса должны совпадать со
|
||||
значениями, заданными в конфиге control-api (`validators[].validator_id`,
|
||||
`sites[].site_id`) — иначе методы, требующие существующую сущность,
|
||||
вернут `404`.
|
||||
|
||||
## Методы для validator-agent
|
||||
|
||||
Эти методы вызывает бинарник `validator-agent`, работающий на ВМ-валидаторе.
|
||||
Оператору вручную дёргать их обычно не требуется — они приведены для
|
||||
понимания протокола и для отладки через curl.
|
||||
|
||||
### `POST /api/v1/agents/register`
|
||||
|
||||
Регистрация/переактивация валидатора. Вызывается один раз при старте
|
||||
агента (и безопасно при каждом рестарте — идемпотентна).
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{
|
||||
"validator_id": "validator_01",
|
||||
"hostname": "vm-validator-01",
|
||||
"agent_version": "1.0.0"
|
||||
}
|
||||
```
|
||||
|
||||
Ответ:
|
||||
```json
|
||||
{"ok": true, "poll_interval_seconds": 5}
|
||||
```
|
||||
|
||||
`poll_interval_seconds` — рекомендованный интервал опроса, значение берётся
|
||||
из `orchestrator.poll_interval_seconds` конфига control-api.
|
||||
|
||||
> `validator_id` должен быть заранее описан в конфиге control-api
|
||||
> (`validators[].validator_id`) вместе с `os_port_id` — сам агент порт ID
|
||||
> не передаёт и не может его сменить через API.
|
||||
|
||||
### `POST /api/v1/agents/{id}/heartbeat`
|
||||
|
||||
"Я жив". Обновляет `last_heartbeat_at` валидатора. Если валидатор не
|
||||
присылает heartbeat дольше `orchestrator.heartbeat_timeout_seconds`, он
|
||||
помечается `unreachable`.
|
||||
|
||||
Запрос (тело необязательно, поля информационные):
|
||||
```json
|
||||
{"local_state": "idle"}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true}`. `404`, если `validator_id` не зарегистрирован.
|
||||
|
||||
### `GET /api/v1/agents/{id}/assignment`
|
||||
|
||||
Есть ли у валидатора сейчас работа. Опрашивается в каждом цикле.
|
||||
|
||||
- `204 No Content` — заданий нет.
|
||||
- `200 OK` с телом:
|
||||
```json
|
||||
{
|
||||
"ip_id": 42,
|
||||
"ip_address": "203.0.113.10",
|
||||
"phase": "awaiting_self_check",
|
||||
"check_config": [
|
||||
{"type": "https", "targets": ["https://hub.docker.com", "https://github.com", "https://packages.ubuntu.com"]},
|
||||
{"type": "icmp", "targets": ["https://hub.docker.com", "https://github.com", "https://packages.ubuntu.com"]}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`phase` — `awaiting_self_check` (нужно выполнить self-check) либо
|
||||
`checking` (self-check уже пройден, можно/нужно выполнять проверки).
|
||||
`check_config` — уже развёрнутая конфигурация проверок (тип + список
|
||||
целей), агенту не нужно самому сопоставлять группы целей.
|
||||
|
||||
### `POST /api/v1/agents/{id}/self-check`
|
||||
|
||||
Отчёт о результате self-check — подтверждение, что исходящий трафик
|
||||
валидатора действительно идёт через только что назначенный FIP. Агент
|
||||
определяет это, вызвав `GET /api/v1/whatsmyip` и сравнив ответ с
|
||||
`ip_address` из задания.
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{
|
||||
"ip_id": 42,
|
||||
"detected_egress_ip": "203.0.113.10",
|
||||
"success": true,
|
||||
"detail": "matched"
|
||||
}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true}`. При `success: false` control-api сам решает —
|
||||
повторить попытку назначения FIP или пометить IP как `failed` (после
|
||||
исчерпания `orchestrator.max_self_check_retries`).
|
||||
|
||||
### `POST /api/v1/agents/{id}/events`
|
||||
|
||||
Произвольная запись в журнал аудита, привязанная (опционально) к IP.
|
||||
Используется агентом для событий `config_received`, `self_check_result`
|
||||
и т.п.
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{
|
||||
"event_type": "config_received",
|
||||
"ip_id": 42,
|
||||
"payload": "{\"checks\":2}"
|
||||
}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true}`.
|
||||
|
||||
### `POST /api/v1/agents/{id}/results`
|
||||
|
||||
Отчёт о результатах исходящих (egress) проверок. Можно отправлять по
|
||||
одной проверке сразу после выполнения (рекомендуется — так прогресс не
|
||||
теряется при падении агента) либо пачкой.
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"ip_id": 42,
|
||||
"check_type": "https",
|
||||
"target": "https://github.com",
|
||||
"success": true,
|
||||
"latency_ms": 87,
|
||||
"detail": "ok",
|
||||
"checked_at": "2026-08-21T09:15:00.123Z"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true}`. Повторная отправка того же `(ip_id, check_type,
|
||||
target)` в рамках текущей попытки — безопасна и просто перезапишет
|
||||
результат (upsert по уникальному ключу).
|
||||
|
||||
### `POST /api/v1/agents/{id}/complete`
|
||||
|
||||
Сигнал "все исходящие проверки для этого IP выполнены".
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{"ip_id": 42}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true}`.
|
||||
|
||||
## Методы для prober
|
||||
|
||||
Эти методы вызывает бинарник `prober`, работающий на внешней площадке.
|
||||
|
||||
### `POST /api/v1/probers/register`
|
||||
|
||||
Регистрация пробера. `site_id` должен присутствовать в конфиге control-api
|
||||
(`sites[].site_id`), иначе — `400`.
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{"site_id": "site-1", "hostname": "probe-host-1"}
|
||||
```
|
||||
|
||||
Ответ: `{"ok": true, "poll_interval_seconds": 5}`.
|
||||
|
||||
### `GET /api/v1/probers/{site_id}/assignments`
|
||||
|
||||
Список всех IP, которые сейчас находятся в состоянии `checking` — то есть
|
||||
всё, что нужно прозондировать на этом цикле опроса (валидаторов может
|
||||
работать несколько параллельно, поэтому список, а не один IP).
|
||||
|
||||
Ответ:
|
||||
```json
|
||||
[
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "ports": [22, 80, 443, 8080], "icmp": true}
|
||||
]
|
||||
```
|
||||
|
||||
Пустой список `[]`, если сейчас нечего проверять.
|
||||
|
||||
### `POST /api/v1/probers/{site_id}/results`
|
||||
|
||||
Отчёт о результатах входящих (inbound) проверок с данной площадки.
|
||||
|
||||
Запрос:
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-22", "success": true, "latency_ms": 12, "checked_at": "2026-08-21T09:15:01Z", "complete": false},
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-80", "success": true, "latency_ms": 9, "checked_at": "2026-08-21T09:15:01Z", "complete": false},
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-443", "success": true, "latency_ms": 10, "checked_at": "2026-08-21T09:15:01Z", "complete": false},
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-8080","success": false,"latency_ms": 0, "checked_at": "2026-08-21T09:15:01Z", "complete": false},
|
||||
{"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "icmp", "success": true, "latency_ms": 5, "checked_at": "2026-08-21T09:15:01Z", "complete": true}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
`complete: true` нужно проставить ровно на одном (обычно последнем)
|
||||
результате в пачке — это сигнал "площадка N закончила зондирование этого
|
||||
IP на данном проходе". До этого момента control-api не будет считать
|
||||
данные с этой площадки завершёнными.
|
||||
|
||||
Ответ: `{"ok": true}`.
|
||||
|
||||
## Служебные и административные методы
|
||||
|
||||
### `GET /healthz`
|
||||
|
||||
Проверка живости процесса. Ответ: `{"ok": true}`. Используется в systemd/
|
||||
внешних системах мониторинга.
|
||||
|
||||
### `GET /api/v1/whatsmyip`
|
||||
|
||||
Возвращает IP-адрес, с которого пришёл TCP-запрос (без учёта заголовков
|
||||
`X-Forwarded-For` — специально, чтобы self-check нельзя было подделать).
|
||||
Это основа механизма self-check.
|
||||
|
||||
Ответ:
|
||||
```json
|
||||
{"ip": "203.0.113.10"}
|
||||
```
|
||||
|
||||
### `GET /api/v1/admin/status`
|
||||
|
||||
Сводка по очереди — сколько IP в каком состоянии.
|
||||
|
||||
```json
|
||||
{
|
||||
"total_ips": 25,
|
||||
"ips_by_state": {"queued": 10, "checking": 3, "done": 11, "failed": 1},
|
||||
"total_validators": 4
|
||||
}
|
||||
```
|
||||
|
||||
### `GET /api/v1/admin/ips`
|
||||
|
||||
Полный список всех IP из очереди со всеми полями (см.
|
||||
[USAGE.md](USAGE.md#значения-полей-ip) — расшифровка полей и статусов).
|
||||
|
||||
### `GET /api/v1/admin/ips/{ip}`
|
||||
|
||||
Детали по одному адресу: сам объект IP, все проверки текущей попытки и
|
||||
вся история событий по нему.
|
||||
|
||||
```json
|
||||
{
|
||||
"ip": { "ID": 42, "IPAddress": "203.0.113.10", "State": "done", "OverallResult": "pass", "...": "..." },
|
||||
"checks": [ {"Source": "egress", "CheckType": "https", "Target": "https://github.com", "Success": true, "...": "..."} ],
|
||||
"events": [ {"EventType": "fip_associated", "OccurredAt": "...", "...": "..."} ]
|
||||
}
|
||||
```
|
||||
|
||||
> Обратите внимание: вложенные объекты `ip`/`checks`/`events` сериализуются
|
||||
> без переопределения имён полей (используются имена Go-структур, например
|
||||
> `IPAddress`, `State`, `Success`) — в отличие от методов для
|
||||
> agent/prober, где поля в `snake_case`. Это осознанная асимметрия:
|
||||
> административные методы — для человека/дашборда, а не для машинного
|
||||
> протокола.
|
||||
|
||||
### `GET /api/v1/admin/validators`
|
||||
|
||||
Список всех валидаторов с их текущим состоянием (`unregistered`, `idle`,
|
||||
`assigned`, `checking`, `unreachable`) и `CurrentIPID`, если валидатор
|
||||
сейчас занят.
|
||||
|
||||
## Модель состояний и связь методов с ней
|
||||
|
||||
```
|
||||
queued ──(control-api сам, без вызова API)──▶ assigning_fip
|
||||
│
|
||||
OpenStack FIP associate успешен
|
||||
▼
|
||||
awaiting_self_check
|
||||
│
|
||||
POST .../self-check {success:true}
|
||||
▼
|
||||
checking
|
||||
│ POST .../results (agent) │ POST .../results (prober, x3 площадки)
|
||||
▼ ▼
|
||||
egress_complete=true siteN_complete=true (N=1,2,3)
|
||||
│
|
||||
все complete=true ИЛИ истекло checking_window_seconds
|
||||
▼
|
||||
aggregating
|
||||
│
|
||||
done (pass/partial/fail)
|
||||
или failed
|
||||
```
|
||||
|
||||
Переходы `queued → assigning_fip → awaiting_self_check` и финальная
|
||||
агрегация выполняются control-api самостоятельно по таймеру (см.
|
||||
`orchestrator.poll_interval_seconds`), явного HTTP-метода для их запуска
|
||||
нет — это фоновый цикл (`Tick`), а не запрос/ответ.
|
||||
|
||||
## Сквозной пример работы (curl)
|
||||
|
||||
Ниже — минимальный ручной прогон одного IP через API, как если бы вы
|
||||
писали собственного клиента вместо `validator-agent`/`prober`. Полезно
|
||||
для отладки и для понимания протокола.
|
||||
|
||||
```bash
|
||||
BASE=http://127.0.0.1:8080
|
||||
|
||||
# 1. Регистрация валидатора (validator_01 уже должен быть в конфиге control-api)
|
||||
curl -s -X POST "$BASE/api/v1/agents/register" \
|
||||
-d '{"validator_id":"validator_01","hostname":"debug-host","agent_version":"manual"}'
|
||||
|
||||
# 2. Подождать, пока control-api (фоновым тиком) назначит IP и привяжет FIP —
|
||||
# проверяем через admin/status или admin/ips, либо просто опрашиваем assignment
|
||||
curl -s "$BASE/api/v1/agents/validator_01/assignment"
|
||||
# => {"ip_id":1,"ip_address":"203.0.113.10","phase":"awaiting_self_check","check_config":[...]}
|
||||
|
||||
# 3. Self-check: спросить у control-api, каким адресом мы к нему пришли
|
||||
curl -s "$BASE/api/v1/whatsmyip"
|
||||
# => {"ip":"203.0.113.10"} (в реальном стенде это и есть проверка через FIP)
|
||||
|
||||
curl -s -X POST "$BASE/api/v1/agents/validator_01/self-check" \
|
||||
-d '{"ip_id":1,"detected_egress_ip":"203.0.113.10","success":true,"detail":"matched"}'
|
||||
|
||||
# 4. Отправить результаты egress-проверок (по одному check_config пункту)
|
||||
curl -s -X POST "$BASE/api/v1/agents/validator_01/results" \
|
||||
-d '{"results":[{"ip_id":1,"check_type":"https","target":"https://github.com","success":true,"latency_ms":80,"checked_at":"2026-08-21T09:00:00Z"}]}'
|
||||
|
||||
# 5. Сообщить, что все egress-проверки выполнены
|
||||
curl -s -X POST "$BASE/api/v1/agents/validator_01/complete" -d '{"ip_id":1}'
|
||||
|
||||
# 6. Со стороны пробера: узнать, что сейчас проверяется, и отправить результат
|
||||
curl -s -X POST "$BASE/api/v1/probers/register" -d '{"site_id":"site-1"}'
|
||||
curl -s "$BASE/api/v1/probers/site-1/assignments"
|
||||
curl -s -X POST "$BASE/api/v1/probers/site-1/results" \
|
||||
-d '{"results":[{"ip_id":1,"ip_address":"203.0.113.10","check_type":"icmp","success":true,"latency_ms":5,"checked_at":"2026-08-21T09:00:01Z","complete":true}]}'
|
||||
|
||||
# 7. Проверить итоговый результат (после того как control-api агрегирует)
|
||||
curl -s "$BASE/api/v1/admin/ips/203.0.113.10" | python3 -m json.tool
|
||||
```
|
||||
|
||||
Для полностью автоматизированного локального прогона (без ручных curl)
|
||||
см. `scripts/run-local-e2e.sh` и [docs/LOCAL_E2E.md](LOCAL_E2E.md).
|
||||
@@ -0,0 +1,92 @@
|
||||
# Local end-to-end smoke test
|
||||
|
||||
`scripts/run-local-e2e.sh` runs the full system as local processes with no
|
||||
real OpenStack cloud and no real internet access:
|
||||
|
||||
- **control-api** with `openstack.mode: mock` — the in-memory
|
||||
`openstack.MockClient` stands in for Neutron, pre-seeded with one
|
||||
synthetic floating IP per configured address (see
|
||||
`cmd/control-api/main.go`'s `newOpenStackClient`).
|
||||
- **1 validator-agent** (`validator_01`), started with
|
||||
`-stub-ports 12022,18081,18443,18888` — trivial accept-and-close TCP
|
||||
listeners standing in for the base-minimum services (22/80/443/8080) a
|
||||
real validator would run. ICMP needs no stub: the kernel answers echo
|
||||
requests to any local address (127.0.0.0/8) on its own.
|
||||
- **3 probers** (`site-1`/`site-2`/`site-3`), all probing `127.0.0.1`.
|
||||
- **3 `httpstub` instances** (`scripts/httpstub`) standing in for the real
|
||||
outbound targets (hub.docker.com / github.com / packages.ubuntu.com),
|
||||
always returning `200 OK`.
|
||||
|
||||
The generated config uses a short `lease_ttl_seconds: 8` and
|
||||
`checking_window_seconds: 15` so the whole run finishes in well under a
|
||||
minute instead of using the (much longer) production defaults.
|
||||
|
||||
**Why only one IP address (`127.0.0.1`), and why it must be exactly that
|
||||
one:** the self-check mechanism (`GET /whatsmyip`, see
|
||||
`internal/httpapi/server.go`'s `remoteIP`) compares the TCP source address
|
||||
the validator-agent's own outbound connection to control-api arrives with
|
||||
against the address it was just assigned. In a real deployment, Neutron
|
||||
actually SNATs the validator's egress traffic through whichever floating
|
||||
IP is attached, so any configured address self-checks correctly. This
|
||||
offline harness has no real network-level SNAT — the agent's traffic to
|
||||
control-api always really originates from `127.0.0.1` — so only that
|
||||
literal loopback address can ever pass self-check here. This is a
|
||||
limitation of the harness's fidelity, not of the self-check mechanism
|
||||
itself.
|
||||
|
||||
## Running it
|
||||
|
||||
```
|
||||
scripts/run-local-e2e.sh
|
||||
```
|
||||
|
||||
It will:
|
||||
|
||||
1. Build all three binaries plus `httpstub` into a temp workdir.
|
||||
2. Start the 3 stub HTTP targets, control-api, the validator-agent, and the
|
||||
3 probers.
|
||||
3. Wait for `GET /healthz` to come up.
|
||||
4. A few seconds in, **kill the validator-agent mid-run** and wait past the
|
||||
8s lease TTL, to demonstrate that control-api's lease sweep reclaims the
|
||||
in-flight IP (moves it back to `queued`, bumps `retry_count`) without
|
||||
any special crash-recovery code — it's the same sweep that runs every
|
||||
tick. It then restarts the validator-agent so the queue can finish.
|
||||
5. Poll `GET /api/v1/admin/status` until every configured IP has reached a
|
||||
terminal state (`done` or `failed`).
|
||||
6. Print the final `/api/v1/admin/status` and `/api/v1/admin/ips` output.
|
||||
|
||||
Expect to see `127.0.0.1` end with `"state":"done"` and
|
||||
`"overall_result":"pass"` (all egress checks against the stub targets
|
||||
succeed, and all 3 probers can reach the stub TCP listeners and get ICMP
|
||||
replies from loopback).
|
||||
|
||||
## Inspecting a run
|
||||
|
||||
The workdir (printed at the end, `/tmp/cloud-ip-validator-e2e.XXXXXX`) is
|
||||
**not** deleted automatically, so you can inspect:
|
||||
|
||||
- `logs/control-api.log`, `logs/validator-agent.log`, `logs/prober-site-*.log`
|
||||
- `control-api.db` — open with `sqlite3` to inspect the `checks` and
|
||||
`events` tables directly, e.g.:
|
||||
```
|
||||
sqlite3 /tmp/cloud-ip-validator-e2e.XXXXXX/control-api.db \
|
||||
"select ip_address, source, check_type, success from checks order by id"
|
||||
```
|
||||
|
||||
## What this does *not* cover
|
||||
|
||||
This harness proves the orchestration, HTTP protocol, and check-running
|
||||
logic all work together correctly. It does **not** exercise the real
|
||||
`internal/openstack/client.go` (gophercloud) path — that only runs against
|
||||
`openstack.mode: real` with actual OpenStack credentials. That path has its
|
||||
own read-only smoke test, `internal/openstack/client_live_test.go`, skipped
|
||||
by default and gated behind `OPENSTACK_LIVE_TEST=1`:
|
||||
|
||||
```
|
||||
OPENSTACK_LIVE_TEST=1 \
|
||||
OS_AUTH_URL=https://keystone.example:5000/v3 \
|
||||
OS_TOKEN=... \
|
||||
OS_PROJECT_ID=... \
|
||||
OS_TEST_FLOATING_IP=203.0.113.10 \
|
||||
go test ./internal/openstack/... -run TestClientLive -v
|
||||
```
|
||||
+275
@@ -0,0 +1,275 @@
|
||||
# Подготовка стенда и первичная инициализация
|
||||
|
||||
Документ описывает, как собрать компоненты, подготовить конфигурацию и
|
||||
запустить стенд с нуля — от чистой машины до работающего control-api,
|
||||
валидаторов и проберов. Если нужно просто быстро посмотреть систему в
|
||||
работе без реального OpenStack — сразу переходите к разделу
|
||||
[«Быстрая проверка без OpenStack»](#быстрая-проверка-без-openstack-offline-режим).
|
||||
|
||||
## Содержание
|
||||
|
||||
- [Компоненты и роли машин](#компоненты-и-роли-машин)
|
||||
- [Требования](#требования)
|
||||
- [Сборка бинарников](#сборка-бинарников)
|
||||
- [Быстрая проверка без OpenStack (offline-режим)](#быстрая-проверка-без-openstack-offline-режим)
|
||||
- [Подготовка конфигурации для реального стенда](#подготовка-конфигурации-для-реального-стенда)
|
||||
- [Развёртывание control-api](#развёртывание-control-api)
|
||||
- [Развёртывание validator-agent на ВМ-валидаторах](#развёртывание-validator-agent-на-вм-валидаторах)
|
||||
- [Развёртывание prober на внешних площадках](#развёртывание-prober-на-внешних-площадках)
|
||||
- [Проверка после запуска](#проверка-после-запуска)
|
||||
- [Сетевые доступы](#сетевые-доступы)
|
||||
|
||||
## Компоненты и роли машин
|
||||
|
||||
| Компонент | Где запускается | Кол-во |
|
||||
|---|---|---|
|
||||
| `control-api` | Отдельная управляющая машина/ВМ с доступом к OpenStack API | 1 (без HA) |
|
||||
| `validator-agent` | Каждая ВМ-валидатор в сервисном проекте облака | по числу валидаторов |
|
||||
| `prober` | По одному на каждой из внешних тестовых площадок | 3 (по числу площадок) |
|
||||
|
||||
`control-api` — единственный компонент с состоянием (SQLite). Валидаторы и
|
||||
проберы не хранят локального состояния и полностью управляются через опрос
|
||||
control-api (см. [API.md](API.md)).
|
||||
|
||||
## Требования
|
||||
|
||||
- **Go 1.22+** для сборки (проверено на Go 1.26). Собранные бинарники —
|
||||
статические, дополнительных зависимостей на целевых машинах не требуют
|
||||
(используется чистый Go-драйвер SQLite, без cgo).
|
||||
- Для `control-api` в боевом режиме (`openstack.mode: real`) — учётная
|
||||
запись OpenStack с правами на чтение/изменение floating IP (Neutron)
|
||||
в сервисном проекте, и заранее выделенные (allocated) floating IP —
|
||||
инструмент их **не создаёт**, только привязывает/отвязывает
|
||||
существующие.
|
||||
- Для `validator-agent` и `prober` — возможность отправлять ICMP echo
|
||||
(нужен root либо capability `CAP_NET_RAW`, см. юниты systemd).
|
||||
- `curl`, `sqlite3` (опционально, для ручной инспекции БД) на машине с
|
||||
control-api пригодятся для диагностики.
|
||||
|
||||
## Сборка бинарников
|
||||
|
||||
Из корня репозитория:
|
||||
|
||||
```bash
|
||||
export PATH=$PATH:/usr/local/go/bin # если go не в PATH
|
||||
go build -o bin/control-api ./cmd/control-api
|
||||
go build -o bin/validator-agent ./cmd/validator-agent
|
||||
go build -o bin/prober ./cmd/prober
|
||||
```
|
||||
|
||||
Каждый бинарник самодостаточен — скопируйте нужный файл на
|
||||
соответствующую машину (control-api → управляющая машина, validator-agent
|
||||
→ каждый валидатор, prober → каждая площадка).
|
||||
|
||||
Убедиться, что всё собирается и юнит-тесты проходят:
|
||||
|
||||
```bash
|
||||
go build ./... && go test ./...
|
||||
```
|
||||
|
||||
## Быстрая проверка без OpenStack (offline-режим)
|
||||
|
||||
Прежде чем разворачивать реальный стенд, рекомендуется убедиться, что всё
|
||||
собирается и работает корректно на локальной машине — без облака и
|
||||
внешних площадок:
|
||||
|
||||
```bash
|
||||
scripts/run-local-e2e.sh
|
||||
```
|
||||
|
||||
Скрипт сам поднимает control-api (в режиме `openstack.mode: mock`),
|
||||
одного validator-agent и трёх проберов как локальные процессы, прогоняет
|
||||
один тестовый адрес через полный цикл проверки и печатает итоговый
|
||||
результат. Подробности — в [docs/LOCAL_E2E.md](LOCAL_E2E.md). Это же
|
||||
хороший способ разобраться в поведении системы перед первым боевым
|
||||
запуском.
|
||||
|
||||
## Подготовка конфигурации для реального стенда
|
||||
|
||||
Все три компонента конфигурируются YAML-файлами. Шаблоны лежат в
|
||||
`configs/*.example.yaml` — скопируйте их и заполните под ваш стенд.
|
||||
|
||||
### 1. `control-api.yaml`
|
||||
|
||||
```bash
|
||||
cp configs/control-api.example.yaml /etc/cloud-ip-validator/control-api.yaml
|
||||
```
|
||||
|
||||
Что обязательно нужно заполнить:
|
||||
|
||||
- **`validators`** — список валидаторов, у каждого `validator_id`
|
||||
(произвольное имя, должно совпадать с `validator_id` в конфиге
|
||||
соответствующего `validator-agent`) и `os_port_id` — **ID Neutron-порта**
|
||||
основного сетевого интерфейса ВМ-валидатора (узнать: `openstack port
|
||||
list --server <имя-ВМ>` или в веб-консоли облака).
|
||||
- **`sites`** — три внешние площадки, `site_id` + `index` (1, 2 или 3).
|
||||
`site_id` должен совпадать с `site_id` в конфиге соответствующего
|
||||
`prober`.
|
||||
- **`ip_addresses`** — список публичных IPv4-адресов на проверку, **в
|
||||
порядке обработки**. Адреса должны существовать в сервисном проекте как
|
||||
уже выделенные (allocated) floating IP — инструмент их не создаёт.
|
||||
- **`openstack.mode: "real"`** и `*_env` поля — имена переменных
|
||||
окружения, из которых будут прочитаны реальные учётные данные (сами
|
||||
значения в этот файл **не пишутся**, см. следующий пункт).
|
||||
- **`targets`** и **`check_types`** — при необходимости смените набор
|
||||
целей для egress-проверок (по умолчанию — hub.docker.com, github.com,
|
||||
packages.ubuntu.com) или включите `ssh` (по умолчанию выключен).
|
||||
|
||||
### 2. Переменные окружения для OpenStack
|
||||
|
||||
Учётные данные передаются **только** через переменные окружения — никогда
|
||||
через YAML. Создайте файл (доступный на чтение только сервисному
|
||||
пользователю):
|
||||
|
||||
```bash
|
||||
install -m 0600 -o cloud-ip-validator -g cloud-ip-validator /dev/null /etc/cloud-ip-validator/control-api.env
|
||||
cat >> /etc/cloud-ip-validator/control-api.env <<'EOF'
|
||||
OS_AUTH_URL=https://keystone.example.com:5000/v3
|
||||
OS_TOKEN=<токен администратора с правами на управление floating IP>
|
||||
OS_PROJECT_ID=<id сервисного проекта>
|
||||
OS_REGION_NAME=<регион>
|
||||
EOF
|
||||
```
|
||||
|
||||
Имена переменных должны совпадать с тем, что указано в
|
||||
`control-api.yaml` в секции `openstack` (`auth_url_env`, `token_env` и
|
||||
т.д.) — в шаблоне это ровно `OS_AUTH_URL`, `OS_TOKEN`, `OS_PROJECT_ID`,
|
||||
`OS_PROJECT_NAME`, `OS_PROJECT_DOMAIN_NAME`, `OS_REGION_NAME`.
|
||||
|
||||
### 3. `validator-agent.yaml` (свой на каждом валидаторе)
|
||||
|
||||
```bash
|
||||
cp configs/validator-agent.example.yaml /etc/cloud-ip-validator/validator-agent.yaml
|
||||
```
|
||||
|
||||
Обязательно поменять:
|
||||
- `validator_id` — должен совпадать с одним из `validators[].validator_id`
|
||||
в конфиге control-api.
|
||||
- `control_api_url` — адрес, по которому эта ВМ достучится до control-api.
|
||||
|
||||
### 4. `prober.yaml` (свой на каждой площадке)
|
||||
|
||||
```bash
|
||||
cp configs/prober.example.yaml /etc/cloud-ip-validator/prober.yaml
|
||||
```
|
||||
|
||||
Обязательно поменять:
|
||||
- `site_id` — должен совпадать с одним из `sites[].site_id` в конфиге
|
||||
control-api (для трёх площадок — три разных файла с `site-1`,
|
||||
`site-2`, `site-3` или как вы их назвали).
|
||||
- `control_api_url` — адрес control-api, доступный с площадки (обычно
|
||||
через интернет — площадки внешние).
|
||||
|
||||
## Развёртывание control-api
|
||||
|
||||
```bash
|
||||
useradd --system --no-create-home --shell /usr/sbin/nologin cloud-ip-validator
|
||||
mkdir -p /var/lib/cloud-ip-validator /etc/cloud-ip-validator
|
||||
chown cloud-ip-validator:cloud-ip-validator /var/lib/cloud-ip-validator
|
||||
|
||||
cp bin/control-api /usr/local/bin/control-api
|
||||
cp deploy/systemd/control-api.service /etc/systemd/system/
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now control-api
|
||||
```
|
||||
|
||||
**Первичная инициализация базы данных происходит автоматически** — при
|
||||
первом старте `control-api` создаёт файл SQLite по пути `database.path`
|
||||
из конфига (миграция схемы применяется один раз, повторные запуски —
|
||||
no-op). Отдельной команды "init db" не требуется.
|
||||
|
||||
При каждом старте control-api также:
|
||||
1. Регистрирует в БД всех валидаторов из `validators` конфига (если их
|
||||
там ещё нет).
|
||||
2. Добавляет в очередь все адреса из `ip_addresses`, которых там ещё нет
|
||||
(уже обработанные ранее адреса повторно не добавляются и не
|
||||
сбрасываются — см. [USAGE.md](USAGE.md#добавление-новых-ip-в-очередь)).
|
||||
|
||||
Проверить, что процесс поднялся:
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:8080/healthz
|
||||
# {"ok":true}
|
||||
journalctl -u control-api -f
|
||||
```
|
||||
|
||||
## Развёртывание validator-agent на ВМ-валидаторах
|
||||
|
||||
Повторить на каждой ВМ-валидаторе:
|
||||
|
||||
```bash
|
||||
cp bin/validator-agent /usr/local/bin/validator-agent
|
||||
cp deploy/systemd/validator-agent.service /etc/systemd/system/
|
||||
mkdir -p /etc/cloud-ip-validator
|
||||
# скопировать сюда заполненный validator-agent.yaml с уникальным validator_id
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now validator-agent
|
||||
journalctl -u validator-agent -f
|
||||
```
|
||||
|
||||
Юнит выдаёт процессу capability `CAP_NET_RAW` (без root) — она нужна для
|
||||
отправки ICMP echo в рамках проверок.
|
||||
|
||||
## Развёртывание prober на внешних площадках
|
||||
|
||||
Аналогично, на каждой из трёх площадок:
|
||||
|
||||
```bash
|
||||
cp bin/prober /usr/local/bin/prober
|
||||
cp deploy/systemd/prober.service /etc/systemd/system/
|
||||
mkdir -p /etc/cloud-ip-validator
|
||||
# скопировать сюда prober.yaml с уникальным site_id для этой площадки
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl enable --now prober
|
||||
journalctl -u prober -f
|
||||
```
|
||||
|
||||
## Проверка после запуска
|
||||
|
||||
После того как control-api, все валидаторы и все три пробера запущены:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/status | python3 -m json.tool
|
||||
```
|
||||
|
||||
Ожидаемая картина сразу после старта: часть адресов в состоянии `queued`,
|
||||
часть уже переходит в `assigning_fip`/`awaiting_self_check`/`checking` по
|
||||
мере того, как освобождаются валидаторы. Через некоторое время появляются
|
||||
записи в `done`/`failed`. Подробнее о том, как читать этот вывод и что
|
||||
делать дальше — в [USAGE.md](USAGE.md).
|
||||
|
||||
Также стоит убедиться, что все валидаторы видны и не «зависли»:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/validators | python3 -m json.tool
|
||||
```
|
||||
|
||||
Все зарегистрированные валидаторы должны рано или поздно оказываться в
|
||||
состоянии `idle` (между заданиями) — если валидатор надолго застрял в
|
||||
`unreachable`, проверьте сетевую связность до control-api и логи агента
|
||||
(`journalctl -u validator-agent`).
|
||||
|
||||
## Сетевые доступы
|
||||
|
||||
Минимально необходимая связность:
|
||||
|
||||
- `validator-agent` → `control-api`: TCP, порт из `server.listen_addr`
|
||||
(обычно 8080).
|
||||
- `prober` (на каждой из 3 площадок) → `control-api`: тот же порт, обычно
|
||||
через интернет.
|
||||
- `prober` → адрес, который в данный момент проверяется (динамический,
|
||||
меняется по ходу работы очереди): TCP 22/80/443/8080 + ICMP —
|
||||
собственно и есть проверяемый трафик, его нельзя заранее ограничить
|
||||
одним IP.
|
||||
- `validator-agent` → интернет: HTTPS/ICMP до целей из `targets` конфига
|
||||
(по умолчанию hub.docker.com, github.com, packages.ubuntu.com) — именно
|
||||
через floating IP, который в данный момент привязан к валидатору.
|
||||
- `control-api` → OpenStack Keystone/Neutron API (`OS_AUTH_URL` и далее по
|
||||
каталогу сервисов).
|
||||
|
||||
API control-api сейчас не аутентифицирован (см. предупреждение в начале
|
||||
[API.md](API.md)) — ограничивайте доступ к порту control-api на уровне
|
||||
сети/firewall теми хостами, где реально работают валидаторы и проберы.
|
||||
+269
@@ -0,0 +1,269 @@
|
||||
# Работа со стендом
|
||||
|
||||
Этот документ — для оператора, который уже развернул стенд (см.
|
||||
[SETUP.md](SETUP.md)) и теперь использует его в повседневной работе:
|
||||
добавляет адреса на проверку, следит за очередью, разбирается в
|
||||
результатах и реагирует на проблемы. Прямые вызовы API описаны в
|
||||
[API.md](API.md) — здесь мы используем их только как инструмент, не
|
||||
углубляясь в протокол.
|
||||
|
||||
## Содержание
|
||||
|
||||
- [Как устроена работа с системой](#как-устроена-работа-с-системой)
|
||||
- [Добавление новых IP в очередь](#добавление-новых-ip-в-очередь)
|
||||
- [Наблюдение за очередью](#наблюдение-за-очередью)
|
||||
- [Значения полей IP](#значения-полей-ip)
|
||||
- [Как читать итоговый результат (pass/partial/fail)](#как-читать-итоговый-результат-passpartialfail)
|
||||
- [Просмотр деталей и истории по конкретному адресу](#просмотр-деталей-и-истории-по-конкретному-адресу)
|
||||
- [Управление валидаторами](#управление-валидаторами)
|
||||
- [Управление площадками (проберами)](#управление-площадками-проберами)
|
||||
- [Повторная проверка адреса](#повторная-проверка-адреса)
|
||||
- [Частые проблемы и что с ними делать](#частые-проблемы-и-что-с-ними-делать)
|
||||
|
||||
## Как устроена работа с системой
|
||||
|
||||
Оператор не взаимодействует с валидаторами и проберами напрямую — вся
|
||||
работа идёт через `control-api`. Цикл жизни одного IP-адреса:
|
||||
|
||||
1. Адрес встаёт в очередь (`queued`).
|
||||
2. Control-api сам находит свободный валидатор, привязывает адрес к нему
|
||||
как Floating IP.
|
||||
3. Валидатор проверяет, что действительно вышел в интернет именно через
|
||||
этот адрес (self-check), затем прогоняет исходящие проверки (HTTPS,
|
||||
ICMP, опционально SSH до заданных внешних целей).
|
||||
4. Одновременно три внешние площадки проверяют, что этот адрес доступен
|
||||
*снаружи* (входящие TCP-подключения на 22/80/443/8080 и ICMP) — это
|
||||
ловит блокировки/чёрные списки на конкретных внешних сетях.
|
||||
5. Как только все источники (валидатор + 3 площадки) отчитались — или
|
||||
истекло время ожидания — control-api подводит итог и освобождает
|
||||
адрес (отвязывает Floating IP).
|
||||
|
||||
Всё это происходит автоматически, без участия оператора. Задача оператора
|
||||
— положить адреса в очередь и снять с них результат.
|
||||
|
||||
## Добавление новых IP в очередь
|
||||
|
||||
**В текущей версии добавление адресов происходит только через конфиг
|
||||
control-api**, отдельного API-метода "добавить IP в очередь" нет.
|
||||
|
||||
1. Добавьте новые адреса в список `ip_addresses` в
|
||||
`/etc/cloud-ip-validator/control-api.yaml` (в конец списка, либо в
|
||||
нужном порядке — очередь обрабатывается строго в порядке следования
|
||||
списка, `sequence`).
|
||||
2. Перезапустите control-api:
|
||||
```bash
|
||||
systemctl restart control-api
|
||||
```
|
||||
|
||||
Это безопасно для уже идущей работы: при старте control-api добавляет в
|
||||
очередь только **новые** адреса (те, которых там ещё нет) — уже
|
||||
обработанные ранее адреса не сбрасываются и повторно не проверяются.
|
||||
Адреса, которые были удалены из `ip_addresses`, но уже есть в базе,
|
||||
**не удаляются** из очереди/истории автоматически — если конкретный адрес
|
||||
больше не нужно проверять и его нет в очереди/в процессе, можно просто
|
||||
оставить как есть (историю он не портит).
|
||||
|
||||
> Совет: держите `control-api.yaml` под версионным контролем (git) —
|
||||
> список адресов на проверку тогда одновременно служит и журналом того,
|
||||
> что вообще когда-либо ставилось в очередь.
|
||||
|
||||
## Наблюдение за очередью
|
||||
|
||||
Общая сводка:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/status | python3 -m json.tool
|
||||
```
|
||||
|
||||
```json
|
||||
{
|
||||
"total_ips": 25,
|
||||
"ips_by_state": {"queued": 10, "awaiting_self_check": 1, "checking": 3, "done": 10, "failed": 1},
|
||||
"total_validators": 4
|
||||
}
|
||||
```
|
||||
|
||||
`ips_by_state` — сколько адресов в каждом состоянии прямо сейчас. Если
|
||||
хотите наблюдать за прогрессом в реальном времени:
|
||||
|
||||
```bash
|
||||
watch -n 2 'curl -s http://<control-api>:8080/api/v1/admin/status | python3 -m json.tool'
|
||||
```
|
||||
|
||||
Полный список всех адресов со всеми полями:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/ips | python3 -m json.tool
|
||||
```
|
||||
|
||||
Только финальные результаты (уже готовые адреса), с помощью `jq`:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/ips \
|
||||
| jq '[.[] | select(.State=="done" or .State=="failed") | {IPAddress, State, OverallResult}]'
|
||||
```
|
||||
|
||||
## Значения полей IP
|
||||
|
||||
| Поле | Значение |
|
||||
|---|---|
|
||||
| `IPAddress` | Проверяемый адрес |
|
||||
| `Sequence` | Позиция в очереди (порядок из конфига) |
|
||||
| `State` | Текущий этап: `queued`, `assigning_fip`, `awaiting_self_check`, `checking`, `aggregating`, `done`, `failed` |
|
||||
| `OwnerValidatorID` | Какой валидатор сейчас (или последним) занимался этим адресом |
|
||||
| `FIPID` | Идентификатор Floating IP в OpenStack, к которому привязан адрес (пусто, если ещё/уже не привязан) |
|
||||
| `AttemptNumber` | Номер попытки — растёт при каждом requeue (сбой привязки, сбой self-check, реклейм по таймауту) |
|
||||
| `RetryCount` | Сколько раз адрес уже переставлялся в очередь заново |
|
||||
| `EgressComplete` | Валидатор закончил исходящие проверки |
|
||||
| `Site1Complete` / `Site2Complete` / `Site3Complete` | Соответствующая площадка закончила входящие проверки |
|
||||
| `OverallResult` | Итог: `pass`, `partial`, `fail`, либо пусто, пока проверка не завершена |
|
||||
| `AssignedAt` / `AggregatedAt` / `FIPReleasedAt` | Метки времени соответствующих этапов |
|
||||
|
||||
## Как читать итоговый результат (pass/partial/fail)
|
||||
|
||||
- **`pass`** — прошли все проверки (все исходящие + все три площадки по
|
||||
всем портам и ICMP). Адрес можно считать пригодным к повторной выдаче.
|
||||
- **`partial`** — часть проверок прошла, часть — нет (например, площадка
|
||||
site-2 не смогла достучаться по 8080/tcp, но остальное в порядке).
|
||||
Означает частичную деградацию — например, адрес заблокирован в
|
||||
отдельном сегменте сети/у отдельного провайдера. Требует решения
|
||||
оператора: годится ли адрес для данного случая использования.
|
||||
- **`fail`** — либо ни одна проверка не прошла, либо адрес вообще не
|
||||
дошёл до стадии проверок (например, self-check не подтвердился —
|
||||
трафик валидатора не пошёл через назначенный FIP — и попытки
|
||||
исчерпались). Смотрите `events` по этому адресу (см. ниже), чтобы
|
||||
понять, на каком шаге и почему.
|
||||
|
||||
Отсутствие ответа от источника (площадка не прислала результат до
|
||||
истечения `checking_window_seconds`) засчитывается как провал — это
|
||||
управляется настройкой `aggregation.missing_counts_as_fail` в конфиге
|
||||
control-api (по умолчанию включено).
|
||||
|
||||
## Просмотр деталей и истории по конкретному адресу
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/ips/203.0.113.10 | python3 -m json.tool
|
||||
```
|
||||
|
||||
Ответ содержит три части:
|
||||
- `ip` — те же поля, что и в списке `admin/ips`, но для одного адреса;
|
||||
- `checks` — все отдельные проверки текущей попытки: кто проверял
|
||||
(`Source`: `egress` или `inbound-site-N`), что именно (`CheckType`,
|
||||
`Target`), результат (`Success`), задержка (`LatencyMS`), и
|
||||
человекочитаемая деталь (`Detail`, например текст ошибки при отказе);
|
||||
- `events` — журнал аудита по этому адресу в хронологическом порядке
|
||||
(регистрация, привязка FIP, self-check, агрегация и т.д.) — полезен,
|
||||
чтобы восстановить точную последовательность событий при разборе
|
||||
инцидента.
|
||||
|
||||
Пример: почему адрес получил `fail`?
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/ips/203.0.113.10 \
|
||||
| jq '.checks[] | select(.Success==false)'
|
||||
```
|
||||
|
||||
## Управление валидаторами
|
||||
|
||||
Список валидаторов и их текущее состояние:
|
||||
|
||||
```bash
|
||||
curl -s http://<control-api>:8080/api/v1/admin/validators | python3 -m json.tool
|
||||
```
|
||||
|
||||
Состояния валидатора: `unregistered` (в конфиге есть, агент ещё не
|
||||
подключался), `idle` (свободен, готов взять адрес), `assigned`/`checking`
|
||||
(занят), `unreachable` (пропустил heartbeat дольше
|
||||
`orchestrator.heartbeat_timeout_seconds`).
|
||||
|
||||
**Добавление нового валидатора:**
|
||||
1. Поднимите новую ВМ в сервисном проекте облака, узнайте её Neutron
|
||||
`port_id`.
|
||||
2. Добавьте запись в `validators` в `control-api.yaml`
|
||||
(`validator_id` + `os_port_id`) и перезапустите `control-api`.
|
||||
3. Разверните и запустите `validator-agent` на новой ВМ с тем же
|
||||
`validator_id` в его конфиге (см. [SETUP.md](SETUP.md#развёртывание-validator-agent-на-вм-валидаторах)).
|
||||
|
||||
**Вывод валидатора из эксплуатации:** остановите на нём
|
||||
`validator-agent` (`systemctl stop validator-agent`). Он перестанет
|
||||
получать новые задания после того, как закончит текущее (если оно было);
|
||||
если он был убит посреди работы — control-api сам заберёт у него
|
||||
незавершённый адрес обратно в очередь по истечении
|
||||
`orchestrator.lease_ttl_seconds`. Удалять запись из `control-api.yaml`
|
||||
не обязательно — просто выключенный агент не будет ничего забирать.
|
||||
|
||||
## Управление площадками (проберами)
|
||||
|
||||
Аналогично валидаторам: чтобы добавить площадку, добавьте `site_id` +
|
||||
`index` (свободный от 1 до 3, или больше — но текущая схема БД
|
||||
рассчитана ровно на 3 площадки, см. ниже) в `sites` конфига control-api,
|
||||
разверните на площадке `prober` с тем же `site_id`.
|
||||
|
||||
> Важно: количество площадок в текущей версии жёстко зашито в схему БД
|
||||
> (`Site1Complete`/`Site2Complete`/`Site3Complete`) — система рассчитана
|
||||
> ровно на **три** внешние площадки, как и описано в исходной схеме
|
||||
> процесса. Изменение их числа потребует доработки схемы данных, это не
|
||||
> делается только правкой конфига.
|
||||
|
||||
## Повторная проверка адреса
|
||||
|
||||
Если адрес завершился с `failed` или `partial`, а вы хотите перепроверить
|
||||
его ещё раз (например, после устранения блокировки на стороне сети):
|
||||
на данный момент нет отдельного API-метода "перезапустить проверку".
|
||||
Самый простой путь:
|
||||
1. Убедитесь, что адрес не находится в активном состоянии (`checking`
|
||||
и т.п.) — то есть уже `done`/`failed`.
|
||||
2. Временно уберите и снова добавьте адрес в список `ip_addresses`
|
||||
(либо просто пересоздайте запись в БД вручную, если это единичный
|
||||
случай и у вас есть доступ к SQLite) и перезапустите `control-api`.
|
||||
|
||||
Поскольку сидирование очереди идёт по уникальности `ip_address`
|
||||
(конфликт по уже существующей записи просто игнорируется), самый чистый
|
||||
способ гарантированно перепроверить конкретный адрес — обратиться к
|
||||
администратору БД (см. следующий раздел) либо дождаться штатной
|
||||
доработки API под повторные проверки.
|
||||
|
||||
## Частые проблемы и что с ними делать
|
||||
|
||||
**Валидатор долго висит в `unreachable`.**
|
||||
Проверьте сетевую связность ВМ-валидатора до `control-api` (порт из
|
||||
`server.listen_addr`) и что процесс `validator-agent` вообще запущен
|
||||
(`systemctl status validator-agent`, `journalctl -u validator-agent`).
|
||||
|
||||
**Адрес постоянно проваливает self-check.**
|
||||
Смотрите `events` по адресу (`GET /api/v1/admin/ips/{ip}`) — в детали
|
||||
события `self_check_result` будет указан обнаруженный исходящий адрес.
|
||||
Если он не совпадает с ожидаемым — вероятно, на ВМ-валидаторе есть другой
|
||||
маршрут наружу (не через назначенный Floating IP), либо привязка FIP на
|
||||
стороне OpenStack не применилась. Проверьте вручную в OpenStack
|
||||
(`openstack floating ip show <адрес>`), что `port_id` совпадает с портом
|
||||
валидатора.
|
||||
|
||||
**Площадка (`site-N`) никогда не отчитывается (`SiteNComplete` всегда
|
||||
`false`).**
|
||||
Проверьте, что `prober` на этой площадке запущен и его `site_id` в
|
||||
конфиге совпадает с `site_id` в конфиге control-api. Проверьте, что
|
||||
площадка имеет сетевой доступ и до `control-api`, и до проверяемого
|
||||
адреса (входящий трафик на 22/80/443/8080 + ICMP — это отдельная
|
||||
связность от связи с control-api, см.
|
||||
[SETUP.md](SETUP.md#сетевые-доступы)).
|
||||
|
||||
**Много адресов зависло в `checking` дольше ожидаемого.**
|
||||
Это нормально, если ещё не истёк `orchestrator.checking_window_seconds` —
|
||||
агрегация ждёт либо полного набора ответов, либо истечения окна. Если
|
||||
адрес завис заметно дольше окна — проверьте, что фоновый цикл control-api
|
||||
вообще работает (смотрите `journalctl -u control-api` на предмет ошибок
|
||||
в `sweep checking window`).
|
||||
|
||||
**Нужно посмотреть на данные "из первых рук", в обход API.**
|
||||
`control-api` использует SQLite, файл — по пути `database.path` из
|
||||
конфига. Можно (только для чтения, на **той же машине**, где крутится
|
||||
control-api) открыть его `sqlite3` в режиме WAL — это безопасно для
|
||||
чтения параллельно с работающим процессом:
|
||||
```bash
|
||||
sqlite3 /var/lib/cloud-ip-validator/control-api.db \
|
||||
"select ip_address, state, overall_result from ip_queue order by sequence"
|
||||
```
|
||||
Не редактируйте эту базу вручную во время работы control-api — это может
|
||||
рассинхронизировать состояние с реальными привязками Floating IP в
|
||||
OpenStack.
|
||||
@@ -0,0 +1,22 @@
|
||||
module cloudipvalidator
|
||||
|
||||
go 1.26.2
|
||||
|
||||
require (
|
||||
github.com/gophercloud/gophercloud/v2 v2.14.0
|
||||
golang.org/x/net v0.58.0
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
modernc.org/sqlite v1.57.0
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/dustin/go-humanize v1.0.1 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/mattn/go-isatty v0.0.24 // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
modernc.org/libc v1.74.4 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
)
|
||||
@@ -0,0 +1,58 @@
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo=
|
||||
github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/gophercloud/gophercloud/v2 v2.14.0 h1:xGxKCvyaOxJDc5FqrnKDNqtdYn43ocQPuJ2Cm4KX/cs=
|
||||
github.com/gophercloud/gophercloud/v2 v2.14.0/go.mod h1:4fs5I9VH6Wg2LyocDL9xf0ASb8VD63tyLA8sgAX/69U=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
|
||||
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
|
||||
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
|
||||
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To=
|
||||
golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU=
|
||||
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
|
||||
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI=
|
||||
modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=
|
||||
modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk=
|
||||
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
|
||||
modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU=
|
||||
modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI=
|
||||
modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito=
|
||||
modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI=
|
||||
modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
|
||||
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
|
||||
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
|
||||
modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k=
|
||||
modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co=
|
||||
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
|
||||
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
|
||||
modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI=
|
||||
modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
|
||||
modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
|
||||
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.57.0 h1:qNQP6xnx5M0ISNtlnxoOX0+cD5bJ0/gr9aMmndFczzg=
|
||||
modernc.org/sqlite v1.57.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
|
||||
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
|
||||
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
|
||||
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
|
||||
modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM=
|
||||
@@ -0,0 +1,280 @@
|
||||
// Package agentcore implements the validator-agent's poll loop: register,
|
||||
// heartbeat, wait for an assignment, self-check that egress actually flows
|
||||
// through the newly attached FIP, run the configured outbound/egress
|
||||
// checks, and report results — all driven entirely by the Control API, so
|
||||
// the process itself holds no durable state (constraint: the agent must be
|
||||
// safely restartable at any point without losing correctness, only
|
||||
// possibly re-doing in-flight work, which the Control API's idempotent
|
||||
// upserts tolerate).
|
||||
package agentcore
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/apiclient"
|
||||
"cloudipvalidator/internal/checkrunner"
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
type Agent struct {
|
||||
cfg *config.ValidatorAgent
|
||||
client *apiclient.Client
|
||||
log *slog.Logger
|
||||
|
||||
lastHandledIPID int64
|
||||
}
|
||||
|
||||
func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent {
|
||||
timeout := time.Duration(cfg.Checks.HTTPSTimeoutSeconds) * time.Second
|
||||
if timeout <= 0 {
|
||||
timeout = 10 * time.Second
|
||||
}
|
||||
return &Agent{
|
||||
cfg: cfg,
|
||||
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
|
||||
log: log,
|
||||
}
|
||||
}
|
||||
|
||||
// Run registers with the Control API and polls forever until ctx is
|
||||
// cancelled.
|
||||
func (a *Agent) Run(ctx context.Context) error {
|
||||
if err := a.register(ctx); err != nil {
|
||||
return fmt.Errorf("register: %w", err)
|
||||
}
|
||||
|
||||
interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
a.pollOnce(ctx)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-ticker.C:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type registerReq struct {
|
||||
ValidatorID string `json:"validator_id"`
|
||||
Hostname string `json:"hostname"`
|
||||
AgentVersion string `json:"agent_version"`
|
||||
}
|
||||
type registerResp struct {
|
||||
OK bool `json:"ok"`
|
||||
PollIntervalSeconds int `json:"poll_interval_seconds"`
|
||||
}
|
||||
|
||||
func (a *Agent) register(ctx context.Context) error {
|
||||
hostname, _ := hostnameOrDefault()
|
||||
var resp registerResp
|
||||
_, err := a.client.Do(ctx, "POST", "/api/v1/agents/register", registerReq{
|
||||
ValidatorID: a.cfg.ValidatorID, Hostname: hostname, AgentVersion: "dev",
|
||||
}, &resp)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
a.log.Info("registered", "validator_id", a.cfg.ValidatorID)
|
||||
return nil
|
||||
}
|
||||
|
||||
type heartbeatReq struct {
|
||||
LocalState string `json:"local_state"`
|
||||
}
|
||||
|
||||
type assignmentResp struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
Phase string `json:"phase"`
|
||||
CheckConfig []checkConfigDTO `json:"check_config"`
|
||||
}
|
||||
|
||||
type checkConfigDTO struct {
|
||||
Type string `json:"type"`
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
func (a *Agent) pollOnce(ctx context.Context) {
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: "idle"}, nil); err != nil {
|
||||
a.log.Error("heartbeat", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
var assignment assignmentResp
|
||||
ok, err := a.client.Do(ctx, "GET", "/api/v1/agents/"+a.cfg.ValidatorID+"/assignment", nil, &assignment)
|
||||
if err != nil {
|
||||
a.log.Error("get assignment", "err", err)
|
||||
return
|
||||
}
|
||||
if !ok {
|
||||
a.lastHandledIPID = 0
|
||||
return // nothing assigned right now
|
||||
}
|
||||
|
||||
if assignment.IPID == a.lastHandledIPID {
|
||||
return // already handled this IP's work this attempt
|
||||
}
|
||||
|
||||
switch assignment.Phase {
|
||||
case "awaiting_self_check":
|
||||
a.handleSelfCheckAndRun(ctx, assignment)
|
||||
case "checking":
|
||||
// Agent restarted (or a prior response was lost) after self-check
|
||||
// already succeeded server-side: just (re-)run checks, which is
|
||||
// safe since results are upserted idempotently.
|
||||
a.runChecks(ctx, assignment)
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) handleSelfCheckAndRun(ctx context.Context, assignment assignmentResp) {
|
||||
a.postEvent(ctx, assignment.IPID, "config_received", "")
|
||||
|
||||
timeout := time.Duration(a.cfg.SelfCheck.TimeoutSeconds) * time.Second
|
||||
selfCtx, cancel := context.WithTimeout(ctx, timeout)
|
||||
defer cancel()
|
||||
|
||||
var whoami struct {
|
||||
IP string `json:"ip"`
|
||||
}
|
||||
_, err := a.client.Do(selfCtx, "GET", "/api/v1/whatsmyip", nil, &whoami)
|
||||
success := err == nil && whoami.IP == assignment.IPAddress
|
||||
detail := "matched"
|
||||
if err != nil {
|
||||
detail = "whatsmyip request failed: " + err.Error()
|
||||
} else if !success {
|
||||
detail = fmt.Sprintf("egress ip %q does not match assigned fip %q", whoami.IP, assignment.IPAddress)
|
||||
}
|
||||
|
||||
a.postSelfCheck(ctx, assignment.IPID, whoami.IP, success, detail)
|
||||
a.postEvent(ctx, assignment.IPID, "self_check_result", fmt.Sprintf(`{"success":%t}`, success))
|
||||
|
||||
if !success {
|
||||
a.log.Warn("self-check failed", "ip", assignment.IPAddress, "detail", detail)
|
||||
return
|
||||
}
|
||||
a.runChecks(ctx, assignment)
|
||||
}
|
||||
|
||||
func (a *Agent) runChecks(ctx context.Context, assignment assignmentResp) {
|
||||
var results []checkResultDTO
|
||||
for _, ct := range assignment.CheckConfig {
|
||||
for _, target := range ct.Targets {
|
||||
var fn func(context.Context) checkrunner.Result
|
||||
switch ct.Type {
|
||||
case "https":
|
||||
fn = checkrunner.HTTPS(target, time.Duration(a.cfg.Checks.HTTPSTimeoutSeconds)*time.Second)
|
||||
case "icmp":
|
||||
fn = checkrunner.ICMPEcho(hostOnly(target), a.cfg.Checks.ICMPCount, time.Duration(a.cfg.Checks.ICMPTimeoutSeconds)*time.Second)
|
||||
case "ssh":
|
||||
if !a.cfg.Checks.SSH.Enabled {
|
||||
continue
|
||||
}
|
||||
fn = checkrunner.SSHBanner(hostOnly(target), time.Duration(a.cfg.Checks.SSH.TimeoutSeconds)*time.Second)
|
||||
default:
|
||||
a.log.Warn("unknown check type", "type", ct.Type)
|
||||
continue
|
||||
}
|
||||
res := fn(ctx)
|
||||
// Record the originally configured target (a full URL for
|
||||
// https, e.g.), not checkrunner's internal host-only value
|
||||
// used for icmp/ssh — otherwise two configured targets that
|
||||
// happen to share a bare host (as can occur, e.g., in the
|
||||
// loopback-only local e2e harness) would collide on the
|
||||
// checks table's UNIQUE(ip_id, attempt, source, type,
|
||||
// target) key and silently overwrite each other.
|
||||
results = append(results, checkResultDTO{
|
||||
IPID: assignment.IPID, CheckType: res.CheckType, Target: target,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail,
|
||||
CheckedAt: res.CheckedAt.Format(time.RFC3339Nano),
|
||||
})
|
||||
// Report progressively rather than batching until the end, so
|
||||
// a crash mid-run doesn't lose already-completed check results.
|
||||
a.postResults(ctx, []checkResultDTO{results[len(results)-1]})
|
||||
}
|
||||
}
|
||||
|
||||
a.postComplete(ctx, assignment.IPID)
|
||||
a.lastHandledIPID = assignment.IPID
|
||||
}
|
||||
|
||||
type checkResultDTO struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
CheckType string `json:"check_type"`
|
||||
Target string `json:"target,omitempty"`
|
||||
Success bool `json:"success"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
CheckedAt string `json:"checked_at"`
|
||||
}
|
||||
|
||||
func (a *Agent) postResults(ctx context.Context, results []checkResultDTO) {
|
||||
body := struct {
|
||||
Results []checkResultDTO `json:"results"`
|
||||
}{results}
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/results", body, nil); err != nil {
|
||||
a.log.Error("post results", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) postComplete(ctx context.Context, ipID int64) {
|
||||
body := struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
}{ipID}
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/complete", body, nil); err != nil {
|
||||
a.log.Error("post complete", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) postSelfCheck(ctx context.Context, ipID int64, detected string, success bool, detail string) {
|
||||
body := struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
DetectedEgress string `json:"detected_egress_ip"`
|
||||
Success bool `json:"success"`
|
||||
Detail string `json:"detail"`
|
||||
}{ipID, detected, success, detail}
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/self-check", body, nil); err != nil {
|
||||
a.log.Error("post self-check", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) postEvent(ctx context.Context, ipID int64, eventType, payload string) {
|
||||
body := struct {
|
||||
EventType string `json:"event_type"`
|
||||
IPID int64 `json:"ip_id"`
|
||||
Payload string `json:"payload"`
|
||||
}{eventType, ipID, payload}
|
||||
if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/events", body, nil); err != nil {
|
||||
a.log.Error("post event", "type", eventType, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
func hostnameOrDefault() (string, error) {
|
||||
h, err := os.Hostname()
|
||||
if err != nil || h == "" {
|
||||
return "unknown", err
|
||||
}
|
||||
return h, nil
|
||||
}
|
||||
|
||||
// hostOnly strips a URL scheme (https://) from a target, since ICMP/SSH
|
||||
// checks operate on bare hostnames while HTTPS checks take a full URL.
|
||||
func hostOnly(target string) string {
|
||||
for _, prefix := range []string{"https://", "http://"} {
|
||||
if len(target) > len(prefix) && target[:len(prefix)] == prefix {
|
||||
target = target[len(prefix):]
|
||||
break
|
||||
}
|
||||
}
|
||||
for i := 0; i < len(target); i++ {
|
||||
if target[i] == '/' || target[i] == ':' {
|
||||
return target[:i]
|
||||
}
|
||||
}
|
||||
return target
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
// Package apiclient is a thin HTTP client for the Control API's /api/v1
|
||||
// surface, shared by the validator-agent and prober binaries. Neither
|
||||
// binary talks to the database directly — this is their only channel to
|
||||
// shared state, keeping both genuinely stateless.
|
||||
package apiclient
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"time"
|
||||
)
|
||||
|
||||
type Client struct {
|
||||
BaseURL string
|
||||
HTTPClient *http.Client
|
||||
}
|
||||
|
||||
func New(baseURL string, timeout time.Duration) *Client {
|
||||
return &Client{BaseURL: baseURL, HTTPClient: &http.Client{Timeout: timeout}}
|
||||
}
|
||||
|
||||
// Do issues a JSON request and decodes a JSON response into out (if
|
||||
// non-nil). A 204 response is treated as "no content" and out is left
|
||||
// untouched, with ok=false — used for the assignment poll's empty case.
|
||||
func (c *Client) Do(ctx context.Context, method, path string, body, out interface{}) (ok bool, err error) {
|
||||
var reader io.Reader
|
||||
if body != nil {
|
||||
b, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("marshal request: %w", err)
|
||||
}
|
||||
reader = bytes.NewReader(b)
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, method, c.BaseURL+path, reader)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("build request: %w", err)
|
||||
}
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
|
||||
resp, err := c.HTTPClient.Do(req)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("%s %s: %w", method, path, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode == http.StatusNoContent {
|
||||
return false, nil
|
||||
}
|
||||
respBody, _ := io.ReadAll(resp.Body)
|
||||
if resp.StatusCode >= 300 {
|
||||
return false, fmt.Errorf("%s %s: status %d: %s", method, path, resp.StatusCode, string(respBody))
|
||||
}
|
||||
if out != nil && len(respBody) > 0 {
|
||||
if err := json.Unmarshal(respBody, out); err != nil {
|
||||
return false, fmt.Errorf("%s %s: decode response: %w", method, path, err)
|
||||
}
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
package checkrunner
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
)
|
||||
|
||||
// HTTPS performs a GET against target (expected to be a full URL, e.g.
|
||||
// https://github.com) and reports success for any response with status
|
||||
// < 400. It never follows the check to a different host on a check-type
|
||||
// boundary — redirects within the same request are followed by the
|
||||
// standard http.Client default policy, which is what we want for
|
||||
// reachability checks.
|
||||
func HTTPS(target string, timeout time.Duration) func(ctx context.Context) Result {
|
||||
return run("https", target, func(ctx context.Context) error {
|
||||
ctx, cancel := context.WithTimeout(ctx, timeout)
|
||||
defer cancel()
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, target, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("build request: %w", err)
|
||||
}
|
||||
client := &http.Client{Timeout: timeout}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode >= 400 {
|
||||
return fmt.Errorf("unexpected status %d", resp.StatusCode)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package checkrunner
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"golang.org/x/net/icmp"
|
||||
"golang.org/x/net/ipv4"
|
||||
)
|
||||
|
||||
// ICMPEcho sends up to `count` ICMP echo requests to host and reports
|
||||
// success if at least one echo reply is received before timeout. Uses a
|
||||
// raw ICMP socket (ip4:icmp), which requires either running as root or
|
||||
// (on Linux, as deployed here) the CAP_NET_RAW capability — see the
|
||||
// validator-agent and prober systemd units.
|
||||
func ICMPEcho(host string, count int, timeout time.Duration) func(ctx context.Context) Result {
|
||||
return run("icmp", host, func(ctx context.Context) error {
|
||||
conn, err := icmp.ListenPacket("ip4:icmp", "0.0.0.0")
|
||||
if err != nil {
|
||||
return fmt.Errorf("open icmp socket (needs CAP_NET_RAW or root): %w", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
dst, err := net.ResolveIPAddr("ip4", host)
|
||||
if err != nil {
|
||||
return fmt.Errorf("resolve %s: %w", host, err)
|
||||
}
|
||||
|
||||
id := os.Getpid() & 0xffff
|
||||
var lastErr error
|
||||
for seq := 1; seq <= count; seq++ {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
msg := icmp.Message{
|
||||
Type: ipv4.ICMPTypeEcho,
|
||||
Code: 0,
|
||||
Body: &icmp.Echo{
|
||||
ID: id,
|
||||
Seq: seq,
|
||||
Data: []byte("cloud-ip-validator"),
|
||||
},
|
||||
}
|
||||
wb, err := msg.Marshal(nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal echo request: %w", err)
|
||||
}
|
||||
if _, err := conn.WriteTo(wb, dst); err != nil {
|
||||
lastErr = fmt.Errorf("write echo request: %w", err)
|
||||
continue
|
||||
}
|
||||
|
||||
perAttempt := timeout / time.Duration(count)
|
||||
if perAttempt <= 0 {
|
||||
perAttempt = timeout
|
||||
}
|
||||
conn.SetReadDeadline(time.Now().Add(perAttempt))
|
||||
rb := make([]byte, 1500)
|
||||
n, _, err := conn.ReadFrom(rb)
|
||||
if err != nil {
|
||||
lastErr = fmt.Errorf("read echo reply: %w", err)
|
||||
continue
|
||||
}
|
||||
rm, err := icmp.ParseMessage(1 /* protocolICMP */, rb[:n])
|
||||
if err != nil {
|
||||
lastErr = fmt.Errorf("parse reply: %w", err)
|
||||
continue
|
||||
}
|
||||
if rm.Type == ipv4.ICMPTypeEchoReply {
|
||||
return nil
|
||||
}
|
||||
lastErr = fmt.Errorf("unexpected icmp type %v", rm.Type)
|
||||
}
|
||||
if lastErr == nil {
|
||||
lastErr = fmt.Errorf("no reply received")
|
||||
}
|
||||
return lastErr
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
package checkrunner
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TCPConnect reports success if a TCP handshake against host:port
|
||||
// completes within timeout. This is the primitive behind the prober's
|
||||
// per-port inbound reachability checks (22/80/443/8080).
|
||||
func TCPConnect(host string, port int, timeout time.Duration) func(ctx context.Context) Result {
|
||||
target := net.JoinHostPort(host, strconv.Itoa(port))
|
||||
checkType := "tcp-" + strconv.Itoa(port)
|
||||
return run(checkType, target, func(ctx context.Context) error {
|
||||
d := net.Dialer{Timeout: timeout}
|
||||
conn, err := d.DialContext(ctx, "tcp", target)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return conn.Close()
|
||||
})
|
||||
}
|
||||
|
||||
// SSHBanner performs a TCP connect to host:22 and additionally verifies the
|
||||
// remote sends an "SSH-2.0-" banner, without performing any auth handshake.
|
||||
// Used for the optional ssh check type.
|
||||
func SSHBanner(host string, timeout time.Duration) func(ctx context.Context) Result {
|
||||
target := net.JoinHostPort(host, "22")
|
||||
return run("ssh", target, func(ctx context.Context) error {
|
||||
d := net.Dialer{Timeout: timeout}
|
||||
conn, err := d.DialContext(ctx, "tcp", target)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
conn.SetReadDeadline(time.Now().Add(timeout))
|
||||
buf := make([]byte, 8)
|
||||
n, err := conn.Read(buf)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read banner: %w", err)
|
||||
}
|
||||
if n < 8 || string(buf[:8]) != "SSH-2.0-" {
|
||||
return fmt.Errorf("unexpected banner prefix %q", string(buf[:n]))
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
// Package checkrunner implements the actual network probes (HTTPS, TCP
|
||||
// connect, ICMP echo, SSH banner) shared by both the validator-agent
|
||||
// (outbound/egress checks) and the prober (inbound/reachability checks).
|
||||
// The two binaries use the same primitives against different targets and
|
||||
// in different directions, but never share process state.
|
||||
package checkrunner
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Result is the outcome of a single check, in a form that maps directly
|
||||
// onto a db.Check row (minus the fields the caller already knows: ip_id,
|
||||
// attempt_number, validator_id, source).
|
||||
type Result struct {
|
||||
CheckType string
|
||||
Target string
|
||||
Success bool
|
||||
LatencyMS int64
|
||||
Detail string
|
||||
CheckedAt time.Time
|
||||
}
|
||||
|
||||
func run(checkType, target string, fn func(ctx context.Context) error) func(ctx context.Context) Result {
|
||||
return func(ctx context.Context) Result {
|
||||
start := time.Now()
|
||||
err := fn(ctx)
|
||||
latency := time.Since(start).Milliseconds()
|
||||
res := Result{
|
||||
CheckType: checkType,
|
||||
Target: target,
|
||||
Success: err == nil,
|
||||
LatencyMS: latency,
|
||||
CheckedAt: time.Now().UTC(),
|
||||
}
|
||||
if err != nil {
|
||||
res.Detail = err.Error()
|
||||
} else {
|
||||
res.Detail = "ok"
|
||||
}
|
||||
return res
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,236 @@
|
||||
// Package config defines the YAML configuration structures for all three
|
||||
// binaries and loads them from disk. OpenStack credentials are deliberately
|
||||
// never part of these structs — only the *names* of environment variables
|
||||
// to read them from — so secrets never land in a config file on disk.
|
||||
package config
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// ---- control-api ----
|
||||
|
||||
type ControlAPI struct {
|
||||
Server ServerConfig `yaml:"server"`
|
||||
Database DatabaseConfig `yaml:"database"`
|
||||
OpenStack OpenStackConfig `yaml:"openstack"`
|
||||
Orchestrator OrchestratorConfig `yaml:"orchestrator"`
|
||||
Aggregation AggregationConfig `yaml:"aggregation"`
|
||||
Validators []ValidatorConfig `yaml:"validators"`
|
||||
Sites []SiteConfig `yaml:"sites"`
|
||||
CheckTypes []CheckTypeConfig `yaml:"check_types"`
|
||||
Targets map[string][]string `yaml:"targets"`
|
||||
Inbound InboundConfig `yaml:"inbound_checks"`
|
||||
IPAddresses []string `yaml:"ip_addresses"`
|
||||
}
|
||||
|
||||
type ServerConfig struct {
|
||||
ListenAddr string `yaml:"listen_addr"`
|
||||
}
|
||||
|
||||
type DatabaseConfig struct {
|
||||
Path string `yaml:"path"`
|
||||
}
|
||||
|
||||
// OpenStackConfig names the environment variables control-api reads its
|
||||
// OpenStack admin credential from at startup. Mode "mock" skips all of this
|
||||
// and uses an in-memory FloatingIPClient instead — used for local dev and
|
||||
// the offline end-to-end harness.
|
||||
type OpenStackConfig struct {
|
||||
Mode string `yaml:"mode"` // "mock" | "real"
|
||||
AuthURLEnv string `yaml:"auth_url_env"`
|
||||
TokenEnv string `yaml:"token_env"`
|
||||
ProjectIDEnv string `yaml:"project_id_env"`
|
||||
ProjectNameEnv string `yaml:"project_name_env"`
|
||||
ProjectDomainEnv string `yaml:"project_domain_env"`
|
||||
RegionEnv string `yaml:"region_env"`
|
||||
}
|
||||
|
||||
type OrchestratorConfig struct {
|
||||
PollIntervalSeconds int `yaml:"poll_interval_seconds"`
|
||||
SelfCheckTimeoutSeconds int `yaml:"self_check_timeout_seconds"`
|
||||
MaxSelfCheckRetries int `yaml:"max_self_check_retries"`
|
||||
CheckingWindowSeconds int `yaml:"checking_window_seconds"`
|
||||
MaxRetries int `yaml:"max_retries"`
|
||||
LeaseTTLSeconds int `yaml:"lease_ttl_seconds"`
|
||||
HeartbeatTimeoutSeconds int `yaml:"heartbeat_timeout_seconds"`
|
||||
}
|
||||
|
||||
type AggregationConfig struct {
|
||||
MissingCountsAsFail bool `yaml:"missing_counts_as_fail"`
|
||||
}
|
||||
|
||||
type ValidatorConfig struct {
|
||||
ValidatorID string `yaml:"validator_id"`
|
||||
OSPortID string `yaml:"os_port_id"`
|
||||
}
|
||||
|
||||
type SiteConfig struct {
|
||||
SiteID string `yaml:"site_id"`
|
||||
Index int `yaml:"index"` // 1, 2, or 3 — maps to ip_queue.siteN_complete
|
||||
}
|
||||
|
||||
// CheckTypeConfig maps a check type (https, icmp, ssh) to the named target
|
||||
// groups (keys into ControlAPI.Targets) it should run against. This drives
|
||||
// the egress/outbound checks the validator-agent performs.
|
||||
type CheckTypeConfig struct {
|
||||
Name string `yaml:"name"`
|
||||
Enabled bool `yaml:"enabled"`
|
||||
Targets []string `yaml:"targets"`
|
||||
}
|
||||
|
||||
type InboundConfig struct {
|
||||
Ports []int `yaml:"ports"`
|
||||
ICMP bool `yaml:"icmp"`
|
||||
}
|
||||
|
||||
func LoadControlAPI(path string) (*ControlAPI, error) {
|
||||
var c ControlAPI
|
||||
if err := loadYAML(path, &c); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.Server.ListenAddr == "" {
|
||||
c.Server.ListenAddr = ":8080"
|
||||
}
|
||||
if c.Database.Path == "" {
|
||||
c.Database.Path = "control-api.db"
|
||||
}
|
||||
if c.OpenStack.Mode == "" {
|
||||
c.OpenStack.Mode = "mock"
|
||||
}
|
||||
if c.Orchestrator.PollIntervalSeconds == 0 {
|
||||
c.Orchestrator.PollIntervalSeconds = 5
|
||||
}
|
||||
if c.Orchestrator.SelfCheckTimeoutSeconds == 0 {
|
||||
c.Orchestrator.SelfCheckTimeoutSeconds = 60
|
||||
}
|
||||
if c.Orchestrator.MaxSelfCheckRetries == 0 {
|
||||
c.Orchestrator.MaxSelfCheckRetries = 3
|
||||
}
|
||||
if c.Orchestrator.CheckingWindowSeconds == 0 {
|
||||
c.Orchestrator.CheckingWindowSeconds = 120
|
||||
}
|
||||
if c.Orchestrator.MaxRetries == 0 {
|
||||
c.Orchestrator.MaxRetries = 3
|
||||
}
|
||||
if c.Orchestrator.LeaseTTLSeconds == 0 {
|
||||
c.Orchestrator.LeaseTTLSeconds = 180
|
||||
}
|
||||
if c.Orchestrator.HeartbeatTimeoutSeconds == 0 {
|
||||
c.Orchestrator.HeartbeatTimeoutSeconds = 30
|
||||
}
|
||||
return &c, nil
|
||||
}
|
||||
|
||||
// ---- validator-agent ----
|
||||
|
||||
type ValidatorAgent struct {
|
||||
ValidatorID string `yaml:"validator_id"`
|
||||
ControlAPIURL string `yaml:"control_api_url"`
|
||||
PollIntervalSeconds int `yaml:"poll_interval_seconds"`
|
||||
SelfCheck SelfCheckCfg `yaml:"self_check"`
|
||||
Checks AgentChecks `yaml:"checks"`
|
||||
}
|
||||
|
||||
type SelfCheckCfg struct {
|
||||
TimeoutSeconds int `yaml:"timeout_seconds"`
|
||||
}
|
||||
|
||||
type AgentChecks struct {
|
||||
HTTPSTimeoutSeconds int `yaml:"https_timeout_seconds"`
|
||||
ICMPTimeoutSeconds int `yaml:"icmp_timeout_seconds"`
|
||||
ICMPCount int `yaml:"icmp_count"`
|
||||
SSH SSHCfg `yaml:"ssh"`
|
||||
}
|
||||
|
||||
type SSHCfg struct {
|
||||
Enabled bool `yaml:"enabled"`
|
||||
TimeoutSeconds int `yaml:"timeout_seconds"`
|
||||
}
|
||||
|
||||
func LoadValidatorAgent(path string) (*ValidatorAgent, error) {
|
||||
var c ValidatorAgent
|
||||
if err := loadYAML(path, &c); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.PollIntervalSeconds == 0 {
|
||||
c.PollIntervalSeconds = 5
|
||||
}
|
||||
if c.SelfCheck.TimeoutSeconds == 0 {
|
||||
c.SelfCheck.TimeoutSeconds = 10
|
||||
}
|
||||
if c.Checks.HTTPSTimeoutSeconds == 0 {
|
||||
c.Checks.HTTPSTimeoutSeconds = 10
|
||||
}
|
||||
if c.Checks.ICMPTimeoutSeconds == 0 {
|
||||
c.Checks.ICMPTimeoutSeconds = 5
|
||||
}
|
||||
if c.Checks.ICMPCount == 0 {
|
||||
c.Checks.ICMPCount = 3
|
||||
}
|
||||
if c.Checks.SSH.TimeoutSeconds == 0 {
|
||||
c.Checks.SSH.TimeoutSeconds = 5
|
||||
}
|
||||
if c.ValidatorID == "" {
|
||||
return nil, fmt.Errorf("validator_id is required")
|
||||
}
|
||||
if c.ControlAPIURL == "" {
|
||||
return nil, fmt.Errorf("control_api_url is required")
|
||||
}
|
||||
return &c, nil
|
||||
}
|
||||
|
||||
// ---- prober ----
|
||||
|
||||
type Prober struct {
|
||||
SiteID string `yaml:"site_id"`
|
||||
ControlAPIURL string `yaml:"control_api_url"`
|
||||
PollIntervalSeconds int `yaml:"poll_interval_seconds"`
|
||||
Checks ProberChecks `yaml:"checks"`
|
||||
}
|
||||
|
||||
type ProberChecks struct {
|
||||
TCPTimeoutSeconds int `yaml:"tcp_timeout_seconds"`
|
||||
ICMPTimeoutSeconds int `yaml:"icmp_timeout_seconds"`
|
||||
ICMPCount int `yaml:"icmp_count"`
|
||||
}
|
||||
|
||||
func LoadProber(path string) (*Prober, error) {
|
||||
var c Prober
|
||||
if err := loadYAML(path, &c); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.PollIntervalSeconds == 0 {
|
||||
c.PollIntervalSeconds = 5
|
||||
}
|
||||
if c.Checks.TCPTimeoutSeconds == 0 {
|
||||
c.Checks.TCPTimeoutSeconds = 5
|
||||
}
|
||||
if c.Checks.ICMPTimeoutSeconds == 0 {
|
||||
c.Checks.ICMPTimeoutSeconds = 5
|
||||
}
|
||||
if c.Checks.ICMPCount == 0 {
|
||||
c.Checks.ICMPCount = 3
|
||||
}
|
||||
if c.SiteID == "" {
|
||||
return nil, fmt.Errorf("site_id is required")
|
||||
}
|
||||
if c.ControlAPIURL == "" {
|
||||
return nil, fmt.Errorf("control_api_url is required")
|
||||
}
|
||||
return &c, nil
|
||||
}
|
||||
|
||||
func loadYAML(path string, out interface{}) error {
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read config %s: %w", path, err)
|
||||
}
|
||||
if err := yaml.Unmarshal(data, out); err != nil {
|
||||
return fmt.Errorf("parse config %s: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
// Package db owns the SQLite connection, schema migrations, and all queries
|
||||
// used by the Control API. It is the only package in the system that talks
|
||||
// to the database directly — agents and probers never connect to it.
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
_ "embed"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
//go:embed migrations/0001_init.sql
|
||||
var initSchema string
|
||||
|
||||
type DB struct {
|
||||
*sql.DB
|
||||
}
|
||||
|
||||
// Open opens (creating if necessary) the SQLite database at path, applies
|
||||
// pragmas suited to a single-writer WAL workload, and runs any pending
|
||||
// schema migrations.
|
||||
func Open(ctx context.Context, path string) (*DB, error) {
|
||||
sqlDB, err := sql.Open("sqlite", path+"?_pragma=busy_timeout(5000)")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open sqlite: %w", err)
|
||||
}
|
||||
// Control API is the sole writer; one connection avoids SQLITE_BUSY
|
||||
// entirely for writes while still allowing concurrent reads via WAL.
|
||||
sqlDB.SetMaxOpenConns(1)
|
||||
|
||||
for _, pragma := range []string{
|
||||
"PRAGMA journal_mode=WAL",
|
||||
"PRAGMA synchronous=NORMAL",
|
||||
"PRAGMA foreign_keys=ON",
|
||||
"PRAGMA busy_timeout=5000",
|
||||
} {
|
||||
if _, err := sqlDB.ExecContext(ctx, pragma); err != nil {
|
||||
sqlDB.Close()
|
||||
return nil, fmt.Errorf("apply pragma %q: %w", pragma, err)
|
||||
}
|
||||
}
|
||||
|
||||
d := &DB{DB: sqlDB}
|
||||
if err := d.migrate(ctx); err != nil {
|
||||
sqlDB.Close()
|
||||
return nil, fmt.Errorf("migrate: %w", err)
|
||||
}
|
||||
return d, nil
|
||||
}
|
||||
|
||||
// migrate applies the embedded schema exactly once, tracked via
|
||||
// PRAGMA user_version so repeated startups are no-ops.
|
||||
func (d *DB) migrate(ctx context.Context) error {
|
||||
var version int
|
||||
if err := d.QueryRowContext(ctx, "PRAGMA user_version").Scan(&version); err != nil {
|
||||
return fmt.Errorf("read user_version: %w", err)
|
||||
}
|
||||
if version >= 1 {
|
||||
return nil
|
||||
}
|
||||
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
if _, err := tx.ExecContext(ctx, initSchema); err != nil {
|
||||
return fmt.Errorf("apply 0001_init.sql: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, "PRAGMA user_version=1"); err != nil {
|
||||
return fmt.Errorf("set user_version: %w", err)
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// Now returns the current time truncated to millisecond precision, the
|
||||
// granularity used consistently for all timestamp columns.
|
||||
func Now() time.Time {
|
||||
return time.Now().UTC().Truncate(time.Millisecond)
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
CREATE TABLE validators (
|
||||
validator_id TEXT PRIMARY KEY,
|
||||
hostname TEXT NOT NULL DEFAULT '',
|
||||
os_port_id TEXT NOT NULL DEFAULT '',
|
||||
state TEXT NOT NULL DEFAULT 'unregistered',
|
||||
current_ip_id INTEGER REFERENCES ip_queue(id),
|
||||
agent_version TEXT NOT NULL DEFAULT '',
|
||||
last_heartbeat_at TIMESTAMP,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
CREATE TABLE ip_queue (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
ip_address TEXT NOT NULL UNIQUE,
|
||||
sequence INTEGER NOT NULL,
|
||||
state TEXT NOT NULL DEFAULT 'queued',
|
||||
owner_validator_id TEXT REFERENCES validators(validator_id),
|
||||
fip_id TEXT NOT NULL DEFAULT '',
|
||||
attempt_number INTEGER NOT NULL DEFAULT 1,
|
||||
retry_count INTEGER NOT NULL DEFAULT 0,
|
||||
lease_expires_at TIMESTAMP,
|
||||
egress_complete BOOLEAN NOT NULL DEFAULT 0,
|
||||
site1_complete BOOLEAN NOT NULL DEFAULT 0,
|
||||
site2_complete BOOLEAN NOT NULL DEFAULT 0,
|
||||
site3_complete BOOLEAN NOT NULL DEFAULT 0,
|
||||
overall_result TEXT NOT NULL DEFAULT '',
|
||||
assigned_at TIMESTAMP,
|
||||
aggregated_at TIMESTAMP,
|
||||
fip_released_at TIMESTAMP,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
CREATE INDEX idx_ip_queue_state_seq ON ip_queue(state, sequence);
|
||||
|
||||
CREATE TABLE checks (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
ip_id INTEGER NOT NULL REFERENCES ip_queue(id),
|
||||
ip_address TEXT NOT NULL,
|
||||
attempt_number INTEGER NOT NULL,
|
||||
validator_id TEXT NOT NULL DEFAULT '',
|
||||
source TEXT NOT NULL,
|
||||
check_type TEXT NOT NULL,
|
||||
target TEXT NOT NULL DEFAULT '',
|
||||
success BOOLEAN NOT NULL,
|
||||
latency_ms INTEGER NOT NULL DEFAULT 0,
|
||||
detail TEXT NOT NULL DEFAULT '',
|
||||
checked_at TIMESTAMP NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
UNIQUE(ip_id, attempt_number, source, check_type, target)
|
||||
);
|
||||
CREATE INDEX idx_checks_ip_attempt ON checks(ip_id, attempt_number);
|
||||
|
||||
CREATE TABLE events (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
source_type TEXT NOT NULL,
|
||||
source_id TEXT NOT NULL DEFAULT '',
|
||||
ip_id INTEGER REFERENCES ip_queue(id),
|
||||
event_type TEXT NOT NULL,
|
||||
payload TEXT NOT NULL DEFAULT '',
|
||||
occurred_at TIMESTAMP NOT NULL,
|
||||
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
CREATE INDEX idx_events_ip ON events(ip_id);
|
||||
CREATE INDEX idx_events_source ON events(source_type, source_id);
|
||||
@@ -0,0 +1,117 @@
|
||||
package db
|
||||
|
||||
import "time"
|
||||
|
||||
// Validator and IP lifecycle states. Kept as typed string constants rather
|
||||
// than a Go enum type so they round-trip through SQLite TEXT columns and
|
||||
// JSON without conversion.
|
||||
const (
|
||||
ValidatorUnregistered = "unregistered"
|
||||
ValidatorIdle = "idle"
|
||||
ValidatorAssigned = "assigned"
|
||||
ValidatorChecking = "checking"
|
||||
ValidatorUnreachable = "unreachable"
|
||||
|
||||
IPQueued = "queued"
|
||||
IPAssigningFIP = "assigning_fip"
|
||||
IPAwaitingSelfCheck = "awaiting_self_check"
|
||||
IPChecking = "checking"
|
||||
IPAggregating = "aggregating"
|
||||
IPDone = "done"
|
||||
IPFailed = "failed"
|
||||
|
||||
ResultPass = "pass"
|
||||
ResultPartial = "partial"
|
||||
ResultFail = "fail"
|
||||
|
||||
SourceEgress = "egress"
|
||||
)
|
||||
|
||||
// InboundSource returns the checks.source value for the given prober site
|
||||
// index (1-based), e.g. InboundSource(1) == "inbound-site-1".
|
||||
func InboundSource(siteIndex int) string {
|
||||
return "inbound-site-" + itoa(siteIndex)
|
||||
}
|
||||
|
||||
func itoa(n int) string {
|
||||
if n == 0 {
|
||||
return "0"
|
||||
}
|
||||
neg := n < 0
|
||||
if neg {
|
||||
n = -n
|
||||
}
|
||||
var buf [20]byte
|
||||
i := len(buf)
|
||||
for n > 0 {
|
||||
i--
|
||||
buf[i] = byte('0' + n%10)
|
||||
n /= 10
|
||||
}
|
||||
if neg {
|
||||
i--
|
||||
buf[i] = '-'
|
||||
}
|
||||
return string(buf[i:])
|
||||
}
|
||||
|
||||
type Validator struct {
|
||||
ValidatorID string
|
||||
Hostname string
|
||||
OSPortID string
|
||||
State string
|
||||
CurrentIPID *int64
|
||||
AgentVersion string
|
||||
LastHeartbeatAt *time.Time
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
type IPQueueItem struct {
|
||||
ID int64
|
||||
IPAddress string
|
||||
Sequence int
|
||||
State string
|
||||
OwnerValidatorID *string
|
||||
FIPID string
|
||||
AttemptNumber int
|
||||
RetryCount int
|
||||
LeaseExpiresAt *time.Time
|
||||
EgressComplete bool
|
||||
Site1Complete bool
|
||||
Site2Complete bool
|
||||
Site3Complete bool
|
||||
OverallResult string
|
||||
AssignedAt *time.Time
|
||||
AggregatedAt *time.Time
|
||||
FIPReleasedAt *time.Time
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
type Check struct {
|
||||
ID int64
|
||||
IPID int64
|
||||
IPAddress string
|
||||
AttemptNumber int
|
||||
ValidatorID string
|
||||
Source string
|
||||
CheckType string
|
||||
Target string
|
||||
Success bool
|
||||
LatencyMS int64
|
||||
Detail string
|
||||
CheckedAt time.Time
|
||||
CreatedAt time.Time
|
||||
}
|
||||
|
||||
type Event struct {
|
||||
ID int64
|
||||
SourceType string
|
||||
SourceID string
|
||||
IPID *int64
|
||||
EventType string
|
||||
Payload string
|
||||
OccurredAt time.Time
|
||||
CreatedAt time.Time
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
)
|
||||
|
||||
// UpsertCheck records (or, on retry, overwrites) a single check result. The
|
||||
// UNIQUE(ip_id, attempt_number, source, check_type, target) constraint plus
|
||||
// this upsert is what makes agent/prober result submission safely
|
||||
// retryable without producing duplicate rows.
|
||||
func (d *DB) UpsertCheck(ctx context.Context, c Check) error {
|
||||
_, err := d.ExecContext(ctx, `
|
||||
INSERT INTO checks (ip_id, ip_address, attempt_number, validator_id, source, check_type, target,
|
||||
success, latency_ms, detail, checked_at, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_id, attempt_number, source, check_type, target) DO UPDATE SET
|
||||
validator_id=excluded.validator_id,
|
||||
success=excluded.success,
|
||||
latency_ms=excluded.latency_ms,
|
||||
detail=excluded.detail,
|
||||
checked_at=excluded.checked_at
|
||||
`, c.IPID, c.IPAddress, c.AttemptNumber, c.ValidatorID, c.Source, c.CheckType, c.Target,
|
||||
c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), timeToDB(Now()))
|
||||
return err
|
||||
}
|
||||
|
||||
// ListChecksForAttempt returns every check recorded for an IP's current
|
||||
// attempt — the input to overall-result aggregation.
|
||||
func (d *DB) ListChecksForAttempt(ctx context.Context, ipID int64, attemptNumber int) ([]Check, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target,
|
||||
success, latency_ms, detail, checked_at, created_at
|
||||
FROM checks WHERE ip_id=? AND attempt_number=?
|
||||
ORDER BY source, check_type, target
|
||||
`, ipID, attemptNumber)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []Check
|
||||
for rows.Next() {
|
||||
var c Check
|
||||
var checkedAt, createdAt string
|
||||
if err := rows.Scan(&c.ID, &c.IPID, &c.IPAddress, &c.AttemptNumber, &c.ValidatorID, &c.Source,
|
||||
&c.CheckType, &c.Target, &c.Success, &c.LatencyMS, &c.Detail, &checkedAt, &createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.CheckedAt, err = dbToTime(checkedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if c.CreatedAt, err = dbToTime(createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, c)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
package db
|
||||
|
||||
import "context"
|
||||
|
||||
func (d *DB) InsertEvent(ctx context.Context, e Event) error {
|
||||
_, err := d.ExecContext(ctx, `
|
||||
INSERT INTO events (source_type, source_id, ip_id, event_type, payload, occurred_at, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
`, e.SourceType, e.SourceID, e.IPID, e.EventType, e.Payload, timeToDB(e.OccurredAt), timeToDB(Now()))
|
||||
return err
|
||||
}
|
||||
|
||||
// ListEventsForIP returns the audit trail for a single IP, most recent
|
||||
// first — used by the admin detail endpoint.
|
||||
func (d *DB) ListEventsForIP(ctx context.Context, ipID int64) ([]Event, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT id, source_type, source_id, ip_id, event_type, payload, occurred_at, created_at
|
||||
FROM events WHERE ip_id=? ORDER BY occurred_at DESC
|
||||
`, ipID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanEvents(rows)
|
||||
}
|
||||
|
||||
func (d *DB) ListRecentEvents(ctx context.Context, limit int) ([]Event, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT id, source_type, source_id, ip_id, event_type, payload, occurred_at, created_at
|
||||
FROM events ORDER BY id DESC LIMIT ?
|
||||
`, limit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanEvents(rows)
|
||||
}
|
||||
|
||||
func scanEvents(rows interface {
|
||||
Next() bool
|
||||
Scan(...interface{}) error
|
||||
Err() error
|
||||
}) ([]Event, error) {
|
||||
var out []Event
|
||||
for rows.Next() {
|
||||
var e Event
|
||||
var ipID *int64
|
||||
var occurredAt, createdAt string
|
||||
if err := rows.Scan(&e.ID, &e.SourceType, &e.SourceID, &ipID, &e.EventType, &e.Payload, &occurredAt, &createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
e.IPID = ipID
|
||||
var err error
|
||||
if e.OccurredAt, err = dbToTime(occurredAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if e.CreatedAt, err = dbToTime(createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, e)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
@@ -0,0 +1,353 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// SeedQueue inserts the configured IP address list in order, assigning each
|
||||
// a stable sequence number. Re-running with the same list is a no-op for
|
||||
// addresses already present (ON CONFLICT DO NOTHING keyed by the UNIQUE
|
||||
// ip_address column), so restarting control-api against the same config
|
||||
// never re-queues already-processed addresses.
|
||||
func (d *DB) SeedQueue(ctx context.Context, addresses []string) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
now := timeToDB(Now())
|
||||
for i, addr := range addresses {
|
||||
_, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO ip_queue (ip_address, sequence, state, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?)
|
||||
ON CONFLICT(ip_address) DO NOTHING
|
||||
`, addr, i, IPQueued, now, now)
|
||||
if err != nil {
|
||||
return fmt.Errorf("seed %s: %w", addr, err)
|
||||
}
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to
|
||||
// the given idle validator. It returns (nil, nil) if the validator isn't
|
||||
// idle or no IP is queued. The DB connection pool is capped at one physical
|
||||
// connection (see Open), so this transaction already has exclusive access
|
||||
// to the database for its duration — no other claim, requeue, or update can
|
||||
// interleave — which combined with the conditional UPDATEs (checked via
|
||||
// RowsAffected) guarantees a single IP is never claimed by two validators.
|
||||
func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL time.Duration) (*IPQueueItem, error) {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
var state string
|
||||
err = tx.QueryRowContext(ctx, `SELECT state FROM validators WHERE validator_id=?`, validatorID).Scan(&state)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if state != ValidatorIdle {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var item IPQueueItem
|
||||
err = tx.QueryRowContext(ctx, `
|
||||
SELECT id, ip_address, sequence, attempt_number, retry_count
|
||||
FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1
|
||||
`, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
now := Now()
|
||||
lease := now.Add(leaseTTL)
|
||||
res, err := tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, owner_validator_id=?, assigned_at=?, lease_expires_at=?, updated_at=?
|
||||
WHERE id=? AND state=?
|
||||
`, IPAssigningFIP, validatorID, timeToDB(now), timeToDB(lease), timeToDB(now), item.ID, IPQueued)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n != 1 {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
res, err = tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=?, updated_at=?
|
||||
WHERE validator_id=? AND state=?
|
||||
`, ValidatorAssigned, item.ID, timeToDB(now), validatorID, ValidatorIdle)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n != 1 {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
if err := tx.Commit(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
item.State = IPAssigningFIP
|
||||
ownerID := validatorID
|
||||
item.OwnerValidatorID = &ownerID
|
||||
item.AssignedAt = &now
|
||||
item.LeaseExpiresAt = &lease
|
||||
return &item, nil
|
||||
}
|
||||
|
||||
func (d *DB) SetFIPAssociated(ctx context.Context, ipID int64, fipID string, leaseTTL time.Duration) error {
|
||||
now := Now()
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, fip_id=?, lease_expires_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPAwaitingSelfCheck, fipID, timeToDB(now.Add(leaseTTL)), timeToDB(now), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
func (d *DB) SetChecking(ctx context.Context, ipID int64, leaseTTL time.Duration) error {
|
||||
now := Now()
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, lease_expires_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, IPChecking, timeToDB(now.Add(leaseTTL)), timeToDB(now), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
func (d *DB) SetAggregating(ctx context.Context, ipID int64) error {
|
||||
_, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, updated_at=? WHERE id=?`,
|
||||
IPAggregating, timeToDB(Now()), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
// FinishIP records the aggregated result and marks the IP done or failed.
|
||||
func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error {
|
||||
state := IPDone
|
||||
if result == ResultFail {
|
||||
state = IPFailed
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET state=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, state, result, now, now, ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
// ReleaseFIP records that the floating IP has been disassociated and frees
|
||||
// the owning validator back to idle, in one transaction.
|
||||
func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
now := timeToDB(Now())
|
||||
if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET fip_released_at=?, updated_at=? WHERE id=?`, now, now, ipID); err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, now, validatorID); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// RequeueOrFail is used by both the retry path (association/self-check
|
||||
// failure) and the lease-sweep reclaim path. It clears ownership and
|
||||
// per-attempt progress, bumps attempt_number and retry_count, and either
|
||||
// sends the IP back to the queue or marks it permanently failed once
|
||||
// maxRetries is exceeded. The owning validator (if any) is freed in the
|
||||
// same transaction.
|
||||
func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string, maxRetries int) error {
|
||||
tx, err := d.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback()
|
||||
|
||||
var retryCount int
|
||||
if err := tx.QueryRowContext(ctx, `SELECT retry_count FROM ip_queue WHERE id=?`, ipID).Scan(&retryCount); err != nil {
|
||||
return err
|
||||
}
|
||||
retryCount++
|
||||
|
||||
now := timeToDB(Now())
|
||||
nextState := IPQueued
|
||||
if retryCount > maxRetries {
|
||||
nextState = IPFailed
|
||||
}
|
||||
|
||||
if nextState == IPQueued {
|
||||
_, err = tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, owner_validator_id=NULL, fip_id='', retry_count=?, attempt_number=attempt_number+1,
|
||||
lease_expires_at=NULL, egress_complete=0, site1_complete=0, site2_complete=0, site3_complete=0,
|
||||
overall_result='', assigned_at=NULL, updated_at=?
|
||||
WHERE id=?
|
||||
`, nextState, retryCount, now, ipID)
|
||||
} else {
|
||||
_, err = tx.ExecContext(ctx, `
|
||||
UPDATE ip_queue SET
|
||||
state=?, retry_count=?, overall_result=?, aggregated_at=?, updated_at=?
|
||||
WHERE id=?
|
||||
`, nextState, retryCount, ResultFail, now, now, ipID)
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if validatorID != "" {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, now, validatorID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
func (d *DB) SetEgressComplete(ctx context.Context, ipID int64) error {
|
||||
_, err := d.ExecContext(ctx, `UPDATE ip_queue SET egress_complete=1, updated_at=? WHERE id=?`, timeToDB(Now()), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
// SetSiteComplete marks completion for prober site 1, 2, or 3.
|
||||
func (d *DB) SetSiteComplete(ctx context.Context, ipID int64, siteIndex int) error {
|
||||
col := map[int]string{1: "site1_complete", 2: "site2_complete", 3: "site3_complete"}[siteIndex]
|
||||
if col == "" {
|
||||
return fmt.Errorf("invalid site index %d", siteIndex)
|
||||
}
|
||||
_, err := d.ExecContext(ctx, fmt.Sprintf(`UPDATE ip_queue SET %s=1, updated_at=? WHERE id=?`, col), timeToDB(Now()), ipID)
|
||||
return err
|
||||
}
|
||||
|
||||
func (d *DB) GetIP(ctx context.Context, ipID int64) (*IPQueueItem, error) {
|
||||
row := d.QueryRowContext(ctx, ipQueueSelect+`WHERE id=?`, ipID)
|
||||
return scanIPQueueItem(row)
|
||||
}
|
||||
|
||||
func (d *DB) GetIPByAddress(ctx context.Context, address string) (*IPQueueItem, error) {
|
||||
row := d.QueryRowContext(ctx, ipQueueSelect+`WHERE ip_address=?`, address)
|
||||
return scanIPQueueItem(row)
|
||||
}
|
||||
|
||||
func (d *DB) ListIPs(ctx context.Context) ([]IPQueueItem, error) {
|
||||
rows, err := d.QueryContext(ctx, ipQueueSelect+`ORDER BY sequence`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
// ListChecking returns all IPs currently in the checking state — the set a
|
||||
// prober should be actively probing.
|
||||
func (d *DB) ListChecking(ctx context.Context) ([]IPQueueItem, error) {
|
||||
rows, err := d.QueryContext(ctx, ipQueueSelect+`WHERE state=? ORDER BY sequence`, IPChecking)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
// ListReadyToAggregate returns checking-state IPs where every source has
|
||||
// reported completion, or whose checking window has expired.
|
||||
func (d *DB) ListReadyToAggregate(ctx context.Context, windowDeadline time.Time) ([]IPQueueItem, error) {
|
||||
rows, err := d.QueryContext(ctx, ipQueueSelect+`
|
||||
WHERE state=? AND (
|
||||
(egress_complete=1 AND site1_complete=1 AND site2_complete=1 AND site3_complete=1)
|
||||
OR assigned_at < ?
|
||||
)`, IPChecking, timeToDB(windowDeadline))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
// ListExpiredLeases returns non-terminal IPs whose lease has expired —
|
||||
// candidates for the lease sweep (crash recovery + stuck-validator reclaim).
|
||||
func (d *DB) ListExpiredLeases(ctx context.Context, now time.Time) ([]IPQueueItem, error) {
|
||||
rows, err := d.QueryContext(ctx, ipQueueSelect+`
|
||||
WHERE state NOT IN (?, ?) AND lease_expires_at IS NOT NULL AND lease_expires_at < ?
|
||||
`, IPDone, IPFailed, timeToDB(now))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
return scanIPQueueItems(rows)
|
||||
}
|
||||
|
||||
const ipQueueSelect = `
|
||||
SELECT id, ip_address, sequence, state, owner_validator_id, fip_id, attempt_number, retry_count,
|
||||
lease_expires_at, egress_complete, site1_complete, site2_complete, site3_complete, overall_result,
|
||||
assigned_at, aggregated_at, fip_released_at, created_at, updated_at
|
||||
FROM ip_queue
|
||||
`
|
||||
|
||||
func scanIPQueueItems(rows *sql.Rows) ([]IPQueueItem, error) {
|
||||
var out []IPQueueItem
|
||||
for rows.Next() {
|
||||
item, err := scanIPQueueItem(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *item)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
func scanIPQueueItem(row rowScanner) (*IPQueueItem, error) {
|
||||
var item IPQueueItem
|
||||
var owner sql.NullString
|
||||
var leaseExpires, assignedAt, aggregatedAt, fipReleasedAt sql.NullString
|
||||
var createdAt, updatedAt string
|
||||
if err := row.Scan(
|
||||
&item.ID, &item.IPAddress, &item.Sequence, &item.State, &owner, &item.FIPID,
|
||||
&item.AttemptNumber, &item.RetryCount, &leaseExpires,
|
||||
&item.EgressComplete, &item.Site1Complete, &item.Site2Complete, &item.Site3Complete,
|
||||
&item.OverallResult, &assignedAt, &aggregatedAt, &fipReleasedAt, &createdAt, &updatedAt,
|
||||
); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if owner.Valid {
|
||||
item.OwnerValidatorID = &owner.String
|
||||
}
|
||||
var err error
|
||||
if item.LeaseExpiresAt, err = nullStringToTimePtr(leaseExpires); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.AssignedAt, err = nullStringToTimePtr(assignedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.AggregatedAt, err = nullStringToTimePtr(aggregatedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.FIPReleasedAt, err = nullStringToTimePtr(fipReleasedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.CreatedAt, err = dbToTime(createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if item.UpdatedAt, err = dbToTime(updatedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &item, nil
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// RegisterValidator inserts a new validator or updates an existing one's
|
||||
// hostname/agent_version on re-registration. It never overwrites the state
|
||||
// of a validator that's mid-assignment, so an agent restarting while it
|
||||
// owns an IP doesn't silently lose that ownership.
|
||||
func (d *DB) RegisterValidator(ctx context.Context, validatorID, hostname, osPortID, agentVersion string) error {
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
INSERT INTO validators (validator_id, hostname, os_port_id, agent_version, state, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(validator_id) DO UPDATE SET
|
||||
hostname=excluded.hostname,
|
||||
os_port_id=excluded.os_port_id,
|
||||
agent_version=excluded.agent_version,
|
||||
updated_at=excluded.updated_at
|
||||
`, validatorID, hostname, osPortID, agentVersion, ValidatorIdle, now, now)
|
||||
if err != nil {
|
||||
return fmt.Errorf("register validator: %w", err)
|
||||
}
|
||||
// A brand-new row already lands in ValidatorIdle via the INSERT branch;
|
||||
// a re-registering validator that was 'unregistered' or 'unreachable'
|
||||
// (but not mid-assignment) should also come back to idle.
|
||||
_, err = d.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, updated_at=?
|
||||
WHERE validator_id=? AND state IN (?, ?)
|
||||
`, ValidatorIdle, now, validatorID, ValidatorUnregistered, ValidatorUnreachable)
|
||||
if err != nil {
|
||||
return fmt.Errorf("register validator (reactivate): %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (d *DB) Heartbeat(ctx context.Context, validatorID string) error {
|
||||
now := timeToDB(Now())
|
||||
res, err := d.ExecContext(ctx, `
|
||||
UPDATE validators SET last_heartbeat_at=?, updated_at=?,
|
||||
state = CASE WHEN state=? THEN ? ELSE state END
|
||||
WHERE validator_id=?
|
||||
`, now, now, ValidatorUnreachable, ValidatorIdle, validatorID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("heartbeat: %w", err)
|
||||
}
|
||||
n, _ := res.RowsAffected()
|
||||
if n == 0 {
|
||||
return sql.ErrNoRows
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (d *DB) GetValidator(ctx context.Context, validatorID string) (*Validator, error) {
|
||||
row := d.QueryRowContext(ctx, `
|
||||
SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version,
|
||||
last_heartbeat_at, created_at, updated_at
|
||||
FROM validators WHERE validator_id=?
|
||||
`, validatorID)
|
||||
return scanValidator(row)
|
||||
}
|
||||
|
||||
func (d *DB) ListValidators(ctx context.Context) ([]Validator, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version,
|
||||
last_heartbeat_at, created_at, updated_at
|
||||
FROM validators ORDER BY validator_id
|
||||
`)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Validator
|
||||
for rows.Next() {
|
||||
v, err := scanValidator(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// ListIdleValidators returns validators currently eligible to be handed a
|
||||
// new IP to work on.
|
||||
func (d *DB) ListIdleValidators(ctx context.Context) ([]Validator, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version,
|
||||
last_heartbeat_at, created_at, updated_at
|
||||
FROM validators WHERE state=? ORDER BY validator_id
|
||||
`, ValidatorIdle)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Validator
|
||||
for rows.Next() {
|
||||
v, err := scanValidator(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// ListStaleHeartbeats returns validators whose last heartbeat predates the
|
||||
// given cutoff and that aren't already marked unreachable.
|
||||
func (d *DB) ListStaleHeartbeats(ctx context.Context, cutoff time.Time) ([]Validator, error) {
|
||||
rows, err := d.QueryContext(ctx, `
|
||||
SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version,
|
||||
last_heartbeat_at, created_at, updated_at
|
||||
FROM validators
|
||||
WHERE state != ? AND last_heartbeat_at IS NOT NULL AND last_heartbeat_at < ?
|
||||
`, ValidatorUnreachable, timeToDB(cutoff))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Validator
|
||||
for rows.Next() {
|
||||
v, err := scanValidator(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *v)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
func (d *DB) MarkValidatorUnreachable(ctx context.Context, validatorID string) error {
|
||||
_, err := d.ExecContext(ctx, `UPDATE validators SET state=?, updated_at=? WHERE validator_id=?`,
|
||||
ValidatorUnreachable, timeToDB(Now()), validatorID)
|
||||
return err
|
||||
}
|
||||
|
||||
// FreeValidator returns a validator to idle with no assigned IP. Used after
|
||||
// an IP finishes (success or failure) or is reclaimed by the lease sweep.
|
||||
func (d *DB) FreeValidator(ctx context.Context, validatorID string) error {
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE validators SET state=?, current_ip_id=NULL, updated_at=?
|
||||
WHERE validator_id=?
|
||||
`, ValidatorIdle, timeToDB(Now()), validatorID)
|
||||
return err
|
||||
}
|
||||
|
||||
type rowScanner interface {
|
||||
Scan(dest ...interface{}) error
|
||||
}
|
||||
|
||||
func scanValidator(row rowScanner) (*Validator, error) {
|
||||
var v Validator
|
||||
var currentIPID sql.NullInt64
|
||||
var lastHeartbeat sql.NullString
|
||||
var createdAt, updatedAt string
|
||||
if err := row.Scan(&v.ValidatorID, &v.Hostname, &v.OSPortID, &v.State, ¤tIPID,
|
||||
&v.AgentVersion, &lastHeartbeat, &createdAt, &updatedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if currentIPID.Valid {
|
||||
v.CurrentIPID = ¤tIPID.Int64
|
||||
}
|
||||
hb, err := nullStringToTimePtr(lastHeartbeat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
v.LastHeartbeatAt = hb
|
||||
if v.CreatedAt, err = dbToTime(createdAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if v.UpdatedAt, err = dbToTime(updatedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &v, nil
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Timestamps are stored as RFC3339Nano TEXT explicitly (rather than relying
|
||||
// on driver-specific time.Time marshaling) so the on-disk format is stable
|
||||
// and easy to inspect with the sqlite3 CLI.
|
||||
|
||||
const timeLayout = time.RFC3339Nano
|
||||
|
||||
func timeToDB(t time.Time) string {
|
||||
return t.UTC().Format(timeLayout)
|
||||
}
|
||||
|
||||
func timePtrToDB(t *time.Time) interface{} {
|
||||
if t == nil {
|
||||
return nil
|
||||
}
|
||||
return timeToDB(*t)
|
||||
}
|
||||
|
||||
func dbToTime(s string) (time.Time, error) {
|
||||
return time.Parse(timeLayout, s)
|
||||
}
|
||||
|
||||
func nullStringToTimePtr(ns sql.NullString) (*time.Time, error) {
|
||||
if !ns.Valid || ns.String == "" {
|
||||
return nil, nil
|
||||
}
|
||||
t, err := dbToTime(ns.String)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &t, nil
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
package httpapi
|
||||
|
||||
// Request/response bodies for the /api/v1 surface. Kept in one file since
|
||||
// they're small and mostly 1:1 with a single handler each.
|
||||
|
||||
type registerAgentRequest struct {
|
||||
ValidatorID string `json:"validator_id"`
|
||||
Hostname string `json:"hostname"`
|
||||
AgentVersion string `json:"agent_version"`
|
||||
}
|
||||
|
||||
type registerAgentResponse struct {
|
||||
OK bool `json:"ok"`
|
||||
PollIntervalSeconds int `json:"poll_interval_seconds"`
|
||||
}
|
||||
|
||||
type heartbeatRequest struct {
|
||||
LocalState string `json:"local_state"`
|
||||
CurrentIPAddress string `json:"current_ip_address,omitempty"`
|
||||
}
|
||||
|
||||
type okResponse struct {
|
||||
OK bool `json:"ok"`
|
||||
}
|
||||
|
||||
type checkConfigDTO struct {
|
||||
Type string `json:"type"`
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
type assignmentResponse struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
Phase string `json:"phase"`
|
||||
CheckConfig []checkConfigDTO `json:"check_config,omitempty"`
|
||||
}
|
||||
|
||||
type selfCheckRequest struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
DetectedEgress string `json:"detected_egress_ip"`
|
||||
Success bool `json:"success"`
|
||||
Detail string `json:"detail"`
|
||||
}
|
||||
|
||||
type agentEventRequest struct {
|
||||
EventType string `json:"event_type"`
|
||||
IPID *int64 `json:"ip_id,omitempty"`
|
||||
Payload string `json:"payload,omitempty"`
|
||||
}
|
||||
|
||||
type checkResultDTO struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
CheckType string `json:"check_type"`
|
||||
Target string `json:"target,omitempty"`
|
||||
Success bool `json:"success"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
CheckedAt string `json:"checked_at"`
|
||||
}
|
||||
|
||||
type agentResultsRequest struct {
|
||||
Results []checkResultDTO `json:"results"`
|
||||
}
|
||||
|
||||
type agentCompleteRequest struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
}
|
||||
|
||||
type registerProberRequest struct {
|
||||
SiteID string `json:"site_id"`
|
||||
Hostname string `json:"hostname"`
|
||||
}
|
||||
|
||||
type proberAssignment struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
Ports []int `json:"ports"`
|
||||
ICMP bool `json:"icmp"`
|
||||
}
|
||||
|
||||
type proberResultDTO struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
CheckType string `json:"check_type"`
|
||||
Success bool `json:"success"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
CheckedAt string `json:"checked_at"`
|
||||
Complete bool `json:"complete"`
|
||||
}
|
||||
|
||||
type proberResultsRequest struct {
|
||||
Results []proberResultDTO `json:"results"`
|
||||
}
|
||||
|
||||
type errorResponse struct {
|
||||
Error string `json:"error"`
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminStatus(w http.ResponseWriter, r *http.Request) {
|
||||
ips, err := s.DB.ListIPs(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
validators, err := s.DB.ListValidators(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
byState := map[string]int{}
|
||||
for _, ip := range ips {
|
||||
byState[ip.State]++
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{
|
||||
"total_ips": len(ips),
|
||||
"ips_by_state": byState,
|
||||
"total_validators": len(validators),
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminIPs(w http.ResponseWriter, r *http.Request) {
|
||||
ips, err := s.DB.ListIPs(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, ips)
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminIPDetail(w http.ResponseWriter, r *http.Request) {
|
||||
address := r.PathValue("ip")
|
||||
item, err := s.DB.GetIPByAddress(r.Context(), address)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusNotFound, "unknown ip: "+address)
|
||||
return
|
||||
}
|
||||
checks, err := s.DB.ListChecksForAttempt(r.Context(), item.ID, item.AttemptNumber)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
events, err := s.DB.ListEventsForIP(r.Context(), item.ID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, struct {
|
||||
IP *db.IPQueueItem `json:"ip"`
|
||||
Checks []db.Check `json:"checks"`
|
||||
Events []db.Event `json:"events"`
|
||||
}{item, checks, events})
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) {
|
||||
validators, err := s.DB.ListValidators(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, validators)
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func (s *Server) handleAgentRegister(w http.ResponseWriter, r *http.Request) {
|
||||
var req registerAgentRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
if req.ValidatorID == "" {
|
||||
writeError(w, http.StatusBadRequest, "validator_id is required")
|
||||
return
|
||||
}
|
||||
// os_port_id is supplied via control-api's own config (config.ValidatorConfig),
|
||||
// not by the agent, so registration only touches hostname/version here;
|
||||
// RegisterValidator preserves any existing os_port_id row.
|
||||
existing, _ := s.DB.GetValidator(r.Context(), req.ValidatorID)
|
||||
osPortID := ""
|
||||
if existing != nil {
|
||||
osPortID = existing.OSPortID
|
||||
}
|
||||
if err := s.DB.RegisterValidator(r.Context(), req.ValidatorID, req.Hostname, osPortID, req.AgentVersion); err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
s.Orch.RecordEvent(r.Context(), "validator-agent", req.ValidatorID, nil, "registered", "")
|
||||
writeJSON(w, http.StatusOK, registerAgentResponse{OK: true, PollIntervalSeconds: s.Orch.Cfg.PollIntervalSeconds})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentHeartbeat(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req heartbeatRequest
|
||||
_ = readJSON(r, &req) // heartbeat body is informational only; tolerate empty/missing
|
||||
if err := s.DB.Heartbeat(r.Context(), id); err != nil {
|
||||
writeError(w, http.StatusNotFound, "unknown validator: "+id)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentAssignment(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
item, checks, err := s.Orch.AssignmentForValidator(r.Context(), id)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusNotFound, "unknown validator: "+id)
|
||||
return
|
||||
}
|
||||
if item == nil {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
return
|
||||
}
|
||||
var cfg []checkConfigDTO
|
||||
for _, c := range checks {
|
||||
cfg = append(cfg, checkConfigDTO{Type: c.Type, Targets: c.Targets})
|
||||
}
|
||||
writeJSON(w, http.StatusOK, assignmentResponse{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, Phase: item.State, CheckConfig: cfg,
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentSelfCheck(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req selfCheckRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
detail := req.Detail
|
||||
if req.DetectedEgress != "" {
|
||||
detail = "detected_egress_ip=" + req.DetectedEgress + " " + detail
|
||||
}
|
||||
if err := s.Orch.SelfCheckResult(r.Context(), id, req.IPID, req.Success, detail); err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentEvent(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req agentEventRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
if req.EventType == "" {
|
||||
writeError(w, http.StatusBadRequest, "event_type is required")
|
||||
return
|
||||
}
|
||||
s.Orch.RecordEvent(r.Context(), "validator-agent", id, req.IPID, req.EventType, req.Payload)
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentResults(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req agentResultsRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
for _, res := range req.Results {
|
||||
item, err := s.DB.GetIP(r.Context(), res.IPID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusNotFound, "unknown ip_id")
|
||||
return
|
||||
}
|
||||
checkedAt, err := time.Parse(time.RFC3339Nano, res.CheckedAt)
|
||||
if err != nil {
|
||||
checkedAt = db.Now()
|
||||
}
|
||||
err = s.Orch.RecordCheck(r.Context(), db.Check{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
|
||||
ValidatorID: id, Source: db.SourceEgress, CheckType: res.CheckType, Target: res.Target,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
|
||||
})
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleAgentComplete(w http.ResponseWriter, r *http.Request) {
|
||||
var req agentCompleteRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
if err := s.Orch.MarkEgressComplete(r.Context(), req.IPID); err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
|
||||
func (s *Server) handleWhatsMyIP(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, map[string]string{"ip": remoteIP(r)})
|
||||
}
|
||||
@@ -0,0 +1,92 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func (s *Server) handleProberRegister(w http.ResponseWriter, r *http.Request) {
|
||||
var req registerProberRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
if s.Orch.SiteIndexForID(req.SiteID) == 0 {
|
||||
writeError(w, http.StatusBadRequest, "unknown site_id: "+req.SiteID)
|
||||
return
|
||||
}
|
||||
s.Orch.RecordEvent(r.Context(), "prober", req.SiteID, nil, "registered", "")
|
||||
writeJSON(w, http.StatusOK, registerAgentResponse{OK: true, PollIntervalSeconds: s.Orch.Cfg.PollIntervalSeconds})
|
||||
}
|
||||
|
||||
// handleProberAssignments returns every IP currently in the checking
|
||||
// state — probers work the whole active set each poll, not one IP at a
|
||||
// time, since multiple validators run in parallel.
|
||||
func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) {
|
||||
siteID := r.PathValue("site_id")
|
||||
if s.Orch.SiteIndexForID(siteID) == 0 {
|
||||
writeError(w, http.StatusNotFound, "unknown site_id: "+siteID)
|
||||
return
|
||||
}
|
||||
items, err := s.DB.ListChecking(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
out := make([]proberAssignment, 0, len(items))
|
||||
for _, item := range items {
|
||||
out = append(out, proberAssignment{
|
||||
IPID: item.ID, IPAddress: item.IPAddress,
|
||||
Ports: s.Orch.Inbound.Ports, ICMP: s.Orch.Inbound.ICMP,
|
||||
})
|
||||
}
|
||||
writeJSON(w, http.StatusOK, out)
|
||||
}
|
||||
|
||||
func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) {
|
||||
siteID := r.PathValue("site_id")
|
||||
siteIndex := s.Orch.SiteIndexForID(siteID)
|
||||
if siteIndex == 0 {
|
||||
writeError(w, http.StatusNotFound, "unknown site_id: "+siteID)
|
||||
return
|
||||
}
|
||||
var req proberResultsRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
completed := map[int64]bool{}
|
||||
for _, res := range req.Results {
|
||||
item, err := s.DB.GetIP(r.Context(), res.IPID)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusNotFound, "unknown ip_id")
|
||||
return
|
||||
}
|
||||
checkedAt, err := time.Parse(time.RFC3339Nano, res.CheckedAt)
|
||||
if err != nil {
|
||||
checkedAt = db.Now()
|
||||
}
|
||||
err = s.Orch.RecordCheck(r.Context(), db.Check{
|
||||
IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber,
|
||||
Source: db.InboundSource(siteIndex), CheckType: res.CheckType, Target: res.IPAddress,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt,
|
||||
})
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
if res.Complete {
|
||||
completed[res.IPID] = true
|
||||
}
|
||||
}
|
||||
for ipID := range completed {
|
||||
if err := s.Orch.MarkSiteComplete(r.Context(), ipID, siteIndex); err != nil {
|
||||
writeError(w, http.StatusInternalServerError, err.Error())
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, okResponse{OK: true})
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
"cloudipvalidator/internal/orchestrator"
|
||||
)
|
||||
|
||||
// fakeClient plays both a validator-agent and the three probers against a
|
||||
// real httptest server, driving the full protocol exactly as the real
|
||||
// binaries would, to prove the HTTP layer and orchestrator agree on state
|
||||
// transitions end-to-end.
|
||||
type fakeClient struct {
|
||||
t *testing.T
|
||||
base string
|
||||
client *http.Client
|
||||
}
|
||||
|
||||
func (f *fakeClient) do(method, path string, body interface{}) (*http.Response, []byte) {
|
||||
f.t.Helper()
|
||||
var reader io.Reader
|
||||
if body != nil {
|
||||
b, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
f.t.Fatalf("marshal body: %v", err)
|
||||
}
|
||||
reader = bytes.NewReader(b)
|
||||
}
|
||||
req, err := http.NewRequest(method, f.base+path, reader)
|
||||
if err != nil {
|
||||
f.t.Fatalf("new request: %v", err)
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
resp, err := f.client.Do(req)
|
||||
if err != nil {
|
||||
f.t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
respBody, _ := io.ReadAll(resp.Body)
|
||||
return resp, respBody
|
||||
}
|
||||
|
||||
func TestEndToEndHTTPFlow(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
dbPath := filepath.Join(t.TempDir(), "test.db")
|
||||
d, err := db.Open(ctx, dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
defer d.Close()
|
||||
|
||||
mock := openstack.NewMockClient()
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
|
||||
cfg := &config.ControlAPI{
|
||||
Orchestrator: config.OrchestratorConfig{
|
||||
PollIntervalSeconds: 1, SelfCheckTimeoutSeconds: 10, MaxSelfCheckRetries: 3,
|
||||
CheckingWindowSeconds: 120, MaxRetries: 3, LeaseTTLSeconds: 180, HeartbeatTimeoutSeconds: 30,
|
||||
},
|
||||
Aggregation: config.AggregationConfig{MissingCountsAsFail: true},
|
||||
Sites: []config.SiteConfig{
|
||||
{SiteID: "site-1", Index: 1}, {SiteID: "site-2", Index: 2}, {SiteID: "site-3", Index: 3},
|
||||
},
|
||||
CheckTypes: []config.CheckTypeConfig{{Name: "https", Enabled: true, Targets: []string{"web"}}},
|
||||
Targets: map[string][]string{"web": {"https://example.test"}},
|
||||
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
|
||||
}
|
||||
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
||||
orch := orchestrator.New(d, mock, cfg, log)
|
||||
|
||||
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
||||
t.Fatalf("seed queue: %v", err)
|
||||
}
|
||||
|
||||
srv := New(d, orch, log)
|
||||
ts := httptest.NewServer(srv.Handler())
|
||||
defer ts.Close()
|
||||
|
||||
fc := &fakeClient{t: t, base: ts.URL, client: ts.Client()}
|
||||
|
||||
// Register the validator directly via DB (os_port_id comes from
|
||||
// control-api config, not the agent's own registration call).
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
||||
t.Fatalf("register validator: %v", err)
|
||||
}
|
||||
|
||||
// Agent re-registers over HTTP (as the real binary would at startup).
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{
|
||||
ValidatorID: "validator-1", Hostname: "host-1", AgentVersion: "v0.1",
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("register: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
// Orchestrator claims the IP and associates the FIP.
|
||||
orch.Tick(ctx)
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/agents/validator-1/assignment", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("assignment: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
var assignment assignmentResponse
|
||||
if err := json.Unmarshal(body, &assignment); err != nil {
|
||||
t.Fatalf("unmarshal assignment: %v", err)
|
||||
}
|
||||
if assignment.Phase != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected awaiting_self_check, got %s", assignment.Phase)
|
||||
}
|
||||
if assignment.IPAddress != "1.2.3.4" {
|
||||
t.Fatalf("expected 1.2.3.4, got %s", assignment.IPAddress)
|
||||
}
|
||||
|
||||
// Self-check via /whatsmyip: in the real deployment this would equal
|
||||
// the FIP; here we just exercise the endpoint and always report success.
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/whatsmyip", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("whatsmyip: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/self-check", selfCheckRequest{
|
||||
IPID: assignment.IPID, DetectedEgress: "1.2.3.4", Success: true, Detail: "matched",
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("self-check: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
// Agent runs its configured egress check and reports the result.
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{
|
||||
Results: []checkResultDTO{{
|
||||
IPID: assignment.IPID, CheckType: "https", Target: "https://example.test",
|
||||
Success: true, LatencyMS: 12, CheckedAt: time.Now().Format(time.RFC3339Nano),
|
||||
}},
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("results: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/complete", agentCompleteRequest{IPID: assignment.IPID})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("complete: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
// Three probers register, poll, and report inbound results.
|
||||
for _, site := range []string{"site-1", "site-2", "site-3"} {
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/probers/register", registerProberRequest{SiteID: site})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("prober register %s: status=%d body=%s", site, resp.StatusCode, body)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/probers/"+site+"/assignments", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("prober assignments %s: status=%d body=%s", site, resp.StatusCode, body)
|
||||
}
|
||||
var assignments []proberAssignment
|
||||
if err := json.Unmarshal(body, &assignments); err != nil {
|
||||
t.Fatalf("unmarshal assignments: %v", err)
|
||||
}
|
||||
if len(assignments) != 1 || assignments[0].IPAddress != "1.2.3.4" {
|
||||
t.Fatalf("expected 1 assignment for 1.2.3.4, got %+v", assignments)
|
||||
}
|
||||
|
||||
now := time.Now().Format(time.RFC3339Nano)
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/probers/"+site+"/results", proberResultsRequest{
|
||||
Results: []proberResultDTO{
|
||||
{IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "tcp-22", Success: true, CheckedAt: now},
|
||||
{IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "tcp-80", Success: true, CheckedAt: now},
|
||||
{IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "icmp", Success: true, CheckedAt: now, Complete: true},
|
||||
},
|
||||
})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("prober results %s: status=%d body=%s", site, resp.StatusCode, body)
|
||||
}
|
||||
}
|
||||
|
||||
// Orchestrator sweep should now aggregate and release.
|
||||
orch.Tick(ctx)
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/ips/1.2.3.4", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("admin ip detail: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
var detail struct {
|
||||
IP db.IPQueueItem `json:"ip"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &detail); err != nil {
|
||||
t.Fatalf("unmarshal detail: %v", err)
|
||||
}
|
||||
if detail.IP.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", detail.IP.State)
|
||||
}
|
||||
if detail.IP.OverallResult != db.ResultPass {
|
||||
t.Fatalf("expected pass, got %s", detail.IP.OverallResult)
|
||||
}
|
||||
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated at end of run")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package httpapi
|
||||
|
||||
import "net/http"
|
||||
|
||||
func (s *Server) routes(mux *http.ServeMux) {
|
||||
mux.HandleFunc("GET /healthz", s.handleHealthz)
|
||||
mux.HandleFunc("GET /api/v1/whatsmyip", s.handleWhatsMyIP)
|
||||
|
||||
mux.HandleFunc("POST /api/v1/agents/register", s.handleAgentRegister)
|
||||
mux.HandleFunc("POST /api/v1/agents/{id}/heartbeat", s.handleAgentHeartbeat)
|
||||
mux.HandleFunc("GET /api/v1/agents/{id}/assignment", s.handleAgentAssignment)
|
||||
mux.HandleFunc("POST /api/v1/agents/{id}/self-check", s.handleAgentSelfCheck)
|
||||
mux.HandleFunc("POST /api/v1/agents/{id}/events", s.handleAgentEvent)
|
||||
mux.HandleFunc("POST /api/v1/agents/{id}/results", s.handleAgentResults)
|
||||
mux.HandleFunc("POST /api/v1/agents/{id}/complete", s.handleAgentComplete)
|
||||
|
||||
mux.HandleFunc("POST /api/v1/probers/register", s.handleProberRegister)
|
||||
mux.HandleFunc("GET /api/v1/probers/{site_id}/assignments", s.handleProberAssignments)
|
||||
mux.HandleFunc("POST /api/v1/probers/{site_id}/results", s.handleProberResults)
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/status", s.handleAdminStatus)
|
||||
mux.HandleFunc("GET /api/v1/admin/ips", s.handleAdminIPs)
|
||||
mux.HandleFunc("GET /api/v1/admin/ips/{ip}", s.handleAdminIPDetail)
|
||||
mux.HandleFunc("GET /api/v1/admin/validators", s.handleAdminValidators)
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
// Package httpapi exposes the Control API's HTTP surface — the only way
|
||||
// validator-agents, probers, and operators interact with the system. All
|
||||
// business logic lives in internal/orchestrator; handlers here do request
|
||||
// parsing/validation, call into the orchestrator or db package, and shape
|
||||
// the JSON response.
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"log/slog"
|
||||
"net"
|
||||
"net/http"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/orchestrator"
|
||||
)
|
||||
|
||||
type Server struct {
|
||||
DB *db.DB
|
||||
Orch *orchestrator.Orchestrator
|
||||
Log *slog.Logger
|
||||
}
|
||||
|
||||
func New(d *db.DB, o *orchestrator.Orchestrator, log *slog.Logger) *Server {
|
||||
return &Server{DB: d, Orch: o, Log: log}
|
||||
}
|
||||
|
||||
func (s *Server) Handler() http.Handler {
|
||||
mux := http.NewServeMux()
|
||||
s.routes(mux)
|
||||
return loggingMiddleware(s.Log, mux)
|
||||
}
|
||||
|
||||
func loggingMiddleware(log *slog.Logger, next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
next.ServeHTTP(w, r)
|
||||
log.Debug("request", "method", r.Method, "path", r.URL.Path, "remote", r.RemoteAddr)
|
||||
})
|
||||
}
|
||||
|
||||
func writeJSON(w http.ResponseWriter, status int, v interface{}) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(status)
|
||||
if v != nil {
|
||||
_ = json.NewEncoder(w).Encode(v)
|
||||
}
|
||||
}
|
||||
|
||||
func writeError(w http.ResponseWriter, status int, msg string) {
|
||||
writeJSON(w, status, errorResponse{Error: msg})
|
||||
}
|
||||
|
||||
func readJSON(r *http.Request, v interface{}) error {
|
||||
if r.Body == nil || r.ContentLength == 0 {
|
||||
return nil
|
||||
}
|
||||
dec := json.NewDecoder(r.Body)
|
||||
return dec.Decode(v)
|
||||
}
|
||||
|
||||
// remoteIP returns the caller's source IP with any port stripped. Used by
|
||||
// /whatsmyip — the validator-agent's self-check mechanism relies on this
|
||||
// being the actual TCP peer address (as SNAT'd by the newly associated
|
||||
// FIP), never a client-supplied header.
|
||||
func remoteIP(r *http.Request) string {
|
||||
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
||||
if err != nil {
|
||||
return r.RemoteAddr
|
||||
}
|
||||
return host
|
||||
}
|
||||
@@ -0,0 +1,104 @@
|
||||
package openstack
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"github.com/gophercloud/gophercloud/v2"
|
||||
osauth "github.com/gophercloud/gophercloud/v2/openstack"
|
||||
"github.com/gophercloud/gophercloud/v2/openstack/networking/v2/extensions/layer3/floatingips"
|
||||
)
|
||||
|
||||
// ClientConfig carries pre-resolved credential values (already read from
|
||||
// environment variables by the caller — see config.OpenStackAuth). Token
|
||||
// auth is required per the deployment constraint that the admin credential
|
||||
// is supplied via the process environment, not a config file.
|
||||
type ClientConfig struct {
|
||||
AuthURL string
|
||||
Token string
|
||||
ProjectID string
|
||||
ProjectName string
|
||||
DomainName string
|
||||
Region string
|
||||
}
|
||||
|
||||
type Client struct {
|
||||
networking *gophercloud.ServiceClient
|
||||
}
|
||||
|
||||
// NewClient authenticates against Keystone using a pre-issued admin token
|
||||
// and returns a Client scoped to the given project/region, backed by the
|
||||
// Neutron (networking v2) service catalog entry.
|
||||
func NewClient(ctx context.Context, cfg ClientConfig) (*Client, error) {
|
||||
if cfg.AuthURL == "" || cfg.Token == "" {
|
||||
return nil, fmt.Errorf("openstack: auth URL and token are required")
|
||||
}
|
||||
|
||||
authOpts := gophercloud.AuthOptions{
|
||||
IdentityEndpoint: cfg.AuthURL,
|
||||
TokenID: cfg.Token,
|
||||
TenantID: cfg.ProjectID,
|
||||
TenantName: cfg.ProjectName,
|
||||
DomainName: cfg.DomainName,
|
||||
}
|
||||
if cfg.ProjectID != "" || cfg.ProjectName != "" {
|
||||
authOpts.Scope = &gophercloud.AuthScope{
|
||||
ProjectID: cfg.ProjectID,
|
||||
ProjectName: cfg.ProjectName,
|
||||
DomainName: cfg.DomainName,
|
||||
}
|
||||
}
|
||||
|
||||
provider, err := osauth.AuthenticatedClient(ctx, authOpts)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("openstack: authenticate: %w", err)
|
||||
}
|
||||
|
||||
networking, err := osauth.NewNetworkV2(provider, gophercloud.EndpointOpts{Region: cfg.Region})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("openstack: networking client: %w", err)
|
||||
}
|
||||
|
||||
return &Client{networking: networking}, nil
|
||||
}
|
||||
|
||||
func (c *Client) GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error) {
|
||||
pages, err := floatingips.List(c.networking, floatingips.ListOpts{FloatingIP: address}).AllPages(ctx)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("openstack: list floating ips: %w", err)
|
||||
}
|
||||
list, err := floatingips.ExtractFloatingIPs(pages)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("openstack: extract floating ips: %w", err)
|
||||
}
|
||||
if len(list) == 0 {
|
||||
return nil, ErrNotFound(address)
|
||||
}
|
||||
f := list[0]
|
||||
return &FloatingIP{ID: f.ID, Address: f.FloatingIP, PortID: f.PortID, ProjectID: f.TenantID}, nil
|
||||
}
|
||||
|
||||
func (c *Client) AssociateFloatingIP(ctx context.Context, fipID, portID string) error {
|
||||
_, err := floatingips.Update(ctx, c.networking, fipID, floatingips.UpdateOpts{
|
||||
PortID: &portID,
|
||||
}).Extract()
|
||||
if err != nil {
|
||||
return fmt.Errorf("openstack: associate floating ip %s -> port %s: %w", fipID, portID, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *Client) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
// gophercloud's UpdateOpts.PortID is *string with `omitempty`: a nil
|
||||
// pointer is dropped from the request body entirely (no-op), while a
|
||||
// pointer to "" is what actually serializes as port_id:null and
|
||||
// disassociates the floating IP. See floatingips.UpdateOpts godoc.
|
||||
empty := ""
|
||||
_, err := floatingips.Update(ctx, c.networking, fipID, floatingips.UpdateOpts{
|
||||
PortID: &empty,
|
||||
}).Extract()
|
||||
if err != nil {
|
||||
return fmt.Errorf("openstack: disassociate floating ip %s: %w", fipID, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package openstack
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestClientLive is a smoke test against a real OpenStack deployment. It's
|
||||
// skipped unless OPENSTACK_LIVE_TEST=1 and the usual OS_* credential env
|
||||
// vars are set, so the default `go test ./...` run needs no cloud access.
|
||||
// It only exercises GetFloatingIPByAddress (read-only) against
|
||||
// OS_TEST_FLOATING_IP, to avoid mutating real infrastructure in CI.
|
||||
func TestClientLive(t *testing.T) {
|
||||
if os.Getenv("OPENSTACK_LIVE_TEST") != "1" {
|
||||
t.Skip("set OPENSTACK_LIVE_TEST=1 (and OS_AUTH_URL, OS_TOKEN, OS_TEST_FLOATING_IP) to run")
|
||||
}
|
||||
|
||||
testIP := os.Getenv("OS_TEST_FLOATING_IP")
|
||||
if testIP == "" {
|
||||
t.Fatal("OS_TEST_FLOATING_IP must name a floating IP address that exists in the target project")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
client, err := NewClient(ctx, ClientConfig{
|
||||
AuthURL: os.Getenv("OS_AUTH_URL"),
|
||||
Token: os.Getenv("OS_TOKEN"),
|
||||
ProjectID: os.Getenv("OS_PROJECT_ID"),
|
||||
ProjectName: os.Getenv("OS_PROJECT_NAME"),
|
||||
DomainName: os.Getenv("OS_PROJECT_DOMAIN_NAME"),
|
||||
Region: os.Getenv("OS_REGION_NAME"),
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("new client: %v", err)
|
||||
}
|
||||
|
||||
fip, err := client.GetFloatingIPByAddress(ctx, testIP)
|
||||
if err != nil {
|
||||
t.Fatalf("get floating ip %s: %v", testIP, err)
|
||||
}
|
||||
if fip.Address != testIP {
|
||||
t.Fatalf("expected address %s, got %s", testIP, fip.Address)
|
||||
}
|
||||
t.Logf("found floating ip %s: id=%s port_id=%q", fip.Address, fip.ID, fip.PortID)
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
// Package openstack isolates all OpenStack/Neutron interaction behind a
|
||||
// small interface, so the orchestrator's business logic can be unit tested
|
||||
// without a real cloud, and so the real implementation is swappable for a
|
||||
// mock via config alone.
|
||||
package openstack
|
||||
|
||||
import "context"
|
||||
|
||||
type FloatingIP struct {
|
||||
ID string
|
||||
Address string
|
||||
PortID string // empty when not associated to any port
|
||||
ProjectID string
|
||||
}
|
||||
|
||||
// FloatingIPClient is the only surface the orchestrator uses to manage
|
||||
// Floating IPs. It intentionally does not expose allocation/deallocation —
|
||||
// this tool only associates/disassociates pre-existing floating IPs with
|
||||
// validator ports; returning an address to the customer-facing pool is an
|
||||
// external business process out of scope here.
|
||||
type FloatingIPClient interface {
|
||||
// GetFloatingIPByAddress looks up the Neutron floating-ip resource for
|
||||
// a given public address. Returns ErrNotFound if no such floating IP
|
||||
// is registered in the service project.
|
||||
GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error)
|
||||
|
||||
// AssociateFloatingIP attaches the floating IP to the given Neutron
|
||||
// port (the validator's primary NIC port).
|
||||
AssociateFloatingIP(ctx context.Context, fipID, portID string) error
|
||||
|
||||
// DisassociateFloatingIP detaches the floating IP from whatever port
|
||||
// it's currently attached to, if any. Disassociating an already-free
|
||||
// floating IP is a no-op, not an error.
|
||||
DisassociateFloatingIP(ctx context.Context, fipID string) error
|
||||
}
|
||||
|
||||
type notFoundError struct{ address string }
|
||||
|
||||
func (e *notFoundError) Error() string { return "floating ip not found: " + e.address }
|
||||
|
||||
// ErrNotFound wraps address into an error satisfying IsNotFound.
|
||||
func ErrNotFound(address string) error { return ¬FoundError{address} }
|
||||
|
||||
// IsNotFound reports whether err indicates the address has no matching
|
||||
// Neutron floating-ip resource.
|
||||
func IsNotFound(err error) bool {
|
||||
_, ok := err.(*notFoundError)
|
||||
return ok
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
package openstack
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// MockClient is an in-memory FloatingIPClient used by unit tests and the
|
||||
// offline end-to-end harness. Seed it with the pool of floating IPs the
|
||||
// scenario expects to exist before use.
|
||||
type MockClient struct {
|
||||
mu sync.Mutex
|
||||
fips map[string]*FloatingIP // keyed by ID
|
||||
byIP map[string]string // address -> ID
|
||||
|
||||
// AssociateFailures/DisassociateFailures let tests force a failure for
|
||||
// a specific floating-IP ID on its next call, to exercise retry paths.
|
||||
AssociateFailures map[string]error
|
||||
DisassociateFailures map[string]error
|
||||
}
|
||||
|
||||
func NewMockClient() *MockClient {
|
||||
return &MockClient{
|
||||
fips: make(map[string]*FloatingIP),
|
||||
byIP: make(map[string]string),
|
||||
}
|
||||
}
|
||||
|
||||
// Seed registers a floating IP as if pre-allocated in the service project.
|
||||
func (m *MockClient) Seed(id, address, projectID string) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.fips[id] = &FloatingIP{ID: id, Address: address, ProjectID: projectID}
|
||||
m.byIP[address] = id
|
||||
}
|
||||
|
||||
func (m *MockClient) GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
id, ok := m.byIP[address]
|
||||
if !ok {
|
||||
return nil, ErrNotFound(address)
|
||||
}
|
||||
f := *m.fips[id]
|
||||
return &f, nil
|
||||
}
|
||||
|
||||
func (m *MockClient) AssociateFloatingIP(ctx context.Context, fipID, portID string) error {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if err := m.AssociateFailures[fipID]; err != nil {
|
||||
delete(m.AssociateFailures, fipID)
|
||||
return err
|
||||
}
|
||||
f, ok := m.fips[fipID]
|
||||
if !ok {
|
||||
return fmt.Errorf("mock openstack: unknown floating ip %q", fipID)
|
||||
}
|
||||
f.PortID = portID
|
||||
return nil
|
||||
}
|
||||
|
||||
func (m *MockClient) DisassociateFloatingIP(ctx context.Context, fipID string) error {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if err := m.DisassociateFailures[fipID]; err != nil {
|
||||
delete(m.DisassociateFailures, fipID)
|
||||
return err
|
||||
}
|
||||
f, ok := m.fips[fipID]
|
||||
if !ok {
|
||||
return fmt.Errorf("mock openstack: unknown floating ip %q", fipID)
|
||||
}
|
||||
f.PortID = ""
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,361 @@
|
||||
// Package orchestrator implements the Control API's core scheduling loop:
|
||||
// claiming queued IPs onto idle validators, driving each IP through
|
||||
// FIP-association -> self-check -> checking -> aggregation -> release, and
|
||||
// reclaiming work from crashed/stuck validators via a lease sweep. It has
|
||||
// no HTTP dependency — internal/httpapi calls into this package, and it can
|
||||
// be exercised directly in tests against an in-memory OpenStack mock and a
|
||||
// temp-file SQLite database.
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
// CheckConfig is the check-type/target configuration handed to a
|
||||
// validator-agent once its IP has passed self-check. It mirrors
|
||||
// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a
|
||||
// flat list so the agent doesn't need its own copy of the target-group
|
||||
// mapping.
|
||||
type CheckConfig struct {
|
||||
Type string `json:"type"`
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
type Orchestrator struct {
|
||||
DB *db.DB
|
||||
OS openstack.FloatingIPClient
|
||||
Cfg config.OrchestratorConfig
|
||||
Agg config.AggregationConfig
|
||||
Checks []CheckConfig
|
||||
Sites []config.SiteConfig
|
||||
Inbound config.InboundConfig
|
||||
Log *slog.Logger
|
||||
}
|
||||
|
||||
func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator {
|
||||
var checks []CheckConfig
|
||||
for _, ct := range cfg.CheckTypes {
|
||||
if !ct.Enabled {
|
||||
continue
|
||||
}
|
||||
var targets []string
|
||||
for _, group := range ct.Targets {
|
||||
targets = append(targets, cfg.Targets[group]...)
|
||||
}
|
||||
checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets})
|
||||
}
|
||||
return &Orchestrator{
|
||||
DB: d,
|
||||
OS: osClient,
|
||||
Cfg: cfg.Orchestrator,
|
||||
Agg: cfg.Aggregation,
|
||||
Checks: checks,
|
||||
Sites: cfg.Sites,
|
||||
Inbound: cfg.Inbound,
|
||||
Log: log,
|
||||
}
|
||||
}
|
||||
|
||||
func (o *Orchestrator) leaseTTL() time.Duration {
|
||||
return time.Duration(o.Cfg.LeaseTTLSeconds) * time.Second
|
||||
}
|
||||
|
||||
// Tick runs one pass of the scheduling loop: claim+associate for idle
|
||||
// validators, sweep the checking window for ready-to-aggregate IPs, and
|
||||
// reclaim expired leases. Intended to be called on a fixed interval
|
||||
// (Cfg.PollIntervalSeconds) by the caller (cmd/control-api/main.go).
|
||||
func (o *Orchestrator) Tick(ctx context.Context) {
|
||||
if err := o.assignIdleValidators(ctx); err != nil {
|
||||
o.Log.Error("assign idle validators", "err", err)
|
||||
}
|
||||
if err := o.sweepCheckingWindow(ctx); err != nil {
|
||||
o.Log.Error("sweep checking window", "err", err)
|
||||
}
|
||||
if err := o.sweepExpiredLeases(ctx); err != nil {
|
||||
o.Log.Error("sweep expired leases", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// assignIdleValidators claims the next queued IP for every currently idle
|
||||
// validator and kicks off FIP association for each newly claimed IP.
|
||||
func (o *Orchestrator) assignIdleValidators(ctx context.Context) error {
|
||||
idle, err := o.DB.ListIdleValidators(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list idle validators: %w", err)
|
||||
}
|
||||
for _, v := range idle {
|
||||
item, err := o.DB.ClaimNextQueued(ctx, v.ValidatorID, o.leaseTTL())
|
||||
if err != nil {
|
||||
o.Log.Error("claim next queued", "validator", v.ValidatorID, "err", err)
|
||||
continue
|
||||
}
|
||||
if item == nil {
|
||||
continue // no work available for this validator right now
|
||||
}
|
||||
o.Log.Info("claimed ip", "validator", v.ValidatorID, "ip", item.IPAddress, "ip_id", item.ID)
|
||||
if err := o.associateFIP(ctx, v.ValidatorID, v.OSPortID, item); err != nil {
|
||||
o.Log.Error("associate fip", "validator", v.ValidatorID, "ip", item.IPAddress, "err", err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) associateFIP(ctx context.Context, validatorID, osPortID string, item *db.IPQueueItem) error {
|
||||
fip, err := o.OS.GetFloatingIPByAddress(ctx, item.IPAddress)
|
||||
if err != nil {
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("lookup floating ip: %v", err))
|
||||
return err
|
||||
}
|
||||
if err := o.OS.AssociateFloatingIP(ctx, fip.ID, osPortID); err != nil {
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("associate floating ip: %v", err))
|
||||
return err
|
||||
}
|
||||
if err := o.DB.SetFIPAssociated(ctx, item.ID, fip.ID, o.leaseTTL()); err != nil {
|
||||
return fmt.Errorf("set fip associated: %w", err)
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "fip_associated", fmt.Sprintf(`{"fip_id":%q,"validator_id":%q}`, fip.ID, validatorID))
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) requeueOrFail(ctx context.Context, ipID int64, validatorID, reason string) {
|
||||
if err := o.DB.RequeueOrFail(ctx, ipID, validatorID, o.Cfg.MaxRetries); err != nil {
|
||||
o.Log.Error("requeue or fail", "ip_id", ipID, "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", &ipID, "retry_or_fail", fmt.Sprintf(`{"reason":%q}`, reason))
|
||||
}
|
||||
|
||||
// SelfCheckResult is called by the httpapi layer when a validator-agent
|
||||
// reports its post-association self-check outcome.
|
||||
func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string, ipID int64, success bool, detail string) error {
|
||||
o.event(ctx, "validator-agent", validatorID, &ipID, "self_check_result",
|
||||
fmt.Sprintf(`{"success":%t,"detail":%q}`, success, detail))
|
||||
|
||||
if !success {
|
||||
item, err := o.DB.GetIP(ctx, ipID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries {
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail)
|
||||
return nil
|
||||
}
|
||||
// Retry association without fully requeuing: re-drive the same
|
||||
// claim by cycling back through requeue/claim keeps the logic in
|
||||
// one place at the cost of the IP briefly returning to `queued`.
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail)
|
||||
return nil
|
||||
}
|
||||
|
||||
return o.DB.SetChecking(ctx, ipID, o.leaseTTL())
|
||||
}
|
||||
|
||||
// AssignmentForValidator returns the check config for a validator's current
|
||||
// IP if it's ready to be worked on (awaiting_self_check or checking),
|
||||
// or nil if the validator has nothing to do right now.
|
||||
func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) {
|
||||
v, err := o.DB.GetValidator(ctx, validatorID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if v.CurrentIPID == nil {
|
||||
return nil, nil, nil
|
||||
}
|
||||
item, err := o.DB.GetIP(ctx, *v.CurrentIPID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking {
|
||||
return nil, nil, nil
|
||||
}
|
||||
return item, o.Checks, nil
|
||||
}
|
||||
|
||||
// SiteIndexForID resolves a configured site_id to its 1/2/3 index.
|
||||
func (o *Orchestrator) SiteIndexForID(siteID string) int {
|
||||
for _, s := range o.Sites {
|
||||
if s.SiteID == siteID {
|
||||
return s.Index
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
// completion signal (egress or a given site's full port+icmp sweep),
|
||||
// updates the corresponding *_complete flag.
|
||||
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
|
||||
return o.DB.UpsertCheck(ctx, c)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
|
||||
return o.DB.SetEgressComplete(ctx, ipID)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteIndex int) error {
|
||||
return o.DB.SetSiteComplete(ctx, ipID, siteIndex)
|
||||
}
|
||||
|
||||
// sweepCheckingWindow moves IPs that have either finished reporting from
|
||||
// every source, or hit the checking-window deadline, into aggregation.
|
||||
func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
deadline := db.Now().Add(-time.Duration(o.Cfg.CheckingWindowSeconds) * time.Second)
|
||||
ready, err := o.DB.ListReadyToAggregate(ctx, deadline)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list ready to aggregate: %w", err)
|
||||
}
|
||||
for _, item := range ready {
|
||||
if err := o.aggregateAndRelease(ctx, item); err != nil {
|
||||
o.Log.Error("aggregate and release", "ip_id", item.ID, "err", err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueItem) error {
|
||||
if err := o.DB.SetAggregating(ctx, item.ID); err != nil {
|
||||
return err
|
||||
}
|
||||
checks, err := o.DB.ListChecksForAttempt(ctx, item.ID, item.AttemptNumber)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
expected := o.expectedCheckCount()
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing := expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
var result string
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
|
||||
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
|
||||
return err
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "aggregated",
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
|
||||
|
||||
if item.FIPID != "" {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip", "ip_id", item.ID, "fip_id", item.FIPID, "err", err)
|
||||
// Fall through and still free the validator/DB state — the
|
||||
// lease sweep or an operator can reconcile a stuck Neutron
|
||||
// association separately; we must not leave the validator
|
||||
// wedged as "checking" forever over a cloud API hiccup.
|
||||
}
|
||||
}
|
||||
if item.OwnerValidatorID != nil {
|
||||
if err := o.DB.ReleaseFIP(ctx, item.ID, *item.OwnerValidatorID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe).
|
||||
func (o *Orchestrator) expectedCheckCount() int {
|
||||
egress := 0
|
||||
for _, c := range o.Checks {
|
||||
egress += len(c.Targets)
|
||||
}
|
||||
inboundPerSite := len(o.Inbound.Ports)
|
||||
if o.Inbound.ICMP {
|
||||
inboundPerSite++
|
||||
}
|
||||
return egress + inboundPerSite*len(o.Sites)
|
||||
}
|
||||
|
||||
// sweepExpiredLeases reclaims non-terminal IPs whose lease has passed —
|
||||
// this is both the "stuck/crashed validator" reclaim path and, since all
|
||||
// state lives in SQLite, the control-api crash-recovery path: a freshly
|
||||
// restarted process finds the same expired leases and reclaims them the
|
||||
// same way, with no separate recovery code required.
|
||||
func (o *Orchestrator) sweepExpiredLeases(ctx context.Context) error {
|
||||
expired, err := o.DB.ListExpiredLeases(ctx, db.Now())
|
||||
if err != nil {
|
||||
return fmt.Errorf("list expired leases: %w", err)
|
||||
}
|
||||
for _, item := range expired {
|
||||
validatorID := ""
|
||||
if item.OwnerValidatorID != nil {
|
||||
validatorID = *item.OwnerValidatorID
|
||||
}
|
||||
o.Log.Info("lease expired, reclaiming", "ip_id", item.ID, "ip", item.IPAddress, "validator", validatorID)
|
||||
if item.FIPID != "" {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on lease reclaim", "ip_id", item.ID, "err", err)
|
||||
}
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "lease_expired", fmt.Sprintf(`{"validator_id":%q}`, validatorID))
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, "lease expired")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// sweepStaleHeartbeats marks validators unreachable if they haven't
|
||||
// heartbeated within HeartbeatTimeoutSeconds. It does not itself reclaim
|
||||
// their in-flight IP — that happens independently via lease expiry, so a
|
||||
// validator that stops heartbeating but whose lease hasn't yet expired
|
||||
// still finishes its current check window if it recovers in time.
|
||||
func (o *Orchestrator) SweepStaleHeartbeats(ctx context.Context) error {
|
||||
cutoff := db.Now().Add(-time.Duration(o.Cfg.HeartbeatTimeoutSeconds) * time.Second)
|
||||
stale, err := o.DB.ListStaleHeartbeats(ctx, cutoff)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, v := range stale {
|
||||
if err := o.DB.MarkValidatorUnreachable(ctx, v.ValidatorID); err != nil {
|
||||
o.Log.Error("mark validator unreachable", "validator", v.ValidatorID, "err", err)
|
||||
continue
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "validator_unreachable", fmt.Sprintf(`{"validator_id":%q}`, v.ValidatorID))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// RecordEvent is the exported entry point httpapi uses to log
|
||||
// agent/prober-reported audit events (config_received, fip_changed,
|
||||
// error, etc.) through the same path as internally generated events.
|
||||
func (o *Orchestrator) RecordEvent(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) {
|
||||
o.event(ctx, sourceType, sourceID, ipID, eventType, payload)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) event(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) {
|
||||
if err := o.DB.InsertEvent(ctx, db.Event{
|
||||
SourceType: sourceType,
|
||||
SourceID: sourceID,
|
||||
IPID: ipID,
|
||||
EventType: eventType,
|
||||
Payload: payload,
|
||||
OccurredAt: db.Now(),
|
||||
}); err != nil {
|
||||
o.Log.Error("insert event", "type", eventType, "err", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,277 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
func newTestOrchestrator(t *testing.T, leaseTTLSeconds int) (*Orchestrator, *db.DB, *openstack.MockClient) {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
dbPath := filepath.Join(t.TempDir(), "test.db")
|
||||
d, err := db.Open(ctx, dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { d.Close() })
|
||||
|
||||
mock := openstack.NewMockClient()
|
||||
|
||||
cfg := &config.ControlAPI{
|
||||
Orchestrator: config.OrchestratorConfig{
|
||||
PollIntervalSeconds: 1,
|
||||
SelfCheckTimeoutSeconds: 10,
|
||||
MaxSelfCheckRetries: 3,
|
||||
CheckingWindowSeconds: 120,
|
||||
MaxRetries: 3,
|
||||
LeaseTTLSeconds: leaseTTLSeconds,
|
||||
HeartbeatTimeoutSeconds: 30,
|
||||
},
|
||||
Aggregation: config.AggregationConfig{MissingCountsAsFail: true},
|
||||
Sites: []config.SiteConfig{
|
||||
{SiteID: "site-1", Index: 1},
|
||||
{SiteID: "site-2", Index: 2},
|
||||
{SiteID: "site-3", Index: 3},
|
||||
},
|
||||
CheckTypes: []config.CheckTypeConfig{
|
||||
{Name: "https", Enabled: true, Targets: []string{"web"}},
|
||||
{Name: "ssh", Enabled: false, Targets: []string{"web"}},
|
||||
},
|
||||
Targets: map[string][]string{
|
||||
"web": {"https://example.test"},
|
||||
},
|
||||
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
|
||||
}
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
||||
return New(d, mock, cfg, log), d, mock
|
||||
}
|
||||
|
||||
func TestHappyPath(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
||||
t.Fatalf("register validator: %v", err)
|
||||
}
|
||||
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
||||
t.Fatalf("seed queue: %v", err)
|
||||
}
|
||||
|
||||
// 1. claim + associate
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if err != nil {
|
||||
t.Fatalf("get ip: %v", err)
|
||||
}
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
||||
}
|
||||
if ip.FIPID != "fip-1" {
|
||||
t.Fatalf("expected fip-1 associated, got %q", ip.FIPID)
|
||||
}
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
||||
t.Fatalf("expected fip associated to port-1, got %q", fip.PortID)
|
||||
}
|
||||
|
||||
v, err := d.GetValidator(ctx, "validator-1")
|
||||
if err != nil {
|
||||
t.Fatalf("get validator: %v", err)
|
||||
}
|
||||
if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != ip.ID {
|
||||
t.Fatalf("expected validator assigned to ip %d, got state=%s current_ip=%v", ip.ID, v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
// 2. self-check success
|
||||
if err := o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "egress matched"); err != nil {
|
||||
t.Fatalf("self check result: %v", err)
|
||||
}
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPChecking {
|
||||
t.Fatalf("expected checking, got %s", ip.State)
|
||||
}
|
||||
|
||||
// 3. egress result + completion
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record egress check: %v", err)
|
||||
}
|
||||
if err := o.MarkEgressComplete(ctx, ip.ID); err != nil {
|
||||
t.Fatalf("mark egress complete: %v", err)
|
||||
}
|
||||
|
||||
// 4. inbound results from all 3 sites
|
||||
for site := 1; site <= 3; site++ {
|
||||
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress,
|
||||
Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record inbound check site %d: %v", site, err)
|
||||
}
|
||||
}
|
||||
if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil {
|
||||
t.Fatalf("mark site %d complete: %v", site, err)
|
||||
}
|
||||
}
|
||||
|
||||
// 5. sweep should now aggregate + release
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
if ip.OverallResult != db.ResultPass {
|
||||
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
||||
}
|
||||
if ip.FIPReleasedAt == nil {
|
||||
t.Fatalf("expected fip_released_at to be set")
|
||||
}
|
||||
|
||||
v, _ = d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("expected validator idle with no current ip, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated, still on port %q", fip.PortID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPartialResult(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
|
||||
// Egress passes...
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.MarkEgressComplete(ctx, ip.ID)
|
||||
// ...but only site-1 reports, and one of its checks fails.
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "tcp-22", Target: ip.IPAddress, Success: false, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "tcp-80", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "icmp", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
||||
|
||||
// Force the checking window to have elapsed so aggregation proceeds
|
||||
// even though site-2/site-3 never reported.
|
||||
o.Cfg.CheckingWindowSeconds = 0
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
if ip.OverallResult != db.ResultPartial {
|
||||
t.Fatalf("expected partial, got %s", ip.OverallResult)
|
||||
}
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated even on partial result")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseReclaim(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
// A 1s lease (rather than 0) avoids a race within the very first Tick:
|
||||
// with a 0s TTL the item's lease can already look expired by the time
|
||||
// the same Tick's lease-sweep phase runs, depending on how much
|
||||
// wall-clock time the claim+associate phase happened to take.
|
||||
o, d, mock := newTestOrchestrator(t, 1)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx) // claims + associates; validator never self-checks
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
||||
}
|
||||
|
||||
time.Sleep(1100 * time.Millisecond) // let the 1s lease expire
|
||||
o.Tick(ctx) // should reclaim via lease sweep
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPQueued {
|
||||
t.Fatalf("expected requeued after lease reclaim, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
||||
}
|
||||
if ip.RetryCount != 1 {
|
||||
t.Fatalf("expected retry_count=1, got %d", ip.RetryCount)
|
||||
}
|
||||
|
||||
v, _ := d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated on reclaim, still on port %q", fip.PortID)
|
||||
}
|
||||
|
||||
// A subsequent tick should re-claim and re-associate the same IP for
|
||||
// the now-idle validator, proving the queue keeps making progress.
|
||||
// Give this attempt a real lease so it isn't immediately re-expired by
|
||||
// the same tick's lease sweep (a 0s TTL, as above, expires instantly).
|
||||
o.Cfg.LeaseTTLSeconds = 180
|
||||
o.Tick(ctx)
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected re-claimed ip to be awaiting_self_check again, got %s", ip.State)
|
||||
}
|
||||
if ip.AttemptNumber != 2 {
|
||||
t.Fatalf("expected attempt_number=2 after reclaim+reassign, got %d", ip.AttemptNumber)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMaxRetriesExhausted(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 0)
|
||||
o.Cfg.MaxRetries = 1
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
for i := 0; i < 3; i++ {
|
||||
o.Tick(ctx)
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if ip.State != db.IPFailed {
|
||||
t.Fatalf("expected failed after exhausting retries, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
// Package probercore implements a prober's poll loop: register once with
|
||||
// its site identity, then each cycle fetch the set of IPs currently under
|
||||
// test and run inbound reachability checks (TCP connect on each configured
|
||||
// port, ICMP echo) directly against each one — this is the real,
|
||||
// unmediated network test; only the control channel goes through the
|
||||
// Control API.
|
||||
package probercore
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/apiclient"
|
||||
"cloudipvalidator/internal/checkrunner"
|
||||
"cloudipvalidator/internal/config"
|
||||
)
|
||||
|
||||
type Prober struct {
|
||||
cfg *config.Prober
|
||||
client *apiclient.Client
|
||||
log *slog.Logger
|
||||
}
|
||||
|
||||
func New(cfg *config.Prober, log *slog.Logger) *Prober {
|
||||
timeout := time.Duration(cfg.Checks.TCPTimeoutSeconds) * time.Second
|
||||
if timeout <= 0 {
|
||||
timeout = 10 * time.Second
|
||||
}
|
||||
return &Prober{
|
||||
cfg: cfg,
|
||||
client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second),
|
||||
log: log,
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Prober) Run(ctx context.Context) error {
|
||||
if err := p.register(ctx); err != nil {
|
||||
return fmt.Errorf("register: %w", err)
|
||||
}
|
||||
|
||||
interval := time.Duration(p.cfg.PollIntervalSeconds) * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
p.pollOnce(ctx)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-ticker.C:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type registerReq struct {
|
||||
SiteID string `json:"site_id"`
|
||||
Hostname string `json:"hostname"`
|
||||
}
|
||||
|
||||
func (p *Prober) register(ctx context.Context) error {
|
||||
hostname, _ := os.Hostname()
|
||||
_, err := p.client.Do(ctx, "POST", "/api/v1/probers/register", registerReq{SiteID: p.cfg.SiteID, Hostname: hostname}, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
p.log.Info("registered", "site_id", p.cfg.SiteID)
|
||||
return nil
|
||||
}
|
||||
|
||||
type assignment struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
Ports []int `json:"ports"`
|
||||
ICMP bool `json:"icmp"`
|
||||
}
|
||||
|
||||
type resultDTO struct {
|
||||
IPID int64 `json:"ip_id"`
|
||||
IPAddress string `json:"ip_address"`
|
||||
CheckType string `json:"check_type"`
|
||||
Success bool `json:"success"`
|
||||
LatencyMS int64 `json:"latency_ms"`
|
||||
Detail string `json:"detail,omitempty"`
|
||||
CheckedAt string `json:"checked_at"`
|
||||
Complete bool `json:"complete"`
|
||||
}
|
||||
|
||||
func (p *Prober) pollOnce(ctx context.Context) {
|
||||
var assignments []assignment
|
||||
ok, err := p.client.Do(ctx, "GET", "/api/v1/probers/"+p.cfg.SiteID+"/assignments", nil, &assignments)
|
||||
if err != nil {
|
||||
p.log.Error("get assignments", "err", err)
|
||||
return
|
||||
}
|
||||
if !ok || len(assignments) == 0 {
|
||||
return
|
||||
}
|
||||
|
||||
for _, a := range assignments {
|
||||
p.probeOne(ctx, a)
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Prober) probeOne(ctx context.Context, a assignment) {
|
||||
tcpTimeout := time.Duration(p.cfg.Checks.TCPTimeoutSeconds) * time.Second
|
||||
icmpTimeout := time.Duration(p.cfg.Checks.ICMPTimeoutSeconds) * time.Second
|
||||
|
||||
var results []resultDTO
|
||||
for _, port := range a.Ports {
|
||||
res := checkrunner.TCPConnect(a.IPAddress, port, tcpTimeout)(ctx)
|
||||
results = append(results, resultDTO{
|
||||
IPID: a.IPID, IPAddress: a.IPAddress, CheckType: res.CheckType,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail,
|
||||
CheckedAt: res.CheckedAt.Format(time.RFC3339Nano),
|
||||
})
|
||||
}
|
||||
if a.ICMP {
|
||||
res := checkrunner.ICMPEcho(a.IPAddress, p.cfg.Checks.ICMPCount, icmpTimeout)(ctx)
|
||||
results = append(results, resultDTO{
|
||||
IPID: a.IPID, IPAddress: a.IPAddress, CheckType: res.CheckType,
|
||||
Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail,
|
||||
CheckedAt: res.CheckedAt.Format(time.RFC3339Nano),
|
||||
})
|
||||
}
|
||||
if len(results) > 0 {
|
||||
results[len(results)-1].Complete = true
|
||||
}
|
||||
|
||||
body := struct {
|
||||
Results []resultDTO `json:"results"`
|
||||
}{results}
|
||||
if _, err := p.client.Do(ctx, "POST", "/api/v1/probers/"+p.cfg.SiteID+"/results", body, nil); err != nil {
|
||||
p.log.Error("post results", "ip", a.IPAddress, "err", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
// Command httpstub is a trivial "always 200 OK" HTTP server used only by
|
||||
// scripts/run-local-e2e.sh, standing in for the real internet targets
|
||||
// (hub.docker.com, github.com, packages.ubuntu.com) so the offline
|
||||
// end-to-end harness needs no real internet access.
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"log"
|
||||
"net/http"
|
||||
)
|
||||
|
||||
func main() {
|
||||
addr := flag.String("addr", ":9090", "address to listen on")
|
||||
flag.Parse()
|
||||
|
||||
http.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
_, _ = w.Write([]byte("stub ok\n"))
|
||||
})
|
||||
log.Printf("httpstub listening on %s", *addr)
|
||||
log.Fatal(http.ListenAndServe(*addr, nil))
|
||||
}
|
||||
Executable
+209
@@ -0,0 +1,209 @@
|
||||
#!/usr/bin/env bash
|
||||
# Offline end-to-end smoke test for the Cloud IP Validator (see
|
||||
# docs/LOCAL_E2E.md). Runs control-api + 1 validator-agent + 3 probers as
|
||||
# local processes against a mock OpenStack backend, with loopback stand-ins
|
||||
# for both the "floating IP" under test and the outbound targets — no real
|
||||
# cloud or internet access required.
|
||||
#
|
||||
# Usage: scripts/run-local-e2e.sh
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$REPO_ROOT"
|
||||
|
||||
GO="${GO:-go}"
|
||||
if ! command -v "$GO" >/dev/null 2>&1 && [ -x /usr/local/go/bin/go ]; then
|
||||
GO=/usr/local/go/bin/go
|
||||
fi
|
||||
|
||||
WORK_DIR="$(mktemp -d /tmp/cloud-ip-validator-e2e.XXXXXX)"
|
||||
BIN_DIR="$WORK_DIR/bin"
|
||||
LOG_DIR="$WORK_DIR/logs"
|
||||
mkdir -p "$BIN_DIR" "$LOG_DIR"
|
||||
|
||||
PIDS=()
|
||||
cleanup() {
|
||||
echo "--- cleaning up (workdir: $WORK_DIR) ---"
|
||||
for pid in "${PIDS[@]:-}"; do
|
||||
kill "$pid" >/dev/null 2>&1 || true
|
||||
done
|
||||
wait >/dev/null 2>&1 || true
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
echo "--- building binaries ---"
|
||||
"$GO" build -o "$BIN_DIR/control-api" ./cmd/control-api
|
||||
"$GO" build -o "$BIN_DIR/validator-agent" ./cmd/validator-agent
|
||||
"$GO" build -o "$BIN_DIR/prober" ./cmd/prober
|
||||
"$GO" build -o "$BIN_DIR/httpstub" ./scripts/httpstub
|
||||
|
||||
echo "--- starting stub outbound targets (stand-ins for hub.docker.com / github.com / packages.ubuntu.com) ---"
|
||||
"$BIN_DIR/httpstub" -addr 127.0.0.1:29091 >"$LOG_DIR/stub-1.log" 2>&1 & PIDS+=($!)
|
||||
"$BIN_DIR/httpstub" -addr 127.0.0.1:29092 >"$LOG_DIR/stub-2.log" 2>&1 & PIDS+=($!)
|
||||
"$BIN_DIR/httpstub" -addr 127.0.0.1:29093 >"$LOG_DIR/stub-3.log" 2>&1 & PIDS+=($!)
|
||||
sleep 0.5
|
||||
|
||||
cat >"$WORK_DIR/control-api.yaml" <<EOF
|
||||
server:
|
||||
listen_addr: "127.0.0.1:28080"
|
||||
database:
|
||||
path: "$WORK_DIR/control-api.db"
|
||||
openstack:
|
||||
mode: "mock"
|
||||
orchestrator:
|
||||
poll_interval_seconds: 1
|
||||
self_check_timeout_seconds: 5
|
||||
max_self_check_retries: 3
|
||||
checking_window_seconds: 15
|
||||
max_retries: 3
|
||||
lease_ttl_seconds: 8
|
||||
heartbeat_timeout_seconds: 10
|
||||
aggregation:
|
||||
missing_counts_as_fail: true
|
||||
validators:
|
||||
- validator_id: "validator_01"
|
||||
os_port_id: "mock-port-1"
|
||||
sites:
|
||||
- site_id: "site-1"
|
||||
index: 1
|
||||
- site_id: "site-2"
|
||||
index: 2
|
||||
- site_id: "site-3"
|
||||
index: 3
|
||||
check_types:
|
||||
- name: "https"
|
||||
enabled: true
|
||||
targets: ["stub-targets"]
|
||||
- name: "icmp"
|
||||
enabled: true
|
||||
targets: ["stub-targets"]
|
||||
targets:
|
||||
stub-targets:
|
||||
- "http://127.0.0.1:29091"
|
||||
- "http://127.0.0.1:29092"
|
||||
- "http://127.0.0.1:29093"
|
||||
inbound_checks:
|
||||
ports: [22022, 28081, 28443, 28888]
|
||||
icmp: true
|
||||
# Only 127.0.0.1 is usable here: the self-check mechanism compares the
|
||||
# TCP source address the validator-agent's own outbound connection to
|
||||
# control-api arrives with (see httpapi.remoteIP) against the assigned
|
||||
# address. In a real deployment Neutron actually SNATs traffic through the
|
||||
# newly attached FIP, so any configured address works; this offline
|
||||
# harness has no real network-level SNAT, so only the literal loopback
|
||||
# address the agent's traffic really originates from can ever pass.
|
||||
ip_addresses:
|
||||
- "127.0.0.1"
|
||||
EOF
|
||||
|
||||
cat >"$WORK_DIR/validator-agent.yaml" <<EOF
|
||||
validator_id: "validator_01"
|
||||
control_api_url: "http://127.0.0.1:28080"
|
||||
poll_interval_seconds: 1
|
||||
self_check:
|
||||
timeout_seconds: 5
|
||||
checks:
|
||||
https_timeout_seconds: 5
|
||||
icmp_timeout_seconds: 3
|
||||
icmp_count: 2
|
||||
ssh:
|
||||
enabled: false
|
||||
timeout_seconds: 3
|
||||
EOF
|
||||
|
||||
for site in site-1 site-2 site-3; do
|
||||
cat >"$WORK_DIR/prober-$site.yaml" <<EOF
|
||||
site_id: "$site"
|
||||
control_api_url: "http://127.0.0.1:28080"
|
||||
poll_interval_seconds: 1
|
||||
checks:
|
||||
tcp_timeout_seconds: 3
|
||||
icmp_timeout_seconds: 3
|
||||
icmp_count: 2
|
||||
EOF
|
||||
done
|
||||
|
||||
echo "--- starting control-api ---"
|
||||
"$BIN_DIR/control-api" -config "$WORK_DIR/control-api.yaml" >"$LOG_DIR/control-api.log" 2>&1 & CONTROL_PID=$!
|
||||
PIDS+=("$CONTROL_PID")
|
||||
|
||||
echo -n "waiting for control-api /healthz "
|
||||
for i in $(seq 1 30); do
|
||||
if curl -fs "http://127.0.0.1:28080/healthz" >/dev/null 2>&1; then
|
||||
echo "ok"
|
||||
break
|
||||
fi
|
||||
echo -n "."
|
||||
sleep 0.5
|
||||
if [ "$i" -eq 30 ]; then
|
||||
echo "FAILED"
|
||||
cat "$LOG_DIR/control-api.log"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
start_validator_agent() {
|
||||
"$BIN_DIR/validator-agent" -config "$WORK_DIR/validator-agent.yaml" -stub-ports "22022,28081,28443,28888" \
|
||||
>>"$LOG_DIR/validator-agent.log" 2>&1 & VALIDATOR_PID=$!
|
||||
PIDS+=("$VALIDATOR_PID")
|
||||
echo "validator-agent started (pid $VALIDATOR_PID)"
|
||||
}
|
||||
|
||||
status() { curl -fs "http://127.0.0.1:28080/api/v1/admin/status"; }
|
||||
ips() { curl -fs "http://127.0.0.1:28080/api/v1/admin/ips"; }
|
||||
ip_state() { ips | python3 -c "import json,sys; d=json.load(sys.stdin); print(d[0]['State'] if d else 'none')" 2>/dev/null || echo "?"; }
|
||||
|
||||
echo "--- starting validator-agent ---"
|
||||
start_validator_agent
|
||||
|
||||
# Poll tightly (every 0.1s) for the IP to leave 'queued' — i.e. control-api
|
||||
# has claimed it and associated the mock FIP — then kill the agent right
|
||||
# there, before it has had a chance to self-check or run any checks. This
|
||||
# proves the lease sweep reclaims genuinely in-flight, not-yet-reported
|
||||
# work, rather than just racing against how fast the local checks happen
|
||||
# to run.
|
||||
echo "--- waiting for control-api to claim+associate the ip, to kill the agent mid-flight ---"
|
||||
CAUGHT_MIDFLIGHT=0
|
||||
for i in $(seq 1 50); do
|
||||
st="$(ip_state)"
|
||||
if [ "$st" != "queued" ] && [ "$st" != "none" ] && [ "$st" != "?" ]; then
|
||||
CAUGHT_MIDFLIGHT=1
|
||||
break
|
||||
fi
|
||||
sleep 0.1
|
||||
done
|
||||
|
||||
if [ "$CAUGHT_MIDFLIGHT" -eq 1 ]; then
|
||||
echo ">>> caught ip in state '$st'; killing validator-agent mid-run to demonstrate lease reclaim <<<"
|
||||
kill "$VALIDATOR_PID" >/dev/null 2>&1 || true
|
||||
wait "$VALIDATOR_PID" 2>/dev/null || true
|
||||
sleep 9 # let the 8s lease_ttl_seconds expire and get reclaimed
|
||||
echo ">>> restarting validator-agent <<<"
|
||||
start_validator_agent
|
||||
else
|
||||
echo ">>> never observed the ip leave 'queued' in time; skipping the reclaim demonstration <<<"
|
||||
fi
|
||||
|
||||
echo "--- starting 3 probers ---"
|
||||
for site in site-1 site-2 site-3; do
|
||||
"$BIN_DIR/prober" -config "$WORK_DIR/prober-$site.yaml" >"$LOG_DIR/prober-$site.log" 2>&1 & PIDS+=($!)
|
||||
done
|
||||
|
||||
echo "--- waiting for the queue to drain (up to 60s) ---"
|
||||
for i in $(seq 1 120); do
|
||||
sleep 0.5
|
||||
remaining=$(status | python3 -c "import json,sys; d=json.load(sys.stdin); s=d['ips_by_state']; print(sum(v for k,v in s.items() if k not in ('done','failed')))" 2>/dev/null || echo "?")
|
||||
if [ "$remaining" = "0" ]; then
|
||||
echo "queue drained after ~$((i / 2))s"
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
echo "--- final status ---"
|
||||
status | python3 -m json.tool
|
||||
echo "--- final ip results ---"
|
||||
ips | python3 -m json.tool
|
||||
|
||||
echo "--- logs are in $WORK_DIR (kept for inspection; workdir NOT auto-deleted) ---"
|
||||
trap - EXIT
|
||||
cleanup
|
||||
Reference in new issue
Block a user