From 7e44db87b2d0947b60936034141ef95a55855230 Mon Sep 17 00:00:00 2001 From: ayurishchev Date: Fri, 21 Aug 2026 07:34:45 +0300 Subject: [PATCH] repo init --- README.md | 53 +++ blueprint.svg | 4 + cmd/control-api/main.go | 139 ++++++++ cmd/prober/main.go | 39 +++ cmd/validator-agent/main.go | 85 +++++ configs/control-api.example.yaml | 84 +++++ configs/prober.example.yaml | 11 + configs/validator-agent.example.yaml | 17 + deploy/systemd/control-api.service | 24 ++ deploy/systemd/prober.service | 22 ++ deploy/systemd/validator-agent.service | 22 ++ docs/API.md | 376 +++++++++++++++++++++ docs/LOCAL_E2E.md | 92 +++++ docs/SETUP.md | 275 +++++++++++++++ docs/USAGE.md | 269 +++++++++++++++ go.mod | 22 ++ go.sum | 58 ++++ internal/agentcore/agentcore.go | 280 +++++++++++++++ internal/apiclient/apiclient.go | 65 ++++ internal/checkrunner/https.go | 36 ++ internal/checkrunner/icmp.go | 82 +++++ internal/checkrunner/tcp.go | 51 +++ internal/checkrunner/types.go | 44 +++ internal/config/config.go | 236 +++++++++++++ internal/db/db.go | 85 +++++ internal/db/migrations/0001_init.sql | 65 ++++ internal/db/models.go | 117 +++++++ internal/db/queries_checks.go | 58 ++++ internal/db/queries_events.go | 63 ++++ internal/db/queries_ipqueue.go | 353 +++++++++++++++++++ internal/db/queries_validators.go | 179 ++++++++++ internal/db/timeutil.go | 38 +++ internal/httpapi/dto.go | 98 ++++++ internal/httpapi/handlers_admin.go | 75 ++++ internal/httpapi/handlers_agent.go | 145 ++++++++ internal/httpapi/handlers_prober.go | 92 +++++ internal/httpapi/httpapi_test.go | 210 ++++++++++++ internal/httpapi/routes.go | 25 ++ internal/httpapi/server.go | 71 ++++ internal/openstack/client.go | 104 ++++++ internal/openstack/client_live_test.go | 48 +++ internal/openstack/interface.go | 49 +++ internal/openstack/mock.go | 77 +++++ internal/orchestrator/orchestrator.go | 361 ++++++++++++++++++++ internal/orchestrator/orchestrator_test.go | 277 +++++++++++++++ internal/probercore/probercore.go | 138 ++++++++ scripts/httpstub/main.go | 23 ++ scripts/run-local-e2e.sh | 209 ++++++++++++ 48 files changed, 5346 insertions(+) create mode 100644 README.md create mode 100644 blueprint.svg create mode 100644 cmd/control-api/main.go create mode 100644 cmd/prober/main.go create mode 100644 cmd/validator-agent/main.go create mode 100644 configs/control-api.example.yaml create mode 100644 configs/prober.example.yaml create mode 100644 configs/validator-agent.example.yaml create mode 100644 deploy/systemd/control-api.service create mode 100644 deploy/systemd/prober.service create mode 100644 deploy/systemd/validator-agent.service create mode 100644 docs/API.md create mode 100644 docs/LOCAL_E2E.md create mode 100644 docs/SETUP.md create mode 100644 docs/USAGE.md create mode 100644 go.mod create mode 100644 go.sum create mode 100644 internal/agentcore/agentcore.go create mode 100644 internal/apiclient/apiclient.go create mode 100644 internal/checkrunner/https.go create mode 100644 internal/checkrunner/icmp.go create mode 100644 internal/checkrunner/tcp.go create mode 100644 internal/checkrunner/types.go create mode 100644 internal/config/config.go create mode 100644 internal/db/db.go create mode 100644 internal/db/migrations/0001_init.sql create mode 100644 internal/db/models.go create mode 100644 internal/db/queries_checks.go create mode 100644 internal/db/queries_events.go create mode 100644 internal/db/queries_ipqueue.go create mode 100644 internal/db/queries_validators.go create mode 100644 internal/db/timeutil.go create mode 100644 internal/httpapi/dto.go create mode 100644 internal/httpapi/handlers_admin.go create mode 100644 internal/httpapi/handlers_agent.go create mode 100644 internal/httpapi/handlers_prober.go create mode 100644 internal/httpapi/httpapi_test.go create mode 100644 internal/httpapi/routes.go create mode 100644 internal/httpapi/server.go create mode 100644 internal/openstack/client.go create mode 100644 internal/openstack/client_live_test.go create mode 100644 internal/openstack/interface.go create mode 100644 internal/openstack/mock.go create mode 100644 internal/orchestrator/orchestrator.go create mode 100644 internal/orchestrator/orchestrator_test.go create mode 100644 internal/probercore/probercore.go create mode 100644 scripts/httpstub/main.go create mode 100755 scripts/run-local-e2e.sh diff --git a/README.md b/README.md new file mode 100644 index 0000000..1fa724f --- /dev/null +++ b/README.md @@ -0,0 +1,53 @@ +# Cloud IP Validator + +Система проверки освобождённых публичных IPv4-адресов перед их повторной +выдачей: каждый адрес привязывается как Floating IP к ВМ-валидатору в +облаке (OpenStack), после чего проверяется одновременно в двух +направлениях — исходящий трафик валидатора (egress: HTTPS/ICMP/опционально +SSH до внешних целей) и входящая доступность самого адреса с трёх +независимых внешних площадок (inbound: TCP 22/80/443/8080 + ICMP). Итог по +каждому адресу — `pass`/`partial`/`fail`, с полной историей проверок в +базе данных. + +Три компонента: `control-api` (управляющий сервис, единственный со +состоянием), `validator-agent` (работает на каждой ВМ-валидаторе, без +состояния) и `prober` (работает на каждой из трёх внешних площадок, без +состояния). Все три общаются между собой только через HTTP API +control-api. + +## Документация + +| Документ | Для чего | +|---|---| +| [docs/SETUP.md](docs/SETUP.md) | Сборка, конфигурация, первый запуск стенда — с нуля | +| [docs/USAGE.md](docs/USAGE.md) | Повседневная работа: постановка адресов в очередь, наблюдение за статусом, разбор результатов | +| [docs/API.md](docs/API.md) | Спецификация HTTP API control-api и примеры запросов (curl) | +| [docs/LOCAL_E2E.md](docs/LOCAL_E2E.md) | Полностью офлайн-прогон всей системы одним скриптом — без реального облака и интернета | + +## Быстрый старт (60 секунд, без OpenStack) + +Хотите просто увидеть систему в работе — без реального облака: + +```bash +go build ./... && go test ./... +scripts/run-local-e2e.sh +``` + +Скрипт сам поднимет все три компонента как локальные процессы (в режиме +`openstack.mode: mock`) и прогонит один тестовый адрес через полный цикл +проверки, включая демонстрацию восстановления после сбоя валидатора. +Подробности — в [docs/LOCAL_E2E.md](docs/LOCAL_E2E.md). + +## Быстрый старт (реальный стенд) + +1. Соберите три бинарника и подготовьте конфиги — + [docs/SETUP.md](docs/SETUP.md). +2. Разверните `control-api` на управляющей машине, `validator-agent` — + на каждой ВМ-валидаторе, `prober` — на каждой из трёх площадок + ([пошагово в docs/SETUP.md](docs/SETUP.md#развёртывание-control-api)). +3. Добавьте адреса в очередь и наблюдайте за результатом — + [docs/USAGE.md](docs/USAGE.md). + +```bash +curl -s http://:8080/api/v1/admin/status | python3 -m json.tool +``` diff --git a/blueprint.svg b/blueprint.svg new file mode 100644 index 0000000..d431c7f --- /dev/null +++ b/blueprint.svg @@ -0,0 +1,4 @@ + + + +
Список целей
https://hub.docker.com
https://github.com
https://packages.ubuntu.com
Валидаторы
валидатор_01
валидатор_02
валидатор_NN
Проверки
ssh (опционально)
https
icmp
Список валидаторов определяется конфигурационным файлом
Список типов проверок определяется конфигурационным файлом
Список целей определяется конфигурационным файлом
Публичные IPv4 адреса
A.A.A.B
A.A.A.C
A.A.A.D
Список адресов определяется конфигурационным файлом
Каждый публичный IPv4 проходит через серию проверок на валидаторе с фиксацией результата проверки в базе данных

IP адреса могут распределяться между группой валидаторов, но проверки не должны дублироваться (один IPv4 обрабатывается на одном валидаторе)

IP адреса из загруженного списка проверяются все до последнего.
Образ конечного результата
  1. Управляющее API
  2. Управляющее API может считывать конфигурацию из конфигурационного файла для профилирования агентов
  3. Управляющее API взаимодействует с API облачной платформы для управления FIP (назначение FIP на вилидаторы)
  4. Управляющее API взаимодействует с сервисом агента
  5. Управляющее API регистрирует события в базе данных
  6. Сервис агента, который работает на валидаторах и интегрирован с API
  7. База данных аудита или телеметрических данных: содержит информацию, которую агент отправляет во внешнее API
  8. База данных проверок содержит: отметку времени, сведения о тестируемом IP адресе, ID валидатора, тип проверки, результат проверки, общий результат (по всем проверкам)
  9. Если все проверки пройдены, то это фиксируется в общем результате (итоговое значение общего результата)
  10. Если проверки пройдены частично, то это фиксируется в общем результате (итоговое значение общего результата)
  11. Если проверки не пройдены, то это фиксируется в общем результате (итоговое значение общего результата)

Как работать с валидаторами.

Предложить оптимальный способ взаимодействия исходя из следующих условий:
  1. валидаторы - виртуальные машины в составе облачной платформы
  2. публичные IPv4 - адреса облачной платформы, которые подключаются к валидатору в качестве FIP (Floating IP)
  3. подключение адреса к валидатору происходит через API вызов к облачной платформе
  4. для запуска программы (сценария) нужно передать через env токен администратора для управления
  5. валидатор может начинать проверки только после того, как он выполнит самопроверку и определит, что он выходит в интернет именно через назначенный FIP (выполняется агентом)
  6. на валидаторе работает агент, который сигнализирует о доступности валидатора, а также результат самопроверки по FIP адресу
  7. агент отправляет запись в журнал (вызов внешнего API), когда видит изменение состояния валидатора (FIP изменился)
  8. агент получает конфигурацию для запуска проверок и отправляет запись в журнал (вызов внешнего API), что конфигурация получена
  9. агент запускает проверки согласно полученной конфигурации
  10. агент зависит (получает команды управления и конфигурацию) от управляющего API, что позволяет валидатору работать как Stateless сервис
\ No newline at end of file diff --git a/cmd/control-api/main.go b/cmd/control-api/main.go new file mode 100644 index 0000000..0988c83 --- /dev/null +++ b/cmd/control-api/main.go @@ -0,0 +1,139 @@ +// Command control-api is the orchestrator/control-plane binary: it reads +// the deployment config, opens the SQLite database, connects to OpenStack +// (or a mock, per config), seeds the IP work queue, and serves the HTTP API +// that validator-agents and probers poll against. +package main + +import ( + "context" + "flag" + "fmt" + "log/slog" + "net/http" + "os" + "os/signal" + "syscall" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/httpapi" + "cloudipvalidator/internal/openstack" + "cloudipvalidator/internal/orchestrator" +) + +func main() { + configPath := flag.String("config", "configs/control-api.yaml", "path to control-api config file") + flag.Parse() + + log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo})) + + if err := run(*configPath, log); err != nil { + log.Error("fatal", "err", err) + os.Exit(1) + } +} + +func run(configPath string, log *slog.Logger) error { + cfg, err := config.LoadControlAPI(configPath) + if err != nil { + return fmt.Errorf("load config: %w", err) + } + + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + + database, err := db.Open(ctx, cfg.Database.Path) + if err != nil { + return fmt.Errorf("open database: %w", err) + } + defer database.Close() + + for _, v := range cfg.Validators { + if err := database.RegisterValidator(ctx, v.ValidatorID, "", v.OSPortID, ""); err != nil { + return fmt.Errorf("seed validator %s: %w", v.ValidatorID, err) + } + } + if err := database.SeedQueue(ctx, cfg.IPAddresses); err != nil { + return fmt.Errorf("seed ip queue: %w", err) + } + + osClient, err := newOpenStackClient(ctx, cfg) + if err != nil { + return fmt.Errorf("init openstack client: %w", err) + } + + orch := orchestrator.New(database, osClient, cfg, log) + + srv := httpapi.New(database, orch, log) + httpServer := &http.Server{Addr: cfg.Server.ListenAddr, Handler: srv.Handler()} + + go runOrchestratorLoop(ctx, orch, cfg, log) + + errCh := make(chan error, 1) + go func() { + log.Info("listening", "addr", cfg.Server.ListenAddr) + if err := httpServer.ListenAndServe(); err != nil && err != http.ErrServerClosed { + errCh <- err + } + }() + + select { + case <-ctx.Done(): + log.Info("shutting down") + shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + return httpServer.Shutdown(shutdownCtx) + case err := <-errCh: + return err + } +} + +func runOrchestratorLoop(ctx context.Context, orch *orchestrator.Orchestrator, cfg *config.ControlAPI, log *slog.Logger) { + interval := time.Duration(cfg.Orchestrator.PollIntervalSeconds) * time.Second + ticker := time.NewTicker(interval) + defer ticker.Stop() + + heartbeatTicker := time.NewTicker(interval * 2) + defer heartbeatTicker.Stop() + + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + orch.Tick(ctx) + case <-heartbeatTicker.C: + if err := orch.SweepStaleHeartbeats(ctx); err != nil { + log.Error("sweep stale heartbeats", "err", err) + } + } + } +} + +func newOpenStackClient(ctx context.Context, cfg *config.ControlAPI) (openstack.FloatingIPClient, error) { + if cfg.OpenStack.Mode == "real" { + clientCfg := openstack.ClientConfig{ + AuthURL: os.Getenv(cfg.OpenStack.AuthURLEnv), + Token: os.Getenv(cfg.OpenStack.TokenEnv), + ProjectID: os.Getenv(cfg.OpenStack.ProjectIDEnv), + ProjectName: os.Getenv(cfg.OpenStack.ProjectNameEnv), + DomainName: os.Getenv(cfg.OpenStack.ProjectDomainEnv), + Region: os.Getenv(cfg.OpenStack.RegionEnv), + } + if clientCfg.AuthURL == "" || clientCfg.Token == "" { + return nil, fmt.Errorf("openstack.mode=real requires %s and %s to be set in the environment", + cfg.OpenStack.AuthURLEnv, cfg.OpenStack.TokenEnv) + } + return openstack.NewClient(ctx, clientCfg) + } + + // Mock mode: pre-seed one synthetic floating-ip resource per + // configured address, standing in for the pre-allocated Neutron + // floating IPs a real deployment's service project would already have. + mock := openstack.NewMockClient() + for i, addr := range cfg.IPAddresses { + mock.Seed(fmt.Sprintf("mock-fip-%d", i), addr, "mock-project") + } + return mock, nil +} diff --git a/cmd/prober/main.go b/cmd/prober/main.go new file mode 100644 index 0000000..228aa23 --- /dev/null +++ b/cmd/prober/main.go @@ -0,0 +1,39 @@ +// Command prober runs on one of the external test sites. It polls the +// Control API for the set of floating IPs currently under test and probes +// each directly (TCP connect + ICMP echo) to measure inbound reachability +// from this vantage point. +package main + +import ( + "context" + "flag" + "log/slog" + "os" + "os/signal" + "syscall" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/probercore" +) + +func main() { + configPath := flag.String("config", "configs/prober.yaml", "path to prober config file") + flag.Parse() + + log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo})) + + cfg, err := config.LoadProber(*configPath) + if err != nil { + log.Error("load config", "err", err) + os.Exit(1) + } + + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + + prober := probercore.New(cfg, log) + if err := prober.Run(ctx); err != nil && err != context.Canceled { + log.Error("prober stopped", "err", err) + os.Exit(1) + } +} diff --git a/cmd/validator-agent/main.go b/cmd/validator-agent/main.go new file mode 100644 index 0000000..e63eb51 --- /dev/null +++ b/cmd/validator-agent/main.go @@ -0,0 +1,85 @@ +// Command validator-agent runs on a validator VM. It is stateless: every +// decision it makes is driven by polling the Control API, per the +// deployment constraint that validators must be able to restart freely +// without any local state to reconcile. +package main + +import ( + "context" + "flag" + "log/slog" + "net" + "os" + "os/signal" + "strconv" + "strings" + "syscall" + + "cloudipvalidator/internal/agentcore" + "cloudipvalidator/internal/config" +) + +func main() { + configPath := flag.String("config", "configs/validator-agent.yaml", "path to validator-agent config file") + stubPorts := flag.String("stub-ports", "", "TEST ONLY: comma-separated TCP ports to accept-and-close on, standing in for the validator's real listening services in the offline end-to-end harness (see docs/LOCAL_E2E.md)") + flag.Parse() + + log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo})) + + cfg, err := config.LoadValidatorAgent(*configPath) + if err != nil { + log.Error("load config", "err", err) + os.Exit(1) + } + + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + + if *stubPorts != "" { + startStubListeners(ctx, log, *stubPorts) + } + + agent := agentcore.New(cfg, log) + if err := agent.Run(ctx); err != nil && err != context.Canceled { + log.Error("agent stopped", "err", err) + os.Exit(1) + } +} + +// startStubListeners binds trivial accept-and-close TCP listeners on the +// given ports, standing in for the base-minimum services (22/80/443/8080) +// a real validator would already run, so an offline prober has something +// to successfully connect to. ICMP needs no stub: the kernel answers echo +// requests to any locally-bound address on its own. +func startStubListeners(ctx context.Context, log *slog.Logger, portsCSV string) { + for _, p := range strings.Split(portsCSV, ",") { + p = strings.TrimSpace(p) + if p == "" { + continue + } + port, err := strconv.Atoi(p) + if err != nil { + log.Error("invalid stub port", "value", p, "err", err) + continue + } + ln, err := net.Listen("tcp", ":"+strconv.Itoa(port)) + if err != nil { + log.Error("stub listener", "port", port, "err", err) + continue + } + log.Info("stub listener up", "port", port) + go func() { + <-ctx.Done() + ln.Close() + }() + go func() { + for { + conn, err := ln.Accept() + if err != nil { + return + } + conn.Close() + } + }() + } +} diff --git a/configs/control-api.example.yaml b/configs/control-api.example.yaml new file mode 100644 index 0000000..b6223e4 --- /dev/null +++ b/configs/control-api.example.yaml @@ -0,0 +1,84 @@ +# Control API configuration. +# +# OpenStack credentials are never set here — only the *names* of the +# environment variables to read them from. The actual values must be +# supplied by the process environment (see deploy/systemd/control-api.service +# and its EnvironmentFile=). + +server: + listen_addr: ":8080" + +database: + path: "/var/lib/cloud-ip-validator/control-api.db" + +openstack: + mode: "real" # "mock" | "real" — mock uses an in-memory + # OpenStack stand-in for local dev/testing + auth_url_env: "OS_AUTH_URL" + token_env: "OS_TOKEN" + project_id_env: "OS_PROJECT_ID" + project_name_env: "OS_PROJECT_NAME" + project_domain_env: "OS_PROJECT_DOMAIN_NAME" + region_env: "OS_REGION_NAME" + +orchestrator: + poll_interval_seconds: 5 + self_check_timeout_seconds: 60 + max_self_check_retries: 3 + checking_window_seconds: 120 + max_retries: 3 + lease_ttl_seconds: 180 + heartbeat_timeout_seconds: 30 + +aggregation: + missing_counts_as_fail: true + +# Validators are VMs in the service project; os_port_id is the Neutron port +# ID of each validator's primary NIC, used when associating a floating IP. +validators: + - validator_id: "validator_01" + os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_1" + - validator_id: "validator_02" + os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_2" + +# The three external prober sites, indexed 1-3 (matches ip_queue.siteN_complete). +sites: + - site_id: "site-1" + index: 1 + - site_id: "site-2" + index: 2 + - site_id: "site-3" + index: 3 + +# Outbound/egress check types the validator-agent runs, and which target +# group (below) each runs against. +check_types: + - name: "https" + enabled: true + targets: ["default-targets"] + - name: "icmp" + enabled: true + targets: ["default-targets"] + - name: "ssh" + enabled: false + targets: [] + +targets: + default-targets: + - "https://hub.docker.com" + - "https://github.com" + - "https://packages.ubuntu.com" + +# Inbound checks the 3 external-site probers run directly against each +# validator's currently-assigned floating IP. +inbound_checks: + ports: [22, 80, 443, 8080] + icmp: true + +# The pool of public IPv4 addresses to validate, in the order they'll be +# processed (ip_queue.sequence). Every address here is checked through to +# the end of the list. +ip_addresses: + - "203.0.113.10" + - "203.0.113.11" + - "203.0.113.12" diff --git a/configs/prober.example.yaml b/configs/prober.example.yaml new file mode 100644 index 0000000..441b1d0 --- /dev/null +++ b/configs/prober.example.yaml @@ -0,0 +1,11 @@ +# Prober configuration. Runs on one of the 3 external test sites. site_id +# must match one of the control-api config's `sites[].site_id`. + +site_id: "site-1" +control_api_url: "http://control-api.internal:8080" +poll_interval_seconds: 5 + +checks: + tcp_timeout_seconds: 5 + icmp_timeout_seconds: 5 + icmp_count: 3 diff --git a/configs/validator-agent.example.yaml b/configs/validator-agent.example.yaml new file mode 100644 index 0000000..8a7f6bb --- /dev/null +++ b/configs/validator-agent.example.yaml @@ -0,0 +1,17 @@ +# Validator-agent configuration. Runs on each validator VM. validator_id +# must match one of the control-api config's `validators[].validator_id`. + +validator_id: "validator_01" +control_api_url: "http://control-api.internal:8080" +poll_interval_seconds: 5 + +self_check: + timeout_seconds: 10 + +checks: + https_timeout_seconds: 10 + icmp_timeout_seconds: 5 + icmp_count: 3 + ssh: + enabled: false + timeout_seconds: 5 diff --git a/deploy/systemd/control-api.service b/deploy/systemd/control-api.service new file mode 100644 index 0000000..5ffbb2e --- /dev/null +++ b/deploy/systemd/control-api.service @@ -0,0 +1,24 @@ +[Unit] +Description=Cloud IP Validator - Control API +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=cloud-ip-validator +Group=cloud-ip-validator +ExecStart=/usr/local/bin/control-api -config /etc/cloud-ip-validator/control-api.yaml +# Holds OS_AUTH_URL / OS_TOKEN / OS_PROJECT_ID / etc — the admin credential +# for OpenStack Floating IP management. Keep this file mode 0600, owned by +# the service user; never commit it or put credentials in the YAML config. +EnvironmentFile=/etc/cloud-ip-validator/control-api.env +WorkingDirectory=/var/lib/cloud-ip-validator +Restart=on-failure +RestartSec=5 +NoNewPrivileges=true +ProtectSystem=strict +ReadWritePaths=/var/lib/cloud-ip-validator +PrivateTmp=true + +[Install] +WantedBy=multi-user.target diff --git a/deploy/systemd/prober.service b/deploy/systemd/prober.service new file mode 100644 index 0000000..dab51a4 --- /dev/null +++ b/deploy/systemd/prober.service @@ -0,0 +1,22 @@ +[Unit] +Description=Cloud IP Validator - Prober +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=cloud-ip-validator +Group=cloud-ip-validator +ExecStart=/usr/local/bin/prober -config /etc/cloud-ip-validator/prober.yaml +Restart=on-failure +RestartSec=5 +NoNewPrivileges=true +# Unprivileged ICMP echo requires either root or CAP_NET_RAW; grant only +# the latter rather than running the prober as root. +AmbientCapabilities=CAP_NET_RAW +CapabilityBoundingSet=CAP_NET_RAW +ProtectSystem=strict +PrivateTmp=true + +[Install] +WantedBy=multi-user.target diff --git a/deploy/systemd/validator-agent.service b/deploy/systemd/validator-agent.service new file mode 100644 index 0000000..4629a50 --- /dev/null +++ b/deploy/systemd/validator-agent.service @@ -0,0 +1,22 @@ +[Unit] +Description=Cloud IP Validator - Validator Agent +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=cloud-ip-validator +Group=cloud-ip-validator +ExecStart=/usr/local/bin/validator-agent -config /etc/cloud-ip-validator/validator-agent.yaml +Restart=on-failure +RestartSec=5 +NoNewPrivileges=true +# Unprivileged ICMP echo requires either root or CAP_NET_RAW; grant only +# the latter rather than running the agent as root. +AmbientCapabilities=CAP_NET_RAW +CapabilityBoundingSet=CAP_NET_RAW +ProtectSystem=strict +PrivateTmp=true + +[Install] +WantedBy=multi-user.target diff --git a/docs/API.md b/docs/API.md new file mode 100644 index 0000000..f4406d9 --- /dev/null +++ b/docs/API.md @@ -0,0 +1,376 @@ +# API Control API + +Control API — единственная точка входа в систему для `validator-agent`, +`prober` и оператора (администратора). Все данные передаются в формате +JSON, базовый префикс прикладных методов — `/api/v1`. + +> **Важно.** На данный момент API не защищён аутентификацией/авторизацией +> — эндпоинты доступны любому, кто может достучаться до порта control-api +> по сети. Для эксплуатации за пределами доверенного сегмента сети +> обязательно ограничьте доступ на уровне сети/файрвола (см. +> [SETUP.md](SETUP.md#сетевые-доступы)). Добавление bearer-токена — известное +> направление доработки, в текущей версии не реализовано. + +Базовый URL в примерах — `http://control-api.internal:8080`, замените на +адрес вашего стенда (см. `server.listen_addr` в конфиге control-api). + +## Содержание + +- [Общие соглашения](#общие-соглашения) +- [Методы для validator-agent](#методы-для-validator-agent) +- [Методы для prober](#методы-для-prober) +- [Служебные и административные методы](#служебные-и-административные-методы) +- [Модель состояний и связь методов с ней](#модель-состояний-и-связь-методов-с-ней) +- [Сквозной пример работы (curl)](#сквозной-пример-работы-curl) + +## Общие соглашения + +- Тело запроса и ответа — JSON (`Content-Type: application/json`). +- Успешные ответы возвращают `200 OK`, либо `204 No Content` (когда + данных нет — например, у валидатора сейчас нет назначения). +- Ошибки возвращают `4xx`/`5xx` и тело вида: + ```json + {"error": "текст ошибки"} + ``` +- Временные метки (`checked_at` в запросах) передаются в формате + RFC3339/RFC3339Nano, например `2026-08-21T09:15:00.123456789Z`. Если поле + не удалось распарсить, сервер молча подставит текущее время сервера — не + полагайтесь на это в продакшене, всегда передавайте валидную метку. +- `validator_id` и `site_id` в пути запроса должны совпадать со + значениями, заданными в конфиге control-api (`validators[].validator_id`, + `sites[].site_id`) — иначе методы, требующие существующую сущность, + вернут `404`. + +## Методы для validator-agent + +Эти методы вызывает бинарник `validator-agent`, работающий на ВМ-валидаторе. +Оператору вручную дёргать их обычно не требуется — они приведены для +понимания протокола и для отладки через curl. + +### `POST /api/v1/agents/register` + +Регистрация/переактивация валидатора. Вызывается один раз при старте +агента (и безопасно при каждом рестарте — идемпотентна). + +Запрос: +```json +{ + "validator_id": "validator_01", + "hostname": "vm-validator-01", + "agent_version": "1.0.0" +} +``` + +Ответ: +```json +{"ok": true, "poll_interval_seconds": 5} +``` + +`poll_interval_seconds` — рекомендованный интервал опроса, значение берётся +из `orchestrator.poll_interval_seconds` конфига control-api. + +> `validator_id` должен быть заранее описан в конфиге control-api +> (`validators[].validator_id`) вместе с `os_port_id` — сам агент порт ID +> не передаёт и не может его сменить через API. + +### `POST /api/v1/agents/{id}/heartbeat` + +"Я жив". Обновляет `last_heartbeat_at` валидатора. Если валидатор не +присылает heartbeat дольше `orchestrator.heartbeat_timeout_seconds`, он +помечается `unreachable`. + +Запрос (тело необязательно, поля информационные): +```json +{"local_state": "idle"} +``` + +Ответ: `{"ok": true}`. `404`, если `validator_id` не зарегистрирован. + +### `GET /api/v1/agents/{id}/assignment` + +Есть ли у валидатора сейчас работа. Опрашивается в каждом цикле. + +- `204 No Content` — заданий нет. +- `200 OK` с телом: + ```json + { + "ip_id": 42, + "ip_address": "203.0.113.10", + "phase": "awaiting_self_check", + "check_config": [ + {"type": "https", "targets": ["https://hub.docker.com", "https://github.com", "https://packages.ubuntu.com"]}, + {"type": "icmp", "targets": ["https://hub.docker.com", "https://github.com", "https://packages.ubuntu.com"]} + ] + } + ``` + +`phase` — `awaiting_self_check` (нужно выполнить self-check) либо +`checking` (self-check уже пройден, можно/нужно выполнять проверки). +`check_config` — уже развёрнутая конфигурация проверок (тип + список +целей), агенту не нужно самому сопоставлять группы целей. + +### `POST /api/v1/agents/{id}/self-check` + +Отчёт о результате self-check — подтверждение, что исходящий трафик +валидатора действительно идёт через только что назначенный FIP. Агент +определяет это, вызвав `GET /api/v1/whatsmyip` и сравнив ответ с +`ip_address` из задания. + +Запрос: +```json +{ + "ip_id": 42, + "detected_egress_ip": "203.0.113.10", + "success": true, + "detail": "matched" +} +``` + +Ответ: `{"ok": true}`. При `success: false` control-api сам решает — +повторить попытку назначения FIP или пометить IP как `failed` (после +исчерпания `orchestrator.max_self_check_retries`). + +### `POST /api/v1/agents/{id}/events` + +Произвольная запись в журнал аудита, привязанная (опционально) к IP. +Используется агентом для событий `config_received`, `self_check_result` +и т.п. + +Запрос: +```json +{ + "event_type": "config_received", + "ip_id": 42, + "payload": "{\"checks\":2}" +} +``` + +Ответ: `{"ok": true}`. + +### `POST /api/v1/agents/{id}/results` + +Отчёт о результатах исходящих (egress) проверок. Можно отправлять по +одной проверке сразу после выполнения (рекомендуется — так прогресс не +теряется при падении агента) либо пачкой. + +Запрос: +```json +{ + "results": [ + { + "ip_id": 42, + "check_type": "https", + "target": "https://github.com", + "success": true, + "latency_ms": 87, + "detail": "ok", + "checked_at": "2026-08-21T09:15:00.123Z" + } + ] +} +``` + +Ответ: `{"ok": true}`. Повторная отправка того же `(ip_id, check_type, +target)` в рамках текущей попытки — безопасна и просто перезапишет +результат (upsert по уникальному ключу). + +### `POST /api/v1/agents/{id}/complete` + +Сигнал "все исходящие проверки для этого IP выполнены". + +Запрос: +```json +{"ip_id": 42} +``` + +Ответ: `{"ok": true}`. + +## Методы для prober + +Эти методы вызывает бинарник `prober`, работающий на внешней площадке. + +### `POST /api/v1/probers/register` + +Регистрация пробера. `site_id` должен присутствовать в конфиге control-api +(`sites[].site_id`), иначе — `400`. + +Запрос: +```json +{"site_id": "site-1", "hostname": "probe-host-1"} +``` + +Ответ: `{"ok": true, "poll_interval_seconds": 5}`. + +### `GET /api/v1/probers/{site_id}/assignments` + +Список всех IP, которые сейчас находятся в состоянии `checking` — то есть +всё, что нужно прозондировать на этом цикле опроса (валидаторов может +работать несколько параллельно, поэтому список, а не один IP). + +Ответ: +```json +[ + {"ip_id": 42, "ip_address": "203.0.113.10", "ports": [22, 80, 443, 8080], "icmp": true} +] +``` + +Пустой список `[]`, если сейчас нечего проверять. + +### `POST /api/v1/probers/{site_id}/results` + +Отчёт о результатах входящих (inbound) проверок с данной площадки. + +Запрос: +```json +{ + "results": [ + {"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-22", "success": true, "latency_ms": 12, "checked_at": "2026-08-21T09:15:01Z", "complete": false}, + {"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-80", "success": true, "latency_ms": 9, "checked_at": "2026-08-21T09:15:01Z", "complete": false}, + {"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-443", "success": true, "latency_ms": 10, "checked_at": "2026-08-21T09:15:01Z", "complete": false}, + {"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "tcp-8080","success": false,"latency_ms": 0, "checked_at": "2026-08-21T09:15:01Z", "complete": false}, + {"ip_id": 42, "ip_address": "203.0.113.10", "check_type": "icmp", "success": true, "latency_ms": 5, "checked_at": "2026-08-21T09:15:01Z", "complete": true} + ] +} +``` + +`complete: true` нужно проставить ровно на одном (обычно последнем) +результате в пачке — это сигнал "площадка N закончила зондирование этого +IP на данном проходе". До этого момента control-api не будет считать +данные с этой площадки завершёнными. + +Ответ: `{"ok": true}`. + +## Служебные и административные методы + +### `GET /healthz` + +Проверка живости процесса. Ответ: `{"ok": true}`. Используется в systemd/ +внешних системах мониторинга. + +### `GET /api/v1/whatsmyip` + +Возвращает IP-адрес, с которого пришёл TCP-запрос (без учёта заголовков +`X-Forwarded-For` — специально, чтобы self-check нельзя было подделать). +Это основа механизма self-check. + +Ответ: +```json +{"ip": "203.0.113.10"} +``` + +### `GET /api/v1/admin/status` + +Сводка по очереди — сколько IP в каком состоянии. + +```json +{ + "total_ips": 25, + "ips_by_state": {"queued": 10, "checking": 3, "done": 11, "failed": 1}, + "total_validators": 4 +} +``` + +### `GET /api/v1/admin/ips` + +Полный список всех IP из очереди со всеми полями (см. +[USAGE.md](USAGE.md#значения-полей-ip) — расшифровка полей и статусов). + +### `GET /api/v1/admin/ips/{ip}` + +Детали по одному адресу: сам объект IP, все проверки текущей попытки и +вся история событий по нему. + +```json +{ + "ip": { "ID": 42, "IPAddress": "203.0.113.10", "State": "done", "OverallResult": "pass", "...": "..." }, + "checks": [ {"Source": "egress", "CheckType": "https", "Target": "https://github.com", "Success": true, "...": "..."} ], + "events": [ {"EventType": "fip_associated", "OccurredAt": "...", "...": "..."} ] +} +``` + +> Обратите внимание: вложенные объекты `ip`/`checks`/`events` сериализуются +> без переопределения имён полей (используются имена Go-структур, например +> `IPAddress`, `State`, `Success`) — в отличие от методов для +> agent/prober, где поля в `snake_case`. Это осознанная асимметрия: +> административные методы — для человека/дашборда, а не для машинного +> протокола. + +### `GET /api/v1/admin/validators` + +Список всех валидаторов с их текущим состоянием (`unregistered`, `idle`, +`assigned`, `checking`, `unreachable`) и `CurrentIPID`, если валидатор +сейчас занят. + +## Модель состояний и связь методов с ней + +``` +queued ──(control-api сам, без вызова API)──▶ assigning_fip + │ + OpenStack FIP associate успешен + ▼ + awaiting_self_check + │ + POST .../self-check {success:true} + ▼ + checking + │ POST .../results (agent) │ POST .../results (prober, x3 площадки) + ▼ ▼ + egress_complete=true siteN_complete=true (N=1,2,3) + │ + все complete=true ИЛИ истекло checking_window_seconds + ▼ + aggregating + │ + done (pass/partial/fail) + или failed +``` + +Переходы `queued → assigning_fip → awaiting_self_check` и финальная +агрегация выполняются control-api самостоятельно по таймеру (см. +`orchestrator.poll_interval_seconds`), явного HTTP-метода для их запуска +нет — это фоновый цикл (`Tick`), а не запрос/ответ. + +## Сквозной пример работы (curl) + +Ниже — минимальный ручной прогон одного IP через API, как если бы вы +писали собственного клиента вместо `validator-agent`/`prober`. Полезно +для отладки и для понимания протокола. + +```bash +BASE=http://127.0.0.1:8080 + +# 1. Регистрация валидатора (validator_01 уже должен быть в конфиге control-api) +curl -s -X POST "$BASE/api/v1/agents/register" \ + -d '{"validator_id":"validator_01","hostname":"debug-host","agent_version":"manual"}' + +# 2. Подождать, пока control-api (фоновым тиком) назначит IP и привяжет FIP — +# проверяем через admin/status или admin/ips, либо просто опрашиваем assignment +curl -s "$BASE/api/v1/agents/validator_01/assignment" +# => {"ip_id":1,"ip_address":"203.0.113.10","phase":"awaiting_self_check","check_config":[...]} + +# 3. Self-check: спросить у control-api, каким адресом мы к нему пришли +curl -s "$BASE/api/v1/whatsmyip" +# => {"ip":"203.0.113.10"} (в реальном стенде это и есть проверка через FIP) + +curl -s -X POST "$BASE/api/v1/agents/validator_01/self-check" \ + -d '{"ip_id":1,"detected_egress_ip":"203.0.113.10","success":true,"detail":"matched"}' + +# 4. Отправить результаты egress-проверок (по одному check_config пункту) +curl -s -X POST "$BASE/api/v1/agents/validator_01/results" \ + -d '{"results":[{"ip_id":1,"check_type":"https","target":"https://github.com","success":true,"latency_ms":80,"checked_at":"2026-08-21T09:00:00Z"}]}' + +# 5. Сообщить, что все egress-проверки выполнены +curl -s -X POST "$BASE/api/v1/agents/validator_01/complete" -d '{"ip_id":1}' + +# 6. Со стороны пробера: узнать, что сейчас проверяется, и отправить результат +curl -s -X POST "$BASE/api/v1/probers/register" -d '{"site_id":"site-1"}' +curl -s "$BASE/api/v1/probers/site-1/assignments" +curl -s -X POST "$BASE/api/v1/probers/site-1/results" \ + -d '{"results":[{"ip_id":1,"ip_address":"203.0.113.10","check_type":"icmp","success":true,"latency_ms":5,"checked_at":"2026-08-21T09:00:01Z","complete":true}]}' + +# 7. Проверить итоговый результат (после того как control-api агрегирует) +curl -s "$BASE/api/v1/admin/ips/203.0.113.10" | python3 -m json.tool +``` + +Для полностью автоматизированного локального прогона (без ручных curl) +см. `scripts/run-local-e2e.sh` и [docs/LOCAL_E2E.md](LOCAL_E2E.md). diff --git a/docs/LOCAL_E2E.md b/docs/LOCAL_E2E.md new file mode 100644 index 0000000..f9dfccd --- /dev/null +++ b/docs/LOCAL_E2E.md @@ -0,0 +1,92 @@ +# Local end-to-end smoke test + +`scripts/run-local-e2e.sh` runs the full system as local processes with no +real OpenStack cloud and no real internet access: + +- **control-api** with `openstack.mode: mock` — the in-memory + `openstack.MockClient` stands in for Neutron, pre-seeded with one + synthetic floating IP per configured address (see + `cmd/control-api/main.go`'s `newOpenStackClient`). +- **1 validator-agent** (`validator_01`), started with + `-stub-ports 12022,18081,18443,18888` — trivial accept-and-close TCP + listeners standing in for the base-minimum services (22/80/443/8080) a + real validator would run. ICMP needs no stub: the kernel answers echo + requests to any local address (127.0.0.0/8) on its own. +- **3 probers** (`site-1`/`site-2`/`site-3`), all probing `127.0.0.1`. +- **3 `httpstub` instances** (`scripts/httpstub`) standing in for the real + outbound targets (hub.docker.com / github.com / packages.ubuntu.com), + always returning `200 OK`. + +The generated config uses a short `lease_ttl_seconds: 8` and +`checking_window_seconds: 15` so the whole run finishes in well under a +minute instead of using the (much longer) production defaults. + +**Why only one IP address (`127.0.0.1`), and why it must be exactly that +one:** the self-check mechanism (`GET /whatsmyip`, see +`internal/httpapi/server.go`'s `remoteIP`) compares the TCP source address +the validator-agent's own outbound connection to control-api arrives with +against the address it was just assigned. In a real deployment, Neutron +actually SNATs the validator's egress traffic through whichever floating +IP is attached, so any configured address self-checks correctly. This +offline harness has no real network-level SNAT — the agent's traffic to +control-api always really originates from `127.0.0.1` — so only that +literal loopback address can ever pass self-check here. This is a +limitation of the harness's fidelity, not of the self-check mechanism +itself. + +## Running it + +``` +scripts/run-local-e2e.sh +``` + +It will: + +1. Build all three binaries plus `httpstub` into a temp workdir. +2. Start the 3 stub HTTP targets, control-api, the validator-agent, and the + 3 probers. +3. Wait for `GET /healthz` to come up. +4. A few seconds in, **kill the validator-agent mid-run** and wait past the + 8s lease TTL, to demonstrate that control-api's lease sweep reclaims the + in-flight IP (moves it back to `queued`, bumps `retry_count`) without + any special crash-recovery code — it's the same sweep that runs every + tick. It then restarts the validator-agent so the queue can finish. +5. Poll `GET /api/v1/admin/status` until every configured IP has reached a + terminal state (`done` or `failed`). +6. Print the final `/api/v1/admin/status` and `/api/v1/admin/ips` output. + +Expect to see `127.0.0.1` end with `"state":"done"` and +`"overall_result":"pass"` (all egress checks against the stub targets +succeed, and all 3 probers can reach the stub TCP listeners and get ICMP +replies from loopback). + +## Inspecting a run + +The workdir (printed at the end, `/tmp/cloud-ip-validator-e2e.XXXXXX`) is +**not** deleted automatically, so you can inspect: + +- `logs/control-api.log`, `logs/validator-agent.log`, `logs/prober-site-*.log` +- `control-api.db` — open with `sqlite3` to inspect the `checks` and + `events` tables directly, e.g.: + ``` + sqlite3 /tmp/cloud-ip-validator-e2e.XXXXXX/control-api.db \ + "select ip_address, source, check_type, success from checks order by id" + ``` + +## What this does *not* cover + +This harness proves the orchestration, HTTP protocol, and check-running +logic all work together correctly. It does **not** exercise the real +`internal/openstack/client.go` (gophercloud) path — that only runs against +`openstack.mode: real` with actual OpenStack credentials. That path has its +own read-only smoke test, `internal/openstack/client_live_test.go`, skipped +by default and gated behind `OPENSTACK_LIVE_TEST=1`: + +``` +OPENSTACK_LIVE_TEST=1 \ +OS_AUTH_URL=https://keystone.example:5000/v3 \ +OS_TOKEN=... \ +OS_PROJECT_ID=... \ +OS_TEST_FLOATING_IP=203.0.113.10 \ +go test ./internal/openstack/... -run TestClientLive -v +``` diff --git a/docs/SETUP.md b/docs/SETUP.md new file mode 100644 index 0000000..7b81a96 --- /dev/null +++ b/docs/SETUP.md @@ -0,0 +1,275 @@ +# Подготовка стенда и первичная инициализация + +Документ описывает, как собрать компоненты, подготовить конфигурацию и +запустить стенд с нуля — от чистой машины до работающего control-api, +валидаторов и проберов. Если нужно просто быстро посмотреть систему в +работе без реального OpenStack — сразу переходите к разделу +[«Быстрая проверка без OpenStack»](#быстрая-проверка-без-openstack-offline-режим). + +## Содержание + +- [Компоненты и роли машин](#компоненты-и-роли-машин) +- [Требования](#требования) +- [Сборка бинарников](#сборка-бинарников) +- [Быстрая проверка без OpenStack (offline-режим)](#быстрая-проверка-без-openstack-offline-режим) +- [Подготовка конфигурации для реального стенда](#подготовка-конфигурации-для-реального-стенда) +- [Развёртывание control-api](#развёртывание-control-api) +- [Развёртывание validator-agent на ВМ-валидаторах](#развёртывание-validator-agent-на-вм-валидаторах) +- [Развёртывание prober на внешних площадках](#развёртывание-prober-на-внешних-площадках) +- [Проверка после запуска](#проверка-после-запуска) +- [Сетевые доступы](#сетевые-доступы) + +## Компоненты и роли машин + +| Компонент | Где запускается | Кол-во | +|---|---|---| +| `control-api` | Отдельная управляющая машина/ВМ с доступом к OpenStack API | 1 (без HA) | +| `validator-agent` | Каждая ВМ-валидатор в сервисном проекте облака | по числу валидаторов | +| `prober` | По одному на каждой из внешних тестовых площадок | 3 (по числу площадок) | + +`control-api` — единственный компонент с состоянием (SQLite). Валидаторы и +проберы не хранят локального состояния и полностью управляются через опрос +control-api (см. [API.md](API.md)). + +## Требования + +- **Go 1.22+** для сборки (проверено на Go 1.26). Собранные бинарники — + статические, дополнительных зависимостей на целевых машинах не требуют + (используется чистый Go-драйвер SQLite, без cgo). +- Для `control-api` в боевом режиме (`openstack.mode: real`) — учётная + запись OpenStack с правами на чтение/изменение floating IP (Neutron) + в сервисном проекте, и заранее выделенные (allocated) floating IP — + инструмент их **не создаёт**, только привязывает/отвязывает + существующие. +- Для `validator-agent` и `prober` — возможность отправлять ICMP echo + (нужен root либо capability `CAP_NET_RAW`, см. юниты systemd). +- `curl`, `sqlite3` (опционально, для ручной инспекции БД) на машине с + control-api пригодятся для диагностики. + +## Сборка бинарников + +Из корня репозитория: + +```bash +export PATH=$PATH:/usr/local/go/bin # если go не в PATH +go build -o bin/control-api ./cmd/control-api +go build -o bin/validator-agent ./cmd/validator-agent +go build -o bin/prober ./cmd/prober +``` + +Каждый бинарник самодостаточен — скопируйте нужный файл на +соответствующую машину (control-api → управляющая машина, validator-agent +→ каждый валидатор, prober → каждая площадка). + +Убедиться, что всё собирается и юнит-тесты проходят: + +```bash +go build ./... && go test ./... +``` + +## Быстрая проверка без OpenStack (offline-режим) + +Прежде чем разворачивать реальный стенд, рекомендуется убедиться, что всё +собирается и работает корректно на локальной машине — без облака и +внешних площадок: + +```bash +scripts/run-local-e2e.sh +``` + +Скрипт сам поднимает control-api (в режиме `openstack.mode: mock`), +одного validator-agent и трёх проберов как локальные процессы, прогоняет +один тестовый адрес через полный цикл проверки и печатает итоговый +результат. Подробности — в [docs/LOCAL_E2E.md](LOCAL_E2E.md). Это же +хороший способ разобраться в поведении системы перед первым боевым +запуском. + +## Подготовка конфигурации для реального стенда + +Все три компонента конфигурируются YAML-файлами. Шаблоны лежат в +`configs/*.example.yaml` — скопируйте их и заполните под ваш стенд. + +### 1. `control-api.yaml` + +```bash +cp configs/control-api.example.yaml /etc/cloud-ip-validator/control-api.yaml +``` + +Что обязательно нужно заполнить: + +- **`validators`** — список валидаторов, у каждого `validator_id` + (произвольное имя, должно совпадать с `validator_id` в конфиге + соответствующего `validator-agent`) и `os_port_id` — **ID Neutron-порта** + основного сетевого интерфейса ВМ-валидатора (узнать: `openstack port + list --server <имя-ВМ>` или в веб-консоли облака). +- **`sites`** — три внешние площадки, `site_id` + `index` (1, 2 или 3). + `site_id` должен совпадать с `site_id` в конфиге соответствующего + `prober`. +- **`ip_addresses`** — список публичных IPv4-адресов на проверку, **в + порядке обработки**. Адреса должны существовать в сервисном проекте как + уже выделенные (allocated) floating IP — инструмент их не создаёт. +- **`openstack.mode: "real"`** и `*_env` поля — имена переменных + окружения, из которых будут прочитаны реальные учётные данные (сами + значения в этот файл **не пишутся**, см. следующий пункт). +- **`targets`** и **`check_types`** — при необходимости смените набор + целей для egress-проверок (по умолчанию — hub.docker.com, github.com, + packages.ubuntu.com) или включите `ssh` (по умолчанию выключен). + +### 2. Переменные окружения для OpenStack + +Учётные данные передаются **только** через переменные окружения — никогда +через YAML. Создайте файл (доступный на чтение только сервисному +пользователю): + +```bash +install -m 0600 -o cloud-ip-validator -g cloud-ip-validator /dev/null /etc/cloud-ip-validator/control-api.env +cat >> /etc/cloud-ip-validator/control-api.env <<'EOF' +OS_AUTH_URL=https://keystone.example.com:5000/v3 +OS_TOKEN=<токен администратора с правами на управление floating IP> +OS_PROJECT_ID= +OS_REGION_NAME=<регион> +EOF +``` + +Имена переменных должны совпадать с тем, что указано в +`control-api.yaml` в секции `openstack` (`auth_url_env`, `token_env` и +т.д.) — в шаблоне это ровно `OS_AUTH_URL`, `OS_TOKEN`, `OS_PROJECT_ID`, +`OS_PROJECT_NAME`, `OS_PROJECT_DOMAIN_NAME`, `OS_REGION_NAME`. + +### 3. `validator-agent.yaml` (свой на каждом валидаторе) + +```bash +cp configs/validator-agent.example.yaml /etc/cloud-ip-validator/validator-agent.yaml +``` + +Обязательно поменять: +- `validator_id` — должен совпадать с одним из `validators[].validator_id` + в конфиге control-api. +- `control_api_url` — адрес, по которому эта ВМ достучится до control-api. + +### 4. `prober.yaml` (свой на каждой площадке) + +```bash +cp configs/prober.example.yaml /etc/cloud-ip-validator/prober.yaml +``` + +Обязательно поменять: +- `site_id` — должен совпадать с одним из `sites[].site_id` в конфиге + control-api (для трёх площадок — три разных файла с `site-1`, + `site-2`, `site-3` или как вы их назвали). +- `control_api_url` — адрес control-api, доступный с площадки (обычно + через интернет — площадки внешние). + +## Развёртывание control-api + +```bash +useradd --system --no-create-home --shell /usr/sbin/nologin cloud-ip-validator +mkdir -p /var/lib/cloud-ip-validator /etc/cloud-ip-validator +chown cloud-ip-validator:cloud-ip-validator /var/lib/cloud-ip-validator + +cp bin/control-api /usr/local/bin/control-api +cp deploy/systemd/control-api.service /etc/systemd/system/ + +systemctl daemon-reload +systemctl enable --now control-api +``` + +**Первичная инициализация базы данных происходит автоматически** — при +первом старте `control-api` создаёт файл SQLite по пути `database.path` +из конфига (миграция схемы применяется один раз, повторные запуски — +no-op). Отдельной команды "init db" не требуется. + +При каждом старте control-api также: +1. Регистрирует в БД всех валидаторов из `validators` конфига (если их + там ещё нет). +2. Добавляет в очередь все адреса из `ip_addresses`, которых там ещё нет + (уже обработанные ранее адреса повторно не добавляются и не + сбрасываются — см. [USAGE.md](USAGE.md#добавление-новых-ip-в-очередь)). + +Проверить, что процесс поднялся: + +```bash +curl -s http://localhost:8080/healthz +# {"ok":true} +journalctl -u control-api -f +``` + +## Развёртывание validator-agent на ВМ-валидаторах + +Повторить на каждой ВМ-валидаторе: + +```bash +cp bin/validator-agent /usr/local/bin/validator-agent +cp deploy/systemd/validator-agent.service /etc/systemd/system/ +mkdir -p /etc/cloud-ip-validator +# скопировать сюда заполненный validator-agent.yaml с уникальным validator_id + +systemctl daemon-reload +systemctl enable --now validator-agent +journalctl -u validator-agent -f +``` + +Юнит выдаёт процессу capability `CAP_NET_RAW` (без root) — она нужна для +отправки ICMP echo в рамках проверок. + +## Развёртывание prober на внешних площадках + +Аналогично, на каждой из трёх площадок: + +```bash +cp bin/prober /usr/local/bin/prober +cp deploy/systemd/prober.service /etc/systemd/system/ +mkdir -p /etc/cloud-ip-validator +# скопировать сюда prober.yaml с уникальным site_id для этой площадки + +systemctl daemon-reload +systemctl enable --now prober +journalctl -u prober -f +``` + +## Проверка после запуска + +После того как control-api, все валидаторы и все три пробера запущены: + +```bash +curl -s http://:8080/api/v1/admin/status | python3 -m json.tool +``` + +Ожидаемая картина сразу после старта: часть адресов в состоянии `queued`, +часть уже переходит в `assigning_fip`/`awaiting_self_check`/`checking` по +мере того, как освобождаются валидаторы. Через некоторое время появляются +записи в `done`/`failed`. Подробнее о том, как читать этот вывод и что +делать дальше — в [USAGE.md](USAGE.md). + +Также стоит убедиться, что все валидаторы видны и не «зависли»: + +```bash +curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool +``` + +Все зарегистрированные валидаторы должны рано или поздно оказываться в +состоянии `idle` (между заданиями) — если валидатор надолго застрял в +`unreachable`, проверьте сетевую связность до control-api и логи агента +(`journalctl -u validator-agent`). + +## Сетевые доступы + +Минимально необходимая связность: + +- `validator-agent` → `control-api`: TCP, порт из `server.listen_addr` + (обычно 8080). +- `prober` (на каждой из 3 площадок) → `control-api`: тот же порт, обычно + через интернет. +- `prober` → адрес, который в данный момент проверяется (динамический, + меняется по ходу работы очереди): TCP 22/80/443/8080 + ICMP — + собственно и есть проверяемый трафик, его нельзя заранее ограничить + одним IP. +- `validator-agent` → интернет: HTTPS/ICMP до целей из `targets` конфига + (по умолчанию hub.docker.com, github.com, packages.ubuntu.com) — именно + через floating IP, который в данный момент привязан к валидатору. +- `control-api` → OpenStack Keystone/Neutron API (`OS_AUTH_URL` и далее по + каталогу сервисов). + +API control-api сейчас не аутентифицирован (см. предупреждение в начале +[API.md](API.md)) — ограничивайте доступ к порту control-api на уровне +сети/firewall теми хостами, где реально работают валидаторы и проберы. diff --git a/docs/USAGE.md b/docs/USAGE.md new file mode 100644 index 0000000..364e56d --- /dev/null +++ b/docs/USAGE.md @@ -0,0 +1,269 @@ +# Работа со стендом + +Этот документ — для оператора, который уже развернул стенд (см. +[SETUP.md](SETUP.md)) и теперь использует его в повседневной работе: +добавляет адреса на проверку, следит за очередью, разбирается в +результатах и реагирует на проблемы. Прямые вызовы API описаны в +[API.md](API.md) — здесь мы используем их только как инструмент, не +углубляясь в протокол. + +## Содержание + +- [Как устроена работа с системой](#как-устроена-работа-с-системой) +- [Добавление новых IP в очередь](#добавление-новых-ip-в-очередь) +- [Наблюдение за очередью](#наблюдение-за-очередью) +- [Значения полей IP](#значения-полей-ip) +- [Как читать итоговый результат (pass/partial/fail)](#как-читать-итоговый-результат-passpartialfail) +- [Просмотр деталей и истории по конкретному адресу](#просмотр-деталей-и-истории-по-конкретному-адресу) +- [Управление валидаторами](#управление-валидаторами) +- [Управление площадками (проберами)](#управление-площадками-проберами) +- [Повторная проверка адреса](#повторная-проверка-адреса) +- [Частые проблемы и что с ними делать](#частые-проблемы-и-что-с-ними-делать) + +## Как устроена работа с системой + +Оператор не взаимодействует с валидаторами и проберами напрямую — вся +работа идёт через `control-api`. Цикл жизни одного IP-адреса: + +1. Адрес встаёт в очередь (`queued`). +2. Control-api сам находит свободный валидатор, привязывает адрес к нему + как Floating IP. +3. Валидатор проверяет, что действительно вышел в интернет именно через + этот адрес (self-check), затем прогоняет исходящие проверки (HTTPS, + ICMP, опционально SSH до заданных внешних целей). +4. Одновременно три внешние площадки проверяют, что этот адрес доступен + *снаружи* (входящие TCP-подключения на 22/80/443/8080 и ICMP) — это + ловит блокировки/чёрные списки на конкретных внешних сетях. +5. Как только все источники (валидатор + 3 площадки) отчитались — или + истекло время ожидания — control-api подводит итог и освобождает + адрес (отвязывает Floating IP). + +Всё это происходит автоматически, без участия оператора. Задача оператора +— положить адреса в очередь и снять с них результат. + +## Добавление новых IP в очередь + +**В текущей версии добавление адресов происходит только через конфиг +control-api**, отдельного API-метода "добавить IP в очередь" нет. + +1. Добавьте новые адреса в список `ip_addresses` в + `/etc/cloud-ip-validator/control-api.yaml` (в конец списка, либо в + нужном порядке — очередь обрабатывается строго в порядке следования + списка, `sequence`). +2. Перезапустите control-api: + ```bash + systemctl restart control-api + ``` + +Это безопасно для уже идущей работы: при старте control-api добавляет в +очередь только **новые** адреса (те, которых там ещё нет) — уже +обработанные ранее адреса не сбрасываются и повторно не проверяются. +Адреса, которые были удалены из `ip_addresses`, но уже есть в базе, +**не удаляются** из очереди/истории автоматически — если конкретный адрес +больше не нужно проверять и его нет в очереди/в процессе, можно просто +оставить как есть (историю он не портит). + +> Совет: держите `control-api.yaml` под версионным контролем (git) — +> список адресов на проверку тогда одновременно служит и журналом того, +> что вообще когда-либо ставилось в очередь. + +## Наблюдение за очередью + +Общая сводка: + +```bash +curl -s http://:8080/api/v1/admin/status | python3 -m json.tool +``` + +```json +{ + "total_ips": 25, + "ips_by_state": {"queued": 10, "awaiting_self_check": 1, "checking": 3, "done": 10, "failed": 1}, + "total_validators": 4 +} +``` + +`ips_by_state` — сколько адресов в каждом состоянии прямо сейчас. Если +хотите наблюдать за прогрессом в реальном времени: + +```bash +watch -n 2 'curl -s http://:8080/api/v1/admin/status | python3 -m json.tool' +``` + +Полный список всех адресов со всеми полями: + +```bash +curl -s http://:8080/api/v1/admin/ips | python3 -m json.tool +``` + +Только финальные результаты (уже готовые адреса), с помощью `jq`: + +```bash +curl -s http://:8080/api/v1/admin/ips \ + | jq '[.[] | select(.State=="done" or .State=="failed") | {IPAddress, State, OverallResult}]' +``` + +## Значения полей IP + +| Поле | Значение | +|---|---| +| `IPAddress` | Проверяемый адрес | +| `Sequence` | Позиция в очереди (порядок из конфига) | +| `State` | Текущий этап: `queued`, `assigning_fip`, `awaiting_self_check`, `checking`, `aggregating`, `done`, `failed` | +| `OwnerValidatorID` | Какой валидатор сейчас (или последним) занимался этим адресом | +| `FIPID` | Идентификатор Floating IP в OpenStack, к которому привязан адрес (пусто, если ещё/уже не привязан) | +| `AttemptNumber` | Номер попытки — растёт при каждом requeue (сбой привязки, сбой self-check, реклейм по таймауту) | +| `RetryCount` | Сколько раз адрес уже переставлялся в очередь заново | +| `EgressComplete` | Валидатор закончил исходящие проверки | +| `Site1Complete` / `Site2Complete` / `Site3Complete` | Соответствующая площадка закончила входящие проверки | +| `OverallResult` | Итог: `pass`, `partial`, `fail`, либо пусто, пока проверка не завершена | +| `AssignedAt` / `AggregatedAt` / `FIPReleasedAt` | Метки времени соответствующих этапов | + +## Как читать итоговый результат (pass/partial/fail) + +- **`pass`** — прошли все проверки (все исходящие + все три площадки по + всем портам и ICMP). Адрес можно считать пригодным к повторной выдаче. +- **`partial`** — часть проверок прошла, часть — нет (например, площадка + site-2 не смогла достучаться по 8080/tcp, но остальное в порядке). + Означает частичную деградацию — например, адрес заблокирован в + отдельном сегменте сети/у отдельного провайдера. Требует решения + оператора: годится ли адрес для данного случая использования. +- **`fail`** — либо ни одна проверка не прошла, либо адрес вообще не + дошёл до стадии проверок (например, self-check не подтвердился — + трафик валидатора не пошёл через назначенный FIP — и попытки + исчерпались). Смотрите `events` по этому адресу (см. ниже), чтобы + понять, на каком шаге и почему. + +Отсутствие ответа от источника (площадка не прислала результат до +истечения `checking_window_seconds`) засчитывается как провал — это +управляется настройкой `aggregation.missing_counts_as_fail` в конфиге +control-api (по умолчанию включено). + +## Просмотр деталей и истории по конкретному адресу + +```bash +curl -s http://:8080/api/v1/admin/ips/203.0.113.10 | python3 -m json.tool +``` + +Ответ содержит три части: +- `ip` — те же поля, что и в списке `admin/ips`, но для одного адреса; +- `checks` — все отдельные проверки текущей попытки: кто проверял + (`Source`: `egress` или `inbound-site-N`), что именно (`CheckType`, + `Target`), результат (`Success`), задержка (`LatencyMS`), и + человекочитаемая деталь (`Detail`, например текст ошибки при отказе); +- `events` — журнал аудита по этому адресу в хронологическом порядке + (регистрация, привязка FIP, self-check, агрегация и т.д.) — полезен, + чтобы восстановить точную последовательность событий при разборе + инцидента. + +Пример: почему адрес получил `fail`? + +```bash +curl -s http://:8080/api/v1/admin/ips/203.0.113.10 \ + | jq '.checks[] | select(.Success==false)' +``` + +## Управление валидаторами + +Список валидаторов и их текущее состояние: + +```bash +curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool +``` + +Состояния валидатора: `unregistered` (в конфиге есть, агент ещё не +подключался), `idle` (свободен, готов взять адрес), `assigned`/`checking` +(занят), `unreachable` (пропустил heartbeat дольше +`orchestrator.heartbeat_timeout_seconds`). + +**Добавление нового валидатора:** +1. Поднимите новую ВМ в сервисном проекте облака, узнайте её Neutron + `port_id`. +2. Добавьте запись в `validators` в `control-api.yaml` + (`validator_id` + `os_port_id`) и перезапустите `control-api`. +3. Разверните и запустите `validator-agent` на новой ВМ с тем же + `validator_id` в его конфиге (см. [SETUP.md](SETUP.md#развёртывание-validator-agent-на-вм-валидаторах)). + +**Вывод валидатора из эксплуатации:** остановите на нём +`validator-agent` (`systemctl stop validator-agent`). Он перестанет +получать новые задания после того, как закончит текущее (если оно было); +если он был убит посреди работы — control-api сам заберёт у него +незавершённый адрес обратно в очередь по истечении +`orchestrator.lease_ttl_seconds`. Удалять запись из `control-api.yaml` +не обязательно — просто выключенный агент не будет ничего забирать. + +## Управление площадками (проберами) + +Аналогично валидаторам: чтобы добавить площадку, добавьте `site_id` + +`index` (свободный от 1 до 3, или больше — но текущая схема БД +рассчитана ровно на 3 площадки, см. ниже) в `sites` конфига control-api, +разверните на площадке `prober` с тем же `site_id`. + +> Важно: количество площадок в текущей версии жёстко зашито в схему БД +> (`Site1Complete`/`Site2Complete`/`Site3Complete`) — система рассчитана +> ровно на **три** внешние площадки, как и описано в исходной схеме +> процесса. Изменение их числа потребует доработки схемы данных, это не +> делается только правкой конфига. + +## Повторная проверка адреса + +Если адрес завершился с `failed` или `partial`, а вы хотите перепроверить +его ещё раз (например, после устранения блокировки на стороне сети): +на данный момент нет отдельного API-метода "перезапустить проверку". +Самый простой путь: +1. Убедитесь, что адрес не находится в активном состоянии (`checking` + и т.п.) — то есть уже `done`/`failed`. +2. Временно уберите и снова добавьте адрес в список `ip_addresses` + (либо просто пересоздайте запись в БД вручную, если это единичный + случай и у вас есть доступ к SQLite) и перезапустите `control-api`. + +Поскольку сидирование очереди идёт по уникальности `ip_address` +(конфликт по уже существующей записи просто игнорируется), самый чистый +способ гарантированно перепроверить конкретный адрес — обратиться к +администратору БД (см. следующий раздел) либо дождаться штатной +доработки API под повторные проверки. + +## Частые проблемы и что с ними делать + +**Валидатор долго висит в `unreachable`.** +Проверьте сетевую связность ВМ-валидатора до `control-api` (порт из +`server.listen_addr`) и что процесс `validator-agent` вообще запущен +(`systemctl status validator-agent`, `journalctl -u validator-agent`). + +**Адрес постоянно проваливает self-check.** +Смотрите `events` по адресу (`GET /api/v1/admin/ips/{ip}`) — в детали +события `self_check_result` будет указан обнаруженный исходящий адрес. +Если он не совпадает с ожидаемым — вероятно, на ВМ-валидаторе есть другой +маршрут наружу (не через назначенный Floating IP), либо привязка FIP на +стороне OpenStack не применилась. Проверьте вручную в OpenStack +(`openstack floating ip show <адрес>`), что `port_id` совпадает с портом +валидатора. + +**Площадка (`site-N`) никогда не отчитывается (`SiteNComplete` всегда +`false`).** +Проверьте, что `prober` на этой площадке запущен и его `site_id` в +конфиге совпадает с `site_id` в конфиге control-api. Проверьте, что +площадка имеет сетевой доступ и до `control-api`, и до проверяемого +адреса (входящий трафик на 22/80/443/8080 + ICMP — это отдельная +связность от связи с control-api, см. +[SETUP.md](SETUP.md#сетевые-доступы)). + +**Много адресов зависло в `checking` дольше ожидаемого.** +Это нормально, если ещё не истёк `orchestrator.checking_window_seconds` — +агрегация ждёт либо полного набора ответов, либо истечения окна. Если +адрес завис заметно дольше окна — проверьте, что фоновый цикл control-api +вообще работает (смотрите `journalctl -u control-api` на предмет ошибок +в `sweep checking window`). + +**Нужно посмотреть на данные "из первых рук", в обход API.** +`control-api` использует SQLite, файл — по пути `database.path` из +конфига. Можно (только для чтения, на **той же машине**, где крутится +control-api) открыть его `sqlite3` в режиме WAL — это безопасно для +чтения параллельно с работающим процессом: +```bash +sqlite3 /var/lib/cloud-ip-validator/control-api.db \ + "select ip_address, state, overall_result from ip_queue order by sequence" +``` +Не редактируйте эту базу вручную во время работы control-api — это может +рассинхронизировать состояние с реальными привязками Floating IP в +OpenStack. diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..3cb163c --- /dev/null +++ b/go.mod @@ -0,0 +1,22 @@ +module cloudipvalidator + +go 1.26.2 + +require ( + github.com/gophercloud/gophercloud/v2 v2.14.0 + golang.org/x/net v0.58.0 + gopkg.in/yaml.v3 v3.0.1 + modernc.org/sqlite v1.57.0 +) + +require ( + github.com/dustin/go-humanize v1.0.1 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/mattn/go-isatty v0.0.24 // indirect + github.com/ncruces/go-strftime v1.0.0 // indirect + github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect + golang.org/x/sys v0.47.0 // indirect + modernc.org/libc v1.74.4 // indirect + modernc.org/mathutil v1.7.1 // indirect + modernc.org/memory v1.11.0 // indirect +) diff --git a/go.sum b/go.sum new file mode 100644 index 0000000..f6a224c --- /dev/null +++ b/go.sum @@ -0,0 +1,58 @@ +github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY= +github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/gophercloud/gophercloud/v2 v2.14.0 h1:xGxKCvyaOxJDc5FqrnKDNqtdYn43ocQPuJ2Cm4KX/cs= +github.com/gophercloud/gophercloud/v2 v2.14.0/go.mod h1:4fs5I9VH6Wg2LyocDL9xf0ASb8VD63tyLA8sgAX/69U= +github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k= +github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM= +github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= +github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w= +github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo= +golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ= +golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0= +golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= +golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= +golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM= +golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q= +golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= +gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI= +modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI= +modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU= +modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk= +modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM= +modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU= +modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI= +modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito= +modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI= +modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY= +modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks= +modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI= +modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k= +modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co= +modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU= +modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg= +modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI= +modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw= +modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg= +modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns= +modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w= +modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE= +modernc.org/sqlite v1.57.0 h1:qNQP6xnx5M0ISNtlnxoOX0+cD5bJ0/gr9aMmndFczzg= +modernc.org/sqlite v1.57.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ= +modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0= +modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A= +modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y= +modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM= diff --git a/internal/agentcore/agentcore.go b/internal/agentcore/agentcore.go new file mode 100644 index 0000000..40c113f --- /dev/null +++ b/internal/agentcore/agentcore.go @@ -0,0 +1,280 @@ +// Package agentcore implements the validator-agent's poll loop: register, +// heartbeat, wait for an assignment, self-check that egress actually flows +// through the newly attached FIP, run the configured outbound/egress +// checks, and report results — all driven entirely by the Control API, so +// the process itself holds no durable state (constraint: the agent must be +// safely restartable at any point without losing correctness, only +// possibly re-doing in-flight work, which the Control API's idempotent +// upserts tolerate). +package agentcore + +import ( + "context" + "fmt" + "log/slog" + "os" + "time" + + "cloudipvalidator/internal/apiclient" + "cloudipvalidator/internal/checkrunner" + "cloudipvalidator/internal/config" +) + +type Agent struct { + cfg *config.ValidatorAgent + client *apiclient.Client + log *slog.Logger + + lastHandledIPID int64 +} + +func New(cfg *config.ValidatorAgent, log *slog.Logger) *Agent { + timeout := time.Duration(cfg.Checks.HTTPSTimeoutSeconds) * time.Second + if timeout <= 0 { + timeout = 10 * time.Second + } + return &Agent{ + cfg: cfg, + client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second), + log: log, + } +} + +// Run registers with the Control API and polls forever until ctx is +// cancelled. +func (a *Agent) Run(ctx context.Context) error { + if err := a.register(ctx); err != nil { + return fmt.Errorf("register: %w", err) + } + + interval := time.Duration(a.cfg.PollIntervalSeconds) * time.Second + ticker := time.NewTicker(interval) + defer ticker.Stop() + + for { + a.pollOnce(ctx) + select { + case <-ctx.Done(): + return ctx.Err() + case <-ticker.C: + } + } +} + +type registerReq struct { + ValidatorID string `json:"validator_id"` + Hostname string `json:"hostname"` + AgentVersion string `json:"agent_version"` +} +type registerResp struct { + OK bool `json:"ok"` + PollIntervalSeconds int `json:"poll_interval_seconds"` +} + +func (a *Agent) register(ctx context.Context) error { + hostname, _ := hostnameOrDefault() + var resp registerResp + _, err := a.client.Do(ctx, "POST", "/api/v1/agents/register", registerReq{ + ValidatorID: a.cfg.ValidatorID, Hostname: hostname, AgentVersion: "dev", + }, &resp) + if err != nil { + return err + } + a.log.Info("registered", "validator_id", a.cfg.ValidatorID) + return nil +} + +type heartbeatReq struct { + LocalState string `json:"local_state"` +} + +type assignmentResp struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + Phase string `json:"phase"` + CheckConfig []checkConfigDTO `json:"check_config"` +} + +type checkConfigDTO struct { + Type string `json:"type"` + Targets []string `json:"targets"` +} + +func (a *Agent) pollOnce(ctx context.Context) { + if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/heartbeat", heartbeatReq{LocalState: "idle"}, nil); err != nil { + a.log.Error("heartbeat", "err", err) + return + } + + var assignment assignmentResp + ok, err := a.client.Do(ctx, "GET", "/api/v1/agents/"+a.cfg.ValidatorID+"/assignment", nil, &assignment) + if err != nil { + a.log.Error("get assignment", "err", err) + return + } + if !ok { + a.lastHandledIPID = 0 + return // nothing assigned right now + } + + if assignment.IPID == a.lastHandledIPID { + return // already handled this IP's work this attempt + } + + switch assignment.Phase { + case "awaiting_self_check": + a.handleSelfCheckAndRun(ctx, assignment) + case "checking": + // Agent restarted (or a prior response was lost) after self-check + // already succeeded server-side: just (re-)run checks, which is + // safe since results are upserted idempotently. + a.runChecks(ctx, assignment) + } +} + +func (a *Agent) handleSelfCheckAndRun(ctx context.Context, assignment assignmentResp) { + a.postEvent(ctx, assignment.IPID, "config_received", "") + + timeout := time.Duration(a.cfg.SelfCheck.TimeoutSeconds) * time.Second + selfCtx, cancel := context.WithTimeout(ctx, timeout) + defer cancel() + + var whoami struct { + IP string `json:"ip"` + } + _, err := a.client.Do(selfCtx, "GET", "/api/v1/whatsmyip", nil, &whoami) + success := err == nil && whoami.IP == assignment.IPAddress + detail := "matched" + if err != nil { + detail = "whatsmyip request failed: " + err.Error() + } else if !success { + detail = fmt.Sprintf("egress ip %q does not match assigned fip %q", whoami.IP, assignment.IPAddress) + } + + a.postSelfCheck(ctx, assignment.IPID, whoami.IP, success, detail) + a.postEvent(ctx, assignment.IPID, "self_check_result", fmt.Sprintf(`{"success":%t}`, success)) + + if !success { + a.log.Warn("self-check failed", "ip", assignment.IPAddress, "detail", detail) + return + } + a.runChecks(ctx, assignment) +} + +func (a *Agent) runChecks(ctx context.Context, assignment assignmentResp) { + var results []checkResultDTO + for _, ct := range assignment.CheckConfig { + for _, target := range ct.Targets { + var fn func(context.Context) checkrunner.Result + switch ct.Type { + case "https": + fn = checkrunner.HTTPS(target, time.Duration(a.cfg.Checks.HTTPSTimeoutSeconds)*time.Second) + case "icmp": + fn = checkrunner.ICMPEcho(hostOnly(target), a.cfg.Checks.ICMPCount, time.Duration(a.cfg.Checks.ICMPTimeoutSeconds)*time.Second) + case "ssh": + if !a.cfg.Checks.SSH.Enabled { + continue + } + fn = checkrunner.SSHBanner(hostOnly(target), time.Duration(a.cfg.Checks.SSH.TimeoutSeconds)*time.Second) + default: + a.log.Warn("unknown check type", "type", ct.Type) + continue + } + res := fn(ctx) + // Record the originally configured target (a full URL for + // https, e.g.), not checkrunner's internal host-only value + // used for icmp/ssh — otherwise two configured targets that + // happen to share a bare host (as can occur, e.g., in the + // loopback-only local e2e harness) would collide on the + // checks table's UNIQUE(ip_id, attempt, source, type, + // target) key and silently overwrite each other. + results = append(results, checkResultDTO{ + IPID: assignment.IPID, CheckType: res.CheckType, Target: target, + Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, + CheckedAt: res.CheckedAt.Format(time.RFC3339Nano), + }) + // Report progressively rather than batching until the end, so + // a crash mid-run doesn't lose already-completed check results. + a.postResults(ctx, []checkResultDTO{results[len(results)-1]}) + } + } + + a.postComplete(ctx, assignment.IPID) + a.lastHandledIPID = assignment.IPID +} + +type checkResultDTO struct { + IPID int64 `json:"ip_id"` + CheckType string `json:"check_type"` + Target string `json:"target,omitempty"` + Success bool `json:"success"` + LatencyMS int64 `json:"latency_ms"` + Detail string `json:"detail,omitempty"` + CheckedAt string `json:"checked_at"` +} + +func (a *Agent) postResults(ctx context.Context, results []checkResultDTO) { + body := struct { + Results []checkResultDTO `json:"results"` + }{results} + if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/results", body, nil); err != nil { + a.log.Error("post results", "err", err) + } +} + +func (a *Agent) postComplete(ctx context.Context, ipID int64) { + body := struct { + IPID int64 `json:"ip_id"` + }{ipID} + if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/complete", body, nil); err != nil { + a.log.Error("post complete", "err", err) + } +} + +func (a *Agent) postSelfCheck(ctx context.Context, ipID int64, detected string, success bool, detail string) { + body := struct { + IPID int64 `json:"ip_id"` + DetectedEgress string `json:"detected_egress_ip"` + Success bool `json:"success"` + Detail string `json:"detail"` + }{ipID, detected, success, detail} + if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/self-check", body, nil); err != nil { + a.log.Error("post self-check", "err", err) + } +} + +func (a *Agent) postEvent(ctx context.Context, ipID int64, eventType, payload string) { + body := struct { + EventType string `json:"event_type"` + IPID int64 `json:"ip_id"` + Payload string `json:"payload"` + }{eventType, ipID, payload} + if _, err := a.client.Do(ctx, "POST", "/api/v1/agents/"+a.cfg.ValidatorID+"/events", body, nil); err != nil { + a.log.Error("post event", "type", eventType, "err", err) + } +} + +func hostnameOrDefault() (string, error) { + h, err := os.Hostname() + if err != nil || h == "" { + return "unknown", err + } + return h, nil +} + +// hostOnly strips a URL scheme (https://) from a target, since ICMP/SSH +// checks operate on bare hostnames while HTTPS checks take a full URL. +func hostOnly(target string) string { + for _, prefix := range []string{"https://", "http://"} { + if len(target) > len(prefix) && target[:len(prefix)] == prefix { + target = target[len(prefix):] + break + } + } + for i := 0; i < len(target); i++ { + if target[i] == '/' || target[i] == ':' { + return target[:i] + } + } + return target +} diff --git a/internal/apiclient/apiclient.go b/internal/apiclient/apiclient.go new file mode 100644 index 0000000..b088e8d --- /dev/null +++ b/internal/apiclient/apiclient.go @@ -0,0 +1,65 @@ +// Package apiclient is a thin HTTP client for the Control API's /api/v1 +// surface, shared by the validator-agent and prober binaries. Neither +// binary talks to the database directly — this is their only channel to +// shared state, keeping both genuinely stateless. +package apiclient + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "time" +) + +type Client struct { + BaseURL string + HTTPClient *http.Client +} + +func New(baseURL string, timeout time.Duration) *Client { + return &Client{BaseURL: baseURL, HTTPClient: &http.Client{Timeout: timeout}} +} + +// Do issues a JSON request and decodes a JSON response into out (if +// non-nil). A 204 response is treated as "no content" and out is left +// untouched, with ok=false — used for the assignment poll's empty case. +func (c *Client) Do(ctx context.Context, method, path string, body, out interface{}) (ok bool, err error) { + var reader io.Reader + if body != nil { + b, err := json.Marshal(body) + if err != nil { + return false, fmt.Errorf("marshal request: %w", err) + } + reader = bytes.NewReader(b) + } + req, err := http.NewRequestWithContext(ctx, method, c.BaseURL+path, reader) + if err != nil { + return false, fmt.Errorf("build request: %w", err) + } + if body != nil { + req.Header.Set("Content-Type", "application/json") + } + + resp, err := c.HTTPClient.Do(req) + if err != nil { + return false, fmt.Errorf("%s %s: %w", method, path, err) + } + defer resp.Body.Close() + + if resp.StatusCode == http.StatusNoContent { + return false, nil + } + respBody, _ := io.ReadAll(resp.Body) + if resp.StatusCode >= 300 { + return false, fmt.Errorf("%s %s: status %d: %s", method, path, resp.StatusCode, string(respBody)) + } + if out != nil && len(respBody) > 0 { + if err := json.Unmarshal(respBody, out); err != nil { + return false, fmt.Errorf("%s %s: decode response: %w", method, path, err) + } + } + return true, nil +} diff --git a/internal/checkrunner/https.go b/internal/checkrunner/https.go new file mode 100644 index 0000000..92c082c --- /dev/null +++ b/internal/checkrunner/https.go @@ -0,0 +1,36 @@ +package checkrunner + +import ( + "context" + "fmt" + "net/http" + "time" +) + +// HTTPS performs a GET against target (expected to be a full URL, e.g. +// https://github.com) and reports success for any response with status +// < 400. It never follows the check to a different host on a check-type +// boundary — redirects within the same request are followed by the +// standard http.Client default policy, which is what we want for +// reachability checks. +func HTTPS(target string, timeout time.Duration) func(ctx context.Context) Result { + return run("https", target, func(ctx context.Context) error { + ctx, cancel := context.WithTimeout(ctx, timeout) + defer cancel() + + req, err := http.NewRequestWithContext(ctx, http.MethodGet, target, nil) + if err != nil { + return fmt.Errorf("build request: %w", err) + } + client := &http.Client{Timeout: timeout} + resp, err := client.Do(req) + if err != nil { + return err + } + defer resp.Body.Close() + if resp.StatusCode >= 400 { + return fmt.Errorf("unexpected status %d", resp.StatusCode) + } + return nil + }) +} diff --git a/internal/checkrunner/icmp.go b/internal/checkrunner/icmp.go new file mode 100644 index 0000000..7807e66 --- /dev/null +++ b/internal/checkrunner/icmp.go @@ -0,0 +1,82 @@ +package checkrunner + +import ( + "context" + "fmt" + "net" + "os" + "time" + + "golang.org/x/net/icmp" + "golang.org/x/net/ipv4" +) + +// ICMPEcho sends up to `count` ICMP echo requests to host and reports +// success if at least one echo reply is received before timeout. Uses a +// raw ICMP socket (ip4:icmp), which requires either running as root or +// (on Linux, as deployed here) the CAP_NET_RAW capability — see the +// validator-agent and prober systemd units. +func ICMPEcho(host string, count int, timeout time.Duration) func(ctx context.Context) Result { + return run("icmp", host, func(ctx context.Context) error { + conn, err := icmp.ListenPacket("ip4:icmp", "0.0.0.0") + if err != nil { + return fmt.Errorf("open icmp socket (needs CAP_NET_RAW or root): %w", err) + } + defer conn.Close() + + dst, err := net.ResolveIPAddr("ip4", host) + if err != nil { + return fmt.Errorf("resolve %s: %w", host, err) + } + + id := os.Getpid() & 0xffff + var lastErr error + for seq := 1; seq <= count; seq++ { + if err := ctx.Err(); err != nil { + return err + } + msg := icmp.Message{ + Type: ipv4.ICMPTypeEcho, + Code: 0, + Body: &icmp.Echo{ + ID: id, + Seq: seq, + Data: []byte("cloud-ip-validator"), + }, + } + wb, err := msg.Marshal(nil) + if err != nil { + return fmt.Errorf("marshal echo request: %w", err) + } + if _, err := conn.WriteTo(wb, dst); err != nil { + lastErr = fmt.Errorf("write echo request: %w", err) + continue + } + + perAttempt := timeout / time.Duration(count) + if perAttempt <= 0 { + perAttempt = timeout + } + conn.SetReadDeadline(time.Now().Add(perAttempt)) + rb := make([]byte, 1500) + n, _, err := conn.ReadFrom(rb) + if err != nil { + lastErr = fmt.Errorf("read echo reply: %w", err) + continue + } + rm, err := icmp.ParseMessage(1 /* protocolICMP */, rb[:n]) + if err != nil { + lastErr = fmt.Errorf("parse reply: %w", err) + continue + } + if rm.Type == ipv4.ICMPTypeEchoReply { + return nil + } + lastErr = fmt.Errorf("unexpected icmp type %v", rm.Type) + } + if lastErr == nil { + lastErr = fmt.Errorf("no reply received") + } + return lastErr + }) +} diff --git a/internal/checkrunner/tcp.go b/internal/checkrunner/tcp.go new file mode 100644 index 0000000..31d243d --- /dev/null +++ b/internal/checkrunner/tcp.go @@ -0,0 +1,51 @@ +package checkrunner + +import ( + "context" + "fmt" + "net" + "strconv" + "time" +) + +// TCPConnect reports success if a TCP handshake against host:port +// completes within timeout. This is the primitive behind the prober's +// per-port inbound reachability checks (22/80/443/8080). +func TCPConnect(host string, port int, timeout time.Duration) func(ctx context.Context) Result { + target := net.JoinHostPort(host, strconv.Itoa(port)) + checkType := "tcp-" + strconv.Itoa(port) + return run(checkType, target, func(ctx context.Context) error { + d := net.Dialer{Timeout: timeout} + conn, err := d.DialContext(ctx, "tcp", target) + if err != nil { + return err + } + return conn.Close() + }) +} + +// SSHBanner performs a TCP connect to host:22 and additionally verifies the +// remote sends an "SSH-2.0-" banner, without performing any auth handshake. +// Used for the optional ssh check type. +func SSHBanner(host string, timeout time.Duration) func(ctx context.Context) Result { + target := net.JoinHostPort(host, "22") + return run("ssh", target, func(ctx context.Context) error { + d := net.Dialer{Timeout: timeout} + conn, err := d.DialContext(ctx, "tcp", target) + if err != nil { + return err + } + defer conn.Close() + + conn.SetReadDeadline(time.Now().Add(timeout)) + buf := make([]byte, 8) + n, err := conn.Read(buf) + if err != nil { + return fmt.Errorf("read banner: %w", err) + } + if n < 8 || string(buf[:8]) != "SSH-2.0-" { + return fmt.Errorf("unexpected banner prefix %q", string(buf[:n])) + } + return nil + }) +} diff --git a/internal/checkrunner/types.go b/internal/checkrunner/types.go new file mode 100644 index 0000000..343687d --- /dev/null +++ b/internal/checkrunner/types.go @@ -0,0 +1,44 @@ +// Package checkrunner implements the actual network probes (HTTPS, TCP +// connect, ICMP echo, SSH banner) shared by both the validator-agent +// (outbound/egress checks) and the prober (inbound/reachability checks). +// The two binaries use the same primitives against different targets and +// in different directions, but never share process state. +package checkrunner + +import ( + "context" + "time" +) + +// Result is the outcome of a single check, in a form that maps directly +// onto a db.Check row (minus the fields the caller already knows: ip_id, +// attempt_number, validator_id, source). +type Result struct { + CheckType string + Target string + Success bool + LatencyMS int64 + Detail string + CheckedAt time.Time +} + +func run(checkType, target string, fn func(ctx context.Context) error) func(ctx context.Context) Result { + return func(ctx context.Context) Result { + start := time.Now() + err := fn(ctx) + latency := time.Since(start).Milliseconds() + res := Result{ + CheckType: checkType, + Target: target, + Success: err == nil, + LatencyMS: latency, + CheckedAt: time.Now().UTC(), + } + if err != nil { + res.Detail = err.Error() + } else { + res.Detail = "ok" + } + return res + } +} diff --git a/internal/config/config.go b/internal/config/config.go new file mode 100644 index 0000000..924f7b2 --- /dev/null +++ b/internal/config/config.go @@ -0,0 +1,236 @@ +// Package config defines the YAML configuration structures for all three +// binaries and loads them from disk. OpenStack credentials are deliberately +// never part of these structs — only the *names* of environment variables +// to read them from — so secrets never land in a config file on disk. +package config + +import ( + "fmt" + "os" + + "gopkg.in/yaml.v3" +) + +// ---- control-api ---- + +type ControlAPI struct { + Server ServerConfig `yaml:"server"` + Database DatabaseConfig `yaml:"database"` + OpenStack OpenStackConfig `yaml:"openstack"` + Orchestrator OrchestratorConfig `yaml:"orchestrator"` + Aggregation AggregationConfig `yaml:"aggregation"` + Validators []ValidatorConfig `yaml:"validators"` + Sites []SiteConfig `yaml:"sites"` + CheckTypes []CheckTypeConfig `yaml:"check_types"` + Targets map[string][]string `yaml:"targets"` + Inbound InboundConfig `yaml:"inbound_checks"` + IPAddresses []string `yaml:"ip_addresses"` +} + +type ServerConfig struct { + ListenAddr string `yaml:"listen_addr"` +} + +type DatabaseConfig struct { + Path string `yaml:"path"` +} + +// OpenStackConfig names the environment variables control-api reads its +// OpenStack admin credential from at startup. Mode "mock" skips all of this +// and uses an in-memory FloatingIPClient instead — used for local dev and +// the offline end-to-end harness. +type OpenStackConfig struct { + Mode string `yaml:"mode"` // "mock" | "real" + AuthURLEnv string `yaml:"auth_url_env"` + TokenEnv string `yaml:"token_env"` + ProjectIDEnv string `yaml:"project_id_env"` + ProjectNameEnv string `yaml:"project_name_env"` + ProjectDomainEnv string `yaml:"project_domain_env"` + RegionEnv string `yaml:"region_env"` +} + +type OrchestratorConfig struct { + PollIntervalSeconds int `yaml:"poll_interval_seconds"` + SelfCheckTimeoutSeconds int `yaml:"self_check_timeout_seconds"` + MaxSelfCheckRetries int `yaml:"max_self_check_retries"` + CheckingWindowSeconds int `yaml:"checking_window_seconds"` + MaxRetries int `yaml:"max_retries"` + LeaseTTLSeconds int `yaml:"lease_ttl_seconds"` + HeartbeatTimeoutSeconds int `yaml:"heartbeat_timeout_seconds"` +} + +type AggregationConfig struct { + MissingCountsAsFail bool `yaml:"missing_counts_as_fail"` +} + +type ValidatorConfig struct { + ValidatorID string `yaml:"validator_id"` + OSPortID string `yaml:"os_port_id"` +} + +type SiteConfig struct { + SiteID string `yaml:"site_id"` + Index int `yaml:"index"` // 1, 2, or 3 — maps to ip_queue.siteN_complete +} + +// CheckTypeConfig maps a check type (https, icmp, ssh) to the named target +// groups (keys into ControlAPI.Targets) it should run against. This drives +// the egress/outbound checks the validator-agent performs. +type CheckTypeConfig struct { + Name string `yaml:"name"` + Enabled bool `yaml:"enabled"` + Targets []string `yaml:"targets"` +} + +type InboundConfig struct { + Ports []int `yaml:"ports"` + ICMP bool `yaml:"icmp"` +} + +func LoadControlAPI(path string) (*ControlAPI, error) { + var c ControlAPI + if err := loadYAML(path, &c); err != nil { + return nil, err + } + if c.Server.ListenAddr == "" { + c.Server.ListenAddr = ":8080" + } + if c.Database.Path == "" { + c.Database.Path = "control-api.db" + } + if c.OpenStack.Mode == "" { + c.OpenStack.Mode = "mock" + } + if c.Orchestrator.PollIntervalSeconds == 0 { + c.Orchestrator.PollIntervalSeconds = 5 + } + if c.Orchestrator.SelfCheckTimeoutSeconds == 0 { + c.Orchestrator.SelfCheckTimeoutSeconds = 60 + } + if c.Orchestrator.MaxSelfCheckRetries == 0 { + c.Orchestrator.MaxSelfCheckRetries = 3 + } + if c.Orchestrator.CheckingWindowSeconds == 0 { + c.Orchestrator.CheckingWindowSeconds = 120 + } + if c.Orchestrator.MaxRetries == 0 { + c.Orchestrator.MaxRetries = 3 + } + if c.Orchestrator.LeaseTTLSeconds == 0 { + c.Orchestrator.LeaseTTLSeconds = 180 + } + if c.Orchestrator.HeartbeatTimeoutSeconds == 0 { + c.Orchestrator.HeartbeatTimeoutSeconds = 30 + } + return &c, nil +} + +// ---- validator-agent ---- + +type ValidatorAgent struct { + ValidatorID string `yaml:"validator_id"` + ControlAPIURL string `yaml:"control_api_url"` + PollIntervalSeconds int `yaml:"poll_interval_seconds"` + SelfCheck SelfCheckCfg `yaml:"self_check"` + Checks AgentChecks `yaml:"checks"` +} + +type SelfCheckCfg struct { + TimeoutSeconds int `yaml:"timeout_seconds"` +} + +type AgentChecks struct { + HTTPSTimeoutSeconds int `yaml:"https_timeout_seconds"` + ICMPTimeoutSeconds int `yaml:"icmp_timeout_seconds"` + ICMPCount int `yaml:"icmp_count"` + SSH SSHCfg `yaml:"ssh"` +} + +type SSHCfg struct { + Enabled bool `yaml:"enabled"` + TimeoutSeconds int `yaml:"timeout_seconds"` +} + +func LoadValidatorAgent(path string) (*ValidatorAgent, error) { + var c ValidatorAgent + if err := loadYAML(path, &c); err != nil { + return nil, err + } + if c.PollIntervalSeconds == 0 { + c.PollIntervalSeconds = 5 + } + if c.SelfCheck.TimeoutSeconds == 0 { + c.SelfCheck.TimeoutSeconds = 10 + } + if c.Checks.HTTPSTimeoutSeconds == 0 { + c.Checks.HTTPSTimeoutSeconds = 10 + } + if c.Checks.ICMPTimeoutSeconds == 0 { + c.Checks.ICMPTimeoutSeconds = 5 + } + if c.Checks.ICMPCount == 0 { + c.Checks.ICMPCount = 3 + } + if c.Checks.SSH.TimeoutSeconds == 0 { + c.Checks.SSH.TimeoutSeconds = 5 + } + if c.ValidatorID == "" { + return nil, fmt.Errorf("validator_id is required") + } + if c.ControlAPIURL == "" { + return nil, fmt.Errorf("control_api_url is required") + } + return &c, nil +} + +// ---- prober ---- + +type Prober struct { + SiteID string `yaml:"site_id"` + ControlAPIURL string `yaml:"control_api_url"` + PollIntervalSeconds int `yaml:"poll_interval_seconds"` + Checks ProberChecks `yaml:"checks"` +} + +type ProberChecks struct { + TCPTimeoutSeconds int `yaml:"tcp_timeout_seconds"` + ICMPTimeoutSeconds int `yaml:"icmp_timeout_seconds"` + ICMPCount int `yaml:"icmp_count"` +} + +func LoadProber(path string) (*Prober, error) { + var c Prober + if err := loadYAML(path, &c); err != nil { + return nil, err + } + if c.PollIntervalSeconds == 0 { + c.PollIntervalSeconds = 5 + } + if c.Checks.TCPTimeoutSeconds == 0 { + c.Checks.TCPTimeoutSeconds = 5 + } + if c.Checks.ICMPTimeoutSeconds == 0 { + c.Checks.ICMPTimeoutSeconds = 5 + } + if c.Checks.ICMPCount == 0 { + c.Checks.ICMPCount = 3 + } + if c.SiteID == "" { + return nil, fmt.Errorf("site_id is required") + } + if c.ControlAPIURL == "" { + return nil, fmt.Errorf("control_api_url is required") + } + return &c, nil +} + +func loadYAML(path string, out interface{}) error { + data, err := os.ReadFile(path) + if err != nil { + return fmt.Errorf("read config %s: %w", path, err) + } + if err := yaml.Unmarshal(data, out); err != nil { + return fmt.Errorf("parse config %s: %w", path, err) + } + return nil +} diff --git a/internal/db/db.go b/internal/db/db.go new file mode 100644 index 0000000..de3eceb --- /dev/null +++ b/internal/db/db.go @@ -0,0 +1,85 @@ +// Package db owns the SQLite connection, schema migrations, and all queries +// used by the Control API. It is the only package in the system that talks +// to the database directly — agents and probers never connect to it. +package db + +import ( + "context" + "database/sql" + _ "embed" + "fmt" + "time" + + _ "modernc.org/sqlite" +) + +//go:embed migrations/0001_init.sql +var initSchema string + +type DB struct { + *sql.DB +} + +// Open opens (creating if necessary) the SQLite database at path, applies +// pragmas suited to a single-writer WAL workload, and runs any pending +// schema migrations. +func Open(ctx context.Context, path string) (*DB, error) { + sqlDB, err := sql.Open("sqlite", path+"?_pragma=busy_timeout(5000)") + if err != nil { + return nil, fmt.Errorf("open sqlite: %w", err) + } + // Control API is the sole writer; one connection avoids SQLITE_BUSY + // entirely for writes while still allowing concurrent reads via WAL. + sqlDB.SetMaxOpenConns(1) + + for _, pragma := range []string{ + "PRAGMA journal_mode=WAL", + "PRAGMA synchronous=NORMAL", + "PRAGMA foreign_keys=ON", + "PRAGMA busy_timeout=5000", + } { + if _, err := sqlDB.ExecContext(ctx, pragma); err != nil { + sqlDB.Close() + return nil, fmt.Errorf("apply pragma %q: %w", pragma, err) + } + } + + d := &DB{DB: sqlDB} + if err := d.migrate(ctx); err != nil { + sqlDB.Close() + return nil, fmt.Errorf("migrate: %w", err) + } + return d, nil +} + +// migrate applies the embedded schema exactly once, tracked via +// PRAGMA user_version so repeated startups are no-ops. +func (d *DB) migrate(ctx context.Context) error { + var version int + if err := d.QueryRowContext(ctx, "PRAGMA user_version").Scan(&version); err != nil { + return fmt.Errorf("read user_version: %w", err) + } + if version >= 1 { + return nil + } + + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() + + if _, err := tx.ExecContext(ctx, initSchema); err != nil { + return fmt.Errorf("apply 0001_init.sql: %w", err) + } + if _, err := tx.ExecContext(ctx, "PRAGMA user_version=1"); err != nil { + return fmt.Errorf("set user_version: %w", err) + } + return tx.Commit() +} + +// Now returns the current time truncated to millisecond precision, the +// granularity used consistently for all timestamp columns. +func Now() time.Time { + return time.Now().UTC().Truncate(time.Millisecond) +} diff --git a/internal/db/migrations/0001_init.sql b/internal/db/migrations/0001_init.sql new file mode 100644 index 0000000..30c62e0 --- /dev/null +++ b/internal/db/migrations/0001_init.sql @@ -0,0 +1,65 @@ +CREATE TABLE validators ( + validator_id TEXT PRIMARY KEY, + hostname TEXT NOT NULL DEFAULT '', + os_port_id TEXT NOT NULL DEFAULT '', + state TEXT NOT NULL DEFAULT 'unregistered', + current_ip_id INTEGER REFERENCES ip_queue(id), + agent_version TEXT NOT NULL DEFAULT '', + last_heartbeat_at TIMESTAMP, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +CREATE TABLE ip_queue ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + ip_address TEXT NOT NULL UNIQUE, + sequence INTEGER NOT NULL, + state TEXT NOT NULL DEFAULT 'queued', + owner_validator_id TEXT REFERENCES validators(validator_id), + fip_id TEXT NOT NULL DEFAULT '', + attempt_number INTEGER NOT NULL DEFAULT 1, + retry_count INTEGER NOT NULL DEFAULT 0, + lease_expires_at TIMESTAMP, + egress_complete BOOLEAN NOT NULL DEFAULT 0, + site1_complete BOOLEAN NOT NULL DEFAULT 0, + site2_complete BOOLEAN NOT NULL DEFAULT 0, + site3_complete BOOLEAN NOT NULL DEFAULT 0, + overall_result TEXT NOT NULL DEFAULT '', + assigned_at TIMESTAMP, + aggregated_at TIMESTAMP, + fip_released_at TIMESTAMP, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP +); +CREATE INDEX idx_ip_queue_state_seq ON ip_queue(state, sequence); + +CREATE TABLE checks ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + ip_id INTEGER NOT NULL REFERENCES ip_queue(id), + ip_address TEXT NOT NULL, + attempt_number INTEGER NOT NULL, + validator_id TEXT NOT NULL DEFAULT '', + source TEXT NOT NULL, + check_type TEXT NOT NULL, + target TEXT NOT NULL DEFAULT '', + success BOOLEAN NOT NULL, + latency_ms INTEGER NOT NULL DEFAULT 0, + detail TEXT NOT NULL DEFAULT '', + checked_at TIMESTAMP NOT NULL, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, + UNIQUE(ip_id, attempt_number, source, check_type, target) +); +CREATE INDEX idx_checks_ip_attempt ON checks(ip_id, attempt_number); + +CREATE TABLE events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + source_type TEXT NOT NULL, + source_id TEXT NOT NULL DEFAULT '', + ip_id INTEGER REFERENCES ip_queue(id), + event_type TEXT NOT NULL, + payload TEXT NOT NULL DEFAULT '', + occurred_at TIMESTAMP NOT NULL, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP +); +CREATE INDEX idx_events_ip ON events(ip_id); +CREATE INDEX idx_events_source ON events(source_type, source_id); diff --git a/internal/db/models.go b/internal/db/models.go new file mode 100644 index 0000000..75565ce --- /dev/null +++ b/internal/db/models.go @@ -0,0 +1,117 @@ +package db + +import "time" + +// Validator and IP lifecycle states. Kept as typed string constants rather +// than a Go enum type so they round-trip through SQLite TEXT columns and +// JSON without conversion. +const ( + ValidatorUnregistered = "unregistered" + ValidatorIdle = "idle" + ValidatorAssigned = "assigned" + ValidatorChecking = "checking" + ValidatorUnreachable = "unreachable" + + IPQueued = "queued" + IPAssigningFIP = "assigning_fip" + IPAwaitingSelfCheck = "awaiting_self_check" + IPChecking = "checking" + IPAggregating = "aggregating" + IPDone = "done" + IPFailed = "failed" + + ResultPass = "pass" + ResultPartial = "partial" + ResultFail = "fail" + + SourceEgress = "egress" +) + +// InboundSource returns the checks.source value for the given prober site +// index (1-based), e.g. InboundSource(1) == "inbound-site-1". +func InboundSource(siteIndex int) string { + return "inbound-site-" + itoa(siteIndex) +} + +func itoa(n int) string { + if n == 0 { + return "0" + } + neg := n < 0 + if neg { + n = -n + } + var buf [20]byte + i := len(buf) + for n > 0 { + i-- + buf[i] = byte('0' + n%10) + n /= 10 + } + if neg { + i-- + buf[i] = '-' + } + return string(buf[i:]) +} + +type Validator struct { + ValidatorID string + Hostname string + OSPortID string + State string + CurrentIPID *int64 + AgentVersion string + LastHeartbeatAt *time.Time + CreatedAt time.Time + UpdatedAt time.Time +} + +type IPQueueItem struct { + ID int64 + IPAddress string + Sequence int + State string + OwnerValidatorID *string + FIPID string + AttemptNumber int + RetryCount int + LeaseExpiresAt *time.Time + EgressComplete bool + Site1Complete bool + Site2Complete bool + Site3Complete bool + OverallResult string + AssignedAt *time.Time + AggregatedAt *time.Time + FIPReleasedAt *time.Time + CreatedAt time.Time + UpdatedAt time.Time +} + +type Check struct { + ID int64 + IPID int64 + IPAddress string + AttemptNumber int + ValidatorID string + Source string + CheckType string + Target string + Success bool + LatencyMS int64 + Detail string + CheckedAt time.Time + CreatedAt time.Time +} + +type Event struct { + ID int64 + SourceType string + SourceID string + IPID *int64 + EventType string + Payload string + OccurredAt time.Time + CreatedAt time.Time +} diff --git a/internal/db/queries_checks.go b/internal/db/queries_checks.go new file mode 100644 index 0000000..3d5038d --- /dev/null +++ b/internal/db/queries_checks.go @@ -0,0 +1,58 @@ +package db + +import ( + "context" +) + +// UpsertCheck records (or, on retry, overwrites) a single check result. The +// UNIQUE(ip_id, attempt_number, source, check_type, target) constraint plus +// this upsert is what makes agent/prober result submission safely +// retryable without producing duplicate rows. +func (d *DB) UpsertCheck(ctx context.Context, c Check) error { + _, err := d.ExecContext(ctx, ` + INSERT INTO checks (ip_id, ip_address, attempt_number, validator_id, source, check_type, target, + success, latency_ms, detail, checked_at, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(ip_id, attempt_number, source, check_type, target) DO UPDATE SET + validator_id=excluded.validator_id, + success=excluded.success, + latency_ms=excluded.latency_ms, + detail=excluded.detail, + checked_at=excluded.checked_at + `, c.IPID, c.IPAddress, c.AttemptNumber, c.ValidatorID, c.Source, c.CheckType, c.Target, + c.Success, c.LatencyMS, c.Detail, timeToDB(c.CheckedAt), timeToDB(Now())) + return err +} + +// ListChecksForAttempt returns every check recorded for an IP's current +// attempt — the input to overall-result aggregation. +func (d *DB) ListChecksForAttempt(ctx context.Context, ipID int64, attemptNumber int) ([]Check, error) { + rows, err := d.QueryContext(ctx, ` + SELECT id, ip_id, ip_address, attempt_number, validator_id, source, check_type, target, + success, latency_ms, detail, checked_at, created_at + FROM checks WHERE ip_id=? AND attempt_number=? + ORDER BY source, check_type, target + `, ipID, attemptNumber) + if err != nil { + return nil, err + } + defer rows.Close() + + var out []Check + for rows.Next() { + var c Check + var checkedAt, createdAt string + if err := rows.Scan(&c.ID, &c.IPID, &c.IPAddress, &c.AttemptNumber, &c.ValidatorID, &c.Source, + &c.CheckType, &c.Target, &c.Success, &c.LatencyMS, &c.Detail, &checkedAt, &createdAt); err != nil { + return nil, err + } + if c.CheckedAt, err = dbToTime(checkedAt); err != nil { + return nil, err + } + if c.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + out = append(out, c) + } + return out, rows.Err() +} diff --git a/internal/db/queries_events.go b/internal/db/queries_events.go new file mode 100644 index 0000000..73b3d2e --- /dev/null +++ b/internal/db/queries_events.go @@ -0,0 +1,63 @@ +package db + +import "context" + +func (d *DB) InsertEvent(ctx context.Context, e Event) error { + _, err := d.ExecContext(ctx, ` + INSERT INTO events (source_type, source_id, ip_id, event_type, payload, occurred_at, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?) + `, e.SourceType, e.SourceID, e.IPID, e.EventType, e.Payload, timeToDB(e.OccurredAt), timeToDB(Now())) + return err +} + +// ListEventsForIP returns the audit trail for a single IP, most recent +// first — used by the admin detail endpoint. +func (d *DB) ListEventsForIP(ctx context.Context, ipID int64) ([]Event, error) { + rows, err := d.QueryContext(ctx, ` + SELECT id, source_type, source_id, ip_id, event_type, payload, occurred_at, created_at + FROM events WHERE ip_id=? ORDER BY occurred_at DESC + `, ipID) + if err != nil { + return nil, err + } + defer rows.Close() + return scanEvents(rows) +} + +func (d *DB) ListRecentEvents(ctx context.Context, limit int) ([]Event, error) { + rows, err := d.QueryContext(ctx, ` + SELECT id, source_type, source_id, ip_id, event_type, payload, occurred_at, created_at + FROM events ORDER BY id DESC LIMIT ? + `, limit) + if err != nil { + return nil, err + } + defer rows.Close() + return scanEvents(rows) +} + +func scanEvents(rows interface { + Next() bool + Scan(...interface{}) error + Err() error +}) ([]Event, error) { + var out []Event + for rows.Next() { + var e Event + var ipID *int64 + var occurredAt, createdAt string + if err := rows.Scan(&e.ID, &e.SourceType, &e.SourceID, &ipID, &e.EventType, &e.Payload, &occurredAt, &createdAt); err != nil { + return nil, err + } + e.IPID = ipID + var err error + if e.OccurredAt, err = dbToTime(occurredAt); err != nil { + return nil, err + } + if e.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + out = append(out, e) + } + return out, rows.Err() +} diff --git a/internal/db/queries_ipqueue.go b/internal/db/queries_ipqueue.go new file mode 100644 index 0000000..75268c9 --- /dev/null +++ b/internal/db/queries_ipqueue.go @@ -0,0 +1,353 @@ +package db + +import ( + "context" + "database/sql" + "fmt" + "time" +) + +// SeedQueue inserts the configured IP address list in order, assigning each +// a stable sequence number. Re-running with the same list is a no-op for +// addresses already present (ON CONFLICT DO NOTHING keyed by the UNIQUE +// ip_address column), so restarting control-api against the same config +// never re-queues already-processed addresses. +func (d *DB) SeedQueue(ctx context.Context, addresses []string) error { + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() + + now := timeToDB(Now()) + for i, addr := range addresses { + _, err := tx.ExecContext(ctx, ` + INSERT INTO ip_queue (ip_address, sequence, state, created_at, updated_at) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(ip_address) DO NOTHING + `, addr, i, IPQueued, now, now) + if err != nil { + return fmt.Errorf("seed %s: %w", addr, err) + } + } + return tx.Commit() +} + +// ClaimNextQueued atomically hands the next queued IP (lowest sequence) to +// the given idle validator. It returns (nil, nil) if the validator isn't +// idle or no IP is queued. The DB connection pool is capped at one physical +// connection (see Open), so this transaction already has exclusive access +// to the database for its duration — no other claim, requeue, or update can +// interleave — which combined with the conditional UPDATEs (checked via +// RowsAffected) guarantees a single IP is never claimed by two validators. +func (d *DB) ClaimNextQueued(ctx context.Context, validatorID string, leaseTTL time.Duration) (*IPQueueItem, error) { + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return nil, err + } + defer tx.Rollback() + + var state string + err = tx.QueryRowContext(ctx, `SELECT state FROM validators WHERE validator_id=?`, validatorID).Scan(&state) + if err == sql.ErrNoRows { + return nil, nil + } + if err != nil { + return nil, err + } + if state != ValidatorIdle { + return nil, nil + } + + var item IPQueueItem + err = tx.QueryRowContext(ctx, ` + SELECT id, ip_address, sequence, attempt_number, retry_count + FROM ip_queue WHERE state=? ORDER BY sequence LIMIT 1 + `, IPQueued).Scan(&item.ID, &item.IPAddress, &item.Sequence, &item.AttemptNumber, &item.RetryCount) + if err == sql.ErrNoRows { + return nil, nil + } + if err != nil { + return nil, err + } + + now := Now() + lease := now.Add(leaseTTL) + res, err := tx.ExecContext(ctx, ` + UPDATE ip_queue SET state=?, owner_validator_id=?, assigned_at=?, lease_expires_at=?, updated_at=? + WHERE id=? AND state=? + `, IPAssigningFIP, validatorID, timeToDB(now), timeToDB(lease), timeToDB(now), item.ID, IPQueued) + if err != nil { + return nil, err + } + if n, _ := res.RowsAffected(); n != 1 { + return nil, nil + } + + res, err = tx.ExecContext(ctx, ` + UPDATE validators SET state=?, current_ip_id=?, updated_at=? + WHERE validator_id=? AND state=? + `, ValidatorAssigned, item.ID, timeToDB(now), validatorID, ValidatorIdle) + if err != nil { + return nil, err + } + if n, _ := res.RowsAffected(); n != 1 { + return nil, nil + } + + if err := tx.Commit(); err != nil { + return nil, err + } + + item.State = IPAssigningFIP + ownerID := validatorID + item.OwnerValidatorID = &ownerID + item.AssignedAt = &now + item.LeaseExpiresAt = &lease + return &item, nil +} + +func (d *DB) SetFIPAssociated(ctx context.Context, ipID int64, fipID string, leaseTTL time.Duration) error { + now := Now() + _, err := d.ExecContext(ctx, ` + UPDATE ip_queue SET state=?, fip_id=?, lease_expires_at=?, updated_at=? + WHERE id=? + `, IPAwaitingSelfCheck, fipID, timeToDB(now.Add(leaseTTL)), timeToDB(now), ipID) + return err +} + +func (d *DB) SetChecking(ctx context.Context, ipID int64, leaseTTL time.Duration) error { + now := Now() + _, err := d.ExecContext(ctx, ` + UPDATE ip_queue SET state=?, lease_expires_at=?, updated_at=? + WHERE id=? + `, IPChecking, timeToDB(now.Add(leaseTTL)), timeToDB(now), ipID) + return err +} + +func (d *DB) SetAggregating(ctx context.Context, ipID int64) error { + _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=?, updated_at=? WHERE id=?`, + IPAggregating, timeToDB(Now()), ipID) + return err +} + +// FinishIP records the aggregated result and marks the IP done or failed. +func (d *DB) FinishIP(ctx context.Context, ipID int64, result string) error { + state := IPDone + if result == ResultFail { + state = IPFailed + } + now := timeToDB(Now()) + _, err := d.ExecContext(ctx, ` + UPDATE ip_queue SET state=?, overall_result=?, aggregated_at=?, updated_at=? + WHERE id=? + `, state, result, now, now, ipID) + return err +} + +// ReleaseFIP records that the floating IP has been disassociated and frees +// the owning validator back to idle, in one transaction. +func (d *DB) ReleaseFIP(ctx context.Context, ipID int64, validatorID string) error { + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() + + now := timeToDB(Now()) + if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET fip_released_at=?, updated_at=? WHERE id=?`, now, now, ipID); err != nil { + return err + } + if _, err := tx.ExecContext(ctx, ` + UPDATE validators SET state=?, current_ip_id=NULL, updated_at=? + WHERE validator_id=? + `, ValidatorIdle, now, validatorID); err != nil { + return err + } + return tx.Commit() +} + +// RequeueOrFail is used by both the retry path (association/self-check +// failure) and the lease-sweep reclaim path. It clears ownership and +// per-attempt progress, bumps attempt_number and retry_count, and either +// sends the IP back to the queue or marks it permanently failed once +// maxRetries is exceeded. The owning validator (if any) is freed in the +// same transaction. +func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string, maxRetries int) error { + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() + + var retryCount int + if err := tx.QueryRowContext(ctx, `SELECT retry_count FROM ip_queue WHERE id=?`, ipID).Scan(&retryCount); err != nil { + return err + } + retryCount++ + + now := timeToDB(Now()) + nextState := IPQueued + if retryCount > maxRetries { + nextState = IPFailed + } + + if nextState == IPQueued { + _, err = tx.ExecContext(ctx, ` + UPDATE ip_queue SET + state=?, owner_validator_id=NULL, fip_id='', retry_count=?, attempt_number=attempt_number+1, + lease_expires_at=NULL, egress_complete=0, site1_complete=0, site2_complete=0, site3_complete=0, + overall_result='', assigned_at=NULL, updated_at=? + WHERE id=? + `, nextState, retryCount, now, ipID) + } else { + _, err = tx.ExecContext(ctx, ` + UPDATE ip_queue SET + state=?, retry_count=?, overall_result=?, aggregated_at=?, updated_at=? + WHERE id=? + `, nextState, retryCount, ResultFail, now, now, ipID) + } + if err != nil { + return err + } + + if validatorID != "" { + if _, err := tx.ExecContext(ctx, ` + UPDATE validators SET state=?, current_ip_id=NULL, updated_at=? + WHERE validator_id=? + `, ValidatorIdle, now, validatorID); err != nil { + return err + } + } + return tx.Commit() +} + +func (d *DB) SetEgressComplete(ctx context.Context, ipID int64) error { + _, err := d.ExecContext(ctx, `UPDATE ip_queue SET egress_complete=1, updated_at=? WHERE id=?`, timeToDB(Now()), ipID) + return err +} + +// SetSiteComplete marks completion for prober site 1, 2, or 3. +func (d *DB) SetSiteComplete(ctx context.Context, ipID int64, siteIndex int) error { + col := map[int]string{1: "site1_complete", 2: "site2_complete", 3: "site3_complete"}[siteIndex] + if col == "" { + return fmt.Errorf("invalid site index %d", siteIndex) + } + _, err := d.ExecContext(ctx, fmt.Sprintf(`UPDATE ip_queue SET %s=1, updated_at=? WHERE id=?`, col), timeToDB(Now()), ipID) + return err +} + +func (d *DB) GetIP(ctx context.Context, ipID int64) (*IPQueueItem, error) { + row := d.QueryRowContext(ctx, ipQueueSelect+`WHERE id=?`, ipID) + return scanIPQueueItem(row) +} + +func (d *DB) GetIPByAddress(ctx context.Context, address string) (*IPQueueItem, error) { + row := d.QueryRowContext(ctx, ipQueueSelect+`WHERE ip_address=?`, address) + return scanIPQueueItem(row) +} + +func (d *DB) ListIPs(ctx context.Context) ([]IPQueueItem, error) { + rows, err := d.QueryContext(ctx, ipQueueSelect+`ORDER BY sequence`) + if err != nil { + return nil, err + } + defer rows.Close() + return scanIPQueueItems(rows) +} + +// ListChecking returns all IPs currently in the checking state — the set a +// prober should be actively probing. +func (d *DB) ListChecking(ctx context.Context) ([]IPQueueItem, error) { + rows, err := d.QueryContext(ctx, ipQueueSelect+`WHERE state=? ORDER BY sequence`, IPChecking) + if err != nil { + return nil, err + } + defer rows.Close() + return scanIPQueueItems(rows) +} + +// ListReadyToAggregate returns checking-state IPs where every source has +// reported completion, or whose checking window has expired. +func (d *DB) ListReadyToAggregate(ctx context.Context, windowDeadline time.Time) ([]IPQueueItem, error) { + rows, err := d.QueryContext(ctx, ipQueueSelect+` + WHERE state=? AND ( + (egress_complete=1 AND site1_complete=1 AND site2_complete=1 AND site3_complete=1) + OR assigned_at < ? + )`, IPChecking, timeToDB(windowDeadline)) + if err != nil { + return nil, err + } + defer rows.Close() + return scanIPQueueItems(rows) +} + +// ListExpiredLeases returns non-terminal IPs whose lease has expired — +// candidates for the lease sweep (crash recovery + stuck-validator reclaim). +func (d *DB) ListExpiredLeases(ctx context.Context, now time.Time) ([]IPQueueItem, error) { + rows, err := d.QueryContext(ctx, ipQueueSelect+` + WHERE state NOT IN (?, ?) AND lease_expires_at IS NOT NULL AND lease_expires_at < ? + `, IPDone, IPFailed, timeToDB(now)) + if err != nil { + return nil, err + } + defer rows.Close() + return scanIPQueueItems(rows) +} + +const ipQueueSelect = ` + SELECT id, ip_address, sequence, state, owner_validator_id, fip_id, attempt_number, retry_count, + lease_expires_at, egress_complete, site1_complete, site2_complete, site3_complete, overall_result, + assigned_at, aggregated_at, fip_released_at, created_at, updated_at + FROM ip_queue +` + +func scanIPQueueItems(rows *sql.Rows) ([]IPQueueItem, error) { + var out []IPQueueItem + for rows.Next() { + item, err := scanIPQueueItem(rows) + if err != nil { + return nil, err + } + out = append(out, *item) + } + return out, rows.Err() +} + +func scanIPQueueItem(row rowScanner) (*IPQueueItem, error) { + var item IPQueueItem + var owner sql.NullString + var leaseExpires, assignedAt, aggregatedAt, fipReleasedAt sql.NullString + var createdAt, updatedAt string + if err := row.Scan( + &item.ID, &item.IPAddress, &item.Sequence, &item.State, &owner, &item.FIPID, + &item.AttemptNumber, &item.RetryCount, &leaseExpires, + &item.EgressComplete, &item.Site1Complete, &item.Site2Complete, &item.Site3Complete, + &item.OverallResult, &assignedAt, &aggregatedAt, &fipReleasedAt, &createdAt, &updatedAt, + ); err != nil { + return nil, err + } + if owner.Valid { + item.OwnerValidatorID = &owner.String + } + var err error + if item.LeaseExpiresAt, err = nullStringToTimePtr(leaseExpires); err != nil { + return nil, err + } + if item.AssignedAt, err = nullStringToTimePtr(assignedAt); err != nil { + return nil, err + } + if item.AggregatedAt, err = nullStringToTimePtr(aggregatedAt); err != nil { + return nil, err + } + if item.FIPReleasedAt, err = nullStringToTimePtr(fipReleasedAt); err != nil { + return nil, err + } + if item.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + if item.UpdatedAt, err = dbToTime(updatedAt); err != nil { + return nil, err + } + return &item, nil +} diff --git a/internal/db/queries_validators.go b/internal/db/queries_validators.go new file mode 100644 index 0000000..ea494e6 --- /dev/null +++ b/internal/db/queries_validators.go @@ -0,0 +1,179 @@ +package db + +import ( + "context" + "database/sql" + "fmt" + "time" +) + +// RegisterValidator inserts a new validator or updates an existing one's +// hostname/agent_version on re-registration. It never overwrites the state +// of a validator that's mid-assignment, so an agent restarting while it +// owns an IP doesn't silently lose that ownership. +func (d *DB) RegisterValidator(ctx context.Context, validatorID, hostname, osPortID, agentVersion string) error { + now := timeToDB(Now()) + _, err := d.ExecContext(ctx, ` + INSERT INTO validators (validator_id, hostname, os_port_id, agent_version, state, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(validator_id) DO UPDATE SET + hostname=excluded.hostname, + os_port_id=excluded.os_port_id, + agent_version=excluded.agent_version, + updated_at=excluded.updated_at + `, validatorID, hostname, osPortID, agentVersion, ValidatorIdle, now, now) + if err != nil { + return fmt.Errorf("register validator: %w", err) + } + // A brand-new row already lands in ValidatorIdle via the INSERT branch; + // a re-registering validator that was 'unregistered' or 'unreachable' + // (but not mid-assignment) should also come back to idle. + _, err = d.ExecContext(ctx, ` + UPDATE validators SET state=?, updated_at=? + WHERE validator_id=? AND state IN (?, ?) + `, ValidatorIdle, now, validatorID, ValidatorUnregistered, ValidatorUnreachable) + if err != nil { + return fmt.Errorf("register validator (reactivate): %w", err) + } + return nil +} + +func (d *DB) Heartbeat(ctx context.Context, validatorID string) error { + now := timeToDB(Now()) + res, err := d.ExecContext(ctx, ` + UPDATE validators SET last_heartbeat_at=?, updated_at=?, + state = CASE WHEN state=? THEN ? ELSE state END + WHERE validator_id=? + `, now, now, ValidatorUnreachable, ValidatorIdle, validatorID) + if err != nil { + return fmt.Errorf("heartbeat: %w", err) + } + n, _ := res.RowsAffected() + if n == 0 { + return sql.ErrNoRows + } + return nil +} + +func (d *DB) GetValidator(ctx context.Context, validatorID string) (*Validator, error) { + row := d.QueryRowContext(ctx, ` + SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version, + last_heartbeat_at, created_at, updated_at + FROM validators WHERE validator_id=? + `, validatorID) + return scanValidator(row) +} + +func (d *DB) ListValidators(ctx context.Context) ([]Validator, error) { + rows, err := d.QueryContext(ctx, ` + SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version, + last_heartbeat_at, created_at, updated_at + FROM validators ORDER BY validator_id + `) + if err != nil { + return nil, err + } + defer rows.Close() + var out []Validator + for rows.Next() { + v, err := scanValidator(rows) + if err != nil { + return nil, err + } + out = append(out, *v) + } + return out, rows.Err() +} + +// ListIdleValidators returns validators currently eligible to be handed a +// new IP to work on. +func (d *DB) ListIdleValidators(ctx context.Context) ([]Validator, error) { + rows, err := d.QueryContext(ctx, ` + SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version, + last_heartbeat_at, created_at, updated_at + FROM validators WHERE state=? ORDER BY validator_id + `, ValidatorIdle) + if err != nil { + return nil, err + } + defer rows.Close() + var out []Validator + for rows.Next() { + v, err := scanValidator(rows) + if err != nil { + return nil, err + } + out = append(out, *v) + } + return out, rows.Err() +} + +// ListStaleHeartbeats returns validators whose last heartbeat predates the +// given cutoff and that aren't already marked unreachable. +func (d *DB) ListStaleHeartbeats(ctx context.Context, cutoff time.Time) ([]Validator, error) { + rows, err := d.QueryContext(ctx, ` + SELECT validator_id, hostname, os_port_id, state, current_ip_id, agent_version, + last_heartbeat_at, created_at, updated_at + FROM validators + WHERE state != ? AND last_heartbeat_at IS NOT NULL AND last_heartbeat_at < ? + `, ValidatorUnreachable, timeToDB(cutoff)) + if err != nil { + return nil, err + } + defer rows.Close() + var out []Validator + for rows.Next() { + v, err := scanValidator(rows) + if err != nil { + return nil, err + } + out = append(out, *v) + } + return out, rows.Err() +} + +func (d *DB) MarkValidatorUnreachable(ctx context.Context, validatorID string) error { + _, err := d.ExecContext(ctx, `UPDATE validators SET state=?, updated_at=? WHERE validator_id=?`, + ValidatorUnreachable, timeToDB(Now()), validatorID) + return err +} + +// FreeValidator returns a validator to idle with no assigned IP. Used after +// an IP finishes (success or failure) or is reclaimed by the lease sweep. +func (d *DB) FreeValidator(ctx context.Context, validatorID string) error { + _, err := d.ExecContext(ctx, ` + UPDATE validators SET state=?, current_ip_id=NULL, updated_at=? + WHERE validator_id=? + `, ValidatorIdle, timeToDB(Now()), validatorID) + return err +} + +type rowScanner interface { + Scan(dest ...interface{}) error +} + +func scanValidator(row rowScanner) (*Validator, error) { + var v Validator + var currentIPID sql.NullInt64 + var lastHeartbeat sql.NullString + var createdAt, updatedAt string + if err := row.Scan(&v.ValidatorID, &v.Hostname, &v.OSPortID, &v.State, ¤tIPID, + &v.AgentVersion, &lastHeartbeat, &createdAt, &updatedAt); err != nil { + return nil, err + } + if currentIPID.Valid { + v.CurrentIPID = ¤tIPID.Int64 + } + hb, err := nullStringToTimePtr(lastHeartbeat) + if err != nil { + return nil, err + } + v.LastHeartbeatAt = hb + if v.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + if v.UpdatedAt, err = dbToTime(updatedAt); err != nil { + return nil, err + } + return &v, nil +} diff --git a/internal/db/timeutil.go b/internal/db/timeutil.go new file mode 100644 index 0000000..af84b52 --- /dev/null +++ b/internal/db/timeutil.go @@ -0,0 +1,38 @@ +package db + +import ( + "database/sql" + "time" +) + +// Timestamps are stored as RFC3339Nano TEXT explicitly (rather than relying +// on driver-specific time.Time marshaling) so the on-disk format is stable +// and easy to inspect with the sqlite3 CLI. + +const timeLayout = time.RFC3339Nano + +func timeToDB(t time.Time) string { + return t.UTC().Format(timeLayout) +} + +func timePtrToDB(t *time.Time) interface{} { + if t == nil { + return nil + } + return timeToDB(*t) +} + +func dbToTime(s string) (time.Time, error) { + return time.Parse(timeLayout, s) +} + +func nullStringToTimePtr(ns sql.NullString) (*time.Time, error) { + if !ns.Valid || ns.String == "" { + return nil, nil + } + t, err := dbToTime(ns.String) + if err != nil { + return nil, err + } + return &t, nil +} diff --git a/internal/httpapi/dto.go b/internal/httpapi/dto.go new file mode 100644 index 0000000..b5dfa49 --- /dev/null +++ b/internal/httpapi/dto.go @@ -0,0 +1,98 @@ +package httpapi + +// Request/response bodies for the /api/v1 surface. Kept in one file since +// they're small and mostly 1:1 with a single handler each. + +type registerAgentRequest struct { + ValidatorID string `json:"validator_id"` + Hostname string `json:"hostname"` + AgentVersion string `json:"agent_version"` +} + +type registerAgentResponse struct { + OK bool `json:"ok"` + PollIntervalSeconds int `json:"poll_interval_seconds"` +} + +type heartbeatRequest struct { + LocalState string `json:"local_state"` + CurrentIPAddress string `json:"current_ip_address,omitempty"` +} + +type okResponse struct { + OK bool `json:"ok"` +} + +type checkConfigDTO struct { + Type string `json:"type"` + Targets []string `json:"targets"` +} + +type assignmentResponse struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + Phase string `json:"phase"` + CheckConfig []checkConfigDTO `json:"check_config,omitempty"` +} + +type selfCheckRequest struct { + IPID int64 `json:"ip_id"` + DetectedEgress string `json:"detected_egress_ip"` + Success bool `json:"success"` + Detail string `json:"detail"` +} + +type agentEventRequest struct { + EventType string `json:"event_type"` + IPID *int64 `json:"ip_id,omitempty"` + Payload string `json:"payload,omitempty"` +} + +type checkResultDTO struct { + IPID int64 `json:"ip_id"` + CheckType string `json:"check_type"` + Target string `json:"target,omitempty"` + Success bool `json:"success"` + LatencyMS int64 `json:"latency_ms"` + Detail string `json:"detail,omitempty"` + CheckedAt string `json:"checked_at"` +} + +type agentResultsRequest struct { + Results []checkResultDTO `json:"results"` +} + +type agentCompleteRequest struct { + IPID int64 `json:"ip_id"` +} + +type registerProberRequest struct { + SiteID string `json:"site_id"` + Hostname string `json:"hostname"` +} + +type proberAssignment struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + Ports []int `json:"ports"` + ICMP bool `json:"icmp"` +} + +type proberResultDTO struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + CheckType string `json:"check_type"` + Success bool `json:"success"` + LatencyMS int64 `json:"latency_ms"` + Detail string `json:"detail,omitempty"` + CheckedAt string `json:"checked_at"` + Complete bool `json:"complete"` +} + +type proberResultsRequest struct { + Results []proberResultDTO `json:"results"` +} + +type errorResponse struct { + Error string `json:"error"` +} diff --git a/internal/httpapi/handlers_admin.go b/internal/httpapi/handlers_admin.go new file mode 100644 index 0000000..cbba00e --- /dev/null +++ b/internal/httpapi/handlers_admin.go @@ -0,0 +1,75 @@ +package httpapi + +import ( + "net/http" + + "cloudipvalidator/internal/db" +) + +func (s *Server) handleHealthz(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleAdminStatus(w http.ResponseWriter, r *http.Request) { + ips, err := s.DB.ListIPs(r.Context()) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + validators, err := s.DB.ListValidators(r.Context()) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + byState := map[string]int{} + for _, ip := range ips { + byState[ip.State]++ + } + writeJSON(w, http.StatusOK, map[string]interface{}{ + "total_ips": len(ips), + "ips_by_state": byState, + "total_validators": len(validators), + }) +} + +func (s *Server) handleAdminIPs(w http.ResponseWriter, r *http.Request) { + ips, err := s.DB.ListIPs(r.Context()) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + writeJSON(w, http.StatusOK, ips) +} + +func (s *Server) handleAdminIPDetail(w http.ResponseWriter, r *http.Request) { + address := r.PathValue("ip") + item, err := s.DB.GetIPByAddress(r.Context(), address) + if err != nil { + writeError(w, http.StatusNotFound, "unknown ip: "+address) + return + } + checks, err := s.DB.ListChecksForAttempt(r.Context(), item.ID, item.AttemptNumber) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + events, err := s.DB.ListEventsForIP(r.Context(), item.ID) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + writeJSON(w, http.StatusOK, struct { + IP *db.IPQueueItem `json:"ip"` + Checks []db.Check `json:"checks"` + Events []db.Event `json:"events"` + }{item, checks, events}) +} + +func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) { + validators, err := s.DB.ListValidators(r.Context()) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + writeJSON(w, http.StatusOK, validators) +} diff --git a/internal/httpapi/handlers_agent.go b/internal/httpapi/handlers_agent.go new file mode 100644 index 0000000..6275d0b --- /dev/null +++ b/internal/httpapi/handlers_agent.go @@ -0,0 +1,145 @@ +package httpapi + +import ( + "net/http" + "time" + + "cloudipvalidator/internal/db" +) + +func (s *Server) handleAgentRegister(w http.ResponseWriter, r *http.Request) { + var req registerAgentRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if req.ValidatorID == "" { + writeError(w, http.StatusBadRequest, "validator_id is required") + return + } + // os_port_id is supplied via control-api's own config (config.ValidatorConfig), + // not by the agent, so registration only touches hostname/version here; + // RegisterValidator preserves any existing os_port_id row. + existing, _ := s.DB.GetValidator(r.Context(), req.ValidatorID) + osPortID := "" + if existing != nil { + osPortID = existing.OSPortID + } + if err := s.DB.RegisterValidator(r.Context(), req.ValidatorID, req.Hostname, osPortID, req.AgentVersion); err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + s.Orch.RecordEvent(r.Context(), "validator-agent", req.ValidatorID, nil, "registered", "") + writeJSON(w, http.StatusOK, registerAgentResponse{OK: true, PollIntervalSeconds: s.Orch.Cfg.PollIntervalSeconds}) +} + +func (s *Server) handleAgentHeartbeat(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + var req heartbeatRequest + _ = readJSON(r, &req) // heartbeat body is informational only; tolerate empty/missing + if err := s.DB.Heartbeat(r.Context(), id); err != nil { + writeError(w, http.StatusNotFound, "unknown validator: "+id) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleAgentAssignment(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + item, checks, err := s.Orch.AssignmentForValidator(r.Context(), id) + if err != nil { + writeError(w, http.StatusNotFound, "unknown validator: "+id) + return + } + if item == nil { + w.WriteHeader(http.StatusNoContent) + return + } + var cfg []checkConfigDTO + for _, c := range checks { + cfg = append(cfg, checkConfigDTO{Type: c.Type, Targets: c.Targets}) + } + writeJSON(w, http.StatusOK, assignmentResponse{ + IPID: item.ID, IPAddress: item.IPAddress, Phase: item.State, CheckConfig: cfg, + }) +} + +func (s *Server) handleAgentSelfCheck(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + var req selfCheckRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + detail := req.Detail + if req.DetectedEgress != "" { + detail = "detected_egress_ip=" + req.DetectedEgress + " " + detail + } + if err := s.Orch.SelfCheckResult(r.Context(), id, req.IPID, req.Success, detail); err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleAgentEvent(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + var req agentEventRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if req.EventType == "" { + writeError(w, http.StatusBadRequest, "event_type is required") + return + } + s.Orch.RecordEvent(r.Context(), "validator-agent", id, req.IPID, req.EventType, req.Payload) + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleAgentResults(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + var req agentResultsRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + for _, res := range req.Results { + item, err := s.DB.GetIP(r.Context(), res.IPID) + if err != nil { + writeError(w, http.StatusNotFound, "unknown ip_id") + return + } + checkedAt, err := time.Parse(time.RFC3339Nano, res.CheckedAt) + if err != nil { + checkedAt = db.Now() + } + err = s.Orch.RecordCheck(r.Context(), db.Check{ + IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber, + ValidatorID: id, Source: db.SourceEgress, CheckType: res.CheckType, Target: res.Target, + Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt, + }) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleAgentComplete(w http.ResponseWriter, r *http.Request) { + var req agentCompleteRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if err := s.Orch.MarkEgressComplete(r.Context(), req.IPID); err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleWhatsMyIP(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusOK, map[string]string{"ip": remoteIP(r)}) +} diff --git a/internal/httpapi/handlers_prober.go b/internal/httpapi/handlers_prober.go new file mode 100644 index 0000000..577433f --- /dev/null +++ b/internal/httpapi/handlers_prober.go @@ -0,0 +1,92 @@ +package httpapi + +import ( + "net/http" + "time" + + "cloudipvalidator/internal/db" +) + +func (s *Server) handleProberRegister(w http.ResponseWriter, r *http.Request) { + var req registerProberRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if s.Orch.SiteIndexForID(req.SiteID) == 0 { + writeError(w, http.StatusBadRequest, "unknown site_id: "+req.SiteID) + return + } + s.Orch.RecordEvent(r.Context(), "prober", req.SiteID, nil, "registered", "") + writeJSON(w, http.StatusOK, registerAgentResponse{OK: true, PollIntervalSeconds: s.Orch.Cfg.PollIntervalSeconds}) +} + +// handleProberAssignments returns every IP currently in the checking +// state — probers work the whole active set each poll, not one IP at a +// time, since multiple validators run in parallel. +func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) { + siteID := r.PathValue("site_id") + if s.Orch.SiteIndexForID(siteID) == 0 { + writeError(w, http.StatusNotFound, "unknown site_id: "+siteID) + return + } + items, err := s.DB.ListChecking(r.Context()) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + out := make([]proberAssignment, 0, len(items)) + for _, item := range items { + out = append(out, proberAssignment{ + IPID: item.ID, IPAddress: item.IPAddress, + Ports: s.Orch.Inbound.Ports, ICMP: s.Orch.Inbound.ICMP, + }) + } + writeJSON(w, http.StatusOK, out) +} + +func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) { + siteID := r.PathValue("site_id") + siteIndex := s.Orch.SiteIndexForID(siteID) + if siteIndex == 0 { + writeError(w, http.StatusNotFound, "unknown site_id: "+siteID) + return + } + var req proberResultsRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + + completed := map[int64]bool{} + for _, res := range req.Results { + item, err := s.DB.GetIP(r.Context(), res.IPID) + if err != nil { + writeError(w, http.StatusNotFound, "unknown ip_id") + return + } + checkedAt, err := time.Parse(time.RFC3339Nano, res.CheckedAt) + if err != nil { + checkedAt = db.Now() + } + err = s.Orch.RecordCheck(r.Context(), db.Check{ + IPID: item.ID, IPAddress: item.IPAddress, AttemptNumber: item.AttemptNumber, + Source: db.InboundSource(siteIndex), CheckType: res.CheckType, Target: res.IPAddress, + Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, CheckedAt: checkedAt, + }) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + if res.Complete { + completed[res.IPID] = true + } + } + for ipID := range completed { + if err := s.Orch.MarkSiteComplete(r.Context(), ipID, siteIndex); err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} diff --git a/internal/httpapi/httpapi_test.go b/internal/httpapi/httpapi_test.go new file mode 100644 index 0000000..d66bc7b --- /dev/null +++ b/internal/httpapi/httpapi_test.go @@ -0,0 +1,210 @@ +package httpapi + +import ( + "bytes" + "context" + "encoding/json" + "io" + "log/slog" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/openstack" + "cloudipvalidator/internal/orchestrator" +) + +// fakeClient plays both a validator-agent and the three probers against a +// real httptest server, driving the full protocol exactly as the real +// binaries would, to prove the HTTP layer and orchestrator agree on state +// transitions end-to-end. +type fakeClient struct { + t *testing.T + base string + client *http.Client +} + +func (f *fakeClient) do(method, path string, body interface{}) (*http.Response, []byte) { + f.t.Helper() + var reader io.Reader + if body != nil { + b, err := json.Marshal(body) + if err != nil { + f.t.Fatalf("marshal body: %v", err) + } + reader = bytes.NewReader(b) + } + req, err := http.NewRequest(method, f.base+path, reader) + if err != nil { + f.t.Fatalf("new request: %v", err) + } + req.Header.Set("Content-Type", "application/json") + resp, err := f.client.Do(req) + if err != nil { + f.t.Fatalf("%s %s: %v", method, path, err) + } + defer resp.Body.Close() + respBody, _ := io.ReadAll(resp.Body) + return resp, respBody +} + +func TestEndToEndHTTPFlow(t *testing.T) { + ctx := context.Background() + dbPath := filepath.Join(t.TempDir(), "test.db") + d, err := db.Open(ctx, dbPath) + if err != nil { + t.Fatalf("open db: %v", err) + } + defer d.Close() + + mock := openstack.NewMockClient() + mock.Seed("fip-1", "1.2.3.4", "svc-project") + + cfg := &config.ControlAPI{ + Orchestrator: config.OrchestratorConfig{ + PollIntervalSeconds: 1, SelfCheckTimeoutSeconds: 10, MaxSelfCheckRetries: 3, + CheckingWindowSeconds: 120, MaxRetries: 3, LeaseTTLSeconds: 180, HeartbeatTimeoutSeconds: 30, + }, + Aggregation: config.AggregationConfig{MissingCountsAsFail: true}, + Sites: []config.SiteConfig{ + {SiteID: "site-1", Index: 1}, {SiteID: "site-2", Index: 2}, {SiteID: "site-3", Index: 3}, + }, + CheckTypes: []config.CheckTypeConfig{{Name: "https", Enabled: true, Targets: []string{"web"}}}, + Targets: map[string][]string{"web": {"https://example.test"}}, + Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true}, + } + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) + orch := orchestrator.New(d, mock, cfg, log) + + if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil { + t.Fatalf("seed queue: %v", err) + } + + srv := New(d, orch, log) + ts := httptest.NewServer(srv.Handler()) + defer ts.Close() + + fc := &fakeClient{t: t, base: ts.URL, client: ts.Client()} + + // Register the validator directly via DB (os_port_id comes from + // control-api config, not the agent's own registration call). + if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil { + t.Fatalf("register validator: %v", err) + } + + // Agent re-registers over HTTP (as the real binary would at startup). + resp, body := fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ + ValidatorID: "validator-1", Hostname: "host-1", AgentVersion: "v0.1", + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("register: status=%d body=%s", resp.StatusCode, body) + } + + // Orchestrator claims the IP and associates the FIP. + orch.Tick(ctx) + + resp, body = fc.do(http.MethodGet, "/api/v1/agents/validator-1/assignment", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("assignment: status=%d body=%s", resp.StatusCode, body) + } + var assignment assignmentResponse + if err := json.Unmarshal(body, &assignment); err != nil { + t.Fatalf("unmarshal assignment: %v", err) + } + if assignment.Phase != db.IPAwaitingSelfCheck { + t.Fatalf("expected awaiting_self_check, got %s", assignment.Phase) + } + if assignment.IPAddress != "1.2.3.4" { + t.Fatalf("expected 1.2.3.4, got %s", assignment.IPAddress) + } + + // Self-check via /whatsmyip: in the real deployment this would equal + // the FIP; here we just exercise the endpoint and always report success. + resp, body = fc.do(http.MethodGet, "/api/v1/whatsmyip", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("whatsmyip: status=%d body=%s", resp.StatusCode, body) + } + + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/self-check", selfCheckRequest{ + IPID: assignment.IPID, DetectedEgress: "1.2.3.4", Success: true, Detail: "matched", + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("self-check: status=%d body=%s", resp.StatusCode, body) + } + + // Agent runs its configured egress check and reports the result. + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{ + Results: []checkResultDTO{{ + IPID: assignment.IPID, CheckType: "https", Target: "https://example.test", + Success: true, LatencyMS: 12, CheckedAt: time.Now().Format(time.RFC3339Nano), + }}, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("results: status=%d body=%s", resp.StatusCode, body) + } + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/complete", agentCompleteRequest{IPID: assignment.IPID}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("complete: status=%d body=%s", resp.StatusCode, body) + } + + // Three probers register, poll, and report inbound results. + for _, site := range []string{"site-1", "site-2", "site-3"} { + resp, body = fc.do(http.MethodPost, "/api/v1/probers/register", registerProberRequest{SiteID: site}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("prober register %s: status=%d body=%s", site, resp.StatusCode, body) + } + + resp, body = fc.do(http.MethodGet, "/api/v1/probers/"+site+"/assignments", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("prober assignments %s: status=%d body=%s", site, resp.StatusCode, body) + } + var assignments []proberAssignment + if err := json.Unmarshal(body, &assignments); err != nil { + t.Fatalf("unmarshal assignments: %v", err) + } + if len(assignments) != 1 || assignments[0].IPAddress != "1.2.3.4" { + t.Fatalf("expected 1 assignment for 1.2.3.4, got %+v", assignments) + } + + now := time.Now().Format(time.RFC3339Nano) + resp, body = fc.do(http.MethodPost, "/api/v1/probers/"+site+"/results", proberResultsRequest{ + Results: []proberResultDTO{ + {IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "tcp-22", Success: true, CheckedAt: now}, + {IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "tcp-80", Success: true, CheckedAt: now}, + {IPID: assignments[0].IPID, IPAddress: "1.2.3.4", CheckType: "icmp", Success: true, CheckedAt: now, Complete: true}, + }, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("prober results %s: status=%d body=%s", site, resp.StatusCode, body) + } + } + + // Orchestrator sweep should now aggregate and release. + orch.Tick(ctx) + + resp, body = fc.do(http.MethodGet, "/api/v1/admin/ips/1.2.3.4", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("admin ip detail: status=%d body=%s", resp.StatusCode, body) + } + var detail struct { + IP db.IPQueueItem `json:"ip"` + } + if err := json.Unmarshal(body, &detail); err != nil { + t.Fatalf("unmarshal detail: %v", err) + } + if detail.IP.State != db.IPDone { + t.Fatalf("expected done, got %s", detail.IP.State) + } + if detail.IP.OverallResult != db.ResultPass { + t.Fatalf("expected pass, got %s", detail.IP.OverallResult) + } + + if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" { + t.Fatalf("expected fip disassociated at end of run") + } +} diff --git a/internal/httpapi/routes.go b/internal/httpapi/routes.go new file mode 100644 index 0000000..d36b649 --- /dev/null +++ b/internal/httpapi/routes.go @@ -0,0 +1,25 @@ +package httpapi + +import "net/http" + +func (s *Server) routes(mux *http.ServeMux) { + mux.HandleFunc("GET /healthz", s.handleHealthz) + mux.HandleFunc("GET /api/v1/whatsmyip", s.handleWhatsMyIP) + + mux.HandleFunc("POST /api/v1/agents/register", s.handleAgentRegister) + mux.HandleFunc("POST /api/v1/agents/{id}/heartbeat", s.handleAgentHeartbeat) + mux.HandleFunc("GET /api/v1/agents/{id}/assignment", s.handleAgentAssignment) + mux.HandleFunc("POST /api/v1/agents/{id}/self-check", s.handleAgentSelfCheck) + mux.HandleFunc("POST /api/v1/agents/{id}/events", s.handleAgentEvent) + mux.HandleFunc("POST /api/v1/agents/{id}/results", s.handleAgentResults) + mux.HandleFunc("POST /api/v1/agents/{id}/complete", s.handleAgentComplete) + + mux.HandleFunc("POST /api/v1/probers/register", s.handleProberRegister) + mux.HandleFunc("GET /api/v1/probers/{site_id}/assignments", s.handleProberAssignments) + mux.HandleFunc("POST /api/v1/probers/{site_id}/results", s.handleProberResults) + + mux.HandleFunc("GET /api/v1/admin/status", s.handleAdminStatus) + mux.HandleFunc("GET /api/v1/admin/ips", s.handleAdminIPs) + mux.HandleFunc("GET /api/v1/admin/ips/{ip}", s.handleAdminIPDetail) + mux.HandleFunc("GET /api/v1/admin/validators", s.handleAdminValidators) +} diff --git a/internal/httpapi/server.go b/internal/httpapi/server.go new file mode 100644 index 0000000..d7b880c --- /dev/null +++ b/internal/httpapi/server.go @@ -0,0 +1,71 @@ +// Package httpapi exposes the Control API's HTTP surface — the only way +// validator-agents, probers, and operators interact with the system. All +// business logic lives in internal/orchestrator; handlers here do request +// parsing/validation, call into the orchestrator or db package, and shape +// the JSON response. +package httpapi + +import ( + "encoding/json" + "log/slog" + "net" + "net/http" + + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/orchestrator" +) + +type Server struct { + DB *db.DB + Orch *orchestrator.Orchestrator + Log *slog.Logger +} + +func New(d *db.DB, o *orchestrator.Orchestrator, log *slog.Logger) *Server { + return &Server{DB: d, Orch: o, Log: log} +} + +func (s *Server) Handler() http.Handler { + mux := http.NewServeMux() + s.routes(mux) + return loggingMiddleware(s.Log, mux) +} + +func loggingMiddleware(log *slog.Logger, next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + next.ServeHTTP(w, r) + log.Debug("request", "method", r.Method, "path", r.URL.Path, "remote", r.RemoteAddr) + }) +} + +func writeJSON(w http.ResponseWriter, status int, v interface{}) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + if v != nil { + _ = json.NewEncoder(w).Encode(v) + } +} + +func writeError(w http.ResponseWriter, status int, msg string) { + writeJSON(w, status, errorResponse{Error: msg}) +} + +func readJSON(r *http.Request, v interface{}) error { + if r.Body == nil || r.ContentLength == 0 { + return nil + } + dec := json.NewDecoder(r.Body) + return dec.Decode(v) +} + +// remoteIP returns the caller's source IP with any port stripped. Used by +// /whatsmyip — the validator-agent's self-check mechanism relies on this +// being the actual TCP peer address (as SNAT'd by the newly associated +// FIP), never a client-supplied header. +func remoteIP(r *http.Request) string { + host, _, err := net.SplitHostPort(r.RemoteAddr) + if err != nil { + return r.RemoteAddr + } + return host +} diff --git a/internal/openstack/client.go b/internal/openstack/client.go new file mode 100644 index 0000000..6cd7b02 --- /dev/null +++ b/internal/openstack/client.go @@ -0,0 +1,104 @@ +package openstack + +import ( + "context" + "fmt" + + "github.com/gophercloud/gophercloud/v2" + osauth "github.com/gophercloud/gophercloud/v2/openstack" + "github.com/gophercloud/gophercloud/v2/openstack/networking/v2/extensions/layer3/floatingips" +) + +// ClientConfig carries pre-resolved credential values (already read from +// environment variables by the caller — see config.OpenStackAuth). Token +// auth is required per the deployment constraint that the admin credential +// is supplied via the process environment, not a config file. +type ClientConfig struct { + AuthURL string + Token string + ProjectID string + ProjectName string + DomainName string + Region string +} + +type Client struct { + networking *gophercloud.ServiceClient +} + +// NewClient authenticates against Keystone using a pre-issued admin token +// and returns a Client scoped to the given project/region, backed by the +// Neutron (networking v2) service catalog entry. +func NewClient(ctx context.Context, cfg ClientConfig) (*Client, error) { + if cfg.AuthURL == "" || cfg.Token == "" { + return nil, fmt.Errorf("openstack: auth URL and token are required") + } + + authOpts := gophercloud.AuthOptions{ + IdentityEndpoint: cfg.AuthURL, + TokenID: cfg.Token, + TenantID: cfg.ProjectID, + TenantName: cfg.ProjectName, + DomainName: cfg.DomainName, + } + if cfg.ProjectID != "" || cfg.ProjectName != "" { + authOpts.Scope = &gophercloud.AuthScope{ + ProjectID: cfg.ProjectID, + ProjectName: cfg.ProjectName, + DomainName: cfg.DomainName, + } + } + + provider, err := osauth.AuthenticatedClient(ctx, authOpts) + if err != nil { + return nil, fmt.Errorf("openstack: authenticate: %w", err) + } + + networking, err := osauth.NewNetworkV2(provider, gophercloud.EndpointOpts{Region: cfg.Region}) + if err != nil { + return nil, fmt.Errorf("openstack: networking client: %w", err) + } + + return &Client{networking: networking}, nil +} + +func (c *Client) GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error) { + pages, err := floatingips.List(c.networking, floatingips.ListOpts{FloatingIP: address}).AllPages(ctx) + if err != nil { + return nil, fmt.Errorf("openstack: list floating ips: %w", err) + } + list, err := floatingips.ExtractFloatingIPs(pages) + if err != nil { + return nil, fmt.Errorf("openstack: extract floating ips: %w", err) + } + if len(list) == 0 { + return nil, ErrNotFound(address) + } + f := list[0] + return &FloatingIP{ID: f.ID, Address: f.FloatingIP, PortID: f.PortID, ProjectID: f.TenantID}, nil +} + +func (c *Client) AssociateFloatingIP(ctx context.Context, fipID, portID string) error { + _, err := floatingips.Update(ctx, c.networking, fipID, floatingips.UpdateOpts{ + PortID: &portID, + }).Extract() + if err != nil { + return fmt.Errorf("openstack: associate floating ip %s -> port %s: %w", fipID, portID, err) + } + return nil +} + +func (c *Client) DisassociateFloatingIP(ctx context.Context, fipID string) error { + // gophercloud's UpdateOpts.PortID is *string with `omitempty`: a nil + // pointer is dropped from the request body entirely (no-op), while a + // pointer to "" is what actually serializes as port_id:null and + // disassociates the floating IP. See floatingips.UpdateOpts godoc. + empty := "" + _, err := floatingips.Update(ctx, c.networking, fipID, floatingips.UpdateOpts{ + PortID: &empty, + }).Extract() + if err != nil { + return fmt.Errorf("openstack: disassociate floating ip %s: %w", fipID, err) + } + return nil +} diff --git a/internal/openstack/client_live_test.go b/internal/openstack/client_live_test.go new file mode 100644 index 0000000..4e4bbee --- /dev/null +++ b/internal/openstack/client_live_test.go @@ -0,0 +1,48 @@ +package openstack + +import ( + "context" + "os" + "testing" + "time" +) + +// TestClientLive is a smoke test against a real OpenStack deployment. It's +// skipped unless OPENSTACK_LIVE_TEST=1 and the usual OS_* credential env +// vars are set, so the default `go test ./...` run needs no cloud access. +// It only exercises GetFloatingIPByAddress (read-only) against +// OS_TEST_FLOATING_IP, to avoid mutating real infrastructure in CI. +func TestClientLive(t *testing.T) { + if os.Getenv("OPENSTACK_LIVE_TEST") != "1" { + t.Skip("set OPENSTACK_LIVE_TEST=1 (and OS_AUTH_URL, OS_TOKEN, OS_TEST_FLOATING_IP) to run") + } + + testIP := os.Getenv("OS_TEST_FLOATING_IP") + if testIP == "" { + t.Fatal("OS_TEST_FLOATING_IP must name a floating IP address that exists in the target project") + } + + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + + client, err := NewClient(ctx, ClientConfig{ + AuthURL: os.Getenv("OS_AUTH_URL"), + Token: os.Getenv("OS_TOKEN"), + ProjectID: os.Getenv("OS_PROJECT_ID"), + ProjectName: os.Getenv("OS_PROJECT_NAME"), + DomainName: os.Getenv("OS_PROJECT_DOMAIN_NAME"), + Region: os.Getenv("OS_REGION_NAME"), + }) + if err != nil { + t.Fatalf("new client: %v", err) + } + + fip, err := client.GetFloatingIPByAddress(ctx, testIP) + if err != nil { + t.Fatalf("get floating ip %s: %v", testIP, err) + } + if fip.Address != testIP { + t.Fatalf("expected address %s, got %s", testIP, fip.Address) + } + t.Logf("found floating ip %s: id=%s port_id=%q", fip.Address, fip.ID, fip.PortID) +} diff --git a/internal/openstack/interface.go b/internal/openstack/interface.go new file mode 100644 index 0000000..4df55c3 --- /dev/null +++ b/internal/openstack/interface.go @@ -0,0 +1,49 @@ +// Package openstack isolates all OpenStack/Neutron interaction behind a +// small interface, so the orchestrator's business logic can be unit tested +// without a real cloud, and so the real implementation is swappable for a +// mock via config alone. +package openstack + +import "context" + +type FloatingIP struct { + ID string + Address string + PortID string // empty when not associated to any port + ProjectID string +} + +// FloatingIPClient is the only surface the orchestrator uses to manage +// Floating IPs. It intentionally does not expose allocation/deallocation — +// this tool only associates/disassociates pre-existing floating IPs with +// validator ports; returning an address to the customer-facing pool is an +// external business process out of scope here. +type FloatingIPClient interface { + // GetFloatingIPByAddress looks up the Neutron floating-ip resource for + // a given public address. Returns ErrNotFound if no such floating IP + // is registered in the service project. + GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error) + + // AssociateFloatingIP attaches the floating IP to the given Neutron + // port (the validator's primary NIC port). + AssociateFloatingIP(ctx context.Context, fipID, portID string) error + + // DisassociateFloatingIP detaches the floating IP from whatever port + // it's currently attached to, if any. Disassociating an already-free + // floating IP is a no-op, not an error. + DisassociateFloatingIP(ctx context.Context, fipID string) error +} + +type notFoundError struct{ address string } + +func (e *notFoundError) Error() string { return "floating ip not found: " + e.address } + +// ErrNotFound wraps address into an error satisfying IsNotFound. +func ErrNotFound(address string) error { return ¬FoundError{address} } + +// IsNotFound reports whether err indicates the address has no matching +// Neutron floating-ip resource. +func IsNotFound(err error) bool { + _, ok := err.(*notFoundError) + return ok +} diff --git a/internal/openstack/mock.go b/internal/openstack/mock.go new file mode 100644 index 0000000..ea458d1 --- /dev/null +++ b/internal/openstack/mock.go @@ -0,0 +1,77 @@ +package openstack + +import ( + "context" + "fmt" + "sync" +) + +// MockClient is an in-memory FloatingIPClient used by unit tests and the +// offline end-to-end harness. Seed it with the pool of floating IPs the +// scenario expects to exist before use. +type MockClient struct { + mu sync.Mutex + fips map[string]*FloatingIP // keyed by ID + byIP map[string]string // address -> ID + + // AssociateFailures/DisassociateFailures let tests force a failure for + // a specific floating-IP ID on its next call, to exercise retry paths. + AssociateFailures map[string]error + DisassociateFailures map[string]error +} + +func NewMockClient() *MockClient { + return &MockClient{ + fips: make(map[string]*FloatingIP), + byIP: make(map[string]string), + } +} + +// Seed registers a floating IP as if pre-allocated in the service project. +func (m *MockClient) Seed(id, address, projectID string) { + m.mu.Lock() + defer m.mu.Unlock() + m.fips[id] = &FloatingIP{ID: id, Address: address, ProjectID: projectID} + m.byIP[address] = id +} + +func (m *MockClient) GetFloatingIPByAddress(ctx context.Context, address string) (*FloatingIP, error) { + m.mu.Lock() + defer m.mu.Unlock() + id, ok := m.byIP[address] + if !ok { + return nil, ErrNotFound(address) + } + f := *m.fips[id] + return &f, nil +} + +func (m *MockClient) AssociateFloatingIP(ctx context.Context, fipID, portID string) error { + m.mu.Lock() + defer m.mu.Unlock() + if err := m.AssociateFailures[fipID]; err != nil { + delete(m.AssociateFailures, fipID) + return err + } + f, ok := m.fips[fipID] + if !ok { + return fmt.Errorf("mock openstack: unknown floating ip %q", fipID) + } + f.PortID = portID + return nil +} + +func (m *MockClient) DisassociateFloatingIP(ctx context.Context, fipID string) error { + m.mu.Lock() + defer m.mu.Unlock() + if err := m.DisassociateFailures[fipID]; err != nil { + delete(m.DisassociateFailures, fipID) + return err + } + f, ok := m.fips[fipID] + if !ok { + return fmt.Errorf("mock openstack: unknown floating ip %q", fipID) + } + f.PortID = "" + return nil +} diff --git a/internal/orchestrator/orchestrator.go b/internal/orchestrator/orchestrator.go new file mode 100644 index 0000000..8cf4087 --- /dev/null +++ b/internal/orchestrator/orchestrator.go @@ -0,0 +1,361 @@ +// Package orchestrator implements the Control API's core scheduling loop: +// claiming queued IPs onto idle validators, driving each IP through +// FIP-association -> self-check -> checking -> aggregation -> release, and +// reclaiming work from crashed/stuck validators via a lease sweep. It has +// no HTTP dependency — internal/httpapi calls into this package, and it can +// be exercised directly in tests against an in-memory OpenStack mock and a +// temp-file SQLite database. +package orchestrator + +import ( + "context" + "fmt" + "log/slog" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/openstack" +) + +// CheckConfig is the check-type/target configuration handed to a +// validator-agent once its IP has passed self-check. It mirrors +// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a +// flat list so the agent doesn't need its own copy of the target-group +// mapping. +type CheckConfig struct { + Type string `json:"type"` + Targets []string `json:"targets"` +} + +type Orchestrator struct { + DB *db.DB + OS openstack.FloatingIPClient + Cfg config.OrchestratorConfig + Agg config.AggregationConfig + Checks []CheckConfig + Sites []config.SiteConfig + Inbound config.InboundConfig + Log *slog.Logger +} + +func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator { + var checks []CheckConfig + for _, ct := range cfg.CheckTypes { + if !ct.Enabled { + continue + } + var targets []string + for _, group := range ct.Targets { + targets = append(targets, cfg.Targets[group]...) + } + checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets}) + } + return &Orchestrator{ + DB: d, + OS: osClient, + Cfg: cfg.Orchestrator, + Agg: cfg.Aggregation, + Checks: checks, + Sites: cfg.Sites, + Inbound: cfg.Inbound, + Log: log, + } +} + +func (o *Orchestrator) leaseTTL() time.Duration { + return time.Duration(o.Cfg.LeaseTTLSeconds) * time.Second +} + +// Tick runs one pass of the scheduling loop: claim+associate for idle +// validators, sweep the checking window for ready-to-aggregate IPs, and +// reclaim expired leases. Intended to be called on a fixed interval +// (Cfg.PollIntervalSeconds) by the caller (cmd/control-api/main.go). +func (o *Orchestrator) Tick(ctx context.Context) { + if err := o.assignIdleValidators(ctx); err != nil { + o.Log.Error("assign idle validators", "err", err) + } + if err := o.sweepCheckingWindow(ctx); err != nil { + o.Log.Error("sweep checking window", "err", err) + } + if err := o.sweepExpiredLeases(ctx); err != nil { + o.Log.Error("sweep expired leases", "err", err) + } +} + +// assignIdleValidators claims the next queued IP for every currently idle +// validator and kicks off FIP association for each newly claimed IP. +func (o *Orchestrator) assignIdleValidators(ctx context.Context) error { + idle, err := o.DB.ListIdleValidators(ctx) + if err != nil { + return fmt.Errorf("list idle validators: %w", err) + } + for _, v := range idle { + item, err := o.DB.ClaimNextQueued(ctx, v.ValidatorID, o.leaseTTL()) + if err != nil { + o.Log.Error("claim next queued", "validator", v.ValidatorID, "err", err) + continue + } + if item == nil { + continue // no work available for this validator right now + } + o.Log.Info("claimed ip", "validator", v.ValidatorID, "ip", item.IPAddress, "ip_id", item.ID) + if err := o.associateFIP(ctx, v.ValidatorID, v.OSPortID, item); err != nil { + o.Log.Error("associate fip", "validator", v.ValidatorID, "ip", item.IPAddress, "err", err) + } + } + return nil +} + +func (o *Orchestrator) associateFIP(ctx context.Context, validatorID, osPortID string, item *db.IPQueueItem) error { + fip, err := o.OS.GetFloatingIPByAddress(ctx, item.IPAddress) + if err != nil { + o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("lookup floating ip: %v", err)) + return err + } + if err := o.OS.AssociateFloatingIP(ctx, fip.ID, osPortID); err != nil { + o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("associate floating ip: %v", err)) + return err + } + if err := o.DB.SetFIPAssociated(ctx, item.ID, fip.ID, o.leaseTTL()); err != nil { + return fmt.Errorf("set fip associated: %w", err) + } + o.event(ctx, "control-api", "", &item.ID, "fip_associated", fmt.Sprintf(`{"fip_id":%q,"validator_id":%q}`, fip.ID, validatorID)) + return nil +} + +func (o *Orchestrator) requeueOrFail(ctx context.Context, ipID int64, validatorID, reason string) { + if err := o.DB.RequeueOrFail(ctx, ipID, validatorID, o.Cfg.MaxRetries); err != nil { + o.Log.Error("requeue or fail", "ip_id", ipID, "err", err) + return + } + o.event(ctx, "control-api", "", &ipID, "retry_or_fail", fmt.Sprintf(`{"reason":%q}`, reason)) +} + +// SelfCheckResult is called by the httpapi layer when a validator-agent +// reports its post-association self-check outcome. +func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string, ipID int64, success bool, detail string) error { + o.event(ctx, "validator-agent", validatorID, &ipID, "self_check_result", + fmt.Sprintf(`{"success":%t,"detail":%q}`, success, detail)) + + if !success { + item, err := o.DB.GetIP(ctx, ipID) + if err != nil { + return err + } + if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries { + o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail) + return nil + } + // Retry association without fully requeuing: re-drive the same + // claim by cycling back through requeue/claim keeps the logic in + // one place at the cost of the IP briefly returning to `queued`. + o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail) + return nil + } + + return o.DB.SetChecking(ctx, ipID, o.leaseTTL()) +} + +// AssignmentForValidator returns the check config for a validator's current +// IP if it's ready to be worked on (awaiting_self_check or checking), +// or nil if the validator has nothing to do right now. +func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) { + v, err := o.DB.GetValidator(ctx, validatorID) + if err != nil { + return nil, nil, err + } + if v.CurrentIPID == nil { + return nil, nil, nil + } + item, err := o.DB.GetIP(ctx, *v.CurrentIPID) + if err != nil { + return nil, nil, err + } + if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking { + return nil, nil, nil + } + return item, o.Checks, nil +} + +// SiteIndexForID resolves a configured site_id to its 1/2/3 index. +func (o *Orchestrator) SiteIndexForID(siteID string) int { + for _, s := range o.Sites { + if s.SiteID == siteID { + return s.Index + } + } + return 0 +} + +// RecordCheck upserts a single check result and, if it represents a +// completion signal (egress or a given site's full port+icmp sweep), +// updates the corresponding *_complete flag. +func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error { + return o.DB.UpsertCheck(ctx, c) +} + +func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error { + return o.DB.SetEgressComplete(ctx, ipID) +} + +func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteIndex int) error { + return o.DB.SetSiteComplete(ctx, ipID, siteIndex) +} + +// sweepCheckingWindow moves IPs that have either finished reporting from +// every source, or hit the checking-window deadline, into aggregation. +func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error { + deadline := db.Now().Add(-time.Duration(o.Cfg.CheckingWindowSeconds) * time.Second) + ready, err := o.DB.ListReadyToAggregate(ctx, deadline) + if err != nil { + return fmt.Errorf("list ready to aggregate: %w", err) + } + for _, item := range ready { + if err := o.aggregateAndRelease(ctx, item); err != nil { + o.Log.Error("aggregate and release", "ip_id", item.ID, "err", err) + } + } + return nil +} + +func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueItem) error { + if err := o.DB.SetAggregating(ctx, item.ID); err != nil { + return err + } + checks, err := o.DB.ListChecksForAttempt(ctx, item.ID, item.AttemptNumber) + if err != nil { + return err + } + + expected := o.expectedCheckCount() + passCount := 0 + for _, c := range checks { + if c.Success { + passCount++ + } + } + missing := expected - len(checks) + if missing < 0 { + missing = 0 + } + failCount := (len(checks) - passCount) + missing + + var result string + switch { + case passCount > 0 && failCount == 0: + result = db.ResultPass + case passCount == 0: + result = db.ResultFail + default: + result = db.ResultPartial + } + if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass { + result = db.ResultPartial + } + + if err := o.DB.FinishIP(ctx, item.ID, result); err != nil { + return err + } + o.event(ctx, "control-api", "", &item.ID, "aggregated", + fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing)) + + if item.FIPID != "" { + if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil { + o.Log.Error("disassociate fip", "ip_id", item.ID, "fip_id", item.FIPID, "err", err) + // Fall through and still free the validator/DB state — the + // lease sweep or an operator can reconcile a stuck Neutron + // association separately; we must not leave the validator + // wedged as "checking" forever over a cloud API hiccup. + } + } + if item.OwnerValidatorID != nil { + if err := o.DB.ReleaseFIP(ctx, item.ID, *item.OwnerValidatorID); err != nil { + return err + } + } + return nil +} + +// expectedCheckCount is the number of check rows a fully-reported IP should +// have: one per (egress check-type x target) plus one per (site x inbound +// port/icmp probe). +func (o *Orchestrator) expectedCheckCount() int { + egress := 0 + for _, c := range o.Checks { + egress += len(c.Targets) + } + inboundPerSite := len(o.Inbound.Ports) + if o.Inbound.ICMP { + inboundPerSite++ + } + return egress + inboundPerSite*len(o.Sites) +} + +// sweepExpiredLeases reclaims non-terminal IPs whose lease has passed — +// this is both the "stuck/crashed validator" reclaim path and, since all +// state lives in SQLite, the control-api crash-recovery path: a freshly +// restarted process finds the same expired leases and reclaims them the +// same way, with no separate recovery code required. +func (o *Orchestrator) sweepExpiredLeases(ctx context.Context) error { + expired, err := o.DB.ListExpiredLeases(ctx, db.Now()) + if err != nil { + return fmt.Errorf("list expired leases: %w", err) + } + for _, item := range expired { + validatorID := "" + if item.OwnerValidatorID != nil { + validatorID = *item.OwnerValidatorID + } + o.Log.Info("lease expired, reclaiming", "ip_id", item.ID, "ip", item.IPAddress, "validator", validatorID) + if item.FIPID != "" { + if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil { + o.Log.Error("disassociate fip on lease reclaim", "ip_id", item.ID, "err", err) + } + } + o.event(ctx, "control-api", "", &item.ID, "lease_expired", fmt.Sprintf(`{"validator_id":%q}`, validatorID)) + o.requeueOrFail(ctx, item.ID, validatorID, "lease expired") + } + return nil +} + +// sweepStaleHeartbeats marks validators unreachable if they haven't +// heartbeated within HeartbeatTimeoutSeconds. It does not itself reclaim +// their in-flight IP — that happens independently via lease expiry, so a +// validator that stops heartbeating but whose lease hasn't yet expired +// still finishes its current check window if it recovers in time. +func (o *Orchestrator) SweepStaleHeartbeats(ctx context.Context) error { + cutoff := db.Now().Add(-time.Duration(o.Cfg.HeartbeatTimeoutSeconds) * time.Second) + stale, err := o.DB.ListStaleHeartbeats(ctx, cutoff) + if err != nil { + return err + } + for _, v := range stale { + if err := o.DB.MarkValidatorUnreachable(ctx, v.ValidatorID); err != nil { + o.Log.Error("mark validator unreachable", "validator", v.ValidatorID, "err", err) + continue + } + o.event(ctx, "control-api", "", nil, "validator_unreachable", fmt.Sprintf(`{"validator_id":%q}`, v.ValidatorID)) + } + return nil +} + +// RecordEvent is the exported entry point httpapi uses to log +// agent/prober-reported audit events (config_received, fip_changed, +// error, etc.) through the same path as internally generated events. +func (o *Orchestrator) RecordEvent(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) { + o.event(ctx, sourceType, sourceID, ipID, eventType, payload) +} + +func (o *Orchestrator) event(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) { + if err := o.DB.InsertEvent(ctx, db.Event{ + SourceType: sourceType, + SourceID: sourceID, + IPID: ipID, + EventType: eventType, + Payload: payload, + OccurredAt: db.Now(), + }); err != nil { + o.Log.Error("insert event", "type", eventType, "err", err) + } +} diff --git a/internal/orchestrator/orchestrator_test.go b/internal/orchestrator/orchestrator_test.go new file mode 100644 index 0000000..fa50f84 --- /dev/null +++ b/internal/orchestrator/orchestrator_test.go @@ -0,0 +1,277 @@ +package orchestrator + +import ( + "context" + "log/slog" + "os" + "path/filepath" + "testing" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/openstack" +) + +func newTestOrchestrator(t *testing.T, leaseTTLSeconds int) (*Orchestrator, *db.DB, *openstack.MockClient) { + t.Helper() + ctx := context.Background() + dbPath := filepath.Join(t.TempDir(), "test.db") + d, err := db.Open(ctx, dbPath) + if err != nil { + t.Fatalf("open db: %v", err) + } + t.Cleanup(func() { d.Close() }) + + mock := openstack.NewMockClient() + + cfg := &config.ControlAPI{ + Orchestrator: config.OrchestratorConfig{ + PollIntervalSeconds: 1, + SelfCheckTimeoutSeconds: 10, + MaxSelfCheckRetries: 3, + CheckingWindowSeconds: 120, + MaxRetries: 3, + LeaseTTLSeconds: leaseTTLSeconds, + HeartbeatTimeoutSeconds: 30, + }, + Aggregation: config.AggregationConfig{MissingCountsAsFail: true}, + Sites: []config.SiteConfig{ + {SiteID: "site-1", Index: 1}, + {SiteID: "site-2", Index: 2}, + {SiteID: "site-3", Index: 3}, + }, + CheckTypes: []config.CheckTypeConfig{ + {Name: "https", Enabled: true, Targets: []string{"web"}}, + {Name: "ssh", Enabled: false, Targets: []string{"web"}}, + }, + Targets: map[string][]string{ + "web": {"https://example.test"}, + }, + Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true}, + } + + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) + return New(d, mock, cfg, log), d, mock +} + +func TestHappyPath(t *testing.T) { + ctx := context.Background() + o, d, mock := newTestOrchestrator(t, 180) + + mock.Seed("fip-1", "1.2.3.4", "svc-project") + if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil { + t.Fatalf("register validator: %v", err) + } + if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil { + t.Fatalf("seed queue: %v", err) + } + + // 1. claim + associate + o.Tick(ctx) + + ip, err := d.GetIPByAddress(ctx, "1.2.3.4") + if err != nil { + t.Fatalf("get ip: %v", err) + } + if ip.State != db.IPAwaitingSelfCheck { + t.Fatalf("expected awaiting_self_check, got %s", ip.State) + } + if ip.FIPID != "fip-1" { + t.Fatalf("expected fip-1 associated, got %q", ip.FIPID) + } + if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" { + t.Fatalf("expected fip associated to port-1, got %q", fip.PortID) + } + + v, err := d.GetValidator(ctx, "validator-1") + if err != nil { + t.Fatalf("get validator: %v", err) + } + if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != ip.ID { + t.Fatalf("expected validator assigned to ip %d, got state=%s current_ip=%v", ip.ID, v.State, v.CurrentIPID) + } + + // 2. self-check success + if err := o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "egress matched"); err != nil { + t.Fatalf("self check result: %v", err) + } + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != db.IPChecking { + t.Fatalf("expected checking, got %s", ip.State) + } + + // 3. egress result + completion + if err := o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https", + Target: "https://example.test", Success: true, CheckedAt: db.Now(), + }); err != nil { + t.Fatalf("record egress check: %v", err) + } + if err := o.MarkEgressComplete(ctx, ip.ID); err != nil { + t.Fatalf("mark egress complete: %v", err) + } + + // 4. inbound results from all 3 sites + for site := 1; site <= 3; site++ { + for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} { + if err := o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress, + Success: true, CheckedAt: db.Now(), + }); err != nil { + t.Fatalf("record inbound check site %d: %v", site, err) + } + } + if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil { + t.Fatalf("mark site %d complete: %v", site, err) + } + } + + // 5. sweep should now aggregate + release + o.Tick(ctx) + + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != db.IPDone { + t.Fatalf("expected done, got %s", ip.State) + } + if ip.OverallResult != db.ResultPass { + t.Fatalf("expected pass, got %s", ip.OverallResult) + } + if ip.FIPReleasedAt == nil { + t.Fatalf("expected fip_released_at to be set") + } + + v, _ = d.GetValidator(ctx, "validator-1") + if v.State != db.ValidatorIdle || v.CurrentIPID != nil { + t.Fatalf("expected validator idle with no current ip, got state=%s current_ip=%v", v.State, v.CurrentIPID) + } + + if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" { + t.Fatalf("expected fip disassociated, still on port %q", fip.PortID) + } +} + +func TestPartialResult(t *testing.T) { + ctx := context.Background() + o, d, mock := newTestOrchestrator(t, 180) + mock.Seed("fip-1", "1.2.3.4", "svc-project") + _ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1") + _ = d.SeedQueue(ctx, []string{"1.2.3.4"}) + + o.Tick(ctx) + ip, _ := d.GetIPByAddress(ctx, "1.2.3.4") + _ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok") + ip, _ = d.GetIP(ctx, ip.ID) + + // Egress passes... + _ = o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https", + Target: "https://example.test", Success: true, CheckedAt: db.Now(), + }) + _ = o.MarkEgressComplete(ctx, ip.ID) + // ...but only site-1 reports, and one of its checks fails. + _ = o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + Source: db.InboundSource(1), CheckType: "tcp-22", Target: ip.IPAddress, Success: false, CheckedAt: db.Now(), + }) + _ = o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + Source: db.InboundSource(1), CheckType: "tcp-80", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(), + }) + _ = o.RecordCheck(ctx, db.Check{ + IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber, + Source: db.InboundSource(1), CheckType: "icmp", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(), + }) + _ = o.MarkSiteComplete(ctx, ip.ID, 1) + + // Force the checking window to have elapsed so aggregation proceeds + // even though site-2/site-3 never reported. + o.Cfg.CheckingWindowSeconds = 0 + time.Sleep(5 * time.Millisecond) + o.Tick(ctx) + + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != db.IPDone { + t.Fatalf("expected done, got %s", ip.State) + } + if ip.OverallResult != db.ResultPartial { + t.Fatalf("expected partial, got %s", ip.OverallResult) + } + if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" { + t.Fatalf("expected fip disassociated even on partial result") + } +} + +func TestLeaseReclaim(t *testing.T) { + ctx := context.Background() + // A 1s lease (rather than 0) avoids a race within the very first Tick: + // with a 0s TTL the item's lease can already look expired by the time + // the same Tick's lease-sweep phase runs, depending on how much + // wall-clock time the claim+associate phase happened to take. + o, d, mock := newTestOrchestrator(t, 1) + mock.Seed("fip-1", "1.2.3.4", "svc-project") + _ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1") + _ = d.SeedQueue(ctx, []string{"1.2.3.4"}) + + o.Tick(ctx) // claims + associates; validator never self-checks + ip, _ := d.GetIPByAddress(ctx, "1.2.3.4") + if ip.State != db.IPAwaitingSelfCheck { + t.Fatalf("expected awaiting_self_check, got %s", ip.State) + } + + time.Sleep(1100 * time.Millisecond) // let the 1s lease expire + o.Tick(ctx) // should reclaim via lease sweep + + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != db.IPQueued { + t.Fatalf("expected requeued after lease reclaim, got %s (retry_count=%d)", ip.State, ip.RetryCount) + } + if ip.RetryCount != 1 { + t.Fatalf("expected retry_count=1, got %d", ip.RetryCount) + } + + v, _ := d.GetValidator(ctx, "validator-1") + if v.State != db.ValidatorIdle || v.CurrentIPID != nil { + t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID) + } + + if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" { + t.Fatalf("expected fip disassociated on reclaim, still on port %q", fip.PortID) + } + + // A subsequent tick should re-claim and re-associate the same IP for + // the now-idle validator, proving the queue keeps making progress. + // Give this attempt a real lease so it isn't immediately re-expired by + // the same tick's lease sweep (a 0s TTL, as above, expires instantly). + o.Cfg.LeaseTTLSeconds = 180 + o.Tick(ctx) + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != db.IPAwaitingSelfCheck { + t.Fatalf("expected re-claimed ip to be awaiting_self_check again, got %s", ip.State) + } + if ip.AttemptNumber != 2 { + t.Fatalf("expected attempt_number=2 after reclaim+reassign, got %d", ip.AttemptNumber) + } +} + +func TestMaxRetriesExhausted(t *testing.T) { + ctx := context.Background() + o, d, mock := newTestOrchestrator(t, 0) + o.Cfg.MaxRetries = 1 + mock.Seed("fip-1", "1.2.3.4", "svc-project") + _ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1") + _ = d.SeedQueue(ctx, []string{"1.2.3.4"}) + + for i := 0; i < 3; i++ { + o.Tick(ctx) + time.Sleep(5 * time.Millisecond) + } + + ip, _ := d.GetIPByAddress(ctx, "1.2.3.4") + if ip.State != db.IPFailed { + t.Fatalf("expected failed after exhausting retries, got %s (retry_count=%d)", ip.State, ip.RetryCount) + } +} diff --git a/internal/probercore/probercore.go b/internal/probercore/probercore.go new file mode 100644 index 0000000..dbbf159 --- /dev/null +++ b/internal/probercore/probercore.go @@ -0,0 +1,138 @@ +// Package probercore implements a prober's poll loop: register once with +// its site identity, then each cycle fetch the set of IPs currently under +// test and run inbound reachability checks (TCP connect on each configured +// port, ICMP echo) directly against each one — this is the real, +// unmediated network test; only the control channel goes through the +// Control API. +package probercore + +import ( + "context" + "fmt" + "log/slog" + "os" + "time" + + "cloudipvalidator/internal/apiclient" + "cloudipvalidator/internal/checkrunner" + "cloudipvalidator/internal/config" +) + +type Prober struct { + cfg *config.Prober + client *apiclient.Client + log *slog.Logger +} + +func New(cfg *config.Prober, log *slog.Logger) *Prober { + timeout := time.Duration(cfg.Checks.TCPTimeoutSeconds) * time.Second + if timeout <= 0 { + timeout = 10 * time.Second + } + return &Prober{ + cfg: cfg, + client: apiclient.New(cfg.ControlAPIURL, timeout+5*time.Second), + log: log, + } +} + +func (p *Prober) Run(ctx context.Context) error { + if err := p.register(ctx); err != nil { + return fmt.Errorf("register: %w", err) + } + + interval := time.Duration(p.cfg.PollIntervalSeconds) * time.Second + ticker := time.NewTicker(interval) + defer ticker.Stop() + + for { + p.pollOnce(ctx) + select { + case <-ctx.Done(): + return ctx.Err() + case <-ticker.C: + } + } +} + +type registerReq struct { + SiteID string `json:"site_id"` + Hostname string `json:"hostname"` +} + +func (p *Prober) register(ctx context.Context) error { + hostname, _ := os.Hostname() + _, err := p.client.Do(ctx, "POST", "/api/v1/probers/register", registerReq{SiteID: p.cfg.SiteID, Hostname: hostname}, nil) + if err != nil { + return err + } + p.log.Info("registered", "site_id", p.cfg.SiteID) + return nil +} + +type assignment struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + Ports []int `json:"ports"` + ICMP bool `json:"icmp"` +} + +type resultDTO struct { + IPID int64 `json:"ip_id"` + IPAddress string `json:"ip_address"` + CheckType string `json:"check_type"` + Success bool `json:"success"` + LatencyMS int64 `json:"latency_ms"` + Detail string `json:"detail,omitempty"` + CheckedAt string `json:"checked_at"` + Complete bool `json:"complete"` +} + +func (p *Prober) pollOnce(ctx context.Context) { + var assignments []assignment + ok, err := p.client.Do(ctx, "GET", "/api/v1/probers/"+p.cfg.SiteID+"/assignments", nil, &assignments) + if err != nil { + p.log.Error("get assignments", "err", err) + return + } + if !ok || len(assignments) == 0 { + return + } + + for _, a := range assignments { + p.probeOne(ctx, a) + } +} + +func (p *Prober) probeOne(ctx context.Context, a assignment) { + tcpTimeout := time.Duration(p.cfg.Checks.TCPTimeoutSeconds) * time.Second + icmpTimeout := time.Duration(p.cfg.Checks.ICMPTimeoutSeconds) * time.Second + + var results []resultDTO + for _, port := range a.Ports { + res := checkrunner.TCPConnect(a.IPAddress, port, tcpTimeout)(ctx) + results = append(results, resultDTO{ + IPID: a.IPID, IPAddress: a.IPAddress, CheckType: res.CheckType, + Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, + CheckedAt: res.CheckedAt.Format(time.RFC3339Nano), + }) + } + if a.ICMP { + res := checkrunner.ICMPEcho(a.IPAddress, p.cfg.Checks.ICMPCount, icmpTimeout)(ctx) + results = append(results, resultDTO{ + IPID: a.IPID, IPAddress: a.IPAddress, CheckType: res.CheckType, + Success: res.Success, LatencyMS: res.LatencyMS, Detail: res.Detail, + CheckedAt: res.CheckedAt.Format(time.RFC3339Nano), + }) + } + if len(results) > 0 { + results[len(results)-1].Complete = true + } + + body := struct { + Results []resultDTO `json:"results"` + }{results} + if _, err := p.client.Do(ctx, "POST", "/api/v1/probers/"+p.cfg.SiteID+"/results", body, nil); err != nil { + p.log.Error("post results", "ip", a.IPAddress, "err", err) + } +} diff --git a/scripts/httpstub/main.go b/scripts/httpstub/main.go new file mode 100644 index 0000000..9c51b7b --- /dev/null +++ b/scripts/httpstub/main.go @@ -0,0 +1,23 @@ +// Command httpstub is a trivial "always 200 OK" HTTP server used only by +// scripts/run-local-e2e.sh, standing in for the real internet targets +// (hub.docker.com, github.com, packages.ubuntu.com) so the offline +// end-to-end harness needs no real internet access. +package main + +import ( + "flag" + "log" + "net/http" +) + +func main() { + addr := flag.String("addr", ":9090", "address to listen on") + flag.Parse() + + http.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusOK) + _, _ = w.Write([]byte("stub ok\n")) + }) + log.Printf("httpstub listening on %s", *addr) + log.Fatal(http.ListenAndServe(*addr, nil)) +} diff --git a/scripts/run-local-e2e.sh b/scripts/run-local-e2e.sh new file mode 100755 index 0000000..2479abb --- /dev/null +++ b/scripts/run-local-e2e.sh @@ -0,0 +1,209 @@ +#!/usr/bin/env bash +# Offline end-to-end smoke test for the Cloud IP Validator (see +# docs/LOCAL_E2E.md). Runs control-api + 1 validator-agent + 3 probers as +# local processes against a mock OpenStack backend, with loopback stand-ins +# for both the "floating IP" under test and the outbound targets — no real +# cloud or internet access required. +# +# Usage: scripts/run-local-e2e.sh +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$REPO_ROOT" + +GO="${GO:-go}" +if ! command -v "$GO" >/dev/null 2>&1 && [ -x /usr/local/go/bin/go ]; then + GO=/usr/local/go/bin/go +fi + +WORK_DIR="$(mktemp -d /tmp/cloud-ip-validator-e2e.XXXXXX)" +BIN_DIR="$WORK_DIR/bin" +LOG_DIR="$WORK_DIR/logs" +mkdir -p "$BIN_DIR" "$LOG_DIR" + +PIDS=() +cleanup() { + echo "--- cleaning up (workdir: $WORK_DIR) ---" + for pid in "${PIDS[@]:-}"; do + kill "$pid" >/dev/null 2>&1 || true + done + wait >/dev/null 2>&1 || true +} +trap cleanup EXIT + +echo "--- building binaries ---" +"$GO" build -o "$BIN_DIR/control-api" ./cmd/control-api +"$GO" build -o "$BIN_DIR/validator-agent" ./cmd/validator-agent +"$GO" build -o "$BIN_DIR/prober" ./cmd/prober +"$GO" build -o "$BIN_DIR/httpstub" ./scripts/httpstub + +echo "--- starting stub outbound targets (stand-ins for hub.docker.com / github.com / packages.ubuntu.com) ---" +"$BIN_DIR/httpstub" -addr 127.0.0.1:29091 >"$LOG_DIR/stub-1.log" 2>&1 & PIDS+=($!) +"$BIN_DIR/httpstub" -addr 127.0.0.1:29092 >"$LOG_DIR/stub-2.log" 2>&1 & PIDS+=($!) +"$BIN_DIR/httpstub" -addr 127.0.0.1:29093 >"$LOG_DIR/stub-3.log" 2>&1 & PIDS+=($!) +sleep 0.5 + +cat >"$WORK_DIR/control-api.yaml" <"$WORK_DIR/validator-agent.yaml" <"$WORK_DIR/prober-$site.yaml" <"$LOG_DIR/control-api.log" 2>&1 & CONTROL_PID=$! +PIDS+=("$CONTROL_PID") + +echo -n "waiting for control-api /healthz " +for i in $(seq 1 30); do + if curl -fs "http://127.0.0.1:28080/healthz" >/dev/null 2>&1; then + echo "ok" + break + fi + echo -n "." + sleep 0.5 + if [ "$i" -eq 30 ]; then + echo "FAILED" + cat "$LOG_DIR/control-api.log" + exit 1 + fi +done + +start_validator_agent() { + "$BIN_DIR/validator-agent" -config "$WORK_DIR/validator-agent.yaml" -stub-ports "22022,28081,28443,28888" \ + >>"$LOG_DIR/validator-agent.log" 2>&1 & VALIDATOR_PID=$! + PIDS+=("$VALIDATOR_PID") + echo "validator-agent started (pid $VALIDATOR_PID)" +} + +status() { curl -fs "http://127.0.0.1:28080/api/v1/admin/status"; } +ips() { curl -fs "http://127.0.0.1:28080/api/v1/admin/ips"; } +ip_state() { ips | python3 -c "import json,sys; d=json.load(sys.stdin); print(d[0]['State'] if d else 'none')" 2>/dev/null || echo "?"; } + +echo "--- starting validator-agent ---" +start_validator_agent + +# Poll tightly (every 0.1s) for the IP to leave 'queued' — i.e. control-api +# has claimed it and associated the mock FIP — then kill the agent right +# there, before it has had a chance to self-check or run any checks. This +# proves the lease sweep reclaims genuinely in-flight, not-yet-reported +# work, rather than just racing against how fast the local checks happen +# to run. +echo "--- waiting for control-api to claim+associate the ip, to kill the agent mid-flight ---" +CAUGHT_MIDFLIGHT=0 +for i in $(seq 1 50); do + st="$(ip_state)" + if [ "$st" != "queued" ] && [ "$st" != "none" ] && [ "$st" != "?" ]; then + CAUGHT_MIDFLIGHT=1 + break + fi + sleep 0.1 +done + +if [ "$CAUGHT_MIDFLIGHT" -eq 1 ]; then + echo ">>> caught ip in state '$st'; killing validator-agent mid-run to demonstrate lease reclaim <<<" + kill "$VALIDATOR_PID" >/dev/null 2>&1 || true + wait "$VALIDATOR_PID" 2>/dev/null || true + sleep 9 # let the 8s lease_ttl_seconds expire and get reclaimed + echo ">>> restarting validator-agent <<<" + start_validator_agent +else + echo ">>> never observed the ip leave 'queued' in time; skipping the reclaim demonstration <<<" +fi + +echo "--- starting 3 probers ---" +for site in site-1 site-2 site-3; do + "$BIN_DIR/prober" -config "$WORK_DIR/prober-$site.yaml" >"$LOG_DIR/prober-$site.log" 2>&1 & PIDS+=($!) +done + +echo "--- waiting for the queue to drain (up to 60s) ---" +for i in $(seq 1 120); do + sleep 0.5 + remaining=$(status | python3 -c "import json,sys; d=json.load(sys.stdin); s=d['ips_by_state']; print(sum(v for k,v in s.items() if k not in ('done','failed')))" 2>/dev/null || echo "?") + if [ "$remaining" = "0" ]; then + echo "queue drained after ~$((i / 2))s" + break + fi +done + +echo "--- final status ---" +status | python3 -m json.tool +echo "--- final ip results ---" +ips | python3 -m json.tool + +echo "--- logs are in $WORK_DIR (kept for inspection; workdir NOT auto-deleted) ---" +trap - EXIT +cleanup