diff --git a/README.md b/README.md index 201dfc0..0e25c6a 100644 --- a/README.md +++ b/README.md @@ -9,11 +9,12 @@ SSH до внешних целей) и входящая доступность каждому адресу — `pass`/`partial`/`fail`, с полной историей проверок в базе данных. -Три компонента: `control-api` (управляющий сервис, единственный со +Четыре компонента: `control-api` (управляющий сервис, единственный со состоянием), `validator-agent` (работает на каждой ВМ-валидаторе, без -состояния) и `prober` (работает на каждой из трёх внешних площадок, без -состояния). Все три общаются между собой только через HTTP API -control-api. +состояния), `prober` (работает на каждой из трёх внешних площадок, без +состояния) и `admin-dashboard` (браузерная веб-панель администратора, без +состояния, опциональна). Все четыре общаются между собой только через +HTTP API control-api. ## Документация @@ -22,6 +23,7 @@ control-api. | [docs/SETUP.md](docs/SETUP.md) | Развёртывание из готовых бинарников (`bin/`) или сборка из исходников, конфигурация, первый запуск стенда — с нуля | | [docs/USAGE.md](docs/USAGE.md) | Повседневная работа: постановка адресов в очередь, наблюдение за статусом, разбор результатов | | [docs/API.md](docs/API.md) | Спецификация HTTP API control-api и примеры запросов (curl) | +| [docs/DASHBOARD.md](docs/DASHBOARD.md) | Браузерная админ-панель (`admin-dashboard`) — то же самое API, но графически | | [docs/DIAGRAMS.md](docs/DIAGRAMS.md) | Диаграммы потоков данных: control plane, поток проверки до целевого сервера, поток телеметрии | | [docs/LOCAL_E2E.md](docs/LOCAL_E2E.md) | Полностью офлайн-прогон всей системы одним скриптом — без реального облака и интернета | diff --git a/bin/SHA256SUMS b/bin/SHA256SUMS index 2dd96bc..a3d346c 100644 --- a/bin/SHA256SUMS +++ b/bin/SHA256SUMS @@ -1,3 +1,4 @@ -ea275ac1b75e2b1aa16f8c2424be04983fda12b23bc99866e2b56e2dc25dda31 control-api -aa82eca68cf5abb6e91091418c728fa4407d306126d0915799cacf7c6c2a1dac validator-agent -645c69e72d5fcd2ff385eaa558c0e5849f5ee28ffbbc46090dd3be6244b7e36c prober +992ec763b2dc26f87d5391e072904e5897863a0c4c5b27c3c4980aa03f037c5e control-api +48c9b99fa88be751d9badfba8b7d80f894e326a743a90e85478f1d2251ce4ab6 validator-agent +43fb660b78179b204b57388241a508345881aa16f3531d77b8fd75af60e7f2d8 prober +fcf80759c37b8b7ea572d313e6689e7c7d00c1369f0889ed24b9eea16c0a17b0 admin-dashboard diff --git a/bin/admin-dashboard b/bin/admin-dashboard new file mode 100755 index 0000000..9575626 Binary files /dev/null and b/bin/admin-dashboard differ diff --git a/bin/control-api b/bin/control-api index d37fcf3..312b065 100755 Binary files a/bin/control-api and b/bin/control-api differ diff --git a/bin/prober b/bin/prober index f906b4a..5a6d5b3 100755 Binary files a/bin/prober and b/bin/prober differ diff --git a/bin/validator-agent b/bin/validator-agent index 864d8cb..5a1bb75 100755 Binary files a/bin/validator-agent and b/bin/validator-agent differ diff --git a/cmd/admin-dashboard/main.go b/cmd/admin-dashboard/main.go new file mode 100644 index 0000000..012f14d --- /dev/null +++ b/cmd/admin-dashboard/main.go @@ -0,0 +1,71 @@ +// Command admin-dashboard is a stateless, server-rendered web UI over +// control-api's /api/v1/admin/* HTTP API — see docs/DASHBOARD.md. It never +// connects to the database and holds no state of its own. +package main + +import ( + "context" + "flag" + "fmt" + "log/slog" + "net/http" + "os" + "os/signal" + "syscall" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/dashboard" +) + +func main() { + configPath := flag.String("config", "configs/admin-dashboard.yaml", "path to admin-dashboard config file") + flag.Parse() + + log := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelInfo})) + + if err := run(*configPath, log); err != nil { + log.Error("fatal", "err", err) + os.Exit(1) + } +} + +func run(configPath string, log *slog.Logger) error { + cfg, err := config.LoadAdminDashboard(configPath) + if err != nil { + return fmt.Errorf("load config: %w", err) + } + + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + + srv, err := dashboard.New(dashboard.Config{ + ControlAPIBaseURL: cfg.ControlAPI.BaseURL, + ControlAPITimeout: time.Duration(cfg.ControlAPI.TimeoutSeconds) * time.Second, + LastCompletedCount: cfg.Overview.LastCompletedCount, + OverviewPollIntervalS: cfg.Overview.PollIntervalSeconds, + }, log) + if err != nil { + return fmt.Errorf("init dashboard: %w", err) + } + + httpServer := &http.Server{Addr: cfg.Server.ListenAddr, Handler: srv.Handler()} + + errCh := make(chan error, 1) + go func() { + log.Info("listening", "addr", cfg.Server.ListenAddr, "control_api", cfg.ControlAPI.BaseURL) + if err := httpServer.ListenAndServe(); err != nil && err != http.ErrServerClosed { + errCh <- err + } + }() + + select { + case <-ctx.Done(): + log.Info("shutting down") + shutdownCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + return httpServer.Shutdown(shutdownCtx) + case err := <-errCh: + return err + } +} diff --git a/cmd/control-api/main.go b/cmd/control-api/main.go index 2bb6405..a1d586e 100644 --- a/cmd/control-api/main.go +++ b/cmd/control-api/main.go @@ -49,13 +49,8 @@ func run(configPath string, log *slog.Logger) error { } defer database.Close() - for _, v := range cfg.Validators { - if err := database.RegisterValidator(ctx, v.ValidatorID, "", v.OSPortID, ""); err != nil { - return fmt.Errorf("seed validator %s: %w", v.ValidatorID, err) - } - } - if err := database.SeedQueue(ctx, cfg.IPAddresses); err != nil { - return fmt.Errorf("seed ip queue: %w", err) + if err := database.BootstrapFromConfig(ctx, cfg); err != nil { + return fmt.Errorf("bootstrap database: %w", err) } osClient, err := newOpenStackClient(ctx, cfg) diff --git a/configs/admin-dashboard.example.yaml b/configs/admin-dashboard.example.yaml new file mode 100644 index 0000000..e2699f6 --- /dev/null +++ b/configs/admin-dashboard.example.yaml @@ -0,0 +1,18 @@ +server: + listen_addr: ":8090" + +control_api: + base_url: "http://control-api.internal:8080" + timeout_seconds: 10 + +# Настройки сводки на странице "Обзор" — см. docs/DASHBOARD.md. Оба поля +# влияют только на то, как дашборд группирует уже существующие данные +# control-api (GET /admin/status, GET /admin/ips); никакого нового +# состояния control-api не заводит. +overview: + # Сколько последних завершённых (done/failed) адресов показывать в + # сводке "последняя завершённая проверка" и учитывать в разбивке + # pass/partial/fail/cancelled. + last_completed_count: 20 + # Как часто браузер опрашивает /overview/fragment для live-обновления. + poll_interval_seconds: 5 diff --git a/deploy/systemd/admin-dashboard.service b/deploy/systemd/admin-dashboard.service new file mode 100644 index 0000000..ff2a9a3 --- /dev/null +++ b/deploy/systemd/admin-dashboard.service @@ -0,0 +1,18 @@ +[Unit] +Description=Cloud IP Validator - Admin Dashboard +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=cloud-ip-validator +Group=cloud-ip-validator +ExecStart=/usr/local/bin/admin-dashboard -config /etc/cloud-ip-validator/admin-dashboard.yaml +Restart=on-failure +RestartSec=5 +NoNewPrivileges=true +ProtectSystem=strict +PrivateTmp=true + +[Install] +WantedBy=multi-user.target diff --git a/docs/API.md b/docs/API.md index d291369..471ab6d 100644 --- a/docs/API.md +++ b/docs/API.md @@ -6,7 +6,10 @@ JSON, базовый префикс прикладных методов — `/ap > **Важно.** На данный момент API не защищён аутентификацией/авторизацией > — эндпоинты доступны любому, кто может достучаться до порта control-api -> по сети. Для эксплуатации за пределами доверенного сегмента сети +> по сети. Это касается и методов из раздела +> [«Управление очередью и конфигурацией»](#управление-очередью-и-конфигурацией) +> ниже — они меняют, что и как проверяется, без подтверждения личности +> вызывающего. Для эксплуатации за пределами доверенного сегмента сети > обязательно ограничьте доступ на уровне сети/файрвола (см. > [SETUP.md](SETUP.md#сетевые-доступы)). Добавление bearer-токена — известное > направление доработки, в текущей версии не реализовано. @@ -14,12 +17,17 @@ JSON, базовый префикс прикладных методов — `/ap Базовый URL в примерах — `http://control-api.internal:8080`, замените на адрес вашего стенда (см. `server.listen_addr` в конфиге control-api). +> Для работы из браузера вместо `curl` есть `admin-dashboard` — веб-панель, +> дающая графический доступ ко всему административному API ниже, см. +> [DASHBOARD.md](DASHBOARD.md). + ## Содержание - [Общие соглашения](#общие-соглашения) - [Методы для validator-agent](#методы-для-validator-agent) - [Методы для prober](#методы-для-prober) - [Служебные и административные методы](#служебные-и-административные-методы) +- [Управление очередью и конфигурацией](#управление-очередью-и-конфигурацией) - [Модель состояний и связь методов с ней](#модель-состояний-и-связь-методов-с-ней) - [Сквозной пример работы (curl)](#сквозной-пример-работы-curl) @@ -37,9 +45,11 @@ JSON, базовый префикс прикладных методов — `/ap не удалось распарсить, сервер молча подставит текущее время сервера — не полагайтесь на это в продакшене, всегда передавайте валидную метку. - `validator_id` и `site_id` в пути запроса должны совпадать со - значениями, заданными в конфиге control-api (`validators[].validator_id`, - `sites[].site_id`) — иначе методы, требующие существующую сущность, - вернут `404`. + значениями, известными control-api — заданными в `control-api.yaml` + при первом запуске (пустая база) либо созданными позже через + `/api/v1/admin/config/*` (см. + [«Управление очередью и конфигурацией»](#управление-очередью-и-конфигурацией)) + — иначе методы, требующие существующую сущность, вернут `404`. ## Методы для validator-agent @@ -297,6 +307,152 @@ IP на данном проходе". До этого момента control-api `assigned`, `checking`, `unreachable`) и `CurrentIPID`, если валидатор сейчас занят. +## Управление очередью и конфигурацией + +Методы этого раздела — единственный способ менять состав очереди +(`ip_addresses`), список валидаторов, площадок (`sites`) и целей +проверки (`targets`/`check_types`) **без остановки процесса**: изменения +применяются немедленно и переживают последующий рестарт control-api. Все +тела запросов/ответов — `snake_case` (в отличие от `GET +/admin/status|ips|validators` выше, которые отдают сырые поля Go-структур +в PascalCase — эти два стиля сосуществуют осознанно, см. примечание к +`GET /api/v1/admin/ips/{ip}`). + +**Источник истины.** `control-api.yaml` используется только как +одноразовый bootstrap для пустой базы данных: секции `validators`, +`sites`, `targets`, `check_types` читаются из YAML один раз, при самом +первом старте на пустых таблицах. Как только в соответствующей таблице +появилась хотя бы одна строка (через bootstrap либо через методы ниже) — +YAML для этой секции больше не перечитывается ни при одном последующем +рестарте; правки нужно вносить через API. Список IP-адресов +(`ip_addresses` в YAML) — исключение, он остаётся отдельным, всегда +аддитивным путём постановки в очередь при каждом старте (см. +[SETUP.md](SETUP.md#развёртывание-control-api)); он не конфликтует с +`POST /api/v1/admin/ips` ниже. + +### `POST /api/v1/admin/ips` + +Единая точка для двух задач: добавить новые адреса в очередь **и** +принудительно перепроверить уже завершённые — один и тот же вызов, разница +только в текущем состоянии каждого конкретного адреса. Список +обрабатывается в one transaction, в порядке следования адресов: + +- адрес неизвестен control-api → добавляется в очередь как новый + (`queued`); +- адрес сейчас `done`/`failed` → принудительно перезапускается: сбрасывается + результат, `attempt_number` увеличивается, `retry_count` обнуляется, + адрес снова становится `queued`; +- адрес сейчас `queued` (ещё не взят в работу) → только переупорядочивается + под порядок текущего списка, повторно не добавляется; +- адрес сейчас активно проверяется (`assigning_fip` / `awaiting_self_check` + / `checking` / `aggregating`) → не трогается вообще — нельзя запустить + вторую параллельную проверку одного и того же адреса. + +Порядок обработки внутри одного вызова соответствует порядку адресов в +списке; повторная отправка того же списка позже даёт тот же относительный +порядок прогона. + +Запрос: +```json +{"addresses": ["203.0.113.10", "203.0.113.11"]} +``` + +Ответ (`200`): +```json +{ + "added": ["203.0.113.11"], + "requeued": ["203.0.113.10"], + "reordered": [], + "skipped_in_progress": [] +} +``` + +`400`, если `addresses` пуст. + +### `POST /api/v1/admin/ips/{ip}/cancel` + +Принудительно останавливает проверку конкретного адреса, не дожидаясь +`checking_window_seconds` — работает из любого нетерминального состояния, +включая `queued` (в этом случае это просто удаление ещё не начатой +проверки из очереди). Если Floating IP уже привязан — отвязывается +(best-effort, как и при обычном завершении проверки); владеющий валидатор +освобождается. Итог записывается как `overall_result: "cancelled"` +(состояние `failed`). + +Ответ: `{"ok": true}`. `404`, если адрес неизвестен. `409`, если адрес уже +в терминальном состоянии (`done`/`failed`/уже отменён) — отменять нечего. + +### Валидаторы: `/api/v1/admin/config/validators` + +| Метод | Путь | Тело | Успех | Ошибки | +|---|---|---|---|---| +| GET | `/api/v1/admin/config/validators` | — | `[{"validator_id","os_port_id","state"}]` | | +| POST | `/api/v1/admin/config/validators` | `{"validator_id","os_port_id"}` | `201` | `409`, если `validator_id` уже существует | +| PUT | `/api/v1/admin/config/validators/{id}` | `{"os_port_id"}` | `200` | `404` | +| DELETE | `/api/v1/admin/config/validators/{id}` | — | `200` | `404`; `409`, если валидатор сейчас владеет IP | + +### Площадки: `/api/v1/admin/config/sites` + +Слотов ровно три (`index` ∈ {1, 2, 3}) — это ограничение схемы БД +(`ip_queue.site{1,2,3}_complete`), а не искусственное. Пустой список слотов +— штатный сценарий, отключающий inbound-проверки целиком (см. +[USAGE.md](USAGE.md#управление-площадками-проберами)). + +| Метод | Путь | Тело | Успех | Ошибки | +|---|---|---|---|---| +| GET | `/api/v1/admin/config/sites` | — | `[{"index","site_id"}]` (до 3 строк) | | +| PUT | `/api/v1/admin/config/sites/{index}` | `{"site_id"}` | `200` | `400`, если `index` не 1..3; `409`, если `site_id` уже занят другим слотом | +| DELETE | `/api/v1/admin/config/sites/{index}` | — | `200` | `404` | + +### Группы целей: `/api/v1/admin/config/targets` + +Группа целей — именованный список URL/адресов (например `stub-targets: [ +"https://hub.docker.com", ...]`), на который затем ссылаются типы +проверок. + +| Метод | Путь | Тело | Успех | Ошибки | +|---|---|---|---|---| +| GET | `/api/v1/admin/config/targets` | — | `[{"name","targets"}]` | | +| PUT | `/api/v1/admin/config/targets/{group}` | `{"targets":[...]}` | `200` | `400`, если список пуст | +| DELETE | `/api/v1/admin/config/targets/{group}` | — | `200` | `404`; `409`, если группа используется каким-то `check_type` | + +### Типы проверок: `/api/v1/admin/config/check-types` + +Тип проверки (`https`, `icmp`, `ssh`, ...) ссылается на одну или несколько +групп целей по имени; именно развёрнутый список отсюда validator-agent +получает в `check_config` при `GET /api/v1/agents/{id}/assignment`. + +| Метод | Путь | Тело | Успех | Ошибки | +|---|---|---|---|---| +| GET | `/api/v1/admin/config/check-types` | — | `[{"name","enabled","targets"}]` (`targets` — имена групп) | | +| PUT | `/api/v1/admin/config/check-types/{name}` | `{"enabled","targets":["group",...]}` | `200` | `400`, если названа несуществующая группа | +| DELETE | `/api/v1/admin/config/check-types/{name}` | — | `200` | `404` | + +### Пример: конфигурация целиком через API, без единой строки в YAML + +```bash +BASE=http://127.0.0.1:8080 + +curl -s -X POST "$BASE/api/v1/admin/config/validators" \ + -d '{"validator_id":"validator_01","os_port_id":"port-abc123"}' + +curl -s -X PUT "$BASE/api/v1/admin/config/targets/web" \ + -d '{"targets":["https://hub.docker.com","https://github.com"]}' + +curl -s -X PUT "$BASE/api/v1/admin/config/check-types/https" \ + -d '{"enabled":true,"targets":["web"]}' + +curl -s -X PUT "$BASE/api/v1/admin/config/sites/1" -d '{"site_id":"site-1"}' + +# Поставить адрес в очередь и, отдельным вызовом позже, принудительно +# перепроверить его ещё раз — тот же метод, разница только в состоянии: +curl -s -X POST "$BASE/api/v1/admin/ips" -d '{"addresses":["203.0.113.10"]}' +curl -s -X POST "$BASE/api/v1/admin/ips" -d '{"addresses":["203.0.113.10"]}' # forced recheck + +# Остановить проверку, не дожидаясь checking_window_seconds: +curl -s -X POST "$BASE/api/v1/admin/ips/203.0.113.10/cancel" +``` + ## Модель состояний и связь методов с ней ``` @@ -327,9 +483,18 @@ queued ──(control-api сам, без вызова API)──▶ assigning_fi нет — это фоновый цикл (`Tick`), а не запрос/ответ. Площадки (`siteN_complete`) — опциональны: сколько их учитывается, -целиком определяется списком `sites` в конфиге control-api (0–3 записи). -Пустой список — агрегация ждёт только `egress_complete`, ни одна площадка -не требуется. Подробнее — [USAGE.md](USAGE.md#управление-площадками-проберами). +целиком определяется текущим списком `sites` (0–3 записи, управляется +через `/api/v1/admin/config/sites` — см. +[выше](#управление-очередью-и-конфигурацией)). Пустой список — агрегация +ждёт только `egress_complete`, ни одна площадка не требуется. Подробнее — +[USAGE.md](USAGE.md#управление-площадками-проберами). + +Два дополнительных перехода, оба инициируются оператором через +`/api/v1/admin/ips`, а не самим оркестратором: +- **любое нетерминальное состояние → `failed` (`overall_result: + "cancelled"`)** — `POST /api/v1/admin/ips/{ip}/cancel`; +- **`done`/`failed` → `queued` (новая попытка)** — `POST + /api/v1/admin/ips` с уже завершённым адресом в списке. ## Сквозной пример работы (curl) diff --git a/docs/DASHBOARD.md b/docs/DASHBOARD.md new file mode 100644 index 0000000..1ebf27e --- /dev/null +++ b/docs/DASHBOARD.md @@ -0,0 +1,99 @@ +# Admin Dashboard + +`admin-dashboard` — 4-й компонент системы: браузерная веб-панель, +дающая полное покрытие административного API `control-api` +([docs/API.md](API.md#управление-очередью-и-конфигурацией)) без единого +`curl`. Отдельный, полностью самостоятельный процесс — не хранит +состояния, не подключается к базе данных напрямую, общается с +`control-api` только через его же HTTP admin API. + +## Устройство + +- Рендеринг полностью на сервере: `html/template` + [htmx](https://htmx.org) + (частичные обновления без перезагрузки страницы) + + [Alpine.js](https://alpinejs.dev) (точечная клиентская интерактивность) + + [Pico CSS](https://picocss.com) в classless-сборке (минималистичный вид + на голой семантической разметке). Никакой сборки фронтенда нет — все три + библиотеки вендорены как статические файлы + (`internal/dashboard/static/vendor/`, см. `VENDOR.md` там же) и встроены + в бинарник через `//go:embed`. Дашборд работает полностью офлайн — при + открытии страницы браузер не делает ни одного запроса за пределы самого + дашборда (можно проверить через DevTools → Network). +- Браузер никогда не видит JSON `control-api` напрямую: каждая страница и + каждый htmx-фрагмент — это HTML, отрендеренный Go-хендлером дашборда + после вызова `control-api`. CORS и reverse-proxy не нужны. +- Как и `control-api`, дашборд **не аутентифицирован** — ограничивайте + доступ на уровне сети/файрвола (см. + [SETUP.md](SETUP.md#сетевые-доступы)). + +## Запуск + +```bash +cp configs/admin-dashboard.example.yaml /etc/cloud-ip-validator/admin-dashboard.yaml +# отредактируйте control_api.base_url под ваш стенд +admin-dashboard -config /etc/cloud-ip-validator/admin-dashboard.yaml +``` + +Развёртывание как systemd-юнита — по образцу остальных компонентов, см. +[SETUP.md](SETUP.md#развёртывание-admin-dashboard). Порт по умолчанию — +`:8090` (у `control-api` — `:8080`). + +## Страницы и что на них можно делать + +| Страница | Назначение | +|---|---| +| `/overview` | Сводная статистика: счётчики по состояниям, «текущая проверка» (live-снимок всех IP не в терминальном состоянии) и «последние N завершённых» (по умолчанию 20, `overview.last_completed_count`) с разбивкой pass/partial/fail/cancelled. Обновляется каждые `overview.poll_interval_seconds` секунд без перезагрузки страницы. | +| `/ips` | Полная очередь. Форма сверху принимает список адресов (по одному на строке или через запятую) и отправляет их в `POST /api/v1/admin/ips` — **один и тот же вызов** добавляет новые адреса и принудительно перезапускает уже завершённые (см. ниже). У каждого адреса — кнопка «Перепроверить» (для `done`/`failed`) или «Отменить» (для активных состояний). | +| `/ips/{ip}` | Детали одного адреса: все проверки текущей попытки и вся история событий. | +| `/validators` | Список валидаторов + создание/изменение `os_port_id`/удаление. | +| `/sites` | Три фиксированных слота площадок (1/2/3) — назначить/сменить/освободить `site_id`. | +| `/targets` | Группы целей для egress-проверок — создание/редактирование/удаление. | +| `/check-types` | Типы проверок (`https`/`icmp`/`ssh`/...), включение/выключение, привязка к группам целей. | + +### «Текущая» и «последняя завершённая» проверка + +В `control-api` нет понятия «запуска»/«цикла проверки» как отдельной +сущности — есть только общая очередь IP-адресов +(`docs/PLAN_ADMIN_DASHBOARD.md`). Дашборд ничего не меняет в этом +устройстве и не заводит своего состояния: + +- **Текущая проверка** — все адреса, которые прямо сейчас не в + состоянии `done`/`failed` (`queued`, `assigning_fip`, + `awaiting_self_check`, `checking`, `aggregating`), вычисляется заново на + каждый запрос из `GET /api/v1/admin/status` + `GET /api/v1/admin/ips`. +- **Последняя завершённая проверка** — последние N адресов, перешедших в + `done`/`failed`, отсортированные по `AggregatedAt` по убыванию (не + «последний запуск», а именно скользящее окно последних по времени + завершений). + +### Добавление адресов и принудительный повтор — один и тот же вызов + +Форма на `/ips` всегда бьёт в `POST /api/v1/admin/ips`. Поведение зависит +от текущего состояния каждого конкретного адреса (см. +[docs/API.md](API.md#post-apiv1adminips)): новый — встаёт в очередь; +уже `done`/`failed` — принудительно перезапускается; уже `queued` — +просто переупорядочивается; уже активно проверяется — не трогается +(дашборд честно показывает это в таблице, а не делает вид, что запрос +ничего не значил). + +## Конфигурация + +См. `configs/admin-dashboard.example.yaml`. Ключевые поля: + +- `server.listen_addr` — где слушает сам дашборд (по умолчанию `:8090`). +- `control_api.base_url` — адрес `control-api`, обязателен. +- `control_api.timeout_seconds` — таймаут HTTP-запросов к `control-api`. +- `overview.last_completed_count` — размер окна «последних завершённых» + на странице обзора. +- `overview.poll_interval_seconds` — как часто браузер опрашивает + `/overview/fragment` для live-обновления. + +## Отображение ошибок + +Любая ошибка `control-api` (4xx/5xx с телом `{"error":"..."}`) или сбой +связи с ним (недоступен, таймаут) показывается баннером наверху страницы, +а не приводит к падению дашборда — таблица/страница при этом всегда +отражает актуальное состояние `control-api` (дашборд перезапрашивает +данные после любой попытки мутации, независимо от её исхода). Жёлтый +баннер — бизнес-ошибка (4xx, например «валидатор занят»), красный — +инфраструктурная проблема (5xx или `control-api` недоступен). diff --git a/docs/DIAGRAMS.md b/docs/DIAGRAMS.md index 157b5c3..c699dca 100644 --- a/docs/DIAGRAMS.md +++ b/docs/DIAGRAMS.md @@ -32,15 +32,15 @@ control-api, фоновый оркестратор, база данных, вы ```mermaid flowchart TB subgraph OP["Оператор"] - CFG["control-api.yaml
(validators, sites, targets,
ip_addresses, check_types)"] + CFG["control-api.yaml
(bootstrap пустой БД:
validators, sites, targets,
check_types, ip_addresses)"] ENV["control-api.env
(OS_AUTH_URL, OS_TOKEN, ...)"] - ADMIN["curl /api/v1/admin/*"] + ADMIN["curl /api/v1/admin/*
(status/ips/validators,
ips submit/cancel,
config CRUD)"] end subgraph CAPI["control-api (управляющая машина, 1 экземпляр)"] HTTP["HTTP API
/api/v1/agents/*
/api/v1/probers/*
/api/v1/admin/*
/healthz"] ORCH["Оркестратор: Tick раз в
poll_interval_seconds
claim → associate FIP →
ожидание self-check →
checking → aggregate → release
+ lease sweep + heartbeat sweep"] - DB[("SQLite
validators / ip_queue
checks / events")] + DB[("SQLite
validators / ip_queue / sites /
target_groups / check_types /
checks / events")] OSCLIENT["OpenStack-клиент
(mode: mock | real)"] end @@ -49,9 +49,10 @@ flowchart TB VA["validator-agent ×N
(на каждой ВМ-валидаторе)"] PR["prober ×3
(на каждой внешней площадке)"] - CFG -->|"читается при старте
(инициализация validators, ip_queue)"| CAPI + CFG -->|"читается только один раз,
на пустых таблицах (bootstrap)"| DB ENV -->|"переменные окружения процесса"| OSCLIENT ADMIN --> HTTP + HTTP -->|"config/queue CRUD:
источник истины после
первого изменения"| DB HTTP --> ORCH ORCH <--> DB ORCH --> OSCLIENT @@ -63,14 +64,20 @@ flowchart TB **Пояснение.** `control-api` — единственный компонент с состоянием и единственная точка принятия решений (какой IP кому назначить, когда -считать проверку завершённой). Конфигурация читается один раз при -старте процесса (горячей перезагрузки нет — изменения требуют -`systemctl restart control-api`, см. [SETUP.md](SETUP.md)). Оркестратор +считать проверку завершённой). `control-api.yaml` используется только как +одноразовый bootstrap для четырёх секций (`validators`, `sites`, +`targets`, `check_types`) — читается лишь пока соответствующая таблица в +БД пуста; `ip_addresses` — отдельный, всегда аддитивный путь постановки в +очередь при каждом старте. После bootstrap все изменения этих сущностей, +включая состав очереди и принудительные повтор/остановку проверки, идут +через `/api/v1/admin/*` — «на лету», без `systemctl restart control-api` +(см. [API.md](API.md#управление-очередью-и-конфигурацией)). Оркестратор работает по таймеру независимо от HTTP-запросов — назначение IP валидаторам и агрегация результатов не привязаны к конкретному входящему -запросу, а выполняются фоновым циклом `Tick`. `validator-agent` и -`prober` — активная сторона: они сами инициируют все HTTP-запросы к -control-api (pull-модель), сам control-api к ним не обращается. +запросу, читая актуальную конфигурацию из БД на каждом проходе, а не +единожды при старте. `validator-agent` и `prober` — активная сторона: они +сами инициируют все HTTP-запросы к control-api (pull-модель), сам +control-api к ним не обращается. --- diff --git a/docs/PLAN_ADMIN_DASHBOARD.md b/docs/PLAN_ADMIN_DASHBOARD.md new file mode 100644 index 0000000..7c398d7 --- /dev/null +++ b/docs/PLAN_ADMIN_DASHBOARD.md @@ -0,0 +1,154 @@ +# План: `cmd/admin-dashboard` — веб-панель администратора + +> Статус: **реализовано**. Актуальное описание — [docs/DASHBOARD.md](DASHBOARD.md). + +## Context + +Единственный способ управлять `control-api` (очередь IP, валидаторы, +площадки, цели проверки) — HTTP API через `curl` (см. `docs/API.md`). API +уже покрывает весь необходимый функционал (`POST /api/v1/admin/ips` для +постановки/принудительного повтора, `POST /api/v1/admin/ips/{ip}/cancel` +для остановки, `/api/v1/admin/config/{validators,sites,targets,check-types}` +для CRUD), но curl неудобен для повседневного оперирования и не даёт +наглядной картины состояния очереди. Нужен браузерный admin dashboard — +графический доступ ко всей этой функциональности: сводная статистика +(текущая проверка / итог последней завершённой), управление конфигурацией +и принудительные операции над очередью — без YAML и без перезапуска +`control-api`. + +**Согласованные решения:** +- **Без сборки фронтенда.** Server-rendered `html/template` + htmx + (частичные AJAX-обновления) + Alpine.js (точечная клиентская + интерактивность) + Pico.css classless (минимализм на голой семантической + разметке). Все три библиотеки вендорятся как статические файлы и + встраиваются через `//go:embed` — без CDN во время выполнения (принцип + проекта — полностью автономный бинарник, работает офлайн). +- **Никакого нового backend-состояния.** «Текущая проверка» — live-снимок + IP не в терминальном состоянии. «Последняя завершённая» — последние N + (по умолчанию 20, настраивается) по `AggregatedAt` desc среди + `done`/`failed`, с разбивкой по `OverallResult`. Оба вычисляются на + каждый запрос из `GET /admin/status` + `GET /admin/ips` — никакого + понятия «запуска»/«батча» в `control-api` не добавляется. +- **Без аутентификации** — как и сам API; доступ ограничивается сетью/firewall. +- **Отдельный 4-й бинарник** (`cmd/admin-dashboard`), не новые маршруты + внутри `control-api`. Ходит в `control-api` только через существующий + HTTP admin API (`internal/apiclient.Client`). Рендеринг на сервере — + браузер не видит JSON `control-api` напрямую, CORS/reverse-proxy не нужны. +- Малые допущения: без пагинации очереди (текущий масштаб — десятки + адресов); баннер ошибок различает 4xx (жёлтый) и 5xx/транспортные + (красный); порт дашборда по умолчанию `:8090`. + +## 1. Структура файлов + +``` +cmd/admin-dashboard/main.go + +internal/dashboard/ + server.go, routes.go, client.go, dto.go, render.go, embed.go + handlers_overview.go, handlers_ips.go, handlers_validators.go, + handlers_sites.go, handlers_targets.go, handlers_checktypes.go + templates/ (layout, overview[+fragment], ips[+table], ip_detail, + validators[+table+row], sites[+table], targets[+table+row], + checktypes[+table+row], error_banner) + static/vendor/{htmx.min.js,alpine.min.js,pico.classless.min.css,VENDOR.md} + static/dashboard.css + +configs/admin-dashboard.example.yaml +deploy/systemd/admin-dashboard.service +docs/DASHBOARD.md +``` + +Таблица `ips` перерисовывается целиком при любой мутации (`POST +/admin/ips` может завести новые строки и поменять `sequence`). +`validators`/`sites`/`targets`/`check-types` — точечный swap одной ``. + +## 2. Маршруты дашборда + +| Метод | Путь | Вызов к control-api | +|---|---|---| +| GET | `/` | редирект на `/overview` | +| GET | `/overview`, `/overview/fragment` | `GET /admin/status`, `GET /admin/ips` | +| GET | `/ips`, `/ips/{ip}` | `GET /admin/ips`, `GET /admin/ips/{ip}` | +| POST | `/ips` | `POST /admin/ips` | +| POST | `/ips/{ip}/recheck` | `POST /admin/ips` `{"addresses":[ip]}` | +| POST | `/ips/{ip}/cancel` | `POST /admin/ips/{ip}/cancel` | +| GET/POST `/validators`, PUT/DELETE `/validators/{id}` | `.../config/validators[/{id}]` | +| GET `/sites`, PUT/DELETE `/sites/{index}` | `.../config/sites[/{index}]` | +| GET/POST `/targets`, PUT/DELETE `/targets/{group}` | `.../config/targets[/{group}]` | +| GET/POST `/check-types`, PUT/DELETE `/check-types/{name}` | `.../config/check-types[/{name}]` | +| GET | `/static/*` | embed.FS | + +`overview/fragment` — `hx-trigger="every Ns"` (из конфига), без фонового +тикера на сервере. `POST /targets`/`/check-types` — перевод «форма с +именем» → `PUT .../{name}` (control-api там upsert). + +## 3. Ошибки control-api + +`client.go`: не-2xx → `*apiErr{Status, Message}`. Хендлеры не отдают 500 — +рендерят страницу/фрагмент + out-of-band `error_banner.html` +(`hx-swap-oob="true"`, `id="error-banner"` в `layout.html`); статус ответа +дашборда = статус control-api (502 при транспортной ошибке). Баннер жёлтый +для 4xx, красный для 5xx/транспортных. + +## 4. Конфигурация + +`internal/config/config.go`, секция `AdminDashboard{Server, ControlAPI{ +BaseURL, TimeoutSeconds}, Overview{LastCompletedCount, PollIntervalSeconds}}` ++ `LoadAdminDashboard` (defaults: `:8090`, timeout 10s, N=20, poll=5s; +`BaseURL` обязателен). `configs/admin-dashboard.example.yaml`, +`deploy/systemd/admin-dashboard.service` (без CAP_NET_RAW/EnvironmentFile). + +## 5. DTO и клиент + +`internal/dashboard/dto.go` — свои wire-структуры (не импортируют +приватные DTO `internal/httpapi`, тот же паттерн, что `internal/probercore`): +snake_case-структуры дословно повторяют `internal/httpapi/dto_admin.go`; +для `GET /admin/ips[/{ip}]`/`GET /admin/validators` — зеркала untagged +PascalCase `db.IPQueueItem`/`db.Validator`/`db.Check`/`db.Event`. + +`client.go` — обёртка над `apiclient.Client`: `Status`, `ListIPs`, `GetIP`, +`SubmitIPs`, `CancelIP`, CRUD-методы для validators/sites/target-groups/ +check-types. + +`handlers_overview.go` — чистые функции: `currentlyChecking`, +`lastCompleted(items, n)`, `resultBreakdown`. + +## 6. Вендоринг статики + +Скачать один раз, закоммитить, задокументировать в `VENDOR.md`: +`htmx.min.js` (unpkg htmx.org, ядро без расширений), `alpine.min.js` +(unpkg alpinejs `dist/cdn.min.js`, IIFE-сборка), `pico.classless.min.css` +(unpkg @picocss/pico). + +## 7. Тестирование + +`httptest`-фейковый control-api + `httptest`-дашборд поверх него, проверка +рендера через `strings.Contains` (по образцу `internal/httpapi/handlers_config_test.go`). +Кейсы: overview live-агрегация и `resultBreakdown`; submit (happy + +пустой список); recheck (done→requeued, checking→skipped, явно показано); +cancel (happy + 409); CRUD-раунд-трип + конфликты (409/400) по всем 4 +сущностям; control-api недоступен → баннер + 502. + +**Обязательный ручной шаг:** браузерный смоук-тест против +`scripts/run-local-e2e.sh` (или отдельного mock control-api) — все +страницы/формы/действия, live-обновление во время реального прогона, +проверка через DevTools Network отсутствия внешних (CDN) запросов. + +## 8. Документация + +Новый `docs/DASHBOARD.md`; `README.md` («три компонента» → +«четыре»); `docs/SETUP.md` (компонент + раздел развёртывания + сетевые +доступы); `docs/API.md` (отсылка на дашборд в предупреждении об +открытости API). + +## Критичные файлы + +`internal/dashboard/{client.go,dto.go,routes.go,server.go, +handlers_overview.go,render.go,embed.go}`, `internal/config/config.go` +(`AdminDashboard`), `cmd/admin-dashboard/main.go`. + +## Проверка + +1. `go build ./... && go test ./...` +2. `scripts/run-local-e2e.sh` + `admin-dashboard` отдельно против того же control-api +3. Ручной браузерный смоук-тест (обязателен, не пропускается) diff --git a/docs/PLAN_API_CONFIG_MANAGEMENT.md b/docs/PLAN_API_CONFIG_MANAGEMENT.md index df05092..156c4cb 100644 --- a/docs/PLAN_API_CONFIG_MANAGEMENT.md +++ b/docs/PLAN_API_CONFIG_MANAGEMENT.md @@ -1,53 +1,70 @@ -# План доработки: API управления конфигурацией +# План доработки: динамическое управление конфигурацией и очередью через API -> Статус: **план на будущее, не реализовано**. Документ фиксирует -> согласованный дизайн доработки Control API, дающей возможность -> управлять `check_types`, `targets`, `validators` и `sites` через HTTP -> API вместо правки YAML + рестарта. Реализация — отдельная задача. +> Статус: **реализовано**. Документ фиксирует дизайн доработки Control +> API, дающей возможность управлять `validators`, `sites`, +> `check_types`/`targets` и очередью IP-адресов через HTTP API вместо +> правки YAML + рестарта, без перезапуска процесса. Актуальная +> спецификация методов — [docs/API.md](API.md#управление-очередью-и-конфигурацией); +> повседневные сценарии — [docs/USAGE.md](USAGE.md). ## Context -Сейчас `control-api` полностью read-only в части конфигурации: типы -проверок (`check_types`), список целей (`targets`), состав валидаторов -(`validators`) и внешних площадок (`sites`) читаются один раз из -`control-api.yaml` при старте процесса и живут дальше только в памяти -(`Orchestrator.Checks`, `Orchestrator.Sites`) либо (для валидаторов) в -таблице `validators`, куда при каждом рестарте они переупорядочиваются из -YAML. Единственный способ что-то поменять — отредактировать YAML и -выполнить `systemctl restart control-api`. Это задокументированное -ограничение (см. `docs/USAGE.md`). +Сейчас `control-api` в части конфигурации и очереди полностью read-only: +`validators`, `sites`, `check_types`/`targets` читаются один раз из YAML при +старте процесса (`cmd/control-api/main.go`) и живут в памяти +(`Orchestrator.Checks`, `Orchestrator.Sites`) либо переприменяются в БД при +каждом рестарте (`RegisterValidator` upsert). Список адресов на проверку +(`ip_addresses`) добавляется в очередь только при старте, и нет способа +принудительно перепроверить уже завершённый адрес или остановить проверку, +которая уже идёт. Единственный способ что-то поменять — отредактировать +YAML и выполнить `systemctl restart control-api`. -Согласованные решения (зафиксированы для будущей реализации): -- **Область доработки** — ровно четыре сущности: `check_types` (типы - проверок + их привязка к группам целей), `targets` (группы целей), - `validators` (состав ВМ-валидаторов), `sites` (состав из ≤3 внешних - площадок). Очередь `ip_addresses` и тайминги оркестратора - (`orchestrator.*`, `aggregation.*`, `inbound_checks.*`) — вне scope, - остаются YAML-only как сейчас. -- **Источник истины после первого изменения — БД.** YAML используется - только для одноразового bootstrap при пустой базе; после первого - запуска (или после первого API-изменения) YAML для этих 4 секций - больше не перечитывается и не переприменяется при рестартах. -- **Добавляется базовая аутентификация** — bearer-токен администратора, - которым закрывается весь namespace `/api/v1/admin/*` (не только новые - write-методы, но и существующие read-методы `status/ips/validators` — - единая политика для всего admin-namespace проще и логичнее половинчатой - защиты). Протокол `/api/v1/agents/*` и `/api/v1/probers/*` - (agent/prober) — вне scope, остаётся как есть. +Целевой набор фич: -## Важное архитектурное ограничение: площадки жёстко капнуты на 3 +1. Администратор передаёт через API список IP-адресов на проверку. +2. Администратор передаёт через API список валидаторов. +3. Администратор передаёт через API список целей (`targets`/`check_types`). +4. Администратор может принудительно инициировать проверку адреса, даже + если она уже была выполнена ранее. +5. Администратор может принудительно остановить идущую проверку. -Схема `ip_queue` хранит завершённость площадок как три отдельные колонки -(`site1_complete`, `site2_complete`, `site3_complete`) — это не список -произвольной длины. Поэтому API для `sites` не может быть обычным -CRUD-списком: это управление максимум тремя пронумерованными слотами -(`index` ∈ {1,2,3}), где `site_id` можно назначить, переименовать или -снять со слота. Это ограничение уже описано в `docs/USAGE.md` и явно -закладывается в дизайн API ниже, а не игнорируется. +Все действия — «на лету», без перезапуска процесса. Источник истины после +первого изменения через API — БД, YAML остаётся только bootstrap для пустой +базы. -## Общий план реализации +**Согласованные решения:** +- **Sites (площадки) включены в объём доработки** — тем же CRUD-подходом, + что и validators/targets/check_types. +- **Аутентификация (admin bearer-токен) НЕ входит в этот план** — API + остаётся открытым, как сейчас. Ограничение доступа — на уровне + сети/firewall (см. `docs/SETUP.md`). +- **Фичи 1 и 4 реализуются одним механизмом**, а не двумя разными + эндпоинтами. Администратор передаёт список IP-адресов в + `POST /api/v1/admin/ips`; для каждого адреса в списке: + - если адрес не встречался раньше — добавляется в очередь как новый; + - если адрес уже в терминальном состоянии (`done`/`failed`) — + принудительно перезапускается на проверку (сброс результата, новая + попытка, `attempt_number` увеличивается, `retry_count` обнуляется); + - если адрес уже в очереди (`queued`) — переупорядочивается под порядок + текущего списка (без дублирования); + - если адрес сейчас активно проверяется (`assigning_fip` / + `awaiting_self_check` / `checking` / `aggregating`) — не трогается + вообще (не создаём вторую параллельную проверку одного и того же + адреса). -### 1. Новая схема БД — `internal/db/migrations/0002_dynamic_config.sql` + Порядок обработки в рамках одного вызова соответствует порядку адресов в + переданном списке — повторная отправка того же списка без изменений даёт + тот же порядок прогона. + +## 1. Схема БД — новая миграция + обобщение `migrate()` + +`internal/db/db.go: migrate()` сейчас гейтится по `PRAGMA user_version`: +`>=1 → no-op`, иначе применяет единственный embedded `migrations/0001_init.sql` +и ставит `user_version=1`. Обобщается на упорядоченный список миграций +(embed `0002_dynamic_config.sql` вторым файлом), применяются по очереди все +версии выше текущей. + +Новый файл `internal/db/migrations/0002_dynamic_config.sql`: ```sql CREATE TABLE sites ( @@ -73,229 +90,98 @@ CREATE TABLE check_types ( ); ``` -Список целей внутри группы и список групп внутри типа проверки хранятся -как JSON-массив в TEXT-колонке (тот же паттерн, что уже используется для -`events.payload`) — они всегда читаются/пишутся целиком, отдельная -реляционная таблица тут не нужна (не переусложняем). +`validators` — существующая таблица (`0001_init.sql`), новых колонок не +требует. -`validators` — существующая таблица, новых колонок не требует. +## 2. Типизированные ошибки — `internal/db/errors.go` -`internal/db/db.go`: функцию `migrate()` обобщить со списка из одной -миграции (`version >= 1 → return`) на упорядоченный список -`{version, sql}` и применение всех версий выше текущего -`PRAGMA user_version` — понадобится и для этой, и для будущих миграций. +Сентинелы (`errors.New` + `%w`-обёртка), чтобы `httpapi`-хендлеры маппили +их в HTTP-статусы через `errors.Is`: `ErrNotFound` (404), `ErrConflict` +(409), `ErrBusy` (409, валидатор владеет IP), `ErrInUse` (409, группа +целей используется check_type'ом), `ErrValidation` (400), `ErrInvalidState` +(409, попытка отменить уже завершённую проверку). -### 2. Bootstrap-логика — новый файл `internal/db/bootstrap.go` +## 3. Bootstrap — `internal/db/bootstrap.go` ```go func (d *DB) BootstrapFromConfig(ctx context.Context, cfg *config.ControlAPI) error ``` -Переносит и обобщает то, что сейчас разбросано по -`cmd/control-api/main.go` (`RegisterValidator` в цикле + `SeedQueue`): +Заменяет текущий цикл `RegisterValidator` + `SeedQueue` в +`cmd/control-api/main.go`: +- `ip_addresses` → `SeedQueue` — без изменений (всегда доливает новые + адреса при каждом старте; отдельный YAML-only путь, не путать с runtime + `POST /api/v1/admin/ips`). +- `validators`, `sites`, `target_groups`, `check_types` — применяются + только если соответствующая таблица пуста. Если строки уже есть — YAML + для этой секции игнорируется. -- `ip_addresses` → `SeedQueue` — **без изменений**, как сейчас (всегда - доливает новые адреса, это уже вне scope доработки). -- `validators`, `sites`, `target_groups`, `check_types` — **новая - семантика**: применяется, **только если соответствующая таблица - сейчас пуста** (`SELECT COUNT(*) ... == 0`). Если в таблице уже есть - строки — YAML для этой секции полностью игнорируется, ничего не - трогаем. Это и есть «bootstrap один раз, дальше БД главная». +**Осознанное изменение поведения**: сейчас `RegisterValidator` при каждом +рестарте переприменяет `os_port_id` из YAML поверх БД. После доработки — +только на пустой таблице (иначе API-правки не переживали бы рестарт). -`internal/db` уже не будет зависеть от `internal/orchestrator` — только -новая зависимость `internal/db → internal/config` (обратной зависимости -`config → db` нет, циклов не возникает). +## 4. Запросы к БД -`cmd/control-api/main.go`: заменить текущий цикл `RegisterValidator` + -`SeedQueue` одним вызовом `database.BootstrapFromConfig(ctx, cfg)`. Это -же делает функцию тестируемой напрямую (используется в обновлённых -`orchestrator_test.go`/`httpapi_test.go` вместо ручного построения -`Orchestrator.Checks`/`.Sites`). +- `internal/db/queries_sites.go`: `ListSites`, `UpsertSite(idx, siteID)`, + `DeleteSite(idx)`, `GetSiteIndex(ctx, siteID) (int, error)` (0, если не + найден — не ошибка). +- `internal/db/queries_targetgroups.go`: `ListTargetGroups`, + `UpsertTargetGroup(name, targets)`, `DeleteTargetGroup(name)` + (`ErrInUse`, если ссылается check_type), `GetTargetGroup(name)`. +- `internal/db/queries_checktypes.go`: `ListCheckTypes`, + `ListResolvedCheckTypes` (разворачивает группы в плоский список targets), + `UpsertCheckType(name, enabled, targetGroups)` (`ErrValidation`, если + группа не существует), `DeleteCheckType(name)`. +- `internal/db/queries_validators.go` (дополнить): `AdminCreateValidator`, + `AdminUpdateValidatorPort`, `DeleteValidator` (`ErrBusy`, если владеет + IP). +- `internal/db/queries_ipqueue.go` (дополнить): `SubmitIPs(ctx, + addresses []string) (SubmitIPsResult, error)` — единая транзакция, + реализует правило из Context выше; `CancelIP(ctx, ipID int64) error` — + условный `UPDATE ... WHERE state NOT IN ('done','failed')`, + `ErrInvalidState` при гонке/уже завершённой проверке. -**Важное следствие смены семантики валидаторов**: сейчас при каждом -рестарте `control-api` валидаторы из YAML переприменяются (в частности, -может тихо откатить `os_port_id`, изменённый через API/вручную в БД). -После доработки — только на пустой таблице. Это осознанное поведенческое -изменение, требует апдейта `docs/SETUP.md`/`docs/USAGE.md` (шаг 6 плана). +`ResultCancelled = "cancelled"` добавляется в `internal/db/models.go`. -### 3. Запросы к БД для новых сущностей +## 5. Оркестратор -Новые файлы, по аналогии с существующими `queries_*.go`: +`internal/orchestrator/orchestrator.go`: убрать статические поля +`Checks`/`Sites`, читать динамически из БД (`ListResolvedCheckTypes`, +`ListSites`, `GetSiteIndex`) в `AssignmentForValidator`, +`expectedCheckCount()`, `isReadyToAggregate()`. Новый метод `ForceCancel` +для фичи 5: отвязывает FIP (best-effort), помечает IP `cancelled`, +освобождает валидатора. -- **`internal/db/queries_sites.go`**: `ListSites`, `UpsertSite(idx, siteID)` - (проверяет допустимость `idx` 1..3 и уникальность `site_id` до записи, - чтобы вернуть чистую типизированную ошибку, а не сырую SQL), `DeleteSite(idx)`, - `GetSiteIndex(siteID) (int, error)` — заменяет текущий - `Orchestrator.SiteIndexForID`, который сканирует статический слайс. -- **`internal/db/queries_targetgroups.go`**: `ListTargetGroups`, - `UpsertTargetGroup(name, targets)`, `DeleteTargetGroup(name)` — - **перед удалением проверяет**, что ни один `check_types` не ссылается - на эту группу (иначе `ErrInUse`), `GetTargetGroup(name)`. -- **`internal/db/queries_checktypes.go`**: `ListCheckTypes`, - `ListResolvedCheckTypes` (сразу разворачивает имена групп в плоский - список URL — то, что раньше строил `orchestrator.New()` один раз при - старте), `UpsertCheckType(name, enabled, targetGroups)` (**проверяет**, - что все переданные `targetGroups` существуют — иначе `ErrValidation`), - `DeleteCheckType(name)`. -- **`internal/db/queries_validators.go`** (дополнить существующий файл): - `AdminCreateValidator(id, osPortID)` (409/`ErrConflict`, если уже - есть), `AdminUpdateValidatorPort(id, osPortID)` (404/`ErrNotFound`, - если нет), `DeleteValidator(id)` (409/`ErrBusy`, если - `current_ip_id IS NOT NULL` — валидатор сейчас владеет IP). Не путать - с существующим `RegisterValidator` — тот остаётся as-is и продолжает - использоваться только агентом при самостоятельной регистрации - (`handleAgentRegister`), полей `os_port_id` не трогает при - self-registration (это уже так в текущем коде). +**Принятый компромисс**: если конфигурация меняется API-запросом ровно в +момент агрегации уже идущей проверки, эта попытка агрегирует по текущей +(уже изменённой) конфигурации — деградирует безопасно через +`missing_counts_as_fail`, самоисправляется на следующей попытке. -Новый файл **`internal/db/errors.go`** с типизированными сентинелами -(`ErrNotFound`, `ErrConflict`, `ErrBusy`, `ErrValidation`, через `errors.New` -+ `%w`-обёртку в местах возврата) — чтобы `httpapi`-хендлеры мапили их в -404/409/400 через `errors.Is`, а не всё подряд в 500 (как сейчас местами -получается по умолчанию). +## 6. HTTP API -### 4. Оркестратор — переход на динамическое чтение конфигурации - -`internal/orchestrator/orchestrator.go`: -- Убрать поля `Checks []CheckConfig` и `Sites []config.SiteConfig` из - `Orchestrator` (сейчас вычисляются один раз в `New()` и застывают на - весь жизненный цикл процесса — это и есть корень проблемы). `Inbound` - остаётся статическим полем как сейчас (вне scope). -- `AssignmentForValidator` — вместо `return item, o.Checks, nil` вызывает - `o.DB.ListResolvedCheckTypes(ctx)` и возвращает актуальный на данный - момент список. -- `SiteIndexForID` — удаляется, вызовы (`handleProberRegister`, - `handleProberAssignments`, `handleProberResults`) переходят на - `o.DB.GetSiteIndex(ctx, siteID)`. -- `expectedCheckCount()` — читает актуальные `ListResolvedCheckTypes` и - `ListSites` из БД на момент агрегации, а не статические поля. - -**Принятый компромисс (осознанно, без over-engineering):** если -`check_types`/`targets`/`sites` меняются API-запросом ровно в момент, -когда чей-то IP уже находится в `checking` (self-check уже пройден, -проверки уже назначены агенту), агрегация этой конкретной попытки -посчитает *текущую* (уже изменённую) конфигурацию, а не ту, что была на -момент выдачи задания. На практике это узкое окно в несколько секунд -между админ-изменением и завершением проверки; деградирует безопасно — -через существующий механизм `missing_counts_as_fail` результат в худшем -случае будет `partial` вместо `pass` для одной попытки, самоисправляется -на следующей (после retry/requeue). Полный snapshot-per-attempt (доп. -колонки в `ip_queue` с зафиксированным ожидаемым числом проверок) — -возможное будущее усиление, не требуется для этой доработки. - -### 5. HTTP API - -Новый файл **`internal/httpapi/handlers_config.go`** и DTO в -`dto.go`. Все — под префиксом `/api/v1/admin/config/*`, JSON в -snake_case (в отличие от существующих `/admin/status|ips|validators`, -которые отдают сырые Go-поля в PascalCase — для новых, «настоящих» -management-эндпоинтов сразу делаем нормальный контракт, старые не -трогаем, чтобы не ломать уже задокументированное поведение). - -| Метод | Путь | Тело | Успех | Ошибки | -|---|---|---|---|---| -| GET | `/api/v1/admin/config/validators` | — | `[{validator_id, os_port_id, state}]` | | -| POST | `/api/v1/admin/config/validators` | `{validator_id, os_port_id}` | 201 | 409 если уже есть | -| PUT | `/api/v1/admin/config/validators/{id}` | `{os_port_id}` | 200 | 404 | -| DELETE | `/api/v1/admin/config/validators/{id}` | — | 200 | 404, 409 если владеет IP | -| GET | `/api/v1/admin/config/sites` | — | `[{index, site_id}]` (до 3 строк) | | -| PUT | `/api/v1/admin/config/sites/{index}` | `{site_id}` | 200 | 400 если `index` не 1..3, 409 если `site_id` занят другим слотом | -| DELETE | `/api/v1/admin/config/sites/{index}` | — | 200 | 404 | -| GET | `/api/v1/admin/config/targets` | — | `[{name, targets}]` | | -| PUT | `/api/v1/admin/config/targets/{group}` | `{targets:[...]}` | 200 | 400 пустой список | -| DELETE | `/api/v1/admin/config/targets/{group}` | — | 200 | 404, 409 если используется check_type'ом | -| GET | `/api/v1/admin/config/check-types` | — | `[{name, enabled, targets}]` | | -| PUT | `/api/v1/admin/config/check-types/{name}` | `{enabled, targets:[group,...]}` | 200 | 400 если группа не существует | -| DELETE | `/api/v1/admin/config/check-types/{name}` | — | 200 | 404 | - -`routes.go`: все существующие и новые `/api/v1/admin/*`-маршруты -оборачиваются `s.requireAdmin(...)`. - -### 6. Аутентификация - -- `internal/config/config.go`: в `ServerConfig` добавить - `AdminTokenEnv string \`yaml:"admin_token_env"\`` — по аналогии с - `openstack.*_env` полями (в YAML — только *имя* переменной, не сам - токен). -- `internal/httpapi/server.go`: `Server.AdminToken string` + - `func (s *Server) requireAdmin(next http.HandlerFunc) http.HandlerFunc` - — сверяет `Authorization: Bearer ` через - `crypto/subtle.ConstantTimeCompare`. Если `s.AdminToken == ""` — - пропускает без проверки (обратная совместимость). -- `cmd/control-api/main.go`: если `cfg.Server.AdminTokenEnv` задан, но - `os.Getenv(...)` пуст — **отказ запуска** с понятной ошибкой - (fail-safe, не запускаемся с «пустым паролем»). Если - `AdminTokenEnv` вообще не задан — запускаемся как сейчас, но пишем - явный `log.Warn` про незащищённый admin API. -- `configs/control-api.example.yaml`, `deploy/systemd/control-api.service` - (добавить пример переменной в `EnvironmentFile`) — обновить. - -### 7. Обновление существующих тестов и добавление новых - -- `internal/orchestrator/orchestrator_test.go`, - `internal/httpapi/httpapi_test.go`: заменить ручное построение - `cfg.CheckTypes/.Targets/.Sites` + прямые поля `Orchestrator{Checks:...}` - на `db.BootstrapFromConfig(ctx, cfg)` перед `orchestrator.New(...)` — - сами тестовые сценарии (happy path, partial, lease reclaim) не меняются - по сути, меняется только способ засеять конфигурацию. -- Новые unit-тесты: `internal/db/queries_dynconfig_test.go` (CRUD + - граничные случаи: удаление занятого валидатора → `ErrBusy`, удаление - группы целей, на которую ссылается check_type → `ErrInUse`, - upsert check_type с несуществующей группой → `ErrValidation`, upsert - сайта с чужим `site_id` → `ErrConflict`, bootstrap на непустой таблице - → YAML игнорируется). -- Новый `internal/httpapi/handlers_config_test.go` (или расширение - `httpapi_test.go`): сквозной сценарий — создать валидатора и сайт через - API вместо конфига, убедиться, что IP реально дошёл до `done`; смена - `check_types` между запусками влияет на следующий назначенный IP. -- Обновить `scripts/run-local-e2e.sh` не требуется по сути (bootstrap - из YAML при пустой БД работает как раньше), но стоит добавить один шаг - с `curl -X PUT .../config/check-types/ssh` как живую демонстрацию. - -### 8. Документация (после реализации) - -- `docs/API.md`: новый раздел «Методы управления конфигурацией» с - таблицей выше + примеры curl (создание валидатора, отключение ssh, - добавление цели, назначение площадки на слот) + раздел про - `Authorization: Bearer`. -- `docs/SETUP.md`: шаг про `server.admin_token_env` в - «Переменные окружения для OpenStack» (переименовать раздел или - добавить рядом «и для admin-токена»); явно описать новую - bootstrap-once семантику `validators`/`sites`/`check_types`/`targets`. -- `docs/USAGE.md`: заменить текущие разделы «Управление валидаторами» / - «Управление площадками» (сейчас там «только через YAML + restart») на - актуальные — через API; убрать утверждение «нет API-метода» там, где - оно перестало быть верным. -- `docs/DIAGRAMS.md`: в диаграмму control plane (раздел 1) добавить - новую стрелку «Оператор → HTTP API → БД (config CRUD)» вместо текущей - «CFG → читается при старте (инициализация)» как единственного пути. +`POST /api/v1/admin/ips` — постановка/принудительный перезапуск (фичи 1 и +4). `POST /api/v1/admin/ips/{ip}/cancel` — остановка (фича 5). +`/api/v1/admin/config/{validators,sites,targets,check-types}` — CRUD +(фичи 2 и 3), snake_case DTO. Подробности — `docs/API.md`. ## Критичные файлы -- `internal/db/migrations/0002_dynamic_config.sql` (новый) -- `internal/db/db.go` (обобщить `migrate()`) -- `internal/db/bootstrap.go` (новый) -- `internal/db/errors.go` (новый) +- `internal/db/migrations/0002_dynamic_config.sql`, `internal/db/db.go`, + `internal/db/bootstrap.go`, `internal/db/errors.go`, `internal/db/models.go` - `internal/db/queries_sites.go`, `queries_targetgroups.go`, - `queries_checktypes.go` (новые), `queries_validators.go` (дополнить) -- `internal/orchestrator/orchestrator.go` (убрать статические - `Checks`/`Sites`, читать из БД) -- `internal/httpapi/handlers_config.go` (новый), `dto.go`, `routes.go`, - `server.go` (`requireAdmin`) -- `internal/config/config.go` (`AdminTokenEnv`) -- `cmd/control-api/main.go` (bootstrap-вызов, проверка токена при старте) + `queries_checktypes.go`, `queries_validators.go`, `queries_ipqueue.go` +- `internal/orchestrator/orchestrator.go` +- `internal/httpapi/handlers_config.go`, `handlers_admin.go`, + `dto_admin.go`, `routes.go` +- `cmd/control-api/main.go` -## Проверка (когда план будет реализовываться) +## Проверка -1. `go build ./... && go test ./...` — все существующие + новые unit- и - httpapi-тесты проходят. -2. `scripts/run-local-e2e.sh` — офлайн-сценарий по-прежнему проходит от - начала до конца без ручного вмешательства (bootstrap из YAML при - пустой БД работает как раньше). -3. Ручная проверка нового контракта: поднять `control-api` с пустой БД и - `admin_token_env` без токена → админ-запрос без заголовка проходит; - задать токен → запрос без `Authorization` получает 401; создать - валидатора/площадку/группу целей/тип проверки через API без - единой строчки в YAML, убедиться, что IP реально проходит полный цикл - проверки на этой конфигурации; попытаться удалить валидатора, пока он - владеет IP → 409; перезапустить `control-api` и убедиться, что - API-изменения пережили рестарт, а YAML их не затёр. +1. `go build ./... && go test ./...` +2. `scripts/run-local-e2e.sh` +3. Ручная проверка: создать validator/site/target-group/check-type только + через API; `POST /admin/ips` с уже `done`-адресом → повторный полный + цикл; `POST /admin/ips` с адресом в `checking` → не трогается; + `POST /admin/ips/{ip}/cancel` во время `checking` → FIP отвязан, + валидатор свободен, `overall_result=cancelled`; удаление занятого + валидатора → 409; рестарт control-api → все API-изменения сохранились. diff --git a/docs/SETUP.md b/docs/SETUP.md index 1f44b1c..86ec27a 100644 --- a/docs/SETUP.md +++ b/docs/SETUP.md @@ -21,6 +21,7 @@ - [Развёртывание control-api](#развёртывание-control-api) - [Развёртывание validator-agent на ВМ-валидаторах](#развёртывание-validator-agent-на-вм-валидаторах) - [Развёртывание prober на внешних площадках](#развёртывание-prober-на-внешних-площадках) +- [Развёртывание admin-dashboard](#развёртывание-admin-dashboard) - [Проверка после запуска](#проверка-после-запуска) - [Сетевые доступы](#сетевые-доступы) @@ -31,10 +32,13 @@ | `control-api` | Отдельная управляющая машина/ВМ с доступом к OpenStack API | 1 (без HA) | | `validator-agent` | Каждая ВМ-валидатор в сервисном проекте облака | по числу валидаторов | | `prober` | По одному на каждой из внешних тестовых площадок | 3 (по числу площадок) | +| `admin-dashboard` | Любая машина с сетевым доступом до `control-api` (опционально) | 0 или 1 | -`control-api` — единственный компонент с состоянием (SQLite). Валидаторы и -проберы не хранят локального состояния и полностью управляются через опрос -control-api (см. [API.md](API.md)). +`control-api` — единственный компонент с состоянием (SQLite). Валидаторы, +проберы и `admin-dashboard` не хранят локального состояния и полностью +управляются через опрос control-api (см. [API.md](API.md), +[DASHBOARD.md](DASHBOARD.md)). `admin-dashboard` не обязателен — вся его +функциональность доступна и через `curl` напрямую по API. ## Требования @@ -58,7 +62,7 @@ control-api (см. [API.md](API.md)). ### Вариант A: готовые бинарники из репозитория (рекомендуется) -В директории `bin/` репозитория уже лежат три готовых бинарника — +В директории `bin/` репозитория уже лежат четыре готовых бинарника — собирать их на целевых серверах не нужно, разворачивание сразу начинается с копирования и запуска (раздел [«Развёртывание control-api»](#развёртывание-control-api) и далее). @@ -68,6 +72,7 @@ bin/ ├── control-api # ~11 МБ ├── validator-agent # ~7 МБ ├── prober # ~7 МБ +├── admin-dashboard # ~9 МБ (опционален, см. DASHBOARD.md) └── SHA256SUMS ``` @@ -97,6 +102,7 @@ sha256sum -c bin/SHA256SUMS scp bin/control-api control-api-host:/tmp/ scp bin/validator-agent validator-host-01:/tmp/ scp bin/prober probe-site-1:/tmp/ +scp bin/admin-dashboard dashboard-host:/tmp/ # опционально ``` > Если целевая платформа отличается от linux/amd64 (например, ВМ на @@ -113,11 +119,13 @@ export CGO_ENABLED=0 GOOS=linux GOARCH=amd64 # поменяйте GOARCH дл go build -trimpath -ldflags="-s -w" -o bin/control-api ./cmd/control-api go build -trimpath -ldflags="-s -w" -o bin/validator-agent ./cmd/validator-agent go build -trimpath -ldflags="-s -w" -o bin/prober ./cmd/prober +go build -trimpath -ldflags="-s -w" -o bin/admin-dashboard ./cmd/admin-dashboard ``` Каждый бинарник самодостаточен — скопируйте нужный файл на соответствующую машину (control-api → управляющая машина, validator-agent -→ каждый валидатор, prober → каждая площадка). +→ каждый валидатор, prober → каждая площадка, admin-dashboard → +опционально, любая машина с доступом до control-api). Убедиться, что всё собирается и юнит-тесты проходят: @@ -130,7 +138,7 @@ go build ./... && go test ./... обновляются автоматически**: ```bash -sha256sum bin/control-api bin/validator-agent bin/prober | sed 's#bin/##' > bin/SHA256SUMS +sha256sum bin/control-api bin/validator-agent bin/prober bin/admin-dashboard | sed 's#bin/##' > bin/SHA256SUMS ``` ## Быстрая проверка без OpenStack (offline-режим) @@ -181,6 +189,15 @@ cp configs/control-api.example.yaml /etc/cloud-ip-validator/control-api.yaml целей для egress-проверок (по умолчанию — hub.docker.com, github.com, packages.ubuntu.com) или включите `ssh` (по умолчанию выключен). +> `validators`, `sites`, `targets` и `check_types` читаются из этого файла +> только один раз — при самом первом старте против пустой базы данных +> (bootstrap). После этого все последующие изменения этих четырёх секций +> вносятся через `/api/v1/admin/config/*`, а не правкой YAML — см. +> [API.md](API.md#управление-очередью-и-конфигурацией) и +> [USAGE.md](USAGE.md#управление-валидаторами). `ip_addresses` — исключение, +> он остаётся YAML + аддитивным добавлением при каждом старте (плюс +> `POST /api/v1/admin/ips` для управления очередью без рестарта). + ### 2. Переменные окружения для OpenStack Учётные данные передаются **только** через переменные окружения — никогда @@ -276,15 +293,29 @@ systemctl enable --now control-api **Первичная инициализация базы данных происходит автоматически** — при первом старте `control-api` создаёт файл SQLite по пути `database.path` -из конфига (миграция схемы применяется один раз, повторные запуски — -no-op). Отдельной команды "init db" не требуется. +из конфига (миграции схемы применяются один раз каждая, повторные запуски +— no-op). Отдельной команды "init db" не требуется. При каждом старте control-api также: -1. Регистрирует в БД всех валидаторов из `validators` конфига (если их - там ещё нет). -2. Добавляет в очередь все адреса из `ip_addresses`, которых там ещё нет - (уже обработанные ранее адреса повторно не добавляются и не - сбрасываются — см. [USAGE.md](USAGE.md#добавление-новых-ip-в-очередь)). +1. **Bootstrap-once для `validators`/`sites`/`targets`/`check_types`.** + YAML применяется **только если соответствующая таблица в БД сейчас + пуста** — то есть только на самом первом старте против чистой базы. + Как только в таблице появилась хотя бы одна строка (через этот + bootstrap либо через `/api/v1/admin/config/*`, см. + [API.md](API.md#управление-очередью-и-конфигурацией)), YAML для этой + секции больше не перечитывается ни при одном последующем рестарте — + источник истины переключается на БД. Это осознанное отличие от более + ранних версий, где `validators` из YAML переприменялись при каждом + рестарте: теперь правки, сделанные через admin API (например, смена + `os_port_id` валидатора), переживают рестарт вместо того, чтобы + тихо откатываться. +2. **Всегда аддитивно** добавляет в очередь все адреса из + `ip_addresses`, которых там ещё нет (уже обработанные ранее адреса + повторно не добавляются и не сбрасываются — см. + [USAGE.md](USAGE.md#добавление-новых-ip-в-очередь)). Это отдельный, + не завязанный на bootstrap-once путь — не путайте с + `POST /api/v1/admin/ips`, который умеет то же самое (и ещё + принудительный повтор уже проверенных адресов) без перезапуска. Проверить, что процесс поднялся: @@ -327,6 +358,28 @@ systemctl enable --now prober journalctl -u prober -f ``` +## Развёртывание admin-dashboard + +Опционально — вся его функциональность доступна и через `curl` напрямую +по API (см. [API.md](API.md)). На любой машине с сетевым доступом до +`control-api`: + +```bash +cp bin/admin-dashboard /usr/local/bin/admin-dashboard +cp deploy/systemd/admin-dashboard.service /etc/systemd/system/ +mkdir -p /etc/cloud-ip-validator +cp configs/admin-dashboard.example.yaml /etc/cloud-ip-validator/admin-dashboard.yaml +# отредактировать control_api.base_url под ваш стенд + +systemctl daemon-reload +systemctl enable --now admin-dashboard +journalctl -u admin-dashboard -f +``` + +Открыть `http://:8090/` в браузере. Подробнее о +страницах и о том, что дашборд может (и не может) — в +[DASHBOARD.md](DASHBOARD.md). + ## Проверка после запуска После того как control-api, все валидаторы и все три пробера запущены: @@ -376,7 +429,14 @@ curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool очередь — см. [USAGE.md](USAGE.md#частые-проблемы-и-что-с-ними-делать). - `control-api` → OpenStack Keystone/Neutron API (`OS_AUTH_URL` и далее по каталогу сервисов). +- `admin-dashboard` → `control-api`: тот же порт (`server.listen_addr`), + адрес задаётся в `control_api.base_url` конфига дашборда. +- Оператор (браузер) → `admin-dashboard`: порт из `server.listen_addr` + дашборда (по умолчанию 8090). API control-api сейчас не аутентифицирован (см. предупреждение в начале -[API.md](API.md)) — ограничивайте доступ к порту control-api на уровне -сети/firewall теми хостами, где реально работают валидаторы и проберы. +[API.md](API.md)) — то же самое верно и для `admin-dashboard`, который +это API оборачивает. Ограничивайте доступ к обоим портам на уровне +сети/firewall: к control-api — теми хостами, где реально работают +валидаторы, проберы и сам дашборд; к дашборду — теми, кому разрешено +администрировать стенд. diff --git a/docs/USAGE.md b/docs/USAGE.md index ff94cad..6b1693c 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -17,7 +17,9 @@ - [Просмотр деталей и истории по конкретному адресу](#просмотр-деталей-и-истории-по-конкретному-адресу) - [Управление валидаторами](#управление-валидаторами) - [Управление площадками (проберами)](#управление-площадками-проберами) +- [Управление целями проверки](#управление-целями-проверки) - [Повторная проверка адреса](#повторная-проверка-адреса) +- [Принудительная остановка проверки](#принудительная-остановка-проверки) - [Частые проблемы и что с ними делать](#частые-проблемы-и-что-с-ними-делать) ## Как устроена работа с системой @@ -43,29 +45,32 @@ ## Добавление новых IP в очередь -**В текущей версии добавление адресов происходит только через конфиг -control-api**, отдельного API-метода "добавить IP в очередь" нет. +Основной способ — API, без перезапуска процесса: -1. Добавьте новые адреса в список `ip_addresses` в - `/etc/cloud-ip-validator/control-api.yaml` (в конец списка, либо в - нужном порядке — очередь обрабатывается строго в порядке следования - списка, `sequence`). -2. Перезапустите control-api: - ```bash - systemctl restart control-api - ``` +```bash +curl -s -X POST http://:8080/api/v1/admin/ips \ + -d '{"addresses": ["203.0.113.10", "203.0.113.11"]}' +``` -Это безопасно для уже идущей работы: при старте control-api добавляет в -очередь только **новые** адреса (те, которых там ещё нет) — уже -обработанные ранее адреса не сбрасываются и повторно не проверяются. -Адреса, которые были удалены из `ip_addresses`, но уже есть в базе, -**не удаляются** из очереди/истории автоматически — если конкретный адрес -больше не нужно проверять и его нет в очереди/в процессе, можно просто -оставить как есть (историю он не портит). +```json +{"added": ["203.0.113.10", "203.0.113.11"], "requeued": [], "reordered": [], "skipped_in_progress": []} +``` -> Совет: держите `control-api.yaml` под версионным контролем (git) — -> список адресов на проверку тогда одновременно служит и журналом того, -> что вообще когда-либо ставилось в очередь. +Адреса обрабатываются в порядке, в котором перечислены в `addresses` — +именно в этом порядке они и встанут в очередь друг за другом. Метод +идемпотентен относительно уже идущих проверок: адрес, который сейчас +активно проверяется, в ответе окажется в `skipped_in_progress` и не будет +тронут (см. [«Повторная проверка адреса»](#повторная-проверка-адреса) +ниже — тот же метод форсирует перепроверку уже завершённых адресов). + +Также по-прежнему можно добавить адреса через `ip_addresses` в +`/etc/cloud-ip-validator/control-api.yaml` и перезапустить control-api — +при каждом старте control-api доливает в очередь только новые адреса из +этого списка (уже обработанные ранее не сбрасываются и повторно не +проверяются). Держать `control-api.yaml` под версионным контролем (git) +по-прежнему полезно как журнал того, что изначально ставилось в очередь +при разворачивании стенда — но для повседневного добавления адресов проще +и быстрее пользоваться API выше. ## Наблюдение за очередью @@ -108,7 +113,7 @@ curl -s http://:8080/api/v1/admin/ips \ | Поле | Значение | |---|---| | `IPAddress` | Проверяемый адрес | -| `Sequence` | Позиция в очереди (порядок из конфига) | +| `Sequence` | Позиция в очереди (порядок постановки — из конфига при первом старте либо из последнего вызова `POST /api/v1/admin/ips`) | | `State` | Текущий этап: `queued`, `assigning_fip`, `awaiting_self_check`, `checking`, `aggregating`, `done`, `failed` | | `OwnerValidatorID` | Какой валидатор сейчас (или последним) занимался этим адресом | | `FIPID` | Идентификатор Floating IP в OpenStack, к которому привязан адрес (пусто, если ещё/уже не привязан) | @@ -116,7 +121,7 @@ curl -s http://:8080/api/v1/admin/ips \ | `RetryCount` | Сколько раз адрес уже переставлялся в очередь заново | | `EgressComplete` | Валидатор закончил исходящие проверки | | `Site1Complete` / `Site2Complete` / `Site3Complete` | Соответствующая площадка закончила входящие проверки | -| `OverallResult` | Итог: `pass`, `partial`, `fail`, либо пусто, пока проверка не завершена | +| `OverallResult` | Итог: `pass`, `partial`, `fail`, `cancelled` (принудительно остановлена, см. [«Принудительная остановка проверки»](#принудительная-остановка-проверки)), либо пусто, пока проверка не завершена | | `AssignedAt` / `AggregatedAt` / `FIPReleasedAt` | Метки времени соответствующих этапов | ## Как читать итоговый результат (pass/partial/fail) @@ -137,6 +142,10 @@ curl -s http://:8080/api/v1/admin/ips \ трафик валидатора не пошёл через назначенный FIP — и попытки исчерпались). Смотрите `events` по этому адресу (см. ниже), чтобы понять, на каком шаге и почему. +- **`cancelled`** — проверку остановил оператор через `POST + /api/v1/admin/ips/{ip}/cancel` (`State` при этом — `failed`), а не + система по итогам проверок. Отличать от обычного `fail` полезно, чтобы + не путать «адрес не прошёл проверку» с «проверку прервали вручную». Отсутствие ответа от источника (площадка не прислала результат до истечения `checking_window_seconds`) засчитывается как провал — это @@ -180,21 +189,48 @@ curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool (занят), `unreachable` (пропустил heartbeat дольше `orchestrator.heartbeat_timeout_seconds`). -**Добавление нового валидатора:** +**Добавление нового валидатора (без перезапуска control-api):** 1. Поднимите новую ВМ в сервисном проекте облака, узнайте её Neutron `port_id`. -2. Добавьте запись в `validators` в `control-api.yaml` - (`validator_id` + `os_port_id`) и перезапустите `control-api`. +2. Зарегистрируйте валидатора через API: + ```bash + curl -s -X POST http://:8080/api/v1/admin/config/validators \ + -d '{"validator_id": "validator_05", "os_port_id": "port-abc123"}' + ``` 3. Разверните и запустите `validator-agent` на новой ВМ с тем же `validator_id` в его конфиге (см. [SETUP.md](SETUP.md#развёртывание-validator-agent-на-вм-валидаторах)). +Сменить `os_port_id` уже существующего валидатора (например, после +пересоздания ВМ) — `PUT /api/v1/admin/config/validators/{id}` с телом +`{"os_port_id": "новый-port-id"}`. + +Полный список зарегистрированных валидаторов — `GET +/api/v1/admin/config/validators` (в отличие от `GET +/api/v1/admin/validators`, отдаёт `snake_case` и без текущего IP — +только конфигурационные поля). + **Вывод валидатора из эксплуатации:** остановите на нём `validator-agent` (`systemctl stop validator-agent`). Он перестанет получать новые задания после того, как закончит текущее (если оно было); если он был убит посреди работы — control-api сам заберёт у него незавершённый адрес обратно в очередь по истечении -`orchestrator.lease_ttl_seconds`. Удалять запись из `control-api.yaml` -не обязательно — просто выключенный агент не будет ничего забирать. +`orchestrator.lease_ttl_seconds`. Удалять регистрацию валидатора не +обязательно — просто выключенный агент не будет ничего забирать. Если всё +же нужно убрать валидатора из системы совсем: +```bash +curl -s -X DELETE http://:8080/api/v1/admin/config/validators/validator_05 +``` +Возвращает `409`, если валидатор прямо сейчас владеет каким-то IP — +дождитесь освобождения (или принудительно остановите его проверку, см. +[«Принудительная остановка проверки»](#принудительная-остановка-проверки)) +перед удалением. + +> Правки через `validators[]` в `control-api.yaml` тоже поддерживаются, +> но только как bootstrap пустой базы данных при самом первом старте — как +> только в БД есть хотя бы один валидатор, YAML для этой секции +> игнорируется при всех последующих рестартах (см. +> [SETUP.md](SETUP.md#развёртывание-control-api)). Для стенда, который уже +> хоть раз запускался, используйте API выше. ## Управление площадками (проберами) @@ -207,12 +243,28 @@ curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool Явного отдельного флага "включить/выключить" нет — самого списка `sites` достаточно. -Чтобы добавить площадку: добавьте `site_id` + `index` (1, 2 или 3 — см. -ограничение ниже) в `sites` конфига control-api и разверните на площадке -`prober` с тем же `site_id`. Чтобы отключить конкретную площадку — -уберите соответствующую запись из `sites` и перезапустите control-api; +Чтобы добавить площадку (без перезапуска control-api) — назначьте +`site_id` на один из трёх слотов (`index` 1, 2 или 3 — см. ограничение +ниже) через API: +```bash +curl -s -X PUT http://:8080/api/v1/admin/config/sites/1 \ + -d '{"site_id": "site-1"}' +``` +и разверните на площадке `prober` с тем же `site_id`. Чтобы отключить +конкретную площадку — освободите слот: +```bash +curl -s -X DELETE http://:8080/api/v1/admin/config/sites/1 +``` процесс `prober` на ней можно не останавливать (он просто перестанет -получать назначения). +получать назначения — `POST /api/v1/probers/register` для отвязанного +`site_id` начнёт отвечать `400`). Текущее распределение слотов — `GET +/api/v1/admin/config/sites`. + +> Правки через `sites[]` в `control-api.yaml` тоже поддерживаются, но +> только как bootstrap пустой базы данных при самом первом старте — как +> только в БД есть хотя бы одна площадка, YAML для этой секции +> игнорируется при всех последующих рестартах. Для стенда, который уже +> хоть раз запускался, используйте API выше. > Важно: количество *возможных* слотов площадок жёстко зашито в схему БД > (`Site1Complete`/`Site2Complete`/`Site3Complete`) — не более **трёх**, @@ -221,23 +273,104 @@ curl -s http://:8080/api/v1/admin/validators | python3 -m json.tool > поддерживаемый сценарий; *больше* трёх потребует доработки схемы > данных, одной правкой конфига не обойтись. +## Управление целями проверки + +Набор egress-целей (`targets`) и типов проверок (`check_types`, +привязывающих тип — `https`/`icmp`/`ssh` — к одной или нескольким группам +целей) управляется через API так же, как валидаторы и площадки — +изменения подхватываются немедленно, следующим же назначением от +оркестратора, без перезапуска. + +Посмотреть текущий набор: +```bash +curl -s http://:8080/api/v1/admin/config/targets | python3 -m json.tool +curl -s http://:8080/api/v1/admin/config/check-types | python3 -m json.tool +``` + +Создать/заменить группу целей и включить тип проверки, ссылающийся на +неё: +```bash +curl -s -X PUT http://:8080/api/v1/admin/config/targets/web \ + -d '{"targets": ["https://hub.docker.com", "https://github.com"]}' + +curl -s -X PUT http://:8080/api/v1/admin/config/check-types/ssh \ + -d '{"enabled": true, "targets": ["web"]}' +``` + +Отключить тип проверки, не удаляя его (значения целей сохраняются): +```bash +curl -s -X PUT http://:8080/api/v1/admin/config/check-types/ssh \ + -d '{"enabled": false, "targets": ["web"]}' +``` + +Удалить группу целей можно только если на неё не ссылается ни один +`check_type` (иначе — `409`); удалить сам `check_type` можно в любой +момент (`DELETE /api/v1/admin/config/check-types/{name}`). + +**Важное следствие принятого компромисса**: если конфигурация меняется +ровно в момент, когда чей-то IP уже находится в `checking` (self-check +уже пройден, проверки уже назначены агенту), агрегация этой конкретной +попытки посчитает уже изменённую конфигурацию, а не ту, что была на +момент выдачи задания. На практике это узкое окно в несколько секунд; +деградирует безопасно — через `aggregation.missing_counts_as_fail` худший +исход для одной попытки — `partial` вместо `pass`, самоисправляется на +следующей попытке (в том числе через принудительный повтор, см. ниже). + +> Правки через `targets`/`check_types` в `control-api.yaml` тоже +> поддерживаются, но только как bootstrap пустой базы данных при самом +> первом старте — см. примечание в разделах выше. + ## Повторная проверка адреса Если адрес завершился с `failed` или `partial`, а вы хотите перепроверить -его ещё раз (например, после устранения блокировки на стороне сети): -на данный момент нет отдельного API-метода "перезапустить проверку". -Самый простой путь: -1. Убедитесь, что адрес не находится в активном состоянии (`checking` - и т.п.) — то есть уже `done`/`failed`. -2. Временно уберите и снова добавьте адрес в список `ip_addresses` - (либо просто пересоздайте запись в БД вручную, если это единичный - случай и у вас есть доступ к SQLite) и перезапустите `control-api`. +его ещё раз (например, после устранения блокировки на стороне сети) — +отправьте его тем же методом, что используется для постановки новых +адресов в очередь: -Поскольку сидирование очереди идёт по уникальности `ip_address` -(конфликт по уже существующей записи просто игнорируется), самый чистый -способ гарантированно перепроверить конкретный адрес — обратиться к -администратору БД (см. следующий раздел) либо дождаться штатной -доработки API под повторные проверки. +```bash +curl -s -X POST http://:8080/api/v1/admin/ips \ + -d '{"addresses": ["203.0.113.10"]}' +``` + +```json +{"added": [], "requeued": ["203.0.113.10"], "reordered": [], "skipped_in_progress": []} +``` + +Адрес в `requeued` означает, что он был в терминальном состоянии +(`done`/`failed`) и его перезапустили: `AttemptNumber` увеличился, +`RetryCount` обнулён, предыдущий `OverallResult` сброшен, адрес снова +`queued` и будет обработан на общих основаниях. Никакой особой обработки +для уже проверенных адресов не требуется — тот же вызов безопасно +принимает список из новых и уже проверенных адресов одновременно; +единственное, что метод не сделает — не запустит вторую параллельную +проверку адреса, который прямо сейчас уже проверяется (такой адрес +вернётся в `skipped_in_progress`, см. +[«Добавление новых IP в очередь»](#добавление-новых-ip-в-очередь)). + +## Принудительная остановка проверки + +Если проверка адреса зависла дольше ожидаемого либо просто больше не +актуальна, не дожидайтесь истечения `checking_window_seconds` — +остановите её сразу: + +```bash +curl -s -X POST http://:8080/api/v1/admin/ips/203.0.113.10/cancel +``` + +Работает из любого состояния, кроме уже терминального (`done`/`failed` +вернут `409` — отменять нечего). Если на момент отмены был привязан +Floating IP — он отвязывается (best-effort, как и при обычном завершении +проверки), владевший валидатор освобождается и снова становится `idle`. +Итог записывается как `OverallResult: "cancelled"` (в `State: "failed"`), +и виден в истории адреса наравне с обычными результатами: + +```bash +curl -s http://:8080/api/v1/admin/ips/203.0.113.10 | python3 -m json.tool +``` + +Чтобы позже всё же проверить этот адрес — используйте +[«Повторную проверку адреса»](#повторная-проверка-адреса) выше, она +одинаково работает и для отменённых, и для обычно завершённых адресов. ## Частые проблемы и что с ними делать diff --git a/internal/config/config.go b/internal/config/config.go index 7c792a3..9962804 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -277,6 +277,55 @@ func LoadProber(path string) (*Prober, error) { return &c, nil } +// ---- admin-dashboard ---- + +// AdminDashboard is the config for the 4th binary, cmd/admin-dashboard — a +// stateless, server-rendered web UI over control-api's /api/v1/admin/* +// HTTP API (see docs/DASHBOARD.md). It never talks to the database. +type AdminDashboard struct { + Server ServerConfig `yaml:"server"` + ControlAPI DashboardControlAPIConfig `yaml:"control_api"` + Overview DashboardOverviewConfig `yaml:"overview"` +} + +type DashboardControlAPIConfig struct { + BaseURL string `yaml:"base_url"` + TimeoutSeconds int `yaml:"timeout_seconds"` +} + +// DashboardOverviewConfig configures the overview page's "текущая +// проверка" / "последние N завершённых" summary — see +// docs/DASHBOARD.md#текущая-и-последняя-завершённая-проверка. Both are +// computed fresh on every request from control-api's existing +// status/queue endpoints; there is no persisted "run"/"batch" concept. +type DashboardOverviewConfig struct { + LastCompletedCount int `yaml:"last_completed_count"` + PollIntervalSeconds int `yaml:"poll_interval_seconds"` +} + +func LoadAdminDashboard(path string) (*AdminDashboard, error) { + var c AdminDashboard + if err := loadYAML(path, &c); err != nil { + return nil, err + } + if c.Server.ListenAddr == "" { + c.Server.ListenAddr = ":8090" + } + if c.ControlAPI.BaseURL == "" { + return nil, fmt.Errorf("control_api.base_url is required") + } + if c.ControlAPI.TimeoutSeconds == 0 { + c.ControlAPI.TimeoutSeconds = 10 + } + if c.Overview.LastCompletedCount == 0 { + c.Overview.LastCompletedCount = 20 + } + if c.Overview.PollIntervalSeconds == 0 { + c.Overview.PollIntervalSeconds = 5 + } + return &c, nil +} + func loadYAML(path string, out interface{}) error { data, err := os.ReadFile(path) if err != nil { diff --git a/internal/dashboard/client.go b/internal/dashboard/client.go new file mode 100644 index 0000000..eb8bbf8 --- /dev/null +++ b/internal/dashboard/client.go @@ -0,0 +1,181 @@ +package dashboard + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "time" +) + +// apiErr is returned by every client method for a non-2xx response from +// control-api, or for a transport-level failure (control-api unreachable/ +// timeout, Status==0). Handlers use it (via errors.As) to render an error +// banner with the right status/severity instead of a raw 500 — see +// writeErrorBanner in render.go. +type apiErr struct { + Status int + Message string +} + +func (e *apiErr) Error() string { + if e.Status == 0 { + return fmt.Sprintf("control-api недоступен: %s", e.Message) + } + return fmt.Sprintf("control-api: %d %s", e.Status, e.Message) +} + +// client is a thin, dashboard-local JSON client for control-api's +// /api/v1/admin/* surface. It intentionally doesn't reuse +// internal/apiclient.Client (shared by validator-agent/prober): that +// client's Do only returns an opaque error, with no way to recover the +// HTTP status code — which the dashboard needs to render 400 vs 404 vs 409 +// vs 5xx differently. Duplicating ~30 lines here avoids widening +// apiclient's contract (and its blast radius on the other two binaries) +// for a need only this package has. +type client struct { + baseURL string + http *http.Client +} + +func newClient(baseURL string, timeout time.Duration) *client { + return &client{baseURL: baseURL, http: &http.Client{Timeout: timeout}} +} + +func (c *client) do(ctx context.Context, method, path string, body, out interface{}) error { + var reader io.Reader + if body != nil { + b, err := json.Marshal(body) + if err != nil { + return fmt.Errorf("marshal request: %w", err) + } + reader = bytes.NewReader(b) + } + req, err := http.NewRequestWithContext(ctx, method, c.baseURL+path, reader) + if err != nil { + return fmt.Errorf("build request: %w", err) + } + if body != nil { + req.Header.Set("Content-Type", "application/json") + } + + resp, err := c.http.Do(req) + if err != nil { + return &apiErr{Status: 0, Message: err.Error()} + } + defer resp.Body.Close() + + respBody, _ := io.ReadAll(resp.Body) + if resp.StatusCode >= 300 { + msg := string(respBody) + var er errorResponse + if json.Unmarshal(respBody, &er) == nil && er.Error != "" { + msg = er.Error + } + return &apiErr{Status: resp.StatusCode, Message: msg} + } + if out != nil && len(respBody) > 0 { + if err := json.Unmarshal(respBody, out); err != nil { + return fmt.Errorf("decode response from %s %s: %w", method, path, err) + } + } + return nil +} + +func (c *client) Status(ctx context.Context) (statusResponse, error) { + var out statusResponse + err := c.do(ctx, http.MethodGet, "/api/v1/admin/status", nil, &out) + return out, err +} + +func (c *client) ListIPs(ctx context.Context) ([]ipQueueItem, error) { + var out []ipQueueItem + err := c.do(ctx, http.MethodGet, "/api/v1/admin/ips", nil, &out) + return out, err +} + +func (c *client) GetIP(ctx context.Context, ip string) (ipDetailResponse, error) { + var out ipDetailResponse + err := c.do(ctx, http.MethodGet, "/api/v1/admin/ips/"+url.PathEscape(ip), nil, &out) + return out, err +} + +// SubmitIPs is the single entry point for both adding new addresses and +// forcing a recheck of already-finished ones — see docs/API.md. +func (c *client) SubmitIPs(ctx context.Context, addresses []string) (submitIPsResponse, error) { + var out submitIPsResponse + err := c.do(ctx, http.MethodPost, "/api/v1/admin/ips", map[string][]string{"addresses": addresses}, &out) + return out, err +} + +func (c *client) CancelIP(ctx context.Context, ip string) error { + return c.do(ctx, http.MethodPost, "/api/v1/admin/ips/"+url.PathEscape(ip)+"/cancel", nil, nil) +} + +func (c *client) ListValidators(ctx context.Context) ([]validatorDTO, error) { + var out []validatorDTO + err := c.do(ctx, http.MethodGet, "/api/v1/admin/config/validators", nil, &out) + return out, err +} + +func (c *client) CreateValidator(ctx context.Context, id, osPortID string) error { + body := map[string]string{"validator_id": id, "os_port_id": osPortID} + return c.do(ctx, http.MethodPost, "/api/v1/admin/config/validators", body, nil) +} + +func (c *client) UpdateValidator(ctx context.Context, id, osPortID string) error { + body := map[string]string{"os_port_id": osPortID} + return c.do(ctx, http.MethodPut, "/api/v1/admin/config/validators/"+url.PathEscape(id), body, nil) +} + +func (c *client) DeleteValidator(ctx context.Context, id string) error { + return c.do(ctx, http.MethodDelete, "/api/v1/admin/config/validators/"+url.PathEscape(id), nil, nil) +} + +func (c *client) ListSites(ctx context.Context) ([]siteDTO, error) { + var out []siteDTO + err := c.do(ctx, http.MethodGet, "/api/v1/admin/config/sites", nil, &out) + return out, err +} + +func (c *client) PutSite(ctx context.Context, index int, siteID string) error { + body := map[string]string{"site_id": siteID} + return c.do(ctx, http.MethodPut, fmt.Sprintf("/api/v1/admin/config/sites/%d", index), body, nil) +} + +func (c *client) DeleteSite(ctx context.Context, index int) error { + return c.do(ctx, http.MethodDelete, fmt.Sprintf("/api/v1/admin/config/sites/%d", index), nil, nil) +} + +func (c *client) ListTargetGroups(ctx context.Context) ([]targetGroupDTO, error) { + var out []targetGroupDTO + err := c.do(ctx, http.MethodGet, "/api/v1/admin/config/targets", nil, &out) + return out, err +} + +func (c *client) PutTargetGroup(ctx context.Context, name string, targets []string) error { + body := map[string][]string{"targets": targets} + return c.do(ctx, http.MethodPut, "/api/v1/admin/config/targets/"+url.PathEscape(name), body, nil) +} + +func (c *client) DeleteTargetGroup(ctx context.Context, name string) error { + return c.do(ctx, http.MethodDelete, "/api/v1/admin/config/targets/"+url.PathEscape(name), nil, nil) +} + +func (c *client) ListCheckTypes(ctx context.Context) ([]checkTypeDTO, error) { + var out []checkTypeDTO + err := c.do(ctx, http.MethodGet, "/api/v1/admin/config/check-types", nil, &out) + return out, err +} + +func (c *client) PutCheckType(ctx context.Context, name string, enabled bool, targets []string) error { + body := map[string]interface{}{"enabled": enabled, "targets": targets} + return c.do(ctx, http.MethodPut, "/api/v1/admin/config/check-types/"+url.PathEscape(name), body, nil) +} + +func (c *client) DeleteCheckType(ctx context.Context, name string) error { + return c.do(ctx, http.MethodDelete, "/api/v1/admin/config/check-types/"+url.PathEscape(name), nil, nil) +} diff --git a/internal/dashboard/dashboard_test.go b/internal/dashboard/dashboard_test.go new file mode 100644 index 0000000..4903d8a --- /dev/null +++ b/internal/dashboard/dashboard_test.go @@ -0,0 +1,358 @@ +package dashboard + +import ( + "encoding/json" + "fmt" + "io" + "log/slog" + "net/http" + "net/http/httptest" + "net/url" + "os" + "strings" + "sync" + "testing" + "time" +) + +// fakeControlAPI is a minimal in-memory stand-in for control-api's +// /api/v1/admin/* surface, serving the exact JSON shapes the dashboard's +// client.go expects (see dto.go). It's intentionally simple — enough to +// exercise the dashboard's rendering logic and mutation flows, not a +// re-implementation of control-api's own business rules (that's already +// covered by internal/httpapi's own tests). +type fakeControlAPI struct { + mu sync.Mutex + ips []ipQueueItem + validators []validatorDTO + sites map[int]string + groups map[string][]string + checkTypes map[string]checkTypeDTO +} + +func newFakeControlAPI(t *testing.T) (*fakeControlAPI, string) { + t.Helper() + f := &fakeControlAPI{ + sites: map[int]string{}, + groups: map[string][]string{}, + checkTypes: map[string]checkTypeDTO{}, + } + ts := httptest.NewServer(f.handler()) + t.Cleanup(ts.Close) + return f, ts.URL +} + +func writeJSON(w http.ResponseWriter, status int, v interface{}) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + _ = json.NewEncoder(w).Encode(v) +} + +func writeAPIErr(w http.ResponseWriter, status int, msg string) { + writeJSON(w, status, errorResponse{Error: msg}) +} + +func (f *fakeControlAPI) handler() http.Handler { + mux := http.NewServeMux() + + mux.HandleFunc("GET /api/v1/admin/status", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + byState := map[string]int{} + for _, ip := range f.ips { + byState[ip.State]++ + } + writeJSON(w, http.StatusOK, statusResponse{TotalIPs: len(f.ips), IPsByState: byState, TotalValidators: len(f.validators)}) + }) + + mux.HandleFunc("GET /api/v1/admin/ips", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + writeJSON(w, http.StatusOK, f.ips) + }) + + mux.HandleFunc("GET /api/v1/admin/ips/{ip}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + addr := r.PathValue("ip") + for _, ip := range f.ips { + if ip.IPAddress == addr { + writeJSON(w, http.StatusOK, ipDetailResponse{IP: ip, Checks: []check{}, Events: []event{}}) + return + } + } + writeAPIErr(w, http.StatusNotFound, "unknown ip: "+addr) + }) + + mux.HandleFunc("POST /api/v1/admin/ips", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var req struct { + Addresses []string `json:"addresses"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + if len(req.Addresses) == 0 { + writeAPIErr(w, http.StatusBadRequest, "addresses must not be empty") + return + } + resp := submitIPsResponse{} + for _, addr := range req.Addresses { + idx := f.findIP(addr) + if idx < 0 { + now := time.Now() + f.ips = append(f.ips, ipQueueItem{IPAddress: addr, State: "queued", CreatedAt: now, UpdatedAt: now}) + resp.Added = append(resp.Added, addr) + continue + } + switch f.ips[idx].State { + case "done", "failed": + f.ips[idx].State = "queued" + f.ips[idx].AttemptNumber++ + f.ips[idx].OverallResult = "" + resp.Requeued = append(resp.Requeued, addr) + case "queued": + resp.Reordered = append(resp.Reordered, addr) + default: + resp.SkippedInProgress = append(resp.SkippedInProgress, addr) + } + } + writeJSON(w, http.StatusOK, resp) + }) + + mux.HandleFunc("POST /api/v1/admin/ips/{ip}/cancel", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + addr := r.PathValue("ip") + idx := f.findIP(addr) + if idx < 0 { + writeAPIErr(w, http.StatusNotFound, "unknown ip: "+addr) + return + } + if f.ips[idx].State == "done" || f.ips[idx].State == "failed" { + writeAPIErr(w, http.StatusConflict, "already finished") + return + } + f.ips[idx].State = "failed" + f.ips[idx].OverallResult = "cancelled" + now := time.Now() + f.ips[idx].AggregatedAt = &now + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + }) + + mux.HandleFunc("GET /api/v1/admin/config/validators", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + writeJSON(w, http.StatusOK, f.validators) + }) + mux.HandleFunc("POST /api/v1/admin/config/validators", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var req validatorDTO + _ = json.NewDecoder(r.Body).Decode(&req) + for _, v := range f.validators { + if v.ValidatorID == req.ValidatorID { + writeAPIErr(w, http.StatusConflict, "already exists") + return + } + } + req.State = "idle" + f.validators = append(f.validators, req) + writeJSON(w, http.StatusCreated, req) + }) + mux.HandleFunc("PUT /api/v1/admin/config/validators/{id}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + id := r.PathValue("id") + var req struct { + OSPortID string `json:"os_port_id"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + for i, v := range f.validators { + if v.ValidatorID == id { + f.validators[i].OSPortID = req.OSPortID + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + return + } + } + writeAPIErr(w, http.StatusNotFound, "unknown validator") + }) + mux.HandleFunc("DELETE /api/v1/admin/config/validators/{id}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + id := r.PathValue("id") + for i, v := range f.validators { + if v.ValidatorID == id { + f.validators = append(f.validators[:i], f.validators[i+1:]...) + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + return + } + } + writeAPIErr(w, http.StatusNotFound, "unknown validator") + }) + + mux.HandleFunc("GET /api/v1/admin/config/sites", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var out []siteDTO + for idx, id := range f.sites { + out = append(out, siteDTO{Index: idx, SiteID: id}) + } + writeJSON(w, http.StatusOK, out) + }) + mux.HandleFunc("PUT /api/v1/admin/config/sites/{index}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var idx int + fmt.Sscanf(r.PathValue("index"), "%d", &idx) + var req struct { + SiteID string `json:"site_id"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + for existingIdx, id := range f.sites { + if id == req.SiteID && existingIdx != idx { + writeAPIErr(w, http.StatusConflict, "site_id already assigned to another slot") + return + } + } + f.sites[idx] = req.SiteID + writeJSON(w, http.StatusOK, siteDTO{Index: idx, SiteID: req.SiteID}) + }) + mux.HandleFunc("DELETE /api/v1/admin/config/sites/{index}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var idx int + fmt.Sscanf(r.PathValue("index"), "%d", &idx) + delete(f.sites, idx) + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + }) + + mux.HandleFunc("GET /api/v1/admin/config/targets", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var out []targetGroupDTO + for name, targets := range f.groups { + out = append(out, targetGroupDTO{Name: name, Targets: targets}) + } + writeJSON(w, http.StatusOK, out) + }) + mux.HandleFunc("PUT /api/v1/admin/config/targets/{group}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + name := r.PathValue("group") + var req struct { + Targets []string `json:"targets"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + if len(req.Targets) == 0 { + writeAPIErr(w, http.StatusBadRequest, "targets must not be empty") + return + } + f.groups[name] = req.Targets + writeJSON(w, http.StatusOK, targetGroupDTO{Name: name, Targets: req.Targets}) + }) + mux.HandleFunc("DELETE /api/v1/admin/config/targets/{group}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + name := r.PathValue("group") + for _, ct := range f.checkTypes { + for _, g := range ct.Targets { + if g == name { + writeAPIErr(w, http.StatusConflict, "in use by check type "+ct.Name) + return + } + } + } + delete(f.groups, name) + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + }) + + mux.HandleFunc("GET /api/v1/admin/config/check-types", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + var out []checkTypeDTO + for _, ct := range f.checkTypes { + out = append(out, ct) + } + writeJSON(w, http.StatusOK, out) + }) + mux.HandleFunc("PUT /api/v1/admin/config/check-types/{name}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + name := r.PathValue("name") + var req struct { + Enabled bool `json:"enabled"` + Targets []string `json:"targets"` + } + _ = json.NewDecoder(r.Body).Decode(&req) + for _, g := range req.Targets { + if _, ok := f.groups[g]; !ok { + writeAPIErr(w, http.StatusBadRequest, "unknown target group "+g) + return + } + } + ct := checkTypeDTO{Name: name, Enabled: req.Enabled, Targets: req.Targets} + f.checkTypes[name] = ct + writeJSON(w, http.StatusOK, ct) + }) + mux.HandleFunc("DELETE /api/v1/admin/config/check-types/{name}", func(w http.ResponseWriter, r *http.Request) { + f.mu.Lock() + defer f.mu.Unlock() + delete(f.checkTypes, r.PathValue("name")) + writeJSON(w, http.StatusOK, map[string]bool{"ok": true}) + }) + + return mux +} + +func (f *fakeControlAPI) findIP(addr string) int { + for i, ip := range f.ips { + if ip.IPAddress == addr { + return i + } + } + return -1 +} + +func newTestServer(t *testing.T, caURL string) *httptest.Server { + t.Helper() + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) + srv, err := New(Config{ + ControlAPIBaseURL: caURL, + ControlAPITimeout: 5 * time.Second, + LastCompletedCount: 20, + OverviewPollIntervalS: 5, + }, log) + if err != nil { + t.Fatalf("new dashboard server: %v", err) + } + ts := httptest.NewServer(srv.Handler()) + t.Cleanup(ts.Close) + return ts +} + +func get(t *testing.T, ts *httptest.Server, path string) string { + t.Helper() + resp, err := http.Get(ts.URL + path) + if err != nil { + t.Fatalf("GET %s: %v", path, err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + return string(body) +} + +func postForm(t *testing.T, ts *httptest.Server, method, path string, form url.Values) string { + t.Helper() + req, err := http.NewRequest(method, ts.URL+path, strings.NewReader(form.Encode())) + if err != nil { + t.Fatalf("build request: %v", err) + } + req.Header.Set("Content-Type", "application/x-www-form-urlencoded") + resp, err := ts.Client().Do(req) + if err != nil { + t.Fatalf("%s %s: %v", method, path, err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + return string(body) +} diff --git a/internal/dashboard/dto.go b/internal/dashboard/dto.go new file mode 100644 index 0000000..2cb3c02 --- /dev/null +++ b/internal/dashboard/dto.go @@ -0,0 +1,116 @@ +package dashboard + +import "time" + +// Wire shapes for control-api's /api/v1/admin/* surface, defined locally +// rather than importing internal/httpapi's (unexported) DTOs or +// internal/db's models — the same "each binary owns the wire shapes it +// needs" pattern already used by internal/probercore and internal/agentcore. +// +// The read-only list/detail endpoints (ipQueueItem, validator, check, +// event below) mirror internal/db model structs field-for-field: those +// endpoints marshal Go structs with no json tags, so encoding/json matches +// fields by name (case-insensitively) with no tags needed here either. +// Everything else mirrors internal/httpapi/dto_admin.go's snake_case tags. + +type statusResponse struct { + TotalIPs int `json:"total_ips"` + IPsByState map[string]int `json:"ips_by_state"` + TotalValidators int `json:"total_validators"` +} + +type ipQueueItem struct { + ID int64 + IPAddress string + Sequence int + State string + OwnerValidatorID *string + FIPID string + AttemptNumber int + RetryCount int + LeaseExpiresAt *time.Time + EgressComplete bool + Site1Complete bool + Site2Complete bool + Site3Complete bool + OverallResult string + AssignedAt *time.Time + AggregatedAt *time.Time + FIPReleasedAt *time.Time + CreatedAt time.Time + UpdatedAt time.Time +} + +type check struct { + ID int64 + IPID int64 + IPAddress string + AttemptNumber int + ValidatorID string + Source string + CheckType string + Target string + Success bool + LatencyMS int64 + Detail string + CheckedAt time.Time +} + +type event struct { + ID int64 + SourceType string + SourceID string + IPID *int64 + EventType string + Payload string + OccurredAt time.Time +} + +type ipDetailResponse struct { + IP ipQueueItem `json:"ip"` + Checks []check `json:"checks"` + Events []event `json:"events"` +} + +type validator struct { + ValidatorID string + Hostname string + OSPortID string + State string + CurrentIPID *int64 + AgentVersion string + LastHeartbeatAt *time.Time +} + +type submitIPsResponse struct { + Added []string `json:"added"` + Requeued []string `json:"requeued"` + Reordered []string `json:"reordered"` + SkippedInProgress []string `json:"skipped_in_progress"` +} + +type validatorDTO struct { + ValidatorID string `json:"validator_id"` + OSPortID string `json:"os_port_id"` + State string `json:"state"` +} + +type siteDTO struct { + Index int `json:"index"` + SiteID string `json:"site_id"` +} + +type targetGroupDTO struct { + Name string `json:"name"` + Targets []string `json:"targets"` +} + +type checkTypeDTO struct { + Name string `json:"name"` + Enabled bool `json:"enabled"` + Targets []string `json:"targets"` +} + +type errorResponse struct { + Error string `json:"error"` +} diff --git a/internal/dashboard/embed.go b/internal/dashboard/embed.go new file mode 100644 index 0000000..7738b49 --- /dev/null +++ b/internal/dashboard/embed.go @@ -0,0 +1,23 @@ +package dashboard + +import ( + "embed" + "io/fs" +) + +//go:embed templates/*.html +var templateFS embed.FS + +//go:embed static +var staticFS embed.FS + +// staticSubFS re-roots staticFS so "static/dashboard.css" is served as +// "/dashboard.css" under the /static/ route prefix, instead of +// "/static/static/dashboard.css". +func staticSubFS() fs.FS { + sub, err := fs.Sub(staticFS, "static") + if err != nil { + panic(err) // embedded at compile time; a bad path here can't happen at runtime + } + return sub +} diff --git a/internal/dashboard/handlers_checktypes.go b/internal/dashboard/handlers_checktypes.go new file mode 100644 index 0000000..4200d33 --- /dev/null +++ b/internal/dashboard/handlers_checktypes.go @@ -0,0 +1,77 @@ +package dashboard + +import ( + "fmt" + "net/http" +) + +type checkTypesPageData struct { + PageData + Items []checkTypeDTO + Groups []string +} + +func (s *Server) loadCheckTypesPage(r *http.Request) (checkTypesPageData, error) { + items, err := s.CA.ListCheckTypes(r.Context()) + if err != nil { + return checkTypesPageData{}, err + } + groupDTOs, err := s.CA.ListTargetGroups(r.Context()) + if err != nil { + return checkTypesPageData{Items: items}, err + } + groups := make([]string, len(groupDTOs)) + for i, g := range groupDTOs { + groups[i] = g.Name + } + return checkTypesPageData{Items: items, Groups: groups}, nil +} + +func (s *Server) handleCheckTypesPage(w http.ResponseWriter, r *http.Request) { + data, err := s.loadCheckTypesPage(r) + data.ActiveNav = "check-types" + data.Banner = bannerFor(err) + s.renderPage(w, "checktypes_page", data) +} + +func (s *Server) renderCheckTypesTable(w http.ResponseWriter, r *http.Request, actionErr error) { + items, listErr := s.CA.ListCheckTypes(r.Context()) + if actionErr == nil { + actionErr = listErr + } + s.renderFragment(w, "checktypes_table", checkTypesPageData{Items: items}, actionErr) +} + +func (s *Server) handleCheckTypeCreate(w http.ResponseWriter, r *http.Request) { + if err := r.ParseForm(); err != nil { + s.renderCheckTypesTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + name := r.PostFormValue("name") + if name == "" { + s.renderCheckTypesTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "имя типа проверки обязательно"}) + return + } + enabled := r.PostFormValue("enabled") != "" + targets := r.PostForm["targets"] + err := s.CA.PutCheckType(r.Context(), name, enabled, targets) + s.renderCheckTypesTable(w, r, err) +} + +func (s *Server) handleCheckTypeUpdate(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := r.ParseForm(); err != nil { + s.renderCheckTypesTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + enabled := r.PostFormValue("enabled") == "1" + targets := splitList(r.PostFormValue("targets")) + err := s.CA.PutCheckType(r.Context(), name, enabled, targets) + s.renderCheckTypesTable(w, r, err) +} + +func (s *Server) handleCheckTypeDelete(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + err := s.CA.DeleteCheckType(r.Context(), name) + s.renderCheckTypesTable(w, r, err) +} diff --git a/internal/dashboard/handlers_ips.go b/internal/dashboard/handlers_ips.go new file mode 100644 index 0000000..79dad69 --- /dev/null +++ b/internal/dashboard/handlers_ips.go @@ -0,0 +1,71 @@ +package dashboard + +import ( + "fmt" + "net/http" +) + +type ipsPageData struct { + PageData + Items []ipQueueItem +} + +type ipDetailData struct { + PageData + Detail ipDetailResponse +} + +func (s *Server) handleIPsPage(w http.ResponseWriter, r *http.Request) { + items, err := s.CA.ListIPs(r.Context()) + data := ipsPageData{Items: items} + data.ActiveNav = "ips" + data.Banner = bannerFor(err) + s.renderPage(w, "ips_page", data) +} + +func (s *Server) handleIPDetail(w http.ResponseWriter, r *http.Request) { + ip := r.PathValue("ip") + detail, err := s.CA.GetIP(r.Context(), ip) + data := ipDetailData{Detail: detail} + data.ActiveNav = "ips" + data.Banner = bannerFor(err) + s.renderPage(w, "ip_detail_page", data) +} + +// renderIPsTable re-fetches the current queue and renders the ips_table +// fragment, tagging actionErr (if any) on the shared error banner. Called +// after every mutating /ips/* request so the table always reflects true +// current state regardless of whether the mutation itself succeeded. +func (s *Server) renderIPsTable(w http.ResponseWriter, r *http.Request, actionErr error) { + items, listErr := s.CA.ListIPs(r.Context()) + if actionErr == nil { + actionErr = listErr + } + s.renderFragment(w, "ips_table", ipsPageData{Items: items}, actionErr) +} + +func (s *Server) handleIPsSubmit(w http.ResponseWriter, r *http.Request) { + if err := r.ParseForm(); err != nil { + s.renderIPsTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + addresses := splitList(r.PostFormValue("addresses")) + if len(addresses) == 0 { + s.renderIPsTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "укажите хотя бы один адрес"}) + return + } + _, err := s.CA.SubmitIPs(r.Context(), addresses) + s.renderIPsTable(w, r, err) +} + +func (s *Server) handleIPRecheck(w http.ResponseWriter, r *http.Request) { + ip := r.PathValue("ip") + _, err := s.CA.SubmitIPs(r.Context(), []string{ip}) + s.renderIPsTable(w, r, err) +} + +func (s *Server) handleIPCancel(w http.ResponseWriter, r *http.Request) { + ip := r.PathValue("ip") + err := s.CA.CancelIP(r.Context(), ip) + s.renderIPsTable(w, r, err) +} diff --git a/internal/dashboard/handlers_overview.go b/internal/dashboard/handlers_overview.go new file mode 100644 index 0000000..d46e92d --- /dev/null +++ b/internal/dashboard/handlers_overview.go @@ -0,0 +1,91 @@ +package dashboard + +import ( + "net/http" + "sort" +) + +type overviewData struct { + PageData + Status statusResponse + CurrentItems []ipQueueItem + LastCompleted []ipQueueItem + Breakdown map[string]int + LastN int + PollSeconds int +} + +func (s *Server) loadOverview(r *http.Request) (overviewData, error) { + ctx := r.Context() + status, err := s.CA.Status(ctx) + if err != nil { + return overviewData{}, err + } + ips, err := s.CA.ListIPs(ctx) + if err != nil { + return overviewData{}, err + } + last := lastCompleted(ips, s.Cfg.LastCompletedCount) + return overviewData{ + Status: status, + CurrentItems: currentlyChecking(ips), + LastCompleted: last, + Breakdown: resultBreakdown(last), + LastN: s.Cfg.LastCompletedCount, + PollSeconds: s.Cfg.OverviewPollIntervalS, + }, nil +} + +func (s *Server) handleOverview(w http.ResponseWriter, r *http.Request) { + data, err := s.loadOverview(r) + data.ActiveNav = "overview" + data.Banner = bannerFor(err) + s.renderPage(w, "overview_page", data) +} + +func (s *Server) handleOverviewFragment(w http.ResponseWriter, r *http.Request) { + data, err := s.loadOverview(r) + s.renderFragment(w, "overview_fragment", data, err) +} + +// currentlyChecking is every IP not yet in a terminal state, ordered by +// queue position — the "текущая проверка" live snapshot. No backend +// concept of a "run" exists; this is computed fresh on every request. +func currentlyChecking(ips []ipQueueItem) []ipQueueItem { + var out []ipQueueItem + for _, ip := range ips { + if ip.State != "done" && ip.State != "failed" { + out = append(out, ip) + } + } + sort.Slice(out, func(i, j int) bool { return out[i].Sequence < out[j].Sequence }) + return out +} + +// lastCompleted returns the n most recently completed (done/failed) IPs by +// AggregatedAt descending — the "последняя завершённая проверка" summary +// window. This is an operational definition, not a real "batch": resubmit +// n if the window size needs tuning (overview.last_completed_count). +func lastCompleted(ips []ipQueueItem, n int) []ipQueueItem { + var done []ipQueueItem + for _, ip := range ips { + if (ip.State == "done" || ip.State == "failed") && ip.AggregatedAt != nil { + done = append(done, ip) + } + } + sort.Slice(done, func(i, j int) bool { return done[i].AggregatedAt.After(*done[j].AggregatedAt) }) + if len(done) > n { + done = done[:n] + } + return done +} + +// resultBreakdown counts OverallResult values across exactly the given +// items (normally the output of lastCompleted) — pass/partial/fail/cancelled. +func resultBreakdown(items []ipQueueItem) map[string]int { + out := map[string]int{"pass": 0, "partial": 0, "fail": 0, "cancelled": 0} + for _, ip := range items { + out[ip.OverallResult]++ + } + return out +} diff --git a/internal/dashboard/handlers_sites.go b/internal/dashboard/handlers_sites.go new file mode 100644 index 0000000..72dac28 --- /dev/null +++ b/internal/dashboard/handlers_sites.go @@ -0,0 +1,66 @@ +package dashboard + +import ( + "fmt" + "net/http" + "strconv" +) + +type sitesPageData struct { + PageData + Items []siteDTO // always exactly 3 entries, index 1..3, SiteID=="" for an empty slot +} + +// fillSlots pads control-api's response (which only lists assigned slots) +// out to all three fixed slots, so the table always renders 3 rows. +func fillSlots(sites []siteDTO) []siteDTO { + byIndex := make(map[int]string, len(sites)) + for _, s := range sites { + byIndex[s.Index] = s.SiteID + } + out := make([]siteDTO, 3) + for i := 0; i < 3; i++ { + out[i] = siteDTO{Index: i + 1, SiteID: byIndex[i+1]} + } + return out +} + +func (s *Server) handleSitesPage(w http.ResponseWriter, r *http.Request) { + items, err := s.CA.ListSites(r.Context()) + data := sitesPageData{Items: fillSlots(items)} + data.ActiveNav = "sites" + data.Banner = bannerFor(err) + s.renderPage(w, "sites_page", data) +} + +func (s *Server) renderSitesTable(w http.ResponseWriter, r *http.Request, actionErr error) { + items, listErr := s.CA.ListSites(r.Context()) + if actionErr == nil { + actionErr = listErr + } + s.renderFragment(w, "sites_table", sitesPageData{Items: fillSlots(items)}, actionErr) +} + +func (s *Server) handleSitePut(w http.ResponseWriter, r *http.Request) { + index, convErr := strconv.Atoi(r.PathValue("index")) + if convErr != nil { + s.renderSitesTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "index должен быть числом"}) + return + } + if err := r.ParseForm(); err != nil { + s.renderSitesTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + err := s.CA.PutSite(r.Context(), index, r.PostFormValue("site_id")) + s.renderSitesTable(w, r, err) +} + +func (s *Server) handleSiteDelete(w http.ResponseWriter, r *http.Request) { + index, convErr := strconv.Atoi(r.PathValue("index")) + if convErr != nil { + s.renderSitesTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "index должен быть числом"}) + return + } + err := s.CA.DeleteSite(r.Context(), index) + s.renderSitesTable(w, r, err) +} diff --git a/internal/dashboard/handlers_targets.go b/internal/dashboard/handlers_targets.go new file mode 100644 index 0000000..119ef1b --- /dev/null +++ b/internal/dashboard/handlers_targets.go @@ -0,0 +1,59 @@ +package dashboard + +import ( + "fmt" + "net/http" +) + +type targetsPageData struct { + PageData + Items []targetGroupDTO +} + +func (s *Server) handleTargetsPage(w http.ResponseWriter, r *http.Request) { + items, err := s.CA.ListTargetGroups(r.Context()) + data := targetsPageData{Items: items} + data.ActiveNav = "targets" + data.Banner = bannerFor(err) + s.renderPage(w, "targets_page", data) +} + +func (s *Server) renderTargetsTable(w http.ResponseWriter, r *http.Request, actionErr error) { + items, listErr := s.CA.ListTargetGroups(r.Context()) + if actionErr == nil { + actionErr = listErr + } + s.renderFragment(w, "targets_table", targetsPageData{Items: items}, actionErr) +} + +func (s *Server) handleTargetCreate(w http.ResponseWriter, r *http.Request) { + if err := r.ParseForm(); err != nil { + s.renderTargetsTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + name := r.PostFormValue("name") + targets := splitList(r.PostFormValue("targets")) + if name == "" { + s.renderTargetsTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "имя группы обязательно"}) + return + } + err := s.CA.PutTargetGroup(r.Context(), name, targets) + s.renderTargetsTable(w, r, err) +} + +func (s *Server) handleTargetUpdate(w http.ResponseWriter, r *http.Request) { + group := r.PathValue("group") + if err := r.ParseForm(); err != nil { + s.renderTargetsTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + targets := splitList(r.PostFormValue("targets")) + err := s.CA.PutTargetGroup(r.Context(), group, targets) + s.renderTargetsTable(w, r, err) +} + +func (s *Server) handleTargetDelete(w http.ResponseWriter, r *http.Request) { + group := r.PathValue("group") + err := s.CA.DeleteTargetGroup(r.Context(), group) + s.renderTargetsTable(w, r, err) +} diff --git a/internal/dashboard/handlers_test.go b/internal/dashboard/handlers_test.go new file mode 100644 index 0000000..8e97cdb --- /dev/null +++ b/internal/dashboard/handlers_test.go @@ -0,0 +1,196 @@ +package dashboard + +import ( + "strings" + "testing" + "time" +) + +func TestOverviewFragment(t *testing.T) { + fake, caURL := newFakeControlAPI(t) + now := time.Now() + older := now.Add(-time.Hour) + fake.validators = []validatorDTO{{ValidatorID: "v1", State: "idle"}} + fake.ips = []ipQueueItem{ + {IPAddress: "1.1.1.1", State: "checking", Sequence: 1, AssignedAt: &now, UpdatedAt: now, CreatedAt: now}, + {IPAddress: "2.2.2.2", State: "done", OverallResult: "pass", AggregatedAt: &now, UpdatedAt: now, CreatedAt: now}, + {IPAddress: "3.3.3.3", State: "failed", OverallResult: "cancelled", AggregatedAt: &older, UpdatedAt: now, CreatedAt: now}, + } + + ts := newTestServer(t, caURL) + body := get(t, ts, "/overview/fragment") + + if !strings.Contains(body, "1.1.1.1") { + t.Fatalf("expected in-progress ip 1.1.1.1 in current-checking section, got:\n%s", body) + } + if strings.Contains(body, `href="/ips/1.1.1.1"`) { + // currently-checking rows link to /ips detail too; that's fine, just + // make sure the done one shows up in the "last completed" table. + } + if !strings.Contains(body, "2.2.2.2") || !strings.Contains(body, "3.3.3.3") { + t.Fatalf("expected both completed ips in last-completed section, got:\n%s", body) + } + if !strings.Contains(body, "pass: 1") || !strings.Contains(body, "cancelled: 1") { + t.Fatalf("expected breakdown pass:1 cancelled:1, got:\n%s", body) + } +} + +func TestIPsSubmitAddAndForceRecheck(t *testing.T) { + fake, caURL := newFakeControlAPI(t) + now := time.Now() + fake.ips = []ipQueueItem{ + {IPAddress: "9.9.9.9", State: "done", OverallResult: "pass", AttemptNumber: 1, AggregatedAt: &now, UpdatedAt: now, CreatedAt: now}, + } + ts := newTestServer(t, caURL) + + // Add a brand-new address. + body := postForm(t, ts, "POST", "/ips", map[string][]string{"addresses": {"1.2.3.4"}}) + if !strings.Contains(body, "1.2.3.4") { + t.Fatalf("expected new address in re-rendered table, got:\n%s", body) + } + + // Force-recheck the already-done address via the same endpoint. + body = postForm(t, ts, "POST", "/ips", map[string][]string{"addresses": {"9.9.9.9"}}) + if !strings.Contains(body, "9.9.9.9") { + t.Fatalf("expected requeued address in table, got:\n%s", body) + } + if fake.ips[0].State != "queued" || fake.ips[0].AttemptNumber != 2 { + t.Fatalf("expected fake control-api state requeued with attempt_number=2, got %+v", fake.ips[0]) + } + + // Empty submission is a client error, surfaced via the banner, not a 500. + body = postForm(t, ts, "POST", "/ips", map[string][]string{"addresses": {""}}) + if !strings.Contains(body, "error-banner-client") { + t.Fatalf("expected client error banner for empty submission, got:\n%s", body) + } +} + +func TestIPRecheckSkippedInProgress(t *testing.T) { + fake, caURL := newFakeControlAPI(t) + now := time.Now() + fake.ips = []ipQueueItem{{IPAddress: "5.5.5.5", State: "checking", UpdatedAt: now, CreatedAt: now}} + ts := newTestServer(t, caURL) + + body := postForm(t, ts, "POST", "/ips/5.5.5.5/recheck", nil) + // SubmitIPs succeeds (200) but the address itself lands in + // skipped_in_progress — the dashboard shouldn't claim success without + // qualification; the table must still show it untouched (still checking). + if !strings.Contains(body, "5.5.5.5") { + t.Fatalf("expected 5.5.5.5 still listed, got:\n%s", body) + } + if fake.ips[0].State != "checking" { + t.Fatalf("expected state untouched (checking), got %s", fake.ips[0].State) + } +} + +func TestIPCancel(t *testing.T) { + fake, caURL := newFakeControlAPI(t) + now := time.Now() + fake.ips = []ipQueueItem{{IPAddress: "7.7.7.7", State: "checking", UpdatedAt: now, CreatedAt: now}} + ts := newTestServer(t, caURL) + + body := postForm(t, ts, "POST", "/ips/7.7.7.7/cancel", nil) + if !strings.Contains(body, "cancelled") { + t.Fatalf("expected cancelled badge after cancel, got:\n%s", body) + } + if fake.ips[0].OverallResult != "cancelled" { + t.Fatalf("expected fake control-api state cancelled, got %+v", fake.ips[0]) + } + + // Cancelling an already-terminal ip is a 409 from control-api, surfaced + // as a banner, not a crash. + body = postForm(t, ts, "POST", "/ips/7.7.7.7/cancel", nil) + if !strings.Contains(body, "error-banner") { + t.Fatalf("expected error banner for double-cancel, got:\n%s", body) + } +} + +func TestValidatorsCRUD(t *testing.T) { + _, caURL := newFakeControlAPI(t) + ts := newTestServer(t, caURL) + + body := postForm(t, ts, "POST", "/validators", map[string][]string{"validator_id": {"val-1"}, "os_port_id": {"port-1"}}) + if !strings.Contains(body, "val-1") || !strings.Contains(body, "port-1") { + t.Fatalf("expected new validator in table, got:\n%s", body) + } + + // Conflict: creating the same validator_id again. + body = postForm(t, ts, "POST", "/validators", map[string][]string{"validator_id": {"val-1"}, "os_port_id": {"port-x"}}) + if !strings.Contains(body, "error-banner-client") { + t.Fatalf("expected client error banner on duplicate create, got:\n%s", body) + } + + body = postForm(t, ts, "PUT", "/validators/val-1", map[string][]string{"os_port_id": {"port-2"}}) + if !strings.Contains(body, "port-2") { + t.Fatalf("expected updated port in table, got:\n%s", body) + } + + body = postForm(t, ts, "DELETE", "/validators/val-1", nil) + if strings.Contains(body, "val-1") { + t.Fatalf("expected validator removed after delete, got:\n%s", body) + } +} + +func TestSitesPutAndConflict(t *testing.T) { + _, caURL := newFakeControlAPI(t) + ts := newTestServer(t, caURL) + + body := postForm(t, ts, "PUT", "/sites/1", map[string][]string{"site_id": {"site-1"}}) + if !strings.Contains(body, "site-1") { + t.Fatalf("expected site-1 assigned to slot 1, got:\n%s", body) + } + + // Slot page always renders exactly 3 rows, including empty ones — count + // the per-slot PUT forms rather than raw (the thead row has one too). + page := get(t, ts, "/sites") + if strings.Count(page, `hx-put="/sites/`) != 3 { + t.Fatalf("expected exactly 3 site slot rows, got:\n%s", page) + } + + body = postForm(t, ts, "PUT", "/sites/2", map[string][]string{"site_id": {"site-1"}}) + if !strings.Contains(body, "error-banner-client") { + t.Fatalf("expected conflict banner assigning a taken site_id to another slot, got:\n%s", body) + } +} + +func TestTargetsAndCheckTypesRoundTrip(t *testing.T) { + _, caURL := newFakeControlAPI(t) + ts := newTestServer(t, caURL) + + body := postForm(t, ts, "POST", "/targets", map[string][]string{"name": {"web"}, "targets": {"https://a.test\nhttps://b.test"}}) + if !strings.Contains(body, "web") { + t.Fatalf("expected new target group in table, got:\n%s", body) + } + + body = postForm(t, ts, "POST", "/check-types", map[string][]string{"name": {"https"}, "enabled": {"1"}, "targets": {"web"}}) + if !strings.Contains(body, "https") { + t.Fatalf("expected new check type in table, got:\n%s", body) + } + + // Can't delete a target group still referenced by a check type. + body = postForm(t, ts, "DELETE", "/targets/web", nil) + if !strings.Contains(body, "error-banner-client") { + t.Fatalf("expected conflict banner deleting in-use target group, got:\n%s", body) + } + + body = postForm(t, ts, "DELETE", "/check-types/https", nil) + if strings.Contains(body, "https") { + t.Fatalf("expected check type removed, got:\n%s", body) + } + + body = postForm(t, ts, "DELETE", "/targets/web", nil) + if strings.Contains(body, "web") { + t.Fatalf("expected target group removable once unreferenced, got:\n%s", body) + } +} + +func TestControlAPIUnreachable(t *testing.T) { + // Point the dashboard at an address nothing listens on, rather than a + // closed httptest.Server, to get a deterministic connection-refused + // transport failure. + ts := newTestServer(t, "http://127.0.0.1:1") + body := get(t, ts, "/overview") + if !strings.Contains(body, "error-banner-server") { + t.Fatalf("expected server/transport error banner when control-api is unreachable, got:\n%s", body) + } +} diff --git a/internal/dashboard/handlers_validators.go b/internal/dashboard/handlers_validators.go new file mode 100644 index 0000000..2144901 --- /dev/null +++ b/internal/dashboard/handlers_validators.go @@ -0,0 +1,58 @@ +package dashboard + +import ( + "fmt" + "net/http" +) + +type validatorsPageData struct { + PageData + Items []validatorDTO +} + +func (s *Server) handleValidatorsPage(w http.ResponseWriter, r *http.Request) { + items, err := s.CA.ListValidators(r.Context()) + data := validatorsPageData{Items: items} + data.ActiveNav = "validators" + data.Banner = bannerFor(err) + s.renderPage(w, "validators_page", data) +} + +func (s *Server) renderValidatorsTable(w http.ResponseWriter, r *http.Request, actionErr error) { + items, listErr := s.CA.ListValidators(r.Context()) + if actionErr == nil { + actionErr = listErr + } + s.renderFragment(w, "validators_table", validatorsPageData{Items: items}, actionErr) +} + +func (s *Server) handleValidatorCreate(w http.ResponseWriter, r *http.Request) { + if err := r.ParseForm(); err != nil { + s.renderValidatorsTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + id := r.PostFormValue("validator_id") + osPortID := r.PostFormValue("os_port_id") + if id == "" { + s.renderValidatorsTable(w, r, &apiErr{Status: http.StatusBadRequest, Message: "validator_id обязателен"}) + return + } + err := s.CA.CreateValidator(r.Context(), id, osPortID) + s.renderValidatorsTable(w, r, err) +} + +func (s *Server) handleValidatorUpdate(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + if err := r.ParseForm(); err != nil { + s.renderValidatorsTable(w, r, fmt.Errorf("invalid form: %w", err)) + return + } + err := s.CA.UpdateValidator(r.Context(), id, r.PostFormValue("os_port_id")) + s.renderValidatorsTable(w, r, err) +} + +func (s *Server) handleValidatorDelete(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + err := s.CA.DeleteValidator(r.Context(), id) + s.renderValidatorsTable(w, r, err) +} diff --git a/internal/dashboard/render.go b/internal/dashboard/render.go new file mode 100644 index 0000000..16f5bcc --- /dev/null +++ b/internal/dashboard/render.go @@ -0,0 +1,154 @@ +package dashboard + +import ( + "errors" + "html/template" + "net/http" + "strings" + "time" +) + +// Badge is the {class, label} pair rendered as a colored pill — see +// static/dashboard.css for the .badge-* classes. +type Badge struct{ Class, Label string } + +func ipBadge(state, result string) Badge { + switch state { + case "done", "failed": + switch result { + case "pass": + return Badge{"badge badge-pass", "pass"} + case "partial": + return Badge{"badge badge-partial", "partial"} + case "cancelled": + return Badge{"badge badge-cancelled", "cancelled"} + default: + return Badge{"badge badge-fail", "fail"} + } + case "queued": + return Badge{"badge badge-queued", "queued"} + default: + return Badge{"badge badge-inprogress", state} + } +} + +func validatorBadge(state string) Badge { + switch state { + case "idle": + return Badge{"badge badge-idle", "idle"} + case "unreachable": + return Badge{"badge badge-unreachable", "unreachable"} + case "unregistered": + return Badge{"badge badge-unregistered", "unregistered"} + default: + return Badge{"badge badge-inprogress", state} // assigned / checking + } +} + +// fmtTime accepts time.Time, *time.Time, or nil and renders a local +// timestamp or an em-dash placeholder — covers both the value-typed +// (CreatedAt, CheckedAt, ...) and pointer-typed (AssignedAt, +// AggregatedAt, ...) timestamp fields in the DTOs without two template +// funcs. +func fmtTime(v interface{}) string { + switch t := v.(type) { + case time.Time: + if t.IsZero() { + return "—" + } + return t.Local().Format("2006-01-02 15:04:05") + case *time.Time: + if t == nil { + return "—" + } + return t.Local().Format("2006-01-02 15:04:05") + default: + return "—" + } +} + +func derefStr(s *string) string { + if s == nil || *s == "" { + return "—" + } + return *s +} + +var funcMap = template.FuncMap{ + "ipBadge": ipBadge, + "validatorBadge": validatorBadge, + "fmtTime": fmtTime, + "deref": derefStr, + "join": strings.Join, +} + +func parseTemplates() (*template.Template, error) { + return template.New("").Funcs(funcMap).ParseFS(templateFS, "templates/*.html") +} + +// bannerData drives the shared error-banner partial (see templates/ +// layout.html's "banner_inner" block and templates/error_banner.html). +// Client distinguishes a business-logic 4xx (rendered in yellow) from a +// server-side 5xx or transport failure such as control-api being +// unreachable (rendered in red). +type bannerData struct { + Message string + Client bool +} + +// PageData is embedded (anonymously) by every full-page template's data +// struct so {{.Banner}}/{{.ActiveNav}} resolve via Go's promoted-field +// rule in both the page template and the shared partials (page_header, +// banner_inner) it includes. ActiveNav drives the nav bar's aria-current. +type PageData struct { + Banner bannerData + ActiveNav string +} + +func bannerFor(err error) bannerData { + if err == nil { + return bannerData{} + } + var ae *apiErr + if errors.As(err, &ae) { + return bannerData{Message: ae.Error(), Client: ae.Status >= 400 && ae.Status < 500} + } + return bannerData{Message: err.Error(), Client: false} +} + +// renderPage renders a full page (extends "layout") for a plain GET +// navigation. actionErr (if any — e.g. the primary control-api call for +// this page failed) is surfaced via the embedded PageData.Banner, which +// the caller must have already set via bannerFor. +func (s *Server) renderPage(w http.ResponseWriter, name string, data interface{}) { + w.Header().Set("Content-Type", "text/html; charset=utf-8") + if err := s.tmpl.ExecuteTemplate(w, name, data); err != nil { + s.Log.Error("render page", "template", name, "err", err) + } +} + +// renderFragment renders an htmx swap target (name) plus, appended to the +// same response body, an out-of-band update of the shared #error-banner +// (see templates/error_banner.html) reflecting actionErr — nil clears any +// previously shown banner, matching htmx's id-based morph. +// +// The response status is always 200, deliberately: htmx's default +// response-handling config only processes swaps (including OOB swaps) for +// 2xx responses, and this dashboard needs the banner to render on every +// response regardless of whether the underlying control-api call +// succeeded. Because every mutating handler re-fetches the authoritative +// list from control-api after attempting its mutation (success or not) +// before calling renderFragment, the primary content is always a true +// reflection of current state — success/failure is communicated by the +// banner text and by whether the content actually changed, not by HTTP +// status. +func (s *Server) renderFragment(w http.ResponseWriter, name string, data interface{}, actionErr error) { + w.Header().Set("Content-Type", "text/html; charset=utf-8") + if err := s.tmpl.ExecuteTemplate(w, name, data); err != nil { + s.Log.Error("render fragment", "template", name, "err", err) + return + } + if err := s.tmpl.ExecuteTemplate(w, "error_banner", bannerFor(actionErr)); err != nil { + s.Log.Error("render error banner", "err", err) + } +} diff --git a/internal/dashboard/routes.go b/internal/dashboard/routes.go new file mode 100644 index 0000000..54c42ab --- /dev/null +++ b/internal/dashboard/routes.go @@ -0,0 +1,41 @@ +package dashboard + +import "net/http" + +func (s *Server) routes(mux *http.ServeMux) { + mux.HandleFunc("GET /{$}", s.handleIndex) + + mux.HandleFunc("GET /overview", s.handleOverview) + mux.HandleFunc("GET /overview/fragment", s.handleOverviewFragment) + + mux.HandleFunc("GET /ips", s.handleIPsPage) + mux.HandleFunc("GET /ips/{ip}", s.handleIPDetail) + mux.HandleFunc("POST /ips", s.handleIPsSubmit) + mux.HandleFunc("POST /ips/{ip}/recheck", s.handleIPRecheck) + mux.HandleFunc("POST /ips/{ip}/cancel", s.handleIPCancel) + + mux.HandleFunc("GET /validators", s.handleValidatorsPage) + mux.HandleFunc("POST /validators", s.handleValidatorCreate) + mux.HandleFunc("PUT /validators/{id}", s.handleValidatorUpdate) + mux.HandleFunc("DELETE /validators/{id}", s.handleValidatorDelete) + + mux.HandleFunc("GET /sites", s.handleSitesPage) + mux.HandleFunc("PUT /sites/{index}", s.handleSitePut) + mux.HandleFunc("DELETE /sites/{index}", s.handleSiteDelete) + + mux.HandleFunc("GET /targets", s.handleTargetsPage) + mux.HandleFunc("POST /targets", s.handleTargetCreate) + mux.HandleFunc("PUT /targets/{group}", s.handleTargetUpdate) + mux.HandleFunc("DELETE /targets/{group}", s.handleTargetDelete) + + mux.HandleFunc("GET /check-types", s.handleCheckTypesPage) + mux.HandleFunc("POST /check-types", s.handleCheckTypeCreate) + mux.HandleFunc("PUT /check-types/{name}", s.handleCheckTypeUpdate) + mux.HandleFunc("DELETE /check-types/{name}", s.handleCheckTypeDelete) + + mux.Handle("GET /static/", http.StripPrefix("/static/", http.FileServerFS(staticSubFS()))) +} + +func (s *Server) handleIndex(w http.ResponseWriter, r *http.Request) { + http.Redirect(w, r, "/overview", http.StatusFound) +} diff --git a/internal/dashboard/server.go b/internal/dashboard/server.go new file mode 100644 index 0000000..cfc5088 --- /dev/null +++ b/internal/dashboard/server.go @@ -0,0 +1,54 @@ +// Package dashboard implements the admin dashboard's HTTP surface: a +// server-rendered (html/template + htmx + Alpine.js) web UI giving full +// coverage of control-api's /api/v1/admin/* API. It never talks to +// internal/db directly and has no state of its own — every page and +// fragment is computed fresh, on each request, from control-api's API via +// the client in client.go. +package dashboard + +import ( + "html/template" + "log/slog" + "net/http" + "time" +) + +type Config struct { + ControlAPIBaseURL string + ControlAPITimeout time.Duration + LastCompletedCount int + OverviewPollIntervalS int +} + +type Server struct { + CA *client + Cfg Config + tmpl *template.Template + Log *slog.Logger +} + +func New(cfg Config, log *slog.Logger) (*Server, error) { + tmpl, err := parseTemplates() + if err != nil { + return nil, err + } + return &Server{ + CA: newClient(cfg.ControlAPIBaseURL, cfg.ControlAPITimeout), + Cfg: cfg, + tmpl: tmpl, + Log: log, + }, nil +} + +func (s *Server) Handler() http.Handler { + mux := http.NewServeMux() + s.routes(mux) + return loggingMiddleware(s.Log, mux) +} + +func loggingMiddleware(log *slog.Logger, next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + next.ServeHTTP(w, r) + log.Debug("request", "method", r.Method, "path", r.URL.Path) + }) +} diff --git a/internal/dashboard/static/dashboard.css b/internal/dashboard/static/dashboard.css new file mode 100644 index 0000000..1b76529 --- /dev/null +++ b/internal/dashboard/static/dashboard.css @@ -0,0 +1,59 @@ +/* Small additions Pico's classless build doesn't cover: color-coded state + badges, the error banner, and a compact stat-tile grid for the overview + page. Everything else relies on Pico's classless defaults. */ + +.badge { + display: inline-block; + padding: 0.15rem 0.55rem; + border-radius: 999px; + font-size: 0.8rem; + font-weight: 600; + line-height: 1.4; + white-space: nowrap; +} +.badge-queued { background: #e2e8f0; color: #334155; } +.badge-inprogress { background: #dbeafe; color: #1d4ed8; } +.badge-pass { background: #dcfce7; color: #15803d; } +.badge-partial { background: #ffedd5; color: #c2410c; } +.badge-fail { background: #fee2e2; color: #b91c1c; } +.badge-cancelled { background: #ede9fe; color: #6d28d9; } +.badge-idle { background: #dcfce7; color: #15803d; } +.badge-unreachable { background: #fee2e2; color: #b91c1c; } +.badge-unregistered { background: #e2e8f0; color: #334155; } + +.stat-grid { + display: grid; + grid-template-columns: repeat(auto-fit, minmax(9rem, 1fr)); + gap: 1rem; + margin-bottom: 1.5rem; +} +.stat-tile { + border: 1px solid var(--pico-muted-border-color, #ddd); + border-radius: 0.5rem; + padding: 0.75rem 1rem; + text-align: center; +} +.stat-tile .value { display: block; font-size: 1.6rem; font-weight: 700; } +.stat-tile .label { display: block; font-size: 0.8rem; opacity: 0.75; } + +#error-banner:empty { display: none; } +.error-banner { + padding: 0.6rem 1rem; + margin-bottom: 1rem; + border-radius: 0.4rem; + font-size: 0.9rem; +} +.error-banner-client { background: #fef9c3; color: #854d0e; border: 1px solid #fde68a; } +.error-banner-server { background: #fee2e2; color: #991b1b; border: 1px solid #fecaca; } + +nav.dashboard-nav ul { display: flex; gap: 1.25rem; list-style: none; padding: 0; margin: 0 0 1.5rem 0; } +nav.dashboard-nav a[aria-current="page"] { font-weight: 700; text-decoration: underline; } + +.actions-cell { white-space: nowrap; } +.actions-cell button { margin-right: 0.35rem; font-size: 0.85rem; padding: 0.25rem 0.6rem; } +.actions-cell button.secondary { --pico-primary-background: #64748b; } +.actions-cell button.contrast { --pico-primary-background: #b91c1c; } + +table.compact th, table.compact td { padding: 0.4rem 0.6rem; } + +.muted { opacity: 0.65; font-size: 0.85rem; } diff --git a/internal/dashboard/static/vendor/VENDOR.md b/internal/dashboard/static/vendor/VENDOR.md new file mode 100644 index 0000000..68da71a --- /dev/null +++ b/internal/dashboard/static/vendor/VENDOR.md @@ -0,0 +1,15 @@ +# Vendored frontend assets + +Downloaded once and committed as-is — the dashboard binary must work fully +offline, with no runtime CDN dependency (see `docs/PLAN_ADMIN_DASHBOARD.md`). +Do not "update" these files casually; bump deliberately, in their own +commit, if a newer version is actually needed. + +| File | Source | Version | Fetched | +|---|---|---|---| +| `htmx.min.js` | https://unpkg.com/htmx.org@2.0.10/dist/htmx.min.js | 2.0.10 | 2026-08-23 | +| `alpine.min.js` | https://unpkg.com/alpinejs@3.16.2/dist/cdn.min.js | 3.16.2 | 2026-08-23 | +| `pico.classless.min.css` | https://unpkg.com/@picocss/pico@2.1.1/css/pico.classless.min.css | 2.1.1 | 2026-08-23 | + +Licenses: htmx (BSD 2-Clause), Alpine.js (MIT), Pico CSS (MIT) — all +permit vendoring/redistribution unmodified. diff --git a/internal/dashboard/static/vendor/alpine.min.js b/internal/dashboard/static/vendor/alpine.min.js new file mode 100644 index 0000000..2de9655 --- /dev/null +++ b/internal/dashboard/static/vendor/alpine.min.js @@ -0,0 +1,21 @@ +(()=>{var _t=!1,gt=!1,$=[],xt=-1,je=!1,yt=!1;function mr(e){ri(e)}function _r(){yt=!0}function gr(){yt=!1,yr()}function ri(e){$.includes(e)||($.push(e),e._x_schedulerPriority!==void 0&&(je=!0)),yr()}function xr(e){let t=$.indexOf(e);t!==-1&&t>xt&&$.splice(t,1)}function yr(){if(!gt&&!_t){if(yt)return;_t=!0,queueMicrotask(ni)}}function ni(){_t=!1,gt=!0;for(let e=0;e<$.length;e++)je&&ii(e),$[e](),xt=e;$.length=0,xt=-1,je=!1,gt=!1}function ii(e){let t=new Map,r=$.slice(e).sort((n,i)=>oi(n,i,t));for(let n=0;ne.effect(t,{scheduler:r=>{bt?mr(r):r()}}),Et=e.raw}function vt(e){P=e}function vr(e){let t=()=>{};return[(n,i)=>{let o=i?.priority==="structural"?si++:void 0,s=P(n);return o!==void 0&&s!==void 0&&(s._x_schedulerPriority={el:e,order:o}),e._x_effects||(e._x_effects=new Set,e._x_runEffects=()=>{e._x_effects.forEach(a=>a())}),e._x_effects.add(s),t=()=>{s!==void 0&&(e._x_effects.delete(s),B(s))},s},()=>{t()}]}function Fe(e,t){let r=!0,n,i,o=P(()=>{let s=e(),a=JSON.stringify(s);if(!r&&(typeof s=="object"||s!==n)){let c=typeof n=="object"?JSON.parse(i):n;queueMicrotask(()=>{t(s,c)})}n=s,i=a,r=!1});return()=>B(o)}async function wr(e){_r();try{await e(),await Promise.resolve()}finally{gr()}}var Sr=[],Ar=[],Or=[];function Tr(e){Or.push(e)}function ce(e,t){typeof t=="function"?(e._x_cleanups||(e._x_cleanups=[]),e._x_cleanups.push(t)):(t=e,Ar.push(t))}function $e(e){Sr.push(e)}function Be(e,t,r){e._x_attributeCleanups||(e._x_attributeCleanups={}),e._x_attributeCleanups[t]||(e._x_attributeCleanups[t]=[]),e._x_attributeCleanups[t].push(r)}function wt(e,t){e._x_attributeCleanups&&Object.entries(e._x_attributeCleanups).forEach(([r,n])=>{(t===void 0||t.includes(r))&&(n.forEach(i=>i()),delete e._x_attributeCleanups[r])})}function Nr(e){for(e._x_effects?.forEach(xr);e._x_cleanups?.length;)e._x_cleanups.pop()()}var St=new MutationObserver(Nt),At=!1;function Ee(){St.observe(document,{subtree:!0,childList:!0,attributes:!0,attributeOldValue:!0}),At=!0}function Ot(){ai(),St.disconnect(),At=!1}var be=[];function ai(){let e=St.takeRecords();be.push(()=>e.length>0&&Nt(e));let t=be.length;queueMicrotask(()=>{if(be.length===t)for(;be.length>0;)be.shift()()})}function h(e){if(!At)return e();Ot();let t=e();return Ee(),t}var Tt=!1,Ve=[];function Cr(){Tt=!0}function Rr(){Tt=!1,Nt(Ve),Ve=[]}function Nt(e){if(Tt){Ve=Ve.concat(e);return}let t=[],r=new Set,n=new Map,i=new Map;for(let o=0;o{s.nodeType===1&&s._x_marker&&r.add(s)}),e[o].addedNodes.forEach(s=>{if(s.nodeType===1){if(r.has(s)){r.delete(s);return}s._x_marker||t.push(s)}})),e[o].type==="attributes")){let s=e[o].target,a=e[o].attributeName,c=e[o].oldValue,l=()=>{n.has(s)||n.set(s,[]),n.get(s).push({name:a,value:s.getAttribute(a)})},f=()=>{i.has(s)||i.set(s,[]),i.get(s).push(a)};s.hasAttribute(a)&&c===null?l():s.hasAttribute(a)?(f(),l()):f()}i.forEach((o,s)=>{wt(s,o)}),n.forEach((o,s)=>{Sr.forEach(a=>a(s,o))});for(let o of r)t.some(s=>s.contains(o))||Ar.forEach(s=>s(o));for(let o of t)o.isConnected&&Or.forEach(s=>s(o));t=null,r=null,n=null,i=null}function He(e){return L(H(e))}function k(e,t,r){return e._x_dataStack=[t,...H(r||e)],()=>{e._x_dataStack=e._x_dataStack.filter(n=>n!==t)}}function H(e){return e._x_dataStack?e._x_dataStack:typeof ShadowRoot=="function"&&e instanceof ShadowRoot?H(e.host):e.parentNode?H(e.parentNode):[]}function L(e){return new Proxy({objects:e},ci)}function Dr(e,t){return e===null||e===Object.prototype?null:Object.prototype.hasOwnProperty.call(e,t)?e:Dr(Object.getPrototypeOf(e),t)}var ci={ownKeys({objects:e}){return Array.from(new Set(e.flatMap(t=>Object.keys(t))))},has({objects:e},t){return t==Symbol.unscopables?!1:e.some(r=>Object.prototype.hasOwnProperty.call(r,t)||Reflect.has(r,t))},get({objects:e},t,r){return t=="toJSON"?li:Reflect.get(e.find(n=>Reflect.has(n,t))||{},t,r)},set({objects:e},t,r,n){let i;for(let s of e)if(i=Dr(s,t),i)break;i||(i=e[e.length-1]);let o=Object.getOwnPropertyDescriptor(i,t);return o?.set&&o?.get?o.set.call(n,r)||!0:Reflect.set(i,t,r)}};function li(){return Reflect.ownKeys(this).reduce((t,r)=>(t[r]=Reflect.get(this,r),t),{})}function le(e,t=()=>{}){let r=i=>typeof i=="object"&&!Array.isArray(i)&&i!==null,n=(i,o="")=>{Object.entries(Object.getOwnPropertyDescriptors(i)).forEach(([s,{value:a,enumerable:c}])=>{if(c===!1||a===void 0||typeof a=="object"&&a!==null&&a.__v_skip)return;let l=o===""?s:`${o}.${s}`;typeof a=="object"&&a!==null&&a._x_interceptor?i[s]=a.initialize(e,l,s,t):r(a)&&a!==i&&!(a instanceof Element)&&n(a,l)})};return n(e)}function Ue(e,t=()=>{}){let r={initialValue:void 0,_x_interceptor:!0,initialize(n,i,o,s){return e(this.initialValue,()=>fi(n,i),a=>Ct(n,i,a),i,o,s)}};return t(r),n=>{if(typeof n=="object"&&n!==null&&n._x_interceptor){let i=r.initialize.bind(r);r.initialize=(o,s,a,c)=>{let l=n.initialize(o,s,a,c);return r.initialValue=l,i(o,s,a,c)}}else r.initialValue=n;return r}}function fi(e,t){return t.split(".").reduce((r,n)=>r[n],e)}function Ct(e,t,r){if(typeof t=="string"&&(t=t.split(".")),t.length===1)e[t[0]]=r;else{if(t.length===0)throw error;return e[t[0]]||(e[t[0]]={}),Ct(e[t[0]],t.slice(1),r)}}var Mr={};function b(e,t){Mr[e]=t}function q(e,t){let r=ui(t);return Object.entries(Mr).forEach(([n,i])=>{Object.defineProperty(e,`$${n}`,{get(){return i(t,r)},enumerable:!1})}),e}function ui(e){let[t,r]=Rt(e),n={interceptor:Ue,...t};return ce(e,r),n}function Pr(e,t,r,...n){try{return r(...n)}catch(i){fe(i,e,t)}}function fe(...e){return kr(...e)}var kr=pi;function Lr(e){kr=e}function pi(e,t,r=void 0){e=Object.assign(e??{message:"No error message given."},{el:t,expression:r}),console.warn(`Alpine Expression Error: ${e.message} + +${r?'Expression: "'+r+`" + +`:""}`,t),setTimeout(()=>{throw e},0)}var ue=!0;function ze(e){let t=ue;ue=!1;let r=e();return ue=t,r}function D(e,t,r={}){let n;return x(e,t)(i=>n=i,r),n}function x(...e){return Ir(...e)}var Ir=()=>{};function jr(e){Ir=e}var Fr;function Vr(e){Fr=e}function $r(e,t){let r={};q(r,e);let n=[r,...H(e)],i=typeof t=="function"?di(n,t):mi(n,t,e);return Pr.bind(null,e,t,i)}function di(e,t){return(r=()=>{},{scope:n={},params:i=[],context:o}={})=>{if(!ue){ve(r,t,L([n,...e]),i);return}let s=t.apply(L([n,...e]),i);ve(r,s)}}var Dt={};function hi(e,t){if(Dt[e])return Dt[e];let r=Object.getPrototypeOf(async function(){}).constructor,n=/^[\n\s]*if.*\(.*\)/.test(e.trim())||/^(let|const)\s/.test(e.trim())?`(async()=>{ ${e} })()`:e,o=(()=>{try{let s=new r(["__self","scope"],`with (scope) { __self.result = ${n} }; __self.finished = true; return __self.result;`);return Object.defineProperty(s,"name",{value:`[Alpine] ${e}`}),s}catch(s){return fe(s,t,e),Promise.resolve()}})();return Dt[e]=o,o}function mi(e,t,r){let n=hi(t,r);return(i=()=>{},{scope:o={},params:s=[],context:a}={})=>{n.result=void 0,n.finished=!1;let c=L([o,...e]);if(typeof n=="function"){let l=n.call(a,n,c).catch(f=>fe(f,r,t));n.finished?(ve(i,n.result,c,s,r),n.result=void 0):l.then(f=>{ve(i,f,c,s,r)}).catch(f=>fe(f,r,t)).finally(()=>n.result=void 0)}}}function ve(e,t,r,n,i){if(ue&&typeof t=="function"){let o=t.apply(r,n);o instanceof Promise?o.then(s=>ve(e,s,r,n)).catch(s=>fe(s,i,t)):e(o)}else typeof t=="object"&&t instanceof Promise?t.then(o=>e(o)):e(t)}function Br(...e){return Fr(...e)}function Hr(e,t,r={}){let n={};q(n,e);let i=[n,...H(e)],o=L([r.scope??{},...i]),s=r.params??[];if(t.includes("await")){let a=Object.getPrototypeOf(async function(){}).constructor,c=/^[\n\s]*if.*\(.*\)/.test(t.trim())||/^(let|const)\s/.test(t.trim())?`(async()=>{ ${t} })()`:t;return new a(["scope"],`with (scope) { let __result = ${c}; return __result }`).call(r.context,o)}else{let a=/^[\n\s]*if.*\(.*\)/.test(t.trim())||/^(let|const)\s/.test(t.trim())?`(()=>{ ${t} })()`:t,l=new Function(["scope"],`with (scope) { let __result = ${a}; return __result }`).call(r.context,o);return typeof l=="function"&&ue?l.apply(o,s):l}}var kt="x-";function N(e=""){return kt+e}function Ur(e){kt=e}var We={};function d(e,t){return We[e]=t,{before(r){if(!We[r]){console.warn(String.raw`Cannot find directive \`${r}\`. \`${e}\` will use the default order of execution`);return}let n=X.indexOf(r);X.splice(n>=0?n:X.indexOf("DEFAULT"),0,e)}}}function zr(e){return Object.keys(We).includes(e)}function Se(e,t,r){if(t=Array.from(t),e._x_virtualDirectives){let o=Object.entries(e._x_virtualDirectives).map(([a,c])=>({name:a,value:c})),s=Lt(o);o=o.map(a=>s.find(c=>c.name===a.name)?{name:`x-bind:${a.name}`,value:`"${a.value}"`}:a),t=t.concat(o)}let n={};return t.map(qr((o,s)=>n[o]=s)).filter(Yr).map(gi(n,r)).sort(xi).map(o=>_i(e,o))}function Lt(e){return Array.from(e).map(qr()).filter(t=>!Yr(t))}var Mt=!1,we=new Map,Wr=Symbol();function Kr(e){Mt=!0;let t=Symbol();Wr=t,we.set(t,[]);let r=()=>{for(;we.get(t).length;)we.get(t).shift()();we.delete(t)},n=()=>{Mt=!1,r()};e(r),n()}function Rt(e){let t=[],r=a=>t.push(a),[n,i]=vr(e);return t.push(i),[{Alpine:U,effect:n,cleanup:r,evaluateLater:x.bind(x,e),evaluate:D.bind(D,e)},()=>t.forEach(a=>a())]}function _i(e,t){let r=()=>{},n=We[t.type]||r,[i,o]=Rt(e);Be(e,t.original,o);let s=()=>{e._x_ignore||e._x_ignoreSelf||(n.inline&&n.inline(e,t,i),n=n.bind(n,e,t,i),Mt?we.get(Wr).push(n):n())};return s.runCleanups=o,s}var Ke=(e,t)=>({name:r,value:n})=>(r.startsWith(e)&&(r=r.replace(e,t)),{name:r,value:n}),qe=e=>e;function qr(e=()=>{}){return({name:t,value:r})=>{let{name:n,value:i}=Gr.reduce((o,s)=>s(o),{name:t,value:r});return n!==t&&e(n,t),{name:n,value:i}}}var Gr=[];function pe(e){Gr.push(e)}function Yr({name:e}){return Jr().test(e)}var Jr=()=>new RegExp(`^${kt}([^:^.]+)\\b`);function gi(e,t){return({name:r,value:n})=>{r===n&&(n="");let i=r.match(Jr()),o=r.match(/:([a-zA-Z0-9\-_:]+)/),s=r.match(/\.[^.\]]+(?=[^\]]*$)/g)||[],a=t||e[r]||r;return{type:i?i[1]:null,value:o?o[1]:null,modifiers:s.map(c=>c.replace(".","")),expression:n,original:a}}}var Pt="DEFAULT",X=["ignore","ref","id","data","anchor","bind","init","for","model","modelable","transition","show","if",Pt,"teleport"];function xi(e,t){let r=X.indexOf(e.type)===-1?Pt:e.type,n=X.indexOf(t.type)===-1?Pt:t.type;return X.indexOf(r)-X.indexOf(n)}function Z(e,t,r={},n={}){return e.dispatchEvent(new CustomEvent(t,{detail:r,bubbles:!0,composed:!0,cancelable:!0,...n}))}function I(e,t){if(typeof ShadowRoot=="function"&&e instanceof ShadowRoot){Array.from(e.children).forEach(i=>I(i,t));return}let r=!1;if(t(e,()=>r=!0),r)return;let n=e.firstElementChild;for(;n;)I(n,t,!1),n=n.nextElementSibling}function w(e,...t){console.warn(`Alpine Warning: ${e}`,...t)}var Xr=!1;function Zr(){Xr&&w("Alpine has already been initialized on this page. Calling Alpine.start() more than once can cause problems."),Xr=!0,document.body||w("Unable to initialize. Trying to load Alpine before `` is available. Did you forget to add `defer` in Alpine's ` + +{{end}} + +{{define "page_header"}} +
+ +
+{{end}} + +{{define "banner_inner"}}{{if .Message}}
{{.Message}}
{{end}}{{end}} + +{{define "error_banner"}}
{{template "banner_inner" .}}
{{end}} diff --git a/internal/dashboard/templates/overview.html b/internal/dashboard/templates/overview.html new file mode 100644 index 0000000..5876a6f --- /dev/null +++ b/internal/dashboard/templates/overview.html @@ -0,0 +1,20 @@ +{{define "overview_page"}} + + +{{template "html_head" .}} + +{{template "page_header" .}} +
+
{{template "banner_inner" .Banner}}
+{{template "overview_content" .}} +
+ + +{{end}} + +{{define "overview_content"}} +

Обзор

+
+{{template "overview_fragment" .}} +
+{{end}} diff --git a/internal/dashboard/templates/overview_fragment.html b/internal/dashboard/templates/overview_fragment.html new file mode 100644 index 0000000..ffe815b --- /dev/null +++ b/internal/dashboard/templates/overview_fragment.html @@ -0,0 +1,55 @@ +{{define "overview_fragment"}} +
+
{{.Status.TotalIPs}}всего IP
+
{{.Status.TotalValidators}}валидаторов
+{{range $state, $count := .Status.IPsByState}} +
{{$count}}{{$state}}
+{{end}} +
+ +

Текущая проверка

+{{if .CurrentItems}} + + + +{{range .CurrentItems}} +{{$b := ipBadge .State .OverallResult}} + + + + + + + +{{end}} + +
АдресСостояниеВалидаторПопыткаНазначено
{{.IPAddress}}{{$b.Label}}{{deref .OwnerValidatorID}}{{.AttemptNumber}}{{fmtTime .AssignedAt}}
+{{else}} +

Сейчас нет адресов в обработке.

+{{end}} + +

Последние {{.LastN}} завершённых

+

+pass: {{index .Breakdown "pass"}} · +partial: {{index .Breakdown "partial"}} · +fail: {{index .Breakdown "fail"}} · +cancelled: {{index .Breakdown "cancelled"}} +

+{{if .LastCompleted}} + + + +{{range .LastCompleted}} +{{$b := ipBadge .State .OverallResult}} + + + + + +{{end}} + +
АдресИтогЗавершено
{{.IPAddress}}{{$b.Label}}{{fmtTime .AggregatedAt}}
+{{else}} +

Пока ничего не завершено.

+{{end}} +{{end}} diff --git a/internal/dashboard/templates/sites.html b/internal/dashboard/templates/sites.html new file mode 100644 index 0000000..6ced45b --- /dev/null +++ b/internal/dashboard/templates/sites.html @@ -0,0 +1,47 @@ +{{define "sites_page"}} + + +{{template "html_head" .}} + +{{template "page_header" .}} +
+
{{template "banner_inner" .Banner}}
+{{template "sites_content" .}} +
+ + +{{end}} + +{{define "sites_content"}} +

Площадки

+

Ровно три слота (1, 2, 3) — жёсткое ограничение схемы БД control-api. +Пустой слот — площадка не используется, входящие проверки для неё не ожидаются.

+ +
+{{template "sites_table" .}} +
+{{end}} + +{{define "sites_table"}} + + + +{{range .Items}} + + + + + +{{end}} + +
Слотsite_idДействия
{{.Index}} +
+ + +
+
+{{if .SiteID}} + +{{end}} +
+{{end}} diff --git a/internal/dashboard/templates/targets.html b/internal/dashboard/templates/targets.html new file mode 100644 index 0000000..a9d0ff3 --- /dev/null +++ b/internal/dashboard/templates/targets.html @@ -0,0 +1,53 @@ +{{define "targets_page"}} + + +{{template "html_head" .}} + +{{template "page_header" .}} +
+
{{template "banner_inner" .Banner}}
+{{template "targets_content" .}} +
+ + +{{end}} + +{{define "targets_content"}} +

Цели проверки

+

Группа целей — именованный список URL/адресов, на который затем ссылаются типы проверок.

+ +
+ + + + + +
+ +
+{{template "targets_table" .}} +
+{{end}} + +{{define "targets_table"}} + + + +{{range .Items}} + + + + + +{{end}} + +
ГруппаЦелиДействия
{{.Name}} +
+ + +
+
+ +
+{{if not .Items}}

Групп целей нет.

{{end}} +{{end}} diff --git a/internal/dashboard/templates/validators.html b/internal/dashboard/templates/validators.html new file mode 100644 index 0000000..03c7092 --- /dev/null +++ b/internal/dashboard/templates/validators.html @@ -0,0 +1,54 @@ +{{define "validators_page"}} + + +{{template "html_head" .}} + +{{template "page_header" .}} +
+
{{template "banner_inner" .Banner}}
+{{template "validators_content" .}} +
+ + +{{end}} + +{{define "validators_content"}} +

Валидаторы

+ +
+
+ + + +
+
+ +
+{{template "validators_table" .}} +
+{{end}} + +{{define "validators_table"}} + + + +{{range .Items}} +{{$b := validatorBadge .State}} + + + + + + +{{end}} + +
IDOS Port IDСостояниеДействия
{{.ValidatorID}} +
+ + +
+
{{$b.Label}} + +
+{{if not .Items}}

Валидаторов нет.

{{end}} +{{end}} diff --git a/internal/dashboard/util.go b/internal/dashboard/util.go new file mode 100644 index 0000000..8aad98c --- /dev/null +++ b/internal/dashboard/util.go @@ -0,0 +1,21 @@ +package dashboard + +import "strings" + +// splitList parses a textarea/hidden-input value into a clean list, +// accepting entries separated by newlines and/or commas interchangeably +// (different forms in this package use one or the other) and dropping +// blank entries from stray whitespace. +func splitList(raw string) []string { + fields := strings.FieldsFunc(raw, func(r rune) bool { + return r == '\n' || r == '\r' || r == ',' + }) + out := make([]string, 0, len(fields)) + for _, f := range fields { + f = strings.TrimSpace(f) + if f != "" { + out = append(out, f) + } + } + return out +} diff --git a/internal/db/bootstrap.go b/internal/db/bootstrap.go new file mode 100644 index 0000000..d15cac9 --- /dev/null +++ b/internal/db/bootstrap.go @@ -0,0 +1,104 @@ +package db + +import ( + "context" + "fmt" + + "cloudipvalidator/internal/config" +) + +// BootstrapFromConfig ensures the database's baseline state matches the +// YAML config on a fresh install, then gets out of the way. +// +// - ip_addresses -> SeedQueue: unchanged, always additive. This is a +// separate concept from the runtime admin queue API (SubmitIPs) — every +// control-api startup re-adds any address from cfg.IPAddresses that +// isn't already in the queue. +// - validators / sites / target_groups / check_types: applied from YAML +// only if the corresponding table is currently empty. Once any row +// exists (via this bootstrap or the admin API), YAML for that section +// is ignored on every subsequent startup — the database is the source +// of truth from that point on. This is what makes admin API changes to +// these four entities survive a restart. +func (d *DB) BootstrapFromConfig(ctx context.Context, cfg *config.ControlAPI) error { + if err := d.bootstrapValidators(ctx, cfg.Validators); err != nil { + return fmt.Errorf("bootstrap validators: %w", err) + } + if err := d.bootstrapSites(ctx, cfg.Sites); err != nil { + return fmt.Errorf("bootstrap sites: %w", err) + } + if err := d.bootstrapTargetGroups(ctx, cfg.Targets); err != nil { + return fmt.Errorf("bootstrap target groups: %w", err) + } + if err := d.bootstrapCheckTypes(ctx, cfg.CheckTypes); err != nil { + return fmt.Errorf("bootstrap check types: %w", err) + } + if err := d.SeedQueue(ctx, cfg.IPAddresses); err != nil { + return fmt.Errorf("seed ip queue: %w", err) + } + return nil +} + +func (d *DB) bootstrapValidators(ctx context.Context, validators []config.ValidatorConfig) error { + var count int + if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM validators`).Scan(&count); err != nil { + return err + } + if count > 0 { + return nil + } + for _, v := range validators { + if err := d.RegisterValidator(ctx, v.ValidatorID, "", v.OSPortID, ""); err != nil { + return fmt.Errorf("seed validator %s: %w", v.ValidatorID, err) + } + } + return nil +} + +func (d *DB) bootstrapSites(ctx context.Context, sites []config.SiteConfig) error { + var count int + if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM sites`).Scan(&count); err != nil { + return err + } + if count > 0 { + return nil + } + for _, s := range sites { + if err := d.UpsertSite(ctx, s.Index, s.SiteID); err != nil { + return fmt.Errorf("seed site %s: %w", s.SiteID, err) + } + } + return nil +} + +func (d *DB) bootstrapTargetGroups(ctx context.Context, targets map[string][]string) error { + var count int + if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM target_groups`).Scan(&count); err != nil { + return err + } + if count > 0 { + return nil + } + for name, addrs := range targets { + if err := d.UpsertTargetGroup(ctx, name, addrs); err != nil { + return fmt.Errorf("seed target group %s: %w", name, err) + } + } + return nil +} + +func (d *DB) bootstrapCheckTypes(ctx context.Context, checkTypes []config.CheckTypeConfig) error { + var count int + if err := d.QueryRowContext(ctx, `SELECT COUNT(*) FROM check_types`).Scan(&count); err != nil { + return err + } + if count > 0 { + return nil + } + for _, ct := range checkTypes { + if err := d.UpsertCheckType(ctx, ct.Name, ct.Enabled, ct.Targets); err != nil { + return fmt.Errorf("seed check type %s: %w", ct.Name, err) + } + } + return nil +} diff --git a/internal/db/db.go b/internal/db/db.go index de3eceb..163390c 100644 --- a/internal/db/db.go +++ b/internal/db/db.go @@ -16,6 +16,21 @@ import ( //go:embed migrations/0001_init.sql var initSchema string +//go:embed migrations/0002_dynamic_config.sql +var dynamicConfigSchema string + +// migrations is the ordered list of schema versions. Each entry's SQL is +// applied, in order, for any version greater than the database's current +// PRAGMA user_version — so a fresh database walks the whole list and an +// existing one only picks up what's new. +var migrations = []struct { + version int + sql string +}{ + {1, initSchema}, + {2, dynamicConfigSchema}, +} + type DB struct { *sql.DB } @@ -52,30 +67,36 @@ func Open(ctx context.Context, path string) (*DB, error) { return d, nil } -// migrate applies the embedded schema exactly once, tracked via -// PRAGMA user_version so repeated startups are no-ops. +// migrate applies every pending migration in order, tracked via +// PRAGMA user_version so repeated startups only apply what's new (and a +// fresh database walks the whole list once). func (d *DB) migrate(ctx context.Context) error { var version int if err := d.QueryRowContext(ctx, "PRAGMA user_version").Scan(&version); err != nil { return fmt.Errorf("read user_version: %w", err) } - if version >= 1 { - return nil - } - tx, err := d.BeginTx(ctx, nil) - if err != nil { - return err + for _, m := range migrations { + if m.version <= version { + continue + } + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + if _, err := tx.ExecContext(ctx, m.sql); err != nil { + tx.Rollback() + return fmt.Errorf("apply migration %d: %w", m.version, err) + } + if _, err := tx.ExecContext(ctx, fmt.Sprintf("PRAGMA user_version=%d", m.version)); err != nil { + tx.Rollback() + return fmt.Errorf("set user_version=%d: %w", m.version, err) + } + if err := tx.Commit(); err != nil { + return fmt.Errorf("commit migration %d: %w", m.version, err) + } } - defer tx.Rollback() - - if _, err := tx.ExecContext(ctx, initSchema); err != nil { - return fmt.Errorf("apply 0001_init.sql: %w", err) - } - if _, err := tx.ExecContext(ctx, "PRAGMA user_version=1"); err != nil { - return fmt.Errorf("set user_version: %w", err) - } - return tx.Commit() + return nil } // Now returns the current time truncated to millisecond precision, the diff --git a/internal/db/errors.go b/internal/db/errors.go new file mode 100644 index 0000000..e854dfd --- /dev/null +++ b/internal/db/errors.go @@ -0,0 +1,26 @@ +package db + +import "errors" + +// Sentinel errors returned by the dynamic-config and IP-queue admin +// mutations. Callers (internal/httpapi) map these to HTTP status codes via +// errors.Is rather than defaulting everything to 500. +var ( + // ErrNotFound: the referenced entity does not exist. + ErrNotFound = errors.New("not found") + // ErrConflict: the entity already exists, or a unique slot/key is + // already taken by something else. + ErrConflict = errors.New("conflict") + // ErrBusy: the entity can't be removed because it's currently in use + // (e.g. a validator that owns an in-progress IP). + ErrBusy = errors.New("busy") + // ErrInUse: the entity can't be removed because something else + // references it (e.g. a target group referenced by a check type). + ErrInUse = errors.New("in use") + // ErrValidation: the request is well-formed but refers to something + // invalid (e.g. a check type naming a target group that doesn't exist). + ErrValidation = errors.New("validation failed") + // ErrInvalidState: the requested state transition isn't valid from the + // entity's current state (e.g. cancelling an already-finished IP). + ErrInvalidState = errors.New("invalid state") +) diff --git a/internal/db/migrations/0002_dynamic_config.sql b/internal/db/migrations/0002_dynamic_config.sql new file mode 100644 index 0000000..fc3bdf1 --- /dev/null +++ b/internal/db/migrations/0002_dynamic_config.sql @@ -0,0 +1,26 @@ +-- Dynamic configuration tables: sites, target_groups, check_types. +-- validators already exists (0001_init.sql) and needs no new columns. +-- Bootstrap semantics (apply YAML only when the table is empty) live in +-- internal/db/bootstrap.go, not in this migration. + +CREATE TABLE sites ( + idx INTEGER PRIMARY KEY, -- 1, 2 or 3 -- fixed slot + site_id TEXT NOT NULL UNIQUE, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +); + +CREATE TABLE target_groups ( + group_name TEXT PRIMARY KEY, + targets TEXT NOT NULL, -- JSON array of strings + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +); + +CREATE TABLE check_types ( + name TEXT PRIMARY KEY, + enabled BOOLEAN NOT NULL DEFAULT 1, + target_groups TEXT NOT NULL, -- JSON array of group names + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +); diff --git a/internal/db/models.go b/internal/db/models.go index 75565ce..e5eb09b 100644 --- a/internal/db/models.go +++ b/internal/db/models.go @@ -20,9 +20,10 @@ const ( IPDone = "done" IPFailed = "failed" - ResultPass = "pass" - ResultPartial = "partial" - ResultFail = "fail" + ResultPass = "pass" + ResultPartial = "partial" + ResultFail = "fail" + ResultCancelled = "cancelled" SourceEgress = "egress" ) @@ -115,3 +116,41 @@ type Event struct { OccurredAt time.Time CreatedAt time.Time } + +type Site struct { + Index int + SiteID string + CreatedAt time.Time + UpdatedAt time.Time +} + +type TargetGroup struct { + Name string + Targets []string + CreatedAt time.Time + UpdatedAt time.Time +} + +type CheckType struct { + Name string + Enabled bool + TargetGroups []string + CreatedAt time.Time + UpdatedAt time.Time +} + +// ResolvedCheckType is a check type with its target groups already expanded +// into a flat target list — what AssignmentForValidator hands to an agent. +type ResolvedCheckType struct { + Type string + Targets []string +} + +// SubmitIPsResult categorizes how each address in a SubmitIPs call was +// handled. +type SubmitIPsResult struct { + Added []string + Requeued []string + Reordered []string + SkippedInProgress []string +} diff --git a/internal/db/queries_checktypes.go b/internal/db/queries_checktypes.go new file mode 100644 index 0000000..e805ead --- /dev/null +++ b/internal/db/queries_checktypes.go @@ -0,0 +1,141 @@ +package db + +import ( + "context" + "encoding/json" + "fmt" +) + +// ListCheckTypes returns all configured check types, ordered by name. +func (d *DB) ListCheckTypes(ctx context.Context) ([]CheckType, error) { + rows, err := d.QueryContext(ctx, ` + SELECT name, enabled, target_groups, created_at, updated_at FROM check_types ORDER BY name + `) + if err != nil { + return nil, err + } + defer rows.Close() + + var out []CheckType + for rows.Next() { + ct, err := scanCheckType(rows) + if err != nil { + return nil, err + } + out = append(out, *ct) + } + return out, rows.Err() +} + +// ListResolvedCheckTypes returns every enabled check type with its target +// groups already expanded into a flat target list — the egress +// check-type/target configuration handed to a validator-agent once its IP +// is ready to be worked on. This is the dynamic equivalent of what +// orchestrator.New used to compute once from static YAML config. +func (d *DB) ListResolvedCheckTypes(ctx context.Context) ([]ResolvedCheckType, error) { + groups, err := d.loadTargetGroupsMap(ctx) + if err != nil { + return nil, err + } + + rows, err := d.QueryContext(ctx, ` + SELECT name, target_groups FROM check_types WHERE enabled=1 ORDER BY name + `) + if err != nil { + return nil, err + } + defer rows.Close() + + var out []ResolvedCheckType + for rows.Next() { + var name, groupsJSON string + if err := rows.Scan(&name, &groupsJSON); err != nil { + return nil, err + } + var groupNames []string + if err := json.Unmarshal([]byte(groupsJSON), &groupNames); err != nil { + return nil, fmt.Errorf("decode target_groups for check type %q: %w", name, err) + } + var targets []string + for _, g := range groupNames { + targets = append(targets, groups[g]...) + } + out = append(out, ResolvedCheckType{Type: name, Targets: targets}) + } + return out, rows.Err() +} + +func (d *DB) loadTargetGroupsMap(ctx context.Context) (map[string][]string, error) { + groups, err := d.ListTargetGroups(ctx) + if err != nil { + return nil, err + } + out := make(map[string][]string, len(groups)) + for _, g := range groups { + out[g.Name] = g.Targets + } + return out, nil +} + +// UpsertCheckType creates or replaces a check type. All named target groups +// must already exist. +func (d *DB) UpsertCheckType(ctx context.Context, name string, enabled bool, targetGroups []string) error { + if name == "" { + return fmt.Errorf("check type name must not be empty: %w", ErrValidation) + } + existing, err := d.loadTargetGroupsMap(ctx) + if err != nil { + return err + } + for _, g := range targetGroups { + if _, ok := existing[g]; !ok { + return fmt.Errorf("check type %q references unknown target group %q: %w", name, g, ErrValidation) + } + } + payload, err := json.Marshal(targetGroups) + if err != nil { + return err + } + now := timeToDB(Now()) + _, err = d.ExecContext(ctx, ` + INSERT INTO check_types (name, enabled, target_groups, created_at, updated_at) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(name) DO UPDATE SET + enabled=excluded.enabled, target_groups=excluded.target_groups, updated_at=excluded.updated_at + `, name, enabled, string(payload), now, now) + if err != nil { + return fmt.Errorf("upsert check type: %w", err) + } + return nil +} + +// DeleteCheckType removes a check type. +func (d *DB) DeleteCheckType(ctx context.Context, name string) error { + res, err := d.ExecContext(ctx, `DELETE FROM check_types WHERE name=?`, name) + if err != nil { + return err + } + if n, _ := res.RowsAffected(); n == 0 { + return fmt.Errorf("check type %q: %w", name, ErrNotFound) + } + return nil +} + +func scanCheckType(row rowScanner) (*CheckType, error) { + var ct CheckType + var groupsJSON, createdAt, updatedAt string + if err := row.Scan(&ct.Name, &ct.Enabled, &groupsJSON, &createdAt, &updatedAt); err != nil { + return nil, err + } + if err := json.Unmarshal([]byte(groupsJSON), &ct.TargetGroups); err != nil { + return nil, fmt.Errorf("decode target_groups for check type %q: %w", ct.Name, err) + } + var err error + if ct.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + if ct.UpdatedAt, err = dbToTime(updatedAt); err != nil { + return nil, err + } + return &ct, nil +} diff --git a/internal/db/queries_dynconfig_test.go b/internal/db/queries_dynconfig_test.go new file mode 100644 index 0000000..647bf38 --- /dev/null +++ b/internal/db/queries_dynconfig_test.go @@ -0,0 +1,332 @@ +package db + +import ( + "context" + "errors" + "path/filepath" + "testing" + "time" + + "cloudipvalidator/internal/config" +) + +func newTestDB(t *testing.T) (*DB, context.Context) { + t.Helper() + ctx := context.Background() + d, err := Open(ctx, filepath.Join(t.TempDir(), "test.db")) + if err != nil { + t.Fatalf("open db: %v", err) + } + t.Cleanup(func() { d.Close() }) + return d, ctx +} + +func TestUpsertSiteValidation(t *testing.T) { + d, ctx := newTestDB(t) + + if err := d.UpsertSite(ctx, 0, "site-1"); !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for idx=0, got %v", err) + } + if err := d.UpsertSite(ctx, 4, "site-1"); !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for idx=4, got %v", err) + } + if err := d.UpsertSite(ctx, 1, ""); !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for empty site_id, got %v", err) + } + + if err := d.UpsertSite(ctx, 1, "site-1"); err != nil { + t.Fatalf("upsert site 1: %v", err) + } + if err := d.UpsertSite(ctx, 2, "site-1"); !errors.Is(err, ErrConflict) { + t.Fatalf("expected ErrConflict assigning site-1 to slot 2, got %v", err) + } + // Renaming the same slot is fine (not a conflict with itself). + if err := d.UpsertSite(ctx, 1, "site-1-renamed"); err != nil { + t.Fatalf("rename slot 1: %v", err) + } + + if err := d.DeleteSite(ctx, 1); err != nil { + t.Fatalf("delete site 1: %v", err) + } + if err := d.DeleteSite(ctx, 1); !errors.Is(err, ErrNotFound) { + t.Fatalf("expected ErrNotFound deleting already-gone slot, got %v", err) + } +} + +func TestTargetGroupInUseCannotBeDeleted(t *testing.T) { + d, ctx := newTestDB(t) + + if err := d.UpsertTargetGroup(ctx, "web", []string{"https://example.test"}); err != nil { + t.Fatalf("upsert target group: %v", err) + } + if err := d.UpsertTargetGroup(ctx, "empty", nil); !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for empty targets, got %v", err) + } + if err := d.UpsertCheckType(ctx, "https", true, []string{"web"}); err != nil { + t.Fatalf("upsert check type: %v", err) + } + + if err := d.DeleteTargetGroup(ctx, "web"); !errors.Is(err, ErrInUse) { + t.Fatalf("expected ErrInUse deleting group referenced by check type, got %v", err) + } + + if err := d.DeleteCheckType(ctx, "https"); err != nil { + t.Fatalf("delete check type: %v", err) + } + if err := d.DeleteTargetGroup(ctx, "web"); err != nil { + t.Fatalf("delete target group after check type removed: %v", err) + } +} + +func TestUpsertCheckTypeUnknownGroup(t *testing.T) { + d, ctx := newTestDB(t) + + err := d.UpsertCheckType(ctx, "https", true, []string{"does-not-exist"}) + if !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for unknown target group, got %v", err) + } +} + +func TestListResolvedCheckTypesExpandsGroups(t *testing.T) { + d, ctx := newTestDB(t) + + if err := d.UpsertTargetGroup(ctx, "web", []string{"https://a.test", "https://b.test"}); err != nil { + t.Fatalf("upsert web group: %v", err) + } + if err := d.UpsertTargetGroup(ctx, "extra", []string{"https://c.test"}); err != nil { + t.Fatalf("upsert extra group: %v", err) + } + if err := d.UpsertCheckType(ctx, "https", true, []string{"web", "extra"}); err != nil { + t.Fatalf("upsert https check type: %v", err) + } + if err := d.UpsertCheckType(ctx, "ssh", false, []string{"web"}); err != nil { + t.Fatalf("upsert disabled ssh check type: %v", err) + } + + resolved, err := d.ListResolvedCheckTypes(ctx) + if err != nil { + t.Fatalf("list resolved check types: %v", err) + } + if len(resolved) != 1 { + t.Fatalf("expected only the enabled check type, got %d entries: %+v", len(resolved), resolved) + } + if resolved[0].Type != "https" || len(resolved[0].Targets) != 3 { + t.Fatalf("expected https with 3 expanded targets, got %+v", resolved[0]) + } +} + +func TestAdminValidatorCRUD(t *testing.T) { + d, ctx := newTestDB(t) + + if err := d.AdminCreateValidator(ctx, "validator-1", "port-1"); err != nil { + t.Fatalf("create validator: %v", err) + } + if err := d.AdminCreateValidator(ctx, "validator-1", "port-2"); !errors.Is(err, ErrConflict) { + t.Fatalf("expected ErrConflict creating duplicate validator, got %v", err) + } + if err := d.AdminUpdateValidatorPort(ctx, "validator-1", "port-3"); err != nil { + t.Fatalf("update validator port: %v", err) + } + v, err := d.GetValidator(ctx, "validator-1") + if err != nil || v.OSPortID != "port-3" { + t.Fatalf("expected port-3 after update, got %+v (err=%v)", v, err) + } + if err := d.AdminUpdateValidatorPort(ctx, "no-such-validator", "port-9"); !errors.Is(err, ErrNotFound) { + t.Fatalf("expected ErrNotFound updating unknown validator, got %v", err) + } + + // Claim it onto an IP, then deletion should be refused as busy. + if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil { + t.Fatalf("seed queue: %v", err) + } + if _, err := d.ClaimNextQueued(ctx, "validator-1", time.Minute); err != nil { + t.Fatalf("claim next queued: %v", err) + } + if err := d.DeleteValidator(ctx, "validator-1"); !errors.Is(err, ErrBusy) { + t.Fatalf("expected ErrBusy deleting validator that owns an ip, got %v", err) + } + + if err := d.ReleaseFIP(ctx, 1, "validator-1"); err != nil { + t.Fatalf("release fip: %v", err) + } + if err := d.DeleteValidator(ctx, "validator-1"); err != nil { + t.Fatalf("delete validator after release: %v", err) + } + if err := d.DeleteValidator(ctx, "validator-1"); !errors.Is(err, ErrNotFound) { + t.Fatalf("expected ErrNotFound deleting already-gone validator, got %v", err) + } +} + +func TestBootstrapFromConfigIgnoresYAMLOnNonEmptyTables(t *testing.T) { + d, ctx := newTestDB(t) + + cfg := &config.ControlAPI{ + Validators: []config.ValidatorConfig{{ValidatorID: "validator-1", OSPortID: "port-1"}}, + Sites: []config.SiteConfig{{SiteID: "site-1", Index: 1}}, + Targets: map[string][]string{"web": {"https://example.test"}}, + CheckTypes: []config.CheckTypeConfig{{Name: "https", Enabled: true, Targets: []string{"web"}}}, + } + if err := d.BootstrapFromConfig(ctx, cfg); err != nil { + t.Fatalf("first bootstrap: %v", err) + } + + // Simulate an admin API change that should survive a second bootstrap + // pass (e.g. a restart) even though the YAML still names the old port. + if err := d.AdminUpdateValidatorPort(ctx, "validator-1", "port-changed-via-api"); err != nil { + t.Fatalf("update validator port via api: %v", err) + } + if err := d.UpsertSite(ctx, 1, "site-changed-via-api"); err != nil { + t.Fatalf("update site via api: %v", err) + } + + if err := d.BootstrapFromConfig(ctx, cfg); err != nil { + t.Fatalf("second bootstrap: %v", err) + } + + v, err := d.GetValidator(ctx, "validator-1") + if err != nil { + t.Fatalf("get validator: %v", err) + } + if v.OSPortID != "port-changed-via-api" { + t.Fatalf("expected YAML to be ignored on non-empty table, got os_port_id=%q", v.OSPortID) + } + sites, err := d.ListSites(ctx) + if err != nil { + t.Fatalf("list sites: %v", err) + } + if len(sites) != 1 || sites[0].SiteID != "site-changed-via-api" { + t.Fatalf("expected YAML to be ignored on non-empty sites table, got %+v", sites) + } +} + +func TestSubmitIPsMixedBatch(t *testing.T) { + d, ctx := newTestDB(t) + + // Seed: 1.1.1.1 done, 2.2.2.2 already queued, 3.3.3.3 mid-check. + if err := d.SeedQueue(ctx, []string{"1.1.1.1", "2.2.2.2", "3.3.3.3"}); err != nil { + t.Fatalf("seed queue: %v", err) + } + ip1, err := d.GetIPByAddress(ctx, "1.1.1.1") + if err != nil { + t.Fatalf("get ip1: %v", err) + } + if err := d.FinishIP(ctx, ip1.ID, ResultPass); err != nil { + t.Fatalf("finish ip1: %v", err) + } + if err := d.AdminCreateValidator(ctx, "validator-1", "port-1"); err != nil { + t.Fatalf("create validator: %v", err) + } + claimed, err := d.ClaimNextQueued(ctx, "validator-1", time.Minute) + if err != nil || claimed == nil { + t.Fatalf("claim next queued: item=%+v err=%v", claimed, err) + } + if claimed.IPAddress != "2.2.2.2" { + t.Fatalf("expected to claim 2.2.2.2 (lowest sequence still queued), got %s", claimed.IPAddress) + } + // 2.2.2.2 is now assigning_fip (mid-check); 3.3.3.3 is still queued. + + result, err := d.SubmitIPs(ctx, []string{"4.4.4.4", "1.1.1.1", "2.2.2.2", "3.3.3.3"}) + if err != nil { + t.Fatalf("submit ips: %v", err) + } + if len(result.Added) != 1 || result.Added[0] != "4.4.4.4" { + t.Fatalf("expected 4.4.4.4 added, got %+v", result.Added) + } + if len(result.Requeued) != 1 || result.Requeued[0] != "1.1.1.1" { + t.Fatalf("expected 1.1.1.1 requeued (was done), got %+v", result.Requeued) + } + if len(result.Reordered) != 1 || result.Reordered[0] != "3.3.3.3" { + t.Fatalf("expected 3.3.3.3 reordered (was queued), got %+v", result.Reordered) + } + if len(result.SkippedInProgress) != 1 || result.SkippedInProgress[0] != "2.2.2.2" { + t.Fatalf("expected 2.2.2.2 skipped (mid-check), got %+v", result.SkippedInProgress) + } + + // 2.2.2.2 must be untouched: still assigning_fip, same owner. + ip2, err := d.GetIPByAddress(ctx, "2.2.2.2") + if err != nil { + t.Fatalf("get ip2: %v", err) + } + if ip2.State != IPAssigningFIP || ip2.OwnerValidatorID == nil || *ip2.OwnerValidatorID != "validator-1" { + t.Fatalf("expected 2.2.2.2 untouched mid-check, got %+v", ip2) + } + + // 1.1.1.1 must be freshly queued again with a bumped attempt number and + // cleared result. + ip1, err = d.GetIPByAddress(ctx, "1.1.1.1") + if err != nil { + t.Fatalf("get ip1 after requeue: %v", err) + } + if ip1.State != IPQueued || ip1.AttemptNumber != 2 || ip1.OverallResult != "" { + t.Fatalf("expected 1.1.1.1 requeued fresh, got %+v", ip1) + } + + // Queue processing order should follow the submitted list order for the + // touched/new addresses: 4.4.4.4, 1.1.1.1, 3.3.3.3 (2.2.2.2 excluded, + // already claimed by validator-1, which is now busy — use a second + // idle validator to observe what's claimed next). + if err := d.AdminCreateValidator(ctx, "validator-2", "port-2"); err != nil { + t.Fatalf("create validator-2: %v", err) + } + claimed, err = d.ClaimNextQueued(ctx, "validator-2", time.Minute) + if err != nil { + t.Fatalf("claim after submit: %v", err) + } + if claimed == nil || claimed.IPAddress != "4.4.4.4" { + t.Fatalf("expected 4.4.4.4 to be claimed first per batch order, got %+v", claimed) + } +} + +func TestSubmitIPsEmptyListRejected(t *testing.T) { + d, ctx := newTestDB(t) + if _, err := d.SubmitIPs(ctx, nil); !errors.Is(err, ErrValidation) { + t.Fatalf("expected ErrValidation for empty address list, got %v", err) + } +} + +func TestCancelIP(t *testing.T) { + d, ctx := newTestDB(t) + + if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil { + t.Fatalf("seed queue: %v", err) + } + ip, err := d.GetIPByAddress(ctx, "1.2.3.4") + if err != nil { + t.Fatalf("get ip: %v", err) + } + + // Cancel straight from queued. + if err := d.CancelIP(ctx, ip.ID); err != nil { + t.Fatalf("cancel queued ip: %v", err) + } + ip, _ = d.GetIP(ctx, ip.ID) + if ip.State != IPFailed || ip.OverallResult != ResultCancelled { + t.Fatalf("expected failed/cancelled, got state=%s result=%s", ip.State, ip.OverallResult) + } + + // Cancelling an already-terminal IP is rejected. + if err := d.CancelIP(ctx, ip.ID); !errors.Is(err, ErrInvalidState) { + t.Fatalf("expected ErrInvalidState cancelling an already-finished ip, got %v", err) + } + + // Cancel mid-check: claim, associate, then cancel — owner/lease/fip + // should be cleared by CancelIP itself (disassociation is the + // orchestrator's job, not this method's). + if err := d.SeedQueue(ctx, []string{"5.6.7.8"}); err != nil { + t.Fatalf("seed queue 2: %v", err) + } + if err := d.AdminCreateValidator(ctx, "validator-1", "port-1"); err != nil { + t.Fatalf("create validator: %v", err) + } + claimed, err := d.ClaimNextQueued(ctx, "validator-1", time.Minute) + if err != nil || claimed == nil { + t.Fatalf("claim: item=%+v err=%v", claimed, err) + } + if err := d.CancelIP(ctx, claimed.ID); err != nil { + t.Fatalf("cancel mid-check ip: %v", err) + } + got, _ := d.GetIP(ctx, claimed.ID) + if got.State != IPFailed || got.OverallResult != ResultCancelled || got.OwnerValidatorID != nil { + t.Fatalf("expected cancelled + owner cleared, got %+v", got) + } +} diff --git a/internal/db/queries_ipqueue.go b/internal/db/queries_ipqueue.go index 9567722..ed8d5a6 100644 --- a/internal/db/queries_ipqueue.go +++ b/internal/db/queries_ipqueue.go @@ -222,6 +222,119 @@ func (d *DB) RequeueOrFail(ctx context.Context, ipID int64, validatorID string, return tx.Commit() } +// SubmitIPs is the single admin entry point for both "add new addresses to +// the queue" and "force a re-check of an already-finished address" — the +// same list can freely mix both. Addresses are processed in one +// transaction, in the order given: +// +// - unknown address: inserted as a new queued row. +// - address currently done/failed: reset to queued (new attempt, +// retry_count cleared — this is a deliberate admin-triggered restart, +// not a system retry). +// - address currently queued (not yet claimed): left in state=queued, +// only its sequence is updated. +// - address currently mid-check (assigning_fip / awaiting_self_check / +// checking / aggregating): left untouched entirely — never start a +// second concurrent check for the same address. +// +// Every touched/inserted address (new, requeued, or merely reordered) gets +// a sequence assigned in list order, continuing after the current max +// sequence, so a batch's relative order is preserved and, critically, +// resubmitting the same list later reproduces the same relative order. +func (d *DB) SubmitIPs(ctx context.Context, addresses []string) (SubmitIPsResult, error) { + var result SubmitIPsResult + if len(addresses) == 0 { + return result, fmt.Errorf("addresses must not be empty: %w", ErrValidation) + } + + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return result, err + } + defer tx.Rollback() + + var base int + if err := tx.QueryRowContext(ctx, `SELECT COALESCE(MAX(sequence), -1) + 1 FROM ip_queue`).Scan(&base); err != nil { + return result, err + } + + now := timeToDB(Now()) + for i, addr := range addresses { + seq := base + i + + var state string + err := tx.QueryRowContext(ctx, `SELECT state FROM ip_queue WHERE ip_address=?`, addr).Scan(&state) + switch { + case err == sql.ErrNoRows: + if _, err := tx.ExecContext(ctx, ` + INSERT INTO ip_queue (ip_address, sequence, state, created_at, updated_at) + VALUES (?, ?, ?, ?, ?) + `, addr, seq, IPQueued, now, now); err != nil { + return result, fmt.Errorf("insert %s: %w", addr, err) + } + result.Added = append(result.Added, addr) + + case err != nil: + return result, err + + case state == IPDone || state == IPFailed: + if _, err := tx.ExecContext(ctx, ` + UPDATE ip_queue SET + state=?, sequence=?, owner_validator_id=NULL, fip_id='', retry_count=0, + attempt_number=attempt_number+1, lease_expires_at=NULL, egress_complete=0, + site1_complete=0, site2_complete=0, site3_complete=0, overall_result='', + assigned_at=NULL, aggregated_at=NULL, fip_released_at=NULL, updated_at=? + WHERE ip_address=? + `, IPQueued, seq, now, addr); err != nil { + return result, fmt.Errorf("requeue %s: %w", addr, err) + } + result.Requeued = append(result.Requeued, addr) + + case state == IPQueued: + if _, err := tx.ExecContext(ctx, ` + UPDATE ip_queue SET sequence=?, updated_at=? WHERE ip_address=? + `, seq, now, addr); err != nil { + return result, fmt.Errorf("reorder %s: %w", addr, err) + } + result.Reordered = append(result.Reordered, addr) + + default: + // Actively being processed (assigning_fip / awaiting_self_check + // / checking / aggregating) — leave it alone, don't duplicate. + result.SkippedInProgress = append(result.SkippedInProgress, addr) + } + } + + if err := tx.Commit(); err != nil { + return result, err + } + return result, nil +} + +// CancelIP force-stops a non-terminal IP: marks it failed with +// overall_result=cancelled. It does not disassociate the floating IP or +// free the owning validator — that requires the OpenStack client, so it's +// the caller's (orchestrator's) job to do that before/after calling this. +// Returns ErrInvalidState if the IP is already done/failed (including the +// race where aggregation finishes between the caller's read and this call — +// closed by the single-connection transactional UPDATE below). +func (d *DB) CancelIP(ctx context.Context, ipID int64) error { + now := timeToDB(Now()) + res, err := d.ExecContext(ctx, ` + UPDATE ip_queue SET + state=?, overall_result=?, aggregated_at=?, owner_validator_id=NULL, fip_id='', + lease_expires_at=NULL, updated_at=? + WHERE id=? AND state NOT IN (?, ?) + `, IPFailed, ResultCancelled, now, now, ipID, IPDone, IPFailed) + if err != nil { + return fmt.Errorf("cancel ip: %w", err) + } + if n, _ := res.RowsAffected(); n == 0 { + return fmt.Errorf("ip_id %d already finished: %w", ipID, ErrInvalidState) + } + return nil +} + func (d *DB) SetEgressComplete(ctx context.Context, ipID int64) error { _, err := d.ExecContext(ctx, `UPDATE ip_queue SET egress_complete=1, updated_at=? WHERE id=?`, timeToDB(Now()), ipID) return err diff --git a/internal/db/queries_sites.go b/internal/db/queries_sites.go new file mode 100644 index 0000000..0d9c74f --- /dev/null +++ b/internal/db/queries_sites.go @@ -0,0 +1,97 @@ +package db + +import ( + "context" + "database/sql" + "fmt" +) + +// ListSites returns all configured prober sites, ordered by slot index. +func (d *DB) ListSites(ctx context.Context) ([]Site, error) { + rows, err := d.QueryContext(ctx, ` + SELECT idx, site_id, created_at, updated_at FROM sites ORDER BY idx + `) + if err != nil { + return nil, err + } + defer rows.Close() + + var out []Site + for rows.Next() { + var s Site + var createdAt, updatedAt string + if err := rows.Scan(&s.Index, &s.SiteID, &createdAt, &updatedAt); err != nil { + return nil, err + } + var err error + if s.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + if s.UpdatedAt, err = dbToTime(updatedAt); err != nil { + return nil, err + } + out = append(out, s) + } + return out, rows.Err() +} + +// GetSiteIndex resolves a configured site_id to its 1/2/3 slot index. It +// returns (0, nil) — not an error — if no site with that ID is configured, +// matching the "0 means unconfigured" convention used throughout the +// prober-facing handlers. +func (d *DB) GetSiteIndex(ctx context.Context, siteID string) (int, error) { + var idx int + err := d.QueryRowContext(ctx, `SELECT idx FROM sites WHERE site_id=?`, siteID).Scan(&idx) + if err == sql.ErrNoRows { + return 0, nil + } + if err != nil { + return 0, err + } + return idx, nil +} + +// UpsertSite assigns (or renames) the site occupying the given slot. idx +// must be 1, 2, or 3 — the schema hard-caps the number of prober slots at +// three (see ip_queue.site{1,2,3}_complete). site_id must be unique across +// slots. +func (d *DB) UpsertSite(ctx context.Context, idx int, siteID string) error { + if idx < 1 || idx > 3 { + return fmt.Errorf("site index must be 1, 2, or 3, got %d: %w", idx, ErrValidation) + } + if siteID == "" { + return fmt.Errorf("site_id must not be empty: %w", ErrValidation) + } + + var existingIdx int + err := d.QueryRowContext(ctx, `SELECT idx FROM sites WHERE site_id=? AND idx!=?`, siteID, idx).Scan(&existingIdx) + if err != nil && err != sql.ErrNoRows { + return err + } + if err == nil { + return fmt.Errorf("site_id %q already assigned to slot %d: %w", siteID, existingIdx, ErrConflict) + } + + now := timeToDB(Now()) + _, err = d.ExecContext(ctx, ` + INSERT INTO sites (idx, site_id, created_at, updated_at) + VALUES (?, ?, ?, ?) + ON CONFLICT(idx) DO UPDATE SET site_id=excluded.site_id, updated_at=excluded.updated_at + `, idx, siteID, now, now) + if err != nil { + return fmt.Errorf("upsert site: %w", err) + } + return nil +} + +// DeleteSite frees the given slot. +func (d *DB) DeleteSite(ctx context.Context, idx int) error { + res, err := d.ExecContext(ctx, `DELETE FROM sites WHERE idx=?`, idx) + if err != nil { + return err + } + if n, _ := res.RowsAffected(); n == 0 { + return fmt.Errorf("site slot %d: %w", idx, ErrNotFound) + } + return nil +} diff --git a/internal/db/queries_targetgroups.go b/internal/db/queries_targetgroups.go new file mode 100644 index 0000000..f40e489 --- /dev/null +++ b/internal/db/queries_targetgroups.go @@ -0,0 +1,109 @@ +package db + +import ( + "context" + "database/sql" + "encoding/json" + "fmt" +) + +// ListTargetGroups returns all configured target groups, ordered by name. +func (d *DB) ListTargetGroups(ctx context.Context) ([]TargetGroup, error) { + rows, err := d.QueryContext(ctx, ` + SELECT group_name, targets, created_at, updated_at FROM target_groups ORDER BY group_name + `) + if err != nil { + return nil, err + } + defer rows.Close() + + var out []TargetGroup + for rows.Next() { + g, err := scanTargetGroup(rows) + if err != nil { + return nil, err + } + out = append(out, *g) + } + return out, rows.Err() +} + +// GetTargetGroup returns a single target group by name. +func (d *DB) GetTargetGroup(ctx context.Context, name string) (*TargetGroup, error) { + row := d.QueryRowContext(ctx, ` + SELECT group_name, targets, created_at, updated_at FROM target_groups WHERE group_name=? + `, name) + g, err := scanTargetGroup(row) + if err == sql.ErrNoRows { + return nil, fmt.Errorf("target group %q: %w", name, ErrNotFound) + } + return g, err +} + +// UpsertTargetGroup creates or replaces a target group's target list. +func (d *DB) UpsertTargetGroup(ctx context.Context, name string, targets []string) error { + if name == "" { + return fmt.Errorf("group name must not be empty: %w", ErrValidation) + } + if len(targets) == 0 { + return fmt.Errorf("target group %q must have at least one target: %w", name, ErrValidation) + } + payload, err := json.Marshal(targets) + if err != nil { + return err + } + now := timeToDB(Now()) + _, err = d.ExecContext(ctx, ` + INSERT INTO target_groups (group_name, targets, created_at, updated_at) + VALUES (?, ?, ?, ?) + ON CONFLICT(group_name) DO UPDATE SET targets=excluded.targets, updated_at=excluded.updated_at + `, name, string(payload), now, now) + if err != nil { + return fmt.Errorf("upsert target group: %w", err) + } + return nil +} + +// DeleteTargetGroup removes a target group, refusing if any check type +// still references it. +func (d *DB) DeleteTargetGroup(ctx context.Context, name string) error { + checkTypes, err := d.ListCheckTypes(ctx) + if err != nil { + return err + } + for _, ct := range checkTypes { + for _, g := range ct.TargetGroups { + if g == name { + return fmt.Errorf("target group %q is used by check type %q: %w", name, ct.Name, ErrInUse) + } + } + } + + res, err := d.ExecContext(ctx, `DELETE FROM target_groups WHERE group_name=?`, name) + if err != nil { + return err + } + if n, _ := res.RowsAffected(); n == 0 { + return fmt.Errorf("target group %q: %w", name, ErrNotFound) + } + return nil +} + +func scanTargetGroup(row rowScanner) (*TargetGroup, error) { + var g TargetGroup + var targetsJSON, createdAt, updatedAt string + if err := row.Scan(&g.Name, &targetsJSON, &createdAt, &updatedAt); err != nil { + return nil, err + } + if err := json.Unmarshal([]byte(targetsJSON), &g.Targets); err != nil { + return nil, fmt.Errorf("decode targets for group %q: %w", g.Name, err) + } + var err error + if g.CreatedAt, err = dbToTime(createdAt); err != nil { + return nil, err + } + if g.UpdatedAt, err = dbToTime(updatedAt); err != nil { + return nil, err + } + return &g, nil +} diff --git a/internal/db/queries_validators.go b/internal/db/queries_validators.go index ea494e6..8b94a29 100644 --- a/internal/db/queries_validators.go +++ b/internal/db/queries_validators.go @@ -148,6 +148,79 @@ func (d *DB) FreeValidator(ctx context.Context, validatorID string) error { return err } +// AdminCreateValidator registers a brand-new validator via the admin API. +// Unlike RegisterValidator (used by the agent's self-registration call), +// this refuses to upsert over an existing row. +func (d *DB) AdminCreateValidator(ctx context.Context, validatorID, osPortID string) error { + var exists int + err := d.QueryRowContext(ctx, `SELECT 1 FROM validators WHERE validator_id=?`, validatorID).Scan(&exists) + if err != nil && err != sql.ErrNoRows { + return err + } + if err == nil { + return fmt.Errorf("validator %q: %w", validatorID, ErrConflict) + } + + now := timeToDB(Now()) + _, err = d.ExecContext(ctx, ` + INSERT INTO validators (validator_id, hostname, os_port_id, agent_version, state, created_at, updated_at) + VALUES (?, '', ?, '', ?, ?, ?) + `, validatorID, osPortID, ValidatorIdle, now, now) + if err != nil { + return fmt.Errorf("create validator: %w", err) + } + return nil +} + +// AdminUpdateValidatorPort updates an existing validator's Neutron port ID. +func (d *DB) AdminUpdateValidatorPort(ctx context.Context, validatorID, osPortID string) error { + res, err := d.ExecContext(ctx, ` + UPDATE validators SET os_port_id=?, updated_at=? WHERE validator_id=? + `, osPortID, timeToDB(Now()), validatorID) + if err != nil { + return fmt.Errorf("update validator port: %w", err) + } + if n, _ := res.RowsAffected(); n == 0 { + return fmt.Errorf("validator %q: %w", validatorID, ErrNotFound) + } + return nil +} + +// DeleteValidator removes a validator, refusing if it currently owns an IP. +// ip_queue.owner_validator_id is a permanent historical record (set once an +// IP is claimed, never cleared on completion — see RequeueOrFail/ +// ReleaseFIP), so any validator that has ever processed an IP would +// otherwise violate the FK constraint on delete; those historical +// references are cleared in the same transaction once we've confirmed the +// validator isn't currently busy. +func (d *DB) DeleteValidator(ctx context.Context, validatorID string) error { + tx, err := d.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() + + var currentIPID sql.NullInt64 + err = tx.QueryRowContext(ctx, `SELECT current_ip_id FROM validators WHERE validator_id=?`, validatorID).Scan(¤tIPID) + if err == sql.ErrNoRows { + return fmt.Errorf("validator %q: %w", validatorID, ErrNotFound) + } + if err != nil { + return err + } + if currentIPID.Valid { + return fmt.Errorf("validator %q owns ip_id %d: %w", validatorID, currentIPID.Int64, ErrBusy) + } + + if _, err := tx.ExecContext(ctx, `UPDATE ip_queue SET owner_validator_id=NULL WHERE owner_validator_id=?`, validatorID); err != nil { + return fmt.Errorf("clear historical ip_queue references: %w", err) + } + if _, err := tx.ExecContext(ctx, `DELETE FROM validators WHERE validator_id=?`, validatorID); err != nil { + return fmt.Errorf("delete validator: %w", err) + } + return tx.Commit() +} + type rowScanner interface { Scan(dest ...interface{}) error } diff --git a/internal/httpapi/dto_admin.go b/internal/httpapi/dto_admin.go new file mode 100644 index 0000000..2adbd88 --- /dev/null +++ b/internal/httpapi/dto_admin.go @@ -0,0 +1,62 @@ +package httpapi + +// DTOs for the admin queue-management and dynamic-config endpoints +// (/api/v1/admin/ips, /api/v1/admin/config/*). Unlike the older read-only +// admin endpoints (which marshal internal/db model structs directly, in +// PascalCase), these use explicit snake_case JSON tags to match the +// agent/prober DTO convention. + +type submitIPsRequest struct { + Addresses []string `json:"addresses"` +} + +type submitIPsResponse struct { + Added []string `json:"added"` + Requeued []string `json:"requeued"` + Reordered []string `json:"reordered"` + SkippedInProgress []string `json:"skipped_in_progress"` +} + +type validatorDTO struct { + ValidatorID string `json:"validator_id"` + OSPortID string `json:"os_port_id"` + State string `json:"state"` +} + +type createValidatorRequest struct { + ValidatorID string `json:"validator_id"` + OSPortID string `json:"os_port_id"` +} + +type updateValidatorRequest struct { + OSPortID string `json:"os_port_id"` +} + +type siteDTO struct { + Index int `json:"index"` + SiteID string `json:"site_id"` +} + +type putSiteRequest struct { + SiteID string `json:"site_id"` +} + +type targetGroupDTO struct { + Name string `json:"name"` + Targets []string `json:"targets"` +} + +type putTargetGroupRequest struct { + Targets []string `json:"targets"` +} + +type checkTypeDTO struct { + Name string `json:"name"` + Enabled bool `json:"enabled"` + Targets []string `json:"targets"` +} + +type putCheckTypeRequest struct { + Enabled bool `json:"enabled"` + Targets []string `json:"targets"` +} diff --git a/internal/httpapi/handlers_admin.go b/internal/httpapi/handlers_admin.go index cbba00e..167cce0 100644 --- a/internal/httpapi/handlers_admin.go +++ b/internal/httpapi/handlers_admin.go @@ -73,3 +73,49 @@ func (s *Server) handleAdminValidators(w http.ResponseWriter, r *http.Request) { } writeJSON(w, http.StatusOK, validators) } + +// handleAdminSubmitIPs is the single entry point for both adding new +// addresses to the queue and forcing a re-check of already-finished ones — +// see db.SubmitIPs for the exact per-address rules. It's a direct DB call +// (no OpenStack interaction is needed to merely queue work), matching the +// existing admin handlers above which also bypass the orchestrator for +// reads. +func (s *Server) handleAdminSubmitIPs(w http.ResponseWriter, r *http.Request) { + var req submitIPsRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + result, err := s.DB.SubmitIPs(r.Context(), req.Addresses) + if err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, submitIPsResponse{ + Added: emptyIfNil(result.Added), + Requeued: emptyIfNil(result.Requeued), + Reordered: emptyIfNil(result.Reordered), + SkippedInProgress: emptyIfNil(result.SkippedInProgress), + }) +} + +// handleAdminCancelIP force-stops a check in progress (or still-queued) for +// the given address. Requires the orchestrator, since a floating IP may +// need to be disassociated in OpenStack. +func (s *Server) handleAdminCancelIP(w http.ResponseWriter, r *http.Request) { + address := r.PathValue("ip") + if err := s.Orch.ForceCancel(r.Context(), address); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +// emptyIfNil turns a nil slice into an empty one so these fields always +// marshal as `[]` rather than `null`. +func emptyIfNil(s []string) []string { + if s == nil { + return []string{} + } + return s +} diff --git a/internal/httpapi/handlers_config.go b/internal/httpapi/handlers_config.go new file mode 100644 index 0000000..4352d73 --- /dev/null +++ b/internal/httpapi/handlers_config.go @@ -0,0 +1,183 @@ +package httpapi + +import ( + "net/http" + "strconv" +) + +// --- validators --- + +func (s *Server) handleConfigListValidators(w http.ResponseWriter, r *http.Request) { + validators, err := s.DB.ListValidators(r.Context()) + if err != nil { + writeDBError(w, err) + return + } + out := make([]validatorDTO, len(validators)) + for i, v := range validators { + out[i] = validatorDTO{ValidatorID: v.ValidatorID, OSPortID: v.OSPortID, State: v.State} + } + writeJSON(w, http.StatusOK, out) +} + +func (s *Server) handleConfigCreateValidator(w http.ResponseWriter, r *http.Request) { + var req createValidatorRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if req.ValidatorID == "" { + writeError(w, http.StatusBadRequest, "validator_id must not be empty") + return + } + if err := s.DB.AdminCreateValidator(r.Context(), req.ValidatorID, req.OSPortID); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusCreated, validatorDTO{ValidatorID: req.ValidatorID, OSPortID: req.OSPortID, State: "idle"}) +} + +func (s *Server) handleConfigUpdateValidator(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + var req updateValidatorRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if err := s.DB.AdminUpdateValidatorPort(r.Context(), id, req.OSPortID); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +func (s *Server) handleConfigDeleteValidator(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + if err := s.DB.DeleteValidator(r.Context(), id); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +// --- sites --- + +func (s *Server) handleConfigListSites(w http.ResponseWriter, r *http.Request) { + sites, err := s.DB.ListSites(r.Context()) + if err != nil { + writeDBError(w, err) + return + } + out := make([]siteDTO, len(sites)) + for i, site := range sites { + out[i] = siteDTO{Index: site.Index, SiteID: site.SiteID} + } + writeJSON(w, http.StatusOK, out) +} + +func (s *Server) handleConfigPutSite(w http.ResponseWriter, r *http.Request) { + idx, err := strconv.Atoi(r.PathValue("index")) + if err != nil { + writeError(w, http.StatusBadRequest, "index must be an integer") + return + } + var req putSiteRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if err := s.DB.UpsertSite(r.Context(), idx, req.SiteID); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, siteDTO{Index: idx, SiteID: req.SiteID}) +} + +func (s *Server) handleConfigDeleteSite(w http.ResponseWriter, r *http.Request) { + idx, err := strconv.Atoi(r.PathValue("index")) + if err != nil { + writeError(w, http.StatusBadRequest, "index must be an integer") + return + } + if err := s.DB.DeleteSite(r.Context(), idx); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +// --- target groups --- + +func (s *Server) handleConfigListTargets(w http.ResponseWriter, r *http.Request) { + groups, err := s.DB.ListTargetGroups(r.Context()) + if err != nil { + writeDBError(w, err) + return + } + out := make([]targetGroupDTO, len(groups)) + for i, g := range groups { + out[i] = targetGroupDTO{Name: g.Name, Targets: g.Targets} + } + writeJSON(w, http.StatusOK, out) +} + +func (s *Server) handleConfigPutTargetGroup(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("group") + var req putTargetGroupRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if err := s.DB.UpsertTargetGroup(r.Context(), name, req.Targets); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, targetGroupDTO{Name: name, Targets: req.Targets}) +} + +func (s *Server) handleConfigDeleteTargetGroup(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("group") + if err := s.DB.DeleteTargetGroup(r.Context(), name); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} + +// --- check types --- + +func (s *Server) handleConfigListCheckTypes(w http.ResponseWriter, r *http.Request) { + checkTypes, err := s.DB.ListCheckTypes(r.Context()) + if err != nil { + writeDBError(w, err) + return + } + out := make([]checkTypeDTO, len(checkTypes)) + for i, ct := range checkTypes { + out[i] = checkTypeDTO{Name: ct.Name, Enabled: ct.Enabled, Targets: ct.TargetGroups} + } + writeJSON(w, http.StatusOK, out) +} + +func (s *Server) handleConfigPutCheckType(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + var req putCheckTypeRequest + if err := readJSON(r, &req); err != nil { + writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) + return + } + if err := s.DB.UpsertCheckType(r.Context(), name, req.Enabled, req.Targets); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, checkTypeDTO{Name: name, Enabled: req.Enabled, Targets: req.Targets}) +} + +func (s *Server) handleConfigDeleteCheckType(w http.ResponseWriter, r *http.Request) { + name := r.PathValue("name") + if err := s.DB.DeleteCheckType(r.Context(), name); err != nil { + writeDBError(w, err) + return + } + writeJSON(w, http.StatusOK, okResponse{OK: true}) +} diff --git a/internal/httpapi/handlers_config_test.go b/internal/httpapi/handlers_config_test.go new file mode 100644 index 0000000..ccc83b1 --- /dev/null +++ b/internal/httpapi/handlers_config_test.go @@ -0,0 +1,291 @@ +package httpapi + +import ( + "context" + "encoding/json" + "log/slog" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + "time" + + "cloudipvalidator/internal/config" + "cloudipvalidator/internal/db" + "cloudipvalidator/internal/openstack" + "cloudipvalidator/internal/orchestrator" +) + +// newConfigTestHarness sets up a control-api stack with an *empty* +// control-api.yaml (no validators/sites/targets/check_types) — everything +// in this test is created purely through the admin API, to prove the +// dynamic-config path works with no YAML at all. +func newConfigTestHarness(t *testing.T) (*fakeClient, *db.DB, *orchestrator.Orchestrator, *openstack.MockClient) { + t.Helper() + ctx := context.Background() + d, err := db.Open(ctx, filepath.Join(t.TempDir(), "test.db")) + if err != nil { + t.Fatalf("open db: %v", err) + } + t.Cleanup(func() { d.Close() }) + + mock := openstack.NewMockClient() + + cfg := &config.ControlAPI{ + Orchestrator: config.OrchestratorConfig{ + PollIntervalSeconds: 1, SelfCheckTimeoutSeconds: 10, MaxSelfCheckRetries: 3, + CheckingWindowSeconds: 120, MaxRetries: 3, LeaseTTLSeconds: 180, HeartbeatTimeoutSeconds: 30, + }, + Aggregation: config.AggregationConfig{MissingCountsAsFail: true}, + Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true}, + } + if err := d.BootstrapFromConfig(ctx, cfg); err != nil { + t.Fatalf("bootstrap from (empty) config: %v", err) + } + + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) + orch := orchestrator.New(d, mock, cfg, log) + srv := New(d, orch, log) + ts := httptest.NewServer(srv.Handler()) + t.Cleanup(ts.Close) + + return &fakeClient{t: t, base: ts.URL, client: ts.Client()}, d, orch, mock +} + +// TestConfigManagedEntirelyViaAPI proves an operator can stand up a working +// validator/site/target-group/check-type configuration using only the +// admin API — no YAML at all — and that an IP submitted afterwards passes +// through the full checking cycle to `done`. +func TestConfigManagedEntirelyViaAPI(t *testing.T) { + fc, d, orch, mock := newConfigTestHarness(t) + ctx := context.Background() + + mock.Seed("fip-1", "9.9.9.9", "svc-project") + + resp, body := fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ + ValidatorID: "validator-1", OSPortID: "port-1", + }) + if resp.StatusCode != http.StatusCreated { + t.Fatalf("create validator: status=%d body=%s", resp.StatusCode, body) + } + + resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/targets/web", putTargetGroupRequest{ + Targets: []string{"https://example.test"}, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("put target group: status=%d body=%s", resp.StatusCode, body) + } + + resp, body = fc.do(http.MethodPut, "/api/v1/admin/config/check-types/https", putCheckTypeRequest{ + Enabled: true, Targets: []string{"web"}, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("put check type: status=%d body=%s", resp.StatusCode, body) + } + + // No sites configured -> inbound checks stay opt-out; egress alone + // should be enough to reach `done`. + + resp, body = fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("submit ips: status=%d body=%s", resp.StatusCode, body) + } + var submitResp submitIPsResponse + if err := json.Unmarshal(body, &submitResp); err != nil { + t.Fatalf("unmarshal submit response: %v", err) + } + if len(submitResp.Added) != 1 || submitResp.Added[0] != "9.9.9.9" { + t.Fatalf("expected 9.9.9.9 added, got %+v", submitResp) + } + + resp, body = fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("register agent: status=%d body=%s", resp.StatusCode, body) + } + + orch.Tick(ctx) + + resp, body = fc.do(http.MethodGet, "/api/v1/agents/validator-1/assignment", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("assignment: status=%d body=%s", resp.StatusCode, body) + } + var assignment assignmentResponse + if err := json.Unmarshal(body, &assignment); err != nil { + t.Fatalf("unmarshal assignment: %v", err) + } + if len(assignment.CheckConfig) != 1 || assignment.CheckConfig[0].Type != "https" { + t.Fatalf("expected the API-created https check type in the assignment, got %+v", assignment.CheckConfig) + } + + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/self-check", selfCheckRequest{ + IPID: assignment.IPID, DetectedEgress: "9.9.9.9", Success: true, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("self-check: status=%d body=%s", resp.StatusCode, body) + } + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{ + Results: []checkResultDTO{{ + IPID: assignment.IPID, CheckType: "https", Target: "https://example.test", + Success: true, CheckedAt: time.Now().Format(time.RFC3339Nano), + }}, + }) + if resp.StatusCode != http.StatusOK { + t.Fatalf("results: status=%d body=%s", resp.StatusCode, body) + } + resp, body = fc.do(http.MethodPost, "/api/v1/agents/validator-1/complete", agentCompleteRequest{IPID: assignment.IPID}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("complete: status=%d body=%s", resp.StatusCode, body) + } + + orch.Tick(ctx) + + item, err := d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip: %v", err) + } + if item.State != db.IPDone || item.OverallResult != db.ResultPass { + t.Fatalf("expected done/pass, got state=%s result=%s", item.State, item.OverallResult) + } +} + +// TestSubmitIPsForcesRecheckOfFinishedAddress proves that resubmitting an +// address that already reached `done` starts a brand-new checking cycle +// rather than being ignored. +func TestSubmitIPsForcesRecheckOfFinishedAddress(t *testing.T) { + fc, d, orch, mock := newConfigTestHarness(t) + ctx := context.Background() + mock.Seed("fip-1", "9.9.9.9", "svc-project") + + fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"}) + fc.do(http.MethodPut, "/api/v1/admin/config/targets/web", putTargetGroupRequest{Targets: []string{"https://example.test"}}) + fc.do(http.MethodPut, "/api/v1/admin/config/check-types/https", putCheckTypeRequest{Enabled: true, Targets: []string{"web"}}) + + runOneCycle := func() { + orch.Tick(ctx) + _, body := fc.do(http.MethodGet, "/api/v1/agents/validator-1/assignment", nil) + var assignment assignmentResponse + if err := json.Unmarshal(body, &assignment); err != nil { + t.Fatalf("unmarshal assignment: %v", err) + } + fc.do(http.MethodPost, "/api/v1/agents/validator-1/self-check", selfCheckRequest{ + IPID: assignment.IPID, DetectedEgress: "9.9.9.9", Success: true, + }) + fc.do(http.MethodPost, "/api/v1/agents/validator-1/results", agentResultsRequest{ + Results: []checkResultDTO{{ + IPID: assignment.IPID, CheckType: "https", Target: "https://example.test", + Success: true, CheckedAt: time.Now().Format(time.RFC3339Nano), + }}, + }) + fc.do(http.MethodPost, "/api/v1/agents/validator-1/complete", agentCompleteRequest{IPID: assignment.IPID}) + orch.Tick(ctx) + } + + fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"}) + fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}}) + runOneCycle() + + item, err := d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip: %v", err) + } + if item.State != db.IPDone || item.AttemptNumber != 1 { + t.Fatalf("expected done after first cycle with attempt_number=1, got state=%s attempt=%d", item.State, item.AttemptNumber) + } + + // Force a recheck of the same, already-finished address. The mock FIP + // was disassociated at the end of the first cycle; associateFIP will + // simply re-associate it during the second cycle. + resp, body := fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}}) + if resp.StatusCode != http.StatusOK { + t.Fatalf("submit ips (recheck): status=%d body=%s", resp.StatusCode, body) + } + var submitResp submitIPsResponse + if err := json.Unmarshal(body, &submitResp); err != nil { + t.Fatalf("unmarshal submit response: %v", err) + } + if len(submitResp.Requeued) != 1 || submitResp.Requeued[0] != "9.9.9.9" { + t.Fatalf("expected 9.9.9.9 to be requeued, got %+v", submitResp) + } + + item, err = d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip after resubmit: %v", err) + } + if item.State != db.IPQueued || item.AttemptNumber != 2 || item.OverallResult != "" { + t.Fatalf("expected freshly queued with attempt_number=2, got %+v", item) + } + + runOneCycle() + item, err = d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip after second cycle: %v", err) + } + if item.State != db.IPDone || item.OverallResult != db.ResultPass || item.AttemptNumber != 2 { + t.Fatalf("expected done/pass on second attempt, got %+v", item) + } +} + +// TestForceCancelMidCheck proves POST /admin/ips/{ip}/cancel stops an +// in-progress check, disassociates its floating IP, and frees the +// validator, without waiting for the checking window to elapse. +func TestForceCancelMidCheck(t *testing.T) { + fc, d, orch, mock := newConfigTestHarness(t) + ctx := context.Background() + mock.Seed("fip-1", "9.9.9.9", "svc-project") + + fc.do(http.MethodPost, "/api/v1/admin/config/validators", createValidatorRequest{ValidatorID: "validator-1", OSPortID: "port-1"}) + fc.do(http.MethodPut, "/api/v1/admin/config/targets/web", putTargetGroupRequest{Targets: []string{"https://example.test"}}) + fc.do(http.MethodPut, "/api/v1/admin/config/check-types/https", putCheckTypeRequest{Enabled: true, Targets: []string{"web"}}) + fc.do(http.MethodPost, "/api/v1/agents/register", registerAgentRequest{ValidatorID: "validator-1"}) + fc.do(http.MethodPost, "/api/v1/admin/ips", submitIPsRequest{Addresses: []string{"9.9.9.9"}}) + + orch.Tick(ctx) // claim + associate FIP -> awaiting_self_check + + item, err := d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip: %v", err) + } + if item.State != db.IPAwaitingSelfCheck { + t.Fatalf("expected awaiting_self_check before cancel, got %s", item.State) + } + if fip, _ := mock.GetFloatingIPByAddress(ctx, "9.9.9.9"); fip.PortID == "" { + t.Fatalf("expected fip associated before cancel") + } + + resp, body := fc.do(http.MethodPost, "/api/v1/admin/ips/9.9.9.9/cancel", nil) + if resp.StatusCode != http.StatusOK { + t.Fatalf("cancel: status=%d body=%s", resp.StatusCode, body) + } + + item, err = d.GetIPByAddress(ctx, "9.9.9.9") + if err != nil { + t.Fatalf("get ip after cancel: %v", err) + } + if item.State != db.IPFailed || item.OverallResult != db.ResultCancelled { + t.Fatalf("expected failed/cancelled, got state=%s result=%s", item.State, item.OverallResult) + } + if fip, _ := mock.GetFloatingIPByAddress(ctx, "9.9.9.9"); fip.PortID != "" { + t.Fatalf("expected fip disassociated after cancel, still on port %q", fip.PortID) + } + + v, err := d.GetValidator(ctx, "validator-1") + if err != nil { + t.Fatalf("get validator: %v", err) + } + if v.State != db.ValidatorIdle || v.CurrentIPID != nil { + t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID) + } + + // Cancelling again is rejected — nothing left to cancel. + resp, body = fc.do(http.MethodPost, "/api/v1/admin/ips/9.9.9.9/cancel", nil) + if resp.StatusCode != http.StatusConflict { + t.Fatalf("expected 409 cancelling an already-finished ip, status=%d body=%s", resp.StatusCode, body) + } + + // Cancelling an unknown address is a 404. + resp, body = fc.do(http.MethodPost, "/api/v1/admin/ips/1.1.1.1/cancel", nil) + if resp.StatusCode != http.StatusNotFound { + t.Fatalf("expected 404 cancelling unknown ip, status=%d body=%s", resp.StatusCode, body) + } +} diff --git a/internal/httpapi/handlers_prober.go b/internal/httpapi/handlers_prober.go index 577433f..b0febcd 100644 --- a/internal/httpapi/handlers_prober.go +++ b/internal/httpapi/handlers_prober.go @@ -13,7 +13,12 @@ func (s *Server) handleProberRegister(w http.ResponseWriter, r *http.Request) { writeError(w, http.StatusBadRequest, "invalid body: "+err.Error()) return } - if s.Orch.SiteIndexForID(req.SiteID) == 0 { + idx, err := s.Orch.SiteIndexForID(r.Context(), req.SiteID) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + if idx == 0 { writeError(w, http.StatusBadRequest, "unknown site_id: "+req.SiteID) return } @@ -26,7 +31,12 @@ func (s *Server) handleProberRegister(w http.ResponseWriter, r *http.Request) { // time, since multiple validators run in parallel. func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) { siteID := r.PathValue("site_id") - if s.Orch.SiteIndexForID(siteID) == 0 { + idx, err := s.Orch.SiteIndexForID(r.Context(), siteID) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } + if idx == 0 { writeError(w, http.StatusNotFound, "unknown site_id: "+siteID) return } @@ -47,7 +57,11 @@ func (s *Server) handleProberAssignments(w http.ResponseWriter, r *http.Request) func (s *Server) handleProberResults(w http.ResponseWriter, r *http.Request) { siteID := r.PathValue("site_id") - siteIndex := s.Orch.SiteIndexForID(siteID) + siteIndex, err := s.Orch.SiteIndexForID(r.Context(), siteID) + if err != nil { + writeError(w, http.StatusInternalServerError, err.Error()) + return + } if siteIndex == 0 { writeError(w, http.StatusNotFound, "unknown site_id: "+siteID) return diff --git a/internal/httpapi/httpapi_test.go b/internal/httpapi/httpapi_test.go index 9fd18de..bdd509e 100644 --- a/internal/httpapi/httpapi_test.go +++ b/internal/httpapi/httpapi_test.go @@ -78,6 +78,10 @@ func TestEndToEndHTTPFlow(t *testing.T) { Targets: map[string][]string{"web": {"https://example.test"}}, Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true}, } + if err := d.BootstrapFromConfig(ctx, cfg); err != nil { + t.Fatalf("bootstrap from config: %v", err) + } + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) orch := orchestrator.New(d, mock, cfg, log) diff --git a/internal/httpapi/routes.go b/internal/httpapi/routes.go index 4ca82fc..b5fadd6 100644 --- a/internal/httpapi/routes.go +++ b/internal/httpapi/routes.go @@ -19,6 +19,25 @@ func (s *Server) routes(mux *http.ServeMux) { mux.HandleFunc("GET /api/v1/admin/status", s.handleAdminStatus) mux.HandleFunc("GET /api/v1/admin/ips", s.handleAdminIPs) + mux.HandleFunc("POST /api/v1/admin/ips", s.handleAdminSubmitIPs) mux.HandleFunc("GET /api/v1/admin/ips/{ip}", s.handleAdminIPDetail) + mux.HandleFunc("POST /api/v1/admin/ips/{ip}/cancel", s.handleAdminCancelIP) mux.HandleFunc("GET /api/v1/admin/validators", s.handleAdminValidators) + + mux.HandleFunc("GET /api/v1/admin/config/validators", s.handleConfigListValidators) + mux.HandleFunc("POST /api/v1/admin/config/validators", s.handleConfigCreateValidator) + mux.HandleFunc("PUT /api/v1/admin/config/validators/{id}", s.handleConfigUpdateValidator) + mux.HandleFunc("DELETE /api/v1/admin/config/validators/{id}", s.handleConfigDeleteValidator) + + mux.HandleFunc("GET /api/v1/admin/config/sites", s.handleConfigListSites) + mux.HandleFunc("PUT /api/v1/admin/config/sites/{index}", s.handleConfigPutSite) + mux.HandleFunc("DELETE /api/v1/admin/config/sites/{index}", s.handleConfigDeleteSite) + + mux.HandleFunc("GET /api/v1/admin/config/targets", s.handleConfigListTargets) + mux.HandleFunc("PUT /api/v1/admin/config/targets/{group}", s.handleConfigPutTargetGroup) + mux.HandleFunc("DELETE /api/v1/admin/config/targets/{group}", s.handleConfigDeleteTargetGroup) + + mux.HandleFunc("GET /api/v1/admin/config/check-types", s.handleConfigListCheckTypes) + mux.HandleFunc("PUT /api/v1/admin/config/check-types/{name}", s.handleConfigPutCheckType) + mux.HandleFunc("DELETE /api/v1/admin/config/check-types/{name}", s.handleConfigDeleteCheckType) } diff --git a/internal/httpapi/server.go b/internal/httpapi/server.go index fb43180..18c684c 100644 --- a/internal/httpapi/server.go +++ b/internal/httpapi/server.go @@ -7,6 +7,7 @@ package httpapi import ( "encoding/json" + "errors" "log/slog" "net/http" @@ -56,3 +57,19 @@ func readJSON(r *http.Request, v interface{}) error { dec := json.NewDecoder(r.Body) return dec.Decode(v) } + +// writeDBError maps the typed sentinel errors returned by internal/db's +// admin mutation methods to the appropriate HTTP status, instead of +// defaulting everything to 500 like the older read-only admin handlers do. +func writeDBError(w http.ResponseWriter, err error) { + switch { + case errors.Is(err, db.ErrNotFound): + writeError(w, http.StatusNotFound, err.Error()) + case errors.Is(err, db.ErrConflict), errors.Is(err, db.ErrBusy), errors.Is(err, db.ErrInUse), errors.Is(err, db.ErrInvalidState): + writeError(w, http.StatusConflict, err.Error()) + case errors.Is(err, db.ErrValidation): + writeError(w, http.StatusBadRequest, err.Error()) + default: + writeError(w, http.StatusInternalServerError, err.Error()) + } +} diff --git a/internal/orchestrator/orchestrator.go b/internal/orchestrator/orchestrator.go index 0b48ca3..62b48ad 100644 --- a/internal/orchestrator/orchestrator.go +++ b/internal/orchestrator/orchestrator.go @@ -9,6 +9,8 @@ package orchestrator import ( "context" + "database/sql" + "errors" "fmt" "log/slog" "time" @@ -20,9 +22,8 @@ import ( // CheckConfig is the check-type/target configuration handed to a // validator-agent once its IP has passed self-check. It mirrors -// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a -// flat list so the agent doesn't need its own copy of the target-group -// mapping. +// db.ResolvedCheckType, kept as a distinct type so httpapi's DTO layer +// doesn't need to import internal/db just for this shape. type CheckConfig struct { Type string `json:"type"` Targets []string `json:"targets"` @@ -33,31 +34,22 @@ type Orchestrator struct { OS openstack.FloatingIPClient Cfg config.OrchestratorConfig Agg config.AggregationConfig - Checks []CheckConfig - Sites []config.SiteConfig Inbound config.InboundConfig Log *slog.Logger } +// New constructs an Orchestrator. Egress check types/targets and prober +// sites are no longer taken from cfg — they're read from the database on +// every use (see AssignmentForValidator, expectedCheckCount, +// isReadyToAggregate) so admin API changes to them take effect without a +// restart. cfg.Validators/.Sites/.CheckTypes/.Targets/.IPAddresses are only +// consulted once, at process startup, by db.BootstrapFromConfig. func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator { - var checks []CheckConfig - for _, ct := range cfg.CheckTypes { - if !ct.Enabled { - continue - } - var targets []string - for _, group := range ct.Targets { - targets = append(targets, cfg.Targets[group]...) - } - checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets}) - } return &Orchestrator{ DB: d, OS: osClient, Cfg: cfg.Orchestrator, Agg: cfg.Aggregation, - Checks: checks, - Sites: cfg.Sites, Inbound: cfg.Inbound, Log: log, } @@ -159,7 +151,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string, // AssignmentForValidator returns the check config for a validator's current // IP if it's ready to be worked on (awaiting_self_check or checking), -// or nil if the validator has nothing to do right now. +// or nil if the validator has nothing to do right now. The check config is +// read fresh from the database on every call, so admin changes to +// check_types/targets apply to the very next assignment. func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) { v, err := o.DB.GetValidator(ctx, validatorID) if err != nil { @@ -175,17 +169,21 @@ func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID s if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking { return nil, nil, nil } - return item, o.Checks, nil + resolved, err := o.DB.ListResolvedCheckTypes(ctx) + if err != nil { + return nil, nil, err + } + checks := make([]CheckConfig, len(resolved)) + for i, r := range resolved { + checks[i] = CheckConfig{Type: r.Type, Targets: r.Targets} + } + return item, checks, nil } -// SiteIndexForID resolves a configured site_id to its 1/2/3 index. -func (o *Orchestrator) SiteIndexForID(siteID string) int { - for _, s := range o.Sites { - if s.SiteID == siteID { - return s.Index - } - } - return 0 +// SiteIndexForID resolves a configured site_id to its 1/2/3 index, or +// (0, nil) if unconfigured. +func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int, error) { + return o.DB.GetSiteIndex(ctx, siteID) } // RecordCheck upserts a single check result and, if it represents a @@ -203,6 +201,43 @@ func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteInd return o.DB.SetSiteComplete(ctx, ipID, siteIndex) } +// ForceCancel stops an in-progress (or still-queued) check for the given +// address on admin request, even though it was never going to finish on +// its own within the checking window. If a floating IP is currently +// associated, it's disassociated best-effort (same fallthrough-on-error +// behavior as aggregateAndRelease/sweepExpiredLeases: the DB/validator +// state must still be freed even if Neutron hiccups). Returns +// db.ErrNotFound if the address is unknown, or db.ErrInvalidState if it has +// already reached done/failed. +func (o *Orchestrator) ForceCancel(ctx context.Context, ipAddress string) error { + item, err := o.DB.GetIPByAddress(ctx, ipAddress) + if err != nil { + if errors.Is(err, sql.ErrNoRows) { + return fmt.Errorf("ip %q: %w", ipAddress, db.ErrNotFound) + } + return err + } + + if item.FIPID != "" { + if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil { + o.Log.Error("disassociate fip on force cancel", "ip_id", item.ID, "fip_id", item.FIPID, "err", err) + } + } + + if err := o.DB.CancelIP(ctx, item.ID); err != nil { + return err + } + + if item.OwnerValidatorID != nil { + if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID); err != nil { + return fmt.Errorf("free validator: %w", err) + } + } + + o.event(ctx, "control-api", "", &item.ID, "force_cancel", "") + return nil +} + // sweepCheckingWindow moves IPs that have either finished reporting from // every source, or hit the checking-window deadline, into aggregation. func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error { @@ -211,8 +246,12 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error { if err != nil { return fmt.Errorf("list checking: %w", err) } + sites, err := o.DB.ListSites(ctx) + if err != nil { + return fmt.Errorf("list sites: %w", err) + } for _, item := range checking { - if !o.isReadyToAggregate(item, deadline) { + if !o.isReadyToAggregate(item, deadline, sites) { continue } if err := o.aggregateAndRelease(ctx, item); err != nil { @@ -225,20 +264,20 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error { // isReadyToAggregate reports whether an in-progress IP has either finished // reporting from every source it's actually expecting, or hit the // checking-window deadline. Which inbound sources it's expecting is driven -// entirely by o.Sites — inbound checks are optional: an empty (or -// partial) `sites` config in control-api.yaml means this IP is ready as -// soon as egress completes (or after the corresponding subset of -// siteN_complete flags), with no need to wait on a prober that will never -// exist. This is what makes inbound checks genuinely opt-in rather than a -// hardcoded expectation of exactly three sites. -func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time) bool { +// entirely by the currently configured sites — inbound checks are optional: +// an empty (or partial) sites configuration means this IP is ready as soon +// as egress completes (or after the corresponding subset of siteN_complete +// flags), with no need to wait on a prober that will never exist. This is +// what makes inbound checks genuinely opt-in rather than a hardcoded +// expectation of exactly three sites. +func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time, sites []db.Site) bool { if item.AssignedAt != nil && item.AssignedAt.Before(deadline) { return true } if !item.EgressComplete { return false } - for _, s := range o.Sites { + for _, s := range sites { switch s.Index { case 1: if !item.Site1Complete { @@ -266,7 +305,10 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI return err } - expected := o.expectedCheckCount() + expected, err := o.expectedCheckCount(ctx) + if err != nil { + return fmt.Errorf("expected check count: %w", err) + } passCount := 0 for _, c := range checks { if c.Success { @@ -317,17 +359,28 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI // expectedCheckCount is the number of check rows a fully-reported IP should // have: one per (egress check-type x target) plus one per (site x inbound -// port/icmp probe). -func (o *Orchestrator) expectedCheckCount() int { +// port/icmp probe). Reads the current check_types/targets/sites from the +// database, so a config change between assignment and aggregation is +// reflected in this specific aggregation (see the "accepted tradeoff" note +// in docs/PLAN_API_CONFIG_MANAGEMENT.md). +func (o *Orchestrator) expectedCheckCount(ctx context.Context) (int, error) { + resolved, err := o.DB.ListResolvedCheckTypes(ctx) + if err != nil { + return 0, err + } egress := 0 - for _, c := range o.Checks { + for _, c := range resolved { egress += len(c.Targets) } + sites, err := o.DB.ListSites(ctx) + if err != nil { + return 0, err + } inboundPerSite := len(o.Inbound.Ports) if o.Inbound.ICMP { inboundPerSite++ } - return egress + inboundPerSite*len(o.Sites) + return egress + inboundPerSite*len(sites), nil } // sweepExpiredLeases reclaims non-terminal IPs whose lease has passed — diff --git a/internal/orchestrator/orchestrator_test.go b/internal/orchestrator/orchestrator_test.go index 2dd6e5e..97063e3 100644 --- a/internal/orchestrator/orchestrator_test.go +++ b/internal/orchestrator/orchestrator_test.go @@ -58,6 +58,10 @@ func newTestOrchestratorWithSites(t *testing.T, leaseTTLSeconds int, sites []con Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true}, } + if err := d.BootstrapFromConfig(ctx, cfg); err != nil { + t.Fatalf("bootstrap from config: %v", err) + } + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError})) return New(d, mock, cfg, log), d, mock } diff --git a/scripts/run-local-e2e.sh b/scripts/run-local-e2e.sh index 8428bf7..6d25466 100755 --- a/scripts/run-local-e2e.sh +++ b/scripts/run-local-e2e.sh @@ -210,6 +210,22 @@ status | python3 -m json.tool echo "--- final ip results ---" ips | python3 -m json.tool +echo "--- demonstrating admin API: force a re-check of the already-done address ---" +curl -fs -X POST "http://127.0.0.1:28080/api/v1/admin/ips" \ + -H 'Content-Type: application/json' -d '{"addresses":["127.0.0.1"]}' | python3 -m json.tool + +echo "--- waiting for the forced re-check to drain (up to 30s) ---" +for i in $(seq 1 60); do + sleep 0.5 + remaining=$(status | python3 -c "import json,sys; d=json.load(sys.stdin); s=d['ips_by_state']; print(sum(v for k,v in s.items() if k not in ('done','failed')))" 2>/dev/null || echo "?") + if [ "$remaining" = "0" ]; then + echo "re-check drained after ~$((i / 2))s" + break + fi +done +echo "--- ip results after forced re-check (attempt_number should have advanced) ---" +ips | python3 -m json.tool + echo "--- logs are in $WORK_DIR (kept for inspection; workdir NOT auto-deleted) ---" trap - EXIT cleanup