Merge pull request 'Add optional automatic check cycle (clear queue -> scan FIPs -> wait -> repeat)' (#2) from feat/auto-cycle into main
Reviewed-on: #2
This commit was merged in pull request #2.
This commit is contained in:
commit
48efa61ef3
35 files changed
+2205
-16
No files matched your search
@@ -28,7 +28,7 @@ control/data plane — в [docs/DIAGRAMS.md](docs/DIAGRAMS.md).
|
||||
| Документ | Для чего |
|
||||
|---|---|
|
||||
| [docs/SETUP.md](docs/SETUP.md) | Развёртывание с нуля: бинарники или сборка из исходников, конфигурация, systemd **и** Docker/docker-compose — пошагово |
|
||||
| [docs/USAGE.md](docs/USAGE.md) | Повседневная работа: постановка адресов в очередь, сканирование Floating IP, наблюдение за статусом, реестр и история, разбор результатов |
|
||||
| [docs/USAGE.md](docs/USAGE.md) | Повседневная работа: постановка адресов в очередь, сканирование Floating IP, автоматический цикл проверок по расписанию, наблюдение за статусом, реестр и история, разбор результатов |
|
||||
| [docs/API.md](docs/API.md) | Спецификация HTTP API `control-api` и примеры запросов (curl) |
|
||||
| [docs/DASHBOARD.md](docs/DASHBOARD.md) | Устройство `admin-dashboard`: страницы, поиск/фильтр, обработка ошибок |
|
||||
| [docs/DIAGRAMS.md](docs/DIAGRAMS.md) | Диаграммы потоков данных: control plane, egress-проверка, телеметрия |
|
||||
|
||||
+4
-4
@@ -1,4 +1,4 @@
|
||||
194bbd250ad93142eb1eb2d8990292327d48842f851f522ddb31f9ddfa8c554f control-api
|
||||
43ca6b15a25e8534ae2d833e8d0575a85da036e17db370c797da613bb51a9c29 validator-agent
|
||||
bde443dd4fd335f9c3c64a2ca04d5db6c9a93fe2f45eb81b1d426581e380cee2 prober
|
||||
e74de9873e9da1ebfa6beeccd34b6fa7c15bf14c2d8efe13f4de7266af141702 admin-dashboard
|
||||
fb8aaab7c07c3702d274aa15fcb299690bb738cbde9faad79e73d726c24e9d1b control-api
|
||||
d7377e90cc34549933f0de8556df113f7ace7425ad0fee96f723ed6a4510d66b validator-agent
|
||||
c3c9e42ad7632cd8a4b88fe5eedf341cf0dcba6ecf5281d247733a98cee91fb6 prober
|
||||
23d1c10c26366dedd790025689eae63677316c7bb9dafb53c8cebca340dfe77f admin-dashboard
|
||||
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
@@ -109,6 +109,7 @@ func runOrchestratorLoop(ctx context.Context, orch *orchestrator.Orchestrator, c
|
||||
return
|
||||
case <-ticker.C:
|
||||
orch.Tick(ctx)
|
||||
orch.AutoCycleStep(ctx)
|
||||
case <-heartbeatTicker.C:
|
||||
if err := orch.SweepStaleHeartbeats(ctx); err != nil {
|
||||
log.Error("sweep stale heartbeats", "err", err)
|
||||
@@ -117,6 +118,14 @@ func runOrchestratorLoop(ctx context.Context, orch *orchestrator.Orchestrator, c
|
||||
log.Error("sweep stale site heartbeats", "err", err)
|
||||
}
|
||||
case <-scanTickerC:
|
||||
// The auto-cycle owns the queue while enabled: a periodic scan
|
||||
// would add addresses in the middle of a cycle.
|
||||
ac, err := orch.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
log.Error("read auto-cycle before periodic scan, scanning anyway", "err", err)
|
||||
} else if ac.Enabled {
|
||||
continue
|
||||
}
|
||||
if _, _, err := orch.ScanFloatingIPs(ctx); err != nil {
|
||||
log.Error("scan floating ips", "err", err)
|
||||
}
|
||||
|
||||
+68
-1
@@ -28,6 +28,7 @@ JSON, базовый префикс прикладных методов — `/ap
|
||||
- [Методы для prober](#методы-для-prober)
|
||||
- [Служебные и административные методы](#служебные-и-административные-методы)
|
||||
- [Управление очередью и конфигурацией](#управление-очередью-и-конфигурацией)
|
||||
- [Автоматический цикл проверок](#автоматический-цикл-проверок)
|
||||
- [Реестр адресов и история проверок](#реестр-адресов-и-история-проверок)
|
||||
- [Модель состояний и связь методов с ней](#модель-состояний-и-связь-методов-с-ней)
|
||||
- [Сквозной пример работы (curl)](#сквозной-пример-работы-curl)
|
||||
@@ -469,7 +470,73 @@ YAML для этой секции больше не перечитывается
|
||||
Помимо ручного вызова, сканирование можно включить по расписанию —
|
||||
`orchestrator.fip_scan_interval_seconds` в `control-api.yaml` (0, по
|
||||
умолчанию, — только по запросу через эту ручку или кнопку «Сканировать
|
||||
Floating IP» в дашборде).
|
||||
Floating IP» в дашборде). Пока включён
|
||||
[автоматический цикл](#автоматический-цикл-проверок), периодический скан
|
||||
не выполняется.
|
||||
|
||||
## Автоматический цикл проверок
|
||||
|
||||
Опциональный повторяющийся сценарий «очистить очередь → просканировать
|
||||
Floating IP → дождаться завершения всех проверок → пауза → заново» (описание
|
||||
для оператора — [USAGE.md](USAGE.md#автоматический-цикл-проверок)). По умолчанию
|
||||
выключен. Состояние и параметры хранятся в базе; перезапуск control-api их
|
||||
не сбрасывает.
|
||||
|
||||
| Метод | Путь | Тело | Успех | Ошибки |
|
||||
|---|---|---|---|---|
|
||||
| `GET` | `/api/v1/admin/auto-cycle` | — | `200`, статус | — |
|
||||
| `PUT` | `/api/v1/admin/auto-cycle` | `{"interval_seconds": N, "max_run_seconds": M}` — любое поле можно опустить | `200`, статус | `400` — `interval_seconds < 60` или `max_run_seconds < 0` (ничего не применяется) |
|
||||
| `POST` | `/api/v1/admin/auto-cycle/start` | — | `200`, статус | — |
|
||||
| `POST` | `/api/v1/admin/auto-cycle/stop` | — | `200`, статус | — |
|
||||
|
||||
Каждый метод отвечает полным объектом статуса:
|
||||
|
||||
```json
|
||||
{
|
||||
"enabled": true,
|
||||
"interval_seconds": 3600,
|
||||
"max_run_seconds": 0,
|
||||
"phase": "waiting",
|
||||
"run_started_at": null,
|
||||
"next_run_at": "2026-10-01T08:00:12.345Z",
|
||||
"last_run_started_at": "2026-10-01T07:00:01.100Z",
|
||||
"last_run_finished_at": "2026-10-01T07:00:12.345Z",
|
||||
"last_outcome": "completed",
|
||||
"last_error": "",
|
||||
"last_scanned_free": 12,
|
||||
"runs_total": 5
|
||||
}
|
||||
```
|
||||
|
||||
- `phase` — `idle` (выключен или ещё не стартовал), `running` (идут
|
||||
проверки), `waiting` (пауза до `next_run_at`).
|
||||
- `last_outcome` — `completed`, `no_free_ips`, `timeout`, `error` или
|
||||
`stopped`; пустая строка, пока не завершился ни один цикл. Для `error`
|
||||
причина — в `last_error`.
|
||||
- `last_scanned_free` — сколько свободных Floating IP нашёл скан последнего
|
||||
цикла; `runs_total` — сколько циклов завершилось исходом `completed`.
|
||||
- `max_run_seconds = 0` — без ограничения времени ожидания проверок.
|
||||
- Времена — RFC 3339 (UTC), `null`, пока не наступили.
|
||||
|
||||
`POST .../start` идемпотентен: у уже включённого цикла ничего не меняется
|
||||
(идущий цикл не перезапускается). Включённый цикл начинает первый прогон на
|
||||
ближайшем шаге оркестратора (порядка `orchestrator.poll_interval_seconds`).
|
||||
`POST .../stop` выключает цикл и переводит его в `idle`; проверки, которые
|
||||
уже идут, не прерываются. Новое значение `interval_seconds` начинает действовать
|
||||
со следующей паузы.
|
||||
|
||||
События цикла (`auto_cycle_started`, `auto_cycle_completed`,
|
||||
`auto_cycle_timeout`, `auto_cycle_error`, `auto_cycle_stopped`) попадают в
|
||||
общий журнал событий; шаги цикла записывают обычные `queue_cleared` и
|
||||
`fip_scan`.
|
||||
|
||||
```bash
|
||||
curl -s -X PUT http://<control-api>:8080/api/v1/admin/auto-cycle \
|
||||
-d '{"interval_seconds": 7200, "max_run_seconds": 1800}'
|
||||
curl -s -X POST http://<control-api>:8080/api/v1/admin/auto-cycle/start
|
||||
curl -s http://<control-api>:8080/api/v1/admin/auto-cycle
|
||||
curl -s -X POST http://<control-api>:8080/api/v1/admin/auto-cycle/stop
|
||||
```
|
||||
|
||||
## Реестр адресов и история проверок
|
||||
|
||||
|
||||
+9
-2
@@ -48,7 +48,7 @@ admin-dashboard -config /etc/cloud-ip-validator/admin-dashboard.yaml
|
||||
|
||||
| Страница | Назначение |
|
||||
|---|---|
|
||||
| `/overview` | Сводная статистика: счётчики по состояниям, «текущая проверка» (live-снимок всех IP не в терминальном состоянии) и «последние N завершённых» (по умолчанию 20, `overview.last_completed_count`) с разбивкой pass/partial/fail/cancelled. Обновляется каждые `overview.poll_interval_seconds` секунд без перезагрузки страницы. Поиск по IP и фильтр по статусу (`pass`/`partial`/`fail`/`cancelled`) над обеими таблицами — набранное/выбранное не сбрасывается очередным обновлением. |
|
||||
| `/overview` | Сводная статистика: счётчики по состояниям, «текущая проверка» (live-снимок всех IP не в терминальном состоянии) и «последние N завершённых» (по умолчанию 20, `overview.last_completed_count`) с разбивкой pass/partial/fail/cancelled. Обновляется каждые `overview.poll_interval_seconds` секунд без перезагрузки страницы. Поиск по IP и фильтр по статусу (`pass`/`partial`/`fail`/`cancelled`) над обеими таблицами — набранное/выбранное не сбрасывается очередным обновлением. Пока включён [автоматический цикл](USAGE.md#автоматический-цикл-проверок), под счётчиками показывается индикатор «Автоцикл активен» с текущей фазой и временем следующего запуска; управляется цикл на `/settings`. |
|
||||
| `/ips` | Полная очередь. Форма сверху принимает список адресов (по одному на строке или через запятую) и отправляет их в `POST /api/v1/admin/ips` — **один и тот же вызов** добавляет новые адреса и принудительно перезапускает уже завершённые (см. ниже). Кнопка «Сканировать Floating IP» делает то же самое автоматически: находит в проекте OpenStack все свободные (не привязанные к порту) Floating IP и сразу ставит их в очередь (`POST /api/v1/admin/ips/scan`, см. [API.md](API.md#post-apiv1adminipsscan)) — то же сканирование можно включить по расписанию через `orchestrator.fip_scan_interval_seconds`. У каждого адреса — кнопка «Перепроверить» (для `done`/`failed`) или «Отменить» (для активных состояний), и всегда — «Удалить» (безвозвратно убирает адрес из очереди, но не из реестра — см. ниже). Чекбоксы у строк + кнопка «Удалить выбранные» удаляют список одним вызовом; «Очистить всё» удаляет вообще всё, включая активные проверки — обе операции требуют явного подтверждения. Пока не истекла настроенная на `/settings` пауза (`fip_settle_seconds`), только что привязавший Floating IP адрес показывает отдельный бейдж «прогрев FIP» вместо обычного статуса. Если на момент попытки привязки Floating IP оказался уже занят другим портом (дрейф состояния облака или ошибочно переданный адрес), цикл проверки для него не запускается — адрес показывает отдельный бейдж «занят» (отличный от «fail») и строку `fip_occupied` в списке событий на его странице; кнопка «Перепроверить» ставит его в очередь заново. |
|
||||
| `/ips/{ip}` | Детали одного адреса, пока он в очереди: все проверки текущей попытки и вся история событий, плюс ссылка на полную историю в реестре (см. ниже). |
|
||||
| `/registry` | **Реестр** — все адреса, когда-либо поставленные на проверку, независимо от того, стоят ли они сейчас в очереди. Переживает удаление адреса из `/ips` и повторное добавление того же адреса позже (см. «Реестр адресов» ниже). Поиск по IP и фильтр по статусу — то же самое, что на `/overview`, плюс отражается в адресной строке (`?q=&status=`), так что отфильтрованную ссылку можно сохранить/переслать. |
|
||||
@@ -57,7 +57,7 @@ admin-dashboard -config /etc/cloud-ip-validator/admin-dashboard.yaml
|
||||
| `/sites` | Площадки — число слотов не ограничено, форма сверху добавляет новый слот, назначить/сменить/освободить `site_id` в каждой строке; колонка «Статус» показывает бейдж подключения пробера (`unregistered`/`idle`/`unreachable`, по аналогии с `/validators`), см. [USAGE.md](USAGE.md#состояния-площадки). |
|
||||
| `/targets` | Группы целей для egress-проверок — создание/редактирование/удаление. |
|
||||
| `/check-types` | Типы проверок (`https`/`icmp`/`ssh`/...), включение/выключение, привязка к группам целей. |
|
||||
| `/settings` | Три формы: `fip_settle_seconds` — пауза (в секундах) между привязкой Floating IP и началом self-check («прогрев» дата-плейна OpenStack, см. [USAGE.md](USAGE.md#пауза-перед-self-check-fip_settle_seconds)); `history_retention_cycles` — сколько последних циклов проверки хранить на адрес в реестре (0 — без ограничения); и типы проверок пробера — TCP-порты (через запятую) + чекбокс ICMP, общие для всех площадок (см. [USAGE.md](USAGE.md#управление-типами-проверок-пробера)). |
|
||||
| `/settings` | Четыре блока. Первый — панель **«Автоматический цикл»**: статус и фаза, время последнего/следующего запуска, результат последнего цикла, поля «Интервал между циклами (мин)» и «Максимальная длительность проверки (мин, 0 = без лимита)» с кнопкой «Сохранить» и кнопка «Включить»/«Выключить» (показывается та, что сейчас применима). Значения вводятся в минутах (допустимы дробные), в control-api уходят секундами; минимум интервала — 1 минута (`60` с), нарушение приходит предупреждением в баннере. Подробности — [USAGE.md](USAGE.md#автоматический-цикл-проверок), API — [API.md](API.md#автоматический-цикл-проверок). Далее три формы: `fip_settle_seconds` — пауза (в секундах) между привязкой Floating IP и началом self-check («прогрев» дата-плейна OpenStack, см. [USAGE.md](USAGE.md#пауза-перед-self-check-fip_settle_seconds)); `history_retention_cycles` — сколько последних циклов проверки хранить на адрес в реестре (0 — без ограничения); и типы проверок пробера — TCP-порты (через запятую) + чекбокс ICMP, общие для всех площадок (см. [USAGE.md](USAGE.md#управление-типами-проверок-пробера)). |
|
||||
|
||||
### «Текущая» и «последняя завершённая» проверка
|
||||
|
||||
@@ -108,6 +108,13 @@ auto-refresh на `/ips`, см. git-историю). Опрашивается т
|
||||
этот порядок и не переносить форму фильтра/панель статистики обратно
|
||||
внутрь опрашиваемого блока.
|
||||
|
||||
Индикатор автоцикла лежит **внутри** панели статистики
|
||||
(`overview_stats`), поэтому обновляется тем же out-of-band swap'ом без
|
||||
отдельного механизма и не меняет порядок блоков. Статус цикла
|
||||
запрашивается у control-api при каждом обновлении; если запрос не удался
|
||||
(например, control-api старой версии без этой ручки), индикатор просто не
|
||||
показывается — остальная страница не страдает, баннер ошибки не выводится.
|
||||
|
||||
### Добавление адресов и принудительный повтор — один и тот же вызов
|
||||
|
||||
Форма на `/ips` всегда бьёт в `POST /api/v1/admin/ips`. Поведение зависит
|
||||
|
||||
+8
-4
@@ -39,13 +39,13 @@ flowchart TB
|
||||
subgraph OP["Оператор"]
|
||||
CFG["control-api.yaml<br/>(bootstrap пустой БД:<br/>validators, sites, targets,<br/>check_types, ip_addresses)"]
|
||||
ENV["control-api.env<br/>(OS_AUTH_URL, OS_TOKEN, ...)"]
|
||||
ADMIN["curl /api/v1/admin/*<br/>(status/ips/validators,<br/>ips submit/cancel,<br/>config CRUD)"]
|
||||
ADMIN["curl /api/v1/admin/*<br/>(status/ips/validators,<br/>ips submit/cancel,<br/>auto-cycle start/stop,<br/>config CRUD)"]
|
||||
end
|
||||
|
||||
subgraph CAPI["control-api (управляющая машина, 1 экземпляр)"]
|
||||
HTTP["HTTP API<br/>/api/v1/agents/*<br/>/api/v1/probers/*<br/>/api/v1/admin/*<br/>/healthz"]
|
||||
ORCH["Оркестратор: Tick раз в<br/>poll_interval_seconds<br/>claim → associate FIP →<br/>ожидание self-check →<br/>checking → aggregate → release<br/>+ lease sweep + heartbeat sweep"]
|
||||
DB[("SQLite<br/>validators / ip_queue / sites /<br/>target_groups / check_types /<br/>checks / events")]
|
||||
ORCH["Оркестратор: Tick раз в<br/>poll_interval_seconds<br/>claim → associate FIP →<br/>ожидание self-check →<br/>checking → aggregate → release<br/>+ lease sweep + heartbeat sweep<br/>+ автоцикл (если включён):<br/>очистка → скан FIP → ожидание →<br/>пауза interval_seconds"]
|
||||
DB[("SQLite<br/>validators / ip_queue / sites /<br/>target_groups / check_types /<br/>checks / events / auto_cycle")]
|
||||
OSCLIENT["OpenStack-клиент<br/>(mode: mock | real)"]
|
||||
end
|
||||
|
||||
@@ -80,7 +80,11 @@ flowchart TB
|
||||
работает по таймеру независимо от HTTP-запросов — назначение IP
|
||||
валидаторам и агрегация результатов не привязаны к конкретному входящему
|
||||
запросу, читая актуальную конфигурацию из БД на каждом проходе, а не
|
||||
единожды при старте. `validator-agent` и `prober` — активная сторона: они
|
||||
единожды при старте. Опциональный автоцикл — часть того же оркестратора:
|
||||
на каждом тике он читает из таблицы `auto_cycle` флаг `enabled`, интервал и
|
||||
фазу (`idle`/`running`/`waiting`), поэтому включение, выключение и смена
|
||||
интервала действуют без перезапуска, а состояние переживает рестарт (см.
|
||||
[USAGE.md](USAGE.md#автоматический-цикл-проверок)). `validator-agent` и `prober` — активная сторона: они
|
||||
сами инициируют все HTTP-запросы к control-api (pull-модель), сам
|
||||
control-api к ним не обращается.
|
||||
|
||||
|
||||
@@ -58,6 +58,18 @@ It will:
|
||||
5. Poll `GET /api/v1/admin/status` until every configured IP has reached a
|
||||
terminal state (`done` or `failed`).
|
||||
6. Print the final `/api/v1/admin/status` and `/api/v1/admin/ips` output.
|
||||
7. Force a re-check of the finished address via `POST /api/v1/admin/ips`
|
||||
and wait for it to drain (`attempt_number` advances).
|
||||
8. Exercise the **automatic cycle** (`/api/v1/admin/auto-cycle`): set the
|
||||
smallest allowed interval (60s) and a 120s run limit, `start` it, and
|
||||
wait for the first cycle to finish. The script then asserts that the
|
||||
outcome is `completed`, the phase is `waiting` with `runs_total=1`, and
|
||||
that the registry's `total_cycles` for `127.0.0.1` grew (the cycle
|
||||
cleared the queue, re-scanned the mock floating IP and re-checked it).
|
||||
Finally it `stop`s the cycle and asserts it is `idle`. The second cycle
|
||||
(the interval wait) is covered by unit tests, so the script does not
|
||||
sit through the 60s pause. The script exits non-zero if any assertion
|
||||
fails.
|
||||
|
||||
Expect to see `127.0.0.1` end with `"state":"done"` and
|
||||
`"overall_result":"pass"` (all egress checks against the stub targets
|
||||
|
||||
+92
-1
@@ -12,6 +12,7 @@
|
||||
- [Как устроена работа с системой](#как-устроена-работа-с-системой)
|
||||
- [Добавление новых IP в очередь](#добавление-новых-ip-в-очередь)
|
||||
- [Сканирование Floating IP из OpenStack](#сканирование-floating-ip-из-openstack)
|
||||
- [Автоматический цикл проверок](#автоматический-цикл-проверок)
|
||||
- [Наблюдение за очередью](#наблюдение-за-очередью)
|
||||
- [Значения полей IP](#значения-полей-ip)
|
||||
- [Как читать итоговый результат (pass/partial/fail)](#как-читать-итоговый-результат-passpartialfail)
|
||||
@@ -100,7 +101,97 @@ curl -s -X POST http://<control-api>:8080/api/v1/admin/ips/scan
|
||||
Если хочется, чтобы сканирование происходило само по расписанию, а не
|
||||
только по запросу — задайте `orchestrator.fip_scan_interval_seconds`
|
||||
(в секундах) в `control-api.yaml`; `0` (по умолчанию) оставляет только
|
||||
ручной запуск через ручку/кнопку выше.
|
||||
ручной запуск через ручку/кнопку выше. Пока включён
|
||||
[автоматический цикл](#автоматический-цикл-проверок), это периодическое
|
||||
сканирование не выполняется — цикл сам управляет очередью.
|
||||
|
||||
## Автоматический цикл проверок
|
||||
|
||||
Опциональный режим, который сам повторяет то, что оператор делает руками:
|
||||
по умолчанию **выключен**, включается и настраивается администратором.
|
||||
Один цикл — это пять шагов:
|
||||
|
||||
1. Очередь очищается целиком — то же, что кнопка «Очистить всё» (см.
|
||||
[«Удаление адресов из очереди»](#удаление-адресов-из-очереди)). История
|
||||
в [реестре](#реестр-адресов-и-глубина-истории) при этом сохраняется.
|
||||
2. Control-api находит все свободные Floating IP и ставит их в очередь —
|
||||
то же, что «Сканировать Floating IP» (см.
|
||||
[выше](#сканирование-floating-ip-из-openstack)).
|
||||
3. Проверки запускаются сами — как для любого адреса в очереди.
|
||||
4. Цикл ждёт, пока **все** адреса очереди дойдут до конечного состояния
|
||||
(`done`, `failed` или `occupied`). К этому моменту результат каждого
|
||||
адреса уже записан в реестр.
|
||||
5. Выдерживается пауза `interval_seconds`, после чего цикл начинается заново
|
||||
с шага 1. Пауза отсчитывается от **завершения** предыдущего цикла, а не
|
||||
от его начала.
|
||||
|
||||
### Параметры
|
||||
|
||||
| Параметр | По умолчанию | Смысл |
|
||||
|---|---|---|
|
||||
| `interval_seconds` | `3600` (1 час) | Пауза между циклами. Не меньше `60`: слишком частые сканы нагружают API OpenStack. |
|
||||
| `max_run_seconds` | `0` (без лимита) | Сколько максимум ждать на шаге 4. По истечении цикл фиксирует `timeout` и переходит к паузе — защита от зависания (нет свободных валидаторов, недоступна площадка). Очередь при этом не трогается: следующий цикл её очистит, а до тех пор видно, что именно не дошло до конца. |
|
||||
|
||||
Параметры хранятся в базе и меняются на лету, без перезапуска; в `control-api.yaml`
|
||||
ничего задавать не нужно. Новый `interval_seconds` применяется к паузе
|
||||
**со следующего цикла** — уже идущая пауза досчитывается по старому значению.
|
||||
|
||||
### Управление
|
||||
|
||||
Через API (подробности — в [API.md](API.md#автоматический-цикл-проверок)):
|
||||
|
||||
```bash
|
||||
# задать параметры (любое из полей можно опустить)
|
||||
curl -s -X PUT http://<control-api>:8080/api/v1/admin/auto-cycle \
|
||||
-d '{"interval_seconds": 7200, "max_run_seconds": 1800}'
|
||||
|
||||
# включить: первый цикл начнётся сразу
|
||||
curl -s -X POST http://<control-api>:8080/api/v1/admin/auto-cycle/start
|
||||
|
||||
# выключить
|
||||
curl -s -X POST http://<control-api>:8080/api/v1/admin/auto-cycle/stop
|
||||
|
||||
# посмотреть состояние
|
||||
curl -s http://<control-api>:8080/api/v1/admin/auto-cycle
|
||||
```
|
||||
|
||||
В `admin-dashboard` — панель «Автоматический цикл» на странице `/settings`: поля
|
||||
«Интервал между циклами» и «Максимальная длительность проверки» (в минутах),
|
||||
кнопки «Включить»/«Выключить». Пока автоцикл включён, на странице `/overview`
|
||||
в блоке статистики показывается индикатор «Автоцикл активен» с фазой и временем
|
||||
следующего запуска.
|
||||
|
||||
### Фазы и результат последнего цикла
|
||||
|
||||
`phase` показывает, что происходит сейчас: `idle` (автоцикл выключен или ещё не
|
||||
стартовал), `running` (идут проверки — шаги 3–4) и `waiting` (пауза между циклами,
|
||||
шаг 5; время следующего запуска — `next_run_at`). Результат последнего цикла
|
||||
(`last_outcome`):
|
||||
|
||||
| Значение | Что произошло |
|
||||
|---|---|
|
||||
| `completed` | Все адреса дошли до конечного состояния; `runs_total` растёт на 1. |
|
||||
| `no_free_ips` | Сканирование не нашло свободных Floating IP — ждать нечего, цикл сразу ушёл в паузу. |
|
||||
| `timeout` | Проверки не уложились в `max_run_seconds`. |
|
||||
| `error` | Не удалось очистить очередь или просканировать облако; причина — в `last_error`. Повтор — через `interval_seconds`. |
|
||||
| `stopped` | Автоцикл выключили в момент, когда шёл цикл. Выключение в паузе предыдущий результат не затирает. |
|
||||
|
||||
### Что важно знать
|
||||
|
||||
- **Выключение не прерывает проверки**, которые уже идут: они закончатся и попадут
|
||||
в реестр, остановится только повторение.
|
||||
- Автоцикл **владеет очередью**: каждый цикл начинается с её полной очистки,
|
||||
поэтому адреса, добавленные вручную, будут удалены (их история в реестре
|
||||
остаётся). Ручные «Очистить всё» и «Сканировать Floating IP» во время цикла
|
||||
не ломают его: если очередь опустела, цикл считается завершённым.
|
||||
- Состояние хранится в базе и **переживает перезапуск** control-api: идущий цикл
|
||||
продолжит ждать, а пауза — досчитается до прежнего `next_run_at`.
|
||||
- События цикла (`auto_cycle_started`, `auto_cycle_completed`, `auto_cycle_timeout`,
|
||||
`auto_cycle_error`, `auto_cycle_stopped`) пишутся в журнал событий вместе с
|
||||
`queue_cleared` и `fip_scan`.
|
||||
- В реальном OpenStack отвязка Floating IP после очистки очереди может
|
||||
отразиться с задержкой; если скан сразу после неё не увидел свободных адресов,
|
||||
цикл завершится с `no_free_ips` и повторится через `interval_seconds`.
|
||||
|
||||
## Наблюдение за очередью
|
||||
|
||||
|
||||
@@ -252,3 +252,28 @@ func (c *client) PutInboundChecks(ctx context.Context, ports []int, icmp bool) (
|
||||
inboundChecksDTO{Ports: ports, ICMP: icmp}, &out)
|
||||
return out, err
|
||||
}
|
||||
|
||||
func (c *client) GetAutoCycle(ctx context.Context) (autoCycleDTO, error) {
|
||||
var out autoCycleDTO
|
||||
err := c.do(ctx, http.MethodGet, "/api/v1/admin/auto-cycle", nil, &out)
|
||||
return out, err
|
||||
}
|
||||
|
||||
func (c *client) PutAutoCycle(ctx context.Context, intervalSeconds, maxRunSeconds int) (autoCycleDTO, error) {
|
||||
var out autoCycleDTO
|
||||
err := c.do(ctx, http.MethodPut, "/api/v1/admin/auto-cycle",
|
||||
map[string]int{"interval_seconds": intervalSeconds, "max_run_seconds": maxRunSeconds}, &out)
|
||||
return out, err
|
||||
}
|
||||
|
||||
func (c *client) StartAutoCycle(ctx context.Context) (autoCycleDTO, error) {
|
||||
var out autoCycleDTO
|
||||
err := c.do(ctx, http.MethodPost, "/api/v1/admin/auto-cycle/start", nil, &out)
|
||||
return out, err
|
||||
}
|
||||
|
||||
func (c *client) StopAutoCycle(ctx context.Context) (autoCycleDTO, error) {
|
||||
var out autoCycleDTO
|
||||
err := c.do(ctx, http.MethodPost, "/api/v1/admin/auto-cycle/stop", nil, &out)
|
||||
return out, err
|
||||
}
|
||||
@@ -41,6 +41,11 @@ type fakeControlAPI struct {
|
||||
scanFreeAddresses []string
|
||||
registry map[string]registryItem
|
||||
registryChecks map[string][]check
|
||||
|
||||
// autoCycle is the state served by /api/v1/admin/auto-cycle*;
|
||||
// autoCycleDown makes all four endpoints answer 500 (unavailable API).
|
||||
autoCycle autoCycleDTO
|
||||
autoCycleDown bool
|
||||
}
|
||||
|
||||
func newFakeControlAPI(t *testing.T) (*fakeControlAPI, string) {
|
||||
@@ -53,6 +58,7 @@ func newFakeControlAPI(t *testing.T) (*fakeControlAPI, string) {
|
||||
checkTypes: map[string]checkTypeDTO{},
|
||||
registry: map[string]registryItem{},
|
||||
registryChecks: map[string][]check{},
|
||||
autoCycle: autoCycleDTO{IntervalSeconds: 3600, Phase: "idle"},
|
||||
}
|
||||
ts := httptest.NewServer(f.handler())
|
||||
t.Cleanup(ts.Close)
|
||||
@@ -250,6 +256,72 @@ func (f *fakeControlAPI) handler() http.Handler {
|
||||
writeJSON(w, http.StatusOK, resp)
|
||||
})
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/auto-cycle", func(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.autoCycleDown {
|
||||
writeAPIErr(w, http.StatusInternalServerError, "auto-cycle unavailable")
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, f.autoCycle)
|
||||
})
|
||||
mux.HandleFunc("PUT /api/v1/admin/auto-cycle", func(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.autoCycleDown {
|
||||
writeAPIErr(w, http.StatusInternalServerError, "auto-cycle unavailable")
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
IntervalSeconds *int `json:"interval_seconds"`
|
||||
MaxRunSeconds *int `json:"max_run_seconds"`
|
||||
}
|
||||
_ = json.NewDecoder(r.Body).Decode(&req)
|
||||
if req.IntervalSeconds != nil && *req.IntervalSeconds < 60 {
|
||||
writeAPIErr(w, http.StatusBadRequest, "interval_seconds must be >= 60: validation failed")
|
||||
return
|
||||
}
|
||||
if req.MaxRunSeconds != nil && *req.MaxRunSeconds < 0 {
|
||||
writeAPIErr(w, http.StatusBadRequest, "max_run_seconds must be >= 0: validation failed")
|
||||
return
|
||||
}
|
||||
if req.IntervalSeconds != nil {
|
||||
f.autoCycle.IntervalSeconds = *req.IntervalSeconds
|
||||
}
|
||||
if req.MaxRunSeconds != nil {
|
||||
f.autoCycle.MaxRunSeconds = *req.MaxRunSeconds
|
||||
}
|
||||
writeJSON(w, http.StatusOK, f.autoCycle)
|
||||
})
|
||||
mux.HandleFunc("POST /api/v1/admin/auto-cycle/start", func(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.autoCycleDown {
|
||||
writeAPIErr(w, http.StatusInternalServerError, "auto-cycle unavailable")
|
||||
return
|
||||
}
|
||||
if !f.autoCycle.Enabled {
|
||||
now := time.Now()
|
||||
f.autoCycle.Enabled = true
|
||||
f.autoCycle.Phase = "idle"
|
||||
f.autoCycle.NextRunAt = &now
|
||||
}
|
||||
writeJSON(w, http.StatusOK, f.autoCycle)
|
||||
})
|
||||
mux.HandleFunc("POST /api/v1/admin/auto-cycle/stop", func(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
if f.autoCycleDown {
|
||||
writeAPIErr(w, http.StatusInternalServerError, "auto-cycle unavailable")
|
||||
return
|
||||
}
|
||||
f.autoCycle.Enabled = false
|
||||
f.autoCycle.Phase = "idle"
|
||||
f.autoCycle.NextRunAt = nil
|
||||
f.autoCycle.LastOutcome = "stopped"
|
||||
writeJSON(w, http.StatusOK, f.autoCycle)
|
||||
})
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/registry", func(w http.ResponseWriter, r *http.Request) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
package dashboard
|
||||
|
||||
import "time"
|
||||
import (
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Wire shapes for control-api's /api/v1/admin/* surface, defined locally
|
||||
// rather than importing internal/httpapi's (unexported) DTOs or
|
||||
@@ -165,3 +168,79 @@ type inboundChecksDTO struct {
|
||||
Ports []int `json:"ports"`
|
||||
ICMP bool `json:"icmp"`
|
||||
}
|
||||
|
||||
// autoCycleDTO mirrors internal/httpapi's autoCycleDTO — the status and
|
||||
// parameters of the automatic check cycle (/api/v1/admin/auto-cycle).
|
||||
type autoCycleDTO struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
IntervalSeconds int `json:"interval_seconds"`
|
||||
MaxRunSeconds int `json:"max_run_seconds"`
|
||||
Phase string `json:"phase"`
|
||||
RunStartedAt *time.Time `json:"run_started_at"`
|
||||
NextRunAt *time.Time `json:"next_run_at"`
|
||||
LastRunStartedAt *time.Time `json:"last_run_started_at"`
|
||||
LastRunFinishedAt *time.Time `json:"last_run_finished_at"`
|
||||
LastOutcome string `json:"last_outcome"`
|
||||
LastError string `json:"last_error"`
|
||||
LastScannedFree int `json:"last_scanned_free"`
|
||||
RunsTotal int `json:"runs_total"`
|
||||
}
|
||||
|
||||
// secondsToMinutes renders seconds as minutes for the settings form: a
|
||||
// whole number when divisible by 60, otherwise a decimal ("1.5").
|
||||
func secondsToMinutes(sec int) string {
|
||||
return strconv.FormatFloat(float64(sec)/60, 'f', -1, 64)
|
||||
}
|
||||
|
||||
// IntervalMinutes and MaxRunMinutes are used by the settings template: the
|
||||
// UI works in minutes, the API in seconds.
|
||||
func (a autoCycleDTO) IntervalMinutes() string { return secondsToMinutes(a.IntervalSeconds) }
|
||||
func (a autoCycleDTO) MaxRunMinutes() string { return secondsToMinutes(a.MaxRunSeconds) }
|
||||
|
||||
// PhaseLabel is the Russian description of the current phase.
|
||||
func (a autoCycleDTO) PhaseLabel() string {
|
||||
switch a.Phase {
|
||||
case "running":
|
||||
return "идёт проверка"
|
||||
case "waiting":
|
||||
return "пауза между циклами"
|
||||
case "idle":
|
||||
return "ожидает запуска"
|
||||
default:
|
||||
return a.Phase
|
||||
}
|
||||
}
|
||||
|
||||
// OutcomeLabel is the Russian description of the last cycle's outcome.
|
||||
func (a autoCycleDTO) OutcomeLabel() string {
|
||||
switch a.LastOutcome {
|
||||
case "completed":
|
||||
return "завершён"
|
||||
case "no_free_ips":
|
||||
return "нет свободных IP"
|
||||
case "timeout":
|
||||
return "превышено время ожидания"
|
||||
case "error":
|
||||
return "ошибка"
|
||||
case "stopped":
|
||||
return "остановлен"
|
||||
case "":
|
||||
return "—"
|
||||
default:
|
||||
return a.LastOutcome
|
||||
}
|
||||
}
|
||||
|
||||
// OutcomePillClass picks the pill style for the last outcome.
|
||||
func (a autoCycleDTO) OutcomePillClass() string {
|
||||
switch a.LastOutcome {
|
||||
case "completed":
|
||||
return "pill-success"
|
||||
case "no_free_ips", "timeout", "stopped":
|
||||
return "pill-warning"
|
||||
case "error":
|
||||
return "pill-danger"
|
||||
default:
|
||||
return "pill-neutral"
|
||||
}
|
||||
}
|
||||
@@ -16,6 +16,9 @@ type overviewData struct {
|
||||
PollSeconds int
|
||||
Query string
|
||||
StatusFilter string
|
||||
// AutoCycle is nil when the auto-cycle status could not be fetched; the
|
||||
// indicator is then simply hidden (a non-fatal failure).
|
||||
AutoCycle *autoCycleDTO
|
||||
}
|
||||
|
||||
func (s *Server) loadOverview(r *http.Request) (overviewData, error) {
|
||||
@@ -28,6 +31,12 @@ func (s *Server) loadOverview(r *http.Request) (overviewData, error) {
|
||||
if err != nil {
|
||||
return overviewData{}, err
|
||||
}
|
||||
var autoCycle *autoCycleDTO
|
||||
if ac, acErr := s.CA.GetAutoCycle(ctx); acErr != nil {
|
||||
s.Log.Warn("overview: auto-cycle status unavailable", "err", acErr)
|
||||
} else {
|
||||
autoCycle = &ac
|
||||
}
|
||||
q := strings.TrimSpace(r.URL.Query().Get("q"))
|
||||
resultFilter := r.URL.Query().Get("status")
|
||||
last := lastCompleted(ips, s.Cfg.LastCompletedCount)
|
||||
@@ -40,6 +49,7 @@ func (s *Server) loadOverview(r *http.Request) (overviewData, error) {
|
||||
PollSeconds: s.Cfg.OverviewPollIntervalS,
|
||||
Query: q,
|
||||
StatusFilter: resultFilter,
|
||||
AutoCycle: autoCycle,
|
||||
}, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ package dashboard
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -11,6 +12,8 @@ type settingsPageData struct {
|
||||
PageData
|
||||
Settings orchestratorSettingsDTO
|
||||
Inbound inboundChecksDTO
|
||||
// AutoCycle is the automatic-check-cycle panel's status/parameters.
|
||||
AutoCycle autoCycleDTO
|
||||
}
|
||||
|
||||
func (s *Server) handleSettingsPage(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -19,7 +22,11 @@ func (s *Server) handleSettingsPage(w http.ResponseWriter, r *http.Request) {
|
||||
if err == nil {
|
||||
err = inboundErr
|
||||
}
|
||||
data := settingsPageData{Settings: settings, Inbound: inbound}
|
||||
autoCycle, autoCycleErr := s.CA.GetAutoCycle(r.Context())
|
||||
if err == nil {
|
||||
err = autoCycleErr
|
||||
}
|
||||
data := settingsPageData{Settings: settings, Inbound: inbound, AutoCycle: autoCycle}
|
||||
data.ActiveNav = "settings"
|
||||
data.Banner = bannerFor(err)
|
||||
s.renderPage(w, "settings_page", data)
|
||||
@@ -38,7 +45,11 @@ func (s *Server) renderSettingsForm(w http.ResponseWriter, r *http.Request, acti
|
||||
if actionErr == nil {
|
||||
actionErr = inboundErr
|
||||
}
|
||||
s.renderFragment(w, "settings_form", settingsPageData{Settings: settings, Inbound: inbound}, actionErr)
|
||||
autoCycle, autoCycleErr := s.CA.GetAutoCycle(r.Context())
|
||||
if actionErr == nil {
|
||||
actionErr = autoCycleErr
|
||||
}
|
||||
s.renderFragment(w, "settings_form", settingsPageData{Settings: settings, Inbound: inbound, AutoCycle: autoCycle}, actionErr)
|
||||
}
|
||||
|
||||
func (s *Server) handleSettingsPut(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -90,3 +101,45 @@ func (s *Server) handleInboundChecksPut(w http.ResponseWriter, r *http.Request)
|
||||
_, err := s.CA.PutInboundChecks(r.Context(), ports, icmp)
|
||||
s.renderSettingsForm(w, r, err)
|
||||
}
|
||||
|
||||
// parseMinutes converts a form field holding a (possibly fractional) number
|
||||
// of minutes into whole seconds.
|
||||
func parseMinutes(raw string) (int, error) {
|
||||
f, err := strconv.ParseFloat(strings.TrimSpace(strings.ReplaceAll(raw, ",", ".")), 64)
|
||||
if err != nil || math.IsNaN(f) || math.IsInf(f, 0) || math.Abs(f) > 1e6 {
|
||||
return 0, fmt.Errorf("not a number of minutes: %q", raw)
|
||||
}
|
||||
return int(math.Round(f * 60)), nil
|
||||
}
|
||||
|
||||
// handleAutoCyclePut saves the auto-cycle interval and maximum run duration
|
||||
// (entered in minutes, sent to control-api in seconds). It does not touch
|
||||
// the enabled flag — that's what the start/stop buttons are for.
|
||||
func (s *Server) handleAutoCyclePut(w http.ResponseWriter, r *http.Request) {
|
||||
if err := r.ParseForm(); err != nil {
|
||||
s.renderSettingsForm(w, r, fmt.Errorf("invalid form: %w", err))
|
||||
return
|
||||
}
|
||||
intervalSec, err := parseMinutes(r.PostFormValue("interval_minutes"))
|
||||
if err != nil {
|
||||
s.renderSettingsForm(w, r, &apiErr{Status: http.StatusBadRequest, Message: "интервал должен быть числом минут"})
|
||||
return
|
||||
}
|
||||
maxRunSec, err := parseMinutes(r.PostFormValue("max_run_minutes"))
|
||||
if err != nil {
|
||||
s.renderSettingsForm(w, r, &apiErr{Status: http.StatusBadRequest, Message: "максимальная длительность должна быть числом минут (0 — без лимита)"})
|
||||
return
|
||||
}
|
||||
_, err = s.CA.PutAutoCycle(r.Context(), intervalSec, maxRunSec)
|
||||
s.renderSettingsForm(w, r, err)
|
||||
}
|
||||
|
||||
func (s *Server) handleAutoCycleStart(w http.ResponseWriter, r *http.Request) {
|
||||
_, err := s.CA.StartAutoCycle(r.Context())
|
||||
s.renderSettingsForm(w, r, err)
|
||||
}
|
||||
|
||||
func (s *Server) handleAutoCycleStop(w http.ResponseWriter, r *http.Request) {
|
||||
_, err := s.CA.StopAutoCycle(r.Context())
|
||||
s.renderSettingsForm(w, r, err)
|
||||
}
|
||||
@@ -724,3 +724,186 @@ func TestFilterRegistryItems(t *testing.T) {
|
||||
t.Fatalf("expected q+status combined with AND to exclude non-matching, got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePanelRenders(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
finished := time.Now().Add(-time.Hour)
|
||||
fake.autoCycle = autoCycleDTO{
|
||||
IntervalSeconds: 5400, MaxRunSeconds: 1800, Phase: "waiting",
|
||||
LastRunFinishedAt: &finished, LastOutcome: "timeout", LastError: "checks did not finish", RunsTotal: 4,
|
||||
}
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
page := get(t, ts, "/settings")
|
||||
for _, want := range []string{
|
||||
"Автоматический цикл",
|
||||
"Интервал между циклами (мин)",
|
||||
"Максимальная длительность проверки (мин, 0 = без лимита)",
|
||||
`name="interval_minutes" min="1" step="any" value="90"`,
|
||||
`name="max_run_minutes" min="0" step="any" value="30"`,
|
||||
"Автоцикл выключен",
|
||||
"пауза между циклами",
|
||||
"превышено время ожидания",
|
||||
"checks did not finish",
|
||||
`hx-put="/settings/auto-cycle"`,
|
||||
`hx-post="/settings/auto-cycle/start"`,
|
||||
"Включить",
|
||||
"Сохранить",
|
||||
} {
|
||||
if !strings.Contains(page, want) {
|
||||
t.Fatalf("expected %q on the settings page, got:\n%s", want, page)
|
||||
}
|
||||
}
|
||||
if strings.Contains(page, "Выключить") {
|
||||
t.Fatalf("stop button must be hidden while the auto-cycle is disabled, got:\n%s", page)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStartAndStopButtons(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
body := postForm(t, ts, "POST", "/settings/auto-cycle/start", nil)
|
||||
if !fake.autoCycle.Enabled {
|
||||
t.Fatalf("expected the fake control-api auto-cycle enabled after start")
|
||||
}
|
||||
if !strings.Contains(body, "Автоцикл включён") || !strings.Contains(body, "Выключить") ||
|
||||
!strings.Contains(body, `hx-post="/settings/auto-cycle/stop"`) {
|
||||
t.Fatalf("expected the re-rendered panel to show the enabled state with a stop button, got:\n%s", body)
|
||||
}
|
||||
if strings.Contains(body, "alert-") {
|
||||
t.Fatalf("expected no error banner after a successful start, got:\n%s", body)
|
||||
}
|
||||
|
||||
body = postForm(t, ts, "POST", "/settings/auto-cycle/stop", nil)
|
||||
if fake.autoCycle.Enabled {
|
||||
t.Fatalf("expected the fake control-api auto-cycle disabled after stop")
|
||||
}
|
||||
if !strings.Contains(body, "Автоцикл выключен") || !strings.Contains(body, "Включить") ||
|
||||
!strings.Contains(body, "остановлен") {
|
||||
t.Fatalf("expected the re-rendered panel to show the stopped state, got:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePutConvertsMinutesToSeconds(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
body := postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"30"}, "max_run_minutes": {"10"},
|
||||
})
|
||||
if fake.autoCycle.IntervalSeconds != 1800 || fake.autoCycle.MaxRunSeconds != 600 {
|
||||
t.Fatalf("expected 1800/600 seconds at control-api, got %d/%d",
|
||||
fake.autoCycle.IntervalSeconds, fake.autoCycle.MaxRunSeconds)
|
||||
}
|
||||
if !strings.Contains(body, `name="interval_minutes" min="1" step="any" value="30"`) ||
|
||||
!strings.Contains(body, `name="max_run_minutes" min="0" step="any" value="10"`) {
|
||||
t.Fatalf("expected updated minutes in the re-rendered form, got:\n%s", body)
|
||||
}
|
||||
if fake.autoCycle.Enabled {
|
||||
t.Fatalf("saving parameters must not toggle the auto-cycle")
|
||||
}
|
||||
|
||||
// 0 = no limit.
|
||||
postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"30"}, "max_run_minutes": {"0"},
|
||||
})
|
||||
if fake.autoCycle.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected max_run_seconds 0, got %d", fake.autoCycle.MaxRunSeconds)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePutValidationErrors(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
// Below the control-api minimum (60 s): rejected by control-api with
|
||||
// 400, surfaced as a warning banner; stored value is untouched.
|
||||
body := postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"0.5"}, "max_run_minutes": {"0"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for a too-short interval, got:\n%s", body)
|
||||
}
|
||||
if fake.autoCycle.IntervalSeconds != 3600 {
|
||||
t.Fatalf("expected interval unchanged, got %d", fake.autoCycle.IntervalSeconds)
|
||||
}
|
||||
|
||||
// Negative max run.
|
||||
body = postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"10"}, "max_run_minutes": {"-5"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") {
|
||||
t.Fatalf("expected client error banner for a negative max run, got:\n%s", body)
|
||||
}
|
||||
|
||||
// Non-numeric input is caught by the dashboard itself.
|
||||
body = postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"abc"}, "max_run_minutes": {"0"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") || !strings.Contains(body, "интервал должен быть числом минут") {
|
||||
t.Fatalf("expected a Russian banner for a non-numeric interval, got:\n%s", body)
|
||||
}
|
||||
body = postForm(t, ts, "PUT", "/settings/auto-cycle", map[string][]string{
|
||||
"interval_minutes": {"10"}, "max_run_minutes": {"abc"},
|
||||
})
|
||||
if !strings.Contains(body, "alert-warning") || !strings.Contains(body, "максимальная длительность") {
|
||||
t.Fatalf("expected a Russian banner for a non-numeric max run, got:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOverviewShowsAutoCycleIndicatorWhenEnabled(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
next := time.Now().Add(30 * time.Minute)
|
||||
fake.autoCycle = autoCycleDTO{Enabled: true, IntervalSeconds: 3600, Phase: "waiting", NextRunAt: &next}
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
page := get(t, ts, "/overview")
|
||||
for _, want := range []string{"Автоцикл активен", "пауза между циклами", "следующий запуск"} {
|
||||
if !strings.Contains(page, want) {
|
||||
t.Fatalf("expected %q in the overview indicator, got:\n%s", want, page)
|
||||
}
|
||||
}
|
||||
// The indicator sits inside the stats block, so the DOM order
|
||||
// stats -> filter -> tables is preserved.
|
||||
indIdx := strings.Index(page, "Автоцикл активен")
|
||||
filterIdx := strings.Index(page, `id="overview-filter"`)
|
||||
tablesIdx := strings.Index(page, `id="overview-tables"`)
|
||||
if !(indIdx < filterIdx && filterIdx < tablesIdx) {
|
||||
t.Fatalf("expected indicator -> filter -> tables order, got %d/%d/%d", indIdx, filterIdx, tablesIdx)
|
||||
}
|
||||
|
||||
// The polled fragment refreshes it through the stats OOB block.
|
||||
frag := get(t, ts, "/overview/fragment")
|
||||
oobIdx := strings.Index(frag, `id="overview-stats" hx-swap-oob="true"`)
|
||||
if oobIdx < 0 || !strings.Contains(frag[oobIdx:], "Автоцикл активен") {
|
||||
t.Fatalf("expected the indicator inside the OOB stats block, got:\n%s", frag)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOverviewHidesAutoCycleIndicatorWhenDisabledOrUnavailable(t *testing.T) {
|
||||
fake, caURL := newFakeControlAPI(t)
|
||||
ts := newTestServer(t, caURL)
|
||||
|
||||
// Disabled: no indicator.
|
||||
if page := get(t, ts, "/overview"); strings.Contains(page, "Автоцикл активен") {
|
||||
t.Fatalf("indicator must be hidden while the auto-cycle is disabled, got:\n%s", page)
|
||||
}
|
||||
|
||||
// Enabled but the auto-cycle API fails: the overview still renders
|
||||
// (non-fatal), without the indicator and without an error banner.
|
||||
fake.mu.Lock()
|
||||
fake.autoCycle.Enabled = true
|
||||
fake.autoCycleDown = true
|
||||
fake.mu.Unlock()
|
||||
page := get(t, ts, "/overview")
|
||||
if strings.Contains(page, "Автоцикл активен") {
|
||||
t.Fatalf("indicator must be hidden when the auto-cycle API fails, got:\n%s", page)
|
||||
}
|
||||
if strings.Contains(page, "alert-") {
|
||||
t.Fatalf("an auto-cycle failure must not raise an error banner on the overview, got:\n%s", page)
|
||||
}
|
||||
if !strings.Contains(page, "всего IP") {
|
||||
t.Fatalf("expected the overview to still render its stats, got:\n%s", page)
|
||||
}
|
||||
}
|
||||
@@ -45,6 +45,9 @@ func (s *Server) routes(mux *http.ServeMux) {
|
||||
mux.HandleFunc("GET /settings", s.handleSettingsPage)
|
||||
mux.HandleFunc("PUT /settings", s.handleSettingsPut)
|
||||
mux.HandleFunc("PUT /settings/inbound-checks", s.handleInboundChecksPut)
|
||||
mux.HandleFunc("PUT /settings/auto-cycle", s.handleAutoCyclePut)
|
||||
mux.HandleFunc("POST /settings/auto-cycle/start", s.handleAutoCycleStart)
|
||||
mux.HandleFunc("POST /settings/auto-cycle/stop", s.handleAutoCycleStop)
|
||||
|
||||
mux.Handle("GET /static/", http.StripPrefix("/static/", http.FileServerFS(staticSubFS())))
|
||||
}
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
<div class="stat-card{{if eq $state "failed"}} bad{{else if eq $state "checking"}} accented{{end}}"><span class="value">{{$count}}</span><span class="label">{{$state}}</span></div>
|
||||
{{end}}
|
||||
</div>
|
||||
{{with .AutoCycle}}{{if .Enabled}}
|
||||
<p class="muted" id="overview-auto-cycle" style="margin-top:12px"><span class="pill pill-info">Автоцикл активен</span> · {{.PhaseLabel}} · следующий запуск: {{fmtTime .NextRunAt}}</p>
|
||||
{{end}}{{end}}
|
||||
{{end}}
|
||||
|
||||
{{/* Out-of-band counterpart of overview_stats, appended to every
|
||||
|
||||
@@ -28,6 +28,47 @@
|
||||
{{end}}
|
||||
|
||||
{{define "settings_form"}}
|
||||
<div class="panel" id="auto-cycle-panel">
|
||||
<div class="panel-body">
|
||||
<h2 class="section-title" style="margin-top:0">Автоматический цикл</h2>
|
||||
<p class="muted" style="margin-bottom:16px">Повторяет сценарий оператора: «Очистить всё» → «Сканировать Floating IP» →
|
||||
ожидание, пока все адреса очереди пройдут проверку → пауза и новый цикл. Интервал отсчитывается от завершения
|
||||
предыдущего цикла. Пока автоцикл включён, периодическое сканирование по <code>fip_scan_interval_seconds</code>
|
||||
не выполняется. Выключение не прерывает проверки, которые уже идут.</p>
|
||||
<p style="margin-bottom:16px">
|
||||
<span class="pill {{if .AutoCycle.Enabled}}pill-success{{else}}pill-neutral{{end}}">{{if .AutoCycle.Enabled}}Автоцикл включён{{else}}Автоцикл выключен{{end}}</span>
|
||||
· фаза: {{.AutoCycle.PhaseLabel}}
|
||||
</p>
|
||||
<div class="muted" style="margin-bottom:16px">
|
||||
<div>Последний запуск: {{fmtTime .AutoCycle.LastRunStartedAt}}</div>
|
||||
<div>Завершён: {{fmtTime .AutoCycle.LastRunFinishedAt}}</div>
|
||||
<div>Следующий запуск: {{fmtTime .AutoCycle.NextRunAt}}</div>
|
||||
<div>Результат последнего цикла: <span class="pill {{.AutoCycle.OutcomePillClass}}">{{.AutoCycle.OutcomeLabel}}</span>{{if .AutoCycle.LastError}} — {{.AutoCycle.LastError}}{{end}}</div>
|
||||
<div>Свободных адресов в последнем скане: {{.AutoCycle.LastScannedFree}} · завершённых циклов: {{.AutoCycle.RunsTotal}}</div>
|
||||
</div>
|
||||
<form hx-put="/settings/auto-cycle" hx-target="#settings-form-wrap" hx-swap="innerHTML">
|
||||
<div class="field-row">
|
||||
<div class="field">
|
||||
<label for="auto_cycle_interval">Интервал между циклами (мин)</label>
|
||||
<input type="number" id="auto_cycle_interval" name="interval_minutes" min="1" step="any" value="{{.AutoCycle.IntervalMinutes}}" required>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="auto_cycle_max_run">Максимальная длительность проверки (мин, 0 = без лимита)</label>
|
||||
<input type="number" id="auto_cycle_max_run" name="max_run_minutes" min="0" step="any" value="{{.AutoCycle.MaxRunMinutes}}" required>
|
||||
</div>
|
||||
<button type="submit" class="btn btn-primary">Сохранить</button>
|
||||
</div>
|
||||
</form>
|
||||
<div style="margin-top:16px">
|
||||
{{if .AutoCycle.Enabled}}
|
||||
<button type="button" class="btn btn-primary" hx-post="/settings/auto-cycle/stop" hx-target="#settings-form-wrap" hx-swap="innerHTML">Выключить</button>
|
||||
{{else}}
|
||||
<button type="button" class="btn btn-primary" hx-post="/settings/auto-cycle/start" hx-target="#settings-form-wrap" hx-swap="innerHTML">Включить</button>
|
||||
{{end}}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="panel">
|
||||
<div class="panel-body">
|
||||
<p class="muted" style="margin-bottom:16px">Пауза между привязкой Floating IP к валидатору и началом self-check —
|
||||
|
||||
@@ -34,6 +34,9 @@ var proberHeartbeatSchema string
|
||||
//go:embed migrations/0007_ip_registry.sql
|
||||
var ipRegistrySchema string
|
||||
|
||||
//go:embed migrations/0008_auto_cycle.sql
|
||||
var autoCycleSchema string
|
||||
|
||||
// migrations is the ordered list of schema versions. Each entry's SQL is
|
||||
// applied, in order, for any version greater than the database's current
|
||||
// PRAGMA user_version — so a fresh database walks the whole list and an
|
||||
@@ -49,6 +52,7 @@ var migrations = []struct {
|
||||
{5, unboundedSitesSchema},
|
||||
{6, proberHeartbeatSchema},
|
||||
{7, ipRegistrySchema},
|
||||
{8, autoCycleSchema},
|
||||
}
|
||||
|
||||
type DB struct {
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
-- Automatic check cycle (see docs/USAGE.md): optionally repeats the
|
||||
-- "clear queue -> scan floating IPs -> wait for all checks to finish ->
|
||||
-- wait interval" scenario. Singleton row, kept apart from `settings` so that
|
||||
-- a full-overwrite PUT /config/orchestrator never resets these fields.
|
||||
-- State is persisted so the cycle survives a control-api restart.
|
||||
|
||||
CREATE TABLE auto_cycle (
|
||||
id INTEGER PRIMARY KEY CHECK (id = 1),
|
||||
enabled INTEGER NOT NULL DEFAULT 0,
|
||||
interval_seconds INTEGER NOT NULL DEFAULT 3600,
|
||||
max_run_seconds INTEGER NOT NULL DEFAULT 0,
|
||||
phase TEXT NOT NULL DEFAULT 'idle',
|
||||
run_started_at TIMESTAMP,
|
||||
next_run_at TIMESTAMP,
|
||||
last_run_started_at TIMESTAMP,
|
||||
last_run_finished_at TIMESTAMP,
|
||||
last_outcome TEXT NOT NULL DEFAULT '',
|
||||
last_error TEXT NOT NULL DEFAULT '',
|
||||
last_scanned_free INTEGER NOT NULL DEFAULT 0,
|
||||
runs_total INTEGER NOT NULL DEFAULT 0,
|
||||
updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
INSERT INTO auto_cycle (id, enabled, interval_seconds, max_run_seconds, phase) VALUES (1, 0, 3600, 0, 'idle');
|
||||
@@ -226,3 +226,49 @@ type InboundChecksSettings struct {
|
||||
CreatedAt time.Time
|
||||
UpdatedAt time.Time
|
||||
}
|
||||
|
||||
// Auto-cycle phases and outcomes (see AutoCycle).
|
||||
const (
|
||||
AutoCyclePhaseIdle = "idle"
|
||||
AutoCyclePhaseRunning = "running"
|
||||
AutoCyclePhaseWaiting = "waiting"
|
||||
|
||||
AutoCycleOutcomeCompleted = "completed"
|
||||
AutoCycleOutcomeNoFreeIPs = "no_free_ips"
|
||||
AutoCycleOutcomeTimeout = "timeout"
|
||||
AutoCycleOutcomeError = "error"
|
||||
AutoCycleOutcomeStopped = "stopped"
|
||||
)
|
||||
|
||||
// AutoCycle is the singleton row describing the automatic check cycle:
|
||||
// its configuration (Enabled, IntervalSeconds, MaxRunSeconds) and its
|
||||
// persisted runtime state (Phase and timestamps). MaxRunSeconds == 0 means
|
||||
// no limit on how long a run may wait for checks to finish.
|
||||
type AutoCycle struct {
|
||||
Enabled bool
|
||||
IntervalSeconds int
|
||||
MaxRunSeconds int
|
||||
Phase string
|
||||
RunStartedAt *time.Time
|
||||
NextRunAt *time.Time
|
||||
LastRunStartedAt *time.Time
|
||||
LastRunFinishedAt *time.Time
|
||||
LastOutcome string
|
||||
LastError string
|
||||
LastScannedFree int
|
||||
RunsTotal int
|
||||
}
|
||||
|
||||
// AutoCycleState is a full replacement of the runtime-state columns of the
|
||||
// auto_cycle row (everything except configuration and enabled flag).
|
||||
type AutoCycleState struct {
|
||||
Phase string
|
||||
RunStartedAt *time.Time
|
||||
NextRunAt *time.Time
|
||||
LastRunStartedAt *time.Time
|
||||
LastRunFinishedAt *time.Time
|
||||
LastOutcome string
|
||||
LastError string
|
||||
LastScannedFree int
|
||||
RunsTotal int
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// MinAutoCycleIntervalSeconds is the smallest allowed pause between
|
||||
// automatic cycles; it protects the OpenStack API from overly frequent
|
||||
// floating-IP scans.
|
||||
const MinAutoCycleIntervalSeconds = 60
|
||||
|
||||
// GetAutoCycle returns the singleton auto_cycle row. Migration 0008 inserts
|
||||
// it, so sql.ErrNoRows would indicate a broken database, not a normal case.
|
||||
func (d *DB) GetAutoCycle(ctx context.Context) (AutoCycle, error) {
|
||||
var a AutoCycle
|
||||
var enabled int
|
||||
var runStarted, nextRun, lastStarted, lastFinished sql.NullString
|
||||
err := d.QueryRowContext(ctx, `
|
||||
SELECT enabled, interval_seconds, max_run_seconds, phase,
|
||||
run_started_at, next_run_at, last_run_started_at, last_run_finished_at,
|
||||
last_outcome, last_error, last_scanned_free, runs_total
|
||||
FROM auto_cycle WHERE id=1
|
||||
`).Scan(&enabled, &a.IntervalSeconds, &a.MaxRunSeconds, &a.Phase,
|
||||
&runStarted, &nextRun, &lastStarted, &lastFinished,
|
||||
&a.LastOutcome, &a.LastError, &a.LastScannedFree, &a.RunsTotal)
|
||||
if err != nil {
|
||||
return AutoCycle{}, err
|
||||
}
|
||||
a.Enabled = enabled != 0
|
||||
if a.RunStartedAt, err = nullStringToTimePtr(runStarted); err != nil {
|
||||
return AutoCycle{}, err
|
||||
}
|
||||
if a.NextRunAt, err = nullStringToTimePtr(nextRun); err != nil {
|
||||
return AutoCycle{}, err
|
||||
}
|
||||
if a.LastRunStartedAt, err = nullStringToTimePtr(lastStarted); err != nil {
|
||||
return AutoCycle{}, err
|
||||
}
|
||||
if a.LastRunFinishedAt, err = nullStringToTimePtr(lastFinished); err != nil {
|
||||
return AutoCycle{}, err
|
||||
}
|
||||
return a, nil
|
||||
}
|
||||
|
||||
// SetAutoCycleParams updates the auto-cycle configuration. A nil argument
|
||||
// leaves the corresponding field unchanged (partial update). Validation:
|
||||
// interval >= MinAutoCycleIntervalSeconds, maxRun >= 0 (0 = no limit).
|
||||
func (d *DB) SetAutoCycleParams(ctx context.Context, intervalSeconds, maxRunSeconds *int) error {
|
||||
if intervalSeconds != nil && *intervalSeconds < MinAutoCycleIntervalSeconds {
|
||||
return fmt.Errorf("interval_seconds must be >= %d: %w", MinAutoCycleIntervalSeconds, ErrValidation)
|
||||
}
|
||||
if maxRunSeconds != nil && *maxRunSeconds < 0 {
|
||||
return fmt.Errorf("max_run_seconds must be >= 0: %w", ErrValidation)
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
if intervalSeconds != nil {
|
||||
if _, err := d.ExecContext(ctx, `
|
||||
UPDATE auto_cycle SET interval_seconds=?, updated_at=? WHERE id=1
|
||||
`, *intervalSeconds, now); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if maxRunSeconds != nil {
|
||||
if _, err := d.ExecContext(ctx, `
|
||||
UPDATE auto_cycle SET max_run_seconds=?, updated_at=? WHERE id=1
|
||||
`, *maxRunSeconds, now); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// SetAutoCycleEnabled flips the enabled flag and resets the runtime phase to
|
||||
// idle (clearing run_started_at). nextRunAt sets next_run_at (nil clears it).
|
||||
// A non-empty outcome is stored as last_outcome (and last_error is cleared);
|
||||
// an empty one leaves the last outcome untouched.
|
||||
func (d *DB) SetAutoCycleEnabled(ctx context.Context, enabled bool, nextRunAt *time.Time, outcome string) error {
|
||||
en := 0
|
||||
if enabled {
|
||||
en = 1
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE auto_cycle SET
|
||||
enabled=?,
|
||||
phase=?,
|
||||
run_started_at=NULL,
|
||||
next_run_at=?,
|
||||
last_outcome=CASE WHEN ?='' THEN last_outcome ELSE ? END,
|
||||
last_error=CASE WHEN ?='' THEN last_error ELSE '' END,
|
||||
updated_at=?
|
||||
WHERE id=1
|
||||
`, en, AutoCyclePhaseIdle, timePtrToDB(nextRunAt), outcome, outcome, outcome, now)
|
||||
return err
|
||||
}
|
||||
|
||||
// UpdateAutoCycleState replaces all runtime-state columns (phase,
|
||||
// timestamps, last outcome/error, counters) in one statement. It does not
|
||||
// touch the configuration or the enabled flag.
|
||||
func (d *DB) UpdateAutoCycleState(ctx context.Context, s AutoCycleState) error {
|
||||
switch s.Phase {
|
||||
case AutoCyclePhaseIdle, AutoCyclePhaseRunning, AutoCyclePhaseWaiting:
|
||||
default:
|
||||
return fmt.Errorf("invalid auto-cycle phase %q: %w", s.Phase, ErrValidation)
|
||||
}
|
||||
now := timeToDB(Now())
|
||||
_, err := d.ExecContext(ctx, `
|
||||
UPDATE auto_cycle SET
|
||||
phase=?, run_started_at=?, next_run_at=?,
|
||||
last_run_started_at=?, last_run_finished_at=?,
|
||||
last_outcome=?, last_error=?, last_scanned_free=?, runs_total=?,
|
||||
updated_at=?
|
||||
WHERE id=1
|
||||
`, s.Phase, timePtrToDB(s.RunStartedAt), timePtrToDB(s.NextRunAt),
|
||||
timePtrToDB(s.LastRunStartedAt), timePtrToDB(s.LastRunFinishedAt),
|
||||
s.LastOutcome, s.LastError, s.LastScannedFree, s.RunsTotal, now)
|
||||
return err
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestAutoCycleDefaultsFromMigration(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
|
||||
a, err := d.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("get auto cycle: %v", err)
|
||||
}
|
||||
if a.Enabled {
|
||||
t.Fatalf("expected disabled by default")
|
||||
}
|
||||
if a.IntervalSeconds != 3600 {
|
||||
t.Fatalf("expected default interval 3600, got %d", a.IntervalSeconds)
|
||||
}
|
||||
if a.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected default max_run_seconds 0, got %d", a.MaxRunSeconds)
|
||||
}
|
||||
if a.Phase != AutoCyclePhaseIdle {
|
||||
t.Fatalf("expected phase idle, got %q", a.Phase)
|
||||
}
|
||||
if a.RunStartedAt != nil || a.NextRunAt != nil || a.LastRunStartedAt != nil || a.LastRunFinishedAt != nil {
|
||||
t.Fatalf("expected all timestamps unset, got %+v", a)
|
||||
}
|
||||
if a.LastOutcome != "" || a.LastError != "" || a.LastScannedFree != 0 || a.RunsTotal != 0 {
|
||||
t.Fatalf("expected empty last-run info, got %+v", a)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleParamsRoundTripAndPartialUpdate(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
|
||||
interval, maxRun := 120, 900
|
||||
if err := d.SetAutoCycleParams(ctx, &interval, &maxRun); err != nil {
|
||||
t.Fatalf("set params: %v", err)
|
||||
}
|
||||
a, err := d.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("get: %v", err)
|
||||
}
|
||||
if a.IntervalSeconds != 120 || a.MaxRunSeconds != 900 {
|
||||
t.Fatalf("expected 120/900, got %d/%d", a.IntervalSeconds, a.MaxRunSeconds)
|
||||
}
|
||||
|
||||
// Partial: only max_run_seconds changes.
|
||||
newMax := 0
|
||||
if err := d.SetAutoCycleParams(ctx, nil, &newMax); err != nil {
|
||||
t.Fatalf("partial set: %v", err)
|
||||
}
|
||||
a, _ = d.GetAutoCycle(ctx)
|
||||
if a.IntervalSeconds != 120 || a.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected 120/0 after partial update, got %d/%d", a.IntervalSeconds, a.MaxRunSeconds)
|
||||
}
|
||||
|
||||
// Partial: only interval changes.
|
||||
newInterval := 60
|
||||
if err := d.SetAutoCycleParams(ctx, &newInterval, nil); err != nil {
|
||||
t.Fatalf("partial set interval: %v", err)
|
||||
}
|
||||
a, _ = d.GetAutoCycle(ctx)
|
||||
if a.IntervalSeconds != 60 || a.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected 60/0, got %d/%d", a.IntervalSeconds, a.MaxRunSeconds)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleParamsValidation(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
|
||||
low := 59
|
||||
if err := d.SetAutoCycleParams(ctx, &low, nil); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation for interval 59, got %v", err)
|
||||
}
|
||||
neg := -1
|
||||
if err := d.SetAutoCycleParams(ctx, nil, &neg); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation for negative max_run_seconds, got %v", err)
|
||||
}
|
||||
// A rejected request must not partially apply the valid half.
|
||||
ok := 300
|
||||
if err := d.SetAutoCycleParams(ctx, &ok, &neg); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation, got %v", err)
|
||||
}
|
||||
a, _ := d.GetAutoCycle(ctx)
|
||||
if a.IntervalSeconds != 3600 {
|
||||
t.Fatalf("interval must stay 3600 after rejected update, got %d", a.IntervalSeconds)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleEnabledAndStateRoundTrip(t *testing.T) {
|
||||
d, ctx := newTestDB(t)
|
||||
|
||||
next := Now()
|
||||
if err := d.SetAutoCycleEnabled(ctx, true, &next, ""); err != nil {
|
||||
t.Fatalf("enable: %v", err)
|
||||
}
|
||||
a, _ := d.GetAutoCycle(ctx)
|
||||
if !a.Enabled || a.Phase != AutoCyclePhaseIdle {
|
||||
t.Fatalf("expected enabled+idle, got %+v", a)
|
||||
}
|
||||
if a.NextRunAt == nil || !a.NextRunAt.Equal(next) {
|
||||
t.Fatalf("expected next_run_at=%v, got %v", next, a.NextRunAt)
|
||||
}
|
||||
|
||||
started := next.Add(time.Second)
|
||||
finished := next.Add(time.Minute)
|
||||
nextRun := finished.Add(time.Hour)
|
||||
st := AutoCycleState{
|
||||
Phase: AutoCyclePhaseWaiting,
|
||||
NextRunAt: &nextRun,
|
||||
LastRunStartedAt: &started,
|
||||
LastRunFinishedAt: &finished,
|
||||
LastOutcome: AutoCycleOutcomeCompleted,
|
||||
LastError: "boom",
|
||||
LastScannedFree: 3,
|
||||
RunsTotal: 7,
|
||||
}
|
||||
if err := d.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
t.Fatalf("update state: %v", err)
|
||||
}
|
||||
a, _ = d.GetAutoCycle(ctx)
|
||||
if !a.Enabled {
|
||||
t.Fatalf("UpdateAutoCycleState must not touch the enabled flag")
|
||||
}
|
||||
if a.Phase != AutoCyclePhaseWaiting || a.LastOutcome != AutoCycleOutcomeCompleted ||
|
||||
a.LastError != "boom" || a.LastScannedFree != 3 || a.RunsTotal != 7 {
|
||||
t.Fatalf("state mismatch: %+v", a)
|
||||
}
|
||||
if a.RunStartedAt != nil {
|
||||
t.Fatalf("expected run_started_at nil, got %v", a.RunStartedAt)
|
||||
}
|
||||
if a.NextRunAt == nil || !a.NextRunAt.Equal(nextRun) ||
|
||||
a.LastRunStartedAt == nil || !a.LastRunStartedAt.Equal(started) ||
|
||||
a.LastRunFinishedAt == nil || !a.LastRunFinishedAt.Equal(finished) {
|
||||
t.Fatalf("timestamp mismatch: %+v", a)
|
||||
}
|
||||
|
||||
// Disable with an outcome: phase back to idle, outcome recorded,
|
||||
// last_error cleared, counters preserved.
|
||||
if err := d.SetAutoCycleEnabled(ctx, false, nil, AutoCycleOutcomeStopped); err != nil {
|
||||
t.Fatalf("disable: %v", err)
|
||||
}
|
||||
a, _ = d.GetAutoCycle(ctx)
|
||||
if a.Enabled || a.Phase != AutoCyclePhaseIdle || a.LastOutcome != AutoCycleOutcomeStopped ||
|
||||
a.LastError != "" || a.NextRunAt != nil || a.RunsTotal != 7 {
|
||||
t.Fatalf("unexpected state after disable: %+v", a)
|
||||
}
|
||||
|
||||
if err := d.UpdateAutoCycleState(ctx, AutoCycleState{Phase: "bogus"}); !errors.Is(err, ErrValidation) {
|
||||
t.Fatalf("expected ErrValidation for bogus phase, got %v", err)
|
||||
}
|
||||
}
|
||||
@@ -117,3 +117,27 @@ type inboundChecksDTO struct {
|
||||
Ports []int `json:"ports"`
|
||||
ICMP bool `json:"icmp"`
|
||||
}
|
||||
|
||||
// autoCycleDTO is the status+parameters object returned by every
|
||||
// /api/v1/admin/auto-cycle endpoint. Times are RFC3339 (null when unset).
|
||||
type autoCycleDTO struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
IntervalSeconds int `json:"interval_seconds"`
|
||||
MaxRunSeconds int `json:"max_run_seconds"`
|
||||
Phase string `json:"phase"`
|
||||
RunStartedAt *time.Time `json:"run_started_at"`
|
||||
NextRunAt *time.Time `json:"next_run_at"`
|
||||
LastRunStartedAt *time.Time `json:"last_run_started_at"`
|
||||
LastRunFinishedAt *time.Time `json:"last_run_finished_at"`
|
||||
LastOutcome string `json:"last_outcome"`
|
||||
LastError string `json:"last_error"`
|
||||
LastScannedFree int `json:"last_scanned_free"`
|
||||
RunsTotal int `json:"runs_total"`
|
||||
}
|
||||
|
||||
// putAutoCycleRequest is the PUT body. Pointers make the update partial:
|
||||
// an omitted field keeps its current value.
|
||||
type putAutoCycleRequest struct {
|
||||
IntervalSeconds *int `json:"interval_seconds"`
|
||||
MaxRunSeconds *int `json:"max_run_seconds"`
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
// Auto-cycle endpoints (see orchestrator/autocycle.go). Every endpoint
|
||||
// answers with the full autoCycleDTO so callers never need a follow-up GET.
|
||||
|
||||
func toAutoCycleDTO(a db.AutoCycle) autoCycleDTO {
|
||||
return autoCycleDTO{
|
||||
Enabled: a.Enabled,
|
||||
IntervalSeconds: a.IntervalSeconds,
|
||||
MaxRunSeconds: a.MaxRunSeconds,
|
||||
Phase: a.Phase,
|
||||
RunStartedAt: a.RunStartedAt,
|
||||
NextRunAt: a.NextRunAt,
|
||||
LastRunStartedAt: a.LastRunStartedAt,
|
||||
LastRunFinishedAt: a.LastRunFinishedAt,
|
||||
LastOutcome: a.LastOutcome,
|
||||
LastError: a.LastError,
|
||||
LastScannedFree: a.LastScannedFree,
|
||||
RunsTotal: a.RunsTotal,
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Server) respondAutoCycle(w http.ResponseWriter, r *http.Request) {
|
||||
ac, err := s.Orch.GetAutoCycle(r.Context())
|
||||
if err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, toAutoCycleDTO(ac))
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminGetAutoCycle(w http.ResponseWriter, r *http.Request) {
|
||||
s.respondAutoCycle(w, r)
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminPutAutoCycle(w http.ResponseWriter, r *http.Request) {
|
||||
var req putAutoCycleRequest
|
||||
if err := readJSON(r, &req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid body: "+err.Error())
|
||||
return
|
||||
}
|
||||
if err := s.DB.SetAutoCycleParams(r.Context(), req.IntervalSeconds, req.MaxRunSeconds); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
s.respondAutoCycle(w, r)
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminStartAutoCycle(w http.ResponseWriter, r *http.Request) {
|
||||
if err := s.Orch.StartAutoCycle(r.Context()); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
s.respondAutoCycle(w, r)
|
||||
}
|
||||
|
||||
func (s *Server) handleAdminStopAutoCycle(w http.ResponseWriter, r *http.Request) {
|
||||
if err := s.Orch.StopAutoCycle(r.Context()); err != nil {
|
||||
writeDBError(w, err)
|
||||
return
|
||||
}
|
||||
s.respondAutoCycle(w, r)
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
package httpapi
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func decodeAutoCycle(t *testing.T, body []byte) autoCycleDTO {
|
||||
t.Helper()
|
||||
var dto autoCycleDTO
|
||||
if err := json.Unmarshal(body, &dto); err != nil {
|
||||
t.Fatalf("unmarshal auto-cycle %q: %v", body, err)
|
||||
}
|
||||
return dto
|
||||
}
|
||||
|
||||
func TestAutoCycleGetDefaults(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
|
||||
resp, body := fc.do(http.MethodGet, "/api/v1/admin/auto-cycle", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto := decodeAutoCycle(t, body)
|
||||
if dto.Enabled || dto.IntervalSeconds != 3600 || dto.MaxRunSeconds != 0 || dto.Phase != "idle" || dto.RunsTotal != 0 {
|
||||
t.Fatalf("unexpected defaults: %+v", dto)
|
||||
}
|
||||
|
||||
// Unset times must be explicit JSON nulls with snake_case keys.
|
||||
var raw map[string]json.RawMessage
|
||||
if err := json.Unmarshal(body, &raw); err != nil {
|
||||
t.Fatalf("unmarshal raw: %v", err)
|
||||
}
|
||||
for _, k := range []string{"enabled", "interval_seconds", "max_run_seconds", "phase", "run_started_at",
|
||||
"next_run_at", "last_run_started_at", "last_run_finished_at", "last_outcome", "last_error",
|
||||
"last_scanned_free", "runs_total"} {
|
||||
v, ok := raw[k]
|
||||
if !ok {
|
||||
t.Fatalf("missing key %q in %s", k, body)
|
||||
}
|
||||
if strings.HasSuffix(k, "_at") && string(v) != "null" {
|
||||
t.Fatalf("expected %s to be null, got %s", k, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePutPartialUpdate(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
|
||||
interval, maxRun := 120, 1800
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", putAutoCycleRequest{IntervalSeconds: &interval, MaxRunSeconds: &maxRun})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto := decodeAutoCycle(t, body)
|
||||
if dto.IntervalSeconds != 120 || dto.MaxRunSeconds != 1800 {
|
||||
t.Fatalf("expected 120/1800, got %+v", dto)
|
||||
}
|
||||
|
||||
// Only max_run_seconds in the body: interval stays.
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", map[string]int{"max_run_seconds": 0})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("partial put: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto = decodeAutoCycle(t, body)
|
||||
if dto.IntervalSeconds != 120 || dto.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected 120/0 after partial update, got %+v", dto)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/auto-cycle", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
if dto = decodeAutoCycle(t, body); dto.IntervalSeconds != 120 {
|
||||
t.Fatalf("expected persisted interval 120, got %+v", dto)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePutValidation(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
|
||||
resp, body := fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", map[string]int{"interval_seconds": 59})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("interval 59: expected 400, got %d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", map[string]int{"max_run_seconds": -1})
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Fatalf("max_run -1: expected 400, got %d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
// Boundary values are accepted.
|
||||
resp, body = fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", map[string]int{"interval_seconds": 60, "max_run_seconds": 0})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("boundary put: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
// Nothing from the rejected requests leaked through.
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/auto-cycle", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
if dto := decodeAutoCycle(t, body); dto.IntervalSeconds != 60 || dto.MaxRunSeconds != 0 {
|
||||
t.Fatalf("expected 60/0, got %+v", dto)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStartStop(t *testing.T) {
|
||||
fc, _, orch, mock := newConfigTestHarness(t)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/start", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("start: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto := decodeAutoCycle(t, body)
|
||||
if !dto.Enabled || dto.Phase != "idle" || dto.NextRunAt == nil {
|
||||
t.Fatalf("expected enabled+idle with next_run_at after start, got %+v", dto)
|
||||
}
|
||||
|
||||
// The engine picks it up on the next step and the API reflects it.
|
||||
orch.AutoCycleStep(t.Context())
|
||||
resp, body = fc.do(http.MethodGet, "/api/v1/admin/auto-cycle", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("get: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto = decodeAutoCycle(t, body)
|
||||
if dto.Phase != "running" || dto.RunStartedAt == nil || dto.LastScannedFree != 1 {
|
||||
t.Fatalf("expected running with 1 scanned FIP, got %+v", dto)
|
||||
}
|
||||
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/stop", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("stop: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto = decodeAutoCycle(t, body)
|
||||
if dto.Enabled || dto.Phase != "idle" || dto.LastOutcome != "stopped" {
|
||||
t.Fatalf("expected disabled/idle/stopped, got %+v", dto)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStartIsIdempotent(t *testing.T) {
|
||||
fc, _, orch, mock := newConfigTestHarness(t)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc-project")
|
||||
|
||||
if resp, body := fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/start", nil); resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("start: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
orch.AutoCycleStep(t.Context())
|
||||
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/start", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("second start: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
dto := decodeAutoCycle(t, body)
|
||||
if !dto.Enabled || dto.Phase != "running" {
|
||||
t.Fatalf("second start must not disturb a running cycle, got %+v", dto)
|
||||
}
|
||||
|
||||
// Stop is idempotent too.
|
||||
for i := 0; i < 2; i++ {
|
||||
resp, body = fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/stop", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("stop #%d: status=%d body=%s", i, resp.StatusCode, body)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCyclePreservesParamsAcrossStartStop(t *testing.T) {
|
||||
fc, _, _, _ := newConfigTestHarness(t)
|
||||
|
||||
interval := 300
|
||||
if resp, body := fc.do(http.MethodPut, "/api/v1/admin/auto-cycle", putAutoCycleRequest{IntervalSeconds: &interval}); resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("put: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/start", nil)
|
||||
resp, body := fc.do(http.MethodPost, "/api/v1/admin/auto-cycle/stop", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("stop: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
if dto := decodeAutoCycle(t, body); dto.IntervalSeconds != 300 {
|
||||
t.Fatalf("expected interval 300 preserved, got %+v", dto)
|
||||
}
|
||||
}
|
||||
@@ -29,6 +29,11 @@ func (s *Server) routes(mux *http.ServeMux) {
|
||||
mux.HandleFunc("POST /api/v1/admin/ips/clear", s.handleAdminClearQueue)
|
||||
mux.HandleFunc("GET /api/v1/admin/validators", s.handleAdminValidators)
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/auto-cycle", s.handleAdminGetAutoCycle)
|
||||
mux.HandleFunc("PUT /api/v1/admin/auto-cycle", s.handleAdminPutAutoCycle)
|
||||
mux.HandleFunc("POST /api/v1/admin/auto-cycle/start", s.handleAdminStartAutoCycle)
|
||||
mux.HandleFunc("POST /api/v1/admin/auto-cycle/stop", s.handleAdminStopAutoCycle)
|
||||
|
||||
mux.HandleFunc("GET /api/v1/admin/registry", s.handleAdminRegistry)
|
||||
mux.HandleFunc("GET /api/v1/admin/registry/{ip}", s.handleAdminRegistryHistory)
|
||||
|
||||
|
||||
@@ -18,6 +18,10 @@ type MockClient struct {
|
||||
// a specific floating-IP ID on its next call, to exercise retry paths.
|
||||
AssociateFailures map[string]error
|
||||
DisassociateFailures map[string]error
|
||||
|
||||
// ListFailure, when non-nil, is returned by every ListFloatingIPs call
|
||||
// until the test resets it to nil — to exercise the scan-error path.
|
||||
ListFailure error
|
||||
}
|
||||
|
||||
func NewMockClient() *MockClient {
|
||||
@@ -59,6 +63,9 @@ func (m *MockClient) GetFloatingIPByAddress(ctx context.Context, address string)
|
||||
func (m *MockClient) ListFloatingIPs(ctx context.Context) ([]FloatingIP, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.ListFailure != nil {
|
||||
return nil, m.ListFailure
|
||||
}
|
||||
out := make([]FloatingIP, 0, len(m.fips))
|
||||
for _, f := range m.fips {
|
||||
out = append(out, *f)
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
// This file implements the automatic check cycle: an optional, repeating
|
||||
// "clear queue -> scan floating IPs -> wait until every queued address has
|
||||
// reached a terminal state -> wait interval" scenario. Steps 1-2 reuse
|
||||
// ClearQueue/ScanFloatingIPs verbatim; step 3 needs no code at all because
|
||||
// Tick already picks up `queued` addresses. All state lives in the database
|
||||
// (db.AutoCycle), so the cycle survives a control-api restart and the
|
||||
// interval/limits can be changed at runtime.
|
||||
|
||||
// GetAutoCycle returns the current auto-cycle configuration and state.
|
||||
func (o *Orchestrator) GetAutoCycle(ctx context.Context) (db.AutoCycle, error) {
|
||||
return o.DB.GetAutoCycle(ctx)
|
||||
}
|
||||
|
||||
// StartAutoCycle enables the auto-cycle; the first cycle begins on the next
|
||||
// AutoCycleStep. It is idempotent: if the cycle is already enabled nothing
|
||||
// changes (in particular a running cycle is not restarted).
|
||||
func (o *Orchestrator) StartAutoCycle(ctx context.Context) error {
|
||||
o.autoCycleMu.Lock()
|
||||
defer o.autoCycleMu.Unlock()
|
||||
|
||||
ac, err := o.DB.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("get auto cycle: %w", err)
|
||||
}
|
||||
if ac.Enabled {
|
||||
return nil
|
||||
}
|
||||
now := db.Now()
|
||||
if err := o.DB.SetAutoCycleEnabled(ctx, true, &now, ""); err != nil {
|
||||
return fmt.Errorf("enable auto cycle: %w", err)
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
||||
"reason": "enabled",
|
||||
"interval_seconds": ac.IntervalSeconds,
|
||||
"max_run_seconds": ac.MaxRunSeconds,
|
||||
}))
|
||||
return nil
|
||||
}
|
||||
|
||||
// StopAutoCycle disables the auto-cycle and returns it to idle. Checks that
|
||||
// are already in flight are NOT cancelled — they finish normally and land in
|
||||
// the registry; only the repetition stops.
|
||||
func (o *Orchestrator) StopAutoCycle(ctx context.Context) error {
|
||||
o.autoCycleMu.Lock()
|
||||
defer o.autoCycleMu.Unlock()
|
||||
|
||||
ac, err := o.DB.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("get auto cycle: %w", err)
|
||||
}
|
||||
if !ac.Enabled {
|
||||
return nil
|
||||
}
|
||||
// "stopped" describes an interrupted run. Stopping during the pause
|
||||
// between cycles must not overwrite the result of the last finished one.
|
||||
outcome := ""
|
||||
if ac.Phase == db.AutoCyclePhaseRunning {
|
||||
outcome = db.AutoCycleOutcomeStopped
|
||||
}
|
||||
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, outcome); err != nil {
|
||||
return fmt.Errorf("disable auto cycle: %w", err)
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_stopped", autoCyclePayload(map[string]any{
|
||||
"phase": ac.Phase,
|
||||
}))
|
||||
return nil
|
||||
}
|
||||
|
||||
// AutoCycleStep advances the auto-cycle state machine by one step. It is
|
||||
// called by the control-api loop right after Tick.
|
||||
func (o *Orchestrator) AutoCycleStep(ctx context.Context) {
|
||||
o.autoCycleStep(ctx, db.Now())
|
||||
}
|
||||
|
||||
// autoCycleStep is AutoCycleStep with an explicit "now", so tests can drive
|
||||
// the state machine deterministically without sleeping.
|
||||
func (o *Orchestrator) autoCycleStep(ctx context.Context, now time.Time) {
|
||||
o.autoCycleMu.Lock()
|
||||
defer o.autoCycleMu.Unlock()
|
||||
|
||||
// Read inside the lock: Start/Stop may have changed the row since the
|
||||
// caller last looked.
|
||||
ac, err := o.DB.GetAutoCycle(ctx)
|
||||
if err != nil {
|
||||
o.Log.Error("auto cycle: read state", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
if !ac.Enabled {
|
||||
if ac.Phase != db.AutoCyclePhaseIdle {
|
||||
if err := o.DB.SetAutoCycleEnabled(ctx, false, nil, ""); err != nil {
|
||||
o.Log.Error("auto cycle: reset phase to idle", "err", err)
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
if ac.Phase == db.AutoCyclePhaseRunning {
|
||||
o.autoCycleCheckRun(ctx, ac, now)
|
||||
return
|
||||
}
|
||||
|
||||
// idle or waiting: start a new cycle once next_run_at has come.
|
||||
if ac.NextRunAt != nil && now.Before(*ac.NextRunAt) {
|
||||
return
|
||||
}
|
||||
o.autoCycleStartRun(ctx, ac, now)
|
||||
}
|
||||
|
||||
// autoCycleStartRun performs steps 1-2 of the scenario (clear the queue,
|
||||
// scan floating IPs) and moves the state to running, or straight to waiting
|
||||
// if there is nothing to wait for.
|
||||
func (o *Orchestrator) autoCycleStartRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
||||
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
||||
next := now.Add(interval)
|
||||
|
||||
st := autoCycleStateOf(ac)
|
||||
st.LastRunStartedAt = &now
|
||||
st.RunStartedAt = nil
|
||||
|
||||
fail := func(step string, err error) {
|
||||
o.Log.Error("auto cycle: step failed", "step", step, "err", err)
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeError
|
||||
st.LastError = fmt.Sprintf("%s: %v", step, err)
|
||||
if uerr := o.DB.UpdateAutoCycleState(ctx, st); uerr != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", uerr)
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_error", autoCyclePayload(map[string]any{
|
||||
"step": step,
|
||||
"error": err.Error(),
|
||||
}))
|
||||
}
|
||||
|
||||
if _, err := o.ClearQueue(ctx); err != nil {
|
||||
fail("clear queue", err)
|
||||
return
|
||||
}
|
||||
_, scanned, err := o.ScanFloatingIPs(ctx)
|
||||
if err != nil {
|
||||
fail("scan floating ips", err)
|
||||
return
|
||||
}
|
||||
st.LastScannedFree = scanned
|
||||
st.LastError = ""
|
||||
|
||||
if scanned == 0 {
|
||||
// Nothing was queued; waiting for completion would never end.
|
||||
o.Log.Info("auto cycle: no free floating ips, waiting for next interval")
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeNoFreeIPs
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
st.Phase = db.AutoCyclePhaseRunning
|
||||
st.RunStartedAt = &now
|
||||
st.NextRunAt = nil
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_started", autoCyclePayload(map[string]any{
|
||||
"reason": "cycle",
|
||||
"scanned_free": scanned,
|
||||
}))
|
||||
}
|
||||
|
||||
// autoCycleCheckRun handles the running phase: finish the cycle once every
|
||||
// queued address is terminal, or give up after max_run_seconds.
|
||||
func (o *Orchestrator) autoCycleCheckRun(ctx context.Context, ac db.AutoCycle, now time.Time) {
|
||||
items, err := o.DB.ListIPs(ctx)
|
||||
if err != nil {
|
||||
o.Log.Error("auto cycle: list ips", "err", err)
|
||||
return
|
||||
}
|
||||
|
||||
interval := time.Duration(ac.IntervalSeconds) * time.Second
|
||||
next := now.Add(interval)
|
||||
|
||||
// An empty queue counts as finished: right after the scan it cannot be
|
||||
// empty (scanned > 0), so it only happens when an operator cleared or
|
||||
// deleted every address mid-cycle — and then there is nothing to wait for
|
||||
// (with max_run_seconds=0 the cycle would otherwise hang forever).
|
||||
allTerminal := true
|
||||
for _, it := range items {
|
||||
if !isTerminalIPState(it.State) {
|
||||
allTerminal = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if allTerminal {
|
||||
st := autoCycleStateOf(ac)
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeCompleted
|
||||
st.LastError = ""
|
||||
st.RunsTotal++
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_completed", autoCyclePayload(map[string]any{
|
||||
"addresses": len(items),
|
||||
"runs": st.RunsTotal,
|
||||
}))
|
||||
return
|
||||
}
|
||||
|
||||
if ac.MaxRunSeconds > 0 && ac.RunStartedAt != nil &&
|
||||
now.Sub(*ac.RunStartedAt) > time.Duration(ac.MaxRunSeconds)*time.Second {
|
||||
// The queue is left untouched: the next cycle clears it anyway, and
|
||||
// an operator can still inspect what got stuck.
|
||||
st := autoCycleStateOf(ac)
|
||||
st.Phase = db.AutoCyclePhaseWaiting
|
||||
st.RunStartedAt = nil
|
||||
st.NextRunAt = &next
|
||||
st.LastRunFinishedAt = &now
|
||||
st.LastOutcome = db.AutoCycleOutcomeTimeout
|
||||
st.LastError = fmt.Sprintf("checks did not finish within %d seconds", ac.MaxRunSeconds)
|
||||
if err := o.DB.UpdateAutoCycleState(ctx, st); err != nil {
|
||||
o.Log.Error("auto cycle: save state", "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "auto_cycle_timeout", autoCyclePayload(map[string]any{
|
||||
"max_run_seconds": ac.MaxRunSeconds,
|
||||
"addresses": len(items),
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
// isTerminalIPState reports whether an address has finished its check cycle
|
||||
// for good (its result, if any, is already written to the registry).
|
||||
// db.IPOccupied counts: such an address never enters the check cycle.
|
||||
func isTerminalIPState(state string) bool {
|
||||
switch state {
|
||||
case db.IPDone, db.IPFailed, db.IPOccupied:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func autoCycleStateOf(ac db.AutoCycle) db.AutoCycleState {
|
||||
return db.AutoCycleState{
|
||||
Phase: ac.Phase,
|
||||
RunStartedAt: ac.RunStartedAt,
|
||||
NextRunAt: ac.NextRunAt,
|
||||
LastRunStartedAt: ac.LastRunStartedAt,
|
||||
LastRunFinishedAt: ac.LastRunFinishedAt,
|
||||
LastOutcome: ac.LastOutcome,
|
||||
LastError: ac.LastError,
|
||||
LastScannedFree: ac.LastScannedFree,
|
||||
RunsTotal: ac.RunsTotal,
|
||||
}
|
||||
}
|
||||
|
||||
func autoCyclePayload(m map[string]any) string {
|
||||
b, err := json.Marshal(m)
|
||||
if err != nil {
|
||||
return "{}"
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
@@ -0,0 +1,544 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/db"
|
||||
)
|
||||
|
||||
func getAutoCycle(t *testing.T, d *db.DB) db.AutoCycle {
|
||||
t.Helper()
|
||||
ac, err := d.GetAutoCycle(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("get auto cycle: %v", err)
|
||||
}
|
||||
return ac
|
||||
}
|
||||
|
||||
func setAutoCycleParams(t *testing.T, d *db.DB, interval, maxRun int) {
|
||||
t.Helper()
|
||||
if err := d.SetAutoCycleParams(context.Background(), &interval, &maxRun); err != nil {
|
||||
t.Fatalf("set auto cycle params: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// finishAllIPs simulates completed checks by moving every queued address to
|
||||
// the given terminal state directly (the full check pipeline is covered by
|
||||
// TestHappyPath).
|
||||
func finishAllIPs(t *testing.T, d *db.DB, state string) {
|
||||
t.Helper()
|
||||
if _, err := d.ExecContext(context.Background(), `UPDATE ip_queue SET state=?`, state); err != nil {
|
||||
t.Fatalf("finish ips: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func queuedAddresses(t *testing.T, d *db.DB) map[string]string {
|
||||
t.Helper()
|
||||
items, err := d.ListIPs(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("list ips: %v", err)
|
||||
}
|
||||
out := make(map[string]string, len(items))
|
||||
for _, it := range items {
|
||||
out[it.IPAddress] = it.State
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func countEvents(t *testing.T, d *db.DB, eventType string) int {
|
||||
t.Helper()
|
||||
var n int
|
||||
if err := d.QueryRowContext(context.Background(), `SELECT COUNT(*) FROM events WHERE event_type=?`, eventType).Scan(&n); err != nil {
|
||||
t.Fatalf("count events: %v", err)
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func TestAutoCycleDisabledIsNoOp(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
|
||||
t.Fatalf("seed queue: %v", err)
|
||||
}
|
||||
|
||||
o.autoCycleStep(ctx, db.Now())
|
||||
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle {
|
||||
t.Fatalf("expected disabled+idle, got %+v", ac)
|
||||
}
|
||||
if got := queuedAddresses(t, d); len(got) != 1 || got["9.9.9.9"] != db.IPQueued {
|
||||
t.Fatalf("queue must be untouched while disabled, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleDisabledResetsStalePhase(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, _ := newTestOrchestrator(t, 180)
|
||||
now := db.Now()
|
||||
if err := d.UpdateAutoCycleState(ctx, db.AutoCycleState{Phase: db.AutoCyclePhaseRunning, RunStartedAt: &now}); err != nil {
|
||||
t.Fatalf("update state: %v", err)
|
||||
}
|
||||
|
||||
o.autoCycleStep(ctx, now)
|
||||
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseIdle || ac.RunStartedAt != nil {
|
||||
t.Fatalf("expected phase reset to idle, got %+v", ac)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStartClearsQueueAndScans(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
mock.Seed("fip-2", "2.2.2.2", "svc")
|
||||
mock.SeedWithPort("fip-3", "3.3.3.3", "svc", "someone-elses-port")
|
||||
if err := d.SeedQueue(ctx, []string{"9.9.9.9"}); err != nil {
|
||||
t.Fatalf("seed queue: %v", err)
|
||||
}
|
||||
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
ac := getAutoCycle(t, d)
|
||||
if !ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.NextRunAt == nil {
|
||||
t.Fatalf("expected enabled+idle with next_run_at set after start, got %+v", ac)
|
||||
}
|
||||
|
||||
now := db.Now()
|
||||
o.autoCycleStep(ctx, now)
|
||||
|
||||
got := queuedAddresses(t, d)
|
||||
if _, stale := got["9.9.9.9"]; stale {
|
||||
t.Fatalf("old queue entry must be cleared, got %v", got)
|
||||
}
|
||||
if len(got) != 2 || got["1.1.1.1"] != db.IPQueued || got["2.2.2.2"] != db.IPQueued {
|
||||
t.Fatalf("expected the two free FIPs queued, got %v", got)
|
||||
}
|
||||
ac = getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected running, got %s", ac.Phase)
|
||||
}
|
||||
if ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(now) {
|
||||
t.Fatalf("expected run_started_at=%v, got %v", now, ac.RunStartedAt)
|
||||
}
|
||||
if ac.LastRunStartedAt == nil || !ac.LastRunStartedAt.Equal(now) {
|
||||
t.Fatalf("expected last_run_started_at=%v, got %v", now, ac.LastRunStartedAt)
|
||||
}
|
||||
if ac.LastScannedFree != 2 {
|
||||
t.Fatalf("expected last_scanned_free=2, got %d", ac.LastScannedFree)
|
||||
}
|
||||
if countEvents(t, d, "queue_cleared") != 1 || countEvents(t, d, "fip_scan") != 1 {
|
||||
t.Fatalf("expected queue_cleared and fip_scan events")
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_started") != 2 { // enable + cycle start
|
||||
t.Fatalf("expected two auto_cycle_started events, got %d", countEvents(t, d, "auto_cycle_started"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleWaitsWhileChecksInProgressAndTickPicksUpQueue(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
mock.Seed("fip-2", "2.2.2.2", "svc")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
||||
t.Fatalf("register validator: %v", err)
|
||||
}
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
// Existing Tick logic starts the checks on its own.
|
||||
o.Tick(ctx)
|
||||
states := queuedAddresses(t, d)
|
||||
inProgress := 0
|
||||
for _, s := range states {
|
||||
if s == db.IPAwaitingSelfCheck {
|
||||
inProgress++
|
||||
}
|
||||
}
|
||||
if inProgress != 1 {
|
||||
t.Fatalf("expected exactly one address claimed by Tick, got %v", states)
|
||||
}
|
||||
|
||||
o.autoCycleStep(ctx, t0.Add(time.Minute))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning || ac.RunsTotal != 0 {
|
||||
t.Fatalf("expected still running, got %+v", ac)
|
||||
}
|
||||
|
||||
// One address done, the other still queued: still not finished.
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPDone, "1.1.1.1"); err != nil {
|
||||
t.Fatalf("update: %v", err)
|
||||
}
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPQueued, "2.2.2.2"); err != nil {
|
||||
t.Fatalf("update: %v", err)
|
||||
}
|
||||
o.autoCycleStep(ctx, t0.Add(2*time.Minute))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected still running with a queued address left, got %+v", ac)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleCompletesAndRepeatsAfterInterval(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
mock.Seed("fip-2", "2.2.2.2", "svc")
|
||||
setAutoCycleParams(t, d, 600, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
// done, failed and occupied are all terminal.
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPDone, "1.1.1.1"); err != nil {
|
||||
t.Fatalf("update: %v", err)
|
||||
}
|
||||
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET state=? WHERE ip_address=?`, db.IPOccupied, "2.2.2.2"); err != nil {
|
||||
t.Fatalf("update: %v", err)
|
||||
}
|
||||
t1 := t0.Add(5 * time.Minute)
|
||||
o.autoCycleStep(ctx, t1)
|
||||
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
|
||||
t.Fatalf("expected waiting/completed, got %+v", ac)
|
||||
}
|
||||
if ac.RunsTotal != 1 {
|
||||
t.Fatalf("expected runs_total=1, got %d", ac.RunsTotal)
|
||||
}
|
||||
wantNext := t1.Add(600 * time.Second)
|
||||
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(wantNext) {
|
||||
t.Fatalf("expected next_run_at=%v (completion + interval), got %v", wantNext, ac.NextRunAt)
|
||||
}
|
||||
if ac.LastRunFinishedAt == nil || !ac.LastRunFinishedAt.Equal(t1) {
|
||||
t.Fatalf("expected last_run_finished_at=%v, got %v", t1, ac.LastRunFinishedAt)
|
||||
}
|
||||
if ac.RunStartedAt != nil {
|
||||
t.Fatalf("expected run_started_at cleared, got %v", ac.RunStartedAt)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_completed") != 1 {
|
||||
t.Fatalf("expected one auto_cycle_completed event")
|
||||
}
|
||||
|
||||
// Before next_run_at: nothing happens, the finished queue is kept.
|
||||
o.autoCycleStep(ctx, wantNext.Add(-time.Second))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting || ac.RunsTotal != 1 {
|
||||
t.Fatalf("expected still waiting, got %+v", ac)
|
||||
}
|
||||
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPDone {
|
||||
t.Fatalf("queue must not be touched while waiting, got %v", got)
|
||||
}
|
||||
|
||||
// At next_run_at: new cycle starts, queue is rebuilt from scratch.
|
||||
o.autoCycleStep(ctx, wantNext)
|
||||
ac = getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected running again, got %+v", ac)
|
||||
}
|
||||
if ac.LastRunStartedAt == nil || !ac.LastRunStartedAt.Equal(wantNext) {
|
||||
t.Fatalf("expected last_run_started_at=%v, got %v", wantNext, ac.LastRunStartedAt)
|
||||
}
|
||||
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPQueued || got["2.2.2.2"] != db.IPQueued {
|
||||
t.Fatalf("expected both addresses re-queued, got %v", got)
|
||||
}
|
||||
if ac.RunsTotal != 1 {
|
||||
t.Fatalf("runs_total only counts completed cycles, got %d", ac.RunsTotal)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleTimeout(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
setAutoCycleParams(t, d, 60, 300)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
o.autoCycleStep(ctx, t0.Add(300*time.Second))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected still running exactly at the limit, got %+v", ac)
|
||||
}
|
||||
|
||||
t1 := t0.Add(301 * time.Second)
|
||||
o.autoCycleStep(ctx, t1)
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeTimeout {
|
||||
t.Fatalf("expected waiting/timeout, got %+v", ac)
|
||||
}
|
||||
if ac.RunsTotal != 0 {
|
||||
t.Fatalf("timeout must not count as completed, got runs_total=%d", ac.RunsTotal)
|
||||
}
|
||||
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t1.Add(60*time.Second)) {
|
||||
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
|
||||
}
|
||||
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPQueued {
|
||||
t.Fatalf("queue must be left untouched on timeout, got %v", got)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_timeout") != 1 {
|
||||
t.Fatalf("expected one auto_cycle_timeout event")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleNoLimitNeverTimesOut(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
o.autoCycleStep(ctx, t0.Add(1000*time.Hour))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("max_run_seconds=0 means no limit, got %+v", ac)
|
||||
}
|
||||
}
|
||||
|
||||
// An operator pressing «Очистить всё» in the middle of a cycle leaves nothing
|
||||
// to wait for; with max_run_seconds=0 the cycle would otherwise hang forever.
|
||||
func TestAutoCycleManualClearMidCycleCompletes(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
setAutoCycleParams(t, d, 60, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected running after start, got %+v", ac)
|
||||
}
|
||||
|
||||
if _, err := o.ClearQueue(ctx); err != nil {
|
||||
t.Fatalf("manual clear: %v", err)
|
||||
}
|
||||
|
||||
t1 := t0.Add(5 * time.Second)
|
||||
o.autoCycleStep(ctx, t1)
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
|
||||
t.Fatalf("expected waiting/completed after the queue was emptied, got %+v", ac)
|
||||
}
|
||||
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t1.Add(60*time.Second)) {
|
||||
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleNoFreeIPs(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.SeedWithPort("fip-1", "1.1.1.1", "svc", "someone-elses-port")
|
||||
setAutoCycleParams(t, d, 120, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
|
||||
now := db.Now()
|
||||
o.autoCycleStep(ctx, now)
|
||||
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeNoFreeIPs {
|
||||
t.Fatalf("expected waiting/no_free_ips, got %+v", ac)
|
||||
}
|
||||
if ac.LastScannedFree != 0 {
|
||||
t.Fatalf("expected last_scanned_free=0, got %d", ac.LastScannedFree)
|
||||
}
|
||||
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(now.Add(120*time.Second)) {
|
||||
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
|
||||
}
|
||||
if ac.RunStartedAt != nil {
|
||||
t.Fatalf("run_started_at must stay unset, got %v", ac.RunStartedAt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleOpenStackErrorRetriesNextInterval(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
mock.ListFailure = errors.New("neutron is down")
|
||||
setAutoCycleParams(t, d, 60, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeError {
|
||||
t.Fatalf("expected waiting/error, got %+v", ac)
|
||||
}
|
||||
if !strings.Contains(ac.LastError, "neutron is down") {
|
||||
t.Fatalf("expected last_error to mention the cause, got %q", ac.LastError)
|
||||
}
|
||||
if ac.NextRunAt == nil || !ac.NextRunAt.Equal(t0.Add(60*time.Second)) {
|
||||
t.Fatalf("expected next_run_at=now+interval, got %v", ac.NextRunAt)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_error") != 1 {
|
||||
t.Fatalf("expected one auto_cycle_error event")
|
||||
}
|
||||
if !ac.Enabled {
|
||||
t.Fatalf("an error must not disable the auto-cycle")
|
||||
}
|
||||
|
||||
// OpenStack recovers: the next interval starts a normal cycle and the
|
||||
// stale error is cleared.
|
||||
mock.ListFailure = nil
|
||||
o.autoCycleStep(ctx, t0.Add(60*time.Second))
|
||||
ac = getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseRunning || ac.LastError != "" {
|
||||
t.Fatalf("expected running with cleared error, got %+v", ac)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStartIsIdempotent(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("second start: %v", err)
|
||||
}
|
||||
ac := getAutoCycle(t, d)
|
||||
if !ac.Enabled || ac.Phase != db.AutoCyclePhaseRunning || ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t0) {
|
||||
t.Fatalf("second start must not restart a running cycle, got %+v", ac)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_started") != 2 { // enable + cycle start, not a third
|
||||
t.Fatalf("expected no extra started event, got %d", countEvents(t, d, "auto_cycle_started"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleStopMidCycle(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
||||
t.Fatalf("register validator: %v", err)
|
||||
}
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
o.Tick(ctx) // check in flight
|
||||
|
||||
if err := o.StopAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("stop: %v", err)
|
||||
}
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.LastOutcome != db.AutoCycleOutcomeStopped {
|
||||
t.Fatalf("expected disabled/idle/stopped, got %+v", ac)
|
||||
}
|
||||
if got := queuedAddresses(t, d); got["1.1.1.1"] != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("in-flight check must not be cancelled, got %v", got)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_stopped") != 1 {
|
||||
t.Fatalf("expected one auto_cycle_stopped event")
|
||||
}
|
||||
|
||||
// Further steps do nothing, even far in the future.
|
||||
o.autoCycleStep(ctx, t0.Add(100*time.Hour))
|
||||
ac = getAutoCycle(t, d)
|
||||
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle || ac.RunsTotal != 0 {
|
||||
t.Fatalf("expected no activity after stop, got %+v", ac)
|
||||
}
|
||||
|
||||
// Stopping again is a harmless no-op.
|
||||
if err := o.StopAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("second stop: %v", err)
|
||||
}
|
||||
if countEvents(t, d, "auto_cycle_stopped") != 1 {
|
||||
t.Fatalf("second stop must not emit another event")
|
||||
}
|
||||
}
|
||||
|
||||
// Stopping during the pause between cycles must keep the result of the last
|
||||
// finished cycle visible instead of replacing it with "stopped".
|
||||
func TestAutoCycleStopWhileWaitingKeepsLastOutcome(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
setAutoCycleParams(t, d, 60, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
finishAllIPs(t, d, db.IPDone)
|
||||
o.autoCycleStep(ctx, t0.Add(5*time.Second))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
|
||||
t.Fatalf("precondition: expected waiting/completed, got %+v", ac)
|
||||
}
|
||||
|
||||
if err := o.StopAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("stop: %v", err)
|
||||
}
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Enabled || ac.Phase != db.AutoCyclePhaseIdle {
|
||||
t.Fatalf("expected disabled/idle, got %+v", ac)
|
||||
}
|
||||
if ac.LastOutcome != db.AutoCycleOutcomeCompleted || ac.RunsTotal != 1 {
|
||||
t.Fatalf("stop in the pause must keep last_outcome=completed, got %+v", ac)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAutoCycleSurvivesRestart(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.1.1.1", "svc")
|
||||
setAutoCycleParams(t, d, 300, 0)
|
||||
if err := o.StartAutoCycle(ctx); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
t0 := db.Now()
|
||||
o.autoCycleStep(ctx, t0)
|
||||
|
||||
// A fresh Orchestrator on the same database (a control-api restart)
|
||||
// continues the running phase instead of starting over.
|
||||
o2 := &Orchestrator{DB: d, OS: mock, Cfg: o.Cfg, Agg: o.Agg, Log: o.Log}
|
||||
o2.autoCycleStep(ctx, t0.Add(time.Minute))
|
||||
ac := getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseRunning || ac.RunStartedAt == nil || !ac.RunStartedAt.Equal(t0) {
|
||||
t.Fatalf("expected the running phase to continue, got %+v", ac)
|
||||
}
|
||||
|
||||
finishAllIPs(t, d, db.IPDone)
|
||||
t1 := t0.Add(2 * time.Minute)
|
||||
o2.autoCycleStep(ctx, t1)
|
||||
ac = getAutoCycle(t, d)
|
||||
if ac.Phase != db.AutoCyclePhaseWaiting || ac.LastOutcome != db.AutoCycleOutcomeCompleted {
|
||||
t.Fatalf("expected waiting/completed, got %+v", ac)
|
||||
}
|
||||
|
||||
// A third instance still honours the persisted next_run_at.
|
||||
o3 := &Orchestrator{DB: d, OS: mock, Cfg: o.Cfg, Agg: o.Agg, Log: o.Log}
|
||||
o3.autoCycleStep(ctx, t1.Add(299*time.Second))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseWaiting {
|
||||
t.Fatalf("expected waiting until next_run_at, got %+v", ac)
|
||||
}
|
||||
o3.autoCycleStep(ctx, t1.Add(300*time.Second))
|
||||
if ac := getAutoCycle(t, d); ac.Phase != db.AutoCyclePhaseRunning {
|
||||
t.Fatalf("expected a new cycle at next_run_at, got %+v", ac)
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
@@ -36,6 +37,10 @@ type Orchestrator struct {
|
||||
Cfg config.OrchestratorConfig
|
||||
Agg config.AggregationConfig
|
||||
Log *slog.Logger
|
||||
|
||||
// autoCycleMu serializes AutoCycleStep with StartAutoCycle/StopAutoCycle
|
||||
// so an API call can never interleave with a half-finished step.
|
||||
autoCycleMu sync.Mutex
|
||||
}
|
||||
|
||||
// New constructs an Orchestrator. Egress check types/targets, prober sites,
|
||||
|
||||
@@ -226,6 +226,64 @@ done
|
||||
echo "--- ip results after forced re-check (attempt_number should have advanced) ---"
|
||||
ips | python3 -m json.tool
|
||||
|
||||
echo "--- automatic cycle: enable it and wait for the first cycle (clear queue -> scan FIPs -> checks -> registry) ---"
|
||||
AC_URL="http://127.0.0.1:28080/api/v1/admin/auto-cycle"
|
||||
registry_cycles() {
|
||||
curl -fs "http://127.0.0.1:28080/api/v1/admin/registry" | python3 -c "
|
||||
import json,sys
|
||||
for r in json.load(sys.stdin):
|
||||
if r['ip_address'] == '127.0.0.1':
|
||||
print(r['total_cycles'])
|
||||
break
|
||||
else:
|
||||
print(0)"
|
||||
}
|
||||
ac_field() { curl -fs "$AC_URL" | python3 -c "import json,sys; print(json.load(sys.stdin)['$1'])"; }
|
||||
|
||||
CYCLES_BEFORE="$(registry_cycles)"
|
||||
# 60s is the smallest interval control-api accepts; the second cycle is
|
||||
# covered by unit tests, so the script stops the auto-cycle after the first.
|
||||
curl -fs -X PUT "$AC_URL" -H 'Content-Type: application/json' \
|
||||
-d '{"interval_seconds":60,"max_run_seconds":120}' >/dev/null
|
||||
curl -fs -X POST "$AC_URL/start" | python3 -m json.tool
|
||||
|
||||
echo "--- waiting for the first auto cycle to complete (up to 90s) ---"
|
||||
AC_OUTCOME=""
|
||||
for i in $(seq 1 180); do
|
||||
sleep 0.5
|
||||
AC_OUTCOME="$(ac_field last_outcome 2>/dev/null || true)"
|
||||
if [ -n "$AC_OUTCOME" ]; then
|
||||
echo "first auto cycle finished after ~$((i / 2))s: outcome=$AC_OUTCOME"
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
echo "--- auto-cycle status ---"
|
||||
curl -fs "$AC_URL" | python3 -m json.tool
|
||||
CYCLES_AFTER="$(registry_cycles)"
|
||||
echo "registry total_cycles for 127.0.0.1: $CYCLES_BEFORE -> $CYCLES_AFTER"
|
||||
|
||||
if [ "$AC_OUTCOME" != "completed" ]; then
|
||||
echo "FAIL: expected auto-cycle outcome 'completed', got '$AC_OUTCOME'"
|
||||
exit 1
|
||||
fi
|
||||
if [ "$(ac_field phase)" != "waiting" ] || [ "$(ac_field runs_total)" != "1" ]; then
|
||||
echo "FAIL: expected phase=waiting and runs_total=1 after the first cycle"
|
||||
exit 1
|
||||
fi
|
||||
if [ "$CYCLES_AFTER" -le "$CYCLES_BEFORE" ]; then
|
||||
echo "FAIL: the auto cycle did not add a new check cycle to the registry"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "--- automatic cycle: disable it; no further cycles must start ---"
|
||||
curl -fs -X POST "$AC_URL/stop" | python3 -m json.tool
|
||||
if [ "$(ac_field enabled)" != "False" ] || [ "$(ac_field phase)" != "idle" ]; then
|
||||
echo "FAIL: expected enabled=false and phase=idle after stop"
|
||||
exit 1
|
||||
fi
|
||||
echo "auto-cycle e2e: OK"
|
||||
|
||||
echo "--- logs are in $WORK_DIR (kept for inspection; workdir NOT auto-deleted) ---"
|
||||
trap - EXIT
|
||||
cleanup
|
||||
Reference in new issue
Block a user