The "Scan Floating IP" button failed with a client timeout: the project now holds ~6.4k floating IPs and the scan listed them all in one unpaginated, timeout-less Neutron request on the HTTP request context. openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them small) with per-page retry/backoff on transport errors, 5xx and 429, and every request now has a timeout (also ends hangs inside the orchestrator tick). orchestrator: the scan is a single-flight background job on the process context with progress (clearing/listing/enqueuing/done/error), dry_run, full discovery before anything is enqueued, then SubmitIPs in chunks of 500 in ascending IP order; a failed read leaves the queue untouched. The auto-cycle gets a "scanning" phase that polls the job, so the control loop and autoCycleMu are never held across OpenStack/DB work; it recovers after a restart and waits for (instead of adopting) a scan started by someone else. db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY counters, EXISTS completion check, set-based ClearAllIPs. API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging and filters on /admin/ips and /admin/registry (bare arrays without limit), results_by_overall in /admin/status. dashboard: scan progress panel and dry-run button, paginated /ips and /registry with server-side filters, Overview on counters and capped lists with progress/ETA, "select all N by filter", hx-params fix for per-row buttons, real counts in confirmations. Also: docs (API, USAGE, DASHBOARD, README), plan and review under docs/changes/, bin/ rebuilt with new SHA256SUMS. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
159 lines
7.0 KiB
YAML
159 lines
7.0 KiB
YAML
# Control API configuration.
|
||
#
|
||
# OpenStack credentials are never set here — only the *names* of the
|
||
# environment variables to read them from. The actual values must be
|
||
# supplied by the process environment (see deploy/systemd/control-api.service
|
||
# and its EnvironmentFile=).
|
||
|
||
server:
|
||
listen_addr: ":8080"
|
||
|
||
database:
|
||
path: "/var/lib/cloud-ip-validator/control-api.db"
|
||
|
||
openstack:
|
||
mode: "real" # "mock" | "real" — mock uses an in-memory
|
||
# OpenStack stand-in for local dev/testing
|
||
auth_method: "token" # "token" (default) | "password" — see below
|
||
|
||
auth_url_env: "OS_AUTH_URL"
|
||
project_id_env: "OS_PROJECT_ID"
|
||
region_env: "OS_REGION_NAME"
|
||
interface_env: "OS_INTERFACE" # optional; empty env value defaults to "public"
|
||
|
||
# auth_method: "token" — an admin supplies an already project-scoped
|
||
# token directly; it's used as-is for every call, never exchanged for a
|
||
# new one. Simplest option, but it can't renew itself: when the token
|
||
# expires, control-api starts failing OpenStack calls until the operator
|
||
# reissues OS_TOKEN and restarts the process.
|
||
token_env: "OS_TOKEN"
|
||
|
||
# auth_method: "password" — the client authenticates with a normal
|
||
# Keystone username/password and automatically re-authenticates
|
||
# (mints a fresh token) whenever the current one is rejected, for as
|
||
# long as the process runs. Trade-off: a long-lived password credential
|
||
# sits in the environment file instead of a token.
|
||
username_env: "OS_USERNAME"
|
||
user_domain_name_env: "OS_USER_DOMAIN_NAME"
|
||
password_env: "OS_PASSWORD"
|
||
|
||
# Постраничное чтение Floating IP (скан при тысячах адресов): сколько
|
||
# адресов запрашивать у Neutron за один запрос. Default 200.
|
||
# Page size of the paged floating-IP listing. Default 200.
|
||
list_page_size: 200
|
||
# Таймаут каждого HTTP-запроса к Keystone/Neutron, секунд. Default 60.
|
||
# Per-request HTTP timeout (also protects the orchestrator tick from a
|
||
# hung Neutron call). Default 60.
|
||
request_timeout_seconds: 60
|
||
# Сколько раз повторять неудавшуюся страницу (сетевая ошибка, EOF/
|
||
# RemoteDisconnected, 5xx, 429) с паузами 1,2,4,8,16 с. Default 5;
|
||
# отрицательное значение отключает повторы.
|
||
# Retries per failed listing page. Default 5; negative disables retries.
|
||
list_page_retries: 5
|
||
|
||
# Аутентификация API: здесь только ИМЕНА переменных окружения, значения
|
||
# (статические bearer-токены) задаются окружением процесса — см.
|
||
# deploy/systemd/control-api.service (EnvironmentFile=). Генерация:
|
||
# openssl rand -hex 32
|
||
# Пустой/незаданный токен оставляет соответствующий уровень ОТКРЫТЫМ
|
||
# (в логе при старте предупреждение) — для обратной совместимости.
|
||
auth:
|
||
# Защищает все /api/v1/admin/* (его использует дашборд и оператор: curl -H
|
||
# "Authorization: Bearer $TOKEN").
|
||
admin_token_env: "CONTROL_API_ADMIN_TOKEN"
|
||
# Защищает запись результатов/событий: POST /agents/{id}/self-check|events|
|
||
# results|complete и POST /probers/{site_id}/results. Один общий токен для
|
||
# validator-agent и prober. register/heartbeat/получение задания остаются
|
||
# открытыми.
|
||
agent_token_env: "CONTROL_API_AGENT_TOKEN"
|
||
|
||
orchestrator:
|
||
poll_interval_seconds: 5
|
||
self_check_timeout_seconds: 60
|
||
max_self_check_retries: 3
|
||
checking_window_seconds: 120
|
||
max_retries: 3
|
||
lease_ttl_seconds: 180
|
||
heartbeat_timeout_seconds: 30
|
||
# Pause (seconds) between FIP association and the start of self-check —
|
||
# gives the OpenStack data plane time to start forwarding traffic
|
||
# through the newly attached floating IP. 0 = no pause (default).
|
||
# This is only the one-time seed value used the first time control-api
|
||
# starts against an empty database; after that it's managed at runtime
|
||
# via PUT /api/v1/admin/config/orchestrator (or the dashboard's
|
||
# /settings page) and this field is ignored. Must satisfy
|
||
# fip_settle_seconds + self_check_timeout_seconds < lease_ttl_seconds.
|
||
fip_settle_seconds: 0
|
||
# How often (seconds) to automatically scan the OpenStack project for free
|
||
# (unassociated) floating IPs and submit them to the check queue. 0 (the
|
||
# default) disables periodic scanning — an operator can still trigger a
|
||
# scan on demand via POST /api/v1/admin/ips/scan or the dashboard's
|
||
# "Scan Floating IPs" button.
|
||
fip_scan_interval_seconds: 0
|
||
# Общий таймаут одного фонового скана Floating IP (очистка + чтение всех
|
||
# страниц + постановка в очередь), секунд. Default 1800.
|
||
# Overall deadline of one background floating-IP scan. Default 1800.
|
||
fip_scan_timeout_seconds: 1800
|
||
|
||
aggregation:
|
||
missing_counts_as_fail: true
|
||
|
||
# Validators are VMs in the service project; os_port_id is the Neutron port
|
||
# ID of each validator's primary NIC, used when associating a floating IP.
|
||
validators:
|
||
- validator_id: "validator_01"
|
||
os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_1"
|
||
- validator_id: "validator_02"
|
||
os_port_id: "REPLACE_WITH_NEUTRON_PORT_ID_2"
|
||
|
||
# Inbound (prober) checks are OPTIONAL: list here only the external sites
|
||
# you actually run a `prober` on — index is any integer >= 1, no cap on
|
||
# how many slots you configure. Leave this list empty to disable inbound
|
||
# checks entirely — the overall result is then based on egress checks
|
||
# alone, and aggregation doesn't wait on any prober. A partial list (e.g.
|
||
# just index 1) only waits on that one site.
|
||
sites:
|
||
- site_id: "site-1"
|
||
index: 1
|
||
- site_id: "site-2"
|
||
index: 2
|
||
- site_id: "site-3"
|
||
index: 3
|
||
|
||
# Outbound/egress check types the validator-agent runs, and which target
|
||
# group (below) each runs against.
|
||
check_types:
|
||
- name: "https"
|
||
enabled: true
|
||
targets: ["default-targets"]
|
||
- name: "icmp"
|
||
enabled: true
|
||
targets: ["default-targets"]
|
||
- name: "ssh"
|
||
enabled: false
|
||
targets: []
|
||
|
||
targets:
|
||
default-targets:
|
||
- "https://hub.docker.com"
|
||
- "https://github.com"
|
||
- "https://packages.ubuntu.com"
|
||
|
||
# Inbound checks the 3 external-site probers run directly against each
|
||
# validator's currently-assigned floating IP. Like validators/sites/targets/
|
||
# check_types above, this is only a bootstrap seed for a fresh, empty
|
||
# database; after that it's managed at runtime via PUT
|
||
# /api/v1/admin/config/inbound-checks (or the dashboard's /settings page)
|
||
# and this field is ignored.
|
||
inbound_checks:
|
||
ports: [22, 80, 443, 8080]
|
||
icmp: true
|
||
|
||
# The pool of public IPv4 addresses to validate, in the order they'll be
|
||
# processed (ip_queue.sequence). Every address here is checked through to
|
||
# the end of the list.
|
||
ip_addresses:
|
||
- "203.0.113.10"
|
||
- "203.0.113.11"
|
||
- "203.0.113.12"
|