Files
ayurishchevandClaude Sonnet 5.5 aff8fe38b5 Scan floating IPs in the background, page by page, so thousands of addresses work
The "Scan Floating IP" button failed with a client timeout: the project now
holds ~6.4k floating IPs and the scan listed them all in one unpaginated,
timeout-less Neutron request on the HTTP request context.

openstack: ListFreeFloatingIPs reads marker-based pages (fields= keeps them
small) with per-page retry/backoff on transport errors, 5xx and 429, and every
request now has a timeout (also ends hangs inside the orchestrator tick).

orchestrator: the scan is a single-flight background job on the process
context with progress (clearing/listing/enqueuing/done/error), dry_run, full
discovery before anything is enqueued, then SubmitIPs in chunks of 500 in
ascending IP order; a failed read leaves the queue untouched. The auto-cycle
gets a "scanning" phase that polls the job, so the control loop and
autoCycleMu are never held across OpenStack/DB work; it recovers after a
restart and waits for (instead of adopting) a scan started by someone else.

db: migration 0009 (indexes), paged ListIPsPage/ListRegistryPage, GROUP BY
counters, EXISTS completion check, set-based ClearAllIPs.

API: POST /admin/ips/scan -> 202 (dry_run, wait), GET /admin/ips/scan, paging
and filters on /admin/ips and /admin/registry (bare arrays without limit),
results_by_overall in /admin/status.

dashboard: scan progress panel and dry-run button, paginated /ips and
/registry with server-side filters, Overview on counters and capped lists
with progress/ETA, "select all N by filter", hx-params fix for per-row
buttons, real counts in confirmations.

Also: docs (API, USAGE, DASHBOARD, README), plan and review under
docs/changes/, bin/ rebuilt with new SHA256SUMS.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-01 19:31:11 +03:00

180 lines
5.9 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package dashboard
import (
"net/http"
"strings"
"time"
)
const (
// overviewActiveLimit caps the "В работе" table; overviewQueueLimit caps
// the compact list of the next queued addresses. The queue itself can hold
// thousands of rows, so the overview never lists more than these.
overviewActiveLimit = 100
overviewQueueLimit = 10
// etaMinSamples is how many completed rows (with AggregatedAt) the ETA
// needs to derive a rate from.
etaMinSamples = 5
)
// overviewProgress is the "Готово D из T" indicator of the stats block.
type overviewProgress struct {
Total, Done, Active, Queued, Percent int
// ETA is a rough time-to-finish estimate ("" when not enough data).
ETA string
}
type overviewData struct {
PageData
Status statusResponse
// ActiveItems are the addresses being checked right now (≤ overviewActiveLimit
// of ActiveTotal); QueuedItems are the next few waiting ones (QueuedTotal
// in total). Both stay empty under a result filter: an address without a
// verdict can't match one.
ActiveItems []ipQueueItem
ActiveTotal int
QueuedItems []ipQueueItem
QueuedTotal int
LastCompleted []ipQueueItem
Breakdown map[string]int
LastN int
PollSeconds int
Query string
StatusFilter string
Progress overviewProgress
// AutoCycle is nil when the auto-cycle status could not be fetched; the
// indicator is then simply hidden (a non-fatal failure). Scan is likewise
// nil when the scan status is unavailable.
AutoCycle *autoCycleDTO
Scan *scanStatusDTO
}
func (s *Server) loadOverview(r *http.Request) (overviewData, error) {
ctx := r.Context()
status, err := s.CA.Status(ctx)
if err != nil {
return overviewData{}, err
}
q := strings.TrimSpace(r.URL.Query().Get("q"))
resultFilter := r.URL.Query().Get("status")
if !containsStr(ipResults, resultFilter) {
resultFilter = ""
}
lastN := s.Cfg.LastCompletedCount
if lastN < 1 {
lastN = 20
}
data := overviewData{
Status: status,
LastN: lastN,
PollSeconds: s.Cfg.OverviewPollIntervalS,
Query: q,
StatusFilter: resultFilter,
}
if resultFilter == "" {
active, err := s.CA.ListIPsPage(ctx, ipsQuery{States: activeStates, Q: q, Order: "sequence", Limit: overviewActiveLimit})
if err != nil {
return overviewData{}, err
}
queued, err := s.CA.ListIPsPage(ctx, ipsQuery{States: []string{"queued"}, Q: q, Order: "sequence", Limit: overviewQueueLimit})
if err != nil {
return overviewData{}, err
}
data.ActiveItems, data.ActiveTotal = active.Items, active.Total
data.QueuedItems, data.QueuedTotal = queued.Items, queued.Total
}
completed, err := s.CA.ListIPsPage(ctx, ipsQuery{States: []string{"done", "failed"}, Q: q, Result: resultFilter, Order: "aggregated_at_desc", Limit: lastN})
if err != nil {
return overviewData{}, err
}
data.LastCompleted = completed.Items
data.Breakdown = resultBreakdown(completed.Items)
// ETA from the unfiltered window only: a filtered list is not a sample of
// the checks' real throughput.
data.Progress = computeProgress(status, completed.Items, q == "" && resultFilter == "")
if ac, acErr := s.CA.GetAutoCycle(ctx); acErr != nil {
s.Log.Warn("overview: auto-cycle status unavailable", "err", acErr)
} else {
data.AutoCycle = &ac
}
if sc, scErr := s.CA.ScanStatus(ctx); scErr != nil {
s.Log.Warn("overview: scan status unavailable", "err", scErr)
} else if sc.Running {
data.Scan = &sc
}
return data, nil
}
func (s *Server) handleOverview(w http.ResponseWriter, r *http.Request) {
data, err := s.loadOverview(r)
data.ActiveNav = "overview"
data.Banner = bannerFor(err)
s.renderPage(w, r, "overview_page", data)
}
// handleOverviewFragment serves both the recurring poll and every
// filter-triggered request. Its hx-target is #overview-tables, but the
// stat-grid (#overview-stats) needs to stay in sync too — it lives outside
// #overview-tables in the DOM (see overview.html) so the filter form can
// sit between them, so it's refreshed via an out-of-band swap appended
// after the main content, the same idiom templates/layout.html's
// error_banner already uses for the shared error banner.
func (s *Server) handleOverviewFragment(w http.ResponseWriter, r *http.Request) {
data, err := s.loadOverview(r)
s.renderFragment(w, "overview_tables", data, err)
if tplErr := s.tmpl.ExecuteTemplate(w, "overview_stats_oob", data); tplErr != nil {
s.Log.Error("render overview stats oob", "err", tplErr)
}
}
// computeProgress derives the done/active/queued counters from the status
// breakdown (done = every terminal state, "occupied" included) and, when
// useETA is set and the newest-first list of completed rows has at least
// etaMinSamples with AggregatedAt, a rough ETA from their completion rate.
func computeProgress(st statusResponse, completed []ipQueueItem, useETA bool) overviewProgress {
p := overviewProgress{
Total: st.TotalIPs,
Done: sumStates(st.IPsByState, terminalStates),
Active: sumStates(st.IPsByState, activeStates),
Queued: st.IPsByState["queued"],
}
if p.Total > 0 {
p.Percent = p.Done * 100 / p.Total
}
remaining := p.Active + p.Queued
if !useETA || remaining == 0 {
return p
}
var stamps []time.Time
for _, ip := range completed {
if ip.AggregatedAt != nil {
stamps = append(stamps, *ip.AggregatedAt)
}
}
if len(stamps) < etaMinSamples {
return p
}
// completed is newest-first: stamps[0] is the newest, the last the oldest.
span := stamps[0].Sub(stamps[len(stamps)-1])
if span <= 0 {
return p
}
perItem := span / time.Duration(len(stamps)-1)
p.ETA = fmtDuration(perItem * time.Duration(remaining))
return p
}
// resultBreakdown counts OverallResult values across exactly the given
// items (the "последние N завершённых" window) — pass/partial/fail/cancelled.
func resultBreakdown(items []ipQueueItem) map[string]int {
out := map[string]int{"pass": 0, "partial": 0, "fail": 0, "cancelled": 0}
for _, ip := range items {
out[ip.OverallResult]++
}
return out
}