admin control features and admin dashboard

This commit is contained in:
ayurishchev committed 2026-08-23 20:39:22 +03:00
1 parent c630f13c57
commit 37910e410b
69 files changed
+4959 -400

No files matched your search

+95 -42
View File
@@ -9,6 +9,8 @@ package orchestrator
import (
"context"
"database/sql"
"errors"
"fmt"
"log/slog"
"time"
@@ -20,9 +22,8 @@ import (
// CheckConfig is the check-type/target configuration handed to a
// validator-agent once its IP has passed self-check. It mirrors
// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a
// flat list so the agent doesn't need its own copy of the target-group
// mapping.
// db.ResolvedCheckType, kept as a distinct type so httpapi's DTO layer
// doesn't need to import internal/db just for this shape.
type CheckConfig struct {
Type string `json:"type"`
Targets []string `json:"targets"`
@@ -33,31 +34,22 @@ type Orchestrator struct {
OS openstack.FloatingIPClient
Cfg config.OrchestratorConfig
Agg config.AggregationConfig
Checks []CheckConfig
Sites []config.SiteConfig
Inbound config.InboundConfig
Log *slog.Logger
}
// New constructs an Orchestrator. Egress check types/targets and prober
// sites are no longer taken from cfg — they're read from the database on
// every use (see AssignmentForValidator, expectedCheckCount,
// isReadyToAggregate) so admin API changes to them take effect without a
// restart. cfg.Validators/.Sites/.CheckTypes/.Targets/.IPAddresses are only
// consulted once, at process startup, by db.BootstrapFromConfig.
func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator {
var checks []CheckConfig
for _, ct := range cfg.CheckTypes {
if !ct.Enabled {
continue
}
var targets []string
for _, group := range ct.Targets {
targets = append(targets, cfg.Targets[group]...)
}
checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets})
}
return &Orchestrator{
DB: d,
OS: osClient,
Cfg: cfg.Orchestrator,
Agg: cfg.Aggregation,
Checks: checks,
Sites: cfg.Sites,
Inbound: cfg.Inbound,
Log: log,
}
@@ -159,7 +151,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
// AssignmentForValidator returns the check config for a validator's current
// IP if it's ready to be worked on (awaiting_self_check or checking),
// or nil if the validator has nothing to do right now.
// or nil if the validator has nothing to do right now. The check config is
// read fresh from the database on every call, so admin changes to
// check_types/targets apply to the very next assignment.
func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) {
v, err := o.DB.GetValidator(ctx, validatorID)
if err != nil {
@@ -175,17 +169,21 @@ func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID s
if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking {
return nil, nil, nil
}
return item, o.Checks, nil
resolved, err := o.DB.ListResolvedCheckTypes(ctx)
if err != nil {
return nil, nil, err
}
checks := make([]CheckConfig, len(resolved))
for i, r := range resolved {
checks[i] = CheckConfig{Type: r.Type, Targets: r.Targets}
}
return item, checks, nil
}
// SiteIndexForID resolves a configured site_id to its 1/2/3 index.
func (o *Orchestrator) SiteIndexForID(siteID string) int {
for _, s := range o.Sites {
if s.SiteID == siteID {
return s.Index
}
}
return 0
// SiteIndexForID resolves a configured site_id to its 1/2/3 index, or
// (0, nil) if unconfigured.
func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int, error) {
return o.DB.GetSiteIndex(ctx, siteID)
}
// RecordCheck upserts a single check result and, if it represents a
@@ -203,6 +201,43 @@ func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteInd
return o.DB.SetSiteComplete(ctx, ipID, siteIndex)
}
// ForceCancel stops an in-progress (or still-queued) check for the given
// address on admin request, even though it was never going to finish on
// its own within the checking window. If a floating IP is currently
// associated, it's disassociated best-effort (same fallthrough-on-error
// behavior as aggregateAndRelease/sweepExpiredLeases: the DB/validator
// state must still be freed even if Neutron hiccups). Returns
// db.ErrNotFound if the address is unknown, or db.ErrInvalidState if it has
// already reached done/failed.
func (o *Orchestrator) ForceCancel(ctx context.Context, ipAddress string) error {
item, err := o.DB.GetIPByAddress(ctx, ipAddress)
if err != nil {
if errors.Is(err, sql.ErrNoRows) {
return fmt.Errorf("ip %q: %w", ipAddress, db.ErrNotFound)
}
return err
}
if item.FIPID != "" {
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
o.Log.Error("disassociate fip on force cancel", "ip_id", item.ID, "fip_id", item.FIPID, "err", err)
}
}
if err := o.DB.CancelIP(ctx, item.ID); err != nil {
return err
}
if item.OwnerValidatorID != nil {
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID); err != nil {
return fmt.Errorf("free validator: %w", err)
}
}
o.event(ctx, "control-api", "", &item.ID, "force_cancel", "")
return nil
}
// sweepCheckingWindow moves IPs that have either finished reporting from
// every source, or hit the checking-window deadline, into aggregation.
func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
@@ -211,8 +246,12 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
if err != nil {
return fmt.Errorf("list checking: %w", err)
}
sites, err := o.DB.ListSites(ctx)
if err != nil {
return fmt.Errorf("list sites: %w", err)
}
for _, item := range checking {
if !o.isReadyToAggregate(item, deadline) {
if !o.isReadyToAggregate(item, deadline, sites) {
continue
}
if err := o.aggregateAndRelease(ctx, item); err != nil {
@@ -225,20 +264,20 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
// isReadyToAggregate reports whether an in-progress IP has either finished
// reporting from every source it's actually expecting, or hit the
// checking-window deadline. Which inbound sources it's expecting is driven
// entirely by o.Sites — inbound checks are optional: an empty (or
// partial) `sites` config in control-api.yaml means this IP is ready as
// soon as egress completes (or after the corresponding subset of
// siteN_complete flags), with no need to wait on a prober that will never
// exist. This is what makes inbound checks genuinely opt-in rather than a
// hardcoded expectation of exactly three sites.
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time) bool {
// entirely by the currently configured sites — inbound checks are optional:
// an empty (or partial) sites configuration means this IP is ready as soon
// as egress completes (or after the corresponding subset of siteN_complete
// flags), with no need to wait on a prober that will never exist. This is
// what makes inbound checks genuinely opt-in rather than a hardcoded
// expectation of exactly three sites.
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time, sites []db.Site) bool {
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
return true
}
if !item.EgressComplete {
return false
}
for _, s := range o.Sites {
for _, s := range sites {
switch s.Index {
case 1:
if !item.Site1Complete {
@@ -266,7 +305,10 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
return err
}
expected := o.expectedCheckCount()
expected, err := o.expectedCheckCount(ctx)
if err != nil {
return fmt.Errorf("expected check count: %w", err)
}
passCount := 0
for _, c := range checks {
if c.Success {
@@ -317,17 +359,28 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
// expectedCheckCount is the number of check rows a fully-reported IP should
// have: one per (egress check-type x target) plus one per (site x inbound
// port/icmp probe).
func (o *Orchestrator) expectedCheckCount() int {
// port/icmp probe). Reads the current check_types/targets/sites from the
// database, so a config change between assignment and aggregation is
// reflected in this specific aggregation (see the "accepted tradeoff" note
// in docs/PLAN_API_CONFIG_MANAGEMENT.md).
func (o *Orchestrator) expectedCheckCount(ctx context.Context) (int, error) {
resolved, err := o.DB.ListResolvedCheckTypes(ctx)
if err != nil {
return 0, err
}
egress := 0
for _, c := range o.Checks {
for _, c := range resolved {
egress += len(c.Targets)
}
sites, err := o.DB.ListSites(ctx)
if err != nil {
return 0, err
}
inboundPerSite := len(o.Inbound.Ports)
if o.Inbound.ICMP {
inboundPerSite++
}
return egress + inboundPerSite*len(o.Sites)
return egress + inboundPerSite*len(sites), nil
}
// sweepExpiredLeases reclaims non-terminal IPs whose lease has passed —
@@ -58,6 +58,10 @@ func newTestOrchestratorWithSites(t *testing.T, leaseTTLSeconds int, sites []con
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
}
if err := d.BootstrapFromConfig(ctx, cfg); err != nil {
t.Fatalf("bootstrap from config: %v", err)
}
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
return New(d, mock, cfg, log), d, mock
}