admin control features and admin dashboard
This commit is contained in:
1 parent
c630f13c57
commit
37910e410b
69 files changed
+4959
-400
No files matched your search
@@ -9,6 +9,8 @@ package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"time"
|
||||
@@ -20,9 +22,8 @@ import (
|
||||
|
||||
// CheckConfig is the check-type/target configuration handed to a
|
||||
// validator-agent once its IP has passed self-check. It mirrors
|
||||
// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a
|
||||
// flat list so the agent doesn't need its own copy of the target-group
|
||||
// mapping.
|
||||
// db.ResolvedCheckType, kept as a distinct type so httpapi's DTO layer
|
||||
// doesn't need to import internal/db just for this shape.
|
||||
type CheckConfig struct {
|
||||
Type string `json:"type"`
|
||||
Targets []string `json:"targets"`
|
||||
@@ -33,31 +34,22 @@ type Orchestrator struct {
|
||||
OS openstack.FloatingIPClient
|
||||
Cfg config.OrchestratorConfig
|
||||
Agg config.AggregationConfig
|
||||
Checks []CheckConfig
|
||||
Sites []config.SiteConfig
|
||||
Inbound config.InboundConfig
|
||||
Log *slog.Logger
|
||||
}
|
||||
|
||||
// New constructs an Orchestrator. Egress check types/targets and prober
|
||||
// sites are no longer taken from cfg — they're read from the database on
|
||||
// every use (see AssignmentForValidator, expectedCheckCount,
|
||||
// isReadyToAggregate) so admin API changes to them take effect without a
|
||||
// restart. cfg.Validators/.Sites/.CheckTypes/.Targets/.IPAddresses are only
|
||||
// consulted once, at process startup, by db.BootstrapFromConfig.
|
||||
func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator {
|
||||
var checks []CheckConfig
|
||||
for _, ct := range cfg.CheckTypes {
|
||||
if !ct.Enabled {
|
||||
continue
|
||||
}
|
||||
var targets []string
|
||||
for _, group := range ct.Targets {
|
||||
targets = append(targets, cfg.Targets[group]...)
|
||||
}
|
||||
checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets})
|
||||
}
|
||||
return &Orchestrator{
|
||||
DB: d,
|
||||
OS: osClient,
|
||||
Cfg: cfg.Orchestrator,
|
||||
Agg: cfg.Aggregation,
|
||||
Checks: checks,
|
||||
Sites: cfg.Sites,
|
||||
Inbound: cfg.Inbound,
|
||||
Log: log,
|
||||
}
|
||||
@@ -159,7 +151,9 @@ func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string,
|
||||
|
||||
// AssignmentForValidator returns the check config for a validator's current
|
||||
// IP if it's ready to be worked on (awaiting_self_check or checking),
|
||||
// or nil if the validator has nothing to do right now.
|
||||
// or nil if the validator has nothing to do right now. The check config is
|
||||
// read fresh from the database on every call, so admin changes to
|
||||
// check_types/targets apply to the very next assignment.
|
||||
func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) {
|
||||
v, err := o.DB.GetValidator(ctx, validatorID)
|
||||
if err != nil {
|
||||
@@ -175,17 +169,21 @@ func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID s
|
||||
if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking {
|
||||
return nil, nil, nil
|
||||
}
|
||||
return item, o.Checks, nil
|
||||
resolved, err := o.DB.ListResolvedCheckTypes(ctx)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
checks := make([]CheckConfig, len(resolved))
|
||||
for i, r := range resolved {
|
||||
checks[i] = CheckConfig{Type: r.Type, Targets: r.Targets}
|
||||
}
|
||||
return item, checks, nil
|
||||
}
|
||||
|
||||
// SiteIndexForID resolves a configured site_id to its 1/2/3 index.
|
||||
func (o *Orchestrator) SiteIndexForID(siteID string) int {
|
||||
for _, s := range o.Sites {
|
||||
if s.SiteID == siteID {
|
||||
return s.Index
|
||||
}
|
||||
}
|
||||
return 0
|
||||
// SiteIndexForID resolves a configured site_id to its 1/2/3 index, or
|
||||
// (0, nil) if unconfigured.
|
||||
func (o *Orchestrator) SiteIndexForID(ctx context.Context, siteID string) (int, error) {
|
||||
return o.DB.GetSiteIndex(ctx, siteID)
|
||||
}
|
||||
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
@@ -203,6 +201,43 @@ func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteInd
|
||||
return o.DB.SetSiteComplete(ctx, ipID, siteIndex)
|
||||
}
|
||||
|
||||
// ForceCancel stops an in-progress (or still-queued) check for the given
|
||||
// address on admin request, even though it was never going to finish on
|
||||
// its own within the checking window. If a floating IP is currently
|
||||
// associated, it's disassociated best-effort (same fallthrough-on-error
|
||||
// behavior as aggregateAndRelease/sweepExpiredLeases: the DB/validator
|
||||
// state must still be freed even if Neutron hiccups). Returns
|
||||
// db.ErrNotFound if the address is unknown, or db.ErrInvalidState if it has
|
||||
// already reached done/failed.
|
||||
func (o *Orchestrator) ForceCancel(ctx context.Context, ipAddress string) error {
|
||||
item, err := o.DB.GetIPByAddress(ctx, ipAddress)
|
||||
if err != nil {
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
return fmt.Errorf("ip %q: %w", ipAddress, db.ErrNotFound)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
if item.FIPID != "" {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on force cancel", "ip_id", item.ID, "fip_id", item.FIPID, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
if err := o.DB.CancelIP(ctx, item.ID); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if item.OwnerValidatorID != nil {
|
||||
if err := o.DB.FreeValidator(ctx, *item.OwnerValidatorID); err != nil {
|
||||
return fmt.Errorf("free validator: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
o.event(ctx, "control-api", "", &item.ID, "force_cancel", "")
|
||||
return nil
|
||||
}
|
||||
|
||||
// sweepCheckingWindow moves IPs that have either finished reporting from
|
||||
// every source, or hit the checking-window deadline, into aggregation.
|
||||
func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
@@ -211,8 +246,12 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
if err != nil {
|
||||
return fmt.Errorf("list checking: %w", err)
|
||||
}
|
||||
sites, err := o.DB.ListSites(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list sites: %w", err)
|
||||
}
|
||||
for _, item := range checking {
|
||||
if !o.isReadyToAggregate(item, deadline) {
|
||||
if !o.isReadyToAggregate(item, deadline, sites) {
|
||||
continue
|
||||
}
|
||||
if err := o.aggregateAndRelease(ctx, item); err != nil {
|
||||
@@ -225,20 +264,20 @@ func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
// isReadyToAggregate reports whether an in-progress IP has either finished
|
||||
// reporting from every source it's actually expecting, or hit the
|
||||
// checking-window deadline. Which inbound sources it's expecting is driven
|
||||
// entirely by o.Sites — inbound checks are optional: an empty (or
|
||||
// partial) `sites` config in control-api.yaml means this IP is ready as
|
||||
// soon as egress completes (or after the corresponding subset of
|
||||
// siteN_complete flags), with no need to wait on a prober that will never
|
||||
// exist. This is what makes inbound checks genuinely opt-in rather than a
|
||||
// hardcoded expectation of exactly three sites.
|
||||
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time) bool {
|
||||
// entirely by the currently configured sites — inbound checks are optional:
|
||||
// an empty (or partial) sites configuration means this IP is ready as soon
|
||||
// as egress completes (or after the corresponding subset of siteN_complete
|
||||
// flags), with no need to wait on a prober that will never exist. This is
|
||||
// what makes inbound checks genuinely opt-in rather than a hardcoded
|
||||
// expectation of exactly three sites.
|
||||
func (o *Orchestrator) isReadyToAggregate(item db.IPQueueItem, deadline time.Time, sites []db.Site) bool {
|
||||
if item.AssignedAt != nil && item.AssignedAt.Before(deadline) {
|
||||
return true
|
||||
}
|
||||
if !item.EgressComplete {
|
||||
return false
|
||||
}
|
||||
for _, s := range o.Sites {
|
||||
for _, s := range sites {
|
||||
switch s.Index {
|
||||
case 1:
|
||||
if !item.Site1Complete {
|
||||
@@ -266,7 +305,10 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
return err
|
||||
}
|
||||
|
||||
expected := o.expectedCheckCount()
|
||||
expected, err := o.expectedCheckCount(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("expected check count: %w", err)
|
||||
}
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
@@ -317,17 +359,28 @@ func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueI
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe).
|
||||
func (o *Orchestrator) expectedCheckCount() int {
|
||||
// port/icmp probe). Reads the current check_types/targets/sites from the
|
||||
// database, so a config change between assignment and aggregation is
|
||||
// reflected in this specific aggregation (see the "accepted tradeoff" note
|
||||
// in docs/PLAN_API_CONFIG_MANAGEMENT.md).
|
||||
func (o *Orchestrator) expectedCheckCount(ctx context.Context) (int, error) {
|
||||
resolved, err := o.DB.ListResolvedCheckTypes(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
egress := 0
|
||||
for _, c := range o.Checks {
|
||||
for _, c := range resolved {
|
||||
egress += len(c.Targets)
|
||||
}
|
||||
sites, err := o.DB.ListSites(ctx)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
inboundPerSite := len(o.Inbound.Ports)
|
||||
if o.Inbound.ICMP {
|
||||
inboundPerSite++
|
||||
}
|
||||
return egress + inboundPerSite*len(o.Sites)
|
||||
return egress + inboundPerSite*len(sites), nil
|
||||
}
|
||||
|
||||
// sweepExpiredLeases reclaims non-terminal IPs whose lease has passed —
|
||||
|
||||
@@ -58,6 +58,10 @@ func newTestOrchestratorWithSites(t *testing.T, leaseTTLSeconds int, sites []con
|
||||
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
|
||||
}
|
||||
|
||||
if err := d.BootstrapFromConfig(ctx, cfg); err != nil {
|
||||
t.Fatalf("bootstrap from config: %v", err)
|
||||
}
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
||||
return New(d, mock, cfg, log), d, mock
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user