repo init
This commit is contained in:
commit
7e44db87b2
48 files changed
+5346
No files matched your search
@@ -0,0 +1,361 @@
|
||||
// Package orchestrator implements the Control API's core scheduling loop:
|
||||
// claiming queued IPs onto idle validators, driving each IP through
|
||||
// FIP-association -> self-check -> checking -> aggregation -> release, and
|
||||
// reclaiming work from crashed/stuck validators via a lease sweep. It has
|
||||
// no HTTP dependency — internal/httpapi calls into this package, and it can
|
||||
// be exercised directly in tests against an in-memory OpenStack mock and a
|
||||
// temp-file SQLite database.
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
// CheckConfig is the check-type/target configuration handed to a
|
||||
// validator-agent once its IP has passed self-check. It mirrors
|
||||
// config.CheckTypeConfig + config.ControlAPI.Targets, pre-resolved into a
|
||||
// flat list so the agent doesn't need its own copy of the target-group
|
||||
// mapping.
|
||||
type CheckConfig struct {
|
||||
Type string `json:"type"`
|
||||
Targets []string `json:"targets"`
|
||||
}
|
||||
|
||||
type Orchestrator struct {
|
||||
DB *db.DB
|
||||
OS openstack.FloatingIPClient
|
||||
Cfg config.OrchestratorConfig
|
||||
Agg config.AggregationConfig
|
||||
Checks []CheckConfig
|
||||
Sites []config.SiteConfig
|
||||
Inbound config.InboundConfig
|
||||
Log *slog.Logger
|
||||
}
|
||||
|
||||
func New(d *db.DB, osClient openstack.FloatingIPClient, cfg *config.ControlAPI, log *slog.Logger) *Orchestrator {
|
||||
var checks []CheckConfig
|
||||
for _, ct := range cfg.CheckTypes {
|
||||
if !ct.Enabled {
|
||||
continue
|
||||
}
|
||||
var targets []string
|
||||
for _, group := range ct.Targets {
|
||||
targets = append(targets, cfg.Targets[group]...)
|
||||
}
|
||||
checks = append(checks, CheckConfig{Type: ct.Name, Targets: targets})
|
||||
}
|
||||
return &Orchestrator{
|
||||
DB: d,
|
||||
OS: osClient,
|
||||
Cfg: cfg.Orchestrator,
|
||||
Agg: cfg.Aggregation,
|
||||
Checks: checks,
|
||||
Sites: cfg.Sites,
|
||||
Inbound: cfg.Inbound,
|
||||
Log: log,
|
||||
}
|
||||
}
|
||||
|
||||
func (o *Orchestrator) leaseTTL() time.Duration {
|
||||
return time.Duration(o.Cfg.LeaseTTLSeconds) * time.Second
|
||||
}
|
||||
|
||||
// Tick runs one pass of the scheduling loop: claim+associate for idle
|
||||
// validators, sweep the checking window for ready-to-aggregate IPs, and
|
||||
// reclaim expired leases. Intended to be called on a fixed interval
|
||||
// (Cfg.PollIntervalSeconds) by the caller (cmd/control-api/main.go).
|
||||
func (o *Orchestrator) Tick(ctx context.Context) {
|
||||
if err := o.assignIdleValidators(ctx); err != nil {
|
||||
o.Log.Error("assign idle validators", "err", err)
|
||||
}
|
||||
if err := o.sweepCheckingWindow(ctx); err != nil {
|
||||
o.Log.Error("sweep checking window", "err", err)
|
||||
}
|
||||
if err := o.sweepExpiredLeases(ctx); err != nil {
|
||||
o.Log.Error("sweep expired leases", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// assignIdleValidators claims the next queued IP for every currently idle
|
||||
// validator and kicks off FIP association for each newly claimed IP.
|
||||
func (o *Orchestrator) assignIdleValidators(ctx context.Context) error {
|
||||
idle, err := o.DB.ListIdleValidators(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list idle validators: %w", err)
|
||||
}
|
||||
for _, v := range idle {
|
||||
item, err := o.DB.ClaimNextQueued(ctx, v.ValidatorID, o.leaseTTL())
|
||||
if err != nil {
|
||||
o.Log.Error("claim next queued", "validator", v.ValidatorID, "err", err)
|
||||
continue
|
||||
}
|
||||
if item == nil {
|
||||
continue // no work available for this validator right now
|
||||
}
|
||||
o.Log.Info("claimed ip", "validator", v.ValidatorID, "ip", item.IPAddress, "ip_id", item.ID)
|
||||
if err := o.associateFIP(ctx, v.ValidatorID, v.OSPortID, item); err != nil {
|
||||
o.Log.Error("associate fip", "validator", v.ValidatorID, "ip", item.IPAddress, "err", err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) associateFIP(ctx context.Context, validatorID, osPortID string, item *db.IPQueueItem) error {
|
||||
fip, err := o.OS.GetFloatingIPByAddress(ctx, item.IPAddress)
|
||||
if err != nil {
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("lookup floating ip: %v", err))
|
||||
return err
|
||||
}
|
||||
if err := o.OS.AssociateFloatingIP(ctx, fip.ID, osPortID); err != nil {
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, fmt.Sprintf("associate floating ip: %v", err))
|
||||
return err
|
||||
}
|
||||
if err := o.DB.SetFIPAssociated(ctx, item.ID, fip.ID, o.leaseTTL()); err != nil {
|
||||
return fmt.Errorf("set fip associated: %w", err)
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "fip_associated", fmt.Sprintf(`{"fip_id":%q,"validator_id":%q}`, fip.ID, validatorID))
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) requeueOrFail(ctx context.Context, ipID int64, validatorID, reason string) {
|
||||
if err := o.DB.RequeueOrFail(ctx, ipID, validatorID, o.Cfg.MaxRetries); err != nil {
|
||||
o.Log.Error("requeue or fail", "ip_id", ipID, "err", err)
|
||||
return
|
||||
}
|
||||
o.event(ctx, "control-api", "", &ipID, "retry_or_fail", fmt.Sprintf(`{"reason":%q}`, reason))
|
||||
}
|
||||
|
||||
// SelfCheckResult is called by the httpapi layer when a validator-agent
|
||||
// reports its post-association self-check outcome.
|
||||
func (o *Orchestrator) SelfCheckResult(ctx context.Context, validatorID string, ipID int64, success bool, detail string) error {
|
||||
o.event(ctx, "validator-agent", validatorID, &ipID, "self_check_result",
|
||||
fmt.Sprintf(`{"success":%t,"detail":%q}`, success, detail))
|
||||
|
||||
if !success {
|
||||
item, err := o.DB.GetIP(ctx, ipID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if item.RetryCount+1 > o.Cfg.MaxSelfCheckRetries {
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed: "+detail)
|
||||
return nil
|
||||
}
|
||||
// Retry association without fully requeuing: re-drive the same
|
||||
// claim by cycling back through requeue/claim keeps the logic in
|
||||
// one place at the cost of the IP briefly returning to `queued`.
|
||||
o.requeueOrFail(ctx, ipID, validatorID, "self-check failed, retrying: "+detail)
|
||||
return nil
|
||||
}
|
||||
|
||||
return o.DB.SetChecking(ctx, ipID, o.leaseTTL())
|
||||
}
|
||||
|
||||
// AssignmentForValidator returns the check config for a validator's current
|
||||
// IP if it's ready to be worked on (awaiting_self_check or checking),
|
||||
// or nil if the validator has nothing to do right now.
|
||||
func (o *Orchestrator) AssignmentForValidator(ctx context.Context, validatorID string) (*db.IPQueueItem, []CheckConfig, error) {
|
||||
v, err := o.DB.GetValidator(ctx, validatorID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if v.CurrentIPID == nil {
|
||||
return nil, nil, nil
|
||||
}
|
||||
item, err := o.DB.GetIP(ctx, *v.CurrentIPID)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if item.State != db.IPAwaitingSelfCheck && item.State != db.IPChecking {
|
||||
return nil, nil, nil
|
||||
}
|
||||
return item, o.Checks, nil
|
||||
}
|
||||
|
||||
// SiteIndexForID resolves a configured site_id to its 1/2/3 index.
|
||||
func (o *Orchestrator) SiteIndexForID(siteID string) int {
|
||||
for _, s := range o.Sites {
|
||||
if s.SiteID == siteID {
|
||||
return s.Index
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// RecordCheck upserts a single check result and, if it represents a
|
||||
// completion signal (egress or a given site's full port+icmp sweep),
|
||||
// updates the corresponding *_complete flag.
|
||||
func (o *Orchestrator) RecordCheck(ctx context.Context, c db.Check) error {
|
||||
return o.DB.UpsertCheck(ctx, c)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkEgressComplete(ctx context.Context, ipID int64) error {
|
||||
return o.DB.SetEgressComplete(ctx, ipID)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) MarkSiteComplete(ctx context.Context, ipID int64, siteIndex int) error {
|
||||
return o.DB.SetSiteComplete(ctx, ipID, siteIndex)
|
||||
}
|
||||
|
||||
// sweepCheckingWindow moves IPs that have either finished reporting from
|
||||
// every source, or hit the checking-window deadline, into aggregation.
|
||||
func (o *Orchestrator) sweepCheckingWindow(ctx context.Context) error {
|
||||
deadline := db.Now().Add(-time.Duration(o.Cfg.CheckingWindowSeconds) * time.Second)
|
||||
ready, err := o.DB.ListReadyToAggregate(ctx, deadline)
|
||||
if err != nil {
|
||||
return fmt.Errorf("list ready to aggregate: %w", err)
|
||||
}
|
||||
for _, item := range ready {
|
||||
if err := o.aggregateAndRelease(ctx, item); err != nil {
|
||||
o.Log.Error("aggregate and release", "ip_id", item.ID, "err", err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (o *Orchestrator) aggregateAndRelease(ctx context.Context, item db.IPQueueItem) error {
|
||||
if err := o.DB.SetAggregating(ctx, item.ID); err != nil {
|
||||
return err
|
||||
}
|
||||
checks, err := o.DB.ListChecksForAttempt(ctx, item.ID, item.AttemptNumber)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
expected := o.expectedCheckCount()
|
||||
passCount := 0
|
||||
for _, c := range checks {
|
||||
if c.Success {
|
||||
passCount++
|
||||
}
|
||||
}
|
||||
missing := expected - len(checks)
|
||||
if missing < 0 {
|
||||
missing = 0
|
||||
}
|
||||
failCount := (len(checks) - passCount) + missing
|
||||
|
||||
var result string
|
||||
switch {
|
||||
case passCount > 0 && failCount == 0:
|
||||
result = db.ResultPass
|
||||
case passCount == 0:
|
||||
result = db.ResultFail
|
||||
default:
|
||||
result = db.ResultPartial
|
||||
}
|
||||
if missing > 0 && o.Agg.MissingCountsAsFail && result == db.ResultPass {
|
||||
result = db.ResultPartial
|
||||
}
|
||||
|
||||
if err := o.DB.FinishIP(ctx, item.ID, result); err != nil {
|
||||
return err
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "aggregated",
|
||||
fmt.Sprintf(`{"result":%q,"checks":%d,"passed":%d,"missing":%d}`, result, len(checks), passCount, missing))
|
||||
|
||||
if item.FIPID != "" {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip", "ip_id", item.ID, "fip_id", item.FIPID, "err", err)
|
||||
// Fall through and still free the validator/DB state — the
|
||||
// lease sweep or an operator can reconcile a stuck Neutron
|
||||
// association separately; we must not leave the validator
|
||||
// wedged as "checking" forever over a cloud API hiccup.
|
||||
}
|
||||
}
|
||||
if item.OwnerValidatorID != nil {
|
||||
if err := o.DB.ReleaseFIP(ctx, item.ID, *item.OwnerValidatorID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// expectedCheckCount is the number of check rows a fully-reported IP should
|
||||
// have: one per (egress check-type x target) plus one per (site x inbound
|
||||
// port/icmp probe).
|
||||
func (o *Orchestrator) expectedCheckCount() int {
|
||||
egress := 0
|
||||
for _, c := range o.Checks {
|
||||
egress += len(c.Targets)
|
||||
}
|
||||
inboundPerSite := len(o.Inbound.Ports)
|
||||
if o.Inbound.ICMP {
|
||||
inboundPerSite++
|
||||
}
|
||||
return egress + inboundPerSite*len(o.Sites)
|
||||
}
|
||||
|
||||
// sweepExpiredLeases reclaims non-terminal IPs whose lease has passed —
|
||||
// this is both the "stuck/crashed validator" reclaim path and, since all
|
||||
// state lives in SQLite, the control-api crash-recovery path: a freshly
|
||||
// restarted process finds the same expired leases and reclaims them the
|
||||
// same way, with no separate recovery code required.
|
||||
func (o *Orchestrator) sweepExpiredLeases(ctx context.Context) error {
|
||||
expired, err := o.DB.ListExpiredLeases(ctx, db.Now())
|
||||
if err != nil {
|
||||
return fmt.Errorf("list expired leases: %w", err)
|
||||
}
|
||||
for _, item := range expired {
|
||||
validatorID := ""
|
||||
if item.OwnerValidatorID != nil {
|
||||
validatorID = *item.OwnerValidatorID
|
||||
}
|
||||
o.Log.Info("lease expired, reclaiming", "ip_id", item.ID, "ip", item.IPAddress, "validator", validatorID)
|
||||
if item.FIPID != "" {
|
||||
if err := o.OS.DisassociateFloatingIP(ctx, item.FIPID); err != nil {
|
||||
o.Log.Error("disassociate fip on lease reclaim", "ip_id", item.ID, "err", err)
|
||||
}
|
||||
}
|
||||
o.event(ctx, "control-api", "", &item.ID, "lease_expired", fmt.Sprintf(`{"validator_id":%q}`, validatorID))
|
||||
o.requeueOrFail(ctx, item.ID, validatorID, "lease expired")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// sweepStaleHeartbeats marks validators unreachable if they haven't
|
||||
// heartbeated within HeartbeatTimeoutSeconds. It does not itself reclaim
|
||||
// their in-flight IP — that happens independently via lease expiry, so a
|
||||
// validator that stops heartbeating but whose lease hasn't yet expired
|
||||
// still finishes its current check window if it recovers in time.
|
||||
func (o *Orchestrator) SweepStaleHeartbeats(ctx context.Context) error {
|
||||
cutoff := db.Now().Add(-time.Duration(o.Cfg.HeartbeatTimeoutSeconds) * time.Second)
|
||||
stale, err := o.DB.ListStaleHeartbeats(ctx, cutoff)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, v := range stale {
|
||||
if err := o.DB.MarkValidatorUnreachable(ctx, v.ValidatorID); err != nil {
|
||||
o.Log.Error("mark validator unreachable", "validator", v.ValidatorID, "err", err)
|
||||
continue
|
||||
}
|
||||
o.event(ctx, "control-api", "", nil, "validator_unreachable", fmt.Sprintf(`{"validator_id":%q}`, v.ValidatorID))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// RecordEvent is the exported entry point httpapi uses to log
|
||||
// agent/prober-reported audit events (config_received, fip_changed,
|
||||
// error, etc.) through the same path as internally generated events.
|
||||
func (o *Orchestrator) RecordEvent(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) {
|
||||
o.event(ctx, sourceType, sourceID, ipID, eventType, payload)
|
||||
}
|
||||
|
||||
func (o *Orchestrator) event(ctx context.Context, sourceType, sourceID string, ipID *int64, eventType, payload string) {
|
||||
if err := o.DB.InsertEvent(ctx, db.Event{
|
||||
SourceType: sourceType,
|
||||
SourceID: sourceID,
|
||||
IPID: ipID,
|
||||
EventType: eventType,
|
||||
Payload: payload,
|
||||
OccurredAt: db.Now(),
|
||||
}); err != nil {
|
||||
o.Log.Error("insert event", "type", eventType, "err", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,277 @@
|
||||
package orchestrator
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloudipvalidator/internal/config"
|
||||
"cloudipvalidator/internal/db"
|
||||
"cloudipvalidator/internal/openstack"
|
||||
)
|
||||
|
||||
func newTestOrchestrator(t *testing.T, leaseTTLSeconds int) (*Orchestrator, *db.DB, *openstack.MockClient) {
|
||||
t.Helper()
|
||||
ctx := context.Background()
|
||||
dbPath := filepath.Join(t.TempDir(), "test.db")
|
||||
d, err := db.Open(ctx, dbPath)
|
||||
if err != nil {
|
||||
t.Fatalf("open db: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { d.Close() })
|
||||
|
||||
mock := openstack.NewMockClient()
|
||||
|
||||
cfg := &config.ControlAPI{
|
||||
Orchestrator: config.OrchestratorConfig{
|
||||
PollIntervalSeconds: 1,
|
||||
SelfCheckTimeoutSeconds: 10,
|
||||
MaxSelfCheckRetries: 3,
|
||||
CheckingWindowSeconds: 120,
|
||||
MaxRetries: 3,
|
||||
LeaseTTLSeconds: leaseTTLSeconds,
|
||||
HeartbeatTimeoutSeconds: 30,
|
||||
},
|
||||
Aggregation: config.AggregationConfig{MissingCountsAsFail: true},
|
||||
Sites: []config.SiteConfig{
|
||||
{SiteID: "site-1", Index: 1},
|
||||
{SiteID: "site-2", Index: 2},
|
||||
{SiteID: "site-3", Index: 3},
|
||||
},
|
||||
CheckTypes: []config.CheckTypeConfig{
|
||||
{Name: "https", Enabled: true, Targets: []string{"web"}},
|
||||
{Name: "ssh", Enabled: false, Targets: []string{"web"}},
|
||||
},
|
||||
Targets: map[string][]string{
|
||||
"web": {"https://example.test"},
|
||||
},
|
||||
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
|
||||
}
|
||||
|
||||
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
||||
return New(d, mock, cfg, log), d, mock
|
||||
}
|
||||
|
||||
func TestHappyPath(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
||||
t.Fatalf("register validator: %v", err)
|
||||
}
|
||||
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
||||
t.Fatalf("seed queue: %v", err)
|
||||
}
|
||||
|
||||
// 1. claim + associate
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if err != nil {
|
||||
t.Fatalf("get ip: %v", err)
|
||||
}
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
||||
}
|
||||
if ip.FIPID != "fip-1" {
|
||||
t.Fatalf("expected fip-1 associated, got %q", ip.FIPID)
|
||||
}
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
||||
t.Fatalf("expected fip associated to port-1, got %q", fip.PortID)
|
||||
}
|
||||
|
||||
v, err := d.GetValidator(ctx, "validator-1")
|
||||
if err != nil {
|
||||
t.Fatalf("get validator: %v", err)
|
||||
}
|
||||
if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != ip.ID {
|
||||
t.Fatalf("expected validator assigned to ip %d, got state=%s current_ip=%v", ip.ID, v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
// 2. self-check success
|
||||
if err := o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "egress matched"); err != nil {
|
||||
t.Fatalf("self check result: %v", err)
|
||||
}
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPChecking {
|
||||
t.Fatalf("expected checking, got %s", ip.State)
|
||||
}
|
||||
|
||||
// 3. egress result + completion
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record egress check: %v", err)
|
||||
}
|
||||
if err := o.MarkEgressComplete(ctx, ip.ID); err != nil {
|
||||
t.Fatalf("mark egress complete: %v", err)
|
||||
}
|
||||
|
||||
// 4. inbound results from all 3 sites
|
||||
for site := 1; site <= 3; site++ {
|
||||
for _, ct := range []string{"tcp-22", "tcp-80", "icmp"} {
|
||||
if err := o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress,
|
||||
Success: true, CheckedAt: db.Now(),
|
||||
}); err != nil {
|
||||
t.Fatalf("record inbound check site %d: %v", site, err)
|
||||
}
|
||||
}
|
||||
if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil {
|
||||
t.Fatalf("mark site %d complete: %v", site, err)
|
||||
}
|
||||
}
|
||||
|
||||
// 5. sweep should now aggregate + release
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
if ip.OverallResult != db.ResultPass {
|
||||
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
||||
}
|
||||
if ip.FIPReleasedAt == nil {
|
||||
t.Fatalf("expected fip_released_at to be set")
|
||||
}
|
||||
|
||||
v, _ = d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("expected validator idle with no current ip, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated, still on port %q", fip.PortID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPartialResult(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 180)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx)
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
|
||||
// Egress passes...
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
||||
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.MarkEgressComplete(ctx, ip.ID)
|
||||
// ...but only site-1 reports, and one of its checks fails.
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "tcp-22", Target: ip.IPAddress, Success: false, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "tcp-80", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.RecordCheck(ctx, db.Check{
|
||||
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
||||
Source: db.InboundSource(1), CheckType: "icmp", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
||||
})
|
||||
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
||||
|
||||
// Force the checking window to have elapsed so aggregation proceeds
|
||||
// even though site-2/site-3 never reported.
|
||||
o.Cfg.CheckingWindowSeconds = 0
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
o.Tick(ctx)
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPDone {
|
||||
t.Fatalf("expected done, got %s", ip.State)
|
||||
}
|
||||
if ip.OverallResult != db.ResultPartial {
|
||||
t.Fatalf("expected partial, got %s", ip.OverallResult)
|
||||
}
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated even on partial result")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseReclaim(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
// A 1s lease (rather than 0) avoids a race within the very first Tick:
|
||||
// with a 0s TTL the item's lease can already look expired by the time
|
||||
// the same Tick's lease-sweep phase runs, depending on how much
|
||||
// wall-clock time the claim+associate phase happened to take.
|
||||
o, d, mock := newTestOrchestrator(t, 1)
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
o.Tick(ctx) // claims + associates; validator never self-checks
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
||||
}
|
||||
|
||||
time.Sleep(1100 * time.Millisecond) // let the 1s lease expire
|
||||
o.Tick(ctx) // should reclaim via lease sweep
|
||||
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPQueued {
|
||||
t.Fatalf("expected requeued after lease reclaim, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
||||
}
|
||||
if ip.RetryCount != 1 {
|
||||
t.Fatalf("expected retry_count=1, got %d", ip.RetryCount)
|
||||
}
|
||||
|
||||
v, _ := d.GetValidator(ctx, "validator-1")
|
||||
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
||||
t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
||||
}
|
||||
|
||||
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
||||
t.Fatalf("expected fip disassociated on reclaim, still on port %q", fip.PortID)
|
||||
}
|
||||
|
||||
// A subsequent tick should re-claim and re-associate the same IP for
|
||||
// the now-idle validator, proving the queue keeps making progress.
|
||||
// Give this attempt a real lease so it isn't immediately re-expired by
|
||||
// the same tick's lease sweep (a 0s TTL, as above, expires instantly).
|
||||
o.Cfg.LeaseTTLSeconds = 180
|
||||
o.Tick(ctx)
|
||||
ip, _ = d.GetIP(ctx, ip.ID)
|
||||
if ip.State != db.IPAwaitingSelfCheck {
|
||||
t.Fatalf("expected re-claimed ip to be awaiting_self_check again, got %s", ip.State)
|
||||
}
|
||||
if ip.AttemptNumber != 2 {
|
||||
t.Fatalf("expected attempt_number=2 after reclaim+reassign, got %d", ip.AttemptNumber)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMaxRetriesExhausted(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
o, d, mock := newTestOrchestrator(t, 0)
|
||||
o.Cfg.MaxRetries = 1
|
||||
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
||||
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
||||
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
||||
|
||||
for i := 0; i < 3; i++ {
|
||||
o.Tick(ctx)
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
|
||||
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
||||
if ip.State != db.IPFailed {
|
||||
t.Fatalf("expected failed after exhausting retries, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user