The cloud is live: the address list submitted as "free" (bootstrap config
or POST /api/v1/admin/ips) can drift by the time the orchestrator claims
it, or an operator can queue an already-occupied address by mistake.
Neutron's floating-IP association is a blind "last write wins" PUT with
no conflict error to catch, so associateFIP now checks the FIP's PortID
(already fetched via GetFloatingIPByAddress) before associating, guarded
against the false-positive of the FIP already belonging to this same
validator's own port.
A match routes the address straight to a new terminal ip_queue.state
("occupied", distinct from failed/fail) via db.MarkFIPOccupied — no
retries, since Neutron won't free it on its own and requeuing would let
it be reclaimed again next tick, starving the rest of the queue — plus a
dedicated fip_occupied audit event. Resubmitting the address later (once
the conflict is resolved) resets it to queued via the existing
POST /api/v1/admin/ips resubmit path (CancelIP/ListExpiredLeases updated
to treat occupied as terminal too). admin-dashboard gets its own "занят"
badge, distinct from fail/partial/cancelled.
Rebuilt bin/{control-api,admin-dashboard,prober,validator-agent} and
bin/SHA256SUMS per docs/SETUP.md's documented build recipe, since
control-api and admin-dashboard source changed.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01NeVbMVEiE7XQAkBd7HQgj6
885 lines
32 KiB
Go
885 lines
32 KiB
Go
package orchestrator
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"log/slog"
|
|
"os"
|
|
"path/filepath"
|
|
"testing"
|
|
"time"
|
|
|
|
"cloudipvalidator/internal/config"
|
|
"cloudipvalidator/internal/db"
|
|
"cloudipvalidator/internal/openstack"
|
|
)
|
|
|
|
var threeSites = []config.SiteConfig{
|
|
{SiteID: "site-1", Index: 1},
|
|
{SiteID: "site-2", Index: 2},
|
|
{SiteID: "site-3", Index: 3},
|
|
}
|
|
|
|
func newTestOrchestrator(t *testing.T, leaseTTLSeconds int) (*Orchestrator, *db.DB, *openstack.MockClient) {
|
|
t.Helper()
|
|
return newTestOrchestratorWithSites(t, leaseTTLSeconds, threeSites)
|
|
}
|
|
|
|
func newTestOrchestratorWithSites(t *testing.T, leaseTTLSeconds int, sites []config.SiteConfig) (*Orchestrator, *db.DB, *openstack.MockClient) {
|
|
t.Helper()
|
|
ctx := context.Background()
|
|
dbPath := filepath.Join(t.TempDir(), "test.db")
|
|
d, err := db.Open(ctx, dbPath)
|
|
if err != nil {
|
|
t.Fatalf("open db: %v", err)
|
|
}
|
|
t.Cleanup(func() { d.Close() })
|
|
|
|
mock := openstack.NewMockClient()
|
|
|
|
cfg := &config.ControlAPI{
|
|
Orchestrator: config.OrchestratorConfig{
|
|
PollIntervalSeconds: 1,
|
|
SelfCheckTimeoutSeconds: 10,
|
|
MaxSelfCheckRetries: 3,
|
|
CheckingWindowSeconds: 120,
|
|
MaxRetries: 3,
|
|
LeaseTTLSeconds: leaseTTLSeconds,
|
|
HeartbeatTimeoutSeconds: 30,
|
|
},
|
|
Aggregation: config.AggregationConfig{MissingCountsAsFail: true},
|
|
Sites: sites,
|
|
CheckTypes: []config.CheckTypeConfig{
|
|
{Name: "https", Enabled: true, Targets: []string{"web"}},
|
|
{Name: "ssh", Enabled: false, Targets: []string{"web"}},
|
|
},
|
|
Targets: map[string][]string{
|
|
"web": {"https://example.test"},
|
|
},
|
|
Inbound: config.InboundConfig{Ports: []int{22, 80}, ICMP: true},
|
|
}
|
|
|
|
if err := d.BootstrapFromConfig(ctx, cfg); err != nil {
|
|
t.Fatalf("bootstrap from config: %v", err)
|
|
}
|
|
|
|
log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: slog.LevelError}))
|
|
return New(d, mock, cfg, log), d, mock
|
|
}
|
|
|
|
func TestHappyPath(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
|
t.Fatalf("register validator: %v", err)
|
|
}
|
|
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
|
t.Fatalf("seed queue: %v", err)
|
|
}
|
|
|
|
// 1. claim + associate
|
|
o.Tick(ctx)
|
|
|
|
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if err != nil {
|
|
t.Fatalf("get ip: %v", err)
|
|
}
|
|
if ip.State != db.IPAwaitingSelfCheck {
|
|
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
|
}
|
|
if ip.FIPID != "fip-1" {
|
|
t.Fatalf("expected fip-1 associated, got %q", ip.FIPID)
|
|
}
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "port-1" {
|
|
t.Fatalf("expected fip associated to port-1, got %q", fip.PortID)
|
|
}
|
|
|
|
v, err := d.GetValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("get validator: %v", err)
|
|
}
|
|
if v.State != db.ValidatorAssigned || v.CurrentIPID == nil || *v.CurrentIPID != ip.ID {
|
|
t.Fatalf("expected validator assigned to ip %d, got state=%s current_ip=%v", ip.ID, v.State, v.CurrentIPID)
|
|
}
|
|
|
|
// 2. self-check success
|
|
if err := o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "egress matched"); err != nil {
|
|
t.Fatalf("self check result: %v", err)
|
|
}
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPChecking {
|
|
t.Fatalf("expected checking, got %s", ip.State)
|
|
}
|
|
|
|
// 3. egress result + completion
|
|
if err := o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
}); err != nil {
|
|
t.Fatalf("record egress check: %v", err)
|
|
}
|
|
if err := o.MarkEgressComplete(ctx, ip.ID); err != nil {
|
|
t.Fatalf("mark egress complete: %v", err)
|
|
}
|
|
|
|
// 4. inbound results from all 3 sites
|
|
for site := 1; site <= 3; site++ {
|
|
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
|
|
if err := o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress,
|
|
Success: true, CheckedAt: db.Now(),
|
|
}); err != nil {
|
|
t.Fatalf("record inbound check site %d: %v", site, err)
|
|
}
|
|
}
|
|
if err := o.MarkSiteComplete(ctx, ip.ID, site); err != nil {
|
|
t.Fatalf("mark site %d complete: %v", site, err)
|
|
}
|
|
}
|
|
|
|
// 5. sweep should now aggregate + release
|
|
o.Tick(ctx)
|
|
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
if ip.FIPReleasedAt == nil {
|
|
t.Fatalf("expected fip_released_at to be set")
|
|
}
|
|
|
|
v, _ = d.GetValidator(ctx, "validator-1")
|
|
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
|
t.Fatalf("expected validator idle with no current ip, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
|
}
|
|
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
|
t.Fatalf("expected fip disassociated, still on port %q", fip.PortID)
|
|
}
|
|
}
|
|
|
|
// TestFIPAlreadyOccupiedByAnotherPortSkipsCheckCycle covers the defense
|
|
// against a Floating IP that turns out to already be attached to some other
|
|
// VM's port at claim time — the cloud is live, so the "free" list supplied
|
|
// at bootstrap/via the admin API can drift, or an operator can mistakenly
|
|
// queue an already-occupied address. The address must be terminated as
|
|
// `occupied` immediately, without ever entering the check cycle, without
|
|
// stealing the port, and with the validator freed back to idle so the rest
|
|
// of the queue isn't starved behind it.
|
|
func TestFIPAlreadyOccupiedByAnotherPortSkipsCheckCycle(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
|
|
mock.SeedWithPort("fip-1", "1.2.3.4", "svc-project", "someone-elses-port")
|
|
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
|
t.Fatalf("register validator: %v", err)
|
|
}
|
|
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
|
t.Fatalf("seed queue: %v", err)
|
|
}
|
|
|
|
o.Tick(ctx)
|
|
|
|
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if err != nil {
|
|
t.Fatalf("get ip: %v", err)
|
|
}
|
|
if ip.State != db.IPOccupied {
|
|
t.Fatalf("expected occupied, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != "" {
|
|
t.Fatalf("expected empty overall_result, got %q", ip.OverallResult)
|
|
}
|
|
if ip.OwnerValidatorID != nil {
|
|
t.Fatalf("expected no owning validator, got %v", *ip.OwnerValidatorID)
|
|
}
|
|
if ip.FIPID != "" {
|
|
t.Fatalf("expected no fip_id recorded, got %q", ip.FIPID)
|
|
}
|
|
|
|
v, err := d.GetValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("get validator: %v", err)
|
|
}
|
|
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
|
t.Fatalf("expected validator freed back to idle, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
|
}
|
|
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "someone-elses-port" {
|
|
t.Fatalf("expected fip's port untouched (not stolen), got %q", fip.PortID)
|
|
}
|
|
|
|
events, err := d.ListEventsForIP(ctx, ip.ID)
|
|
if err != nil {
|
|
t.Fatalf("list events: %v", err)
|
|
}
|
|
found := false
|
|
for _, e := range events {
|
|
if e.EventType == "fip_occupied" {
|
|
found = true
|
|
}
|
|
}
|
|
if !found {
|
|
t.Fatalf("expected a fip_occupied event, got %+v", events)
|
|
}
|
|
}
|
|
|
|
// TestFIPAssociatedToOwnValidatorPortIsNotOccupied guards against a false
|
|
// positive: a Floating IP already attached to the very validator we're
|
|
// about to associate it with (e.g. control-api restarted between the
|
|
// OpenStack call and recording it in the DB) is a resume, not a conflict —
|
|
// it must proceed through the normal happy path, not be flagged occupied.
|
|
func TestFIPAssociatedToOwnValidatorPortIsNotOccupied(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
|
|
mock.SeedWithPort("fip-1", "1.2.3.4", "svc-project", "port-1")
|
|
if err := d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1"); err != nil {
|
|
t.Fatalf("register validator: %v", err)
|
|
}
|
|
if err := d.SeedQueue(ctx, []string{"1.2.3.4"}); err != nil {
|
|
t.Fatalf("seed queue: %v", err)
|
|
}
|
|
|
|
o.Tick(ctx)
|
|
|
|
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if err != nil {
|
|
t.Fatalf("get ip: %v", err)
|
|
}
|
|
if ip.State != db.IPAwaitingSelfCheck {
|
|
t.Fatalf("expected awaiting_self_check (not occupied), got %s", ip.State)
|
|
}
|
|
}
|
|
|
|
func TestPartialResult(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
// Egress passes...
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
// ...but only site-1 reports, and one of its checks fails.
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: "tcp-22", Target: ip.IPAddress, Success: false, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: "tcp-80", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: "icmp", Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
|
|
|
// Force the checking window to have elapsed so aggregation proceeds
|
|
// even though site-2/site-3 never reported.
|
|
o.Cfg.CheckingWindowSeconds = 0
|
|
time.Sleep(5 * time.Millisecond)
|
|
o.Tick(ctx)
|
|
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPartial {
|
|
t.Fatalf("expected partial, got %s", ip.OverallResult)
|
|
}
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
|
t.Fatalf("expected fip disassociated even on partial result")
|
|
}
|
|
}
|
|
|
|
func TestLeaseReclaim(t *testing.T) {
|
|
ctx := context.Background()
|
|
// A 1s lease (rather than 0) avoids a race within the very first Tick:
|
|
// with a 0s TTL the item's lease can already look expired by the time
|
|
// the same Tick's lease-sweep phase runs, depending on how much
|
|
// wall-clock time the claim+associate phase happened to take.
|
|
o, d, mock := newTestOrchestrator(t, 1)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx) // claims + associates; validator never self-checks
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if ip.State != db.IPAwaitingSelfCheck {
|
|
t.Fatalf("expected awaiting_self_check, got %s", ip.State)
|
|
}
|
|
|
|
time.Sleep(1100 * time.Millisecond) // let the 1s lease expire
|
|
o.Tick(ctx) // should reclaim via lease sweep
|
|
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPQueued {
|
|
t.Fatalf("expected requeued after lease reclaim, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
|
}
|
|
if ip.RetryCount != 1 {
|
|
t.Fatalf("expected retry_count=1, got %d", ip.RetryCount)
|
|
}
|
|
|
|
v, _ := d.GetValidator(ctx, "validator-1")
|
|
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
|
t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
|
}
|
|
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
|
t.Fatalf("expected fip disassociated on reclaim, still on port %q", fip.PortID)
|
|
}
|
|
|
|
// A subsequent tick should re-claim and re-associate the same IP for
|
|
// the now-idle validator, proving the queue keeps making progress.
|
|
// Give this attempt a real lease so it isn't immediately re-expired by
|
|
// the same tick's lease sweep (a 0s TTL, as above, expires instantly).
|
|
o.Cfg.LeaseTTLSeconds = 180
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPAwaitingSelfCheck {
|
|
t.Fatalf("expected re-claimed ip to be awaiting_self_check again, got %s", ip.State)
|
|
}
|
|
if ip.AttemptNumber != 2 {
|
|
t.Fatalf("expected attempt_number=2 after reclaim+reassign, got %d", ip.AttemptNumber)
|
|
}
|
|
}
|
|
|
|
func TestMaxRetriesExhausted(t *testing.T) {
|
|
ctx := context.Background()
|
|
// A 1s lease (rather than 0) avoids the same intra-tick race noted in
|
|
// TestLeaseReclaim: with a 0s TTL, whether a freshly claimed item is
|
|
// reclaimed within the very same Tick (making each iteration's timing
|
|
// unpredictable) depends on how much wall-clock time claim+associate
|
|
// happened to take, which made this test flaky under load.
|
|
o, d, mock := newTestOrchestrator(t, 1)
|
|
o.Cfg.MaxRetries = 1
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
// Each (claim, sleep past 1s lease) pair reclaims once, bumping
|
|
// retry_count by 1; with MaxRetries=1 that needs two full reclaim
|
|
// cycles (retry_count 0->1 requeues, 1->2 fails) — four ticks total:
|
|
// claim, reclaim+requeue, re-claim, reclaim+fail.
|
|
for i := 0; i < 4; i++ {
|
|
o.Tick(ctx)
|
|
time.Sleep(1100 * time.Millisecond)
|
|
}
|
|
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if ip.State != db.IPFailed {
|
|
t.Fatalf("expected failed after exhausting retries, got %s (retry_count=%d)", ip.State, ip.RetryCount)
|
|
}
|
|
}
|
|
|
|
// TestInboundChecksDisabled confirms inbound (prober) checks are genuinely
|
|
// optional: with no sites configured, an IP must aggregate as soon as
|
|
// egress completes, without ever waiting on siteN_complete flags that
|
|
// nothing will ever set — and, critically, without waiting out the full
|
|
// checking_window_seconds timeout to get there (newTestOrchestrator uses
|
|
// 120s; this test never sleeps, so a pass here proves the "all required
|
|
// sources complete" path fired, not the timeout fallback).
|
|
func TestInboundChecksDisabled(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, nil)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
o.Tick(ctx)
|
|
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done immediately after egress completed with no sites configured, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
}
|
|
|
|
// TestDeleteIPDisassociatesFIP proves DeleteIP disassociates a currently
|
|
// attached floating IP (via the mock) before permanently removing the
|
|
// address — the same resource-freeing ForceCancel does, but going straight
|
|
// to physical deletion instead of a `cancelled` record.
|
|
func TestDeleteIPDisassociatesFIP(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx) // claim + associate -> awaiting_self_check, fip attached
|
|
|
|
ip, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if err != nil {
|
|
t.Fatalf("get ip: %v", err)
|
|
}
|
|
if ip.FIPID == "" {
|
|
t.Fatalf("expected fip associated before delete")
|
|
}
|
|
|
|
if err := o.DeleteIP(ctx, "1.2.3.4"); err != nil {
|
|
t.Fatalf("delete ip: %v", err)
|
|
}
|
|
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
|
t.Fatalf("expected fip disassociated on delete, still on port %q", fip.PortID)
|
|
}
|
|
if _, err := d.GetIPByAddress(ctx, "1.2.3.4"); err == nil {
|
|
t.Fatalf("expected ip row gone after delete")
|
|
}
|
|
v, err := d.GetValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("get validator: %v", err)
|
|
}
|
|
if v.State != db.ValidatorIdle || v.CurrentIPID != nil {
|
|
t.Fatalf("expected validator freed, got state=%s current_ip=%v", v.State, v.CurrentIPID)
|
|
}
|
|
|
|
if err := o.DeleteIP(ctx, "1.2.3.4"); !errors.Is(err, db.ErrNotFound) {
|
|
t.Fatalf("expected ErrNotFound deleting already-gone ip, got %v", err)
|
|
}
|
|
}
|
|
|
|
// TestClearQueueDisassociatesAllFIPs proves ClearQueue deletes every
|
|
// address regardless of state and disassociates any attached floating IPs
|
|
// along the way.
|
|
func TestClearQueueDisassociatesAllFIPs(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4", "5.6.7.8"})
|
|
|
|
o.Tick(ctx) // claims + associates 1.2.3.4; 5.6.7.8 stays queued
|
|
|
|
result, err := o.ClearQueue(ctx)
|
|
if err != nil {
|
|
t.Fatalf("clear queue: %v", err)
|
|
}
|
|
if len(result.Deleted) != 2 {
|
|
t.Fatalf("expected both addresses deleted, got %+v", result)
|
|
}
|
|
if fip, _ := mock.GetFloatingIPByAddress(ctx, "1.2.3.4"); fip.PortID != "" {
|
|
t.Fatalf("expected fip disassociated on clear, still on port %q", fip.PortID)
|
|
}
|
|
ips, err := d.ListIPs(ctx)
|
|
if err != nil {
|
|
t.Fatalf("list ips: %v", err)
|
|
}
|
|
if len(ips) != 0 {
|
|
t.Fatalf("expected empty queue after clear, got %+v", ips)
|
|
}
|
|
}
|
|
|
|
// TestFIPSettleDelayWithholdsAssignment proves a nonzero fip_settle_seconds
|
|
// keeps AssignmentForValidator returning nil (agent keeps polling and
|
|
// getting nothing) until the configured pause has elapsed since the
|
|
// floating IP was associated.
|
|
func TestFIPSettleDelayWithholdsAssignment(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
if err := o.SetFIPSettleSeconds(ctx, 1); err != nil {
|
|
t.Fatalf("set fip settle seconds: %v", err)
|
|
}
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx) // claim + associate -> awaiting_self_check, fip attached
|
|
|
|
item, checks, err := o.AssignmentForValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("assignment for validator: %v", err)
|
|
}
|
|
if item != nil || checks != nil {
|
|
t.Fatalf("expected no assignment during settle window, got item=%+v checks=%+v", item, checks)
|
|
}
|
|
|
|
time.Sleep(1100 * time.Millisecond)
|
|
|
|
item, checks, err = o.AssignmentForValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("assignment for validator after settle: %v", err)
|
|
}
|
|
if item == nil {
|
|
t.Fatalf("expected assignment to be available once settle window elapsed")
|
|
}
|
|
if len(checks) == 0 {
|
|
t.Fatalf("expected check config to be returned alongside the item")
|
|
}
|
|
}
|
|
|
|
// TestFIPSettleDelayZeroIsNoOp proves the default fip_settle_seconds=0
|
|
// never withholds an assignment — no behavior change for upgraded-but-
|
|
// unconfigured installs.
|
|
func TestFIPSettleDelayZeroIsNoOp(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
|
|
item, _, err := o.AssignmentForValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("assignment for validator: %v", err)
|
|
}
|
|
if item == nil {
|
|
t.Fatalf("expected assignment immediately available with fip_settle_seconds=0")
|
|
}
|
|
}
|
|
|
|
// TestFIPSettleDelayIgnoresNilFIPAssociatedAt proves an in-flight row from
|
|
// before this feature's migration (fip_associated_at NULL) is never newly
|
|
// blocked, even with a nonzero configured pause — upgrading control-api
|
|
// must not strand an address already awaiting self-check.
|
|
func TestFIPSettleDelayIgnoresNilFIPAssociatedAt(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, d, mock := newTestOrchestrator(t, 180)
|
|
if err := o.SetFIPSettleSeconds(ctx, 60); err != nil {
|
|
t.Fatalf("set fip settle seconds: %v", err)
|
|
}
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
|
|
item, err := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
if err != nil {
|
|
t.Fatalf("get ip: %v", err)
|
|
}
|
|
if _, err := d.ExecContext(ctx, `UPDATE ip_queue SET fip_associated_at=NULL WHERE id=?`, item.ID); err != nil {
|
|
t.Fatalf("clear fip_associated_at: %v", err)
|
|
}
|
|
|
|
assigned, _, err := o.AssignmentForValidator(ctx, "validator-1")
|
|
if err != nil {
|
|
t.Fatalf("assignment for validator: %v", err)
|
|
}
|
|
if assigned == nil {
|
|
t.Fatalf("expected assignment available for a row with nil FIPAssociatedAt despite nonzero settle delay")
|
|
}
|
|
}
|
|
|
|
// TestSetFIPSettleSecondsValidation proves the cross-field rule against
|
|
// lease_ttl_seconds/self_check_timeout_seconds is enforced.
|
|
func TestSetFIPSettleSecondsValidation(t *testing.T) {
|
|
ctx := context.Background()
|
|
o, _, _ := newTestOrchestrator(t, 180) // SelfCheckTimeoutSeconds: 10 (see newTestOrchestratorWithSites)
|
|
|
|
cases := []struct {
|
|
seconds int
|
|
wantErr bool
|
|
}{
|
|
{-1, true},
|
|
{170, true}, // 170+10 >= 180
|
|
{169, false},
|
|
{0, false},
|
|
}
|
|
for _, c := range cases {
|
|
err := o.SetFIPSettleSeconds(ctx, c.seconds)
|
|
if c.wantErr && !errors.Is(err, db.ErrValidation) {
|
|
t.Errorf("seconds=%d: expected ErrValidation, got %v", c.seconds, err)
|
|
}
|
|
if !c.wantErr && err != nil {
|
|
t.Errorf("seconds=%d: expected success, got %v", c.seconds, err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestInboundChecksPartialSites confirms a partially-configured sites list
|
|
// (fewer than 3 slots assigned) only waits on the sites actually
|
|
// configured — the two unassigned slots are never expected.
|
|
func TestInboundChecksPartialSites(t *testing.T) {
|
|
ctx := context.Background()
|
|
sites := []config.SiteConfig{{SiteID: "site-1", Index: 1}}
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
// Egress is done but the one configured site (site-1) hasn't reported
|
|
// yet — must not aggregate.
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPChecking {
|
|
t.Fatalf("expected still checking (site-1 pending), got %s", ip.State)
|
|
}
|
|
|
|
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
|
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done once the single configured site reported, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
}
|
|
|
|
// TestExpectedCheckCountReflectsInboundChecksConfigChange confirms
|
|
// expectedCheckCount reads inbound_checks (ports/icmp) from the database on
|
|
// every use, not a cached copy — same "accepted tradeoff" already true of
|
|
// check_types/targets/sites (see expectedCheckCount's doc comment).
|
|
func TestExpectedCheckCountReflectsInboundChecksConfigChange(t *testing.T) {
|
|
ctx := context.Background()
|
|
sites := []config.SiteConfig{{SiteID: "site-1", Index: 1}}
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
|
// newTestOrchestratorWithSites seeds Inbound: {Ports: [22, 80], ICMP: true}
|
|
// (4 inbound checks expected per site: tcp-22, ssh, tcp-80, icmp).
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
// Shrink the inbound check config to a single TCP port, no ICMP — an
|
|
// admin API change made between assignment and aggregation. Port 22
|
|
// still auto-triggers the extra "ssh" check (see expectedCheckCount).
|
|
if err := d.SetInboundChecks(ctx, []int{22}, false); err != nil {
|
|
t.Fatalf("set inbound checks: %v", err)
|
|
}
|
|
|
|
// Report only the two now-expected inbound checks.
|
|
for _, ct := range []string{"tcp-22", "ssh"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
|
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done once the (shrunk) inbound config was fully reported, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
}
|
|
|
|
// TestFourSitesAllMustReportBeforeAggregation confirms there's no hardcoded
|
|
// cap of three sites: with 4 sites configured, aggregation must wait on
|
|
// all 4, not silently treat the 4th as always-complete.
|
|
func TestFourSitesAllMustReportBeforeAggregation(t *testing.T) {
|
|
ctx := context.Background()
|
|
sites := []config.SiteConfig{
|
|
{SiteID: "site-1", Index: 1}, {SiteID: "site-2", Index: 2},
|
|
{SiteID: "site-3", Index: 3}, {SiteID: "site-4", Index: 4},
|
|
}
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
for _, site := range []int{1, 2, 3} {
|
|
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(site), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, site)
|
|
}
|
|
|
|
// Sites 1-3 reported, site 4 (the case beyond the old fixed-3 cap)
|
|
// hasn't — must not aggregate yet.
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPChecking {
|
|
t.Fatalf("expected still checking (site-4 pending), got %s", ip.State)
|
|
}
|
|
|
|
for _, ct := range []string{"tcp-22", "ssh", "tcp-80", "icmp"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(4), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 4)
|
|
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done once all 4 sites reported, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
}
|
|
|
|
// TestExpectedCheckCountIncludesTLSAndSSHForPorts443And22 confirms
|
|
// expectedCheckCount counts the auto-triggered tls-443/ssh checks (in
|
|
// addition to the base tcp-443/tcp-22) once those ports are configured —
|
|
// reporting exactly that full set must be enough to aggregate as pass.
|
|
func TestExpectedCheckCountIncludesTLSAndSSHForPorts443And22(t *testing.T) {
|
|
ctx := context.Background()
|
|
sites := []config.SiteConfig{{SiteID: "site-1", Index: 1}}
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
if err := d.SetInboundChecks(ctx, []int{22, 443}, false); err != nil {
|
|
t.Fatalf("set inbound checks: %v", err)
|
|
}
|
|
|
|
for _, ct := range []string{"tcp-22", "ssh", "tcp-443", "tls-443"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
|
|
|
o.Tick(ctx)
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done once tcp+tls+ssh all reported, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPass {
|
|
t.Fatalf("expected pass, got %s", ip.OverallResult)
|
|
}
|
|
}
|
|
|
|
// TestAggregationWaitsForTLSAndSSHResults confirms that reporting only the
|
|
// base tcp-22/tcp-443 checks (without the auto-triggered ssh/tls-443
|
|
// checks) is not enough to aggregate as pass — expectedCheckCount must
|
|
// actually have grown, not just failed to break.
|
|
func TestAggregationWaitsForTLSAndSSHResults(t *testing.T) {
|
|
ctx := context.Background()
|
|
sites := []config.SiteConfig{{SiteID: "site-1", Index: 1}}
|
|
o, d, mock := newTestOrchestratorWithSites(t, 180, sites)
|
|
mock.Seed("fip-1", "1.2.3.4", "svc-project")
|
|
_ = d.RegisterValidator(ctx, "validator-1", "host-1", "port-1", "v0.1")
|
|
_ = d.SeedQueue(ctx, []string{"1.2.3.4"})
|
|
|
|
o.Tick(ctx)
|
|
ip, _ := d.GetIPByAddress(ctx, "1.2.3.4")
|
|
_ = o.SelfCheckResult(ctx, "validator-1", ip.ID, true, "ok")
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
ValidatorID: "validator-1", Source: db.SourceEgress, CheckType: "https",
|
|
Target: "https://example.test", Success: true, CheckedAt: db.Now(),
|
|
})
|
|
_ = o.MarkEgressComplete(ctx, ip.ID)
|
|
|
|
if err := d.SetInboundChecks(ctx, []int{22, 443}, false); err != nil {
|
|
t.Fatalf("set inbound checks: %v", err)
|
|
}
|
|
|
|
// Only the base TCP checks report — ssh/tls-443 never arrive, but the
|
|
// site still (incorrectly, from the operator's point of view) claims
|
|
// completion. Force the checking window to have elapsed so aggregation
|
|
// runs anyway, same as TestPartialResult.
|
|
for _, ct := range []string{"tcp-22", "tcp-443"} {
|
|
_ = o.RecordCheck(ctx, db.Check{
|
|
IPID: ip.ID, IPAddress: ip.IPAddress, AttemptNumber: ip.AttemptNumber,
|
|
Source: db.InboundSource(1), CheckType: ct, Target: ip.IPAddress, Success: true, CheckedAt: db.Now(),
|
|
})
|
|
}
|
|
_ = o.MarkSiteComplete(ctx, ip.ID, 1)
|
|
|
|
o.Cfg.CheckingWindowSeconds = 0
|
|
time.Sleep(5 * time.Millisecond)
|
|
o.Tick(ctx)
|
|
|
|
ip, _ = d.GetIP(ctx, ip.ID)
|
|
if ip.State != db.IPDone {
|
|
t.Fatalf("expected done, got %s", ip.State)
|
|
}
|
|
if ip.OverallResult != db.ResultPartial {
|
|
t.Fatalf("expected partial (missing ssh/tls-443 counted against it), got %s", ip.OverallResult)
|
|
}
|
|
}
|