Files
ayurishchevandClaude Sonnet 5.5 abbee9a08a Add self-check via control-api (self_check.methods)
control-api is hosted outside the cloud and validators reach it directly,
so it sees the floating IP as the connection's source address. New open
route GET /api/v1/agents/{id}/observed-ip returns that address (taken only
from the TCP peer; forwarding headers are ignored so a validator cannot
forge it).

The agent gets self_check.methods, a priority-ordered list of ip_echo
(unchanged) and control_api; the default stays [ip_echo]. The self-check
passes when any method confirms the address; the next method is tried on
no answer and on a mismatch. Each method has its own timeout so a hung
first method cannot starve the fallback, and control_api uses a new TCP
connection per call (a connection opened before the floating IP was
attached would keep reporting the old address).

Also: docker agent template/env, example config, docs, plan in
docs/changes, e2e script switch E2E_SELF_CHECK_METHODS, rebuilt
bin/control-api and bin/validator-agent.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-10-02 03:24:20 +03:00

326 lines
12 KiB
Bash
Executable File

#!/usr/bin/env bash
# Offline end-to-end smoke test for the Cloud IP Validator (see
# docs/LOCAL_E2E.md). Runs control-api + 1 validator-agent + 3 probers as
# local processes against a mock OpenStack backend, with loopback stand-ins
# for both the "floating IP" under test and the outbound targets — no real
# cloud or internet access required.
#
# Usage: scripts/run-local-e2e.sh
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$REPO_ROOT"
GO="${GO:-go}"
if ! command -v "$GO" >/dev/null 2>&1 && [ -x /usr/local/go/bin/go ]; then
GO=/usr/local/go/bin/go
fi
WORK_DIR="$(mktemp -d /tmp/cloud-ip-validator-e2e.XXXXXX)"
BIN_DIR="$WORK_DIR/bin"
LOG_DIR="$WORK_DIR/logs"
mkdir -p "$BIN_DIR" "$LOG_DIR"
# Static bearer tokens (fixed, test-only values). control-api enforces both;
# validator-agent and prober pick the agent token up from the same variable
# (the default name of their control_api_token_env); admin curls below send the
# admin token explicitly.
export CONTROL_API_ADMIN_TOKEN="e2e-admin-token-0123456789abcdef"
export CONTROL_API_AGENT_TOKEN="e2e-agent-token-fedcba9876543210"
ADMIN_AUTH=(-H "Authorization: Bearer $CONTROL_API_ADMIN_TOKEN")
PIDS=()
cleanup() {
echo "--- cleaning up (workdir: $WORK_DIR) ---"
for pid in "${PIDS[@]:-}"; do
kill "$pid" >/dev/null 2>&1 || true
done
wait >/dev/null 2>&1 || true
}
trap cleanup EXIT
echo "--- building binaries ---"
"$GO" build -o "$BIN_DIR/control-api" ./cmd/control-api
"$GO" build -o "$BIN_DIR/validator-agent" ./cmd/validator-agent
"$GO" build -o "$BIN_DIR/prober" ./cmd/prober
"$GO" build -o "$BIN_DIR/httpstub" ./scripts/httpstub
echo "--- starting stub outbound targets (stand-ins for hub.docker.com / github.com / packages.ubuntu.com) ---"
"$BIN_DIR/httpstub" -addr 127.0.0.1:29091 >"$LOG_DIR/stub-1.log" 2>&1 & PIDS+=($!)
"$BIN_DIR/httpstub" -addr 127.0.0.1:29092 >"$LOG_DIR/stub-2.log" 2>&1 & PIDS+=($!)
"$BIN_DIR/httpstub" -addr 127.0.0.1:29093 >"$LOG_DIR/stub-3.log" 2>&1 & PIDS+=($!)
sleep 0.5
cat >"$WORK_DIR/control-api.yaml" <<EOF
server:
listen_addr: "127.0.0.1:28080"
database:
path: "$WORK_DIR/control-api.db"
openstack:
mode: "mock"
orchestrator:
poll_interval_seconds: 1
self_check_timeout_seconds: 5
max_self_check_retries: 3
checking_window_seconds: 15
max_retries: 3
lease_ttl_seconds: 8
heartbeat_timeout_seconds: 10
aggregation:
missing_counts_as_fail: true
validators:
- validator_id: "validator_01"
os_port_id: "mock-port-1"
sites:
- site_id: "site-1"
index: 1
- site_id: "site-2"
index: 2
- site_id: "site-3"
index: 3
check_types:
- name: "https"
enabled: true
targets: ["stub-targets"]
- name: "icmp"
enabled: true
targets: ["stub-targets"]
targets:
stub-targets:
- "http://127.0.0.1:29091"
- "http://127.0.0.1:29092"
- "http://127.0.0.1:29093"
inbound_checks:
ports: [22022, 28081, 28443, 28888]
icmp: true
# Only 127.0.0.1 is usable here: the self-check mechanism compares the
# TCP source address the validator-agent's own outbound connection to
# control-api arrives with (see httpapi.remoteIP) against the assigned
# address. In a real deployment Neutron actually SNATs traffic through the
# newly attached FIP, so any configured address works; this offline
# harness has no real network-level SNAT, so only the literal loopback
# address the agent's traffic really originates from can ever pass.
ip_addresses:
- "127.0.0.1"
EOF
cat >"$WORK_DIR/validator-agent.yaml" <<EOF
validator_id: "validator_01"
control_api_url: "http://127.0.0.1:28080"
poll_interval_seconds: 1
self_check:
timeout_seconds: 5
# ip_echo by default; E2E_SELF_CHECK_METHODS="[control_api]" runs the same
# scenario through control-api's observed-ip route (loopback sees 127.0.0.1).
methods: ${E2E_SELF_CHECK_METHODS:-[ip_echo]}
# Points at the local httpstub's /ip echo route rather than a real
# internet IP-echo service — self-check only produces a meaningful
# signal here because the "floating IP" under test (127.0.0.1) is
# genuinely the address this process connects from; there's no real
# OpenStack SNAT involved in this offline harness. See docs/LOCAL_E2E.md.
ip_echo_urls: ["http://127.0.0.1:29091/ip"]
checks:
https_timeout_seconds: 5
icmp_timeout_seconds: 3
icmp_count: 2
ssh:
enabled: false
timeout_seconds: 3
EOF
for site in site-1 site-2 site-3; do
cat >"$WORK_DIR/prober-$site.yaml" <<EOF
site_id: "$site"
control_api_url: "http://127.0.0.1:28080"
poll_interval_seconds: 1
checks:
tcp_timeout_seconds: 3
icmp_timeout_seconds: 3
icmp_count: 2
EOF
done
echo "--- starting control-api ---"
"$BIN_DIR/control-api" -config "$WORK_DIR/control-api.yaml" >"$LOG_DIR/control-api.log" 2>&1 & CONTROL_PID=$!
PIDS+=("$CONTROL_PID")
echo -n "waiting for control-api /healthz "
for i in $(seq 1 30); do
if curl -fs "http://127.0.0.1:28080/healthz" >/dev/null 2>&1; then
echo "ok"
break
fi
echo -n "."
sleep 0.5
if [ "$i" -eq 30 ]; then
echo "FAILED"
cat "$LOG_DIR/control-api.log"
exit 1
fi
done
echo "--- authentication checks ---"
BASE="http://127.0.0.1:28080"
http_code() { curl -s -o /dev/null -w '%{http_code}' "$@"; }
expect_code() { # expect_code <want> <description> <curl args...>
local want="$1" desc="$2" got
shift 2
got="$(http_code "$@")"
if [ "$got" != "$want" ]; then
echo "FAIL: $desc: expected HTTP $want, got $got"
exit 1
fi
echo "ok: $desc -> $got"
}
expect_code 401 "admin endpoint without token" "$BASE/api/v1/admin/status"
expect_code 200 "admin endpoint with admin token" "${ADMIN_AUTH[@]}" "$BASE/api/v1/admin/status"
expect_code 401 "agent results without token" -X POST -H 'Content-Type: application/json' -d '{}' "$BASE/api/v1/agents/validator_01/results"
expect_code 401 "agent results with the ADMIN token" -X POST "${ADMIN_AUTH[@]}" -H 'Content-Type: application/json' -d '{}' "$BASE/api/v1/agents/validator_01/results"
# Open route: whatever the status (200/204/404 depending on state), it must not be 401.
assignment_code="$(http_code "$BASE/api/v1/agents/validator_01/assignment")"
if [ "$assignment_code" = "401" ]; then
echo "FAIL: GET /agents/validator_01/assignment must be open (no token), got 401"
exit 1
fi
echo "ok: assignment without token is not rejected -> $assignment_code"
start_validator_agent() {
"$BIN_DIR/validator-agent" -config "$WORK_DIR/validator-agent.yaml" -stub-ports "22022,28081,28443,28888" \
>>"$LOG_DIR/validator-agent.log" 2>&1 & VALIDATOR_PID=$!
PIDS+=("$VALIDATOR_PID")
echo "validator-agent started (pid $VALIDATOR_PID)"
}
status() { curl -fs "${ADMIN_AUTH[@]}" "http://127.0.0.1:28080/api/v1/admin/status"; }
ips() { curl -fs "${ADMIN_AUTH[@]}" "http://127.0.0.1:28080/api/v1/admin/ips"; }
ip_state() { ips | python3 -c "import json,sys; d=json.load(sys.stdin); print(d[0]['State'] if d else 'none')" 2>/dev/null || echo "?"; }
echo "--- starting validator-agent ---"
start_validator_agent
# Poll tightly (every 0.1s) for the IP to leave 'queued' — i.e. control-api
# has claimed it and associated the mock FIP — then kill the agent right
# there, before it has had a chance to self-check or run any checks. This
# proves the lease sweep reclaims genuinely in-flight, not-yet-reported
# work, rather than just racing against how fast the local checks happen
# to run.
echo "--- waiting for control-api to claim+associate the ip, to kill the agent mid-flight ---"
CAUGHT_MIDFLIGHT=0
for i in $(seq 1 50); do
st="$(ip_state)"
if [ "$st" != "queued" ] && [ "$st" != "none" ] && [ "$st" != "?" ]; then
CAUGHT_MIDFLIGHT=1
break
fi
sleep 0.1
done
if [ "$CAUGHT_MIDFLIGHT" -eq 1 ]; then
echo ">>> caught ip in state '$st'; killing validator-agent mid-run to demonstrate lease reclaim <<<"
kill "$VALIDATOR_PID" >/dev/null 2>&1 || true
wait "$VALIDATOR_PID" 2>/dev/null || true
sleep 9 # let the 8s lease_ttl_seconds expire and get reclaimed
echo ">>> restarting validator-agent <<<"
start_validator_agent
else
echo ">>> never observed the ip leave 'queued' in time; skipping the reclaim demonstration <<<"
fi
echo "--- starting 3 probers ---"
for site in site-1 site-2 site-3; do
"$BIN_DIR/prober" -config "$WORK_DIR/prober-$site.yaml" >"$LOG_DIR/prober-$site.log" 2>&1 & PIDS+=($!)
done
echo "--- waiting for the queue to drain (up to 60s) ---"
for i in $(seq 1 120); do
sleep 0.5
remaining=$(status | python3 -c "import json,sys; d=json.load(sys.stdin); s=d['ips_by_state']; print(sum(v for k,v in s.items() if k not in ('done','failed')))" 2>/dev/null || echo "?")
if [ "$remaining" = "0" ]; then
echo "queue drained after ~$((i / 2))s"
break
fi
done
echo "--- final status ---"
status | python3 -m json.tool
echo "--- final ip results ---"
ips | python3 -m json.tool
echo "--- demonstrating admin API: force a re-check of the already-done address ---"
curl -fs "${ADMIN_AUTH[@]}" -X POST "http://127.0.0.1:28080/api/v1/admin/ips" \
-H 'Content-Type: application/json' -d '{"addresses":["127.0.0.1"]}' | python3 -m json.tool
echo "--- waiting for the forced re-check to drain (up to 30s) ---"
for i in $(seq 1 60); do
sleep 0.5
remaining=$(status | python3 -c "import json,sys; d=json.load(sys.stdin); s=d['ips_by_state']; print(sum(v for k,v in s.items() if k not in ('done','failed')))" 2>/dev/null || echo "?")
if [ "$remaining" = "0" ]; then
echo "re-check drained after ~$((i / 2))s"
break
fi
done
echo "--- ip results after forced re-check (attempt_number should have advanced) ---"
ips | python3 -m json.tool
echo "--- automatic cycle: enable it and wait for the first cycle (clear queue -> scan FIPs -> checks -> registry) ---"
AC_URL="http://127.0.0.1:28080/api/v1/admin/auto-cycle"
registry_cycles() {
curl -fs "${ADMIN_AUTH[@]}" "http://127.0.0.1:28080/api/v1/admin/registry" | python3 -c "
import json,sys
for r in json.load(sys.stdin):
if r['ip_address'] == '127.0.0.1':
print(r['total_cycles'])
break
else:
print(0)"
}
ac_field() { curl -fs "${ADMIN_AUTH[@]}" "$AC_URL" | python3 -c "import json,sys; print(json.load(sys.stdin)['$1'])"; }
CYCLES_BEFORE="$(registry_cycles)"
# 60s is the smallest interval control-api accepts; the second cycle is
# covered by unit tests, so the script stops the auto-cycle after the first.
curl -fs "${ADMIN_AUTH[@]}" -X PUT "$AC_URL" -H 'Content-Type: application/json' \
-d '{"interval_seconds":60,"max_run_seconds":120}' >/dev/null
curl -fs "${ADMIN_AUTH[@]}" -X POST "$AC_URL/start" | python3 -m json.tool
echo "--- waiting for the first auto cycle to complete (up to 90s) ---"
AC_OUTCOME=""
for i in $(seq 1 180); do
sleep 0.5
AC_OUTCOME="$(ac_field last_outcome 2>/dev/null || true)"
if [ -n "$AC_OUTCOME" ]; then
echo "first auto cycle finished after ~$((i / 2))s: outcome=$AC_OUTCOME"
break
fi
done
echo "--- auto-cycle status ---"
curl -fs "${ADMIN_AUTH[@]}" "$AC_URL" | python3 -m json.tool
CYCLES_AFTER="$(registry_cycles)"
echo "registry total_cycles for 127.0.0.1: $CYCLES_BEFORE -> $CYCLES_AFTER"
if [ "$AC_OUTCOME" != "completed" ]; then
echo "FAIL: expected auto-cycle outcome 'completed', got '$AC_OUTCOME'"
exit 1
fi
if [ "$(ac_field phase)" != "waiting" ] || [ "$(ac_field runs_total)" != "1" ]; then
echo "FAIL: expected phase=waiting and runs_total=1 after the first cycle"
exit 1
fi
if [ "$CYCLES_AFTER" -le "$CYCLES_BEFORE" ]; then
echo "FAIL: the auto cycle did not add a new check cycle to the registry"
exit 1
fi
echo "--- automatic cycle: disable it; no further cycles must start ---"
curl -fs "${ADMIN_AUTH[@]}" -X POST "$AC_URL/stop" | python3 -m json.tool
if [ "$(ac_field enabled)" != "False" ] || [ "$(ac_field phase)" != "idle" ]; then
echo "FAIL: expected enabled=false and phase=idle after stop"
exit 1
fi
echo "auto-cycle e2e: OK"
echo "--- logs are in $WORK_DIR (kept for inspection; workdir NOT auto-deleted) ---"
trap - EXIT
cleanup