Add per-source status tracking, /health sources block and /sources

Collectors record the result of every source (schema v3, table
source_status); /health reports failing sources (failures in a row >=
source_failure_threshold) and turns degraded; GET /sources shows the full
state; runs end with a summary log line instead of "data saved".

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
ayurishchevandClaude Sonnet 5 committed 2026-09-21 10:46:38 +03:00
1 parent ab89c87321
commit cb935fef0d
8 files changed
+351 -13

No files matched your search

+38 -1
View File
@@ -210,6 +210,28 @@ def collector_state():
return alive, updated_at, (status.get("jobs", {}) if status else {})
def source_health():
"""Сводка по источникам для /health: всего и только неисправные (сбоев подряд не меньше порога)."""
try:
config = load_full_config()
except StorageError:
return None, False
threshold = cc.failure_threshold(config)
with db.session() as conn:
statuses = db.source_statuses(conn)
summary, any_failing = {}, False
for kind, key in (("asn", "asns"), ("fqdn", "fqdns")):
configured = [str(source) for source in config.get(key, [])]
failing = [{"source": source, "failures": statuses[kind, source]["failures"],
"last_success": statuses[kind, source]["last_success"],
"error_kind": statuses[kind, source]["error_kind"]}
for source in configured
if (kind, source) in statuses and statuses[kind, source]["failures"] >= threshold]
summary[kind] = {"total": len(configured), "failing": failing}
any_failing = any_failing or bool(failing)
return summary, any_failing
@app.get("/health")
def health():
"""Состояние сборщика: демон пишет status.json (задания + heartbeat), API только читает."""
@@ -232,13 +254,15 @@ def health():
except StorageError:
recreated = None
healthy = alive and not pending and not any(j.get("last_error") for j in jobs.values())
sources, any_failing = source_health()
healthy = alive and not pending and not any_failing and not any(j.get("last_error") for j in jobs.values())
return {
"status": "ok" if healthy else "degraded",
"collector_alive": alive,
"collector_updated_at": updated_at,
"jobs": jobs,
"counts": counts,
"sources": sources,
"last_restore": last_restore,
# Только время и признак ожидания (пути карантина наружу не отдаём)
"db_recreated": {"at": recreated.get("at"), "pending": pending} if recreated else None,
@@ -275,6 +299,19 @@ def _purge(kind, source):
return db.purge_source(conn, kind, source)
@app.get("/sources")
def list_sources():
"""Состояние всех настроенных источников: сколько адресов дал, когда опрашивался, сколько сбоев подряд."""
config = load_full_config()
with db.session() as conn:
statuses, counts = db.source_statuses(conn), db.address_counts(conn)
empty = {"last_attempt": None, "last_success": None, "failures": 0, "error_kind": None}
return {"failure_threshold": cc.failure_threshold(config), "sources": [
{"kind": kind, "source": str(source), "addresses": counts.get((kind, str(source)), 0),
**statuses.get((kind, str(source)), empty)}
for kind, key in (("asn", "asns"), ("fqdn", "fqdns")) for source in config.get(key, [])]}
@app.get("/asns")
def list_asns():
return {"asns": load_full_config().get("asns", [])}