Add per-source status tracking, /health sources block and /sources
Collectors record the result of every source (schema v3, table source_status); /health reports failing sources (failures in a row >= source_failure_threshold) and turns degraded; GET /sources shows the full state; runs end with a summary log line instead of "data saved". Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
ab89c87321
commit
cb935fef0d
8 files changed
+351
-13
No files matched your search
+38
-1
@@ -210,6 +210,28 @@ def collector_state():
|
||||
return alive, updated_at, (status.get("jobs", {}) if status else {})
|
||||
|
||||
|
||||
def source_health():
|
||||
"""Сводка по источникам для /health: всего и только неисправные (сбоев подряд не меньше порога)."""
|
||||
try:
|
||||
config = load_full_config()
|
||||
except StorageError:
|
||||
return None, False
|
||||
threshold = cc.failure_threshold(config)
|
||||
with db.session() as conn:
|
||||
statuses = db.source_statuses(conn)
|
||||
summary, any_failing = {}, False
|
||||
for kind, key in (("asn", "asns"), ("fqdn", "fqdns")):
|
||||
configured = [str(source) for source in config.get(key, [])]
|
||||
failing = [{"source": source, "failures": statuses[kind, source]["failures"],
|
||||
"last_success": statuses[kind, source]["last_success"],
|
||||
"error_kind": statuses[kind, source]["error_kind"]}
|
||||
for source in configured
|
||||
if (kind, source) in statuses and statuses[kind, source]["failures"] >= threshold]
|
||||
summary[kind] = {"total": len(configured), "failing": failing}
|
||||
any_failing = any_failing or bool(failing)
|
||||
return summary, any_failing
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
def health():
|
||||
"""Состояние сборщика: демон пишет status.json (задания + heartbeat), API только читает."""
|
||||
@@ -232,13 +254,15 @@ def health():
|
||||
except StorageError:
|
||||
recreated = None
|
||||
|
||||
healthy = alive and not pending and not any(j.get("last_error") for j in jobs.values())
|
||||
sources, any_failing = source_health()
|
||||
healthy = alive and not pending and not any_failing and not any(j.get("last_error") for j in jobs.values())
|
||||
return {
|
||||
"status": "ok" if healthy else "degraded",
|
||||
"collector_alive": alive,
|
||||
"collector_updated_at": updated_at,
|
||||
"jobs": jobs,
|
||||
"counts": counts,
|
||||
"sources": sources,
|
||||
"last_restore": last_restore,
|
||||
# Только время и признак ожидания (пути карантина наружу не отдаём)
|
||||
"db_recreated": {"at": recreated.get("at"), "pending": pending} if recreated else None,
|
||||
@@ -275,6 +299,19 @@ def _purge(kind, source):
|
||||
return db.purge_source(conn, kind, source)
|
||||
|
||||
|
||||
@app.get("/sources")
|
||||
def list_sources():
|
||||
"""Состояние всех настроенных источников: сколько адресов дал, когда опрашивался, сколько сбоев подряд."""
|
||||
config = load_full_config()
|
||||
with db.session() as conn:
|
||||
statuses, counts = db.source_statuses(conn), db.address_counts(conn)
|
||||
empty = {"last_attempt": None, "last_success": None, "failures": 0, "error_kind": None}
|
||||
return {"failure_threshold": cc.failure_threshold(config), "sources": [
|
||||
{"kind": kind, "source": str(source), "addresses": counts.get((kind, str(source)), 0),
|
||||
**statuses.get((kind, str(source)), empty)}
|
||||
for kind, key in (("asn", "asns"), ("fqdn", "fqdns")) for source in config.get(key, [])]}
|
||||
|
||||
|
||||
@app.get("/asns")
|
||||
def list_asns():
|
||||
return {"asns": load_full_config().get("asns", [])}
|
||||
|
||||
Reference in new issue
Block a user