Stage B of review fixes: daemon state lock, image build check, RIPEstat retries
Job state is changed under one RLock, the Dockerfile copies all root modules and imports them at build time, RIPEstat requests go through a retrying session with the sourceapp parameter (ripestat_sourceapp) and a capped Retry-After. Adds the summary and marks review findings 5-10 fixed. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
d6b69d0842
commit
aaf41adc79
8 files changed
+164
-28
No files matched your search
+19
-11
@@ -41,7 +41,8 @@ LOCK_FILE = os.path.join(cc.DATA_DIR, "collector.daemon.lock")
|
||||
job_state = {name: {"cron": None, "last_run": None, "last_finished": None, "running": False,
|
||||
"last_error": None, "rejected_cron": None}
|
||||
for name in COLLECTORS}
|
||||
_state_lock = threading.Lock()
|
||||
# RLock: изменения состояния и write_status (читает под той же блокировкой) не должны мешать друг другу
|
||||
_state_lock = threading.RLock()
|
||||
# Один запуск за раз на тип: наложение планового и ручного сбора пропускается
|
||||
_run_locks = {name: threading.Lock() for name in COLLECTORS}
|
||||
|
||||
@@ -68,18 +69,22 @@ def run_job(name, scheduler):
|
||||
try:
|
||||
logger.info("Running %s collection...", name)
|
||||
state = job_state[name]
|
||||
state["last_run"] = datetime.datetime.now().isoformat()
|
||||
state["running"] = True
|
||||
with _state_lock:
|
||||
error = state["last_error"] # при прерывании (SystemExit) прежняя ошибка остаётся
|
||||
state["last_run"] = datetime.datetime.now().isoformat()
|
||||
state["running"] = True
|
||||
write_status(scheduler)
|
||||
try:
|
||||
COLLECTORS[name]()
|
||||
state["last_error"] = None
|
||||
error = None
|
||||
except Exception as e:
|
||||
logger.exception("%s collection failed", name)
|
||||
state["last_error"] = str(e)
|
||||
error = str(e)
|
||||
finally:
|
||||
state["running"] = False
|
||||
state["last_finished"] = datetime.datetime.now().isoformat()
|
||||
with _state_lock:
|
||||
state["last_error"] = error
|
||||
state["running"] = False
|
||||
state["last_finished"] = datetime.datetime.now().isoformat()
|
||||
write_status(scheduler)
|
||||
finally:
|
||||
lock.release()
|
||||
@@ -104,7 +109,8 @@ def schedule_jobs(scheduler):
|
||||
cron = schedule.get(name, DEFAULT_CRONS[name])
|
||||
scheduler.add_job(run_job, CronTrigger.from_crontab(cron), args=[name, scheduler],
|
||||
id=f"{name}_job", replace_existing=True, max_instances=1, coalesce=True)
|
||||
job_state[name]["cron"] = cron
|
||||
with _state_lock:
|
||||
job_state[name]["cron"] = cron
|
||||
logger.info("Job %s scheduled: %s", name, cron)
|
||||
|
||||
|
||||
@@ -125,12 +131,14 @@ def sync_schedule(scheduler):
|
||||
trigger = CronTrigger.from_crontab(cron)
|
||||
except ValueError as e:
|
||||
logger.error("Invalid cron for %s (%r), keeping %r: %s", name, cron, state["cron"], e)
|
||||
state["rejected_cron"] = cron
|
||||
with _state_lock:
|
||||
state["rejected_cron"] = cron
|
||||
continue
|
||||
scheduler.reschedule_job(f"{name}_job", trigger=trigger)
|
||||
logger.info("Job %s rescheduled: %s -> %s", name, state["cron"], cron)
|
||||
state["cron"] = cron
|
||||
state["rejected_cron"] = None
|
||||
with _state_lock:
|
||||
state["cron"] = cron
|
||||
state["rejected_cron"] = None
|
||||
|
||||
write_status(scheduler)
|
||||
|
||||
|
||||
Reference in new issue
Block a user