From df08735dbd78296c668f16d490504b2ca3d72abf Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Apr 2026 09:55:42 +0000 Subject: [PATCH 1/3] fix(db): prevent process exit on transient MongoDB outages Two related changes to make startup resilient to a brief DB outage: 1. database/manager.py: reduce initial background-reconnect delay from 30s to 5s. Previously, inside the 120s startup wait only 2 reconnect attempts fit (t=30, t=75). With 5s first delay we get several early attempts where recovery is most likely. 2. main.py: after the 120s window, don't SystemExit(1). Enter a passive wait loop (poll every DB_RECONNECT_POLL_INTERVAL, default 30s) until the background task reconnects. Avoids CrashLoop + Sentry storms on transient outages (e.g. the fatal events on 2026-04-03). https://claude.ai/code/session_01Dgaz4FwhYhLtERMozH2UZF --- database/manager.py | 2 +- main.py | 17 ++++++++++++++--- 2 files changed, 15 insertions(+), 4 deletions(-) diff --git a/database/manager.py b/database/manager.py index 5706800cc..ae2afc232 100644 --- a/database/manager.py +++ b/database/manager.py @@ -1243,7 +1243,7 @@ def _schedule_background_reconnect( kwargs: Dict[str, Any], mongo_url: Optional[str], database_name: Optional[str], - delay: float = 30.0, + delay: float = 5.0, max_bg_attempts: int = 10, _attempt: int = 1, ) -> None: diff --git a/main.py b/main.py index 7df7eb286..1216a454d 100644 --- a/main.py +++ b/main.py @@ -4917,9 +4917,20 @@ def main() -> None: time.sleep(5) _db_waited += 5 if not getattr(db, 'is_connected', True): - logger.critical("DB still unreachable after %ds; exiting to prevent unsafe operation.", _db_wait_max) - raise SystemExit(1) - logger.info("DB reconnected after %ds wait; proceeding with lock acquisition.", _db_waited) + # Passive wait instead of SystemExit: keep the process alive and let the + # background reconnect keep trying. Avoids CrashLoop + Sentry storms when + # the DB has a transient outage. + _poll_interval = int(os.getenv("DB_RECONNECT_POLL_INTERVAL", "30")) + logger.warning( + "DB still unreachable after %ds; entering passive wait (polling every %ds).", + _db_wait_max, _poll_interval, + ) + while not getattr(db, 'is_connected', True): + time.sleep(_poll_interval) + _db_waited += _poll_interval + logger.info("DB reconnected after %ds total wait; proceeding with lock acquisition.", _db_waited) + else: + logger.info("DB reconnected after %ds wait; proceeding with lock acquisition.", _db_waited) # MongoDB connection and lock management if not manage_mongo_lock(): From 158c7324c7da535828723abdf38a1e4bfe1e0bde Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Apr 2026 10:06:57 +0000 Subject: [PATCH 2/3] fix(db): keep background reconnect running forever with capped backoff MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to df08735. Cursor-bot flagged that main.py's new passive wait loop could hang forever: after max_bg_attempts (10) in _schedule_background_reconnect, no more attempts were scheduled, so is_connected would stay False and the main loop would busy-wait doing nothing useful. Fix: remove the hard cap on background reconnect attempts. Exponential backoff is already capped at 300s, so worst-case load is one attempt per 5 minutes — cheap to keep trying. We emit a one-time "escalating" event when the historical max is crossed, so observability is preserved. https://claude.ai/code/session_01Dgaz4FwhYhLtERMozH2UZF --- database/manager.py | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/database/manager.py b/database/manager.py index ae2afc232..a902c7bcc 100644 --- a/database/manager.py +++ b/database/manager.py @@ -1293,13 +1293,24 @@ def _try_reconnect(): client.close() except Exception: pass - if _attempt < max_bg_attempts: - next_delay = min(delay * 1.5, 300.0) - self._schedule_background_reconnect( - kwargs, mongo_url, database_name, - delay=next_delay, max_bg_attempts=max_bg_attempts, - _attempt=_attempt + 1, + # Never give up: main.py now does a passive wait instead of SystemExit, + # so if we stop scheduling here the app would spin forever without any + # active reconnect attempt. Exponential backoff is capped at 300s, so + # load stays bounded. We emit a one-time escalation event when the + # historical max is crossed, for observability. + if _attempt == max_bg_attempts: + emit_event( + "db_background_reconnect_escalating", + severity="error", + attempt=_attempt, + note="continuing with capped backoff", ) + next_delay = min(delay * 1.5, 300.0) + self._schedule_background_reconnect( + kwargs, mongo_url, database_name, + delay=next_delay, max_bg_attempts=max_bg_attempts, + _attempt=_attempt + 1, + ) timer = threading.Timer(delay, _try_reconnect) timer.daemon = True From 3ef0a9d6d01e0241861e00cc95d5b1ecdb67eafb Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Apr 2026 10:16:19 +0000 Subject: [PATCH 3/3] docs(config): register DB_RECONNECT_* env vars in inspector and docs Cursor-bot flagged that the new DB_RECONNECT_POLL_INTERVAL (introduced in df08735) wasn't registered in services/config_inspector_service.py or docs/environment-variables.rst. Also adds the previously-undocumented DB_RECONNECT_WAIT_BEFORE_POLL so both tunables are visible to operators using the config inspector. https://claude.ai/code/session_01Dgaz4FwhYhLtERMozH2UZF --- docs/environment-variables.rst | 12 ++++++++++++ services/config_inspector_service.py | 12 ++++++++++++ 2 files changed, 24 insertions(+) diff --git a/docs/environment-variables.rst b/docs/environment-variables.rst index 93dd71c1e..b8f70ca8c 100644 --- a/docs/environment-variables.rst +++ b/docs/environment-variables.rst @@ -406,6 +406,18 @@ - ``1000`` - ``1500`` - Bot/WebApp + * - ``DB_RECONNECT_WAIT_BEFORE_POLL`` + - זמן המתנה ראשוני (שניות) בעלייה של הבוט עד שה-DB חוזר, לפני שעוברים ל-poll פסיבי. הבוט לא יוצא (SystemExit) אחרי החלון — הוא ממשיך להמתין. + - לא + - ``120`` + - ``120`` + - Bot + * - ``DB_RECONNECT_POLL_INTERVAL`` + - מרווח (שניות) בין בדיקות חיבור ב-poll פסיבי לאחר שחלון ההמתנה הראשוני פג. ה-reconnect ברקע ממשיך לנסות עם backoff מעריכי (מקסימום 300 שניות). + - לא + - ``30`` + - ``30`` + - Bot * - ``DB_HEALTH_POOL_REFRESH_SEC`` - תדירות רענון מומלצת (שניות) לסטטוס ה-pool בדשבורד. (משתנה תיעודי/קונפיגורציה כללית) - לא diff --git a/services/config_inspector_service.py b/services/config_inspector_service.py index b066e5dcc..354b7a6a6 100644 --- a/services/config_inspector_service.py +++ b/services/config_inspector_service.py @@ -189,6 +189,18 @@ class ConfigService: category="database", sensitive=True, ), + "DB_RECONNECT_WAIT_BEFORE_POLL": ConfigDefinition( + key="DB_RECONNECT_WAIT_BEFORE_POLL", + default="120", + description="זמן המתנה ראשוני (שניות) להתחברות מחדש ל-DB בעלייה לפני מעבר ל-poll פסיבי", + category="database", + ), + "DB_RECONNECT_POLL_INTERVAL": ConfigDefinition( + key="DB_RECONNECT_POLL_INTERVAL", + default="30", + description="מרווח (שניות) בין בדיקות חיבור ב-poll פסיבי לאחר שחלון ההמתנה הראשוני פג", + category="database", + ), "DB_HEALTH_SLOW_THRESHOLD_MS": ConfigDefinition( key="DB_HEALTH_SLOW_THRESHOLD_MS", default="1000",