Session-recovery robustness (from PR #1)
- Suppress ConnectionError log noise from in-flight requests draining after a session close (poll_scheduler + keepalive log at DEBUG). - Null KeepaliveTask.on_unreachable during a forced reconnect so the dying session's keepalive can't flip HA availability offline after the new session is already healthy. Ported into PR #5's _maybe_force_reconnect flow. Drops the _session_stop / publish-health force-close mechanism from the original PR — PR #5's last_success_ts + _maybe_force_reconnect already covers the 'session dead, restart it' goal via a different path, and running both means two paths force-closing the same session on the same failure.
This commit is contained in:
@@ -15,6 +15,7 @@ from .oven import OVEN
|
||||
from .fridge import FRIDGE
|
||||
|
||||
|
||||
|
||||
DESCRIPTORS: dict[str, ApplianceDescriptor] = {
|
||||
DRYER.name: DRYER,
|
||||
OVEN.name: OVEN,
|
||||
|
||||
@@ -409,6 +409,11 @@ class PushBridge:
|
||||
self.log.warning(
|
||||
"unreachable for %.0fs — forcing session reconnect", elapsed)
|
||||
self._force_close_in_flight = True
|
||||
# Null the dying session's on_unreachable so its keepalive thread
|
||||
# can't flip availability offline after the new session takes over.
|
||||
ka = self.keepalive
|
||||
if ka is not None:
|
||||
ka.on_unreachable = None
|
||||
try:
|
||||
sess.close()
|
||||
except Exception as e:
|
||||
|
||||
@@ -67,6 +67,8 @@ class KeepaliveTask:
|
||||
try:
|
||||
self.session.ping()
|
||||
ok = True
|
||||
except ConnectionError as e:
|
||||
if self.log: self.log.debug("ping: %s", e)
|
||||
except Exception as e:
|
||||
if self.log: self.log.warning("ping: %s", e)
|
||||
# Real half-open detection: ping sends can succeed against a
|
||||
|
||||
@@ -237,6 +237,11 @@ class PollScheduler:
|
||||
self.log.warning("poll %s timeout (cooldown %.0fs)",
|
||||
href, cooldown)
|
||||
return
|
||||
except ConnectionError as e:
|
||||
self._poll_error_count += 1
|
||||
self._record_rtt((time.monotonic() - t0) * 1000.0)
|
||||
if self.log: self.log.debug("poll %s: %s", href, e)
|
||||
return
|
||||
except Exception as e:
|
||||
self._poll_error_count += 1
|
||||
self._record_rtt((time.monotonic() - t0) * 1000.0)
|
||||
@@ -274,6 +279,11 @@ class PollScheduler:
|
||||
self.log.warning("sweep %s timeout (cooldown %.0fs)",
|
||||
path, cooldown)
|
||||
return
|
||||
except ConnectionError as e:
|
||||
self._poll_error_count += 1
|
||||
self._record_rtt((time.monotonic() - t0) * 1000.0)
|
||||
if self.log: self.log.debug("sweep %s: %s", path, e)
|
||||
return
|
||||
except Exception as e:
|
||||
self._poll_error_count += 1
|
||||
self._record_rtt((time.monotonic() - t0) * 1000.0)
|
||||
@@ -300,4 +310,4 @@ class PollScheduler:
|
||||
if self.log:
|
||||
elapsed_ms = (time.monotonic() - t0) * 1000.0
|
||||
self.log.info("sweep complete (%d links, %.0fms)",
|
||||
len(indexed), elapsed_ms)
|
||||
len(indexed), elapsed_ms)
|
||||
Reference in New Issue
Block a user