Session-recovery robustness (from PR #1)

- Suppress ConnectionError log noise from in-flight requests draining
  after a session close (poll_scheduler + keepalive log at DEBUG).
- Null KeepaliveTask.on_unreachable during a forced reconnect so the
  dying session's keepalive can't flip HA availability offline after
  the new session is already healthy. Ported into PR #5's
  _maybe_force_reconnect flow.

Drops the _session_stop / publish-health force-close mechanism from
the original PR — PR #5's last_success_ts + _maybe_force_reconnect
already covers the 'session dead, restart it' goal via a different
path, and running both means two paths force-closing the same session
on the same failure.
This commit is contained in:
Aminorjourney
2026-07-02 21:15:00 +01:00
committed by Jack Nagy
parent 93f78f99a2
commit 015e4a4e9e
4 changed files with 19 additions and 1 deletions
+1
View File
@@ -15,6 +15,7 @@ from .oven import OVEN
from .fridge import FRIDGE
DESCRIPTORS: dict[str, ApplianceDescriptor] = {
DRYER.name: DRYER,
OVEN.name: OVEN,
+5
View File
@@ -409,6 +409,11 @@ class PushBridge:
self.log.warning(
"unreachable for %.0fs — forcing session reconnect", elapsed)
self._force_close_in_flight = True
# Null the dying session's on_unreachable so its keepalive thread
# can't flip availability offline after the new session takes over.
ka = self.keepalive
if ka is not None:
ka.on_unreachable = None
try:
sess.close()
except Exception as e:
+2
View File
@@ -67,6 +67,8 @@ class KeepaliveTask:
try:
self.session.ping()
ok = True
except ConnectionError as e:
if self.log: self.log.debug("ping: %s", e)
except Exception as e:
if self.log: self.log.warning("ping: %s", e)
# Real half-open detection: ping sends can succeed against a
+11 -1
View File
@@ -237,6 +237,11 @@ class PollScheduler:
self.log.warning("poll %s timeout (cooldown %.0fs)",
href, cooldown)
return
except ConnectionError as e:
self._poll_error_count += 1
self._record_rtt((time.monotonic() - t0) * 1000.0)
if self.log: self.log.debug("poll %s: %s", href, e)
return
except Exception as e:
self._poll_error_count += 1
self._record_rtt((time.monotonic() - t0) * 1000.0)
@@ -274,6 +279,11 @@ class PollScheduler:
self.log.warning("sweep %s timeout (cooldown %.0fs)",
path, cooldown)
return
except ConnectionError as e:
self._poll_error_count += 1
self._record_rtt((time.monotonic() - t0) * 1000.0)
if self.log: self.log.debug("sweep %s: %s", path, e)
return
except Exception as e:
self._poll_error_count += 1
self._record_rtt((time.monotonic() - t0) * 1000.0)
@@ -300,4 +310,4 @@ class PollScheduler:
if self.log:
elapsed_ms = (time.monotonic() - t0) * 1000.0
self.log.info("sweep complete (%d links, %.0fms)",
len(indexed), elapsed_ms)
len(indexed), elapsed_ms)