From 96dc4df3160f55e13ee5c26515423c93bd8c6cbb Mon Sep 17 00:00:00 2001 From: Kristian Bendiksen Date: Mon, 4 May 2026 13:21:00 +0200 Subject: [PATCH] #9336 Python: Tolerate transient heartbeat failures The heartbeat introduced in db871296c closed the gRPC channel after a single missed ping (2s deadline). On Windows CI under load this fired during normal operations, leaving every subsequent test stuck on a dead channel. Track consecutive failures inside the heartbeat loop and only declare the connection lost after failure_threshold (default 3) consecutive failures. A successful ping resets the counter. Per-ping deadline raised from 2s to 5s, and below-threshold failures log a warning so transient slowness is still visible. --- GrpcInterface/Python/rips/instance.py | 29 +++++++++++++++++++++++++-- 1 file changed, 27 insertions(+), 2 deletions(-) diff --git a/GrpcInterface/Python/rips/instance.py b/GrpcInterface/Python/rips/instance.py index 2884eadf6d..bcf1bd5f08 100644 --- a/GrpcInterface/Python/rips/instance.py +++ b/GrpcInterface/Python/rips/instance.py @@ -386,12 +386,18 @@ class Instance: def start_heartbeat( self, interval_sec: float = 5.0, - deadline_sec: float = 2.0, + deadline_sec: float = 5.0, + failure_threshold: int = 3, on_failure: Optional[Callable[["RipsError"], None]] = None, ) -> None: """Start a background thread that periodically pings ResInsight. - On the first failed ping the instance is marked as + A single missed ping is not fatal: the heartbeat tolerates up + to ``failure_threshold - 1`` consecutive failures (e.g. + transient slowness on a loaded CI box) before declaring the + connection lost. A successful ping resets the counter. + + Once the threshold is reached, the instance is marked as ``connection_lost``, the underlying gRPC channel is closed so any in-flight calls fail fast with ``UNAVAILABLE``, and the retry interceptor is told to stop retrying. Subsequent API @@ -405,10 +411,16 @@ class Instance: interval_sec: Seconds between pings. deadline_sec: Per-ping deadline. Pings exceeding this are treated as failures. + failure_threshold: Number of consecutive failed pings + required before the connection is declared lost. + Must be >= 1. on_failure: Optional callback invoked once when the heartbeat detects a lost connection. Receives a :class:`RipsError`. """ + if failure_threshold < 1: + raise RipsError("failure_threshold must be >= 1") + if self._heartbeat_thread is not None and self._heartbeat_thread.is_alive(): return @@ -416,10 +428,23 @@ class Instance: self._heartbeat_stop = stop_event def _run() -> None: + consecutive_failures = 0 while not stop_event.is_set(): try: self.app.GetVersion(Empty(), timeout=deadline_sec) + consecutive_failures = 0 except grpc.RpcError as exc: + consecutive_failures += 1 + if consecutive_failures < failure_threshold: + logger.warning( + "Heartbeat ping failed (%d/%d): %s", + consecutive_failures, + failure_threshold, + exc, + ) + stop_event.wait(interval_sec) + continue + self._connection_lost = True # Close the channel so any pending RPC unblocks # immediately with UNAVAILABLE instead of waiting