keepAliveSweep method

Future<void> keepAliveSweep({
  1. Duration timeout = const Duration(seconds: 4),
  2. Duration dropAfterSilence = _silenceToDrop,
})

Active keepalive + liveness for transient stream transports (ble:// / tcp:// / serial:// carried on a streamableHttp config). Sends a cheap ping round-trip on every such CONNECTED link (any reply counts, see the probe below):

  • the traffic keeps the link warm — an idle BLE session to an ESP32 drops in ~15s, but ~2-3s keepalive traffic stretches it to ~45s (measured), so far fewer reconnect cycles;
  • a probe that fails/times out means the link died silently, so the entry is marked error (via _handleTransportDrop) and the health monitor reconnects it. HTTP/SSE/stdio servers are skipped — they don't suffer idle-drop and a periodic poll would just be noise.

Implementation

Future<void> keepAliveSweep({
  Duration timeout = const Duration(seconds: 4),
  Duration dropAfterSilence = _silenceToDrop,
}) async {
  // One sweep at a time. The health monitor fires on a timer and does not
  // wait for the previous tick, so a slow probe let sweeps pile up; when they
  // expired together each dropped the link again — the same connection was
  // logged as dropped three times in one second, over and over, on an ESP32
  // node that was answering.
  if (_sweeping) return;
  _sweeping = true;
  try {
    final targets = _connections.entries
        .where((e) =>
            e.value.state == ConnectionState.connected &&
            e.value.client != null &&
            _isTransientStream(e.value.serverConfig))
        .toList();
    for (final e in targets) {
      final serverId = e.key;
      final client = e.value.client!;
      // Anything the device sent is proof of life (23 §6.1.2). A link that
      // is streaming does not need a probe, and a probe on a busy single-task
      // device only queues behind the traffic that already answers the
      // question. Measured 2026-09-21: an ESP32 pushing an update every
      // second was dropped 23 times because its ping answers came late, and
      // every drop stopped the stream on all seven devices sharing it.
      final probeFrom = DateTime.now();
      final last = client is SharedClient ? client.lastMessageAt : null;
      if (last != null && probeFrom.difference(last) < timeout) {
        continue;
      }
      final limit = _probeLimit(serverId, timeout);
      final watch = Stopwatch()..start();
      // The probe is `ping`, and any answer is life — a result, or a JSON-RPC
      // error reply (it carries a code), which a device that never
      // implemented ping still sends. Only a timeout or a failure with no
      // reply behind it counts against the link.
      //
      // It used to be `resources/list`, every two seconds. On a small board
      // that serialises its whole resource list per call and serves one
      // request at a time, that was load the check itself added: measured
      // under two clients, list answers drifted to 1.2–2.6 s and sometimes
      // past the limit twice in a row, while the same board answered a ping
      // it does not even implement in 0.2 s.
      //
      // A client may throw on the call itself rather than fail its future;
      // both are the same attempt.
      Future<void> probe;
      try {
        probe = client.ping().then<void>((_) {}, onError: (Object e) {
          if (e is McpError && e.code != null) return; // it answered
          throw e;
        });
      } catch (e) {
        probe = Future<void>.error(e);
      }
      // Whether a missed probe was a lost answer or a late one decides the
      // fix — a longer wait, or a look elsewhere — so keep watching the
      // original request after giving up on it and say when, if ever, it came.
      unawaited(probe.then((_) {
        _recordProbe(serverId, watch.elapsed);
        if (watch.elapsed > limit) {
          // A late answer is still an answer: the device is alive, only slow.
          // Counting the miss anyway dropped an ESP32 that had answered both
          // of its "missed" probes — at 7.9 s and 4.4 s — one tick before the
          // second timeout was judged.
          _logger.info('Keepalive answer arrived late', {
            'serverId': serverId,
            'afterMs': watch.elapsedMilliseconds,
          });
        } else if (watch.elapsedMilliseconds > 1000) {
          _logger.info('Keepalive answer slow', {
            'serverId': serverId,
            'afterMs': watch.elapsedMilliseconds,
          });
        }
      }, onError: (Object _) {}));
      try {
        await probe.timeout(limit);
      } catch (error) {
        // Judge the client that was probed, not whatever holds the id now:
        // by the time a probe times out the link may already have been
        // replaced, and dropping the replacement starts the cycle again.
        if (!identical(_connections[serverId]?.client, client)) {
          _logger.info('Keepalive miss on a replaced client — ignored', {
            'serverId': serverId,
            'afterMs': watch.elapsedMilliseconds,
          });
          continue;
        }
        // A timeout says "slow or dead" and cannot tell which; one slow
        // answer is not a dead link. Measured on an ESP32 node under two
        // clients' load: keepalive answers drifted to 1.2–2.6 s, one crossed
        // the 4 s line, and dropping on that single miss cut a board that
        // answered the next probe normally — failing the press in flight.
        // A request that *errors* is different: the link refused it, so that
        // still drops at once.
        // Late is not dead: if anything arrived while the probe waited, the
        // link is alive and only this answer is slow (23 §6.1.2).
        final heard = client is SharedClient ? client.lastMessageAt : null;
        if (error is TimeoutException &&
            heard != null &&
            heard.isAfter(probeFrom)) {
          _logger.info('Keepalive answer late — link active', {
            'serverId': serverId,
            'afterMs': watch.elapsedMilliseconds,
          });
          continue;
        }
        // A missed probe is not the verdict; silence is. The link is called
        // dead only when nothing at all has come back for [dropAfterSilence].
        // Counting misses against an RTT-scaled wait dropped a board whose
        // answers were only held up by radio retransmission: the wait shrank
        // to 4.2 s on a good minute, two lost frames cost 4.5 s, and the
        // fourth such miss cut a link TCP would have delivered (2026-09-21).
        // A new connection rides the same radio, so it cannot do better.
        if (error is TimeoutException) {
          final since = heard ?? e.value.connectedAt ?? probeFrom;
          final silent = DateTime.now().difference(since);
          if (silent < dropAfterSilence) {
            _logger.info('Keepalive answer missed — keeping the link', {
              'serverId': serverId,
              'afterMs': watch.elapsedMilliseconds,
              'silentMs': silent.inMilliseconds,
            });
            continue;
          }
        }
        // Say why. A drop with no reason left a day of measurement guessing
        // between the board, the network and this side.
        _logger.info('Keepalive probe failed', {
          'serverId': serverId,
          'afterMs': watch.elapsedMilliseconds,
          'error': error.toString(),
        });
        _handleTransportDrop(serverId, DisconnectReason.transportError);
      }
    }
  } finally {
    _sweeping = false;
  }
}