keepAliveSweep method
Active keepalive + liveness for transient stream transports (ble:// /
tcp:// / serial:// carried on a streamableHttp config). Sends a cheap
ping round-trip on every such CONNECTED link (any reply counts, see the
probe below):
- the traffic keeps the link warm — an idle BLE session to an ESP32 drops in ~15s, but ~2-3s keepalive traffic stretches it to ~45s (measured), so far fewer reconnect cycles;
- a probe that fails/times out means the link died silently, so the
entry is marked error (via
_handleTransportDrop) and the health monitor reconnects it. HTTP/SSE/stdio servers are skipped — they don't suffer idle-drop and a periodic poll would just be noise.
Implementation
Future<void> keepAliveSweep({
Duration timeout = const Duration(seconds: 4),
Duration dropAfterSilence = _silenceToDrop,
}) async {
// One sweep at a time. The health monitor fires on a timer and does not
// wait for the previous tick, so a slow probe let sweeps pile up; when they
// expired together each dropped the link again — the same connection was
// logged as dropped three times in one second, over and over, on an ESP32
// node that was answering.
if (_sweeping) return;
_sweeping = true;
try {
final targets = _connections.entries
.where((e) =>
e.value.state == ConnectionState.connected &&
e.value.client != null &&
_isTransientStream(e.value.serverConfig))
.toList();
for (final e in targets) {
final serverId = e.key;
final client = e.value.client!;
// Anything the device sent is proof of life (23 §6.1.2). A link that
// is streaming does not need a probe, and a probe on a busy single-task
// device only queues behind the traffic that already answers the
// question. Measured 2026-09-21: an ESP32 pushing an update every
// second was dropped 23 times because its ping answers came late, and
// every drop stopped the stream on all seven devices sharing it.
final probeFrom = DateTime.now();
final last = client is SharedClient ? client.lastMessageAt : null;
if (last != null && probeFrom.difference(last) < timeout) {
continue;
}
final limit = _probeLimit(serverId, timeout);
final watch = Stopwatch()..start();
// The probe is `ping`, and any answer is life — a result, or a JSON-RPC
// error reply (it carries a code), which a device that never
// implemented ping still sends. Only a timeout or a failure with no
// reply behind it counts against the link.
//
// It used to be `resources/list`, every two seconds. On a small board
// that serialises its whole resource list per call and serves one
// request at a time, that was load the check itself added: measured
// under two clients, list answers drifted to 1.2–2.6 s and sometimes
// past the limit twice in a row, while the same board answered a ping
// it does not even implement in 0.2 s.
//
// A client may throw on the call itself rather than fail its future;
// both are the same attempt.
Future<void> probe;
try {
probe = client.ping().then<void>((_) {}, onError: (Object e) {
if (e is McpError && e.code != null) return; // it answered
throw e;
});
} catch (e) {
probe = Future<void>.error(e);
}
// Whether a missed probe was a lost answer or a late one decides the
// fix — a longer wait, or a look elsewhere — so keep watching the
// original request after giving up on it and say when, if ever, it came.
unawaited(probe.then((_) {
_recordProbe(serverId, watch.elapsed);
if (watch.elapsed > limit) {
// A late answer is still an answer: the device is alive, only slow.
// Counting the miss anyway dropped an ESP32 that had answered both
// of its "missed" probes — at 7.9 s and 4.4 s — one tick before the
// second timeout was judged.
_logger.info('Keepalive answer arrived late', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
});
} else if (watch.elapsedMilliseconds > 1000) {
_logger.info('Keepalive answer slow', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
});
}
}, onError: (Object _) {}));
try {
await probe.timeout(limit);
} catch (error) {
// Judge the client that was probed, not whatever holds the id now:
// by the time a probe times out the link may already have been
// replaced, and dropping the replacement starts the cycle again.
if (!identical(_connections[serverId]?.client, client)) {
_logger.info('Keepalive miss on a replaced client — ignored', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
});
continue;
}
// A timeout says "slow or dead" and cannot tell which; one slow
// answer is not a dead link. Measured on an ESP32 node under two
// clients' load: keepalive answers drifted to 1.2–2.6 s, one crossed
// the 4 s line, and dropping on that single miss cut a board that
// answered the next probe normally — failing the press in flight.
// A request that *errors* is different: the link refused it, so that
// still drops at once.
// Late is not dead: if anything arrived while the probe waited, the
// link is alive and only this answer is slow (23 §6.1.2).
final heard = client is SharedClient ? client.lastMessageAt : null;
if (error is TimeoutException &&
heard != null &&
heard.isAfter(probeFrom)) {
_logger.info('Keepalive answer late — link active', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
});
continue;
}
// A missed probe is not the verdict; silence is. The link is called
// dead only when nothing at all has come back for [dropAfterSilence].
// Counting misses against an RTT-scaled wait dropped a board whose
// answers were only held up by radio retransmission: the wait shrank
// to 4.2 s on a good minute, two lost frames cost 4.5 s, and the
// fourth such miss cut a link TCP would have delivered (2026-09-21).
// A new connection rides the same radio, so it cannot do better.
if (error is TimeoutException) {
final since = heard ?? e.value.connectedAt ?? probeFrom;
final silent = DateTime.now().difference(since);
if (silent < dropAfterSilence) {
_logger.info('Keepalive answer missed — keeping the link', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
'silentMs': silent.inMilliseconds,
});
continue;
}
}
// Say why. A drop with no reason left a day of measurement guessing
// between the board, the network and this side.
_logger.info('Keepalive probe failed', {
'serverId': serverId,
'afterMs': watch.elapsedMilliseconds,
'error': error.toString(),
});
_handleTransportDrop(serverId, DisconnectReason.transportError);
}
}
} finally {
_sweeping = false;
}
}