Skip to content
Merged
Show file tree
Hide file tree
Changes from 3 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
136 changes: 124 additions & 12 deletions src/ble_reticulum/linux_bluetooth_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -387,6 +387,13 @@ def __init__(

# Scanner health tracking
self.consecutive_empty_scans = 0
# Whether the adapter has ever demonstrated a working scan (seen any
# device, or received any Reticulum callback). Used to distinguish a
# genuinely wedged adapter (was healthy, now blind) from a quiet RF
# environment where no BLE devices are present at all (never seen
# anything, not necessarily broken). Only the former warrants a
# "system reboot required" critical.
self.healthy_ever = False

# Apply BlueZ timing patch
apply_bluez_services_resolved_patch()
Expand Down Expand Up @@ -680,22 +687,100 @@ def detection_callback(device, advertisement_data):
raise

# Detect scanner callback corruption
#
# A service-filtered scan seeing nothing is NOT by itself evidence the
# adapter is wedged: at boot (and whenever no Reticulum peer is in
# range) it is normal for the filtered scan to return zero callbacks,
# because the peer's GATT server takes a few seconds to start
# advertising after ours. Declaring that a "corrupted / reboot
# required" stack is a false positive that fires on_error(critical) and
# tears the interface down right in the window the peer is coming up.
#
# Disambiguate with a short UNFILTERED health scan: if the adapter can
# see ANY device at all, it is alive and we are simply waiting for a
# Reticulum peer to advertise. Only a fully blind adapter (unfiltered
# scan also empty) counts as a genuine wedge.
if callback_count[0] == 0:
self.consecutive_empty_scans += 1
self._log(f"⚠️ Scanner corruption detected: 0 callbacks after {scan_time}s scan (streak: {self.consecutive_empty_scans})", "WARNING")

if self.consecutive_empty_scans >= 3:
self._log("⚠️ CRITICAL: Bleak scanner callbacks not firing", "ERROR")
self._log("⚠️ Bluetooth/BlueZ/D-Bus state is corrupted", "ERROR")
self._log("⚠️ System reboot required to restore BLE scanning", "ERROR")
# Re-check connection state before starting a SECOND scan. The
# pause check at the top of _perform_scan ran before the main scan;
# a connection could have started during the main scan window. If so,
# running the unfiltered health scan now would start another BlueZ
# scan mid-connection and trigger an "Operation already in progress"
# error. Skip the health scan this cycle (it will run next cycle);
# treat the filtered result as-is without changing the streak.
if self._should_pause_scanning():
self._log("Pausing health scan: connection(s) in progress", "DEBUG")
return
health_devices = await self._adapter_health_check()
Comment thread
greptile-apps[bot] marked this conversation as resolved.
if health_devices > 0:
self.healthy_ever = True
if self.consecutive_empty_scans > 0:
self._log(f"✓ Scanner callbacks resumed after {self.consecutive_empty_scans} empty scans", "INFO")
self.consecutive_empty_scans = 0
self._log(
f"✓ No Reticulum peer advertising, but adapter is healthy "
f"(unfiltered health scan saw {health_devices} device(s)) - not a wedge",
"INFO"
)
elif health_devices < 0:
# The health scan itself FAILED (D-Bus/adapter error). A scan
# that cannot even run is direct evidence the adapter is in a
# bad state - unlike a clean zero, which in a quiet RF
# environment may be benign. Escalate regardless of whether we
# ever saw a healthy scan.
self.consecutive_empty_scans += 1
self._log(
f"⚠️ Unfiltered health scan FAILED (adapter in fault state) "
f"(streak: {self.consecutive_empty_scans})", "WARNING"
)
if self.consecutive_empty_scans >= 3:
self._log("⚠️ CRITICAL: unfiltered health scan repeatedly failing (adapter wedged)", "ERROR")
self._log("⚠️ Bluetooth/BlueZ/D-Bus state is corrupted", "ERROR")
self._log("⚠️ System reboot required to restore BLE scanning", "ERROR")
if self.on_error:
self.on_error("critical",
f"Adapter health scan has failed for {self.consecutive_empty_scans} "
f"consecutive scans (adapter in fault state). Bluetooth stack requires reboot.",
Exception("BleakScanner health scan failed"))
else:
# Clean zero: unfiltered scan ran but saw no devices at all.
self.consecutive_empty_scans += 1
Comment thread
greptile-apps[bot] marked this conversation as resolved.
self._log(
f"⚠️ Scanner blind: 0 Reticulum callbacks AND 0 devices in unfiltered "
f"health scan (streak: {self.consecutive_empty_scans})", "WARNING"
)

if self.on_error:
self.on_error("critical",
f"Scanner callback failure detected (0 callbacks for {self.consecutive_empty_scans} consecutive scans). "
"Bluetooth stack requires reboot.",
Exception("BleakScanner callbacks not invoked"))
# Only escalate to a "reboot required" critical if the adapter
# was previously proven healthy and has now gone blind. An
# empty unfiltered scan in a genuinely quiet RF environment
# (no BLE devices in range at all) is NOT proof the adapter is
# broken - in that case both scans legitimately return nothing,
# and mandating a system reboot would be a false alarm. A truly
# wedged adapter is one that USED to see devices and no longer
# can, so gate the critical on healthy_ever.
if self.consecutive_empty_scans >= 3 and self.healthy_ever:

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Startup scan fault stays hidden

If an adapter is already blind when the driver starts but both scans finish without errors, healthy_ever stays false. Even after many empty scans, the driver only warns and never raises the critical fault signal. A real startup fault that returns empty results can leave discovery dead without that signal. The driver needs a way to distinguish this case from a quiet room.

Knowledge Base Used: Linux adapter recovery and error handling

Prompt To Fix With AI
This is a comment left during a code review.
Path: src/ble_reticulum/linux_bluetooth_driver.py
Line: 761

Comment:
**Startup scan fault stays hidden**

If an adapter is already blind when the driver starts but both scans finish without errors, `healthy_ever` stays false. Even after many empty scans, the driver only warns and never raises the critical fault signal. A real startup fault that returns empty results can leave discovery dead without that signal. The driver needs a way to distinguish this case from a quiet room.

**Knowledge Base Used:** [Linux adapter recovery and error handling](https://app.greptile.com/torlando-tech/-/custom-context/knowledge-base/torlando-tech/ble-reticulum/-/docs/linux-adapter-recovery.md)

---

For each issue above, determine whether it is valid and should be fixed. If so, fix it directly.

Fix in Claude Code

self._log("⚠️ CRITICAL: adapter was healthy but is now blind to all devices (likely wedged)", "ERROR")
self._log("⚠️ Bluetooth/BlueZ/D-Bus state is corrupted", "ERROR")
self._log("⚠️ System reboot required to restore BLE scanning", "ERROR")

if self.on_error:
self.on_error("critical",
f"Scanner callback failure detected (adapter was healthy but is blind to all devices for "
f"{self.consecutive_empty_scans} consecutive scans). Bluetooth stack requires reboot.",
Exception("BleakScanner callbacks not invoked"))
elif self.consecutive_empty_scans >= 3:
# Never demonstrated a working scan: this may simply be a
# quiet RF environment rather than a fault. Warn without
# mandating a reboot so we do not false-alarm.
self._log(
f"⚠️ Adapter has never seen any device in {self.consecutive_empty_scans} "
f"consecutive scans. This may be a quiet RF environment rather than a "
f"fault; not escalating to a reboot-required critical.",
"WARNING"
)
else:
# Reset counter on successful callback
self.healthy_ever = True
if self.consecutive_empty_scans > 0:
self._log(f"✓ Scanner callbacks resumed after {self.consecutive_empty_scans} empty scans", "INFO")
self.consecutive_empty_scans = 0
Expand Down Expand Up @@ -736,6 +821,33 @@ def detection_callback(device, advertisement_data):
else:
self._log(f"✗ {device.address} ({device.name or 'Unknown'}): service UUID mismatch (has {adv_data.service_uuids}, want {self.service_uuid})", "EXTRA")

async def _adapter_health_check(self) -> int:
"""Run a short UNFILTERED scan to distinguish "no Reticulum peer in
range" (adapter healthy) from "adapter blind" (genuine wedge).

The main discovery scan is service-filtered, so it returns zero
callbacks whenever no peer is advertising our Reticulum service - which
is the normal state at boot before the peer's GATT server starts
advertising. That is not evidence of a wedged adapter. To disambiguate,
we run a brief one-shot scan with NO service filter: if the adapter can
see any BLE device at all (including weak ones we would filter out by
RSSI), it is alive and we are simply waiting for a peer.

Returns the number of distinct devices seen. A negative value (-1)
indicates the health scan itself failed (D-Bus/adapter error) rather
than returning zero devices - the caller treats a failed scan as direct
evidence of an adapter fault (unlike a clean zero, which in a quiet RF
environment may be benign).
"""
try:
devices = await BleakScanner.discover(timeout=1.0)
count = len(devices)
self._log(f"🩺 Adapter health scan: {count} device(s) visible (unfiltered)", "DEBUG")
return count
except Exception as e:
self._log(f"⚠️ Adapter health scan failed: {e}", "WARNING")
return -1

# ========================================================================
# Advertising (Peripheral Mode)
# ========================================================================
Expand Down
Loading
Loading