mirror of
https://github.com/meshcore-dev/meshcore_py.git
synced 2026-08-01 20:56:12 +00:00
fix(ble): bound the write, and correct two reply-path edge cases
Follow-up to review of the two preceding commits.
The write lock added in 00135bb prevented concurrent writes from dropping the
link, but bounded nothing. CommandHandler's own timeout does not cover the
write: send() awaits _sender_func() and only afterwards arms
asyncio.wait(futures, timeout=...). So a stalled write -- observed on hardware
running to minutes -- held the lock indefinitely while every other command
queued behind it, with nothing logged, no error raised and no DISCONNECTED
event. Before the lock only the stalled command hung; the others went out and
could trip the error-19 disconnect, which at least recovered. The lock turned
a bounded, self-healing failure into an unbounded silent one, and also blocked
the post-reconnect CMD_APP_START behind the dead connection's holder.
The write (lock acquisition included) is now bounded by
BLEConnection.WRITE_TIMEOUT. On expiry the link is torn down rather than the
lock merely released: the underlying CoreBluetooth write may still be in
flight, and a second write racing it re-creates the overlap the lock exists to
prevent. Tearing down hands over to the reconnect path, which is bounded and
self-healing.
The lock is now a lazily-created property, mirroring _mesh_request_lock in
commands/base.py, so an instance built without __init__ still works. The two
BLE tests previously assigned _write_lock themselves, which meant deleting the
__init__ line left them green; there is now a test that __init__ provides it.
Also in send_anon_req:
- A failed change_contact_path() no longer proceeds. The device still has the
contact as flood, so sendAnonReq() floods the request, and the server gates
REGIONS/OWNER/BASIC behind isRouteDirect() and drops it -- the caller then
waits out a full path-scaled timeout for a reply that cannot arrive. It now
returns ERROR path_reset_failed.
- encode_reply_path() clamps to the server's 64-byte reply_path buffer, which
MyMesh.cpp memcpys into with no length check, and rejects hash mode 3 (the
4-byte hops Packet::isValidPathLen refuses). Not a regression -- the previous
encoder overflowed identically -- but this function is the chokepoint and its
comment claimed to bound the read.
- The hop count saturates at 63 instead of being masked with & 63, which would
wrap a 64-hop path to zero hops, i.e. request a zero-hop reply from a distant
node. Not reachable from a device-sourced contact (the reader caps the field
at 63) but silent if it ever were.
- The suggested_timeout multiplier now scales by the hops actually emitted
rather than the contact's claimed out_path_len, which can differ once the
encoder clamps or truncates.
Correction to 30446ed's message: the claim that mode 0 is "byte-for-byte
identical" is wrong. Differentially, over 30000 randomised contact fields
restricted to what a device can actually emit, mode 0 diverges in 1177 of
10118 cases -- every one of them a path containing a 0x00 byte, and in every
one the old encoder was the wrong one. The accurate claim is "unchanged for
mode-0 paths containing no 0x00 byte".
This commit is contained in:
@@ -368,34 +368,30 @@ async def test_send_anon_req_flood_contact_still_forces_and_restores_zero_hop(
|
||||
assert sent == b"\x39" + bytes.fromhex(VALID_PUBKEY_HEX) + b"\x01" + b"\x00"
|
||||
|
||||
|
||||
async def test_send_anon_req_survives_change_contact_path_not_mutating(
|
||||
async def test_send_anon_req_aborts_when_change_contact_path_fails(
|
||||
command_handler, mock_connection, mock_dispatcher
|
||||
):
|
||||
"""A failed change_contact_path must not crash and must still restore the path.
|
||||
"""A failed zero-hop switch must abort, not send an unanswerable request.
|
||||
|
||||
update_contact() normally writes out_path_len back onto the dict, but if it
|
||||
errors the value stays -1. Unclamped, the unsigned to_bytes raises
|
||||
OverflowError, which skips reset_path and leaves the contact pinned to
|
||||
zero-hop on the device.
|
||||
If change_contact_path() fails the device still has the contact as flood, so
|
||||
sendAnonReq() floods the request -- and the server gates REGIONS/OWNER/BASIC
|
||||
behind isRouteDirect(), dropping it silently. Sending anyway would burn a
|
||||
full path-scaled timeout waiting for a reply that cannot arrive.
|
||||
"""
|
||||
contact = {"public_key": VALID_PUBKEY_HEX, "out_path_len": -1, "out_path": ""}
|
||||
command_handler._get_contact_by_prefix = MagicMock(return_value=contact)
|
||||
# Returns an error without reflecting the change onto the dict.
|
||||
command_handler.change_contact_path = AsyncMock(
|
||||
return_value=Event(EventType.ERROR, {"reason": "device_query_failed"})
|
||||
)
|
||||
command_handler.reset_path = AsyncMock()
|
||||
setup_event_response(
|
||||
mock_dispatcher, EventType.MSG_SENT,
|
||||
{"expected_ack": b"\x01\x02\x03\x04", "suggested_timeout": 4000},
|
||||
)
|
||||
|
||||
result = await command_handler.send_anon_req(VALID_PUBKEY_HEX, MagicMock(value=1))
|
||||
|
||||
assert not result.is_error()
|
||||
sent = mock_connection.send.await_args.args[0]
|
||||
assert sent == b"\x39" + bytes.fromhex(VALID_PUBKEY_HEX) + b"\x01" + b"\x00"
|
||||
command_handler.reset_path.assert_awaited_once()
|
||||
assert result.is_error()
|
||||
assert result.payload["reason"] == "path_reset_failed"
|
||||
mock_connection.send.assert_not_awaited()
|
||||
# Nothing was changed on the device, so there is nothing to restore.
|
||||
command_handler.reset_path.assert_not_awaited()
|
||||
|
||||
|
||||
# ── send_trace handles unknown path_hash_len without NameError ──
|
||||
@@ -427,7 +423,6 @@ async def test_ble_send_serialises_concurrent_writes():
|
||||
|
||||
conn = BLEConnection.__new__(BLEConnection) # bypass bleak availability check
|
||||
conn._disconnect_callback = None
|
||||
conn._write_lock = None
|
||||
conn.rx_char = object()
|
||||
|
||||
overlap = {"current": 0, "max": 0}
|
||||
@@ -451,7 +446,6 @@ async def test_ble_send_releases_lock_on_write_failure():
|
||||
from meshcore.ble_cx import BLEConnection
|
||||
|
||||
conn = BLEConnection.__new__(BLEConnection)
|
||||
conn._write_lock = None
|
||||
conn.rx_char = object()
|
||||
reasons = []
|
||||
|
||||
@@ -595,3 +589,128 @@ async def test_reader_contact_out_path_keeps_zero_bytes():
|
||||
assert contact["out_path_len"] == 1
|
||||
assert contact["out_path_hash_mode"] == 2
|
||||
assert contact["out_path"] == "aa00bb", "the 0x00 inside the hop hash was lost"
|
||||
|
||||
|
||||
# ── BLE write bound (a stalled write must not wedge the client) ──
|
||||
|
||||
async def test_ble_write_timeout_tears_down_the_link():
|
||||
"""A stalled write must be bounded and must drop the link, not just release.
|
||||
|
||||
CommandHandler's timeout starts only after the write returns, so nothing
|
||||
else bounds this. Releasing the lock alone would be wrong too: the
|
||||
underlying write may still be in flight, and a second write racing it
|
||||
re-creates the overlap the lock exists to prevent.
|
||||
"""
|
||||
from meshcore.ble_cx import BLEConnection
|
||||
|
||||
conn = BLEConnection.__new__(BLEConnection)
|
||||
conn.rx_char = object()
|
||||
reasons = []
|
||||
|
||||
async def capture(reason):
|
||||
reasons.append(reason)
|
||||
|
||||
conn._disconnect_callback = capture
|
||||
conn.WRITE_TIMEOUT = 0.05
|
||||
|
||||
class Stalling:
|
||||
async def write_gatt_char(self, char, data, response=True):
|
||||
await asyncio.Event().wait() # never completes
|
||||
|
||||
conn.client = Stalling()
|
||||
|
||||
result = await asyncio.wait_for(conn.send(b"\x01"), timeout=2.0)
|
||||
assert result is False
|
||||
assert reasons == ["ble_write_timeout"]
|
||||
|
||||
|
||||
async def test_ble_stalled_write_does_not_block_later_commands_forever():
|
||||
"""Head-of-line: queued writes must not inherit an unbounded stall."""
|
||||
from meshcore.ble_cx import BLEConnection
|
||||
|
||||
conn = BLEConnection.__new__(BLEConnection)
|
||||
conn.rx_char = object()
|
||||
conn._disconnect_callback = None
|
||||
conn.WRITE_TIMEOUT = 0.05
|
||||
|
||||
class Stalling:
|
||||
async def write_gatt_char(self, char, data, response=True):
|
||||
await asyncio.Event().wait()
|
||||
|
||||
conn.client = Stalling()
|
||||
|
||||
# Five writers behind one stalled holder must all return, not hang.
|
||||
done = await asyncio.wait_for(
|
||||
asyncio.gather(*(conn.send(bytes([i])) for i in range(5))), timeout=3.0
|
||||
)
|
||||
assert done == [False] * 5
|
||||
|
||||
|
||||
async def test_ble_connection_init_provides_the_lock():
|
||||
"""__init__ must set up the lock slot; the tests must not paper over it."""
|
||||
from meshcore.ble_cx import BLEConnection
|
||||
|
||||
conn = BLEConnection.__new__(BLEConnection)
|
||||
BLEConnection.__init__(conn, address="AA:BB:CC:DD:EE:FF")
|
||||
assert conn._write_lock_obj is None # lazily created
|
||||
assert isinstance(conn._write_lock, asyncio.Lock)
|
||||
assert conn._write_lock is conn._write_lock # stable across reads
|
||||
|
||||
|
||||
# ── reply-path bounds ────────────────────────────────────────
|
||||
|
||||
async def test_encode_reply_path_rejects_unsupported_hash_mode():
|
||||
# mode 3 -> 4-byte hops, which Packet::isValidPathLen refuses outright.
|
||||
assert encode_reply_path(2, "aabbccddeeffaabb", 3) == b"\x00"
|
||||
|
||||
|
||||
async def test_encode_reply_path_never_overflows_the_server_buffer():
|
||||
"""The server memcpys into a fixed 64-byte reply_path with no length check."""
|
||||
for mode in (0, 1, 2):
|
||||
hash_size = mode + 1
|
||||
out = encode_reply_path(63, "ab" * 200, mode)
|
||||
body = out[1:]
|
||||
assert len(body) <= 64, f"mode {mode} emitted {len(body)} bytes"
|
||||
assert (out[0] & 63) * hash_size == len(body), "header disagrees with body"
|
||||
|
||||
|
||||
async def test_send_anon_req_timeout_tracks_emitted_hops_not_claimed_length():
|
||||
"""Scaling must follow the hops actually sent, not a length that got clamped."""
|
||||
command_handler = CommandHandler()
|
||||
command_handler.dispatcher = MagicMock()
|
||||
command_handler._reader = MagicMock()
|
||||
conn = MagicMock()
|
||||
conn.send = AsyncMock()
|
||||
|
||||
async def sender(data):
|
||||
await conn.send(data)
|
||||
|
||||
command_handler._sender_func = sender
|
||||
# Claims 8 hops but carries only 3 bytes -> 3 hops actually emitted.
|
||||
command_handler._get_contact_by_prefix = MagicMock(return_value={
|
||||
"public_key": VALID_PUBKEY_HEX,
|
||||
"out_path_len": 8,
|
||||
"out_path": "aabbcc",
|
||||
"out_path_hash_mode": 0,
|
||||
})
|
||||
setup_event_response(
|
||||
command_handler.dispatcher, EventType.MSG_SENT,
|
||||
{"expected_ack": b"\x01\x02\x03\x04", "suggested_timeout": 1000},
|
||||
)
|
||||
|
||||
result = await command_handler.send_anon_req(VALID_PUBKEY_HEX, MagicMock(value=1))
|
||||
|
||||
sent = conn.send.await_args.args[0]
|
||||
assert sent[-4] & 63 == 3 # 3 hops on the wire
|
||||
assert result.payload["suggested_timeout"] == 4000 # (3 + 1) x 1000, not 9x
|
||||
|
||||
|
||||
async def test_encode_reply_path_saturates_rather_than_wrapping():
|
||||
"""hops is a 6-bit field; masking would wrap 64 to 0.
|
||||
|
||||
A 64-hop path would then be described as zero hops -- a zero-hop reply
|
||||
request for a distant node -- instead of being clamped and logged.
|
||||
"""
|
||||
out = encode_reply_path(64, "ab" * 64, 0)
|
||||
assert out[0] & 63 == 63
|
||||
assert len(out[1:]) == 63
|
||||
|
||||
Reference in New Issue
Block a user