From ee1bf0b18d2648eb32a8233d27694a1ab4b18866 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:02:02 +0530 Subject: [PATCH 01/16] feat(bindings): offline-protocol-verify, and scenario tests for the networking properties --- CHANGELOG.md | 23 ++ .../offline_protocol_sdk/verify/__init__.py | 21 ++ .../python/offline_protocol_sdk/verify/cli.py | 133 +++++++ .../offline_protocol_sdk/verify/client.py | 154 ++++++++ .../offline_protocol_sdk/verify/commands.py | 349 ++++++++++++++++++ bindings/python/pyproject.toml | 3 + bindings/python/tests/scenarios/__init__.py | 0 bindings/python/tests/scenarios/network.py | 181 +++++++++ .../python/tests/scenarios/test_reboot.py | 77 ++++ .../tests/scenarios/test_store_and_forward.py | 98 +++++ .../scenarios/test_through_the_middle.py | 100 +++++ bindings/python/tests/verify/__init__.py | 0 .../tests/verify/test_verify_commands.py | 256 +++++++++++++ docs/local-api.md | 56 +++ 14 files changed, 1451 insertions(+) create mode 100644 bindings/python/offline_protocol_sdk/verify/__init__.py create mode 100644 bindings/python/offline_protocol_sdk/verify/cli.py create mode 100644 bindings/python/offline_protocol_sdk/verify/client.py create mode 100644 bindings/python/offline_protocol_sdk/verify/commands.py create mode 100644 bindings/python/tests/scenarios/__init__.py create mode 100644 bindings/python/tests/scenarios/network.py create mode 100644 bindings/python/tests/scenarios/test_reboot.py create mode 100644 bindings/python/tests/scenarios/test_store_and_forward.py create mode 100644 bindings/python/tests/scenarios/test_through_the_middle.py create mode 100644 bindings/python/tests/verify/__init__.py create mode 100644 bindings/python/tests/verify/test_verify_commands.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 6862183dc..74fddbeab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -79,6 +79,29 @@ archived by series under [docs/changelog/](docs/changelog/); see the registrations survived a restart. Bluetooth LE from a container is untested on hardware. +- **`offline-protocol-verify`: drive a service and read the proof.** A + command for the device a service runs on: `send` a message, `await` its + delivery receipt on the sender or its arrival on the recipient, `pair` + (wait for the session with a peer), `ping` (a message on a cadence, each + reported with the carrier it arrived over), `watch` (every event as a JSON + line, across restarts of the service) and `state` (address, carriers, + neighbours, queues, relay counters, sessions). Each prints JSON lines and + exits 0 when what it waited for happened, 2 when its time ran out and 1 + when the engine gave the message up. It matches events by the identifier + they carry, so it works on a sender restarted since the send. A receipt + the engine emits while no client of the application is connected is not + held, so start `await` or `watch` on the sender before the recipient can + answer. New scenario tests run the networking properties end to end over + loopback peer streams with encryption on: a message to a device that is + off arrives when it returns and the sender gets the receipt; a queued + message survives the sender restarting (and is lost when the saved state + is removed, the control); a message crosses a middle device to one the + sender cannot hear. In that last one the sender's receipt never arrives: + the recipient's acknowledgement is carried back and dropped, because a + message whose only route is the mesh is refused by every carrier of the + sender and so has no pending acknowledgement to settle. The test records + it as an expected failure until the engine fix in #537 lands. + ### Changed - **A peer-stream or relay flag the configuration cannot honour is refused.** diff --git a/bindings/python/offline_protocol_sdk/verify/__init__.py b/bindings/python/offline_protocol_sdk/verify/__init__.py new file mode 100644 index 000000000..d3346522f --- /dev/null +++ b/bindings/python/offline_protocol_sdk/verify/__init__.py @@ -0,0 +1,21 @@ +"""``offline-protocol-verify``: drive a running service and read the proof. + +Every networking property the engine has, it reports as an event on the +local API: ``message_deferred`` while the recipient is away, +``message_relayed`` on a device that carries a frame for someone else, +``message_received`` with its ``hop_count`` and ``transport`` on the far +side, and ``message_delivered`` back on the sender, which is the +recipient's own acknowledgement and names the carrier it arrived on. This +package sends through the local API and waits for those events, so a +scenario passes or fails on what the engine says rather than on what an +observer reads in a log. + +It is a client of the local API like any other (``docs/spec/local-api.md`` +holds for it unchanged), runs on the device whose service it talks to, and +needs only the ``websockets`` package the SDK depends on. Guide: +``docs/local-api.md``. +""" + +from .client import VerifyClient + +__all__ = ["VerifyClient"] diff --git a/bindings/python/offline_protocol_sdk/verify/cli.py b/bindings/python/offline_protocol_sdk/verify/cli.py new file mode 100644 index 000000000..f9563af61 --- /dev/null +++ b/bindings/python/offline_protocol_sdk/verify/cli.py @@ -0,0 +1,133 @@ +"""``offline-protocol-verify``: the command line over ``commands.py``. + +Usage, on each device against its own service: + offline-protocol-verify state --peer off1... + offline-protocol-verify send off1... "hello" + offline-protocol-verify await --until delivered --timeout 120 + offline-protocol-verify await --until received + offline-protocol-verify pair off1... --timeout 60 + offline-protocol-verify ping off1... --every 2 --count 30 + offline-protocol-verify watch --log events.jsonl +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import sys +from pathlib import Path + +from websockets.exceptions import WebSocketException + +from ..local_api.cli import default_socket_path +from . import commands +from .client import DEFAULT_APP_ID, RpcFailure + + +def _positive(text: str) -> float: + value = float(text) + if not value > 0: + raise argparse.ArgumentTypeError(f"must be positive, not {text}") + return value + + +def _count(text: str) -> int: + value = int(text) + if value < 1: + raise argparse.ArgumentTypeError(f"must be at least 1, not {text}") + return value + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="offline-protocol-verify", + description="Drive the local service and wait for the events that prove what happened.", + ) + carrier = parser.add_mutually_exclusive_group() + carrier.add_argument("--socket", help=f"the service's Unix socket (default: {default_socket_path()})") + carrier.add_argument("--tcp", type=int, metavar="PORT", help="the service's loopback TCP port") + parser.add_argument("--token-file", help="the token file the service wrote (with --tcp)") + parser.add_argument( + "--app-id", + default=DEFAULT_APP_ID, + help=f"the application id to declare; the same on every device (default: {DEFAULT_APP_ID})", + ) + sub = parser.add_subparsers(dest="command", required=True) + + state = sub.add_parser("state", help="address, carriers, neighbours, queues, relay counters, sessions") + state.add_argument("--peer", action="append", default=[], help="an off1... address to report the session with") + + send = sub.add_parser("send", help="send one message and print its id") + send.add_argument("recipient") + send.add_argument("content") + + wait = sub.add_parser("await", help="wait for one message's delivery receipt, or its arrival") + wait.add_argument("message_id") + wait.add_argument("--until", choices=("delivered", "received"), default="delivered") + wait.add_argument("--timeout", type=_positive, default=120.0) + + pair = sub.add_parser("pair", help="wait until the session with a peer is confirmed") + pair.add_argument("peer") + pair.add_argument("--timeout", type=_positive, default=60.0) + + ping = sub.add_parser("ping", help="send messages on a cadence and report the carrier of each") + ping.add_argument("recipient") + ping.add_argument("--every", type=_positive, default=2.0) + ping.add_argument("--count", type=_count, default=10) + ping.add_argument("--timeout", type=_positive, default=120.0, help="per message, from its send") + + watch = sub.add_parser("watch", help="print every event, across restarts of the service") + watch.add_argument("--log", type=Path, help="also append each line to this file") + watch.add_argument("--duration", type=_positive, help="stop after this many seconds") + return parser + + +async def run(args: argparse.Namespace) -> int: + target = commands.Target( + socket_path=None if args.tcp is not None else (args.socket or str(default_socket_path())), + tcp_port=args.tcp, + token_file=args.token_file, + app_id=args.app_id, + ) + output = commands.Output() + if args.command == "state": + return await commands.state(target, args.peer, output) + if args.command == "send": + return await commands.send(target, args.recipient, args.content, output) + if args.command == "await": + return await commands.await_message(target, args.message_id, args.until, args.timeout, output) + if args.command == "pair": + return await commands.pair(target, args.peer, args.timeout, output) + if args.command == "ping": + return await commands.ping(target, args.recipient, args.every, args.count, args.timeout, output) + return await commands.watch(target, args.log, args.duration, output) + + +def main(argv: list[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if args.tcp is not None and not args.token_file: + parser.error("--tcp needs --token-file") + try: + return asyncio.run(run(args)) + except RpcFailure as exc: + print(json.dumps({"error": {"code": exc.code, "variant": exc.variant, "message": str(exc)}}), file=sys.stderr) + return commands.FAILED + except (OSError, ConnectionError, WebSocketException) as exc: + # Not a refusal from the engine: the socket, the token file, or the service went away. + print(json.dumps({"error": {"message": str(exc) or type(exc).__name__}}), file=sys.stderr) + return commands.FAILED + except NotImplementedError: + # asyncio has no Unix sockets on Windows, where the service listens on TCP. + print( + json.dumps({"error": {"message": "Unix sockets are not available on this platform; use --tcp and --token-file"}}), + file=sys.stderr, + ) + return commands.FAILED + except KeyboardInterrupt: + return commands.PASS if args.command == "watch" else commands.FAILED + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/bindings/python/offline_protocol_sdk/verify/client.py b/bindings/python/offline_protocol_sdk/verify/client.py new file mode 100644 index 000000000..5415d8ce2 --- /dev/null +++ b/bindings/python/offline_protocol_sdk/verify/client.py @@ -0,0 +1,154 @@ +"""One connection to the local API, subscribed to every event. + +Deliberately small and separate from the HTTP front's persistent client: a +verifier command is short-lived, matches events by their own identifiers, +and must not depend on the front's package. Only :func:`watch` in +``commands.py`` reconnects, around this class. +""" + +from __future__ import annotations + +import asyncio +import itertools +import json +from pathlib import Path +from typing import Any, Callable + +from websockets.asyncio.client import connect, unix_connect + +#: The application id every verifier connection declares. A received +#: message is routed to the clients of the application its sender stamped, +#: so a verifier on the sending device and one on the receiving device must +#: declare the same id. +DEFAULT_APP_ID = "offline-protocol-verify" + + +class RpcFailure(Exception): + """A JSON-RPC error: ``code`` is the number, ``variant`` the engine's name.""" + + def __init__(self, error: dict[str, Any]) -> None: + super().__init__(error.get("message") or "local API error") + self.code = error.get("code") + self.variant = (error.get("data") or {}).get("variant") + + +class VerifyClient: + """``call`` awaits the matching response; every event lands in a queue.""" + + def __init__(self, websocket: Any) -> None: + self._ws = websocket + self._ids = itertools.count(1) + self._pending: dict[int, asyncio.Future[dict[str, Any]]] = {} + self._events: asyncio.Queue[dict[str, Any] | None] = asyncio.Queue() + self.hello_result: dict[str, Any] = {} + self._reader = asyncio.ensure_future(self._read()) + + @classmethod + async def open( + cls, + *, + socket_path: str | Path | None = None, + tcp_port: int | None = None, + token_file: str | Path | None = None, + app_id: str = DEFAULT_APP_ID, + ) -> "VerifyClient": + """Connects, declares ``app_id`` and subscribes to every event.""" + if (socket_path is None) == (tcp_port is None): + raise ValueError("pass exactly one of socket_path or tcp_port") + token: str | None = None + if socket_path is not None: + websocket = await unix_connect(str(socket_path), uri="ws://localhost/", max_size=None) + else: + if token_file is None: + raise ValueError("the TCP carrier needs the service's token file") + # Read at every connect: the service writes a new token at every launch. + token = Path(token_file).read_text(encoding="ascii").strip() + websocket = await connect(f"ws://127.0.0.1:{tcp_port}/", max_size=None) + client = cls(websocket) + try: + params: dict[str, Any] = {"app_id": app_id, "client": "offline-protocol-verify"} + if token is not None: + params["token"] = token + client.hello_result = await client.call("hello", params) + await client.call("subscribe", {"types": "all"}) + except BaseException: + await client.close() + raise + return client + + @property + def local_address(self) -> str | None: + return self.hello_result.get("local_address") + + async def _read(self) -> None: + reason = "the local API connection closed" + try: + async for raw in self._ws: + try: + message = json.loads(raw) + except ValueError: + continue + if not isinstance(message, dict): + continue + if message.get("method") == "event": + params = message.get("params") + if isinstance(params, dict): + self._events.put_nowait(params) + continue + waiter = self._pending.pop(message.get("id"), None) # type: ignore[arg-type] + if waiter is not None and not waiter.done(): + waiter.set_result(message) + except Exception as exc: + reason = f"the local API connection closed: {exc}" + finally: + for waiter in self._pending.values(): + if not waiter.done(): + waiter.set_exception(ConnectionError(reason)) + self._pending.clear() + # Wakes a reader of the queue: nothing more will arrive. + self._events.put_nowait(None) + + async def call(self, method: str, params: dict[str, Any] | None = None) -> Any: + request_id = next(self._ids) + future: asyncio.Future[dict[str, Any]] = asyncio.get_running_loop().create_future() + self._pending[request_id] = future + await self._ws.send(json.dumps({"jsonrpc": "2.0", "id": request_id, "method": method, "params": params or {}})) + response = await future + if "error" in response: + raise RpcFailure(response["error"]) + return response.get("result") + + async def next_event(self, timeout: float | None) -> dict[str, Any]: + """The next event in arrival order. Raises ``TimeoutError`` when + ``timeout`` runs out and ``ConnectionError`` once the socket closed.""" + try: + event = await asyncio.wait_for(self._events.get(), timeout) + except asyncio.TimeoutError: + raise TimeoutError("no event in time") from None + if event is None: + self._events.put_nowait(None) + raise ConnectionError("the local API connection closed") + return event + + async def wait_for( + self, + accept: Callable[[dict[str, Any]], bool], + timeout: float, + on_other: Callable[[dict[str, Any]], None] | None = None, + ) -> dict[str, Any]: + """The first event ``accept`` takes; every other one goes to ``on_other``.""" + loop = asyncio.get_running_loop() + deadline = loop.time() + timeout + while True: + remaining = deadline - loop.time() + if remaining <= 0: + raise TimeoutError("no matching event in time") + event = await self.next_event(remaining) + if accept(event): + return event + if on_other is not None: + on_other(event) + + async def close(self) -> None: + await self._ws.close() + await asyncio.gather(self._reader, return_exceptions=True) diff --git a/bindings/python/offline_protocol_sdk/verify/commands.py b/bindings/python/offline_protocol_sdk/verify/commands.py new file mode 100644 index 000000000..ded36419a --- /dev/null +++ b/bindings/python/offline_protocol_sdk/verify/commands.py @@ -0,0 +1,349 @@ +"""The verifier's commands. Each one prints JSON lines on stdout, a short +summary on stderr, and returns an exit status: 0 when what it waited for +happened, 2 when its time ran out first, 1 on a terminal failure (the engine +gave the message up, refused the call, or the service went away). + +Events are read as the engine serialises them (``docs/spec/local-api.md``, +the event table): ``type`` is the tag, and a carrier is named by its +lowercase label (``wifiDirect``, ``ble``, ``internet``, ``reticulum``, +``nostr``). ``transport_switched`` is never read: the FFI layer emits it +with other names and for no Bluetooth LE edge, so it cannot say which +carrier a message took. ``message_delivered.transport`` can. +""" + +from __future__ import annotations + +import asyncio +import json +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import IO, Any, Callable + +from .client import DEFAULT_APP_ID, RpcFailure, VerifyClient + +PASS = 0 +FAILED = 1 +TIMED_OUT = 2 + +#: Events about a message the sender is still waiting on: printed as status +#: and never the outcome. ``message_undeliverable`` is among them: it is the +#: relay's or gateway's verdict that the recipient is away right now, it +#: repeats while the message is parked, and the message settles only by +#: ``message_delivered`` or ``message_failed``. +STATUS_TAGS = frozenset({"message_sent", "message_deferred", "message_retrying", "message_undeliverable"}) + +#: How often ``pair`` asks for the session state. +PAIR_POLL_INTERVAL = 0.25 + +#: Receipts ``ping`` keeps for ids it has not been handed yet. +EARLY_CAPACITY = 256 + +#: How long ``watch`` waits before reconnecting to a service that went away. +WATCH_RECONNECT_DELAY = 0.5 + + +@dataclass +class Target: + """Where the service listens, and the application id to declare.""" + + socket_path: str | None = None + tcp_port: int | None = None + token_file: str | None = None + app_id: str = DEFAULT_APP_ID + + async def open(self) -> VerifyClient: + return await VerifyClient.open( + socket_path=self.socket_path, + tcp_port=self.tcp_port, + token_file=self.token_file, + app_id=self.app_id, + ) + + +@dataclass +class Output: + """Where lines go; tests swap these for buffers.""" + + out: IO[str] = field(default_factory=lambda: sys.stdout) + err: IO[str] = field(default_factory=lambda: sys.stderr) + + def line(self, payload: dict[str, Any]) -> None: + self.out.write(json.dumps(payload, separators=(",", ":")) + "\n") + self.out.flush() + + def say(self, text: str) -> None: + self.err.write(text + "\n") + self.err.flush() + + +def _names(event: dict[str, Any], message_id: str) -> bool: + return event.get("message_id") == message_id + + +async def state(target: Target, peers: list[str], output: Output) -> int: + """What this device is and holds right now: its address, the carriers + that are up, its direct neighbours, the queues, the relay counters, and + the session state toward each of ``peers``.""" + client = await target.open() + try: + report: dict[str, Any] = { + "local_address": client.local_address, + "active_transports": await client.call("get_active_transports"), + "topology": await client.call("get_topology"), + "pending_ack_count": await client.call("get_pending_ack_count"), + "retry_queue_size": await client.call("get_retry_queue_size"), + "mesh_relay_stats": await client.call("get_mesh_relay_stats"), + "sessions": { + peer: await client.call("get_establishment_state", {"peer_id": peer}) for peer in peers + }, + } + finally: + await client.close() + output.line({"state": report}) + output.say(f"{report['local_address']}: transports {', '.join(report['active_transports']) or 'none'}") + return PASS + + +async def send(target: Target, recipient: str, content: str, output: Output) -> int: + """Sends one message and prints its id. The id is what ``await`` takes.""" + client = await target.open() + try: + message_id = await client.call( + "send_message", {"recipient": recipient, "content": content, "priority": "Medium"} + ) + finally: + await client.close() + output.line({"sent": {"message_id": message_id, "recipient": recipient}}) + output.say(f"sent {message_id} to {recipient}") + return PASS + + +async def await_message( + target: Target, + message_id: str, + until: str, + timeout: float, + output: Output, +) -> int: + """Waits for one message's outcome. + + ``until="delivered"`` runs on the sender: passes on ``message_delivered`` + naming the id, fails on ``message_failed``, and prints every status event + in between. ``until="received"`` runs on the recipient: passes on + ``message_received`` naming the id, which a service holds for the + application while no client of it is connected. + + Events are matched by the id they carry, never by the server's + correlation, which knows only the ids its own clients were handed in + this process. A sender's receipt that fires while no client is connected + is not held: start this before the recipient can answer. + """ + client = await target.open() + try: + return await _await_on(client, message_id, until, timeout, output) + finally: + await client.close() + + +async def _await_on(client: VerifyClient, message_id: str, until: str, timeout: float, output: Output) -> int: + started = time.monotonic() + if until == "received": + settles = {"message_received"} + else: + settles = {"message_delivered", "message_failed"} + + def accept(event: dict[str, Any]) -> bool: + return event.get("type") in settles and _names(event, message_id) + + def status(event: dict[str, Any]) -> None: + if event.get("type") in STATUS_TAGS and _names(event, message_id): + output.line({"status": event}) + + try: + event = await client.wait_for(accept, timeout, on_other=status) + except TimeoutError: + output.line({"result": "timeout", "message_id": message_id, "until": until}) + output.say(f"{message_id}: nothing after {timeout:g} s") + return TIMED_OUT + except ConnectionError as exc: + output.line({"result": "failed", "message_id": message_id, "reason": str(exc)}) + output.say(f"{message_id}: {exc}") + return FAILED + elapsed = round(time.monotonic() - started, 3) + if event["type"] == "message_failed": + output.line({"result": "failed", "event": event, "elapsed_s": elapsed}) + output.say(f"{message_id}: failed, {event.get('reason')}") + return FAILED + output.line({"result": "pass", "event": event, "elapsed_s": elapsed}) + output.say( + f"{message_id}: {event['type'].removeprefix('message_')} over {event.get('transport')}, " + f"{event.get('hop_count')} hop(s), after {elapsed:g} s" + ) + return PASS + + +async def pair(target: Target, peer: str, timeout: float, output: Output) -> int: + """Waits until the session toward ``peer`` is confirmed. + + The engine forms it by itself once the two devices hear each other + directly (``docs/state-machines/session-lifecycle.md``); this only waits. + The automatic key exchange never crosses a hop, so two devices that will + later reach each other only through a third must have met once.""" + client = await target.open() + loop = asyncio.get_running_loop() + deadline = loop.time() + timeout + last = None + try: + while True: + last = await client.call("get_establishment_state", {"peer_id": peer}) + if last == "SessionConfirmed": + output.line({"result": "pass", "peer": peer, "state": last}) + output.say(f"session with {peer} confirmed") + return PASS + if loop.time() >= deadline: + output.line({"result": "timeout", "peer": peer, "state": last}) + output.say(f"no session with {peer} after {timeout:g} s (state {last})") + return TIMED_OUT + await asyncio.sleep(PAIR_POLL_INTERVAL) + finally: + await client.close() + + +async def ping( + target: Target, + recipient: str, + every: float, + count: int, + timeout: float, + output: Output, +) -> int: + """Sends ``count`` messages ``every`` seconds and prints each outcome + with the carrier it arrived over. Sends do not wait for earlier answers, + so a carrier going away while some are in flight shows as those + messages arriving late, over another carrier, rather than as a stall. + Passes when every message is delivered within ``timeout`` of its send.""" + client = await target.open() + loop = asyncio.get_running_loop() + sent: dict[str, tuple[int, float]] = {} + outcomes: dict[str, dict[str, Any]] = {} + early: list[dict[str, Any]] = [] + + def record(event: dict[str, Any]) -> bool: + message_id = event.get("message_id") + if event.get("type") not in ("message_delivered", "message_failed"): + return False + if message_id in sent and message_id not in outcomes: + seq, at = sent[message_id] + outcomes[message_id] = event + output.line( + { + "seq": seq, + "message_id": message_id, + "outcome": event["type"].removeprefix("message_"), + "transport": event.get("transport"), + "hop_count": event.get("hop_count"), + "latency_ms": event.get("latency_ms"), + "elapsed_s": round(loop.time() - at, 3), + } + ) + return True + if message_id is not None and message_id not in sent: + # The receipt beat ``send_message``'s own answer: kept until the + # id is known. Bounded, since other clients' receipts land here too. + early.append(event) + del early[:-EARLY_CAPACITY] + return False + + async def drain(until: float, settled_ends: bool) -> None: + while not (settled_ends and len(outcomes) == len(sent)): + remaining = until - loop.time() + if remaining <= 0: + return + try: + record(await client.next_event(remaining)) + except TimeoutError: + return + + try: + for seq in range(count): + at = loop.time() + message_id = await client.call( + "send_message", {"recipient": recipient, "content": f"ping {seq}", "priority": "Medium"} + ) + sent[message_id] = (seq, at) + for event in [e for e in early if e.get("message_id") == message_id]: + early.remove(event) + record(event) + if seq + 1 < count: + await drain(at + every, settled_ends=False) + await drain(max(at for _, at in sent.values()) + timeout, settled_ends=True) + except ConnectionError as exc: + output.say(f"ping: {exc}") + return FAILED + finally: + await client.close() + delivered = sum(1 for e in outcomes.values() if e["type"] == "message_delivered") + failed = sum(1 for e in outcomes.values() if e["type"] == "message_failed") + carriers = sorted({str(e.get("transport")) for e in outcomes.values() if e["type"] == "message_delivered"}) + status = PASS if delivered == count else FAILED if failed else TIMED_OUT + result = {PASS: "pass", FAILED: "failed", TIMED_OUT: "timeout"}[status] + output.line({"result": result, "sent": count, "delivered": delivered, "failed": failed, "carriers": carriers}) + output.say(f"ping: {delivered}/{count} delivered over {', '.join(carriers) or 'nothing'}") + return status + + +async def watch( + target: Target, + log: Path | None, + duration: float | None, + output: Output, + on_event: Callable[[dict[str, Any]], None] | None = None, +) -> int: + """Prints every event, one JSON line each with the local receive time, + and appends it to ``log`` when given. Reconnects when the service goes + away and comes back, so one watcher spans a restart; events the engine + emits while no client is connected are not seen, except the received + messages the service holds for the application.""" + loop = asyncio.get_running_loop() + deadline = None if duration is None else loop.time() + duration + handle = log.open("a", encoding="utf-8") if log is not None else None + connected = False + try: + while deadline is None or loop.time() < deadline: + try: + client = await target.open() + except (OSError, ConnectionError, RpcFailure) as exc: + if connected: + output.say(f"watch: service gone ({exc}); reconnecting") + connected = False + await asyncio.sleep(WATCH_RECONNECT_DELAY) + continue + connected = True + output.say(f"watch: connected to {client.local_address}") + try: + while True: + remaining = None if deadline is None else deadline - loop.time() + if remaining is not None and remaining <= 0: + return PASS + try: + event = await client.next_event(remaining) + except TimeoutError: + return PASS + line = {"at_ms": int(time.time() * 1000), "event": event} + output.line(line) + if handle is not None: + handle.write(json.dumps(line, separators=(",", ":")) + "\n") + handle.flush() + if on_event is not None: + on_event(event) + except ConnectionError: + output.say("watch: service closed the connection; reconnecting") + connected = False + finally: + await client.close() + return PASS + finally: + if handle is not None: + handle.close() diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml index 4adfa17e3..a67e5f374 100644 --- a/bindings/python/pyproject.toml +++ b/bindings/python/pyproject.toml @@ -73,6 +73,9 @@ http = [ offline-protocol-service = "offline_protocol_sdk.local_api.cli:main" # The HTTP front on its own, against a service already running. offline-protocol-http-front = "offline_protocol_sdk.http_front.cli:main" +# Drives the local service and waits for the events that prove a message +# was held, carried and delivered (docs/local-api.md). +offline-protocol-verify = "offline_protocol_sdk.verify.cli:main" [project.urls] Homepage = "https://github.com/Offline-Protocol/offline-protocol-sdk" diff --git a/bindings/python/tests/scenarios/__init__.py b/bindings/python/tests/scenarios/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/bindings/python/tests/scenarios/network.py b/bindings/python/tests/scenarios/network.py new file mode 100644 index 000000000..39ec70bf0 --- /dev/null +++ b/bindings/python/tests/scenarios/network.py @@ -0,0 +1,181 @@ +"""Devices for the scenario tests: each one a local API server over its own +engine, with its own file stores, a peer-stream listener on loopback, and a +static peer list. A device can be switched off (its server and engine stop, +its stores are released) and on again over the same stores and port, which +is what a power cycle leaves: the identity, the sessions and every queued +message, and nothing that lived in memory. + +Encryption is on and required, as in the container image: a message crosses +only inside an MLS session the engines formed by themselves. +""" + +from __future__ import annotations + +import asyncio +import os +import shutil +import tempfile +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +import pytest + +from offline_protocol_sdk.local_api.server import LocalApiServer +from offline_protocol_sdk.offline_protocol import OverflowPolicy, ProtocolConfig +from offline_protocol_sdk.protocol_manager import ProtocolManager +from offline_protocol_sdk.verify.client import VerifyClient + +#: Generous for loopback, where a session forms in well under a second; a +#: loaded CI runner is the reason for the margin. +SESSION_TIMEOUT = 30.0 +DELIVERY_TIMEOUT = 45.0 + + +def scenario_config(profile: str) -> ProtocolConfig: + return ProtocolConfig( + app_id="scenario", + profile=profile, + ble_enabled=False, + wifi_direct_enabled=True, + internet_enabled=False, + reticulum_enabled=False, + nostr_enabled=False, + prefer_online=False, + initial_ttl=5, + encryption_enabled=True, + auto_key_exchange=True, + store_pending=True, + require_encryption=True, + max_pending_per_peer=100, + max_pending_global=1000, + pending_ttl_ms=604_800_000, + overflow_policy=OverflowPolicy.DROP_OLDEST, + ) + + +@dataclass +class Device: + name: str + root: Path + key: bytes + peers: list["Device"] = field(default_factory=list) + port: int = 0 + address: str | None = None + server: LocalApiServer | None = None + + @property + def socket_path(self) -> Path: + return self.root / "run" / "api.sock" + + @property + def on(self) -> bool: + return self.server is not None + + async def switch_on(self, *, state_root: Path | None = None) -> None: + """Starts the engine over this device's stores, listening on the + port it had before (any free one the first time), dialling its + peers. ``state_root`` replaces the protocol-state directory, for a + test that must show what is lost without it.""" + manager = ProtocolManager( + scenario_config(self.name), + store_key=self.key, + mls_root=self.root / "mls", + state_root=state_root or self.root / "state", + ) + manager.peer_stream.configure( + listen_host="127.0.0.1", + listen_port=self.port, + peers=[f"127.0.0.1:{peer.port}" for peer in self.peers], + ) + server = LocalApiServer(manager, socket_path=self.socket_path, health=False) + await server.start() + self.server = server + self.port = manager.peer_stream.listen_port or 0 + address = manager.local_address + assert address is not None and (self.address is None or address == self.address), ( + f"{self.name} came back as {address}, not {self.address}" + ) + self.address = address + + async def switch_off(self) -> None: + server, self.server = self.server, None + if server is not None: + await server.stop() + + def linked_to(self, other: "Device") -> bool: + return self.server is not None and other.address in self.server.manager.peer_stream.connected_peers() + + async def client(self) -> VerifyClient: + return await VerifyClient.open(socket_path=self.socket_path) + + +class Network: + """Creates devices under one short temporary directory (a Unix socket + path is limited to about a hundred bytes) and switches them all off at + the end.""" + + def __init__(self) -> None: + # The release also runs this suite on Windows, where asyncio has no + # Unix sockets; the verifier's TCP carrier is covered in tests/verify. + if os.name == "nt": + pytest.skip("the scenario devices serve the local API on a Unix socket") + self._tmp =Path(tempfile.mkdtemp(prefix="opnet-", dir="/tmp")) + self.devices: list[Device] = [] + self._clients: list[VerifyClient] = [] + + def device(self, name: str, key_byte: int) -> Device: + root = self._tmp / name + (root / "run").mkdir(parents=True, mode=0o700) + device = Device(name=name, root=root, key=bytes([key_byte] * 32)) + self.devices.append(device) + return device + + async def client(self, device: Device) -> VerifyClient: + client = await device.client() + self._clients.append(client) + return client + + async def close(self) -> None: + for client in self._clients: + await client.close() + for device in self.devices: + await device.switch_off() + shutil.rmtree(self._tmp, ignore_errors=True) + + +async def until(predicate: Any, timeout: float, what: str) -> None: + loop = asyncio.get_running_loop() + deadline = loop.time() + timeout + while not predicate(): + if loop.time() > deadline: + raise AssertionError(f"{what}: not within {timeout:g} s") + await asyncio.sleep(0.05) + + +class Recorder: + """Every event one client sees, kept in order, so a test can assert on + what happened before the outcome as well as on the outcome.""" + + def __init__(self, client: VerifyClient) -> None: + self.client = client + self.events: list[dict[str, Any]] = [] + + async def wait(self, tag: str, timeout: float, **fields: Any) -> dict[str, Any]: + def matches(event: dict[str, Any]) -> bool: + return event.get("type") == tag and all(event.get(k) == v for k, v in fields.items()) + + for event in self.events: + if matches(event): + return event + try: + event = await self.client.wait_for(matches, timeout, on_other=self.events.append) + except TimeoutError: + raise AssertionError(f"no {tag} {fields} within {timeout:g} s; saw {self.tags()}") from None + self.events.append(event) + return event + + def tags(self, message_id: str | None = None) -> list[str]: + return [ + e["type"] for e in self.events if message_id is None or e.get("message_id") == message_id + ] diff --git a/bindings/python/tests/scenarios/test_reboot.py b/bindings/python/tests/scenarios/test_reboot.py new file mode 100644 index 000000000..acf350fd0 --- /dev/null +++ b/bindings/python/tests/scenarios/test_reboot.py @@ -0,0 +1,77 @@ +"""A message queued for an absent recipient survives the sender restarting: +the queue is on disk before ``send_message`` returns, and a new engine over +the same stores sends it once the recipient is back. + +The sender's receipt is awaited by a client of the restarted server, which +was never handed the id: it is matched by the id it carries, not by the +server's correlation, which lives for one process only. +""" + +from __future__ import annotations + +import pytest + +from .network import DELIVERY_TIMEOUT, SESSION_TIMEOUT, Network, Recorder, until +from .test_store_and_forward import _paired + + +@pytest.fixture +async def network(): + net = Network() + try: + yield net + finally: + await net.close() + + +async def _queue_while_b_is_off(network): + a = network.device("alpha", 0x11) + b = network.device("bravo", 0x22) + b.peers = [a] + await a.switch_on() + await b.switch_on() + await _paired(a, b) + await b.switch_off() + await until(lambda: not a.linked_to(b), SESSION_TIMEOUT, "A noticing B is gone") + + before = await network.client(a) + message_id = await before.call( + "send_message", {"recipient": b.address, "content": "queued before the restart", "priority": "Medium"} + ) + await Recorder(before).wait("message_deferred", DELIVERY_TIMEOUT, message_id=message_id) + return a, b, message_id + + +async def test_a_queued_message_survives_the_sender_restarting(network): + a, b, message_id = await _queue_while_b_is_off(network) + + await a.switch_off() + await a.switch_on() + # The server that will deliver it never handed this id to anyone. + assert not a.server.router.knows(message_id) + after = Recorder(await network.client(a)) + await b.switch_on() + + delivered = await after.wait("message_delivered", DELIVERY_TIMEOUT, message_id=message_id) + assert delivered["transport"] == "wifiDirect" + received = Recorder(await network.client(b)) + message = await received.wait("message_received", DELIVERY_TIMEOUT, message_id=message_id) + assert message["content"] == "queued before the restart" + assert message["sender"] == a.address + + +async def test_without_the_saved_state_the_message_is_gone(network): + """The control for the test above: the same restart over an empty + protocol-state directory (the identity and the sessions are in the MLS + store and survive) delivers nothing, so it is the saved queue that + carries the message across the restart, not anything in memory.""" + a, b, message_id = await _queue_while_b_is_off(network) + + await a.switch_off() + await a.switch_on(state_root=a.root / "state-empty") + after = Recorder(await network.client(a)) + await b.switch_on() + await until(lambda: a.linked_to(b), SESSION_TIMEOUT, "A and B linked again") + + with pytest.raises(AssertionError, match="no message_delivered"): + await after.wait("message_delivered", 5.0, message_id=message_id) diff --git a/bindings/python/tests/scenarios/test_store_and_forward.py b/bindings/python/tests/scenarios/test_store_and_forward.py new file mode 100644 index 000000000..9d6aa6cd1 --- /dev/null +++ b/bindings/python/tests/scenarios/test_store_and_forward.py @@ -0,0 +1,98 @@ +"""A message to a device that is switched off is held by the sender and +delivered when the device comes back, and the sender hears the recipient's +own acknowledgement. + +Two devices over loopback peer streams, as two hosts on a LAN: B dials A. +""" + +from __future__ import annotations + +import asyncio +import io +import json + +import pytest + +from offline_protocol_sdk.verify import commands + +from .network import DELIVERY_TIMEOUT, SESSION_TIMEOUT, Network, Recorder, until + + +@pytest.fixture +async def network(): + net = Network() + try: + yield net + finally: + await net.close() + + +def _target(device) -> commands.Target: + return commands.Target(socket_path=str(device.socket_path)) + + +async def _paired(a, b) -> None: + out = commands.Output(out=io.StringIO(), err=io.StringIO()) + assert await commands.pair(_target(a), b.address, SESSION_TIMEOUT, out) == commands.PASS + assert await commands.pair(_target(b), a.address, SESSION_TIMEOUT, out) == commands.PASS + + +async def test_a_message_to_a_device_that_is_off_arrives_when_it_comes_back(network): + a = network.device("alpha", 0x11) + b = network.device("bravo", 0x22) + b.peers = [a] + await a.switch_on() + await b.switch_on() + await _paired(a, b) + + await b.switch_off() + await until(lambda: not a.linked_to(b), SESSION_TIMEOUT, "A noticing B is gone") + + sender = Recorder(await network.client(a)) + message_id = await sender.client.call( + "send_message", {"recipient": b.address, "content": "while you were out", "priority": "Medium"} + ) + # Held, not lost and not failed: the peer stream refused a recipient it + # has no stream to, and the message waits in the outbox. + deferred = await sender.wait("message_deferred", DELIVERY_TIMEOUT, message_id=message_id) + assert deferred["recipient"] == b.address + assert "message_delivered" not in sender.tags(message_id) + + await b.switch_on() + receiver = Recorder(await network.client(b)) + received = await receiver.wait("message_received", DELIVERY_TIMEOUT, message_id=message_id) + assert received["content"] == "while you were out" + assert received["sender"] == a.address + assert received["encrypted"] is True + + delivered = await sender.wait("message_delivered", DELIVERY_TIMEOUT, message_id=message_id) + assert delivered["transport"] == "wifiDirect" + assert delivered["hop_count"] == 0 + assert "message_failed" not in sender.tags(message_id) + + +async def test_the_verifier_waits_out_the_absence_and_passes_on_the_receipt(network): + """The same, through the commands an operator runs: ``await`` started on + the sender while the recipient is still off, and passing once it is on.""" + a =network.device("alpha", 0x11) + b = network.device("bravo", 0x22) + b.peers = [a] + await a.switch_on() + await b.switch_on() + await _paired(a, b) + await b.switch_off() + await until(lambda: not a.linked_to(b), SESSION_TIMEOUT, "A noticing B is gone") + + out = commands.Output(out=io.StringIO(), err=io.StringIO()) + assert await commands.send(_target(a), b.address, "later", out) == commands.PASS + message_id = json.loads(out.out.getvalue().splitlines()[-1])["sent"]["message_id"] + + waiting = asyncio.ensure_future( + commands.await_message(_target(a), message_id, "delivered", DELIVERY_TIMEOUT, out) + ) + await asyncio.sleep(0.5) + assert not waiting.done() + await b.switch_on() + assert await waiting == commands.PASS + assert '"result":"pass"' in out.out.getvalue() + assert await commands.await_message(_target(b), message_id, "received", DELIVERY_TIMEOUT, out) == commands.PASS diff --git a/bindings/python/tests/scenarios/test_through_the_middle.py b/bindings/python/tests/scenarios/test_through_the_middle.py new file mode 100644 index 000000000..df66c3db5 --- /dev/null +++ b/bindings/python/tests/scenarios/test_through_the_middle.py @@ -0,0 +1,100 @@ +"""A reaches C through B when A and C cannot hear each other. + +A and C meet once first, directly, so their engines form the MLS session +(the automatic key exchange and the Welcome travel only over a direct link, +never through the mesh). Then C restarts with B as its only peer, A and C +have no link, and a message from A to C is carried by B: B reports +``message_relayed`` and C receives it one hop away. + +The sender's receipt does not come back, and the second test records that as +an expected failure. C's acknowledgement is carried back by B and reaches A, +but A drops it: every send of a message whose only route is the mesh is +refused by A's own carriers (``handle_send_failure``, ``send.rs``), so no +pending acknowledgement is ever registered for it, and an acknowledgement +with no pending record settles only a DM the relay parked +(``settle_parked_dm_from_ack``, ``send.rs:5185``). A keeps re-offering the +message, C keeps re-acknowledging the duplicates, and the outbox reports the +delivered message failed when its lifetime ends. The Rust neighbourhood +simulator does not catch it: ``the_answer_finds_its_way_back`` asserts that +B transmits toward A, never that A emits ``message_delivered``. PR #537 +fixes it in the engine; the expected failure is strict, so the test turns +red once that lands, and the marker comes off then. +""" + +from __future__ import annotations + +import pytest + +from .network import DELIVERY_TIMEOUT, SESSION_TIMEOUT, Network, Recorder, until +from .test_store_and_forward import _paired + + +@pytest.fixture +async def network(): + net = Network() + try: + yield net + finally: + await net.close() + + +async def _line_of_three(network): + """A - B - C, with A and C paired before their direct link goes away.""" + a = network.device("alpha", 0x11) + b = network.device("bravo", 0x22) + c = network.device("charlie", 0x33) + await b.switch_on() + a.peers = [b] + await a.switch_on() + + # Met once: C dials A as well as B, and the two form their session. + c.peers = [b, a] + await c.switch_on() + await _paired(a, c) + + # From now on C hears only B. + await c.switch_off() + c.peers = [b] + await c.switch_on() + await until(lambda: b.linked_to(c) and b.linked_to(a), SESSION_TIMEOUT, "B linked to A and C") + assert not a.linked_to(c) and not c.linked_to(a) + return a, b, c + + +async def test_a_message_crosses_the_middle_device(network): + a, b, c = await _line_of_three(network) + middle = Recorder(await network.client(b)) + receiver = Recorder(await network.client(c)) + sender = Recorder(await network.client(a)) + message_id = await sender.client.call( + "send_message", {"recipient": c.address, "content": "via bravo", "priority": "Medium"} + ) + + received = await receiver.wait("message_received", DELIVERY_TIMEOUT, message_id=message_id) + assert received["content"] == "via bravo" + assert received["sender"] == a.address + assert received["encrypted"] is True + assert received["hop_count"] == 1 + + relayed = await middle.wait("message_relayed", DELIVERY_TIMEOUT, message_id=message_id) + assert relayed["sender"] == a.address and relayed["recipient"] == c.address + stats = await middle.client.call("get_mesh_relay_stats") + assert stats["forwarded"] >= 1 + assert not a.linked_to(c) + + +@pytest.mark.xfail( + strict=True, + reason="a mesh-only DM registers no pending acknowledgement, so the carried receipt is dropped; fixed by #537 (module docstring)", +) +async def test_the_receipt_comes_back_through_the_middle_device(network): + a, b, c = await _line_of_three(network) + middle = Recorder(await network.client(b)) + sender = Recorder(await network.client(a)) + message_id = await sender.client.call( + "send_message", {"recipient": c.address, "content": "via bravo", "priority": "Medium"} + ) + # The receipt is carried: B forwards a frame from C to A. + await middle.wait("message_relayed", DELIVERY_TIMEOUT, sender=c.address, recipient=a.address) + delivered = await sender.wait("message_delivered", 20.0, message_id=message_id) + assert delivered["hop_count"] == 1 diff --git a/bindings/python/tests/verify/__init__.py b/bindings/python/tests/verify/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py new file mode 100644 index 000000000..98755f256 --- /dev/null +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -0,0 +1,256 @@ +"""The verifier's commands against in-process servers. + +Two engines with encryption off over loopback streams, from the local API +suite's harness: what is under test here is the command, its output and its +exit status, not the engine. The scenarios with encryption on and devices +switched off are in ``tests/scenarios``. +""" + +from __future__ import annotations + +import asyncio +import io +import json + +import pytest + +from offline_protocol_sdk.local_api.server import LocalApiServer +from offline_protocol_sdk.protocol_manager import ProtocolManager +from offline_protocol_sdk.verify import cli, commands +from offline_protocol_sdk.verify.client import VerifyClient + +from local_api.conftest import harness, make_config # noqa: F401 +from local_api.test_examples import two_servers + + +def _output() -> commands.Output: + return commands.Output(out=io.StringIO(), err=io.StringIO()) + + +def _lines(output: commands.Output) -> list[dict]: + return [json.loads(line) for line in output.out.getvalue().splitlines() if line.strip()] + + +def _target(server) -> commands.Target: + return commands.Target(socket_path=str(server.socket_path)) + + +async def test_state_reports_the_address_the_carriers_and_the_session(harness): + server_a, server_b = await two_servers(harness) + out = _output() + status = await commands.state(_target(server_a), [server_b.manager.local_address], out) + assert status == commands.PASS + (line,) = _lines(out) + report = line["state"] + assert report["local_address"] == server_a.manager.local_address + assert "WiFiDirect" in report["active_transports"] + assert set(report) >= { + "topology", + "pending_ack_count", + "retry_queue_size", + "mesh_relay_stats", + "sessions", + } + assert server_b.manager.local_address in report["sessions"] + assert "forwarded" in report["mesh_relay_stats"] + + +async def test_send_then_await_passes_on_the_receipt_and_the_arrival(harness): + server_a, server_b = await two_servers(harness) + recipient = server_b.manager.local_address + out = _output() + # One connection for the send and the wait: a receipt that fires while no + # client of the application is connected is not held. + sender = await VerifyClient.open(socket_path=server_a.socket_path) + try: + message_id = await sender.call( + "send_message", {"recipient": recipient, "content": "checked", "priority": "Medium"} + ) + assert await commands._await_on(sender, message_id, "delivered", 15.0, out) == commands.PASS + finally: + await sender.close() + result = _lines(out)[-1] + assert result["result"] == "pass" + assert result["event"]["type"] == "message_delivered" + assert result["event"]["message_id"] == message_id + assert result["event"]["transport"] == "wifiDirect" + + # The recipient's service held the message for the application id; a + # verifier connecting afterwards still gets it. + arrival = _output() + assert await commands.await_message(_target(server_b), message_id, "received", 15.0, arrival) == commands.PASS + event = _lines(arrival)[-1]["event"] + assert event["type"] == "message_received" and event["content"] == "checked" + + +async def test_send_prints_the_id_await_matches(harness): + server_a, server_b = await two_servers(harness) + out = _output() + assert await commands.send(_target(server_a), server_b.manager.local_address, "one", out) == commands.PASS + message_id = _lines(out)[0]["sent"]["message_id"] + arrival = _output() + assert await commands.await_message(_target(server_b), message_id, "received", 15.0, arrival) == commands.PASS + + +async def test_await_times_out_with_status_two(harness): + server = await harness.server(config=make_config(profile="alone")) + out = _output() + status = await commands.await_message(_target(server), "no-such-message", "delivered", 0.3, out) + assert status == commands.TIMED_OUT + assert _lines(out)[-1] == {"result": "timeout", "message_id": "no-such-message", "until": "delivered"} + + +async def test_await_fails_on_message_failed_and_prints_status_events(harness): + server = await harness.server(config=make_config(profile="alone")) + out = _output() + client = await VerifyClient.open(socket_path=server.socket_path) + try: + # The engine's own events stand in for a real failure: what is under + # test is that the command reads them, in order, by id. + task = asyncio.ensure_future(commands._await_on(client, "m-1", "delivered", 5.0, out)) + await asyncio.sleep(0.1) + for event in ( + {"type": "message_deferred", "message_id": "m-1", "recipient": "x", "reason": "peer_not_reachable"}, + {"type": "message_undeliverable", "message_id": "m-1", "recipient": "x", "reason": "recipient_unreachable"}, + {"type": "message_deferred", "message_id": "m-2", "recipient": "x", "reason": "peer_not_reachable"}, + {"type": "message_failed", "message_id": "m-1", "reason": "Outbox lifetime exceeded", "retry_count": 0}, + ): + client._events.put_nowait(event) + assert await task == commands.FAILED + finally: + await client.close() + lines = _lines(out) + assert [line["status"]["type"] for line in lines if "status" in line] == [ + "message_deferred", + "message_undeliverable", + ] + assert lines[-1]["result"] == "failed" + assert lines[-1]["event"]["reason"] == "Outbox lifetime exceeded" + + +async def test_pair_passes_once_confirmed_and_times_out_otherwise(harness, monkeypatch): + server = await harness.server(config=make_config(profile="alone")) + out = _output() + status = await commands.pair(_target(server), "off1nobody", 0.4, out) + assert status == commands.TIMED_OUT + assert _lines(out)[-1]["result"] == "timeout" + + # A real confirmation needs encryption on and is what the scenarios + # wait for; here the engine's answers are scripted to pin the polling. + states = iter(["NoKeyPackage", "SessionPending", "SessionConfirmed"]) + real_call = VerifyClient.call + + async def fake_call(self, method, params=None): + if method != "get_establishment_state": + return await real_call(self, method, params) + return next(states) + + monkeypatch.setattr(commands, "PAIR_POLL_INTERVAL", 0.01) + monkeypatch.setattr(VerifyClient, "call", fake_call) + out = _output() + assert await commands.pair(_target(server), "off1somebody", 5.0, out) == commands.PASS + assert _lines(out)[-1] == {"result": "pass", "peer": "off1somebody", "state": "SessionConfirmed"} + + +async def test_ping_reports_each_message_and_the_carrier(harness): + server_a, server_b = await two_servers(harness) + out = _output() + status = await commands.ping(_target(server_a), server_b.manager.local_address, 0.1, 3, 15.0, out) + assert status == commands.PASS + lines = _lines(out) + per_message = [line for line in lines if "seq" in line] + assert sorted(line["seq"] for line in per_message) == [0, 1, 2] + assert {line["outcome"] for line in per_message} == {"delivered"} + assert lines[-1] == {"result": "pass", "sent": 3, "delivered": 3, "failed": 0, "carriers": ["wifiDirect"]} + + +async def test_ping_times_out_toward_nobody(harness): + server = await harness.server(config=make_config(profile="alone", internet_enabled=False, wifi_direct_enabled=True)) + out = _output() + status = await commands.ping(_target(server), "off1nobody", 0.05, 2, 0.3, out) + assert status == commands.TIMED_OUT + assert _lines(out)[-1] == {"result": "timeout", "sent": 2, "delivered": 0, "failed": 0, "carriers": []} + + +async def test_watch_logs_every_event_and_spans_a_restart(harness, tmp_path): + server_a, server_b = await two_servers(harness) + log = tmp_path / "events.jsonl" + out = _output() + seen: list[dict] = [] + watcher = asyncio.ensure_future( + commands.watch(_target(server_a), log, 30.0, out, on_event=seen.append) + ) + try: + await asyncio.sleep(0.3) + sender = await VerifyClient.open(socket_path=server_a.socket_path) + try: + message_id = await sender.call( + "send_message", + {"recipient": server_b.manager.local_address, "content": "logged", "priority": "Medium"}, + ) + finally: + await sender.close() + for _ in range(300): + if any(e.get("type") == "message_delivered" and e.get("message_id") == message_id for e in seen): + break + await asyncio.sleep(0.05) + else: + raise AssertionError(f"no receipt in the watch: {[e.get('type') for e in seen]}") + finally: + watcher.cancel() + await asyncio.gather(watcher, return_exceptions=True) + logged = [json.loads(line) for line in log.read_text().splitlines()] + assert any(line["event"].get("message_id") == message_id for line in logged) + assert all("at_ms" in line for line in logged) + + +async def test_watch_reconnects_when_the_service_comes_back(harness, tmp_path, monkeypatch): + monkeypatch.setattr(commands, "WATCH_RECONNECT_DELAY", 0.05) + first = await harness.server(config=make_config(profile="first")) + socket_path = first.socket_path + out = _output() + watcher = asyncio.ensure_future(commands.watch(commands.Target(socket_path=str(socket_path)), None, 30.0, out)) + try: + for _ in range(100): + if "connected" in out.err.getvalue(): + break + await asyncio.sleep(0.05) + await first.stop() + second = LocalApiServer(ProtocolManager(make_config(profile="second")), socket_path=socket_path, health=False) + await second.start() + try: + for _ in range(200): + if out.err.getvalue().count("watch: connected") >= 2: + break + await asyncio.sleep(0.05) + assert out.err.getvalue().count("watch: connected") == 2, out.err.getvalue() + finally: + await second.stop() + finally: + watcher.cancel() + await asyncio.gather(watcher, return_exceptions=True) + + +async def test_the_tcp_carrier_reads_the_token_file(harness): + server = await harness.server(config=make_config(profile="over-tcp"), tcp=True) + target = commands.Target(tcp_port=server.port, token_file=str(server.token_path)) + out = _output() + assert await commands.state(target, [], out) == commands.PASS + assert _lines(out)[0]["state"]["local_address"] == server.manager.local_address + + +def test_the_command_line_refuses_tcp_without_a_token_file(capsys): + with pytest.raises(SystemExit): + cli.main(["--tcp", "7800", "state"]) + assert "--tcp needs --token-file" in capsys.readouterr().err + + +def test_the_command_line_reports_a_missing_service_as_a_failure(tmp_path, capsys): + status = cli.main(["--socket", str(tmp_path / "absent.sock"), "state"]) + assert status == commands.FAILED + assert json.loads(capsys.readouterr().err.strip().splitlines()[-1])["error"]["message"] + + +def test_the_command_line_refuses_a_zero_count(): + with pytest.raises(SystemExit): + cli.build_parser().parse_args(["ping", "off1x", "--count", "0"]) diff --git a/docs/local-api.md b/docs/local-api.md index 23eee217c..9aa61b62d 100644 --- a/docs/local-api.md +++ b/docs/local-api.md @@ -149,6 +149,62 @@ model. [examples/http-front](../examples/http-front) has a provider and a client, and [the container image](../bindings/python/docker) runs one service per host with the front on. +## Checking what the network did: `offline-protocol-verify` + +The engine reports every networking fact as an event: `message_deferred` +while a recipient is away, `message_relayed` on a device that carries a +frame for someone else, `message_received` with its `hop_count` and +`transport` on the far side, and `message_delivered` back on the sender, +which is the recipient's own acknowledgement and names the carrier it +arrived over. `offline-protocol-verify` sends through the service on the +device it runs on and waits for those events, so a check passes or fails on +what the engine says: + +```bash +# on A: send, then wait for B's acknowledgement (up to 10 minutes) +id=$(offline-protocol-verify send off1...B "hello" | jq -r .sent.message_id) +offline-protocol-verify await "$id" --until delivered --timeout 600 +# on B: wait for the message itself +offline-protocol-verify await "$id" --until received +``` + +| Command | Waits for | Passes on | +|---|---|---| +| `state [--peer ADDR]` | nothing | always; prints address, carriers, neighbours, queues, relay counters, the session with each peer | +| `send ADDR TEXT` | nothing | always; prints the message id | +| `await ID --until delivered` | the receipt on the sender | `message_delivered` naming the id; `message_failed` is a failure, `message_deferred`, `message_retrying` and `message_undeliverable` print as status | +| `await ID --until received` | the message on the recipient | `message_received` naming the id | +| `pair ADDR` | the session with a peer | `get_establishment_state` reaching `SessionConfirmed` | +| `ping ADDR --every S --count N` | each message's receipt | every message delivered within `--timeout` of its send; each line names the carrier | +| `watch [--log FILE]` | until stopped or `--duration` | every event as `{"at_ms", "event"}`, reconnecting across restarts of the service | + +Each command prints JSON lines on standard output and a summary on standard +error, and exits 0 on a pass, 2 when its time ran out, and 1 on a terminal +failure (the engine gave the message up, refused a call, or the service went +away). Every verifier declares the application id `offline-protocol-verify` +unless told otherwise with `--app-id`, and the sending and receiving +devices must declare the same one: a received message is routed to the +clients of the application its sender stamped, and held for that +application, 256 deep, while none is connected. + +Two things about events decide how a check is written. A verifier matches +events by the identifier they carry rather than relying on the server's +correlation, which knows only the identifiers its own clients were handed +by this process; so `await` works on a sender restarted since the send. And +an event the engine emits while no client of the application is connected +is gone, except a received message: start `await` or `watch` on the sender +before the recipient can answer, which for a restart test means restarting +the sender while the recipient is still away. Two devices that will reach +each other only through a third must have met directly once: the automatic +key exchange and the Welcome travel over a direct link, never through the +mesh. Today a message that crosses the mesh is received, but the sender's +receipt never arrives: the acknowledgement is carried back and dropped, +because a message whose only route is the mesh has no pending +acknowledgement to settle (an engine defect the scenario tests record as an +expected failure; #537 fixes it). Until then, check a crossing on the +recipient with `await --until received`, and on the middle device in the +`watch` log as `message_relayed`. + ## The policy file With no policy, any well-formed application id is accepted and nothing is From 53cfad2471b8f5fb8bbec1c684aa2f705eb3bab2 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:20:22 +0530 Subject: [PATCH 02/16] fix(bindings): state and pair never take a message the service holds for the verifier --- CHANGELOG.md | 2 +- .../python/offline_protocol_sdk/verify/cli.py | 2 +- .../offline_protocol_sdk/verify/commands.py | 20 ++++++++++++----- .../tests/verify/test_verify_commands.py | 22 ++++++++++++++++++- docs/local-api.md | 13 ++++++++--- 5 files changed, 47 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 74fddbeab..0484cbd4d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -85,7 +85,7 @@ archived by series under [docs/changelog/](docs/changelog/); see the (wait for the session with a peer), `ping` (a message on a cadence, each reported with the carrier it arrived over), `watch` (every event as a JSON line, across restarts of the service) and `state` (address, carriers, - neighbours, queues, relay counters, sessions). Each prints JSON lines and + queues, relay counters, sessions). Each prints JSON lines and exits 0 when what it waited for happened, 2 when its time ran out and 1 when the engine gave the message up. It matches events by the identifier they carry, so it works on a sender restarted since the send. A receipt diff --git a/bindings/python/offline_protocol_sdk/verify/cli.py b/bindings/python/offline_protocol_sdk/verify/cli.py index f9563af61..394351397 100644 --- a/bindings/python/offline_protocol_sdk/verify/cli.py +++ b/bindings/python/offline_protocol_sdk/verify/cli.py @@ -55,7 +55,7 @@ def build_parser() -> argparse.ArgumentParser: ) sub = parser.add_subparsers(dest="command", required=True) - state = sub.add_parser("state", help="address, carriers, neighbours, queues, relay counters, sessions") + state = sub.add_parser("state", help="address, carriers, queues, relay counters, sessions") state.add_argument("--peer", action="append", default=[], help="an off1... address to report the session with") send = sub.add_parser("send", help="send one message and print its id") diff --git a/bindings/python/offline_protocol_sdk/verify/commands.py b/bindings/python/offline_protocol_sdk/verify/commands.py index ded36419a..b8d567974 100644 --- a/bindings/python/offline_protocol_sdk/verify/commands.py +++ b/bindings/python/offline_protocol_sdk/verify/commands.py @@ -34,6 +34,10 @@ #: ``message_delivered`` or ``message_failed``. STATUS_TAGS = frozenset({"message_sent", "message_deferred", "message_retrying", "message_undeliverable"}) +#: Appended to the application id by the commands that only read state +#: (``state``, ``pair``); see :meth:`Target.open`. +OBSERVER_SUFFIX = ".observer" + #: How often ``pair`` asks for the session state. PAIR_POLL_INTERVAL = 0.25 @@ -53,12 +57,17 @@ class Target: token_file: str | None = None app_id: str = DEFAULT_APP_ID - async def open(self) -> VerifyClient: + async def open(self, *, observer: bool = False) -> VerifyClient: + """Connects as the application, or with ``observer`` under an id of + its own. The service hands every message it held for an application + to the first connection of that application, so a command that only + reads state must not declare it: a ``state`` run on the recipient + before ``await --until received`` would take the message and drop it.""" return await VerifyClient.open( socket_path=self.socket_path, tcp_port=self.tcp_port, token_file=self.token_file, - app_id=self.app_id, + app_id=f"{self.app_id}{OBSERVER_SUFFIX}" if observer else self.app_id, ) @@ -84,14 +93,13 @@ def _names(event: dict[str, Any], message_id: str) -> bool: async def state(target: Target, peers: list[str], output: Output) -> int: """What this device is and holds right now: its address, the carriers - that are up, its direct neighbours, the queues, the relay counters, and + that are up, the queues, the relay counters, and the session state toward each of ``peers``.""" - client = await target.open() + client = await target.open(observer=True) try: report: dict[str, Any] = { "local_address": client.local_address, "active_transports": await client.call("get_active_transports"), - "topology": await client.call("get_topology"), "pending_ack_count": await client.call("get_pending_ack_count"), "retry_queue_size": await client.call("get_retry_queue_size"), "mesh_relay_stats": await client.call("get_mesh_relay_stats"), @@ -191,7 +199,7 @@ async def pair(target: Target, peer: str, timeout: float, output: Output) -> int directly (``docs/state-machines/session-lifecycle.md``); this only waits. The automatic key exchange never crosses a hop, so two devices that will later reach each other only through a third must have met once.""" - client = await target.open() + client = await target.open(observer=True) loop = asyncio.get_running_loop() deadline = loop.time() + timeout last = None diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py index 98755f256..6ea980919 100644 --- a/bindings/python/tests/verify/test_verify_commands.py +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -45,7 +45,6 @@ async def test_state_reports_the_address_the_carriers_and_the_session(harness): assert report["local_address"] == server_a.manager.local_address assert "WiFiDirect" in report["active_transports"] assert set(report) >= { - "topology", "pending_ack_count", "retry_queue_size", "mesh_relay_stats", @@ -92,6 +91,27 @@ async def test_send_prints_the_id_await_matches(harness): assert await commands.await_message(_target(server_b), message_id, "received", 15.0, arrival) == commands.PASS +async def test_state_and_pair_on_the_recipient_leave_a_held_message_for_await(harness): + """The service hands what it held for an application to that + application's first connection. Run on the recipient before the + arrival check, ``state`` and ``pair`` must not be it.""" + server_a, server_b = await two_servers(harness) + out = _output() + assert await commands.send(_target(server_a), server_b.manager.local_address, "held", out) == commands.PASS + message_id = _lines(out)[0]["sent"]["message_id"] + for _ in range(300): + if server_b.router.held_count(commands.DEFAULT_APP_ID): + break + await asyncio.sleep(0.05) + assert server_b.router.held_count(commands.DEFAULT_APP_ID) == 1 + + assert await commands.state(_target(server_b), [], _output()) == commands.PASS + assert await commands.pair(_target(server_b), server_a.manager.local_address, 0.2, _output()) == commands.TIMED_OUT + assert server_b.router.held_count(commands.DEFAULT_APP_ID) == 1 + arrival = _output() + assert await commands.await_message(_target(server_b), message_id, "received", 5.0, arrival) == commands.PASS + + async def test_await_times_out_with_status_two(harness): server = await harness.server(config=make_config(profile="alone")) out = _output() diff --git a/docs/local-api.md b/docs/local-api.md index 9aa61b62d..5c1b2e0e9 100644 --- a/docs/local-api.md +++ b/docs/local-api.md @@ -170,7 +170,7 @@ offline-protocol-verify await "$id" --until received | Command | Waits for | Passes on | |---|---|---| -| `state [--peer ADDR]` | nothing | always; prints address, carriers, neighbours, queues, relay counters, the session with each peer | +| `state [--peer ADDR]` | nothing | always; prints address, carriers, queues, relay counters, the session with each peer; it never takes a held message | | `send ADDR TEXT` | nothing | always; prints the message id | | `await ID --until delivered` | the receipt on the sender | `message_delivered` naming the id; `message_failed` is a failure, `message_deferred`, `message_retrying` and `message_undeliverable` print as status | | `await ID --until received` | the message on the recipient | `message_received` naming the id | @@ -185,7 +185,10 @@ away). Every verifier declares the application id `offline-protocol-verify` unless told otherwise with `--app-id`, and the sending and receiving devices must declare the same one: a received message is routed to the clients of the application its sender stamped, and held for that -application, 256 deep, while none is connected. +application, 256 deep, while none is connected. The first connection of the +application takes everything held, so on the recipient run `await --until +received` before `send`, `ping` or `watch`; `state` and `pair` only read, +and connect as `.observer` so they never take a held message. Two things about events decide how a check is written. A verifier matches events by the identifier they carry rather than relying on the server's @@ -203,7 +206,11 @@ because a message whose only route is the mesh has no pending acknowledgement to settle (an engine defect the scenario tests record as an expected failure; #537 fixes it). Until then, check a crossing on the recipient with `await --until received`, and on the middle device in the -`watch` log as `message_relayed`. +`watch` log as `message_relayed`. A second gap: only a send no carrier takes +is handed to the mesh. A message a direct link took and then lost (a stream +to a device that went away without closing it, until keepalive ends it in +about 30 seconds) is retried over direct carriers only, so it never crosses +the mesh; it waits for a direct link to the recipient. ## The policy file From 130099892bccdb1f72c88563a55de44b637dc526 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:18:15 +0530 Subject: [PATCH 03/16] feat(bindings): an offline-first demo on three devices, and OP_GATEWAY in the image --- CHANGELOG.md | 17 ++ bindings/python/docker/Dockerfile | 8 +- bindings/python/docker/README.md | 16 +- bindings/python/docker/config-ble.json | 19 ++ bindings/python/docker/config-relay.json | 19 ++ bindings/python/docker/entrypoint.sh | 7 +- examples/offline-first/.gitignore | 2 + examples/offline-first/README.md | 127 ++++++++ examples/offline-first/compose.yml | 61 ++++ examples/offline-first/run.py | 363 +++++++++++++++++++++++ scripts/tests/test-docker-entrypoint.sh | 12 + 11 files changed, 646 insertions(+), 5 deletions(-) create mode 100644 bindings/python/docker/config-ble.json create mode 100644 bindings/python/docker/config-relay.json create mode 100644 examples/offline-first/.gitignore create mode 100644 examples/offline-first/README.md create mode 100644 examples/offline-first/compose.yml create mode 100644 examples/offline-first/run.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 0484cbd4d..251d0e5e7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -102,6 +102,23 @@ archived by series under [docs/changelog/](docs/changelog/); see the sender and so has no pending acknowledgement to settle. The test records it as an expected failure until the engine fix in #537 lands. +- **An offline-first demo on three devices.** `examples/offline-first` runs + three services from the container image on two bridge networks, so the + middle one is the only way between the other two, and `run.py` runs the + scenarios with `offline-protocol-verify` and prints a table: store and + forward, the sender killed and restarted while its message is queued, and + a message carried through the middle device. Its README is the runbook for + the legs that need hardware (the LAN between hosts, the carrier changing + under a stream of messages, the relay, a gateway daemon, a phone), with a + record of what has been run: scenarios 1, 2 and 4 in containers on one + machine, nothing on hardware yet. The image gains `OP_GATEWAY` (the + gateway daemon, for a configuration with `reticulum_enabled`), two baked + configurations beside the default (`config-ble.json`, `config-relay.json`), + and the verifier, and its build fails when the installed package has none. + Found while running it: a message that a direct stream took and then lost + (a device that went away without closing the stream) is retried over + direct carriers only and never handed to the mesh. + ### Changed - **A peer-stream or relay flag the configuration cannot honour is refused.** diff --git a/bindings/python/docker/Dockerfile b/bindings/python/docker/Dockerfile index 502261677..defd7d692 100644 --- a/bindings/python/docker/Dockerfile +++ b/bindings/python/docker/Dockerfile @@ -27,9 +27,13 @@ RUN set -eu; \ pip install --no-cache-dir --find-links /wheels "${SDK_SPEC}"; \ rm -rf /wheels; \ { offline-protocol-service --help | grep -q -- '--http ' && python -c 'import aiohttp, zeroconf'; } \ - || { echo "the installed offline-protocol-service has no HTTP front: put a wheel built from a checkout in wheels/" >&2; exit 1; } + || { echo "the installed offline-protocol-service has no HTTP front: put a wheel built from a checkout in wheels/" >&2; exit 1; }; \ + command -v offline-protocol-verify >/dev/null \ + || { echo "the installed package has no offline-protocol-verify: put a wheel built from a checkout in wheels/" >&2; exit 1; } -COPY config.json /etc/offline-protocol/config.json +# The baked configuration, and two to mount or name with OP_CONFIG: one +# with Bluetooth LE on, one with the internet relay on. +COPY config.json config-ble.json config-relay.json /etc/offline-protocol/ COPY entrypoint.sh /usr/local/bin/offline-protocol-entrypoint RUN chmod 0755 /usr/local/bin/offline-protocol-entrypoint \ && mkdir -p /var/lib/offline-protocol diff --git a/bindings/python/docker/README.md b/bindings/python/docker/README.md index 5fa1960d6..577fbe95e 100644 --- a/bindings/python/docker/README.md +++ b/bindings/python/docker/README.md @@ -31,8 +31,19 @@ the service these, and the failure each one prevents: `config.json` is the engine's `ProtocolConfig`, baked into the image; mount another at `/etc/offline-protocol/config.json` or point `OP_CONFIG` at one. The default enables the peer stream with encryption required, and leaves -Bluetooth LE and the internet relay off. `profile` is this host's label and -part of its storage namespace: change it before the first start, never after. +Bluetooth LE and the internet relay off. Two more are baked beside it and +differ from it in one field each: `OP_CONFIG=/etc/offline-protocol/config-ble.json` +turns Bluetooth LE on (and needs the D-Bus grant above), and +`config-relay.json` turns the internet transport on for `OP_RELAY`. `profile` +is this host's label and part of its storage namespace: change it before the +first start, never after. + +The image also has `offline-protocol-verify`, which drives the service in +the same container and waits for the events that prove a message was held, +carried and delivered: `docker exec offline-protocol-verify +--socket /run/offline-protocol/api.sock state`. The +[offline-first demo](../../../examples/offline-first) runs its scenarios +with it. `entrypoint.sh` maps environment variables to flags: @@ -47,6 +58,7 @@ part of its storage namespace: change it before the first start, never after. | `OP_HTTP_TOKEN_FILE` | none | `--http-token-file`: the front writes a per-launch token there and requires it; needed off loopback, and on a host where a browser runs (R23 in the threat model). The file is replaced at every start and readable by the container's user only: bind-mount its directory, never the file, and read it as that user | | `OP_HTTP_ALIASES` | none | `--http-aliases` | | `OP_RELAY` | none | `--relay`, token in `OFFLINE_PROTOCOL_RELAY_TOKEN`; the config must set `internet_enabled`, which the default leaves off, or the service refuses to start | +| `OP_GATEWAY` | none | `--gateway`, a gateway daemon's `HOST:PORT`; the config must set `reticulum_enabled`, or the service refuses to start | | `OP_SOCKET` | `/run/offline-protocol/api.sock` | `--socket` | Arguments after the image name come after every flag the environment sets, diff --git a/bindings/python/docker/config-ble.json b/bindings/python/docker/config-ble.json new file mode 100644 index 000000000..70f3a1cdf --- /dev/null +++ b/bindings/python/docker/config-ble.json @@ -0,0 +1,19 @@ +{ + "app_id": "offline-protocol-service", + "profile": "device", + "ble_enabled": true, + "wifi_direct_enabled": true, + "internet_enabled": false, + "reticulum_enabled": false, + "nostr_enabled": false, + "prefer_online": false, + "initial_ttl": 5, + "encryption_enabled": true, + "require_encryption": true, + "auto_key_exchange": true, + "store_pending": true, + "max_pending_per_peer": 100, + "max_pending_global": 1000, + "pending_ttl_ms": 604800000, + "overflow_policy": "DropOldest" +} diff --git a/bindings/python/docker/config-relay.json b/bindings/python/docker/config-relay.json new file mode 100644 index 000000000..b7eec7a09 --- /dev/null +++ b/bindings/python/docker/config-relay.json @@ -0,0 +1,19 @@ +{ + "app_id": "offline-protocol-service", + "profile": "device", + "ble_enabled": false, + "wifi_direct_enabled": true, + "internet_enabled": true, + "reticulum_enabled": false, + "nostr_enabled": false, + "prefer_online": false, + "initial_ttl": 5, + "encryption_enabled": true, + "require_encryption": true, + "auto_key_exchange": true, + "store_pending": true, + "max_pending_per_peer": 100, + "max_pending_global": 1000, + "pending_ttl_ms": 604800000, + "overflow_policy": "DropOldest" +} diff --git a/bindings/python/docker/entrypoint.sh b/bindings/python/docker/entrypoint.sh index e2ce82772..74eb3d4c7 100755 --- a/bindings/python/docker/entrypoint.sh +++ b/bindings/python/docker/entrypoint.sh @@ -15,7 +15,9 @@ # OP_HTTP_ALIASES device alias file for the front # OP_RELAY internet relay URL, with a config that sets # internet_enabled; its token in OFFLINE_PROTOCOL_RELAY_TOKEN -# OP_SOCKET local API socket (default /run/offline-protocol/api.sock) +# OP_GATEWAY gateway daemon HOST:PORT, with a config that sets +# reticulum_enabled +# OP_SOCKET local API socket (default /run/offline-protocol/api.sock) # # Any arguments come after every flag the environment sets, so an explicit # flag wins over its variable. @@ -66,6 +68,9 @@ fi if [ -n "${OP_RELAY:-}" ]; then set -- "$@" --relay "$OP_RELAY" fi +if [ -n "${OP_GATEWAY:-}" ]; then + set -- "$@" --gateway "$OP_GATEWAY" +fi # The operator's arguments last: the service keeps the last occurrence of a # flag, so `docker run IMAGE --http 0.0.0.0:8080 ...` would otherwise lose to diff --git a/examples/offline-first/.gitignore b/examples/offline-first/.gitignore new file mode 100644 index 000000000..9966bfbc7 --- /dev/null +++ b/examples/offline-first/.gitignore @@ -0,0 +1,2 @@ +# The per-device store keys run.py writes: secret, and per checkout. +.env diff --git a/examples/offline-first/README.md b/examples/offline-first/README.md new file mode 100644 index 000000000..80a418048 --- /dev/null +++ b/examples/offline-first/README.md @@ -0,0 +1,127 @@ +# Offline first: what the network does when a device is not there + +A request with a deadline needs both devices on at once. A message does +not: the sender holds it on disk until the recipient can be reached, by +whatever carrier reaches it, through whichever devices are in between, and +the recipient's own acknowledgement comes back as `message_delivered`. These +scenarios show that, and each one passes or fails on the engine's events, +read on the device by +[`offline-protocol-verify`](../../docs/local-api.md#checking-what-the-network-did-offline-protocol-verify). + +| # | Scenario | Passes when | +|---|---|---| +| 1 | Store and forward: B off, A sends, B on | B receives it, and A gets `message_delivered` within 60 s of B coming back | +| 2 | The sender restarts: B off, A sends, A killed (`SIGKILL`) and started, B on | the same, from the restarted A: the queue was on disk | +| 3 | The carrier changes: the link A and B were using goes away | every message of a `ping` run is delivered, and `message_delivered.transport` names the new carrier (hardware only, below) | +| 4 | Through the middle: A and C meet once, then only B hears both | C receives A's message at `hop_count` 1 within 30 s, and B reports `message_relayed` | + +## In containers, on one machine + +``` +a ---- net-ab ---- b ---- net-bc ---- c +``` + +[`compose.yml`](compose.yml) runs three devices from the +[service image](../../bindings/python/docker), each with its own identity +on its own volume. a and c share no network, so b is the only way between +them. Peers are static, so the topology is the file's and not multicast +DNS's. + +```bash +python3 run.py # builds the image, starts the devices, runs 1, 2 and 4 +python3 run.py --fresh # new identities first +python3 run.py --scenario 4 --require-hop-receipt +``` + +The image installs the package from PyPI unless a Linux wheel built from a +checkout is in `bindings/python/docker/wheels/`, and its build fails when +the installed package has no `offline-protocol-verify`, which is true of +every release so far: until one ships, put a wheel there, or build the image +another way and pass `--no-build` to use `offline-protocol-service:offline-first` +as it is. `run.py` writes `.env` with one store key per device the first +time; keep it, or the devices come back with new addresses. It prints each +scenario as it finishes and a table at the end, and exits non-zero if any +failed. + +What `run.py` does, so each step can be run by hand with `docker exec +offline-first- offline-protocol-verify --socket /run/offline-protocol/api.sock ...`: + +1. `pair` A and B (the session forms by itself once they hear each other), + `docker stop` B, `send` from A, start `await --until delivered` on + A, `docker start` B, then `await --until received` on B. +2. The same with `docker kill --signal KILL` and `docker start` on A between + the send and B's return. +4. `docker network connect` C to net-ab and `pair` A and C; `docker stop` C, + disconnect it, `docker start` it; then `watch` on A and B, `send` from A + to C, and `await --until received` on C. + +Two things in that order matter, and both are the engine's rules rather +than the script's. A receipt that fires while no client is connected is not +held, so the wait on the sender starts before the recipient can answer. And +C is stopped before it leaves net-ab: taken off the network first, it would +leave A a stream that looks open for about 30 s, and a message A sent down +it would never be handed to B (see the known gaps). + +## On hardware + +`run.py --ssh a=user@host --ssh b=user@host --ssh c=user@host` runs the same +checks over ssh against hosts that already run the service, and asks the +operator to switch devices off and on, and to move a and c apart, at each +step. Each host runs the [service image](../../bindings/python/docker) under +host networking with `OP_LISTEN` set to its own LAN address, or the service +directly. + +The legs that need real radios or more than one machine, and how to run +each: + +- **LAN between hosts.** Two or three boxes on one segment, `OP_LAN=1` so + they find each other over DNS-SD. Scenarios 1, 2, 4 as above; for 4, put + C on a second segment that only B joins. +- **The carrier changes (3).** Two boxes with `OP_CONFIG=/etc/offline-protocol/config-ble.json` + and the D-Bus grant, so both peer streams and Bluetooth LE are up. Start + `offline-protocol-verify ping --every 2 --count 60` on A, then pull + the cable (or `ip link set down`) on B. Expect the messages in + flight to arrive about 30 to 90 s late over `ble` (keepalive ends the dead + stream in about 30 s, then the next retry picks the carrier that is up), + and later ones over `ble` at once. Plug it back in to see `wifiDirect` + return. +- **Relay.** The relay server with a Postgres database and authentication + off for a closed test, `config-relay.json` and `OP_RELAY=ws://:3000/ws` + on each box. Scenario 1 twice: once with A also off when B returns (the + relay's mailbox holds the frame), once with the relay unreachable from A + (A's outbox holds it). `message_undeliverable` is printed as status on the + way: it is the relay saying B is away now, not a failure. +- **Reticulum.** A gateway daemon and `rnsd` on each of two boxes with a + backbone between them (a TCP interface, or a pair of RNodes), a config + with `reticulum_enabled` and `OP_GATEWAY=127.0.0.1:4242`. Check the attach + first (the transport comes up only after the daemon's capabilities), then + scenario 1. The service's gateway client has not met a real daemon yet. +- **A phone.** The React Native example app with Bluetooth LE on, against a + box running `config-ble.json`: the box's `await --until received`, and + the app's own delivered state, both directions. An iPhone can also reach + a box over the LAN peer stream on the same Wi-Fi. + +## Known gaps + +- **The sender's receipt does not come back across a hop.** In scenario 4, + C's acknowledgement is carried back by B and dropped by A: a message whose + only route is the mesh has no pending acknowledgement on A to settle. + `run.py` reports it and passes unless `--require-hop-receipt`. The engine + fix is #537. +- **A message a direct link took and then lost never crosses the mesh.** + Only a send that no carrier takes is handed to neighbours; a retry that + the carriers refuse is queued again for direct carriers only. So a message + sent down a stream to a device that went away without closing it waits + for a direct link to that device. +- **Bluetooth LE on a Linux box as the peripheral** does not learn which + phone wrote to it, so replies to a phone go over the box's central role. + One Bluetooth LE peer per box until that is fixed. + +## What has been run + +| When | Where | Scenarios | Result | +|---|---|---|---| +| 2026-10-08 | `run.py`, Docker 28.3 on one arm64 laptop, the image built from the 0.28.0 Linux wheel with this branch's Python sources (no native change since 0.28.0) | 1, 2, 4 | pass, twice in a row (receipt latency 29 to 45 ms; 4 without A's receipt, as above) | + +Nothing on this page has been run between separate hosts, over Bluetooth +LE, over a relay, through a gateway daemon, or with a phone. diff --git a/examples/offline-first/compose.yml b/examples/offline-first/compose.yml new file mode 100644 index 000000000..fe47810d8 --- /dev/null +++ b/examples/offline-first/compose.yml @@ -0,0 +1,61 @@ +# Three devices on one machine, for the offline-first scenarios in README.md. +# +# a ---- net-ab ---- b ---- net-bc ---- c +# +# a and c share no network, so b is the only way between them; Docker keeps +# separate bridge networks apart. Each device is the service image from +# bindings/python/docker with its own identity on its own volume. No HTTP +# front: the scenarios use the local API through offline-protocol-verify. +# +# Peers are static (a and c dial b, c also dials a when the scenario puts it +# on net-ab), so the topology is decided here and not by multicast DNS. +# +# run.py writes .env with one store key per device the first time; keep it, +# or the devices come back with new addresses. +x-device: &device + build: ../../bindings/python/docker + image: offline-protocol-service:offline-first + environment: &environment + OP_LAN: "0" + OP_HTTP: "" + OP_LISTEN: 0.0.0.0:7878 + +services: + a: + <<: *device + container_name: offline-first-a + hostname: a + environment: + <<: *environment + OFFLINE_PROTOCOL_STORE_KEY: ${KEY_A:?run run.py once, or set KEY_A} + OP_PEERS: b:7878 + networks: [net-ab] + volumes: [data-a:/var/lib/offline-protocol] + b: + <<: *device + container_name: offline-first-b + hostname: b + environment: + <<: *environment + OFFLINE_PROTOCOL_STORE_KEY: ${KEY_B:?run run.py once, or set KEY_B} + networks: [net-ab, net-bc] + volumes: [data-b:/var/lib/offline-protocol] + c: + <<: *device + container_name: offline-first-c + hostname: c + environment: + <<: *environment + OFFLINE_PROTOCOL_STORE_KEY: ${KEY_C:?run run.py once, or set KEY_C} + OP_PEERS: b:7878 a:7878 + networks: [net-bc] + volumes: [data-c:/var/lib/offline-protocol] + +networks: + net-ab: + net-bc: + +volumes: + data-a: + data-b: + data-c: diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py new file mode 100644 index 000000000..0f4994d00 --- /dev/null +++ b/examples/offline-first/run.py @@ -0,0 +1,363 @@ +#!/usr/bin/env python3 +"""Runs the offline-first scenarios against three devices and prints a table. + +Each check is offline-protocol-verify on the device itself (docs/local-api.md), +so a scenario passes on the engine's own events, not on a log read by eye: + + 1 store and forward B off, A sends, B on: B receives it, A gets B's receipt + 2 sender restarts B off, A sends, A killed and restarted, B on: same + 4 through the middle A and C meet once, then share no network: C receives + A's message one hop away and B reports carrying it + +With --docker (the default) the devices are the containers of compose.yml in +this directory, and switching a device off is `docker stop`. With --ssh each +device is a host running the service, and the operator switches it off and on +when asked. Standard library only. + +Usage: + python3 run.py # build, start, run every scenario + python3 run.py --fresh # new identities first (compose down -v) + python3 run.py --scenario 1 --scenario 2 + python3 run.py --ssh a=pi@10.0.0.11 --ssh b=pi@10.0.0.12 --ssh c=pi@10.0.0.13 +""" + +from __future__ import annotations + +import argparse +import json +import secrets +import shlex +import subprocess +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path + +HERE = Path(__file__).resolve().parent +SOCKET = "/run/offline-protocol/api.sock" +PROJECT = "offline-first" + +#: Pass criteria, from the moment the recipient is switched back on (1 and 2) +#: or from the send (4). Loopback-fast in the lab; the margin is for hardware +#: and for a peer-stream redial ladder that may be part-way up. +DELIVERY_AFTER_RETURN_S = 60 +HOP_RECEIVED_S = 30 +#: How long to wait for the sender's receipt across the hop. It does not come +#: back on an engine without #537 (docs/local-api.md), so by default its +#: absence is reported and does not fail the run. +HOP_RECEIPT_S = 20 +#: Session formation after devices first hear each other, and a restarted +#: service answering on its socket. +PAIR_S = 90 +READY_S = 60 + + +class ScenarioFailed(Exception): + pass + + +@dataclass +class Result: + scenario: str + passed: bool + elapsed_s: float + detail: str + + +@dataclass +class Device: + name: str + runner: "Runner" + address: str = "" + + def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str]: + process = subprocess.run( + self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + capture_output=True, + text=True, + timeout=timeout, + ) + return process.returncode, _json_lines(process.stdout), process.stderr.strip() + + def verify_in_background(self, *args: str) -> subprocess.Popen[str]: + return subprocess.Popen( + self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + + def wait_ready(self) -> None: + deadline = time.monotonic() + READY_S + while True: + status, lines, err = self.verify("state", timeout=30) + if status == 0: + self.address = lines[-1]["state"]["local_address"] + return + if time.monotonic() > deadline: + raise ScenarioFailed(f"{self.name} did not answer on its socket: {err}") + time.sleep(1) + + def send(self, recipient: "Device", text: str) -> str: + status, lines, err = self.verify("send", recipient.address, text) + if status != 0: + raise ScenarioFailed(f"{self.name} could not send: {err}") + return lines[-1]["sent"]["message_id"] + + +def _json_lines(text: str) -> list[dict]: + lines = [] + for line in text.splitlines(): + try: + lines.append(json.loads(line)) + except ValueError: + continue + return lines + + +def _saw(watched: list[dict], tag: str, message_id: str) -> dict | None: + for line in watched: + event = line.get("event", {}) + if event.get("type") == tag and event.get("message_id") == message_id: + return event + return None + + +def _finish(process: subprocess.Popen[str], timeout: float) -> tuple[int, list[dict], str]: + try: + out, err = process.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + process.kill() + out, err = process.communicate() + return 2, _json_lines(out), err.strip() + return process.returncode, _json_lines(out), err.strip() + + +class Runner: + """How to reach a device and how to switch it off and on.""" + + def command(self, device: str, argv: list[str]) -> list[str]: + raise NotImplementedError + + def setup(self, fresh: bool) -> None: + pass + + def off(self, device: str) -> None: + raise NotImplementedError + + def on(self, device: str) -> None: + raise NotImplementedError + + def kill(self, device: str) -> None: + raise NotImplementedError + + def join_a_and_c(self) -> None: + raise NotImplementedError + + def part_a_and_c(self) -> None: + raise NotImplementedError + + +class DockerRunner(Runner): + def _compose(self, *args: str) -> None: + subprocess.run(["docker", "compose", "-p", PROJECT, *args], cwd=HERE, check=True) + + def _docker(self, *args: str) -> None: + subprocess.run(["docker", *args], check=True, stdout=subprocess.DEVNULL) + + def command(self, device: str, argv: list[str]) -> list[str]: + return ["docker", "exec", f"{PROJECT}-{device}", *argv] + + def __init__(self, build: bool) -> None: + self.build = build + + def setup(self, fresh: bool) -> None: + env = HERE / ".env" + if fresh: + self._compose("down", "-v") + if not env.exists(): + # One key per device, kept: the identity and every queued message + # are sealed under it. + env.write_text("".join(f"KEY_{d.upper()}={secrets.token_hex(32)}\n" for d in "abc")) + env.chmod(0o600) + self._compose("up", "-d", "--build" if self.build else "--no-build") + + def off(self, device: str) -> None: + self._docker("stop", f"{PROJECT}-{device}") + + def on(self, device: str) -> None: + self._docker("start", f"{PROJECT}-{device}") + + def kill(self, device: str) -> None: + self._docker("kill", "--signal", "KILL", f"{PROJECT}-{device}") + + def join_a_and_c(self) -> None: + self._docker("network", "connect", f"{PROJECT}_net-ab", f"{PROJECT}-c") + + def part_a_and_c(self) -> None: + # Stopped while still on net-ab, so A sees the stream close. Taken off + # the network first, C would leave A a stream that looks open until + # keepalive ends it (about 30 s), and a message A sends in that window + # goes down it, is retried over direct carriers only, and never + # reaches the mesh (docs/local-api.md). + self._docker("stop", f"{PROJECT}-c") + self._docker("network", "disconnect", f"{PROJECT}_net-ab", f"{PROJECT}-c") + self._docker("start", f"{PROJECT}-c") + + +class SshRunner(Runner): + """Hosts that run the service already; power and links are the operator's.""" + + def __init__(self, hosts: dict[str, str]) -> None: + self.hosts = hosts + + def command(self, device: str, argv: list[str]) -> list[str]: + return ["ssh", self.hosts[device], shlex.join(argv)] + + def _ask(self, text: str) -> None: + input(f"\n>>> {text}, then press Enter: ") + + def off(self, device: str) -> None: + self._ask(f"switch {device} ({self.hosts[device]}) off") + + def on(self, device: str) -> None: + self._ask(f"switch {device} ({self.hosts[device]}) on") + + def kill(self, device: str) -> None: + self._ask(f"cut {device}'s power, or kill -9 its service") + + def join_a_and_c(self) -> None: + self._ask("put a and c where they hear each other directly") + + def part_a_and_c(self) -> None: + self._ask("move a and c apart so only b hears both (restart c's service to drop the old link)") + + +@dataclass +class Lab: + a: Device + b: Device + c: Device + runner: Runner + results: list[Result] = field(default_factory=list) + + def pair(self, x: Device, y: Device) -> None: + status, _, err = x.verify("pair", y.address, "--timeout", str(PAIR_S)) + if status != 0: + raise ScenarioFailed(f"no session between {x.name} and {y.name}: {err}") + + def store_and_forward(self, restart_sender: bool) -> Result: + name = "2 sender restarts" if restart_sender else "1 store and forward" + self.pair(self.a, self.b) + self.runner.off("b") + message_id = self.a.send(self.b, f"{name} {time.strftime('%H:%M:%S')}") + if restart_sender: + self.runner.kill("a") + self.runner.on("a") + self.a.wait_ready() + # The receipt is not held for a client that is not connected: the + # wait starts before B can answer. + receipt = self.a.verify_in_background("await", message_id, "--until", "delivered", + "--timeout", str(DELIVERY_AFTER_RETURN_S + READY_S)) + time.sleep(2) + started = time.monotonic() + self.runner.on("b") + status, lines, err = _finish(receipt, DELIVERY_AFTER_RETURN_S + READY_S + 30) + elapsed = time.monotonic() - started + if status != 0 or elapsed > DELIVERY_AFTER_RETURN_S: + return Result(name, False, elapsed, f"no receipt within {DELIVERY_AFTER_RETURN_S} s: {err}") + event = lines[-1]["event"] + self.b.wait_ready() + status, _, err = self.b.verify("await", message_id, "--until", "received", "--timeout", "30") + if status != 0: + return Result(name, False, elapsed, f"receipt came back but B has no message: {err}") + return Result(name, True, elapsed, f"receipt over {event['transport']}, latency {event['latency_ms']} ms") + + def through_the_middle(self, require_receipt: bool) -> Result: + name = "4 through the middle" + self.runner.join_a_and_c() + self.c.wait_ready() + self.pair(self.a, self.c) + self.runner.part_a_and_c() + self.c.wait_ready() + self.pair(self.b, self.c) + self.pair(self.a, self.b) + + # Both watches start before the send: the receipt on A, and B's + # report of carrying the frame, are not held for a late client. + window = str(HOP_RECEIVED_S + HOP_RECEIPT_S) + middle = self.b.verify_in_background("watch", "--duration", window) + sender = self.a.verify_in_background("watch", "--duration", window) + time.sleep(2) + started = time.monotonic() + message_id = self.a.send(self.c, f"{name} {time.strftime('%H:%M:%S')}") + status, lines, err = self.c.verify("await", message_id, "--until", "received", + "--timeout", str(HOP_RECEIVED_S)) + elapsed = time.monotonic() - started + _, watched_b, _ = _finish(middle, float(window) + 30) + _, watched_a, _ = _finish(sender, float(window) + 30) + if status != 0: + return Result(name, False, elapsed, f"C did not receive it within {HOP_RECEIVED_S} s: {err}") + hops = lines[-1]["event"]["hop_count"] + relayed = _saw(watched_b, "message_relayed", message_id) + if hops != 1 or relayed is None: + return Result(name, False, elapsed, f"received at hop {hops}, B relayed it: {relayed is not None}") + detail = "C received it at hop 1, B relayed it" + receipt = _saw(watched_a, "message_delivered", message_id) + if receipt is not None: + return Result(name, True, elapsed, detail + f"; A's receipt at hop {receipt['hop_count']}") + if require_receipt: + return Result(name, False, elapsed, detail + "; A's receipt never came back") + return Result(name, True, elapsed, detail + "; A's receipt did not come back (known, #537)") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--ssh", action="append", default=[], metavar="NAME=HOST", + help="a device as an ssh destination (a, b and c); without it, Docker") + parser.add_argument("--fresh", action="store_true", help="Docker: remove the devices and their identities first") + parser.add_argument("--no-build", action="store_true", + help="Docker: use the image offline-protocol-service:offline-first as it is") + parser.add_argument("--scenario", action="append", type=int, choices=(1, 2, 4), help="run only these") + parser.add_argument("--require-hop-receipt", action="store_true", + help="fail scenario 4 when A's receipt does not come back") + args = parser.parse_args(argv) + + if args.ssh: + hosts = dict(item.split("=", 1) for item in args.ssh) + if set(hosts) != {"a", "b", "c"}: + parser.error("--ssh needs a=..., b=... and c=...") + runner: Runner = SshRunner(hosts) + else: + runner = DockerRunner(build=not args.no_build) + runner.setup(args.fresh) + + lab = Lab(*(Device(name, runner) for name in "abc"), runner=runner) + for device in (lab.a, lab.b, lab.c): + device.wait_ready() + print(f"{device.name}: {device.address}", file=sys.stderr) + + wanted = args.scenario or [1, 2, 4] + for number, run in ( + (1, lambda: lab.store_and_forward(restart_sender=False)), + (2, lambda: lab.store_and_forward(restart_sender=True)), + (4, lambda: lab.through_the_middle(args.require_hop_receipt)), + ): + if number not in wanted: + continue + try: + result = run() + except (ScenarioFailed, subprocess.SubprocessError) as exc: + result = Result(str(number), False, 0.0, str(exc)) + lab.results.append(result) + print(f"{'PASS' if result.passed else 'FAIL'} {result.scenario}: {result.detail}", file=sys.stderr) + + width = max(len(r.scenario) for r in lab.results) + print(f"\n{'scenario':<{width}} result seconds detail") + for r in lab.results: + print(f"{r.scenario:<{width}} {'pass' if r.passed else 'FAIL':<6} {r.elapsed_s:7.1f} {r.detail}") + return 0 if all(r.passed for r in lab.results) else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test-docker-entrypoint.sh b/scripts/tests/test-docker-entrypoint.sh index 13552a1d8..df46b7036 100755 --- a/scripts/tests/test-docker-entrypoint.sh +++ b/scripts/tests/test-docker-entrypoint.sh @@ -129,6 +129,18 @@ run "the front's flags follow OP_HTTP" "$BASE wss://relay.example" OP_HTTP=0.0.0.0:8080 OP_HTTP_TOKEN_FILE=/run/front/token \ OP_HTTP_ALIASES=/etc/aliases.json OP_RELAY=wss://relay.example +run "OP_GATEWAY names the gateway daemon, before an operator flag" "$BASE +--lan +--http +127.0.0.1:8080 +--gateway +10.0.0.5:4242 +--gateway +127.0.0.1:4242" OP_GATEWAY=10.0.0.5:4242 -- --gateway 127.0.0.1:4242 + +run "an empty OP_GATEWAY names none" "$BASE +--lan" OP_HTTP= OP_GATEWAY= + run "OP_LAN other than 1 or 0 is refused" REFUSED OP_LAN=2 run "an empty OP_LAN is refused" REFUSED OP_LAN= run "a front flag without a front is refused" REFUSED OP_HTTP= OP_HTTP_TOKEN_FILE=/t From c6c0293badf63e0b5bd484b0d84fd621c696ddf6 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:50:59 +0530 Subject: [PATCH 04/16] docs(bindings): realign OP_SOCKET in the entrypoint's variable list Adding OP_GATEWAY to the comment block that lists the entrypoint's variables took a space from the OP_SOCKET line, so its description no longer lined up with every other entry. Restore it. --- bindings/python/docker/entrypoint.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/bindings/python/docker/entrypoint.sh b/bindings/python/docker/entrypoint.sh index 74eb3d4c7..341eefd80 100755 --- a/bindings/python/docker/entrypoint.sh +++ b/bindings/python/docker/entrypoint.sh @@ -17,7 +17,7 @@ # internet_enabled; its token in OFFLINE_PROTOCOL_RELAY_TOKEN # OP_GATEWAY gateway daemon HOST:PORT, with a config that sets # reticulum_enabled -# OP_SOCKET local API socket (default /run/offline-protocol/api.sock) +# OP_SOCKET local API socket (default /run/offline-protocol/api.sock) # # Any arguments come after every flag the environment sets, so an explicit # flag wins over its variable. From 857d8724a638bfc59b19d34a1b09deb4dc039a8a Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:51:11 +0530 Subject: [PATCH 05/16] fix(bindings): keep the offline-first verifiers off the terminal Under --ssh, run.py starts `await --until delivered` on A in the background and then prompts the operator to switch B back on. The ssh client inherited the terminal's stdin and forwards whatever it reads to the remote command, so it competed with input() for the operator's Enter: the keypress could be swallowed by the remote side, leaving the prompt waiting while the receipt window ran out. Run every verifier, foreground and background, with stdin from /dev/null. Nothing the verifier does reads it. --- examples/offline-first/run.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py index 0f4994d00..40675a1e2 100644 --- a/examples/offline-first/run.py +++ b/examples/offline-first/run.py @@ -70,9 +70,14 @@ class Device: runner: "Runner" address: str = "" + # Every verifier runs with stdin closed. Under --ssh the operator answers + # prompts on the terminal while an `await` is in flight, and an ssh + # client that inherits the terminal forwards what it reads to the remote + # side: the Enter that says B is back on would never reach input(). def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str]: process = subprocess.run( self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=timeout, @@ -82,6 +87,7 @@ def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str def verify_in_background(self, *args: str) -> subprocess.Popen[str]: return subprocess.Popen( self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, From 35920c642b6cdb918c9f290fdef6907943e2f880 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:51:43 +0530 Subject: [PATCH 06/16] fix(bindings): let run.py --ssh reach a service inside the image The hardware runbook has each host run the service image under host networking, but run.py --ssh ran `offline-protocol-verify --socket /run/offline-protocol/api.sock` on the host itself. The local API socket is inside the container and the host has no verifier, so every command failed and no scenario could start. A host running the service directly fared no better: its default socket is under XDG_RUNTIME_DIR or ~/.offline-protocol, never the image's path. Add --remote-exec, a command the verifier runs through on each host (`docker exec ` for the image), and --socket for a service started with another path. The README's hardware section says which to use. --- examples/offline-first/README.md | 6 +++++- examples/offline-first/run.py | 31 ++++++++++++++++++++++++------- 2 files changed, 29 insertions(+), 8 deletions(-) diff --git a/examples/offline-first/README.md b/examples/offline-first/README.md index 80a418048..ca6f2cbf8 100644 --- a/examples/offline-first/README.md +++ b/examples/offline-first/README.md @@ -69,7 +69,11 @@ checks over ssh against hosts that already run the service, and asks the operator to switch devices off and on, and to move a and c apart, at each step. Each host runs the [service image](../../bindings/python/docker) under host networking with `OP_LISTEN` set to its own LAN address, or the service -directly. +directly. The verifier has to run where the service's socket is: for the +image that is inside the container, so add `--remote-exec "docker exec +"` (`docker-offline-protocol-1` for the image's `compose.yml` +started from its own directory); for the service run directly, add +`--socket` with the path it was started with. The legs that need real radios or more than one machine, and how to run each: diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py index 40675a1e2..bade72672 100644 --- a/examples/offline-first/run.py +++ b/examples/offline-first/run.py @@ -18,7 +18,8 @@ python3 run.py # build, start, run every scenario python3 run.py --fresh # new identities first (compose down -v) python3 run.py --scenario 1 --scenario 2 - python3 run.py --ssh a=pi@10.0.0.11 --ssh b=pi@10.0.0.12 --ssh c=pi@10.0.0.13 + python3 run.py --ssh a=pi@10.0.0.11 --ssh b=pi@10.0.0.12 --ssh c=pi@10.0.0.13 \\ + --remote-exec "docker exec docker-offline-protocol-1" # hosts run the image """ from __future__ import annotations @@ -76,7 +77,7 @@ class Device: # side: the Enter that says B is back on would never reach input(). def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str]: process = subprocess.run( - self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + self.runner.command(self.name, ["offline-protocol-verify", "--socket", self.runner.socket, *args]), stdin=subprocess.DEVNULL, capture_output=True, text=True, @@ -86,7 +87,7 @@ def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str def verify_in_background(self, *args: str) -> subprocess.Popen[str]: return subprocess.Popen( - self.runner.command(self.name, ["offline-protocol-verify", "--socket", SOCKET, *args]), + self.runner.command(self.name, ["offline-protocol-verify", "--socket", self.runner.socket, *args]), stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, @@ -142,6 +143,9 @@ def _finish(process: subprocess.Popen[str], timeout: float) -> tuple[int, list[d class Runner: """How to reach a device and how to switch it off and on.""" + #: The local API socket, as the verifier sees it where it runs. + socket = SOCKET + def command(self, device: str, argv: list[str]) -> list[str]: raise NotImplementedError @@ -212,13 +216,21 @@ def part_a_and_c(self) -> None: class SshRunner(Runner): - """Hosts that run the service already; power and links are the operator's.""" + """Hosts that run the service already; power and links are the operator's. + + The verifier must run where the socket is: on a host that runs the + service image, that is inside the container (``prefix`` is then + ``docker exec ``), since the socket is in the container and + the host has no verifier; on a host that runs the service directly, on + the host, with ``socket`` naming the service's ``--socket``.""" - def __init__(self, hosts: dict[str, str]) -> None: + def __init__(self, hosts: dict[str, str], prefix: list[str], socket: str) -> None: self.hosts = hosts + self.prefix = prefix + self.socket = socket def command(self, device: str, argv: list[str]) -> list[str]: - return ["ssh", self.hosts[device], shlex.join(argv)] + return ["ssh", self.hosts[device], shlex.join([*self.prefix, *argv])] def _ask(self, text: str) -> None: input(f"\n>>> {text}, then press Enter: ") @@ -321,6 +333,11 @@ def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--ssh", action="append", default=[], metavar="NAME=HOST", help="a device as an ssh destination (a, b and c); without it, Docker") + parser.add_argument("--remote-exec", default="", metavar="COMMAND", + help="ssh: run the verifier through this on each host, e.g. 'docker exec CONTAINER' " + "when the host runs the service image") + parser.add_argument("--socket", default=SOCKET, + help=f"ssh: the service's local API socket where the verifier runs (default: {SOCKET})") parser.add_argument("--fresh", action="store_true", help="Docker: remove the devices and their identities first") parser.add_argument("--no-build", action="store_true", help="Docker: use the image offline-protocol-service:offline-first as it is") @@ -333,7 +350,7 @@ def main(argv: list[str] | None = None) -> int: hosts = dict(item.split("=", 1) for item in args.ssh) if set(hosts) != {"a", "b", "c"}: parser.error("--ssh needs a=..., b=... and c=...") - runner: Runner = SshRunner(hosts) + runner: Runner = SshRunner(hosts, shlex.split(args.remote_exec), args.socket) else: runner = DockerRunner(build=not args.no_build) runner.setup(args.fresh) From d82f22affcc6902a936ee4d50af4a70fb351331f Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:54:53 +0530 Subject: [PATCH 07/16] fix(bindings): stop one failed offline-first scenario failing the rest Two ways a scenario that stopped part-way poisoned what ran after it: - Scenario 1 or 2 raising between switching B off and on (a refused send, a socket that never answered) left B stopped. Every later scenario then began with `pair` toward B and failed after 90 s on a device that was simply off, which reads as a networking failure. - Scenario 4 stopping between joining C to net-ab and parting it left C on net-ab, and `compose up` does not take it off. Every later run then failed scenario 4 at once: `docker network connect` refuses a container that is already on the network. After a scenario fails, switch every device back on (Docker: `docker start` all three, a no-op for a running one; ssh: ask the operator) and wait for each socket before the next. Join C to net-ab only when it is not already there. A failure raised by a scenario now carries its name in the table instead of its bare number, and the docstring no longer names a --docker flag the script never had. Checked against the lab: with C left on net-ab, scenario 4 failed before and passes now; with scenario 1 failing while B is off, scenario 2 passes straight after. --- examples/offline-first/run.py | 40 +++++++++++++++++++++++++++++------ 1 file changed, 33 insertions(+), 7 deletions(-) diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py index bade72672..89c04457f 100644 --- a/examples/offline-first/run.py +++ b/examples/offline-first/run.py @@ -9,7 +9,7 @@ 4 through the middle A and C meet once, then share no network: C receives A's message one hop away and B reports carrying it -With --docker (the default) the devices are the containers of compose.yml in +Without --ssh the devices are the containers of compose.yml in this directory, and switching a device off is `docker stop`. With --ssh each device is a host running the service, and the operator switches it off and on when asked. Standard library only. @@ -167,6 +167,11 @@ def join_a_and_c(self) -> None: def part_a_and_c(self) -> None: raise NotImplementedError + def restore(self) -> None: + """Every device on again after a scenario that failed part-way, so + the next one does not fail for a device the last one left off.""" + raise NotImplementedError + class DockerRunner(Runner): def _compose(self, *args: str) -> None: @@ -202,7 +207,14 @@ def kill(self, device: str) -> None: self._docker("kill", "--signal", "KILL", f"{PROJECT}-{device}") def join_a_and_c(self) -> None: - self._docker("network", "connect", f"{PROJECT}_net-ab", f"{PROJECT}-c") + # A run that stopped between this and part_a_and_c leaves C on net-ab, + # where `compose up` does not take it off; connecting it again fails. + networks = subprocess.run( + ["docker", "inspect", "--format", "{{json .NetworkSettings.Networks}}", f"{PROJECT}-c"], + check=True, capture_output=True, text=True, + ).stdout + if f"{PROJECT}_net-ab" not in json.loads(networks): + self._docker("network", "connect", f"{PROJECT}_net-ab", f"{PROJECT}-c") def part_a_and_c(self) -> None: # Stopped while still on net-ab, so A sees the stream close. Taken off @@ -214,6 +226,10 @@ def part_a_and_c(self) -> None: self._docker("network", "disconnect", f"{PROJECT}_net-ab", f"{PROJECT}-c") self._docker("start", f"{PROJECT}-c") + def restore(self) -> None: + # `docker start` leaves a running container as it is. + self._docker("start", *(f"{PROJECT}-{d}" for d in "abc")) + class SshRunner(Runner): """Hosts that run the service already; power and links are the operator's. @@ -250,6 +266,9 @@ def join_a_and_c(self) -> None: def part_a_and_c(self) -> None: self._ask("move a and c apart so only b hears both (restart c's service to drop the old link)") + def restore(self) -> None: + self._ask("make sure a, b and c are all on") + @dataclass class Lab: @@ -361,17 +380,24 @@ def main(argv: list[str] | None = None) -> int: print(f"{device.name}: {device.address}", file=sys.stderr) wanted = args.scenario or [1, 2, 4] - for number, run in ( - (1, lambda: lab.store_and_forward(restart_sender=False)), - (2, lambda: lab.store_and_forward(restart_sender=True)), - (4, lambda: lab.through_the_middle(args.require_hop_receipt)), + for number, name, run in ( + (1, "1 store and forward", lambda: lab.store_and_forward(restart_sender=False)), + (2, "2 sender restarts", lambda: lab.store_and_forward(restart_sender=True)), + (4, "4 through the middle", lambda: lab.through_the_middle(args.require_hop_receipt)), ): if number not in wanted: continue try: result = run() except (ScenarioFailed, subprocess.SubprocessError) as exc: - result = Result(str(number), False, 0.0, str(exc)) + result = Result(name, False, 0.0, str(exc)) + if not result.passed: + try: + runner.restore() + for device in (lab.a, lab.b, lab.c): + device.wait_ready() + except (ScenarioFailed, subprocess.SubprocessError) as exc: + print(f"could not switch every device back on: {exc}", file=sys.stderr) lab.results.append(result) print(f"{'PASS' if result.passed else 'FAIL'} {result.scenario}: {result.detail}", file=sys.stderr) From 9921e15ae143a58c9a708149645a128213c0a043 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:55:30 +0530 Subject: [PATCH 08/16] docs(bindings): keep the peer stream out of the relay and gateway legs The offline-first runbook had the relay leg run scenario 1 with config-relay.json, which keeps wifi_direct_enabled on, on boxes the other legs put on one LAN. The engine tries a live direct mesh link to the recipient before transport selection (send.rs, the direct_mesh branch), so once B is back on the LAN the outbox retry goes down the peer stream, the receipt names wifiDirect, and the leg passes without the relay having carried anything. The gateway leg has the same hole. Say that both legs need no peer stream between the boxes (OP_LAN=0 and no OP_PEERS, or two networks), and that the receipt names internet when the relay carried it. --- examples/offline-first/README.md | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/examples/offline-first/README.md b/examples/offline-first/README.md index ca6f2cbf8..61ec6dc4a 100644 --- a/examples/offline-first/README.md +++ b/examples/offline-first/README.md @@ -91,13 +91,19 @@ each: return. - **Relay.** The relay server with a Postgres database and authentication off for a closed test, `config-relay.json` and `OP_RELAY=ws://:3000/ws` - on each box. Scenario 1 twice: once with A also off when B returns (the - relay's mailbox holds the frame), once with the relay unreachable from A - (A's outbox holds it). `message_undeliverable` is printed as status on the - way: it is the relay saying B is away now, not a failure. + on each box, and no peer stream between them (`OP_LAN=0` and no + `OP_PEERS`, or two networks): a live direct link to the recipient is + tried before any other carrier, so on one LAN the message comes back over + `wifiDirect` and proves nothing about the relay. The receipt names + `internet` when it does. Scenario 1 twice: once with A also off when B + returns (the relay's mailbox holds the frame), once with the relay + unreachable from A (A's outbox holds it). `message_undeliverable` is + printed as status on the way: it is the relay saying B is away now, not a + failure. - **Reticulum.** A gateway daemon and `rnsd` on each of two boxes with a backbone between them (a TCP interface, or a pair of RNodes), a config - with `reticulum_enabled` and `OP_GATEWAY=127.0.0.1:4242`. Check the attach + with `reticulum_enabled` and `OP_GATEWAY=127.0.0.1:4242`, and no peer + stream between the boxes, as for the relay. Check the attach first (the transport comes up only after the daemon's capabilities), then scenario 1. The service's gateway client has not met a real daemon yet. - **A phone.** The React Native example app with Bluetooth LE on, against a From b1fdecf5f8fccfae2ce909431a7715cb65019647 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:56:01 +0530 Subject: [PATCH 09/16] fix(bindings): send --await waits for the receipt on its own connection The documented check, `send` and then `await --until delivered`, timed out whenever the recipient was reachable. `send` closes its connection once `send_message` answers, and `await` opens a new one; a receipt the engine emits in between reaches no client of the application, and the local API holds only inbound tags. Over loopback the receipt lands in that gap every time (5 of 5 runs exited 2). `send --await [--timeout S]` keeps the connection `send_message` was called on, which `VerifyClient.open` subscribed before the call, and waits on it exactly as `await --until delivered` does, so a receipt that beats the call's own answer is queued rather than lost. The guide's example uses it; a separate `await` stays for a recipient that is away, started before it returns. --- CHANGELOG.md | 5 +-- .../python/offline_protocol_sdk/verify/cli.py | 12 ++++++- .../offline_protocol_sdk/verify/commands.py | 25 ++++++++++++--- .../tests/verify/test_verify_commands.py | 32 +++++++++++++++++++ docs/local-api.md | 20 ++++++++---- 5 files changed, 80 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0484cbd4d..d05b89f66 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -90,8 +90,9 @@ archived by series under [docs/changelog/](docs/changelog/); see the when the engine gave the message up. It matches events by the identifier they carry, so it works on a sender restarted since the send. A receipt the engine emits while no client of the application is connected is not - held, so start `await` or `watch` on the sender before the recipient can - answer. New scenario tests run the networking properties end to end over + held, so `send --await` waits for it on the connection it sent on, and a + separate `await` or `watch` on the sender must start before the recipient + can answer. New scenario tests run the networking properties end to end over loopback peer streams with encryption on: a message to a device that is off arrives when it returns and the sender gets the receipt; a queued message survives the sender restarting (and is lost when the saved state diff --git a/bindings/python/offline_protocol_sdk/verify/cli.py b/bindings/python/offline_protocol_sdk/verify/cli.py index 394351397..af59bce44 100644 --- a/bindings/python/offline_protocol_sdk/verify/cli.py +++ b/bindings/python/offline_protocol_sdk/verify/cli.py @@ -3,6 +3,7 @@ Usage, on each device against its own service: offline-protocol-verify state --peer off1... offline-protocol-verify send off1... "hello" + offline-protocol-verify send off1... "hello" --await --timeout 120 offline-protocol-verify await --until delivered --timeout 120 offline-protocol-verify await --until received offline-protocol-verify pair off1... --timeout 60 @@ -61,6 +62,13 @@ def build_parser() -> argparse.ArgumentParser: send = sub.add_parser("send", help="send one message and print its id") send.add_argument("recipient") send.add_argument("content") + send.add_argument( + "--await", + dest="wait", + action="store_true", + help="then wait for the receipt on the same connection, as `await --until delivered` does", + ) + send.add_argument("--timeout", type=_positive, default=120.0, help="with --await") wait = sub.add_parser("await", help="wait for one message's delivery receipt, or its arrival") wait.add_argument("message_id") @@ -94,7 +102,9 @@ async def run(args: argparse.Namespace) -> int: if args.command == "state": return await commands.state(target, args.peer, output) if args.command == "send": - return await commands.send(target, args.recipient, args.content, output) + return await commands.send( + target, args.recipient, args.content, output, wait=args.wait, timeout=args.timeout + ) if args.command == "await": return await commands.await_message(target, args.message_id, args.until, args.timeout, output) if args.command == "pair": diff --git a/bindings/python/offline_protocol_sdk/verify/commands.py b/bindings/python/offline_protocol_sdk/verify/commands.py index b8d567974..8a3b66515 100644 --- a/bindings/python/offline_protocol_sdk/verify/commands.py +++ b/bindings/python/offline_protocol_sdk/verify/commands.py @@ -114,17 +114,34 @@ async def state(target: Target, peers: list[str], output: Output) -> int: return PASS -async def send(target: Target, recipient: str, content: str, output: Output) -> int: - """Sends one message and prints its id. The id is what ``await`` takes.""" +async def send( + target: Target, + recipient: str, + content: str, + output: Output, + *, + wait: bool = False, + timeout: float = 120.0, +) -> int: + """Sends one message and prints its id. The id is what ``await`` takes. + + With ``wait`` it then waits for the receipt as ``await --until + delivered`` does, on the same connection. That is the only race-free + way to see the receipt of a message to a recipient that is reachable + now: the connection is subscribed before ``send_message`` is called, + whereas a separate ``await`` connects only after this one closed, and + a receipt that fires in between reaches no client and is not held.""" client = await target.open() try: message_id = await client.call( "send_message", {"recipient": recipient, "content": content, "priority": "Medium"} ) + output.line({"sent": {"message_id": message_id, "recipient": recipient}}) + output.say(f"sent {message_id} to {recipient}") + if wait: + return await _await_on(client, message_id, "delivered", timeout, output) finally: await client.close() - output.line({"sent": {"message_id": message_id, "recipient": recipient}}) - output.say(f"sent {message_id} to {recipient}") return PASS diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py index 6ea980919..b58374b0f 100644 --- a/bindings/python/tests/verify/test_verify_commands.py +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -82,6 +82,38 @@ async def test_send_then_await_passes_on_the_receipt_and_the_arrival(harness): assert event["type"] == "message_received" and event["content"] == "checked" +async def test_send_await_waits_for_the_receipt_on_its_own_connection(harness): + """``send --await`` is the race-free way to the receipt of a message to + a reachable recipient: a separate ``await`` connects after the send's + connection closed, and the receipt usually fires in between.""" + server_a, server_b = await two_servers(harness) + out = _output() + status = await commands.send( + _target(server_a), server_b.manager.local_address, "and wait", out, wait=True, timeout=15.0 + ) + assert status == commands.PASS + sent, result = _lines(out)[0], _lines(out)[-1] + message_id = sent["sent"]["message_id"] + assert result["result"] == "pass" + assert result["event"]["type"] == "message_delivered" + assert result["event"]["message_id"] == message_id + + +async def test_send_await_times_out_with_status_two(harness): + server = await harness.server(config=make_config(profile="alone", internet_enabled=False, wifi_direct_enabled=True)) + out = _output() + status = await commands.send(_target(server), "off1nobody", "lost", out, wait=True, timeout=0.3) + assert status == commands.TIMED_OUT + assert _lines(out)[-1]["result"] == "timeout" + + +def test_the_command_line_takes_send_await(): + args = cli.build_parser().parse_args(["send", "off1x", "hi", "--await", "--timeout", "5"]) + assert args.wait is True and args.timeout == 5.0 + args = cli.build_parser().parse_args(["send", "off1x", "hi"]) + assert args.wait is False + + async def test_send_prints_the_id_await_matches(harness): server_a, server_b = await two_servers(harness) out = _output() diff --git a/docs/local-api.md b/docs/local-api.md index 5c1b2e0e9..4d4f35000 100644 --- a/docs/local-api.md +++ b/docs/local-api.md @@ -161,17 +161,23 @@ device it runs on and waits for those events, so a check passes or fails on what the engine says: ```bash -# on A: send, then wait for B's acknowledgement (up to 10 minutes) -id=$(offline-protocol-verify send off1...B "hello" | jq -r .sent.message_id) -offline-protocol-verify await "$id" --until delivered --timeout 600 +# on A: send, and wait for B's acknowledgement on the same connection +# (up to 10 minutes) +id=$(offline-protocol-verify send off1...B "hello" --await --timeout 600 \ + | jq -r 'select(.sent) | .sent.message_id') # on B: wait for the message itself offline-protocol-verify await "$id" --until received ``` +A separate `await --until delivered` connects only after `send` has closed +its connection, and a receipt that arrives in between reaches no client: to +a recipient that is reachable now that is most receipts. Use it for a +message whose recipient is away, started before the recipient returns. + | Command | Waits for | Passes on | |---|---|---| | `state [--peer ADDR]` | nothing | always; prints address, carriers, queues, relay counters, the session with each peer; it never takes a held message | -| `send ADDR TEXT` | nothing | always; prints the message id | +| `send ADDR TEXT [--await]` | nothing, or with `--await` the receipt | prints the message id; with `--await`, as `await --until delivered`, on the connection the send used | | `await ID --until delivered` | the receipt on the sender | `message_delivered` naming the id; `message_failed` is a failure, `message_deferred`, `message_retrying` and `message_undeliverable` print as status | | `await ID --until received` | the message on the recipient | `message_received` naming the id | | `pair ADDR` | the session with a peer | `get_establishment_state` reaching `SessionConfirmed` | @@ -195,9 +201,9 @@ events by the identifier they carry rather than relying on the server's correlation, which knows only the identifiers its own clients were handed by this process; so `await` works on a sender restarted since the send. And an event the engine emits while no client of the application is connected -is gone, except a received message: start `await` or `watch` on the sender -before the recipient can answer, which for a restart test means restarting -the sender while the recipient is still away. Two devices that will reach +is gone, except a received message: use `send --await`, or start `await` +or `watch` on the sender before the recipient can answer, which for a +restart test means restarting the sender while the recipient is still away. Two devices that will reach each other only through a third must have met directly once: the automatic key exchange and the Welcome travel over a direct link, never through the mesh. Today a message that crosses the mesh is received, but the sender's From 94e2bdf3bf8488655d78ec90ed18346030d23441 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:57:11 +0530 Subject: [PATCH 10/16] fix(bindings): watch rides out a dropped handshake, and fails unseen `watch` promises to span a restart of the service, but it retried only an `OSError`, a closed connection or a refused `hello`. A service that is stopping or starting can accept the connection and close it before the WebSocket handshake answers; websockets raises `InvalidMessage` for that (a `WebSocketException`, not an `OSError`), and the watcher died with status 1 in the very gap it exists to cover. An opening handshake that times out raises `asyncio.TimeoutError`, which is not an `OSError` on Python 3.10. Both are now retried. A `watch --duration` that never reached the service also exited 0 with an empty log, which an orchestrator cannot tell from a quiet network. It now says once that it is waiting, and exits 1 if the duration ends without it ever having connected. --- .../offline_protocol_sdk/verify/commands.py | 24 +++++++- .../tests/verify/test_verify_commands.py | 56 +++++++++++++++++++ docs/local-api.md | 2 +- 3 files changed, 78 insertions(+), 4 deletions(-) diff --git a/bindings/python/offline_protocol_sdk/verify/commands.py b/bindings/python/offline_protocol_sdk/verify/commands.py index 8a3b66515..c88de51e8 100644 --- a/bindings/python/offline_protocol_sdk/verify/commands.py +++ b/bindings/python/offline_protocol_sdk/verify/commands.py @@ -21,6 +21,8 @@ from pathlib import Path from typing import IO, Any, Callable +from websockets.exceptions import WebSocketException + from .client import DEFAULT_APP_ID, RpcFailure, VerifyClient PASS = 0 @@ -330,22 +332,35 @@ async def watch( and appends it to ``log`` when given. Reconnects when the service goes away and comes back, so one watcher spans a restart; events the engine emits while no client is connected are not seen, except the received - messages the service holds for the application.""" + messages the service holds for the application. + + Passes when ``duration`` runs out having been connected at some point, + and fails when it never was: an empty watch log is otherwise + indistinguishable from a quiet network.""" loop = asyncio.get_running_loop() deadline = None if duration is None else loop.time() + duration handle = log.open("a", encoding="utf-8") if log is not None else None connected = False + ever_connected = False + waiting_said = False try: while deadline is None or loop.time() < deadline: try: client = await target.open() - except (OSError, ConnectionError, RpcFailure) as exc: + except (OSError, ConnectionError, RpcFailure, WebSocketException, asyncio.TimeoutError) as exc: + # A service that is stopping or starting can refuse the + # connection, drop it inside the handshake (a websockets + # error, not an OSError) or close it during ``hello``; each + # is the gap a restart leaves, so each is retried. if connected: output.say(f"watch: service gone ({exc}); reconnecting") connected = False + elif not ever_connected and not waiting_said: + output.say(f"watch: waiting for the service ({exc or type(exc).__name__})") + waiting_said = True await asyncio.sleep(WATCH_RECONNECT_DELAY) continue - connected = True + connected = ever_connected = True output.say(f"watch: connected to {client.local_address}") try: while True: @@ -368,6 +383,9 @@ async def watch( connected = False finally: await client.close() + if not ever_connected: + output.say("watch: never connected to the service") + return FAILED return PASS finally: if handle is not None: diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py index b58374b0f..7e3f4181f 100644 --- a/bindings/python/tests/verify/test_verify_commands.py +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -283,6 +283,62 @@ async def test_watch_reconnects_when_the_service_comes_back(harness, tmp_path, m await asyncio.gather(watcher, return_exceptions=True) +async def test_watch_retries_a_service_that_drops_the_handshake(harness, monkeypatch): + """A service that is stopping or starting can accept a connection and + close it before the WebSocket handshake answers. That raises a + websockets error rather than an ``OSError``, and is a restart's gap + like any other: ``watch`` keeps retrying and connects once it is up.""" + monkeypatch.setattr(commands, "WATCH_RECONNECT_DELAY", 0.05) + first = await harness.server(config=make_config(profile="first")) + socket_path = first.socket_path + await first.stop() + if socket_path.exists(): + socket_path.unlink() + attempts = 0 + + def drop(reader, writer): + nonlocal attempts + attempts += 1 + writer.close() + + dropping = await asyncio.start_unix_server(drop, str(socket_path)) + out = _output() + watcher = asyncio.ensure_future(commands.watch(commands.Target(socket_path=str(socket_path)), None, 30.0, out)) + try: + for _ in range(100): + if attempts >= 2 or watcher.done(): + break + await asyncio.sleep(0.05) + assert not watcher.done(), watcher.exception() + assert attempts >= 2 + dropping.close() + await dropping.wait_closed() + if socket_path.exists(): + socket_path.unlink() + second = LocalApiServer(ProtocolManager(make_config(profile="second")), socket_path=socket_path, health=False) + await second.start() + try: + for _ in range(200): + if "watch: connected" in out.err.getvalue(): + break + await asyncio.sleep(0.05) + assert "watch: connected" in out.err.getvalue(), out.err.getvalue() + finally: + await second.stop() + finally: + watcher.cancel() + await asyncio.gather(watcher, return_exceptions=True) + + +async def test_watch_that_never_connected_fails(tmp_path, monkeypatch): + monkeypatch.setattr(commands, "WATCH_RECONNECT_DELAY", 0.05) + out = _output() + status = await commands.watch(commands.Target(socket_path=str(tmp_path / "absent.sock")), None, 0.3, out) + assert status == commands.FAILED + assert out.out.getvalue() == "" + assert "never connected" in out.err.getvalue() + + async def test_the_tcp_carrier_reads_the_token_file(harness): server = await harness.server(config=make_config(profile="over-tcp"), tcp=True) target = commands.Target(tcp_port=server.port, token_file=str(server.token_path)) diff --git a/docs/local-api.md b/docs/local-api.md index 4d4f35000..ae6932768 100644 --- a/docs/local-api.md +++ b/docs/local-api.md @@ -182,7 +182,7 @@ message whose recipient is away, started before the recipient returns. | `await ID --until received` | the message on the recipient | `message_received` naming the id | | `pair ADDR` | the session with a peer | `get_establishment_state` reaching `SessionConfirmed` | | `ping ADDR --every S --count N` | each message's receipt | every message delivered within `--timeout` of its send; each line names the carrier | -| `watch [--log FILE]` | until stopped or `--duration` | every event as `{"at_ms", "event"}`, reconnecting across restarts of the service | +| `watch [--log FILE]` | until stopped or `--duration` | every event as `{"at_ms", "event"}`, reconnecting across restarts of the service; fails if it never connected | Each command prints JSON lines on standard output and a summary on standard error, and exits 0 on a pass, 2 when its time ran out, and 1 on a terminal From 40cf2d6a7afa8f1dbbca981a33244cac43041a6d Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:58:05 +0530 Subject: [PATCH 11/16] test(bindings): the reboot control first proves sending still works `test_without_the_saved_state_the_message_is_gone` passed when no receipt for the queued message came within five seconds of the relink. That also holds if wiping the protocol-state directory broke sending altogether (a lost session, a refused carrier), so the control could pass without saying anything about the saved queue. It now sends a fresh message after the restart and waits for both its receipt on A and its arrival on B before asserting the queued one neither settles on A nor arrives on B. Keeping the saved state (the mutation the control exists for) still turns it red. --- bindings/python/tests/scenarios/test_reboot.py | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/bindings/python/tests/scenarios/test_reboot.py b/bindings/python/tests/scenarios/test_reboot.py index acf350fd0..1519cf2cc 100644 --- a/bindings/python/tests/scenarios/test_reboot.py +++ b/bindings/python/tests/scenarios/test_reboot.py @@ -64,14 +64,26 @@ async def test_without_the_saved_state_the_message_is_gone(network): """The control for the test above: the same restart over an empty protocol-state directory (the identity and the sessions are in the MLS store and survive) delivers nothing, so it is the saved queue that - carries the message across the restart, not anything in memory.""" + carries the message across the restart, not anything in memory. + + A message sent after the restart is delivered first: without it the + control would also pass if wiping the state broke sending altogether, + and the absence of the queued message would prove nothing.""" a, b, message_id = await _queue_while_b_is_off(network) await a.switch_off() await a.switch_on(state_root=a.root / "state-empty") after = Recorder(await network.client(a)) await b.switch_on() + received = Recorder(await network.client(b)) await until(lambda: a.linked_to(b), SESSION_TIMEOUT, "A and B linked again") + fresh_id = await after.client.call( + "send_message", {"recipient": b.address, "content": "sent after the restart", "priority": "Medium"} + ) + await after.wait("message_delivered", DELIVERY_TIMEOUT, message_id=fresh_id) + await received.wait("message_received", DELIVERY_TIMEOUT, message_id=fresh_id) + with pytest.raises(AssertionError, match="no message_delivered"): await after.wait("message_delivered", 5.0, message_id=message_id) + assert message_id not in {e.get("message_id") for e in received.events if e.get("type") == "message_received"} From 42ab0e55f6aabbccaa5887d89165157eecbbf9d6 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:59:00 +0530 Subject: [PATCH 12/16] feat(bindings): await and watch say when they are listening `await` and `watch` printed nothing between connecting and the event they wait for, so a script that starts one in the background and then triggers the event (brings the recipient back, sends from another device) could only sleep and hope the subscription was in place. A receipt or a relay report emitted before it is not held, so a short sleep loses it and a long one only makes that less likely. Both now print the line `subscribed`, alone, on standard error once `subscribe` has answered (`watch` again after each reconnect). The string is `commands.READY_LINE`, documented in the guide and pinned in a test, since a script matches it. --- CHANGELOG.md | 3 +- .../offline_protocol_sdk/verify/commands.py | 10 ++++++ .../tests/verify/test_verify_commands.py | 34 +++++++++++++++++++ docs/local-api.md | 7 +++- 4 files changed, 52 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d05b89f66..5a4764672 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -92,7 +92,8 @@ archived by series under [docs/changelog/](docs/changelog/); see the the engine emits while no client of the application is connected is not held, so `send --await` waits for it on the connection it sent on, and a separate `await` or `watch` on the sender must start before the recipient - can answer. New scenario tests run the networking properties end to end over + can answer; both print `subscribed` on standard error once they are + listening, which a script waits for instead of sleeping. New scenario tests run the networking properties end to end over loopback peer streams with encryption on: a message to a device that is off arrives when it returns and the sender gets the receipt; a queued message survives the sender restarting (and is lost when the saved state diff --git a/bindings/python/offline_protocol_sdk/verify/commands.py b/bindings/python/offline_protocol_sdk/verify/commands.py index c88de51e8..010502d62 100644 --- a/bindings/python/offline_protocol_sdk/verify/commands.py +++ b/bindings/python/offline_protocol_sdk/verify/commands.py @@ -49,6 +49,14 @@ #: How long ``watch`` waits before reconnecting to a service that went away. WATCH_RECONNECT_DELAY = 0.5 +#: The stderr line ``await`` and ``watch`` print, alone, once their +#: subscription to every event is confirmed. An event emitted after it is +#: seen; one emitted before it may not be (only received messages are held). +#: A caller that starts one of them and then triggers the event (brings a +#: recipient back, sends from another device) waits for this line rather +#: than sleeping. Part of the command line's contract: a script matches it. +READY_LINE = "subscribed" + @dataclass class Target: @@ -168,6 +176,7 @@ async def await_message( is not held: start this before the recipient can answer. """ client = await target.open() + output.say(READY_LINE) try: return await _await_on(client, message_id, until, timeout, output) finally: @@ -362,6 +371,7 @@ async def watch( continue connected = ever_connected = True output.say(f"watch: connected to {client.local_address}") + output.say(READY_LINE) try: while True: remaining = None if deadline is None else deadline - loop.time() diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py index 7e3f4181f..e6411a36b 100644 --- a/bindings/python/tests/verify/test_verify_commands.py +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -152,6 +152,40 @@ async def test_await_times_out_with_status_two(harness): assert _lines(out)[-1] == {"result": "timeout", "message_id": "no-such-message", "until": "delivered"} +async def test_await_and_watch_say_subscribed_once_ready(harness): + """The readiness line is a contract with scripts (the offline-first + runner waits for it), so its exact text is pinned here.""" + assert commands.READY_LINE == "subscribed" + server_a, server_b = await two_servers(harness) + out = _output() + waiting = asyncio.ensure_future(commands.await_message(_target(server_a), "m-ready", "delivered", 10.0, out)) + try: + for _ in range(200): + if "subscribed\n" in out.err.getvalue(): + break + await asyncio.sleep(0.02) + assert out.err.getvalue().splitlines()[0] == "subscribed" + # Anything emitted from here on is seen: the server pushes it to the + # subscribed connection, not to a hold. + server_a.router.route({"type": "message_delivered", "message_id": "m-ready", "transport": "x", "hop_count": 0}) + assert await waiting == commands.PASS + finally: + waiting.cancel() + await asyncio.gather(waiting, return_exceptions=True) + + out = _output() + watcher = asyncio.ensure_future(commands.watch(_target(server_b), None, 10.0, out)) + try: + for _ in range(200): + if "subscribed\n" in out.err.getvalue(): + break + await asyncio.sleep(0.02) + assert "subscribed" in out.err.getvalue().splitlines() + finally: + watcher.cancel() + await asyncio.gather(watcher, return_exceptions=True) + + async def test_await_fails_on_message_failed_and_prints_status_events(harness): server = await harness.server(config=make_config(profile="alone")) out = _output() diff --git a/docs/local-api.md b/docs/local-api.md index ae6932768..073ad0416 100644 --- a/docs/local-api.md +++ b/docs/local-api.md @@ -187,7 +187,12 @@ message whose recipient is away, started before the recipient returns. Each command prints JSON lines on standard output and a summary on standard error, and exits 0 on a pass, 2 when its time ran out, and 1 on a terminal failure (the engine gave the message up, refused a call, or the service went -away). Every verifier declares the application id `offline-protocol-verify` +away). `await` and `watch` print the line `subscribed`, alone, on standard +error once their subscription to every event is confirmed (`watch` again +after each reconnect). An event emitted after that line is seen, and one +emitted before it may not be, so a script that starts either in the +background and then brings a recipient back or sends from another device +waits for that line instead of sleeping. Every verifier declares the application id `offline-protocol-verify` unless told otherwise with `--app-id`, and the sending and receiving devices must declare the same one: a received message is routed to the clients of the application its sender stamped, and held for that From 514997ac1916c71eecae0af54ab9ac8e65f7bff1 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 13:59:07 +0530 Subject: [PATCH 13/16] chore(bindings): two missing spaces in the scenario tests --- bindings/python/tests/scenarios/network.py | 2 +- bindings/python/tests/scenarios/test_store_and_forward.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/bindings/python/tests/scenarios/network.py b/bindings/python/tests/scenarios/network.py index 39ec70bf0..218ff7c0b 100644 --- a/bindings/python/tests/scenarios/network.py +++ b/bindings/python/tests/scenarios/network.py @@ -120,7 +120,7 @@ def __init__(self) -> None: # Unix sockets; the verifier's TCP carrier is covered in tests/verify. if os.name == "nt": pytest.skip("the scenario devices serve the local API on a Unix socket") - self._tmp =Path(tempfile.mkdtemp(prefix="opnet-", dir="/tmp")) + self._tmp = Path(tempfile.mkdtemp(prefix="opnet-", dir="/tmp")) self.devices: list[Device] = [] self._clients: list[VerifyClient] = [] diff --git a/bindings/python/tests/scenarios/test_store_and_forward.py b/bindings/python/tests/scenarios/test_store_and_forward.py index 9d6aa6cd1..f38d0ec87 100644 --- a/bindings/python/tests/scenarios/test_store_and_forward.py +++ b/bindings/python/tests/scenarios/test_store_and_forward.py @@ -74,7 +74,7 @@ async def test_a_message_to_a_device_that_is_off_arrives_when_it_comes_back(netw async def test_the_verifier_waits_out_the_absence_and_passes_on_the_receipt(network): """The same, through the commands an operator runs: ``await`` started on the sender while the recipient is still off, and passing once it is on.""" - a =network.device("alpha", 0x11) + a = network.device("alpha", 0x11) b = network.device("bravo", 0x22) b.peers = [a] await a.switch_on() From aaa8d0be76b72b141408d5841bb76a85fea5d94e Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 14:02:09 +0530 Subject: [PATCH 14/16] fix(bindings): the lab waits for the verifier to subscribe, not 2 s run.py started each background await or watch and then slept two seconds before switching B on or sending. Neither a receipt nor a relay report is held for a client that is not listening yet, so on a slow box (or over ssh) the event could fire before the verifier had subscribed, and the scenario failed 120 s later for no real reason. The verifier now prints a `subscribed` line on stderr once its subscription is confirmed. Wait for that line instead of guessing. Both pipes are drained on threads from the start, so reading stderr early never loses the tail of it and a long watch never blocks on a full stdout pipe. A verifier that exits before subscribing fails the scenario at once with its error, rather than passing as ready. --- examples/offline-first/run.py | 82 ++++++++++++++++++++++++++++------- 1 file changed, 66 insertions(+), 16 deletions(-) diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py index 89c04457f..8ee961c75 100644 --- a/examples/offline-first/run.py +++ b/examples/offline-first/run.py @@ -30,6 +30,7 @@ import shlex import subprocess import sys +import threading import time from dataclasses import dataclass, field from pathlib import Path @@ -85,14 +86,24 @@ def verify(self, *args: str, timeout: float = 600) -> tuple[int, list[dict], str ) return process.returncode, _json_lines(process.stdout), process.stderr.strip() - def verify_in_background(self, *args: str) -> subprocess.Popen[str]: - return subprocess.Popen( + def verify_in_background(self, *args: str) -> "Background": + """Start an `await` or `watch` and return once it is subscribed. + + Neither a receipt nor a relay report is held for a client that is + not yet listening, so the caller must not trigger the event until + the verifier has printed its readiness line. + """ + background = Background(subprocess.Popen( self.runner.command(self.name, ["offline-protocol-verify", "--socket", self.runner.socket, *args]), stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, - ) + )) + if not background.subscribed.wait(SUBSCRIBE_S) or not background.ready: + status, _, err = background.finish(5) + raise ScenarioFailed(f"{self.name}: {args[0]} never subscribed (exit {status}): {err}") + return background def wait_ready(self) -> None: deadline = time.monotonic() + READY_S @@ -130,14 +141,55 @@ def _saw(watched: list[dict], tag: str, message_id: str) -> dict | None: return None -def _finish(process: subprocess.Popen[str], timeout: float) -> tuple[int, list[dict], str]: - try: - out, err = process.communicate(timeout=timeout) - except subprocess.TimeoutExpired: - process.kill() - out, err = process.communicate() - return 2, _json_lines(out), err.strip() - return process.returncode, _json_lines(out), err.strip() +#: The exact stderr line `await` and `watch` print once subscribed +#: (`offline_protocol_sdk.verify.commands.READY_LINE`). +READY_LINE = "subscribed" +SUBSCRIBE_S = 30 + + +class Background: + """A verifier in flight, its output drained on threads. + + Both pipes are read line by line from the start: stderr to see the + readiness line, stdout so a long `watch` never blocks on a full pipe. + """ + + def __init__(self, process: subprocess.Popen[str]) -> None: + self.process = process + #: Set once the readiness line arrives or stderr closes; `ready` + #: says which. + self.subscribed = threading.Event() + self.ready = False + self._out: list[str] = [] + self._err: list[str] = [] + self._readers = [ + threading.Thread(target=self._drain, args=(process.stdout, self._out, False), daemon=True), + threading.Thread(target=self._drain, args=(process.stderr, self._err, True), daemon=True), + ] + for reader in self._readers: + reader.start() + + def _drain(self, stream, into: list[str], watch_ready: bool) -> None: + for line in stream: + into.append(line) + if watch_ready and line.strip() == READY_LINE: + self.ready = True + self.subscribed.set() + # The pipe closed: nothing will subscribe now, so stop any wait. + if watch_ready: + self.subscribed.set() + + def finish(self, timeout: float) -> tuple[int, list[dict], str]: + try: + status = self.process.wait(timeout=timeout) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait() + status = 2 + for reader in self._readers: + reader.join(timeout=5) + err = "".join(line for line in self._err if line.strip() != READY_LINE) + return status, _json_lines("".join(self._out)), err.strip() class Runner: @@ -296,10 +348,9 @@ def store_and_forward(self, restart_sender: bool) -> Result: # wait starts before B can answer. receipt = self.a.verify_in_background("await", message_id, "--until", "delivered", "--timeout", str(DELIVERY_AFTER_RETURN_S + READY_S)) - time.sleep(2) started = time.monotonic() self.runner.on("b") - status, lines, err = _finish(receipt, DELIVERY_AFTER_RETURN_S + READY_S + 30) + status, lines, err = receipt.finish(DELIVERY_AFTER_RETURN_S + READY_S + 30) elapsed = time.monotonic() - started if status != 0 or elapsed > DELIVERY_AFTER_RETURN_S: return Result(name, False, elapsed, f"no receipt within {DELIVERY_AFTER_RETURN_S} s: {err}") @@ -325,14 +376,13 @@ def through_the_middle(self, require_receipt: bool) -> Result: window = str(HOP_RECEIVED_S + HOP_RECEIPT_S) middle = self.b.verify_in_background("watch", "--duration", window) sender = self.a.verify_in_background("watch", "--duration", window) - time.sleep(2) started = time.monotonic() message_id = self.a.send(self.c, f"{name} {time.strftime('%H:%M:%S')}") status, lines, err = self.c.verify("await", message_id, "--until", "received", "--timeout", str(HOP_RECEIVED_S)) elapsed = time.monotonic() - started - _, watched_b, _ = _finish(middle, float(window) + 30) - _, watched_a, _ = _finish(sender, float(window) + 30) + _, watched_b, _ = middle.finish(float(window) + 30) + _, watched_a, _ = sender.finish(float(window) + 30) if status != 0: return Result(name, False, elapsed, f"C did not receive it within {HOP_RECEIVED_S} s: {err}") hops = lines[-1]["event"]["hop_count"] From 61fb5460b9bd80fd4c7197c1fa3756e8abde53e0 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 14:18:43 +0530 Subject: [PATCH 15/16] test(bindings): skip the never-connected watch test on Windows asyncio has no Unix socket client on Windows: create_unix_connection raises NotImplementedError, so the test failed in the Windows wheel job before the watcher ever got to retry. Every other test that needs a Unix socket already skips there through the harness, and the verifier reaches a Windows service over --tcp. This one built its target by hand and so missed the skip. --- bindings/python/tests/verify/test_verify_commands.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/bindings/python/tests/verify/test_verify_commands.py b/bindings/python/tests/verify/test_verify_commands.py index e6411a36b..56879ede2 100644 --- a/bindings/python/tests/verify/test_verify_commands.py +++ b/bindings/python/tests/verify/test_verify_commands.py @@ -11,6 +11,7 @@ import asyncio import io import json +import os import pytest @@ -364,6 +365,7 @@ def drop(reader, writer): await asyncio.gather(watcher, return_exceptions=True) +@pytest.mark.skipif(os.name == "nt", reason="asyncio has no Unix socket client on Windows") async def test_watch_that_never_connected_fails(tmp_path, monkeypatch): monkeypatch.setattr(commands, "WATCH_RECONNECT_DELAY", 0.05) out = _output() From 61867d63b8f2928c3250af57afd28cd4b8649562 Mon Sep 17 00:00:00 2001 From: bahdotsh Date: Thu, 8 Oct 2026 16:47:45 +0530 Subject: [PATCH 16/16] fix(bindings): require the hop receipt in the offline-first demo The demo opens by saying the recipient's acknowledgement comes back as message_delivered, and then scenario 4 passed without it. That was defensible while #537 was open. #537 is merged now, so a missing receipt across the hop is a regression, and the scenario has to fail on it. --require-hop-receipt becomes --allow-missing-hop-receipt, for an image built from a 0.28.0 or earlier wheel whose engine never sends one. The known gaps were stale too. #536 taught the BlueZ peripheral which phone wrote to it. What is left is narrower: a phone is mapped to its user id only while it is the one central connected. The retry-path gap now has an issue (#541), so the README points at it instead of describing the mechanism a third time. The run record says that its lab run used an engine without #537, which is why that run has no receipt. restore() switched devices back on but left C on net-ab after a scenario 4 that failed half-way. Every later scenario then ran with A and C linked, and the lab exists to keep them apart. It now parts them the same way part_a_and_c does, with C stopped first so A sees the stream close. While at it, --ssh without '=' is a usage error instead of a ValueError traceback, and the image README names the verifier check as a second reason the PyPI build fails. --- bindings/python/docker/README.md | 3 +- examples/offline-first/README.md | 31 ++++++++++----------- examples/offline-first/run.py | 47 +++++++++++++++++++++----------- 3 files changed, 48 insertions(+), 33 deletions(-) diff --git a/bindings/python/docker/README.md b/bindings/python/docker/README.md index 577fbe95e..503a2a94d 100644 --- a/bindings/python/docker/README.md +++ b/bindings/python/docker/README.md @@ -85,7 +85,8 @@ docker build bindings/python/docker Wheels for several architectures may sit there together; pip takes the one that matches the image. The build fails if the installed service has no HTTP -front, which is the case for every release before the `http` extra. +front or the package has no `offline-protocol-verify`, which is the case for +every release so far. The build context is this directory only. Never build from the repository root or mount the checkout: `target/` alone fills the build VM. diff --git a/examples/offline-first/README.md b/examples/offline-first/README.md index 61ec6dc4a..e6737056d 100644 --- a/examples/offline-first/README.md +++ b/examples/offline-first/README.md @@ -13,7 +13,7 @@ read on the device by | 1 | Store and forward: B off, A sends, B on | B receives it, and A gets `message_delivered` within 60 s of B coming back | | 2 | The sender restarts: B off, A sends, A killed (`SIGKILL`) and started, B on | the same, from the restarted A: the queue was on disk | | 3 | The carrier changes: the link A and B were using goes away | every message of a `ping` run is delivered, and `message_delivered.transport` names the new carrier (hardware only, below) | -| 4 | Through the middle: A and C meet once, then only B hears both | C receives A's message at `hop_count` 1 within 30 s, and B reports `message_relayed` | +| 4 | Through the middle: A and C meet once, then only B hears both | C receives A's message at `hop_count` 1 within 30 s, B reports `message_relayed`, and A gets `message_delivered` back the same way | ## In containers, on one machine @@ -30,7 +30,7 @@ DNS's. ```bash python3 run.py # builds the image, starts the devices, runs 1, 2 and 4 python3 run.py --fresh # new identities first -python3 run.py --scenario 4 --require-hop-receipt +python3 run.py --scenario 4 ``` The image installs the package from PyPI unless a Linux wheel built from a @@ -113,25 +113,24 @@ each: ## Known gaps -- **The sender's receipt does not come back across a hop.** In scenario 4, - C's acknowledgement is carried back by B and dropped by A: a message whose - only route is the mesh has no pending acknowledgement on A to settle. - `run.py` reports it and passes unless `--require-hop-receipt`. The engine - fix is #537. -- **A message a direct link took and then lost never crosses the mesh.** - Only a send that no carrier takes is handed to neighbours; a retry that - the carriers refuse is queued again for direct carriers only. So a message - sent down a stream to a device that went away without closing it waits - for a direct link to that device. -- **Bluetooth LE on a Linux box as the peripheral** does not learn which - phone wrote to it, so replies to a phone go over the box's central role. - One Bluetooth LE peer per box until that is fixed. +- **A message a direct link took and then lost never crosses the mesh** + (#541): a message sent down a stream to a device that went away without + closing it waits for a direct link to that device. +- **One phone per Linux box over Bluetooth LE.** The box's peripheral learns + which phone wrote to it, but it maps a phone to its user id only while that + phone is the one central connected, so with two a reply has no route back + through the peripheral. +- **An image built from 0.28.0 or earlier** never gets the receipt in + scenario 4: the engine drops an acknowledgement for a message whose only + route was the mesh (#537). Pass `--allow-missing-hop-receipt` to run the + rest of the scenario on such an image. ## What has been run | When | Where | Scenarios | Result | |---|---|---|---| -| 2026-10-08 | `run.py`, Docker 28.3 on one arm64 laptop, the image built from the 0.28.0 Linux wheel with this branch's Python sources (no native change since 0.28.0) | 1, 2, 4 | pass, twice in a row (receipt latency 29 to 45 ms; 4 without A's receipt, as above) | +| 2026-10-08 | `run.py`, Docker 28.3 on one arm64 laptop, the image built from the 0.28.0 Linux wheel with this branch's Python sources, so an engine without #537 | 1, 2, 4 | pass, twice in a row (receipt latency 29 to 45 ms; 4 without A's receipt, which that engine never sends) | +| 2026-10-08 | `run.py --fresh` then `run.py`, the same machine, the image with the native library built from `main` after #537 and this branch's Python sources | 1, 2, 4 | pass, twice in a row, A's receipt required in 4 and back at hop 1 (latency 33 to 36 ms in 1 and 2) | Nothing on this page has been run between separate hosts, over Bluetooth LE, over a relay, through a gateway daemon, or with a phone. diff --git a/examples/offline-first/run.py b/examples/offline-first/run.py index 8ee961c75..01bbba490 100644 --- a/examples/offline-first/run.py +++ b/examples/offline-first/run.py @@ -7,7 +7,8 @@ 1 store and forward B off, A sends, B on: B receives it, A gets B's receipt 2 sender restarts B off, A sends, A killed and restarted, B on: same 4 through the middle A and C meet once, then share no network: C receives - A's message one hop away and B reports carrying it + A's message one hop away, B reports carrying it, and + A gets C's receipt back the same way Without --ssh the devices are the containers of compose.yml in this directory, and switching a device off is `docker stop`. With --ssh each @@ -44,9 +45,10 @@ #: and for a peer-stream redial ladder that may be part-way up. DELIVERY_AFTER_RETURN_S = 60 HOP_RECEIVED_S = 30 -#: How long to wait for the sender's receipt across the hop. It does not come -#: back on an engine without #537 (docs/local-api.md), so by default its -#: absence is reported and does not fail the run. +#: How much longer than HOP_RECEIVED_S the sender's watch waits for the +#: receipt across the hop. An engine without #537 (0.28.0 and earlier) never +#: settles it, and --allow-missing-hop-receipt lets an image built from such +#: a wheel pass. HOP_RECEIPT_S = 20 #: Session formation after devices first hear each other, and a restarted #: service answering on its socket. @@ -258,14 +260,17 @@ def on(self, device: str) -> None: def kill(self, device: str) -> None: self._docker("kill", "--signal", "KILL", f"{PROJECT}-{device}") - def join_a_and_c(self) -> None: - # A run that stopped between this and part_a_and_c leaves C on net-ab, - # where `compose up` does not take it off; connecting it again fails. + def _c_on_net_ab(self) -> bool: networks = subprocess.run( ["docker", "inspect", "--format", "{{json .NetworkSettings.Networks}}", f"{PROJECT}-c"], check=True, capture_output=True, text=True, ).stdout - if f"{PROJECT}_net-ab" not in json.loads(networks): + return f"{PROJECT}_net-ab" in json.loads(networks) + + def join_a_and_c(self) -> None: + # A run that stopped between this and part_a_and_c leaves C on net-ab, + # where `compose up` does not take it off; connecting it again fails. + if not self._c_on_net_ab(): self._docker("network", "connect", f"{PROJECT}_net-ab", f"{PROJECT}-c") def part_a_and_c(self) -> None: @@ -273,12 +278,19 @@ def part_a_and_c(self) -> None: # the network first, C would leave A a stream that looks open until # keepalive ends it (about 30 s), and a message A sends in that window # goes down it, is retried over direct carriers only, and never - # reaches the mesh (docs/local-api.md). + # reaches the mesh (#541). self._docker("stop", f"{PROJECT}-c") self._docker("network", "disconnect", f"{PROJECT}_net-ab", f"{PROJECT}-c") self._docker("start", f"{PROJECT}-c") def restore(self) -> None: + # A scenario 4 that failed between join_a_and_c and part_a_and_c + # leaves C on net-ab, linked to A: every later scenario would run on + # a graph where B is not the only way between them. Parted the same + # way part_a_and_c does it, stopped first so A sees the stream close. + if self._c_on_net_ab(): + self._docker("stop", f"{PROJECT}-c") + self._docker("network", "disconnect", f"{PROJECT}_net-ab", f"{PROJECT}-c") # `docker start` leaves a running container as it is. self._docker("start", *(f"{PROJECT}-{d}" for d in "abc")) @@ -361,7 +373,7 @@ def store_and_forward(self, restart_sender: bool) -> Result: return Result(name, False, elapsed, f"receipt came back but B has no message: {err}") return Result(name, True, elapsed, f"receipt over {event['transport']}, latency {event['latency_ms']} ms") - def through_the_middle(self, require_receipt: bool) -> Result: + def through_the_middle(self, allow_missing_receipt: bool) -> Result: name = "4 through the middle" self.runner.join_a_and_c() self.c.wait_ready() @@ -393,9 +405,9 @@ def through_the_middle(self, require_receipt: bool) -> Result: receipt = _saw(watched_a, "message_delivered", message_id) if receipt is not None: return Result(name, True, elapsed, detail + f"; A's receipt at hop {receipt['hop_count']}") - if require_receipt: - return Result(name, False, elapsed, detail + "; A's receipt never came back") - return Result(name, True, elapsed, detail + "; A's receipt did not come back (known, #537)") + if allow_missing_receipt: + return Result(name, True, elapsed, detail + "; A's receipt did not come back (allowed)") + return Result(name, False, elapsed, detail + "; A's receipt never came back") def main(argv: list[str] | None = None) -> int: @@ -411,11 +423,14 @@ def main(argv: list[str] | None = None) -> int: parser.add_argument("--no-build", action="store_true", help="Docker: use the image offline-protocol-service:offline-first as it is") parser.add_argument("--scenario", action="append", type=int, choices=(1, 2, 4), help="run only these") - parser.add_argument("--require-hop-receipt", action="store_true", - help="fail scenario 4 when A's receipt does not come back") + parser.add_argument("--allow-missing-hop-receipt", action="store_true", + help="pass scenario 4 without A's receipt, for an image whose engine predates " + "the receipt crossing the mesh (#537)") args = parser.parse_args(argv) if args.ssh: + if any("=" not in item for item in args.ssh): + parser.error("--ssh takes NAME=HOST, e.g. a=pi@10.0.0.11") hosts = dict(item.split("=", 1) for item in args.ssh) if set(hosts) != {"a", "b", "c"}: parser.error("--ssh needs a=..., b=... and c=...") @@ -433,7 +448,7 @@ def main(argv: list[str] | None = None) -> int: for number, name, run in ( (1, "1 store and forward", lambda: lab.store_and_forward(restart_sender=False)), (2, "2 sender restarts", lambda: lab.store_and_forward(restart_sender=True)), - (4, "4 through the middle", lambda: lab.through_the_middle(args.require_hop_receipt)), + (4, "4 through the middle", lambda: lab.through_the_middle(args.allow_missing_hop_receipt)), ): if number not in wanted: continue