diff --git a/bridges/mm_client_check.py b/bridges/mm_client_check.py index bce8921..1471672 100644 --- a/bridges/mm_client_check.py +++ b/bridges/mm_client_check.py @@ -232,6 +232,18 @@ def _decode(blob: bytes, key: bytes, record: bytes) -> None: print(f" saved to {out} for offline analysis") +def _select(refs, which: str): + """Pick which events `--download` fetches.""" + which = which.strip().lower() + if which == "first": + return refs[:1] + if which == "all": + return refs + if which == "largest": + return [max(refs, key=lambda r: r.size)] + return [r for r in refs if r.key_hex.lower() == which] + + def step(label: str, fn): """Run one read, report how long it took and what it returned.""" t0 = time.monotonic() @@ -259,7 +271,12 @@ def main() -> int: "and negotiates its own rate.") ap.add_argument("--timeout", type=float, default=10.0) ap.add_argument("--download", action="store_true", - help="also download the first stored event (read-only)") + help="also walk the event chain and download (read-only)") + ap.add_argument("--event", default="first", metavar="WHICH", + help="which event --download fetches: a key in hex " + "(e.g. 055d4a83), 'first', 'largest', or 'all'. " + "'largest' is the one worth running on an unfamiliar " + "unit -- it is what exercises offsets past 64 KB.") ap.add_argument("--capture", metavar="DIR", help="also write a raw_bw_*/raw_s3_*.bin pair to DIR, so " "this run can become a test fixture. Worth doing on " @@ -291,6 +308,12 @@ def main() -> int: try: info = step("connect() identity", mm.connect) + if info is None: + print("\n Nothing answered. If the port opened but no frame came back,") + print(" check it is actually a Micromate and not another CDC-ACM device —") + print(" /dev/ttyACM* numbering shifts when anything else is plugged in.") + print(" `ls -l /dev/serial/by-id/` names each device and is stable.") + return 3 if info: print(f" serial={info.serial} model={info.model} " f"fw={info.firmware_line} monitoring={info.monitoring}") @@ -314,25 +337,67 @@ def main() -> int: if setups: print(f" first={setups[0]!r} last={setups[-1]!r}") + # ── SUB 0x06: is content[0:4] really the event count? ───────────── + # Two samples on one unit said yes, and THOR reads it BEFORE the chain + # walk then stops without ever reading the sentinel. A third value + # either confirms it or kills it. + claimed = None + raw06 = step("0x06 storage range", mm.protocol.read_storage_range) + if raw06: + c = _content(raw06) + claimed = int.from_bytes(c[0:4], "big") + print(f" content[0:4] = {claimed} <- CANDIDATE: event count") + print(f" content[4:8] = {int.from_bytes(c[4:8],'big')} " + f"<- unexplained (read 9 alongside a 6 on UM12947)") + if a.download: - print("\n event chain (read-only):") - proto = mm.protocol - proto.arm_event() - hdr = _content(proto.read_event_first()) - key, size = hdr[0:4], int.from_bytes(hdr[4:8], "big") - if not size: - print(" no events stored") - else: - rec = _content(proto.read_event_record(key)) - print(f" first event key={key.hex()} size={size} B " - f"type={_event_type(rec)}") - t1 = time.monotonic() - blob = proto.read_event_file(key, size) - dt = time.monotonic() - t1 - print(f" downloaded {len(blob)} B in {dt:.1f} s " - f"({len(blob)/dt/1024:.1f} KiB/s)") - assert len(blob) == size - _decode(blob, key, rec) + print("\n event chain (read-only), via MicromateClient:") + t1 = time.monotonic() + refs = mm.list_events() # walks to the sentinel + walk = time.monotonic() - t1 + print(f" {len(refs)} events in {walk:.1f} s " + f"({walk/max(len(refs),1):.2f} s each, 3 round trips per event)") + + if claimed is not None: + verdict = ("✓ AGREES" if claimed == len(refs) + else f"✗ DISAGREES (0x06 said {claimed})") + print(f" 0x06 count vs chain length: {verdict}") + + for ref in refs: + print(f" {ref}") + print(f" would be filed as {ref.filename}") + + if refs: + wanted = _select(refs, a.event) + if not wanted: + print(f"\n --event {a.event!r} matched nothing") + else: + print(f"\n download ({len(wanted)} of {len(refs)}, " + f"THOR's interleaved order):") + want_keys = {r.key_hex for r in wanted} + # iter_events() walks the chain; download only the selected + # events, at the cursor position THOR would be at. + for ref in mm.iter_events(): + if ref.key_hex not in want_keys: + continue + n_chunks = -(-ref.size // 1024) + note = (" <- past 64 KB, carries into params[1]" + if ref.size > 0x10000 else "") + t1 = time.monotonic() + try: + result = mm.get_event(ref) # verify=True + except Exception as e: + print(f" {ref.key_hex} {ref.size:7} B " + f"FAILED {type(e).__name__}: {e}") + continue + dt = max(time.monotonic() - t1, 1e-6) + n = sum(len(v) for v in getattr(result, "samples", {}).values()) + err = mm.decode_error(ref, result) + check = ("PVS %+.4f%%" % (100 * err) if err is not None + else "PVS n/a (histogram)") + print(f" {ref.key_hex} {ref.record_type:9} " + f"{ref.size:7} B {n_chunks:3} chunks {dt:5.1f} s " + f"{n:6} samples {check}{note}") except ProtocolError as e: print(f"\n ABORTED {type(e).__name__}: {e}") @@ -354,8 +419,11 @@ def main() -> int: print(" RX55 (TCP) 36 reads list_setups 16.05 s download 1.6 KiB/s") print(" The modem needs FEWER reads, not more -- it buffers ~1 s and then") print(" forwards one large segment, where CDC-ACM delivers many small ones.") - print(" Cost is ~0.65 s PER ROUND TRIP regardless of payload size, so what") - print(" matters over cellular is the number of commands, not the bytes.\n") + print(" Cellular cost, measured: ~0.21 s per request + ~2,350 B/s.") + print(" So STATUS work is round-trip bound (minimise commands) but a") + print(" DOWNLOAD is throughput bound -- 16 KB chunks save ~30% on a large") + print(" event, not 14x. An earlier note claiming cost was independent of") + print(" payload was fitted only to sub-1 KB responses.\n") return 0 diff --git a/docs/micromate_protocol_reference.md b/docs/micromate_protocol_reference.md index 06ba509..561daaa 100644 --- a/docs/micromate_protocol_reference.md +++ b/docs/micromate_protocol_reference.md @@ -694,19 +694,94 @@ key size(1E) chunks sum(offsets) last offset 055d4a86 6092 6 6092 0x03cc ``` -**These are probably two different modes, not a contradiction.** THOR's -`offset_hi` is the chunk length (`0x04`, `0x03`, `0x00` …). Our single-request -probes set `offset_hi = 0x10` — which in Series III is precisely the -bulk-stream marker `build_5a_frame()` writes raw. So `0x10XX` plausibly means -"stream until done" and returns **several** frames, which the parser of the day -concatenated into the 11,049 bytes recorded below. That reconciles both -observations, but it is a hypothesis: those 2026-09-23 captures never landed in -the repo, so it cannot be re-derived from bytes on disk. +> #### ⚠ RETRACTED 2026-10-02 — there is no streaming mode +> +> This section previously hypothesised that `offset_hi = 0x10` meant "stream +> until done" and returned several frames, reconciling THOR's chunk loop with +> our 2026-09-23 single-request observation. **Measured directly on UM20147, +> it does not:** +> +> | request | one frame returns | file bytes | +> |---|---|---| +> | `offset = 0x1014` (4,116) | 4,127 B data | **4,116** | +> | `offset = 0x111c` (4,380) | 4,391 B data | **4,380** | +> +> **`offset` is simply a byte count.** The device returns exactly `offset + 11` +> bytes in one frame, and `page_key` is `offset // 256`. `0x1000` is not a +> marker — it is part of the number. One model covers every `0x5A` request ever +> observed, THOR's and ours. +> +> So the 2026-09-23 note that `offset_word = 0x1000 + 2 × pages` returned an +> entire 11 KB event is **wrong**: `0x102C` is 4,140, and 4,140 bytes is what it +> would have returned. Most likely the 11,049 figure was the whole capture +> rather than one frame's payload. Those captures never landed in the repo, so +> the error cannot be traced further — but it does not need to be, because the +> live measurement is unambiguous. +> +> 🔑 **And the negative result turned up something better. See below.** + +### 🔑 1024 bytes per request is THOR's choice, not the device's limit + +The retraction above has a payoff. In establishing that `offset` is a plain byte +count, UM20147 served **4,380 bytes in a single frame** without being asked +twice. THOR uses 1024; nothing about the device requires it. + +That matters because of the round-trip cost: ~0.65 s each over cellular, +regardless of payload. A 72,560-byte event is **71 requests ≈ 46 seconds** at +THOR's chunk size. The offset field is a **uint16**, so the structural ceiling is +**65,535 bytes per request** — which would make that same event **2 requests, +~1.3 seconds**. + +#### ✅ Confirmed over a cellular modem too (2026-10-02) + +UM12947 on an RX55, every size checked byte-for-byte against a 1024 B control: +**1,024 / 2,048 / 4,096 / 8,192 / 14,176 all served in one frame, intact.** Its +largest event is 14,176 B, so the 16,384 ceiling itself was not reached — but a +14 KB single response crossing the PAD unscathed is the thing that needed +proving. The reader reads to frame completion rather than using idle-gap +detection, which is why ~10 TCP segments reassemble without special handling. + +#### ✅ The ceiling is 16,384 bytes, and over it the device CLAMPS SILENTLY + +Measured on UM20147, each size checked byte-for-byte against a known-good +download: + +| requested | served | | +|---|---|---| +| 1,024 | 1,024 | ✅ | +| 2,048 | 2,048 | ✅ | +| 4,096 | 4,096 | ✅ | +| 8,192 | 8,192 | ✅ | +| **16,384** | **16,384** | ✅ | +| 32,768 | **16,384** | ⚠ clamped | +| 65,535 | **16,384** | ⚠ clamped | + +⚠ **The clamp is silent and the bytes are correct.** A 32,768-byte request +returns 16,384 bytes of perfectly good data and no error. There is nothing in the +response to say it was truncated — the length is the only signal. + +**That is what dictates how the download loop must be written.** A loop striding +by a fixed chunk size would either fail on a clamp or, worse, skip the bytes it +never collected. `read_event_file()` therefore tracks its offset by **bytes +received**, which makes a clamp cost one extra request rather than corrupting the +file — and makes the loop self-correcting against any short response, which is +precisely the failure mode that has bitten the Series III side repeatedly. + +`CHUNK_SIZE` is now **16,384**. For UM20147's events: + +| event | THOR's 1024 | 16,384 | cellular | +|---|---|---|---| +| 4,796 B | 5 requests | **1** | ~3.3 s → ~0.7 s | +| 30,230 B | 30 | **2** | ~20 s → ~1.3 s | +| 72,560 B | 71 | **5** | ~46 s → ~3.2 s | + +⚠ **Measured on one unit, over USB.** `THOR_CHUNK_SIZE = 1024` remains available +and the replay tests pin it, so reproducing THOR's exact traffic is still one +argument away. A unit that clamps lower than 16,384 costs extra requests, not a +failure. **Implement THOR's chunked form.** It is verified byte-exact across six events -and five distinct sizes, and it is what the firmware runs every day. The -single-request form is worth one bench test as an optimisation — `offset_hi = -0x10` and count the frames — but not worth depending on first. +and five distinct sizes, and it is what the firmware runs every day. ### The offset word is a LENGTH, not a position @@ -811,7 +886,37 @@ across reads, and the modem splits **less** than USB does. The client reads to frame completion rather than using `read_until_idle`'s idle-gap detection, and handled both without a retry. -### 🔑 ~0.65 s per round trip over cellular, independent of payload +### 🔑 The cellular cost model — ~0.21 s per request + ~2,350 B/s + +> #### ⚠ CORRECTED 2026-10-02 — it is NOT independent of payload +> +> This section previously read **"~0.65 s per round trip, independent of payload +> size"**, and concluded *"minimise round trips, not bytes"*. The first claim is +> wrong and the second is too strong. +> +> Every measurement behind it had a **small** payload — 59 B status, 266 B setup +> records, 1024 B download chunks. With the byte term small and similar across +> all of them, per-command cost looked constant. It was a narrow-range fit +> extrapolated past its evidence. +> +> Measured against a 14,176 B single response on UM12947 over an RX55: +> +> | request | measured | model | +> |---|---|---| +> | 1,024 B | 0.67 s | 0.65 s | +> | 2,048 B | 1.22 s | 1.08 s | +> | 4,096 B | 2.13 s | 1.96 s | +> | 8,192 B | 3.59 s | 3.70 s | +> | 14,176 B | 6.25 s | 6.25 s | +> +> **`t ≈ 0.21 s + bytes / 2,350`** — fits within ±11% across a 14× size range. +> +> The old 0.65 s figure was *right for a 1024-byte response* and is simply that +> model evaluated at 1 KB. Round trips still cost real money (0.21 s each, and +> 24 of them for a setup walk is still 16 s), but **bytes cost more than round +> trips on anything over ~500 B**, and that reverses the design advice: for +> *status* work minimise commands; for *downloads* the floor is throughput and no +> amount of batching beats it. This is the number that matters for SFM's design. @@ -994,6 +1099,45 @@ its counter resets, so keys are reused *within* a unit, which is why `ach_state.json` tracks `max_downloaded_key` per serial. Series IV inherits that and adds cross-unit collision on top. +### ⚠ 🔑 The chunk offset is a uint32 — and a 64 KB cap was one event away + +UM20147 holds a **72,560-byte** event (`055d4a83`). The `0x5A` chunk offset was +implemented as a **uint16 at `params[2:4]`**, because every offset THOR was +observed to send fits in two bytes — the largest is `0x3400` (13,312). That caps +a download at **65,536 bytes**, so that event could not have been fetched at all. + +`params[0:4]` is demonstrably **one 4-byte field**: chunk 0 puts the 4-byte event +key there. Writing the offset as a uint32 BE in the same slot is **byte-identical +for every offset below 65,536** — so it changes nothing that was verified against +THOR's 74 captured frames — and extends the range to 4 GB. + +✅ **Confirmed on hardware 2026-10-01.** UM20147's 72,560-byte event +(`055d4a83`) downloaded in **71 chunks** at exactly its promised size — chunks +64–70 carry a `0x01` in `params[1]`, and the device served them without +complaint. The field is a uint32; it was inference for one commit. + +🔑 **And it prices a large event over cellular.** 71 chunks × ~0.65 s ≈ **46 +seconds** for one 72 KB histogram, against 0.2 s over USB. That is the +round-trip cost model doing real work: the bytes are nothing, the commands are +everything. A unit holding several events this size is a multi-minute call, and +the untested single-request `offset_hi = 0x10` streaming mode would collapse it +to one round trip — which moves that two-minute bench test up the list. + +⚠ **At offset 1 MiB `params[1]` is `0x10`**, which must be escaped on the wire as +`10 10`. Unreachable below 64 KB, so the uint32 change is what first makes that +case possible — and an unescaped `0x10` in `5A` params is the exact bug that cost +the Series III walk a release. + +**This is the Series III 64 KB page-boundary bug wearing a different hat.** There, +`parse_strt_end_offset()` discards the key's page byte and the walk crashes once a +unit's buffer crosses 64 KB; that is **still open**. The transferable lesson: +*an address field whose high bytes are zero in every capture is not a narrow +field, it is an untested one.* Both bugs are the same mistake, found twice, in +code written years apart. + +**The test is sitting on a desk:** download `055d4a83` from UM20147 and see +whether 71 chunks come back. + ### Still not covered - **A unit that is monitoring**, and a unit with a nearly-full event buffer. - **The inbound call-home session** — still the one protocol unknown. diff --git a/micromate/client.py b/micromate/client.py index 8149863..67f47e3 100644 --- a/micromate/client.py +++ b/micromate/client.py @@ -32,11 +32,14 @@ from __future__ import annotations import datetime import logging +import math +import os +import struct from typing import Optional from minimateplus.transport import BaseTransport -from .models import MicromateDeviceInfo, MicromateState +from .models import MicromateDeviceInfo, MicromateEventRef, MicromateState from .protocol import MicromateProtocol, ProtocolError log = logging.getLogger(__name__) @@ -57,6 +60,27 @@ _MS_BATTERY = slice(34, 36) # uint16 BE, volts × 100 _MS_MEM_TOTAL = slice(36, 40) # uint32 BE _MS_MEM_FREE = slice(40, 44) # uint32 BE +# `SUB 0x0C` record, relative to content. Established against 7 events across +# both firmware lines. +_REC_DAY, _REC_MONTH, _REC_YEAR = 0, 1, slice(2, 4) +_REC_UNKNOWN_4 = 4 # ⚠ same shape as 0x1C's content[6]; undecoded +_REC_HOUR, _REC_MIN, _REC_SEC = 5, 6, 7 +_REC_TYPE = 11 # 0x07 waveform, 0x08 histogram +_REC_LOCATION = 12 +_REC_SETUP = 34 +_REC_SERIAL = 76 +_RECORD_TYPES = {0x07: "waveform", 0x08: "histogram"} + +# The channel labels the record carries, in the order they appear. The peak +# float sits `label + 6`; the peak vector sum sits 12 bytes BEFORE "Tran". +_REC_CHANNELS = (b"Tran", b"Vert", b"Long", b"Mic") +_REC_PVS_BACK = 12 + +# A chain walk terminates on an all-zero key. The ceilings below are guards +# against a device cursor that never advances, not fleet limits. +_NULL_KEY = bytes(4) +_MAX_EVENTS = 4096 + # A setup-list walk that does not terminate is a bug, not a big fleet. The # bench unit holds 22 setups; this is a generous ceiling, not a limit. _MAX_SETUPS = 512 @@ -71,6 +95,10 @@ def _cstring(buf: bytes, offset: int = 0) -> str: return buf[offset:].split(b"\x00")[0].decode("ascii", "replace").strip() +class DecodeMismatch(ProtocolError): + """Our decoded peak disagrees with the one the device computed itself.""" + + class MicromateClient: """High-level read-only client for one Micromate. @@ -88,6 +116,7 @@ class MicromateClient: transport, recv_timeout=recv_timeout, strict_checksums=strict_checksums ) self._firmware_line: Optional[str] = None + self._serial: Optional[str] = None # ── Lifecycle ───────────────────────────────────────────────────────────── @@ -140,6 +169,7 @@ class MicromateClient: manufacturer, model = self._parse_poll(poll.data) serial = _cstring(_content(self._proto.read_serial())) + self._serial = serial monitoring = self._parse_state(self._proto.read_state()) info = MicromateDeviceInfo( @@ -293,3 +323,253 @@ class MicromateClient: f"setup list did not terminate after {_MAX_SETUPS} entries — the " f"device cursor is not advancing" ) + + # ── Events ──────────────────────────────────────────────────────────────── + + def serial(self) -> str: + """The unit's serial, cached from `connect()` or read on demand. + + Needed by anything that handles an event, because an event key is + ambiguous without it — see `MicromateEventRef`. + """ + if self._serial is None: + self._serial = _cstring(_content(self._proto.read_serial())) + return self._serial + + def list_events(self, *, with_records: bool = True) -> list[MicromateEventRef]: + """Walk the event chain. `0x93 → 1E`, then `0x93 → 1F` until the sentinel. + + THOR sends `0x93` before **every** chain read, and this mirrors that. + The chain ends on an all-zero key. + + ⚠ `with_records=True` costs **one extra round trip per event** for the + `0x0C` read, and over cellular a round trip is ~0.65 s regardless of + size. On a unit with 40 events that is the difference between ~52 s and + ~78 s. Pass False when you only need "what is here and how big" — but + note the **record is where the type and timestamp live**, so without it + `ref.filename` is None and `get_event()` cannot pick a suffix. + + ⚠ **To download, prefer `iter_events()`.** This walks the whole chain + first; THOR interleaves, downloading each event before advancing. + `0x5A` addresses an event by key, so downloading afterwards *should* + work — but "should" is doing real work in that sentence, and the + interleaved order is the one with captures behind it. See + `iter_events()`. + """ + serial = self.serial() + refs: list[MicromateEventRef] = [] + + for i in range(_MAX_EVENTS): + self._proto.arm_event() + raw = (self._proto.read_event_first() if i == 0 + else self._proto.read_event_next()) + c = _content(raw) + if len(c) < 8: + raise ProtocolError( + f"chain entry {i}: {len(c)} B of content, need 8 (key + size)" + ) + key, size = c[0:4], int.from_bytes(c[4:8], "big") + if key == _NULL_KEY: + return refs # the sentinel, not an error + + ref = MicromateEventRef(index=i, key=key, size=size, serial=serial) + if with_records: + self._read_record_into(ref) + refs.append(ref) + + raise ProtocolError( + f"event chain did not terminate after {_MAX_EVENTS} entries — the " + f"device cursor is not advancing" + ) + + def iter_events(self, *, with_records: bool = True): + """Walk the chain, yielding each event **at the cursor position THOR uses.** + + for ref in mm.iter_events(): + if ref.is_histogram: + continue + data = mm.download_event(ref) # ← safe here + + Why this exists alongside `list_events()`: THOR's captured order is + + 0x93 → 1E → 0C → 5A×n → 0x93 → 1F → 0C → 5A×n → … + + — it downloads each event *before* advancing the chain. `list_events()` + walks to the end first, which is fine for browsing (the browse walk is + separately attested) but means a later download happens with the device + cursor parked past the event. `0x5A` is key-addressed, so it very + probably does not care; nothing observed says it does, and nothing + observed says it does not. + + Downloading inside this loop reproduces THOR's sequence exactly, so it + is the path to use when it matters. ⚠ Do not advance the generator + before finishing with the event it yielded. + """ + serial = self.serial() + for i in range(_MAX_EVENTS): + self._proto.arm_event() + raw = (self._proto.read_event_first() if i == 0 + else self._proto.read_event_next()) + c = _content(raw) + if len(c) < 8: + raise ProtocolError( + f"chain entry {i}: {len(c)} B of content, need 8 (key + size)" + ) + key, size = c[0:4], int.from_bytes(c[4:8], "big") + if key == _NULL_KEY: + return + ref = MicromateEventRef(index=i, key=key, size=size, serial=serial) + if with_records: + self._read_record_into(ref) + yield ref + + raise ProtocolError( + f"event chain did not terminate after {_MAX_EVENTS} entries — the " + f"device cursor is not advancing" + ) + + def _read_record_into(self, ref: MicromateEventRef) -> None: + """`SUB 0x0C` — 210 B of content: timestamp, type, names, peaks.""" + raw = self._proto.read_event_record(ref.key) + c = _content(raw) + if len(c) <= _REC_SERIAL: + raise ProtocolError(f"event record is {len(c)} B, too short to decode") + ref.raw_record = raw + + ref.record_type = _RECORD_TYPES.get(c[_REC_TYPE]) + if ref.record_type is None: + # Worth saying out loud rather than filing the event as a waveform: + # the suffix decides which codec runs. + log.warning("event %s: unknown record type 0x%02x at content[%d]", + ref.key_hex, c[_REC_TYPE], _REC_TYPE) + + try: + ref.timestamp = datetime.datetime( + year=int.from_bytes(c[_REC_YEAR], "big"), + month=c[_REC_MONTH], day=c[_REC_DAY], + hour=c[_REC_HOUR], minute=c[_REC_MIN], second=c[_REC_SEC], + ) + except ValueError as e: + log.warning("event %s: bad timestamp (%s): %s", + ref.key_hex, e, c[:8].hex(" ")) + + ref.sensor_location = _cstring(c, _REC_LOCATION) or None + ref.setup = _cstring(c, _REC_SETUP) or None + # Prefer the record's own serial over the cached one — they have always + # agreed, but the record is the event's own account of where it came from. + if rec_serial := _cstring(c, _REC_SERIAL): + ref.serial = rec_serial + + peaks: dict[str, float] = {} + for label in _REC_CHANNELS: + i = c.find(label) + if i < 0 or i + len(label) + 10 > len(c): + continue + peaks[label.decode()] = struct.unpack( + ">f", c[i + len(label) + 2: i + len(label) + 6])[0] + ref.peaks_ips = peaks or None + + tran = c.find(b"Tran") + if tran >= _REC_PVS_BACK: + ref.peak_vector_sum_ips = struct.unpack( + ">f", c[tran - _REC_PVS_BACK: tran - _REC_PVS_BACK + 4])[0] + + def download_event(self, ref: MicromateEventRef, *, + chunk_size: Optional[int] = None) -> bytes: + """The raw `.IDFW`/`.IDFH` bytes, exactly as THOR would have stored them. + + Feeds `micromate.idf_file.read_idf_file()` and `/db/import/idf_file` + unchanged — no new codec work is needed for a directly downloaded event. + + `chunk_size` defaults to the measured device ceiling (16,384 B), which is + 16x THOR's 1024 and therefore ~14x fewer round trips on a large event. + Pass `THOR_CHUNK_SIZE` to reproduce THOR's wire traffic exactly, or a + smaller value on a link where big responses are not surviving. + """ + kw = {} if chunk_size is None else {"chunk_size": chunk_size} + return self._proto.read_event_file(ref.key, ref.size, **kw) + + def get_event(self, ref: MicromateEventRef, *, verify: bool = True, + tolerance: float = 0.01, chunk_size: Optional[int] = None): + """Download and decode one event. + + Returns the codec's `IdfReadResult`. Needs `ref.record_type`, since + `read_idf_file()` dispatches on the filename suffix and there is no + filename on the wire — so call `list_events(with_records=True)` first. + + ⚠ `verify=True` re-computes the **peak vector sum** from the decoded + samples and compares it against the float the *device* put in the `0x0C` + record. Those are two independent computations over the same samples — + the device's from its own firmware, ours from our codec — so a + disagreement means our decode is wrong. Measured agreement on the bench + events is **0.000%**. + + This is cheap insurance in a codebase whose decode failures have + historically been *silent*: unhandled block tags shorten a channel and + nothing raises. ⚠ It is a decode-correctness check, **not** a + truncation detector — a channel cut after its peak still yields the + right PVS. + + Histograms are not verified: `samples` is empty for them. + """ + if not ref.record_type: + raise ValueError( + f"event {ref.key_hex}: record_type is unknown, so the codec " + f"cannot be dispatched. Use list_events(with_records=True)." + ) + blob = self.download_event(ref, chunk_size=chunk_size) + + import tempfile + from .idf_file import read_idf_file + + # read_idf_file dispatches on the suffix, so the bytes need a name. + with tempfile.NamedTemporaryFile(suffix=ref.suffix, delete=False) as f: + f.write(blob) + tmp = f.name + try: + result = read_idf_file(tmp) + finally: + os.unlink(tmp) + + if verify and not ref.is_histogram: + err = self.decode_error(ref, result) + if err is not None and abs(err) > tolerance: + raise DecodeMismatch( + f"event {ref.uid}: decoded peak vector sum differs from the " + f"device's own by {100 * err:+.3f}% (tolerance " + f"{100 * tolerance:.1f}%) — the decode is suspect, not the " + f"device. Stored {ref.peak_vector_sum_ips:.5f} in/s." + ) + if err is not None: + log.debug("event %s: PVS agrees to %+.4f%%", ref.uid, 100 * err) + return result + + @staticmethod + def decode_error(ref: MicromateEventRef, result) -> Optional[float]: + """Relative error between our decoded PVS and the device's stored one. + + None when either side is unavailable. Positive means the device's + figure is higher than ours. + """ + from .idf_file import geo_count_to_ips + + stored = ref.peak_vector_sum_ips + if not stored or not getattr(result, "samples", None): + return None + + ch = {k.lower(): v for k, v in result.samples.items()} + try: + t, v, l = ch["tran"], ch["vert"], ch["long"] + except KeyError: + return None + n = min(len(t), len(v), len(l)) + if not n: + return None + + pvs = max( + math.sqrt(geo_count_to_ips(t[i]) ** 2 + + geo_count_to_ips(v[i]) ** 2 + + geo_count_to_ips(l[i]) ** 2) + for i in range(n) + ) + return (stored - pvs) / pvs if pvs else None diff --git a/micromate/models.py b/micromate/models.py index 49e4c25..7b56c2d 100644 --- a/micromate/models.py +++ b/micromate/models.py @@ -478,3 +478,76 @@ class MicromateState: if frac is not None: bits.append(f"memory {frac * 100:.1f}% used") return " ".join(bits) + + +@dataclass +class MicromateEventRef: + """One entry in a unit's event chain, from `1E`/`1F` and optionally `0x0C`. + + ⚠ **`key` is NOT unique across units.** The event counter starts from the + same value on every Micromate — UM12947 and UM20147 both have an event + `055d4a81`, with different sizes and different contents. Use `uid`, or key + on `(serial, key_hex)`, for anything that stores or deduplicates. A store + keyed on the event key alone silently treats one unit's event as a duplicate + of another's, and nothing raises. + """ + + index: int + key: bytes # 4-byte event key from the chain walk + size: int # bytes the device will send for this event + serial: Optional[str] = None # the unit, because `key` alone is ambiguous + + # From `SUB 0x0C` — one extra round trip per event, so optional. + record_type: Optional[str] = None # "waveform" | "histogram" + timestamp: Optional[datetime.datetime] = None + setup: Optional[str] = None # setup file name, no extension + sensor_location: Optional[str] = None + peak_vector_sum_ips: Optional[float] = None # per-sample PVS, device-computed + peaks_ips: Optional[Dict[str, float]] = None # {"Tran": …, "Vert": …, …} + raw_record: Optional[bytes] = field(default=None, repr=False) + + @property + def key_hex(self) -> str: + return self.key.hex() + + @property + def uid(self) -> str: + """`SERIAL:key` — safe to use as a primary key. See the class note.""" + return f"{self.serial or '?'}:{self.key_hex}" + + @property + def is_histogram(self) -> Optional[bool]: + if self.record_type is None: + return None + return self.record_type == "histogram" + + @property + def suffix(self) -> Optional[str]: + return {"waveform": ".IDFW", "histogram": ".IDFH"}.get(self.record_type or "") + + @property + def filename(self) -> Optional[str]: + """The name THOR would have given this event. + + `_.IDF{W,H}` — e.g. + `UM12947_20260923163319.IDFW`. Verified against the production store + for all five bench events. + + ⚠ The type comes from the **protocol**, not the payload, so it has to be + carried here from the `0x0C` read. Returns None without it: guessing + the suffix would file a histogram as a waveform, and `read_idf_file()` + dispatches on exactly that. + """ + if not (self.serial and self.timestamp and self.suffix): + return None + return f"{self.serial}_{self.timestamp:%Y%m%d%H%M%S}{self.suffix}" + + def __str__(self) -> str: + bits = [self.uid, f"{self.size} B"] + if self.record_type: + bits.append(self.record_type) + if self.timestamp: + bits.append(self.timestamp.strftime("%Y-%m-%d %H:%M:%S")) + if self.peak_vector_sum_ips is not None: + bits.append(f"PVS {self.peak_vector_sum_ips:.4f} in/s") + return " ".join(bits) diff --git a/micromate/protocol.py b/micromate/protocol.py index 6e35a2f..215b02b 100644 --- a/micromate/protocol.py +++ b/micromate/protocol.py @@ -106,8 +106,30 @@ OBSERVED_DATA_LEN = { # bulk stream; here THOR sends it on every chain read, browse or download. EVENT_TOKEN = 0xFE -# `SUB 0x5A` chunk size, in bytes of file payload per response. -CHUNK_SIZE = 1024 +# `SUB 0x5A` request size, in bytes of file payload per response. +# +# ⚠ **16,384, not THOR's 1024.** Measured on UM20147 (2026-10-02): requests of +# 1024 / 2048 / 4096 / 8192 / 16384 all returned byte-identical data against a +# known-good download, and anything larger is **silently clamped to 16,384** — +# correct bytes, short length, no error. +# +# ⚠ The win is real but MODEST, and smaller than a first reading suggests. +# Measured over an RX55, cost per request is ~0.21 s + bytes/2350 — so a download +# is throughput-bound, not round-trip bound. A 72,560-byte event costs ~46 s at +# THOR's 1024 B and ~32 s at 16,384 B: a **30% saving**, not 14x, because ~31 s of +# it is bytes on the wire and no amount of batching addresses that. +# +# Still worth having: 30% faster, and 14x fewer requests is 14x fewer chances for +# a link to drop mid-download. +# +# ⚠ Measured on ONE unit, over USB. The loop below is driven by bytes received +# rather than chunk index, so a unit that clamps lower simply takes more +# requests instead of failing — which is what makes raising this safe. +CHUNK_SIZE = 16384 + +# THOR's value. Pass `chunk_size=THOR_CHUNK_SIZE` to reproduce its wire traffic +# exactly; the replay test does. +THOR_CHUNK_SIZE = 1024 # Every `0x5A` response prefixes the file bytes with 11 bytes of header. _CHUNK_PREFIX = 11 @@ -170,15 +192,38 @@ def chunk_params(key4: bytes, byte_offset: int) -> bytes: """`0x5A`: the key opens the file, then a byte offset walks it. Chunk 0 carries the event key at params[0:4] — that is what says "from the - beginning". Later chunks carry a uint16 BE byte offset at params[2:4]. + beginning". Later chunks carry the byte offset **in that same 4-byte slot**, + as a uint32 BE. + + ⚠ **Written as a uint32 deliberately, and this matters above 64 KB.** Every + offset THOR was observed to send fits in two bytes — the largest was `0x3400` + (13,312) — so `params[0:2]` was always `00 00` and the field looks like a + uint16 at `params[2:4]`. Reading it that way caps a download at **65,536 + bytes**, and UM20147 currently holds a **72,560-byte** event, so that cap is + not hypothetical. + + A uint32 here is **byte-identical for every offset below 65,536**, so it + changes nothing that was verified against THOR's frames (the replay test + asserts all 74 of them) and extends the range to 4 GB. + + ✅ **Confirmed on hardware 2026-10-01.** UM20147's 72,560-byte event + downloaded in **71 chunks** at the exact promised size, which exercises the + carry into `params[1]` on chunks 64–70. It was inference for one commit and + is now evidence. + + ⚠ This is the **Series III 64 KB page-boundary bug in a new guise** — there, + `parse_strt_end_offset()` discards the key's page byte and the `5A` walk + crashes once a unit's buffer crosses 64 KB, which is *still open* on that + side. The lesson that transfers: an address field whose high bytes are zero + in every capture is not a narrow field, it is an untested one. """ if byte_offset == 0: if len(key4) != 4: raise ValueError(f"key4 must be 4 bytes, got {len(key4)}") return key4 + bytes(6) - if not 0 <= byte_offset <= 0xFFFF: - raise ValueError(f"byte_offset must fit in uint16, got {byte_offset}") - return bytes(2) + struct.pack(">H", byte_offset) + bytes(6) + if not 0 <= byte_offset <= 0xFFFFFFFF: + raise ValueError(f"byte_offset must fit in uint32, got {byte_offset}") + return struct.pack(">I", byte_offset) + bytes(6) # ── Protocol ────────────────────────────────────────────────────────────────── @@ -350,14 +395,15 @@ class MicromateProtocol: # ── Bulk download ───────────────────────────────────────────────────────── - def read_event_file(self, key4: bytes, size: int) -> bytes: + def read_event_file(self, key4: bytes, size: int, *, + chunk_size: int = CHUNK_SIZE) -> bytes: """`0x5A` → `0xA5`. The `.IDFW`/`.IDFH` file, byte for byte. `size` is the 4 bytes after the key in the `1E`/`1F` response. Returns exactly that many bytes, or raises `ShortRead`. - A bounded chunk walk — `ceil(size / 1024)` requests, each asking for - `min(1024, remaining)` bytes: + A bounded walk — `ceil(size / chunk_size)` requests, each asking for + `min(chunk_size, remaining)` bytes: offset = the byte count wanted (NOT an address) params = the key on chunk 0, then a uint16 BE byte offset @@ -378,36 +424,54 @@ class MicromateProtocol: if size <= 0: raise ValueError(f"size must be positive, got {size}") + if chunk_size < 1: + raise ValueError(f"chunk_size must be positive, got {chunk_size}") + + # ⚠ Driven by BYTES RECEIVED, not by chunk index. + # + # The device silently clamps an over-large request: ask for 32,768 and it + # returns exactly 16,384 — correct bytes, short length, no error. A loop + # that strides by a fixed chunk size would either fail on that or, worse, + # skip the bytes it never collected. Tracking the offset by what actually + # arrived makes the clamp a non-event: it just takes another request. + # + # That also makes this self-correcting against any short response, which + # is the failure mode this codebase has been bitten by repeatedly on the + # Series III side — there, a short read surfaced as a silently truncated + # channel. out = bytearray() - n_chunks = math.ceil(size / CHUNK_SIZE) - for i in range(n_chunks): - want = min(CHUNK_SIZE, size - i * CHUNK_SIZE) + requests = 0 + while len(out) < size: + want = min(chunk_size, size - len(out)) data = self._read( SUB_BULK_DOWNLOAD, offset=want, - params=chunk_params(key4, i * CHUNK_SIZE), + params=chunk_params(key4, len(out)), ) + requests += 1 if len(data) < _CHUNK_PREFIX: raise ShortRead( - f"chunk {i + 1}/{n_chunks} of {key4.hex()}: " - f"{len(data)} B is too short to hold a chunk header" + f"{key4.hex()} at offset {len(out)}: {len(data)} B is too " + f"short to hold a chunk header" ) body = data[_CHUNK_PREFIX:] - if len(body) != want: - # Worth being loud: a silently short event is the failure mode - # this project has been bitten by repeatedly on the Series III - # side, and here the expected length is known up front. + if not body: + # No progress at all — continuing would spin forever. raise ShortRead( - f"chunk {i + 1}/{n_chunks} of {key4.hex()}: asked for " - f"{want} B, got {len(body)}" + f"{key4.hex()} at offset {len(out)}: asked for {want} B and " + f"got none; {len(out)} of {size} B assembled" ) + if len(body) < want: + log.debug("%s: asked %d B at offset %d, served %d — clamped", + key4.hex(), want, len(out), len(body)) out += body if len(out) != size: raise ShortRead( f"{key4.hex()}: assembled {len(out)} B, device promised {size}" ) - log.debug("downloaded %s: %d B in %d chunks", key4.hex(), len(out), n_chunks) + log.debug("downloaded %s: %d B in %d request(s)", + key4.hex(), len(out), requests) return bytes(out) # ── Plumbing ────────────────────────────────────────────────────────────── diff --git a/scratch/mm_stream_probe.py b/scratch/mm_stream_probe.py new file mode 100644 index 0000000..7efed83 --- /dev/null +++ b/scratch/mm_stream_probe.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +""" +mm_stream_probe.py — how many bytes will `SUB 0x5A` serve in one request? + +Settled 2026-10-02: there is NO streaming mode +---------------------------------------------- +This script started out asking whether `offset_hi = 0x10` meant "stream until +done", because our 2026-09-23 notes recorded a single request appearing to return +an entire 11 KB event. Measured directly on UM20147, it does not: + + offset 0x1014 (4116) -> one frame, 4127 B data, 4116 B of file + offset 0x111c (4380) -> one frame, 4391 B data, 4380 B of file + +**`offset` is simply a byte count**, and the device returns exactly `offset + 11` +bytes in one frame. `0x1000` is not a marker; it is part of the number. The +2026-09-23 reading was wrong, and the protocol reference now says so. + +The useful question it turned up +------------------------------- +**1024 bytes per request is THOR's choice, not the device's limit.** The unit +served 4,380 bytes in a single frame without being asked twice. Since a round +trip over cellular costs ~0.65 s regardless of payload, and UM20147's +72,560-byte event is 71 chunks ≈ 46 seconds, the ceiling on one request is worth +knowing precisely: every doubling halves the dominant cost. + +So this now walks ascending request sizes against one event and **checks each +against a known-good chunk-loop download** — a pass means byte-identical output, +not merely a plausible length. The offset field is a uint16, so 65,535 is the +structural maximum. + +⚠ **Read-only.** `0x5A` is a read we have sent thousands of times; the only new +thing is a larger value in its offset field. Nothing here writes, erases or +changes monitoring state. It re-POLLs at the end, because the honest risk is +leaving the session in an odd state and the script should say so. + +Usage +----- + python3 scratch/mm_stream_probe.py /dev/ttyACM1 + python3 scratch/mm_stream_probe.py /dev/ttyACM1 --event largest +""" + +from __future__ import annotations + +import argparse +import math +import sys +import time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "bridges")) + +from micromate.client import MicromateClient, _content # noqa: E402 +from micromate.framing import MicromateFrameParser, build_request # noqa: E402 +from micromate.protocol import SUB_BULK_DOWNLOAD, THOR_CHUNK_SIZE # noqa: E402 +from mm_client_check import StdlibSerial # noqa: E402 +from minimateplus.transport import TcpTransport # noqa: E402 + +_CHUNK_PREFIX = 11 + + +def collect(transport, parser, *, idle_gap: float, deadline: float) -> list: + """Read until `idle_gap` seconds pass with no new bytes, or `deadline`.""" + frames, last = [], time.monotonic() + while time.monotonic() < deadline: + chunk = transport.read(4096) + if chunk: + frames += parser.feed(chunk) + last = time.monotonic() + continue + if time.monotonic() - last > idle_gap: + break + time.sleep(0.005) + return frames + + +def main() -> int: + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + ap.add_argument("target", help="host:port, or a serial device path") + ap.add_argument("--baud", type=int, default=115200) + ap.add_argument("--event", default="smallest", + help="a key in hex, 'smallest' (default — the gentlest " + "first test) or 'largest' (the one that matters)") + ap.add_argument("--idle-gap", type=float, default=3.0, + help="seconds of silence that end a read. A 16 KB response " + "is ~1.4 s of serial time at 115200 BEFORE the modem's " + "~1 s forwarding delay, so this is deliberately roomy; " + "too small and a slow link looks like a clamp.") + ap.add_argument("--timeout", type=float, default=90.0) + a = ap.parse_args() + + if ":" in a.target and not Path(a.target).exists(): + host, _, port = a.target.rpartition(":") + inner = TcpTransport(host, int(port), connect_timeout=10.0) + label = f"TCP {host}:{port}" + else: + inner = StdlibSerial(a.target, baud=a.baud) + label = f"serial {a.target} @ {a.baud}" + + mm = MicromateClient(inner, recv_timeout=20.0) + print(f"\n{label} READ-ONLY: chain walk + two downloads of one event\n") + mm.open() + try: + info = mm.connect(with_active_setup=False) + print(f" {info}\n") + + refs = mm.list_events() + if not refs: + print(" no events stored — nothing to download. Record one first.") + return 1 + for r in refs: + print(f" {r}") + + which = a.event.strip().lower() + if which == "smallest": + ref = min(refs, key=lambda r: r.size) + elif which == "largest": + ref = max(refs, key=lambda r: r.size) + else: + matches = [r for r in refs if r.key_hex.lower() == which] + if not matches: + print(f"\n --event {a.event!r} matched nothing") + return 2 + ref = matches[0] + + n_chunks = math.ceil(ref.size / THOR_CHUNK_SIZE) + print(f"\n target: {ref.key_hex} {ref.size} B {ref.record_type} " + f"({n_chunks} requests at THOR's chunk size)") + + # ── 1. the control: THOR's 1024-byte chunk loop ─────────────────────── + # ⚠ Pinned to THOR_CHUNK_SIZE on purpose. The library default is now + # 16,384, and using it here would make the experiment circular -- the + # "known-good" reference would share any fault with the sizes under test. + # 1024 is the size with THOR's own captures behind it and the one proven + # over a modem, so it is the control. + t0 = time.monotonic() + chunked = mm.protocol.read_event_file(ref.key, ref.size, + chunk_size=THOR_CHUNK_SIZE) + dt_chunked = time.monotonic() - t0 + print(f"\n [1] control ......... {len(chunked)} B in {dt_chunked:.2f} s " + f"({n_chunks} requests at THOR's 1024 B)") + + # ── 2. how many bytes will it serve in ONE frame? ───────────────── + # The streaming hypothesis is dead (see the module docstring): `offset` + # is simply a BYTE COUNT, and the device returns `offset + 11` bytes in + # one frame. So the real question is the ceiling -- because 1024 is + # THOR's choice, not the device's limit, and every doubling halves the + # round trips that dominate a cellular download. + print("\n [2] chunk-size ceiling — ascending single requests") + print(" each asks for N bytes from offset 0 and is checked against") + print(" the known-good download, so a pass means identical bytes.\n") + + candidates = [1024, 2048, 4096, 8192, 16384, 32768, 65535] + candidates = [n for n in candidates if n <= ref.size] or [ref.size] + if ref.size not in candidates and ref.size < 65536: + candidates.append(ref.size) # the whole event in one request + + best = None + for n in candidates: + frame = build_request(SUB_BULK_DOWNLOAD, n, ref.key + bytes(6)) + parser = MicromateFrameParser() + t0 = time.monotonic() + mm.protocol._send(frame) + got = collect(inner, parser, idle_gap=a.idle_gap, + deadline=t0 + a.timeout) + dt = time.monotonic() - t0 + + if not got: + print(f" {n:6} B no answer ({dt:.2f} s)") + continue + body = b"".join(f.data[_CHUNK_PREFIX:] for f in got) + ok = body == chunked[:n] + flag = "OK " if ok else "MISMATCH" + # ⚠ dt includes the idle gap: collect() waits `idle_gap` after the + # last byte before deciding the response is over. Subtract it to + # compare against the control, which reads to frame completion. + print(f" {n:6} B {len(got)} frame(s) {len(body):6} B back " + f"{dt:5.2f} s ({max(dt - a.idle_gap, 0):4.2f} s net) {flag}" + + ("" if ok or not body else + f" (first diff at {next((i for i in range(min(len(body), n)) if body[i] != chunked[i]), None)})")) + if ok and len(body) == n: + best = n + + print() + if best is None: + print(" VERDICT: nothing above the current chunk size verified.") + else: + now = math.ceil(ref.size / THOR_CHUNK_SIZE) + then = math.ceil(ref.size / best) + print(f" VERDICT: the device serves at least {best} B per request,") + print(f" verified byte-identical. For this {ref.size} B event that") + print(f" is {then} request(s) instead of {now}.") + if best > THOR_CHUNK_SIZE: + # ⚠ cost is ~0.21 s per request PLUS ~2,350 B/s over a modem -- + # a download is throughput-bound, so do not promise a saving + # proportional to the drop in request count. + fixed, rate = 0.212, 2348.0 + t_now = now * fixed + ref.size / rate + t_then = then * fixed + ref.size / rate + print(f" Modelled over cellular (0.21 s/request + " + f"{rate:.0f} B/s): {t_now:.1f} s -> {t_then:.1f} s, " + f"a {100 * (1 - t_then / t_now):.0f}% saving.") + print(" Throughput-bound, not round-trip bound — most of that is") + print(" bytes on the wire and batching does not touch it.") + if best >= ref.size: + print(" The WHOLE EVENT fits in one request.") + + # ── 3. is the unit still healthy? ───────────────────────────────────── + # The real risk of this experiment is leaving the session wedged, so + # check rather than assume. + print() + try: + p = mm.protocol.poll() + print(f" [3] unit still answering POLL (SUB 0x{p.sub:02x}) — " + f"session is healthy") + except Exception as e: + print(f" [3] ⚠ POLL FAILED after the probe: {type(e).__name__}: {e}") + print(" Reconnect; if that does not help, power-cycle the unit.") + return 4 + finally: + mm.close() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_micromate_events.py b/tests/test_micromate_events.py new file mode 100644 index 0000000..883ed26 --- /dev/null +++ b/tests/test_micromate_events.py @@ -0,0 +1,345 @@ +"""Event-chain tests for the Micromate (series-4) client. + +The load-bearing test here replays THOR's captured six-event download session +through `MicromateClient.iter_events()` + `get_event()` and asserts **every byte +we put on the wire matches what THOR put on the wire** — 99 frames — while also +decoding all six events and cross-checking each waveform's peak vector sum +against the one the device computed itself. + +That capture is gitignored, so those tests skip on a fresh clone; the offline +tests below use embedded real response bytes and cover the same logic. +""" +from __future__ import annotations + +import datetime +import os +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from micromate.client import CONTENT, DecodeMismatch, MicromateClient +from micromate.framing import ACK, DLE, ETX, STX, checksum, stuff +from micromate.models import MicromateEventRef +from micromate.protocol import ProtocolError + +from test_micromate_client import ScriptedTransport, SERIAL, frame + +# The real 0x0C record from UM20147 (11.0BD), captured over USB 2026-09-30. +# A histogram: content[11] = 0x08. +RECORD_BD = bytes.fromhex( + "d200000000055d4a8100001e0907eab30d1b21000000084c6f636174696f6e00" + "0000000000000000000000000074657374320000000000000000000000000000" + "0000000000000000000000000000000000000000000000554d32303134370000" + "003fcb3bde00000000053f000f5472616e00003e698cdb000300005665727400" + "003fc76e65000300004c6f6e6700003e567c21000300004d69630000003956b9" + "7c00050000000000000000000000000000000000000000000000000000000000" + "0000000000000000000000000000000000000000000000000000000000" +) + + +def chain_entry(key: bytes, size: int) -> bytes: + """A 1E/1F response data section: 11-byte prefix then key + size.""" + return (bytes([0x08]) + bytes(7) + bytes([0xFE]) + bytes(2) + + key + size.to_bytes(4, "big")) + + +def client(responses): + t = ScriptedTransport(responses) + return MicromateClient(t, recv_timeout=0.5), t + + +# ── The chain walk ──────────────────────────────────────────────────────────── + +def test_chain_walk_arms_before_every_entry(): + """⚠ THOR sends 0x93 before EVERY 1E/1F, and this mirrors that.""" + responses = [ + frame(0xEA, SERIAL), # serial() + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 4076)), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes.fromhex("055d4a82"), 11032)), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), # sentinel + ] + mm, t = client(responses) + refs = mm.list_events(with_records=False) + + subs = [w[5] for w in t.written] + assert subs == [0x15, 0x93, 0x1E, 0x93, 0x1F, 0x93, 0x1F] + assert [r.key_hex for r in refs] == ["055d4a81", "055d4a82"] + assert [r.size for r in refs] == [4076, 11032] + + +def test_an_all_zero_key_ends_the_chain_and_is_not_an_error(): + mm, _ = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes(4), 0)), + ]) + assert mm.list_events(with_records=False) == [] + + +def test_refs_carry_the_serial_because_a_key_alone_is_ambiguous(): + """⚠ UM12947 and UM20147 BOTH have an event 055d4a81. + + A store keyed on the event key alone treats one unit's event as a duplicate + of the other's, and nothing raises. `uid` is the safe identifier. + """ + mm, _ = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 4076)), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), + ]) + (ref,) = mm.list_events(with_records=False) + assert ref.serial == "UM12947" + assert ref.uid == "UM12947:055d4a81" + + +def test_a_cursor_that_never_advances_raises_rather_than_hanging(): + from micromate import client as C + responses = [frame(0xEA, SERIAL)] + entry = chain_entry(bytes.fromhex("055d4a81"), 4076) + responses += [frame(0x6C, bytes(11)), frame(0xE1, entry)] # the 1E read + for _ in range(C._MAX_EVENTS + 2): # then 1F forever + responses += [frame(0x6C, bytes(11)), frame(0xE0, entry)] + mm, _ = client(responses) + with pytest.raises(ProtocolError, match="not advancing"): + mm.list_events(with_records=False) + + +def test_a_truncated_chain_entry_raises(): + mm, _ = client([ + frame(0xEA, SERIAL), frame(0x6C, bytes(11)), frame(0xE1, bytes(14)), + ]) + with pytest.raises(ProtocolError, match="need 8"): + mm.list_events(with_records=False) + + +def test_iter_events_does_not_read_ahead(): + """It must yield at the cursor position, one arm/advance per event.""" + mm, t = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 4076)), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), + ]) + it = mm.iter_events(with_records=False) + first = next(it) + assert [w[5] for w in t.written] == [0x15, 0x93, 0x1E], "no read-ahead" + assert first.key_hex == "055d4a81" + assert list(it) == [] + + +# ── The 0x0C record ─────────────────────────────────────────────────────────── + +def test_record_decode_on_real_bd_bytes(): + mm, _ = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 4796)), + frame(0xF3, RECORD_BD), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), + ]) + (ref,) = mm.list_events() + + assert ref.record_type == "histogram" # content[11] = 0x08 + assert ref.is_histogram is True + assert ref.suffix == ".IDFH" + assert ref.timestamp == datetime.datetime(2026, 9, 30, 13, 27, 33) + assert ref.sensor_location == "Location" + assert ref.setup == "test2" + assert ref.serial == "UM20147", "the record's own serial wins over the cache" + assert ref.peak_vector_sum_ips == pytest.approx(1.587765, abs=1e-5) + assert ref.peaks_ips["Tran"] == pytest.approx(0.228076, abs=1e-5) + assert ref.peaks_ips["Vert"] == pytest.approx(1.558056, abs=1e-5) + assert ref.peaks_ips["Long"] == pytest.approx(0.209458, abs=1e-5) + + +def test_the_pvs_is_not_reconstructible_from_the_reported_peaks(): + """⚠ Why the PVS field matters: it cannot be recomputed from the peaks. + + It is the PER-SAMPLE peak vector sum. `sqrt(Σpeak²)` is only an upper + bound, because the channel maxima do not occur at the same instant, and + `max(T,V,L)` is a lower bound. On THIS event the two happen to be within + 0.05%, which is exactly the coincidence that made the field look like + `sqrt(Σpeak²)` on first inspection. Six other events separated them. + """ + import math + mm, _ = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 4796)), + frame(0xF3, RECORD_BD), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), + ]) + (ref,) = mm.list_events() + p = ref.peaks_ips + upper = math.sqrt(p["Tran"] ** 2 + p["Vert"] ** 2 + p["Long"] ** 2) + lower = max(p["Tran"], p["Vert"], p["Long"]) + assert lower < ref.peak_vector_sum_ips < upper + + +def test_filename_matches_thors_convention(): + ref = MicromateEventRef(index=0, key=bytes.fromhex("055d4a82"), size=11032, + serial="UM12947", record_type="waveform", + timestamp=datetime.datetime(2026, 9, 23, 16, 33, 19)) + assert ref.filename == "UM12947_20260923163319.IDFW" + ref.record_type = "histogram" + assert ref.filename == "UM12947_20260923163319.IDFH" + + +def test_filename_is_none_without_a_type_rather_than_guessing(): + """⚠ Guessing would file a histogram as a waveform, and read_idf_file() + dispatches on exactly that suffix.""" + ref = MicromateEventRef(index=0, key=bytes(4), size=1, serial="UM12947", + timestamp=datetime.datetime(2026, 1, 1)) + assert ref.record_type is None + assert ref.suffix is None + assert ref.filename is None + + +def test_an_unknown_record_type_warns_and_leaves_it_none(caplog): + bad = bytearray(RECORD_BD) + bad[CONTENT + 11] = 0x99 + mm, _ = client([ + frame(0xEA, SERIAL), + frame(0x6C, bytes(11)), frame(0xE1, chain_entry(bytes.fromhex("055d4a81"), 10)), + frame(0xF3, bytes(bad)), + frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0)), + ]) + with caplog.at_level("WARNING"): + (ref,) = mm.list_events() + assert ref.record_type is None + assert "unknown record type 0x99" in caplog.text + + +def test_get_event_refuses_without_a_record_type(): + mm, _ = client([]) + ref = MicromateEventRef(index=0, key=bytes.fromhex("055d4a81"), size=4076) + with pytest.raises(ValueError, match="record_type is unknown"): + mm.get_event(ref) + + +# ── Against the real captured session ───────────────────────────────────────── + +_CAPTURES = ( + Path(__file__).resolve().parents[1] + / "bridges" / "captures" / "9-24-26 - micromate2" +) +_DOWNLOAD = "raw_bw_20260925_011403_Download_events_then_delete_1_event.bin" + + +def _destuffed(blob: bytes, is_req: bool): + i, n = 0, len(blob) + while i < n: + if is_req: + if not (blob[i] == ACK and i + 1 < n and blob[i + 1] == STX): + i += 1 + continue + j = i + 2 + else: + if blob[i] != STX: + i += 1 + continue + j = i + 1 + out = bytearray() + while j < n: + if blob[j] == DLE and j + 1 < n: + out.append(blob[j + 1]) + j += 2 + continue + if blob[j] == ETX: + break + out.append(blob[j]) + j += 1 + if len(out) >= 6: + yield blob[i:j + 1], bytes(out[:-1]) + i = j + 1 + + +@pytest.mark.skipif( + not (_CAPTURES / _DOWNLOAD).is_file(), + reason="capture is gitignored; present only on a dev box", +) +def test_replaying_thors_session_reproduces_every_byte_and_decodes_every_event(): + """The whole point of step 4. + + Feed THOR's own responses to `iter_events()` + `get_event()`, and assert: + * every request byte we emit matches THOR's, in order + * all six events decode + * each waveform's decoded PVS matches the device's stored float + """ + reqs = list(_destuffed((_CAPTURES / _DOWNLOAD).read_bytes(), True)) + rsps = list(_destuffed( + (_CAPTURES / _DOWNLOAD.replace("raw_bw", "raw_s3")).read_bytes(), False)) + + # THOR's session opens with commands our client does not send (POLL, 0x1C, + # 0x06 …) and ends with a delete. Take the contiguous run from the first + # 0x93 to the last 0x5A -- that is the event walk, and it is what we mirror. + subs = [p[2] for _, p in reqs] + lo = subs.index(0x93) + hi = len(subs) - 1 - subs[::-1].index(0x5A) + want_wire = [w for w, _ in reqs[lo:hi + 1]] + replay = [bytes([STX]) + stuff(p + bytes([checksum(p)])) + bytes([ETX]) + for _, p in rsps[lo:hi + 1]] + + # ⚠ THOR never reads the chain sentinel in this capture -- it downloaded + # exactly six events and stopped, so it knew the count in advance. It read + # `SUB 0x06` (storage range) at frame 7, BEFORE the walk, and that response + # begins `00 00 00 06` -- the event count. Our generator walks until the + # sentinel instead, so append one and compare only the overlapping frames. + replay += [frame(0x6C, bytes(11)), frame(0xE0, chain_entry(bytes(4), 0))] + + # serial() would emit a 0x15 that is not in this slice, so prime the cache. + t = ScriptedTransport(replay) + mm = MicromateClient(t, recv_timeout=1.0) + mm._serial = "UM12947" + + decoded, checked = 0, 0 + for ref in mm.iter_events(): + # Pin THOR's chunk size: this test asserts byte-identity with + # THOR's frames, and our default is 16x larger. + result = mm.get_event(ref, chunk_size=1024) + decoded += 1 + assert ref.serial == "UM12947" + assert ref.filename and ref.filename.endswith(ref.suffix) + if not ref.is_histogram: + err = mm.decode_error(ref, result) + assert err is not None + assert abs(err) < 1e-4, f"{ref.uid}: PVS off by {100 * err:+.4f}%" + checked += 1 + + assert decoded == 6, "four waveforms and two histograms" + assert checked == 4 + # Our trailing sentinel read is two frames THOR did not send; everything + # up to it must match byte for byte, in order. + assert t.written[:len(want_wire)] == want_wire, ( + f"we emitted {len(t.written)} frames, THOR emitted {len(want_wire)}" + ) + assert len(t.written) == len(want_wire) + 2, "only the sentinel read is extra" + + +@pytest.mark.skipif( + not (_CAPTURES / _DOWNLOAD).is_file(), + reason="capture is gitignored; present only on a dev box", +) +def test_a_corrupted_stored_peak_is_caught_by_the_self_check(): + """Prove the verify path actually fires -- otherwise it is decoration.""" + reqs = list(_destuffed((_CAPTURES / _DOWNLOAD).read_bytes(), True)) + rsps = list(_destuffed( + (_CAPTURES / _DOWNLOAD.replace("raw_bw", "raw_s3")).read_bytes(), False)) + subs = [p[2] for _, p in reqs] + lo = subs.index(0x93) + hi = len(subs) - 1 - subs[::-1].index(0x5A) + replay = [bytes([STX]) + stuff(p + bytes([checksum(p)])) + bytes([ETX]) + for _, p in rsps[lo:hi + 1]] + + t = ScriptedTransport(replay) + mm = MicromateClient(t, recv_timeout=1.0) + mm._serial = "UM12947" + + for ref in mm.iter_events(): + if ref.is_histogram: + mm.download_event(ref, chunk_size=1024) # keep the replay in step + continue + ref.peak_vector_sum_ips *= 1.5 # as a bad decode would look + with pytest.raises(DecodeMismatch, match="decode is suspect"): + mm.get_event(ref, chunk_size=1024) + return + pytest.fail("no waveform event found in the capture") diff --git a/tests/test_micromate_protocol.py b/tests/test_micromate_protocol.py index ccb0214..0e652e7 100644 --- a/tests/test_micromate_protocol.py +++ b/tests/test_micromate_protocol.py @@ -174,12 +174,13 @@ def test_download_reproduces_thors_chunk_sequence_byte_for_byte(): payload = payload[:SIZE_4A81] responses = [] for i in range(4): - want = min(P.CHUNK_SIZE, SIZE_4A81 - i * P.CHUNK_SIZE) - body = payload[i * P.CHUNK_SIZE: i * P.CHUNK_SIZE + want] + want = min(P.THOR_CHUNK_SIZE, SIZE_4A81 - i * P.THOR_CHUNK_SIZE) + body = payload[i * P.THOR_CHUNK_SIZE: i * P.THOR_CHUNK_SIZE + want] responses.append(frame(0xA5, bytes(11) + body, page=want // 256)) p, t = proto(responses) - got = p.read_event_file(bytes.fromhex("055d4a81"), SIZE_4A81) + got = p.read_event_file(bytes.fromhex("055d4a81"), SIZE_4A81, + chunk_size=P.THOR_CHUNK_SIZE) assert t.written == REQ_CHUNKS_4A81 assert got == payload @@ -198,11 +199,12 @@ def test_chunk_count_and_final_offset(size, n_chunks, last_offset): """The first six rows are the six bench events, with THOR's real offsets.""" responses = [] for i in range(n_chunks): - want = min(P.CHUNK_SIZE, size - i * P.CHUNK_SIZE) + want = min(P.THOR_CHUNK_SIZE, size - i * P.THOR_CHUNK_SIZE) responses.append(frame(0xA5, bytes(11) + bytes(want))) p, t = proto(responses) - p.read_event_file(bytes.fromhex("055d4a81"), size) + p.read_event_file(bytes.fromhex("055d4a81"), size, + chunk_size=P.THOR_CHUNK_SIZE) assert len(t.written) == n_chunks # offset is payload[4:5] of the request; recover it from the built frame @@ -214,17 +216,29 @@ def test_chunk_count_and_final_offset(size, n_chunks, last_offset): ) -def test_a_short_chunk_raises_rather_than_truncating(): - """A silently short event is the failure mode this codebase keeps hitting.""" - p, _ = proto([frame(0xA5, bytes(11) + bytes(900))]) # asked for 1024 - with pytest.raises(ShortRead, match="asked for 1024 B, got 900"): - p.read_event_file(bytes.fromhex("055d4a81"), 1024) +def test_a_short_chunk_is_absorbed_and_the_remainder_refetched(): + """⚠ Changed 2026-10-02: a short chunk is no longer an error. + + It used to raise, on the principle that a silently short event is the failure + mode this codebase keeps hitting. But the device *legitimately* returns + short — it clamps an over-large request to 16,384 B without saying so — and + the real protection is tracking the offset by bytes received, which makes a + short response cost one extra request instead of corrupting the file. The + total length is still checked, so a genuinely truncated event still raises. + """ + p, t = proto([frame(0xA5, bytes(11) + b"\xaa" * 900), + frame(0xA5, bytes(11) + b"\xbb" * 124)]) + got = p.read_event_file(bytes.fromhex("055d4a81"), 1024, + chunk_size=P.THOR_CHUNK_SIZE) + assert got == b"\xaa" * 900 + b"\xbb" * 124 + assert len(t.written) == 2, "the 124 B remainder was refetched" def test_a_chunk_too_short_to_hold_its_header_raises(): p, _ = proto([frame(0xA5, bytes(4))]) with pytest.raises(ShortRead, match="too short to hold a chunk header"): - p.read_event_file(bytes.fromhex("055d4a81"), 1024) + p.read_event_file(bytes.fromhex("055d4a81"), 1024, + chunk_size=P.THOR_CHUNK_SIZE) def test_download_rejects_a_nonsense_size(): @@ -391,7 +405,8 @@ def test_every_captured_download_frame_is_one_we_would_have_sent(): for e in events: p, t = proto([bytes([STX]) + stuff(r + bytes([checksum(r)])) + bytes([ETX]) for r in e["rsps"]]) - got = p.read_event_file(e["key"], e["size"]) + got = p.read_event_file(e["key"], e["size"], + chunk_size=P.THOR_CHUNK_SIZE) assert t.written == e["reqs"], ( f"event {e['key'].hex()}: our {len(t.written)} frames differ from " f"THOR's {len(e['reqs'])}" @@ -400,3 +415,124 @@ def test_every_captured_download_frame_is_one_we_would_have_sent(): total += len(e["reqs"]) assert total == 56, f"expected 56 download frames across the 6 events, saw {total}" + + +# ── Above 64 KB: the limit UM20147 already exceeds ──────────────────────────── + +def test_chunk_offsets_carry_past_64_kb(): + """⚠ The offset is a uint32 at params[0:4], not a uint16 at params[2:4]. + + Every offset THOR was observed to send fits in two bytes (largest 0x3400), + so params[0:2] was always zero and the field looks narrower than it is. + Reading it as a uint16 caps a download at 65,536 B — and UM20147 holds a + 72,560 B event, so the cap is not hypothetical. + + This is the Series III 64 KB page-boundary bug in a new guise; that one is + still open. An address field whose high bytes are zero in every capture is + not a narrow field, it is an untested one. + """ + key = bytes.fromhex("055d4a83") + # Below the old cap: byte-identical to the uint16 form, so nothing verified + # against THOR's frames changes. + for off in (1024, 13312, 65535): + assert chunk_params(key, off) == bytes(2) + off.to_bytes(2, "big") + bytes(6) + # Above it: the carry lands in params[1]. + assert chunk_params(key, 65536) == bytes.fromhex("00010000") + bytes(6) + assert chunk_params(key, 71680) == bytes.fromhex("00011800") + bytes(6) + + +def test_a_chunk_offset_with_0x10_in_it_is_escaped_on_the_wire(): + """At offset 1 MiB params[1] is 0x10, which must go out as `10 10`. + + Nothing below 64 KB can produce this, so it only became reachable with the + uint32 offset — and an unescaped 0x10 is the bug class that cost the + Series III `5A` walk a release. + """ + params = chunk_params(bytes(4), 1 << 20) + assert params[:4] == bytes.fromhex("00100000") + + from micromate.framing import build_request + frame = build_request(P.SUB_BULK_DOWNLOAD, 0x0400, params) + assert bytes.fromhex("1010") in frame + + +def test_a_72kb_event_downloads_in_71_chunks(): + """UM20147's event 055d4a83, the one that broke the uint16 assumption. + + ✅ Confirmed against the real unit 2026-10-01: 71 chunks, exact size, with + chunks 64-70 carrying 0x01 in params[1]. This test pins the arithmetic the + hardware agreed with. + """ + size = 72560 + n = 71 + responses = [] + for i in range(n): + want = min(P.THOR_CHUNK_SIZE, size - i * P.THOR_CHUNK_SIZE) + responses.append(frame(0xA5, bytes(11) + bytes(want))) + + p, t = proto(responses) + got = p.read_event_file(bytes.fromhex("055d4a83"), size, + chunk_size=P.THOR_CHUNK_SIZE) + + assert len(got) == size + assert len(t.written) == n + # Chunk 64 is the first past the old cap; its params must carry the 0x01. + wire = t.written[64] + assert bytes.fromhex("000100") in wire, "the carry into params[1] is on the wire" + + +# ── The 16 KB ceiling, and the silent clamp ─────────────────────────────────── + +def test_default_chunk_size_is_the_measured_ceiling_not_thors(): + assert P.CHUNK_SIZE == 16384 + assert P.THOR_CHUNK_SIZE == 1024 + + +def test_a_silently_clamped_response_is_absorbed_not_failed(): + """⚠ Ask for more than the device serves and it CLAMPS — silently. + + Measured on UM20147: a 32,768 B request returns exactly 16,384 B of correct + data, no error. A loop striding by a fixed chunk size would either fail on + that or skip the bytes it never collected. Driving by bytes received makes + it a non-event: one more request. + + Here a 20,000 B event is fetched with chunk_size=16384 against a device + pretending to clamp at 8192, so the walk must take 3 requests at + offsets 0 / 8192 / 16384. + """ + size, clamp = 20000, 8192 + payload = bytes(range(256)) * 100 + payload = payload[:size] + + served, responses = 0, [] + while served < size: + n = min(clamp, size - served) + responses.append(frame(0xA5, bytes(11) + payload[served:served + n])) + served += n + + p, t = proto(responses) + got = p.read_event_file(bytes.fromhex("055d4a83"), size, chunk_size=16384) + + assert got == payload + assert len(t.written) == 3, "two clamped requests plus the remainder" + + +def test_a_chunk_that_returns_nothing_raises_rather_than_spinning(): + """A byte-driven loop must not loop forever on zero progress.""" + p, _ = proto([frame(0xA5, bytes(11))] * 4) + with pytest.raises(ShortRead, match="got none"): + p.read_event_file(bytes.fromhex("055d4a83"), 5000) + + +def test_the_72kb_event_now_takes_5_requests_not_71(): + size = 72560 + served, responses = 0, [] + while served < size: + n = min(P.CHUNK_SIZE, size - served) + responses.append(frame(0xA5, bytes(11) + bytes(n))) + served += n + + p, t = proto(responses) + got = p.read_event_file(bytes.fromhex("055d4a83"), size) + assert len(got) == size + assert len(t.written) == 5, "14x fewer round trips than THOR's 71"