From 338ce4fea7945dfb3b51d4ff4836696a5ecc2c9b Mon Sep 17 00:00:00 2001 From: serversdown Date: Fri, 2 Oct 2026 15:09:38 -0400 Subject: [PATCH] docs(series4): CORRECTED -- cellular cost is NOT independent of payload I had it as "~0.65 s per round trip, independent of payload size", and based design advice on it ("minimise round trips, not bytes"). The first claim is wrong and the second is too strong. Every measurement behind it had a SMALL payload -- 59 B status, 266 B setup records, 1024 B download chunks. With the byte term small and similar across all of them, per-command cost looked constant. It was a narrow-range fit extrapolated past its evidence. Measured against a 14,176 B single response on UM12947 over an RX55: 1,024 B 0.67 s 8,192 B 3.59 s 2,048 B 1.22 s 14,176 B 6.25 s 4,096 B 2.13 s t ~ 0.21 s + bytes / 2,350 -- fits within +/-11% over a 14x size range The old 0.65 s figure was right FOR A 1 KB RESPONSE and is that model evaluated at 1 KB. Round trips still cost (0.21 s each; 24 of them for a setup walk is still 16 s), but bytes cost more than round trips on anything over ~500 B, and that reverses the advice: for STATUS work minimise commands, for DOWNLOADS the floor is throughput and batching does not beat it. So the 16 KB chunk size is a ~30% win, not 14x. UM20147's 72,560-byte event is ~46 s at THOR's 1024 B and ~32 s at 16,384 B, because ~31 s of it is bytes on the wire. Still worth keeping -- 30% faster, and 14x fewer requests is 14x fewer chances for a link to drop mid-download -- but the earlier "~3.2 s" projection was wrong and is withdrawn. Corrected in all four places it had propagated: the protocol reference, the CHUNK_SIZE comment, the probe's verdict, and mm_client_check's banner. The probe now also prints a net time per measurement, since its raw timings include the idle gap while the control reads to frame completion -- comparing them directly was misleading. Also confirmed in the same run: 16 KB-class responses survive a cellular PAD. 1,024 / 2,048 / 4,096 / 8,192 / 14,176 B all arrived in one frame, byte-identical, over an RX55. 14,176 B is UM12947's largest event so the ceiling itself was not reached, but a 14 KB response crossing the PAD intact is what needed proving. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Ru8Lg9HkkYvX9VWWo65SmL --- bridges/mm_client_check.py | 7 +++-- docs/micromate_protocol_reference.md | 41 +++++++++++++++++++++++++++- micromate/protocol.py | 11 ++++++-- scratch/mm_stream_probe.py | 20 +++++++++++--- 4 files changed, 69 insertions(+), 10 deletions(-) diff --git a/bridges/mm_client_check.py b/bridges/mm_client_check.py index 6cc71d2..1471672 100644 --- a/bridges/mm_client_check.py +++ b/bridges/mm_client_check.py @@ -419,8 +419,11 @@ def main() -> int: print(" RX55 (TCP) 36 reads list_setups 16.05 s download 1.6 KiB/s") print(" The modem needs FEWER reads, not more -- it buffers ~1 s and then") print(" forwards one large segment, where CDC-ACM delivers many small ones.") - print(" Cost is ~0.65 s PER ROUND TRIP regardless of payload size, so what") - print(" matters over cellular is the number of commands, not the bytes.\n") + print(" Cellular cost, measured: ~0.21 s per request + ~2,350 B/s.") + print(" So STATUS work is round-trip bound (minimise commands) but a") + print(" DOWNLOAD is throughput bound -- 16 KB chunks save ~30% on a large") + print(" event, not 14x. An earlier note claiming cost was independent of") + print(" payload was fitted only to sub-1 KB responses.\n") return 0 diff --git a/docs/micromate_protocol_reference.md b/docs/micromate_protocol_reference.md index 34ad37a..561daaa 100644 --- a/docs/micromate_protocol_reference.md +++ b/docs/micromate_protocol_reference.md @@ -732,6 +732,15 @@ THOR's chunk size. The offset field is a **uint16**, so the structural ceiling i **65,535 bytes per request** — which would make that same event **2 requests, ~1.3 seconds**. +#### ✅ Confirmed over a cellular modem too (2026-10-02) + +UM12947 on an RX55, every size checked byte-for-byte against a 1024 B control: +**1,024 / 2,048 / 4,096 / 8,192 / 14,176 all served in one frame, intact.** Its +largest event is 14,176 B, so the 16,384 ceiling itself was not reached — but a +14 KB single response crossing the PAD unscathed is the thing that needed +proving. The reader reads to frame completion rather than using idle-gap +detection, which is why ~10 TCP segments reassemble without special handling. + #### ✅ The ceiling is 16,384 bytes, and over it the device CLAMPS SILENTLY Measured on UM20147, each size checked byte-for-byte against a known-good @@ -877,7 +886,37 @@ across reads, and the modem splits **less** than USB does. The client reads to frame completion rather than using `read_until_idle`'s idle-gap detection, and handled both without a retry. -### 🔑 ~0.65 s per round trip over cellular, independent of payload +### 🔑 The cellular cost model — ~0.21 s per request + ~2,350 B/s + +> #### ⚠ CORRECTED 2026-10-02 — it is NOT independent of payload +> +> This section previously read **"~0.65 s per round trip, independent of payload +> size"**, and concluded *"minimise round trips, not bytes"*. The first claim is +> wrong and the second is too strong. +> +> Every measurement behind it had a **small** payload — 59 B status, 266 B setup +> records, 1024 B download chunks. With the byte term small and similar across +> all of them, per-command cost looked constant. It was a narrow-range fit +> extrapolated past its evidence. +> +> Measured against a 14,176 B single response on UM12947 over an RX55: +> +> | request | measured | model | +> |---|---|---| +> | 1,024 B | 0.67 s | 0.65 s | +> | 2,048 B | 1.22 s | 1.08 s | +> | 4,096 B | 2.13 s | 1.96 s | +> | 8,192 B | 3.59 s | 3.70 s | +> | 14,176 B | 6.25 s | 6.25 s | +> +> **`t ≈ 0.21 s + bytes / 2,350`** — fits within ±11% across a 14× size range. +> +> The old 0.65 s figure was *right for a 1024-byte response* and is simply that +> model evaluated at 1 KB. Round trips still cost real money (0.21 s each, and +> 24 of them for a setup walk is still 16 s), but **bytes cost more than round +> trips on anything over ~500 B**, and that reverses the design advice: for +> *status* work minimise commands; for *downloads* the floor is throughput and no +> amount of batching beats it. This is the number that matters for SFM's design. diff --git a/micromate/protocol.py b/micromate/protocol.py index 42ca96a..215b02b 100644 --- a/micromate/protocol.py +++ b/micromate/protocol.py @@ -113,9 +113,14 @@ EVENT_TOKEN = 0xFE # known-good download, and anything larger is **silently clamped to 16,384** — # correct bytes, short length, no error. # -# This matters because round trips dominate a cellular download at ~0.65 s each -# regardless of payload. A 72,560-byte event is 71 requests (~46 s) at THOR's -# size and **5 requests (~3.2 s)** at this one. +# ⚠ The win is real but MODEST, and smaller than a first reading suggests. +# Measured over an RX55, cost per request is ~0.21 s + bytes/2350 — so a download +# is throughput-bound, not round-trip bound. A 72,560-byte event costs ~46 s at +# THOR's 1024 B and ~32 s at 16,384 B: a **30% saving**, not 14x, because ~31 s of +# it is bytes on the wire and no amount of batching addresses that. +# +# Still worth having: 30% faster, and 14x fewer requests is 14x fewer chances for +# a link to drop mid-download. # # ⚠ Measured on ONE unit, over USB. The loop below is driven by bytes received # rather than chunk index, so a unit that clamps lower simply takes more diff --git a/scratch/mm_stream_probe.py b/scratch/mm_stream_probe.py index e5c1fc0..7efed83 100644 --- a/scratch/mm_stream_probe.py +++ b/scratch/mm_stream_probe.py @@ -173,8 +173,11 @@ def main() -> int: body = b"".join(f.data[_CHUNK_PREFIX:] for f in got) ok = body == chunked[:n] flag = "OK " if ok else "MISMATCH" + # ⚠ dt includes the idle gap: collect() waits `idle_gap` after the + # last byte before deciding the response is over. Subtract it to + # compare against the control, which reads to frame completion. print(f" {n:6} B {len(got)} frame(s) {len(body):6} B back " - f"{dt:5.2f} s {flag}" + f"{dt:5.2f} s ({max(dt - a.idle_gap, 0):4.2f} s net) {flag}" + ("" if ok or not body else f" (first diff at {next((i for i in range(min(len(body), n)) if body[i] != chunked[i]), None)})")) if ok and len(body) == n: @@ -189,9 +192,18 @@ def main() -> int: print(f" VERDICT: the device serves at least {best} B per request,") print(f" verified byte-identical. For this {ref.size} B event that") print(f" is {then} request(s) instead of {now}.") - if best > 1024: - print(f" Over cellular at ~0.65 s per round trip: " - f"~{now * 0.65:.0f} s -> ~{then * 0.65:.1f} s.") + if best > THOR_CHUNK_SIZE: + # ⚠ cost is ~0.21 s per request PLUS ~2,350 B/s over a modem -- + # a download is throughput-bound, so do not promise a saving + # proportional to the drop in request count. + fixed, rate = 0.212, 2348.0 + t_now = now * fixed + ref.size / rate + t_then = then * fixed + ref.size / rate + print(f" Modelled over cellular (0.21 s/request + " + f"{rate:.0f} B/s): {t_now:.1f} s -> {t_then:.1f} s, " + f"a {100 * (1 - t_then / t_now):.0f}% saving.") + print(" Throughput-bound, not round-trip bound — most of that is") + print(" bytes on the wire and batching does not touch it.") if best >= ref.size: print(" The WHOLE EVENT fits in one request.")