""" micromate/idf_file.py — Thor IDF binary codec. Decodes the Instantel Micromate Series IV ``.IDFW`` (waveform) and ``.IDFH`` (histogram) binary on-disk format. Sister module to ``minimateplus/event_file_io.py``. Status (2026-05-28): - **Genuine Series IV / Thor binaries** are all signed ``00 12 01 00 00 00 Instantel\\0`` (sig-A in earlier notes). Two Series III (Blastware) binaries appear in the example corpus (``BE9439_*``) — they share the ``.IDFW``/``.IDFH`` extension by filing convention but carry a BW STRT header (``10 00 01 80 00 00 Instantel STRT...``) and are NOT Thor data. The reader detects them by signature and raises NotImplementedError pointing callers at ``minimateplus.event_file_io.read_blastware_file()``. - **IDFW waveform body** reuses the BW segment-rotated block codec verbatim. Body always starts at file offset ``0x0f1f``. Samples decoded via ``minimateplus.waveform_codec.decode_waveform_v2`` with 87–99% byte-exact match against ``.IDFW.txt`` sidecar (quiet events). Loud events hit the BW codec's known walker-stops-early limit. Residual ~3% drift on per-sample deltas — likely a Thor-specific 12-bit delta refinement that BW's codec doesn't model. Geo LSB = 0.0003 in/s; mic factor ~2.14e-6 psi/count. - **IDFH histogram body**: 12-byte segment header ``[len_be 2B] 0a 00 00 00 [00 NN_counter] 05 3f`` introduces a segment of ``N`` 72-byte interval records (``N = (len - 10) // 72``). Each record holds 4 × 16-byte per-channel min/max/halfp + 8-byte tail. Geo peaks via ``max(|min|, |max|) / 32768 × 10`` in/s (matches sidecar within ~1.8%), freq via ``512 / halfp`` Hz. **All 859 Thor IDFH files in the corpus decode (181,071 intervals).** - Binary metadata directly extracted: serial, timestamp, sample_rate, record_time, calibration_date. Other fields fall back to the paired ``.IDFW.txt`` / ``.IDFH.txt`` sidecar (consumed by ``WaveformStore.save_imported_idf``). The full reverse-engineering writeup lives in ``docs/idf_protocol_reference.md``. """ from __future__ import annotations import datetime import struct from dataclasses import dataclass from pathlib import Path from typing import Optional, Union # Thor IDFW bodies use the series-3 record-chain decoder. # # This was previously pinned to the SUPERSEDED tag-dispatch walker # (`decode_waveform_legacy`) on the stated grounds that "Thor has no ASCII # ground truth in the corpus and its geo scaling is separately suspect". # Both premises were false: Thor writes a per-sample CSV export next to every # binary (see scratch/verify_thor_against_csv.py), and the scaling is now # resolved (see _GEO_LSB_IPS). Measured against that ground truth on # 2026-09-10, the record chain beats the legacy walker outright: # # channel truncation 55/153 files -> 3/153 # files exact 98/153 -> 150/153 # per-sample exact 99.781% -> 99.854% # # The legacy walker stops at the first unrecognised tag and returns whatever # channels it had, so its failure mode is silent short channels rather than an # error. Do not re-pin it. from minimateplus.waveform_codec import _MODES, decode_waveform_v2, is_record from .models import IdfEvent, IdfPeaks, IdfReport # Genuine Series IV / Thor IDF binary signature: 6 bytes, then ASCII "Instantel". _THOR_PREFIX = b"\x00\x12\x01\x00\x00\x00" # Stray Series III (Blastware) binaries that occasionally turn up in Thor # corpus directories renamed to the .IDFW/.IDFH convention. Their header # (`10 00 01 80 00 00 Instantel STRT ...`) is byte-for-byte a BW SUB 5A # STRT record, not a Thor binary. Detected so we can refuse-and-route # rather than mis-parse. _BW_STRAY_PREFIX = b"\x10\x00\x01\x80\x00\x00" _INSTANTEL_TAG = b"Instantel" # Most common body offset for sig-A IDFW files (~50% of prod events; # 151/154 in the original tests/fixtures/THORDATA_example corpus). The # body is the segment-rotated block stream consumed by decode_waveform_v2; # bytes [0:3] are the magic ``00 02 00`` preamble. Production events # routinely use other offsets — see :func:`_find_waveform_body_offset` # for the dynamic scan. This constant survives only as the priority hint. _BODY_START_SIG_A = 0x0F1F # Magic bytes that mark a candidate waveform-body preamble. _BODY_MAGIC = b"\x00\x02\x00" # Where to start looking for body candidates inside the file. Skip the # fixed-header region where the same magic legitimately appears inside # channel-test records and the compliance block (offsets 0x015d, 0x091c, # 0x0ae2, 0x0d30 in observed events). # Lowered from 0x0E00 to 0x0C00 (2026-09-10). Three-channel events -- mic # disabled -- have a shorter fixed header and put their record chain head at # 0x0dba, below the old floor. The head was therefore invisible to the scan, # which fell through to the *Vert* segment-0 record and decoded a body shifted # one position around the channel rotation. 46 of 139 files in the # 9-10-26-csv-req corpus were affected; all 46 became per-sample exact once # the head was reachable. The floor still skips the fixed-header region, # where `is_record()` can match channel-test records (0x015d, 0x091c, 0x0ae2). _BODY_SCAN_FLOOR = 0x0C00 # Cap on trial decodes per file. Chain-head detection normally yields one # or two candidates; the cap only bounds the worst case on a corrupt file. _MAX_BODY_CANDIDATES = 16 # Geophone count → in/s. # # The old value 0.0003 was read off the smallest non-zero sample in the # sidecar corpus, but that sample is Thor's *4-decimal display rounding* of # the true LSB, not the LSB itself. It read every series-4 geophone sample # 3.3% low. The quantisation ladder gives it away: counts 1..6 export as # 0.0003, 0.0006, 0.0009, 0.0012, 0.0016, 0.0019 — an LSB of exactly 0.0003 # would end 0.0015, 0.0018. # # The value below maximises exact 4-dp agreement over 1,046,016 paired # samples (454 channel-events, 2 units) at 99.854%, versus 50.7% for 0.0003. # It is a global constant, not a per-unit calibration: all 8 UM units in the # production store independently agree to within ±0.07% on their # device-reported PPV. 1/LSB = 3222.6 counts per in/s. # # The value is pinned, not guessed. Each exported sample constrains the LSB # to the window that rounds to the printed 4-dp figure; intersecting 991,415 # such constraints (clean channel-events only) gives # # LSB in [0.000310307933, 0.000310308057] width 1.2e-10 # # 0.000310308 sits at the centre of that window. Equivalent full scale is # 10.0 in/s / 0.000310308 = 32226.05 counts. # # Corroboration from the device: an IDFH interval that never recorded keeps # its min/max accumulator at its ±full-scale seed, and that seed is # (min=+32226, max=-32226) — the same magnitude, independently. Note the # tempting closed form 10.0/32226 is very slightly WRONG: it lands 4.5e-10 # above the feasible window and loses 78 boundary samples to the literal # value while never winning one. Series-3 uses 32000 counts for the same # 10.0 in/s, so the two generations do NOT share a scale. # # Ground truth + harness: scratch/verify_thor_against_csv.py _GEO_LSB_IPS = 0.000310308 # Microphone count → psi, derived from sidecar regression on 50 sample # pairs from UM11719_20231219162723.IDFW (mic-heavy event). _MIC_LSB_PSI = 2.14e-6 # IDFH histogram constants. # Bytes per interval record = 16 per channel + an 8-byte tail, so a # 4-channel unit uses 72 and a mic-disabled 3-channel unit uses 56. It is # NOT a constant: derive it per segment from the interval counter (see # decode_idfh_body). This value survives only as the 4-channel default. _IDFH_INTERVAL_SIZE = 72 # bytes per per-interval record (4 channels) _IDFH_CHANNEL_BLOCK = 16 # bytes per channel inside an interval record _IDFH_INTERVAL_TAIL = 8 # bytes after the per-channel blocks _IDFH_SEGMENT_HEADER = 10 # bytes: [len_be 2B][0a 00 00 00 4B][00 NN 2B][05 3f 2B] _IDFH_SEGMENT_TAIL = 2 # bytes after the interval data block, before next marker _IDFH_HALFP_FREQ_NUM = 512.0 # freq_hz = NUM / halfp; halfp ≤ 5 means ">100 Hz" sentinel _IDFH_CHANNELS = ("Tran", "Vert", "Long", "MicL") # ─── Binary metadata extraction ───────────────────────────────────────────── @dataclass class IdfBinaryMetadata: """Fields recoverable from the sig-A binary header (no .txt needed).""" serial: Optional[str] = None event_datetime: Optional[datetime.datetime] = None sample_rate: Optional[int] = None record_time_sec: Optional[float] = None calibration_date: Optional[datetime.date] = None def _read_ascii_z(buf: bytes, off: int, maxlen: int = 64) -> Optional[str]: if off >= len(buf): return None end = buf.find(b"\x00", off, off + maxlen) if end < 0: end = min(off + maxlen, len(buf)) s = buf[off:end].decode("ascii", errors="replace").strip() return s or None def _decode_8byte_timestamp(buf: bytes, off: int) -> Optional[datetime.datetime]: """Layout: ``[day][month][year_hi][year_lo][unknown][hour][min][sec]``.""" if off + 8 > len(buf): return None day, mon, yh, yl, _unk, hr, mn, sc = buf[off : off + 8] year = (yh << 8) | yl if not (2015 <= year <= 2050 and 1 <= mon <= 12 and 1 <= day <= 31 and 0 <= hr < 24 and 0 <= mn < 60 and 0 <= sc < 60): return None try: return datetime.datetime(year, mon, day, hr, mn, sc) except ValueError: return None def extract_binary_metadata(buf: bytes) -> IdfBinaryMetadata: """Pull serial/timestamp/sample_rate/record_time/calibration from the sig-A binary header. Field positions confirmed against UM11719_20231219162723.IDFW; stable across the 151-file sig-A corpus. """ md = IdfBinaryMetadata() # Serial: null-terminated ASCII at 0x14E. md.serial = _read_ascii_z(buf, 0x14E, maxlen=16) # Sample rate + record time live in a BW-compatible compliance block. # Locate the 6-byte anchor `be 80 00 00 00 00` and read offsets relative # to it: anchor-6 = sample_rate uint16 BE; anchor+6 = record_time float32 BE. anchor = buf.find(b"\xbe\x80\x00\x00\x00\x00", 0x800, 0xA00) if anchor > 0: sr_bytes = buf[anchor - 6 : anchor - 4] if len(sr_bytes) == 2: sr = int.from_bytes(sr_bytes, "big") if sr in (256, 512, 1024, 2048, 4096): md.sample_rate = sr rt_bytes = buf[anchor + 6 : anchor + 10] if len(rt_bytes) == 4: try: rt = struct.unpack(">f", rt_bytes)[0] if 0.1 <= rt <= 600.0: md.record_time_sec = float(rt) except struct.error: pass # Event timestamp: 8 bytes. Position differs between IDFW (0x97A) and # IDFH (0x9F8); scan a small range and accept the first valid decode. for off in (0x97A, 0x9F8): ts = _decode_8byte_timestamp(buf, off) if ts is not None: md.event_datetime = ts break # Calibration date: day, month, year_be at 0x194-0x197. if len(buf) > 0x197: day, mon = buf[0x194], buf[0x195] year = int.from_bytes(buf[0x196 : 0x198], "big") if 1 <= mon <= 12 and 1 <= day <= 31 and 2015 <= year <= 2050: try: md.calibration_date = datetime.date(year, mon, day) except ValueError: pass return md # ─── Sample decoder + unit conversion ─────────────────────────────────────── def _find_waveform_body_offset(buf: bytes) -> Optional[int]: """Pick the file offset of the waveform body by trial-decoding every ``00 02 00`` magic position past the fixed-header region. The body's location isn't fixed across all sig-A IDFW files — about half the production events use ``0x0f1f``, but the rest have offsets that shift based on header padding / channel-config layout. We auto-detect by: 1. Find every ``00 02 00`` occurrence past ``_BODY_SCAN_FLOOR``. 2. Try ``decode_waveform_v2()`` on each candidate. 3. Pick the offset whose decoded sample count is largest. Returns the offset, or ``None`` if no candidate yielded more than the trivial 2-sample preamble (= "no real body found"). Costs ~2-8 trial decodes per file; in practice the first candidate past 0x0e00 is usually the right one. """ if len(buf) < _BODY_SCAN_FLOOR + 8: return None # 1. Locate every plausible per-channel record header. A header carries # [len 2B][channel_id][00][00] at +2..+6, so anchor the search on the # three-byte `` 00 00`` signature and validate with is_record(). # Scanning candidate *preambles* instead is not viable: MODE_RAW16 is # ``00 00``, so every run of three zero bytes would look like a body # start and each would cost a full trial decode (~0.5 s/file measured). floor = max(0, _BODY_SCAN_FLOOR - 7) starts: list = [] for cid in (0x46, 0x47, 0x48, 0x49): sig = bytes((cid, 0x00, 0x00)) i = floor while True: j = buf.find(sig, i) if j < 0: break i = j + 1 q = j - 4 if q >= floor and is_record(buf, q): starts.append(q) if not starts: return None starts.sort() # 2. A body begins at the head of a record chain -- a record that no other # record's length field points at. The head's own payload is the # implicit segment-0 Tran record, and the body offset is head + 7 (past # [len 2B][cid][00][00][seg]) so that body[1:3] lands on the mode. ends = {q + 2 + int.from_bytes(buf[q + 2 : q + 4], "big") for q in starts} heads = [q for q in starts if q not in ends] or starts[:1] # 3. Trial-decode each head and keep the best. Prefer a candidate where # all four channels come out the same length: scoring on raw sample # count alone picks false positives sitting *inside* a record header, # which decode a plausible-looking but rotation-shifted body that # silently drops each channel's segment 0. best = None best_off = None for head in heads[:_MAX_BODY_CANDIDATES]: j = head + 7 if j + 3 > len(buf) or (buf[j + 1], buf[j + 2]) not in _MODES: continue try: decoded = decode_waveform_v2(buf[j:]) except Exception: continue if not decoded: continue lengths = [len(v) for v in decoded.values() if v] total = sum(len(v) for v in decoded.values()) # A "real" body has more than just the 2-sample preamble. if total <= 2: continue # >= 3 rather than == 4: a mic-disabled event has only the three geo # channels, and demanding four made `equal` permanently False for # them, leaving the pick to raw sample count alone. equal = len(lengths) >= 3 and len(set(lengths)) == 1 score = (equal, total) if best is None or score > best: best, best_off = score, j return best_off def _decode_waveform_samples(buf: bytes) -> Optional[dict]: """Decode samples from the sig-A waveform body. Returns the raw decoder counts dict — geo LSB = 0.0003 in/s, mic in its own count unit (see :func:`mic_count_to_psi`). Returns None if no usable body is found. Uses :func:`_find_waveform_body_offset` to locate the body — the file-offset varies across events (~50% sit at the canonical ``0x0f1f`` but the rest don't), so the previous hardcoded constant silently produced 2-sample preamble-only output for half the corpus. """ off = _find_waveform_body_offset(buf) if off is None: return None return decode_waveform_v2(buf[off:]) def geo_count_to_ips(count: int) -> float: """Convert a Thor geo decoder count to in/s. LSB = 0.0003 in/s.""" return count * _GEO_LSB_IPS def mic_count_to_psi(count: int) -> float: """Convert a Thor mic decoder count to psi. Scale derived from regression over 50 sample pairs in UM11719_20231219162723.IDFW; consistent to ~5%. Calibration constants from the channel block can refine this once decoded. """ return count * _MIC_LSB_PSI # ─── IDFH histogram decoder ───────────────────────────────────────────────── @dataclass class IdfhInterval: """One decoded histogram interval (typically one minute of monitoring).""" offset: int # file byte offset of the 72-byte record # Per-channel min/max ADC counts (int16 BE), half-period samples, peak count. # Peak = max(|min|, |max|). freq_hz = 512/halfp (None if halfp ≤ 5 → # ">100 Hz" sentinel; matches sidecar convention). tran_min: int tran_max: int tran_halfp: int vert_min: int vert_max: int vert_halfp: int long_min: int long_max: int long_halfp: int micl_min: int micl_max: int micl_halfp: int # 4 on a normal unit; 3 when the microphone is disabled, in which case the # micl_* fields are absent from the record and read as zero. n_channels: int = 4 def has_channel(self, channel: str) -> bool: return channel != "MicL" or self.n_channels >= 4 def peak_count(self, channel: str) -> int: mn = getattr(self, f"{channel.lower()}_min") mx = getattr(self, f"{channel.lower()}_max") return max(abs(mn), abs(mx)) def peak_ips(self, channel: str) -> float: """Convert peak count to in/s (geo channels only).""" # Same geo LSB as the waveform path — verified independently against # the IDFH exports: as peak magnitude rises (and 4-dp quantisation # noise falls) the implied LSB converges on 0.0003103, matching # _GEO_LSB_IPS. The old 10.0/32768 read histogram peaks 1.7% low. return self.peak_count(channel) * _GEO_LSB_IPS def freq_hz(self, channel: str) -> Optional[float]: halfp = getattr(self, f"{channel.lower()}_halfp") if halfp <= 5: return None return _IDFH_HALFP_FREQ_NUM / halfp def _is_unwritten_interval(interval: "IdfhInterval") -> bool: """True for an interval slot the device reserved but never wrote. Thor seeds each interval's per-channel accumulators at ``min = +full scale`` and ``max = -full scale`` and then narrows them as samples arrive. A slot that never recorded keeps that seed, so ``min > max`` — impossible for real data. Such a record decodes to a full-scale 10.0 in/s peak on every channel and, being a max-over-intervals, poisons the whole file's PPV. Rare but real: exactly 1 of 497,611 corpus intervals, and it inflated that file's Long PPV from 0.0081 to 10.0 in/s. The inversion is always all-or-nothing across channels (0 partial cases in the corpus), so requiring every channel to be inverted keeps this from ever firing on genuine data. """ pairs = [ (interval.tran_min, interval.tran_max), (interval.vert_min, interval.vert_max), (interval.long_min, interval.long_max), ] if interval.has_channel("MicL"): pairs.append((interval.micl_min, interval.micl_max)) return all(mn > mx for mn, mx in pairs) def _decode_idfh_interval(buf72: bytes, offset: int, n_channels: int = 4) -> IdfhInterval: """Decode one interval record into per-channel min/max/halfp. The record is ``n_channels`` × 16-byte blocks plus an 8-byte tail, so it is 72 bytes on a normal unit and 56 when the microphone is disabled. Missing channels read as zero. """ import struct fields = [] for i in range(4): if i >= n_channels: fields.extend([0, 0, 0]) continue block = buf72[i * 16 : (i + 1) * 16] mn = struct.unpack_from(">h", block, 0)[0] mx = struct.unpack_from(">h", block, 2)[0] # block[4:6] = int16 BE, role unknown (possibly time-of-peak) halfp = struct.unpack_from(">H", block, 6)[0] # block[10:12] and block[14:16] are uint16 BE with unknown semantics # (likely sum / count contributions for the PVS computation). fields.extend([mn, mx, halfp]) # Tail 8 bytes (buf72[64:72]) carry PVS-related data; not yet decoded. return IdfhInterval( offset=offset, tran_min=fields[0], tran_max=fields[1], tran_halfp=fields[2], vert_min=fields[3], vert_max=fields[4], vert_halfp=fields[5], long_min=fields[6], long_max=fields[7], long_halfp=fields[8], micl_min=fields[9], micl_max=fields[10], micl_halfp=fields[11], n_channels=n_channels, ) def decode_idfh_body(buf: bytes) -> list: """Walk an IDFH file and decode every interval record. The body has one or more segments; each segment header is 12 bytes: ``[length_be 2B][0a 00 00 00][counter_be 2B][05 3f]`` where ``length`` is bytes from the magic through the end of the interval block (= 10 + 72 × n_intervals). Segments are separated by a 2-byte tail + next-segment 2-byte prefix (the bytes before the next length field). ``counter`` is a **uint16 BE cumulative interval index** — the 0-based index of the LAST interval in this segment. Segments carry 10 intervals each, so it runs 9, 19, 29, ... across the file. ⚠ This validator used to require ``buf[j + 4] == 0x00``, i.e. that the counter's high byte was zero. That silently capped every histogram at **250 intervals**: the moment the cumulative counter passed 255 the high byte went non-zero and every later segment was rejected, so any monitoring run longer than ~4 hours lost its tail — frequently the part holding the event peak, which is why those files' PPV read low. 540 of 858 corpus files were affected. Do not reinstate that check. """ intervals: list = [] i = 0 prev_counter = -1 # so the first segment's n = counter + 1 while True: j = buf.find(b"\x0a\x00\x00\x00", i) if j < 0 or j < 2: break # Validate: [length_be][0a 00 00 00][counter_be][05 3f]. The counter # is deliberately NOT constrained — see the note above. if buf[j + 6 : j + 8] != b"\x05\x3f": i = j + 1 continue length = int.from_bytes(buf[j - 2 : j], "big") counter = int.from_bytes(buf[j + 4 : j + 6], "big") header_start = j - 2 if length < _IDFH_SEGMENT_HEADER or header_start + length > len(buf): # Truncated / bogus length — not a real segment header. i = j + 1 continue # The counter is the cumulative index of this segment's LAST interval, # so the interval count is its delta from the previous segment. That # gives the record stride, which is NOT fixed: 16 bytes per channel # plus an 8-byte tail, so 72 for a 4-channel unit and 56 for a # mic-disabled 3-channel one. Assuming 72 unconditionally made every # 3-channel histogram read 7 intervals per 10-interval segment, # walking off alignment into garbage that decoded as ~10 in/s peaks. n = counter - prev_counter if n <= 0: i = j + 1 continue stride = (length - _IDFH_SEGMENT_HEADER) // n n_channels, remainder = divmod(stride - _IDFH_INTERVAL_TAIL, _IDFH_CHANNEL_BLOCK) if remainder or not (1 <= n_channels <= 4): i = j + 1 continue interval_start = header_start + _IDFH_SEGMENT_HEADER for k in range(n): off = interval_start + k * stride if off + stride > len(buf): break chunk = buf[off : off + stride] interval = _decode_idfh_interval(chunk, off, n_channels) if _is_unwritten_interval(interval): # Reserved-but-never-recorded slot: the min/max accumulators # still hold their ±full-scale seed. Counting it would # fabricate a 10.0 in/s peak on every channel. continue intervals.append(interval) prev_counter = counter # Advance past this segment + the 2-byte tail. i = header_start + length + _IDFH_SEGMENT_TAIL return intervals # ─── Top-level reader ─────────────────────────────────────────────────────── @dataclass class IdfReadResult: """Return type for :func:`read_idf_file`. For waveforms (``.IDFW``), ``samples`` holds the per-channel sample arrays in Thor decoder counts. For histograms (``.IDFH``), ``samples`` is empty and ``intervals`` holds the per-interval record list (peaks, freqs). """ event: IdfEvent samples: dict # {"Tran": [...], ...} for IDFW; empty for IDFH binary_metadata: IdfBinaryMetadata signature: str # always "thor" for now (sig-A genuine Thor) intervals: Optional[list] = None # list[IdfhInterval] for IDFH; None for IDFW def read_idf_file( path: Union[str, Path], *, data: Optional[bytes] = None, ) -> IdfReadResult: """Parse a Thor ``.IDFW`` binary into an ``IdfEvent`` + decoded samples. Currently implements signature-A waveforms only. Signature-B (old-firmware) and ``.IDFH`` histograms raise NotImplementedError; use the paired ``.IDFW.txt`` / ``.IDFH.txt`` sidecar for those via ``parse_idf_report()``. Returns an :class:`IdfReadResult`. The caller converts int sample counts to physical units via :func:`geo_count_to_ips` / :func:`mic_count_to_psi`. ``path`` is used for filename in error messages and ``.IDFH`` vs ``.IDFW`` suffix detection. When ``data`` is supplied the disk read is skipped — useful for ingest paths that already have the bytes in memory and where the file may not exist on disk yet. """ p = Path(path) buf = data if data is not None else p.read_bytes() if len(buf) < 16 or buf[6:16] != _INSTANTEL_TAG + b"\x00": raise ValueError(f"{p.name}: not an IDF file (missing Instantel magic)") sig_prefix = buf[:6] if sig_prefix == _THOR_PREFIX: signature = "thor" elif sig_prefix == _BW_STRAY_PREFIX: raise NotImplementedError( f"{p.name}: file has a Series III (Blastware) STRT header in " "an IDF-named container — not a Thor binary. Route through " "minimateplus.event_file_io.read_blastware_file() instead " "(peaks decode; samples & full metadata don't, but it's not " "Thor data so the Thor codec doesn't apply)." ) else: raise ValueError(f"{p.name}: unknown IDF signature {sig_prefix.hex()}") is_histogram = p.suffix.upper() == ".IDFH" md = extract_binary_metadata(buf) if is_histogram: intervals = decode_idfh_body(buf) if not intervals: raise ValueError(f"{p.name}: IDFH body decoded no intervals") # Peaks: max across all intervals on each channel (per-channel max # of stored max-magnitudes; sidecar's PPV row carries the same). peak_tran = max((iv.peak_ips("Tran") for iv in intervals), default=0.0) peak_vert = max((iv.peak_ips("Vert") for iv in intervals), default=0.0) peak_long = max((iv.peak_ips("Long") for iv in intervals), default=0.0) # Mic peak in psi — Thor stores per-interval mic ADC counts in the # binary; convert the max count to psi via the per-count factor. # Skip on a mic-disabled (3-channel) unit: those records carry no mic # block at all, so peak_count("MicL") would report a synthetic zero. mic_peak_count = max( (iv.peak_count("MicL") for iv in intervals if iv.has_channel("MicL")), default=0, ) mic_peak_psi = mic_count_to_psi(mic_peak_count) if mic_peak_count else None rep = IdfReport( serial_number=md.serial, event_type="Full Histogram", event_datetime=md.event_datetime, filename=p.name, sample_rate=md.sample_rate, record_time_sec=md.record_time_sec, ) peaks = IdfPeaks( transverse_ips=peak_tran, vertical_ips=peak_vert, longitudinal_ips=peak_long, peak_vector_sum_ips=None, mic_pspl_dbl=None, # IDFH binary doesn't carry the dB(L) value mic_pspl_psi=mic_peak_psi, ) event = IdfEvent( serial=md.serial or "UNKNOWN", timestamp=md.event_datetime or datetime.datetime(1970, 1, 1), kind="Histogram", filename=p.name, sample_rate=md.sample_rate, record_time_sec=md.record_time_sec, peaks=peaks, report=rep, ) return IdfReadResult( event=event, samples={}, binary_metadata=md, signature=signature, intervals=intervals, ) # Waveform path. decoded = _decode_waveform_samples(buf) if decoded is None: raise ValueError(f"{p.name}: waveform body codec failed") rep = IdfReport( serial_number=md.serial, event_type="Full Waveform", event_datetime=md.event_datetime, filename=p.name, sample_rate=md.sample_rate, record_time_sec=md.record_time_sec, ) def _peak_ips(ch: str) -> float: arr = decoded.get(ch, []) return geo_count_to_ips(max((abs(v) for v in arr), default=0)) # Mic peak psi from binary: max absolute MicL ADC count × 2.14e-6 psi/count. mic_arr = decoded.get("MicL", []) mic_peak_count = max((abs(v) for v in mic_arr), default=0) mic_peak_psi = mic_count_to_psi(mic_peak_count) if mic_peak_count else None peaks = IdfPeaks( transverse_ips=_peak_ips("Tran"), vertical_ips=_peak_ips("Vert"), longitudinal_ips=_peak_ips("Long"), # PVS requires aligned per-sample √(T²+V²+L²); leave None — the # sidecar carries it and the bridge picks it up if present. peak_vector_sum_ips=None, mic_pspl_dbl=None, # binary IDFW doesn't carry the dB(L) value; # sidecar .txt fills it via IdfReport.from_dict mic_pspl_psi=mic_peak_psi, ) event = IdfEvent( serial=md.serial or "UNKNOWN", timestamp=md.event_datetime or datetime.datetime(1970, 1, 1), kind="Waveform", filename=p.name, sample_rate=md.sample_rate, record_time_sec=md.record_time_sec, peaks=peaks, report=rep, ) return IdfReadResult( event=event, samples=decoded, binary_metadata=md, signature=signature, )