v0.27.0 Decoder fixes, offset exploration and testing. #34
@@ -6,6 +6,17 @@ All notable changes to seismo-relay are documented here.
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
- **Histogram/waveform twin matching is now interval-based** (`find_twins`). A real
|
||||||
|
trigger is recorded twice — as a triggered waveform (stamped at the trigger instant)
|
||||||
|
and inside the scheduled histogram whose interval contains it (stamped at the 7am/7pm
|
||||||
|
interval start) — so the two twins can be **hours apart**. The old ±5-minute window
|
||||||
|
silently missed them, which broke review propagation (flagging one twin didn't flag its
|
||||||
|
twin). Twins are now matched by same serial + identical `peak_vector_sum` + opposite
|
||||||
|
record type + the waveform falling within the histogram's interval (bounded by the next
|
||||||
|
same-serial histogram). `window_seconds` is retained but ignored. Fixes terra-view #102
|
||||||
|
sub-task 2.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## v0.26.0 — 2026-08-27
|
## v0.26.0 — 2026-08-27
|
||||||
|
|||||||
+83
-32
@@ -603,56 +603,107 @@ class SeismoDb:
|
|||||||
).fetchall()
|
).fetchall()
|
||||||
return [dict(r) for r in rows]
|
return [dict(r) for r in rows]
|
||||||
|
|
||||||
def find_twins(self, event_id: str, *, window_seconds: int = 300) -> list[dict]:
|
def find_twins(self, event_id: str, *, window_seconds: int | None = None) -> list[dict]:
|
||||||
"""
|
"""
|
||||||
Find this event's histogram/waveform twin(s): rows sharing the same
|
Find this event's histogram/waveform twin(s): the SAME physical event
|
||||||
serial and an identical peak_vector_sum, whose timestamp falls
|
recorded both as a scheduled histogram and as a triggered waveform.
|
||||||
within ``window_seconds`` of this event's timestamp. Excludes the
|
|
||||||
event itself. Returns [] if the event or any required field
|
|
||||||
(serial / peak_vector_sum / timestamp) is missing.
|
|
||||||
|
|
||||||
Caveat: identical-PVS matching is a proxy for "same physical event
|
A real trigger is captured twice — once as a triggered waveform (stamped
|
||||||
recorded twice," not a guarantee. In the rare case where the device
|
at the trigger instant) and once inside the scheduled histogram whose
|
||||||
clamps/saturates PVS (clamped to sqrt(3) * geo_range), two distinct
|
interval contains it (stamped at the histogram's interval start, e.g. the
|
||||||
saturated events on the same serial within the window can share the
|
7am/7pm call-in). The two can be HOURS apart in time yet report the same
|
||||||
same clamped PVS value and be matched as twins even though they are
|
serial and identical peak_vector_sum. Twins are therefore matched by:
|
||||||
different events. This is harmless in practice — false_trigger/
|
|
||||||
reviewed_real are derived/index columns re-derivable from the
|
* same serial,
|
||||||
sidecar source of truth — but worth knowing if twin counts look
|
* identical peak_vector_sum,
|
||||||
surprising on a saturated/clamped run.
|
* OPPOSITE record type (one histogram, one waveform), and
|
||||||
|
* the waveform's timestamp falls within the histogram's interval —
|
||||||
|
from a histogram's timestamp up to the next histogram (same serial).
|
||||||
|
|
||||||
|
This replaces the old ±``window_seconds`` heuristic, which silently
|
||||||
|
missed twins more than a few minutes apart (a histogram's interval-start
|
||||||
|
stamp and the trigger instant routinely differ by hours). ``window_seconds``
|
||||||
|
is still accepted for backward compatibility but is ignored.
|
||||||
|
|
||||||
|
Returns [] if the event or a required field (serial / peak_vector_sum /
|
||||||
|
timestamp) is missing.
|
||||||
|
|
||||||
|
Caveat: identical-PVS matching remains a proxy for "same physical event"
|
||||||
|
— if the device clamps/saturates PVS (to sqrt(3) * geo_range), two
|
||||||
|
distinct saturated events could share a PVS. The added opposite-type and
|
||||||
|
interval constraints make a false pairing far less likely than the old
|
||||||
|
time-window match, and false_trigger/reviewed_real stay re-derivable from
|
||||||
|
the sidecar source of truth.
|
||||||
"""
|
"""
|
||||||
|
def _parse(ts):
|
||||||
|
if not ts:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return datetime.datetime.fromisoformat(str(ts).replace(" ", "T"))
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _is_hist(rt):
|
||||||
|
return str(rt or "").lower().startswith("hist")
|
||||||
|
|
||||||
row = self.get_event(event_id)
|
row = self.get_event(event_id)
|
||||||
if not row:
|
if not row:
|
||||||
return []
|
return []
|
||||||
serial = row.get("serial"); pvs = row.get("peak_vector_sum"); ts = row.get("timestamp")
|
serial = row.get("serial"); pvs = row.get("peak_vector_sum")
|
||||||
if serial is None or pvs is None or not ts:
|
t_target = _parse(row.get("timestamp"))
|
||||||
|
if serial is None or pvs is None or t_target is None:
|
||||||
return []
|
return []
|
||||||
try:
|
target_hist = _is_hist(row.get("record_type"))
|
||||||
t = datetime.datetime.fromisoformat(ts.replace(" ", "T"))
|
|
||||||
except ValueError:
|
|
||||||
return []
|
|
||||||
lo = (t - datetime.timedelta(seconds=window_seconds)).isoformat()
|
|
||||||
hi = (t + datetime.timedelta(seconds=window_seconds)).isoformat()
|
|
||||||
with self._connect() as conn:
|
|
||||||
rows = conn.execute(
|
|
||||||
"SELECT * FROM events WHERE serial=? AND id!=? AND peak_vector_sum=? "
|
|
||||||
"AND timestamp BETWEEN ? AND ?",
|
|
||||||
(serial, event_id, pvs, lo, hi),
|
|
||||||
).fetchall()
|
|
||||||
return [dict(r) for r in rows]
|
|
||||||
|
|
||||||
def propagate_review_to_twins(self, event_id: str, *, window_seconds: int = 300) -> list[str]:
|
with self._connect() as conn:
|
||||||
|
cand_rows = [dict(r) for r in conn.execute(
|
||||||
|
"SELECT * FROM events WHERE serial=? AND id!=? AND peak_vector_sum=?",
|
||||||
|
(serial, event_id, pvs)).fetchall()]
|
||||||
|
hist_ts = [r["timestamp"] for r in conn.execute(
|
||||||
|
"SELECT timestamp FROM events WHERE serial=? AND lower(record_type) LIKE 'hist%'",
|
||||||
|
(serial,)).fetchall()]
|
||||||
|
|
||||||
|
# Histogram interval-start times for this serial, sorted, to bound intervals.
|
||||||
|
starts = sorted(x for x in (_parse(t) for t in hist_ts) if x is not None)
|
||||||
|
|
||||||
|
def _interval_end(h_start):
|
||||||
|
# The next histogram strictly after h_start bounds the interval; else open-ended.
|
||||||
|
for x in starts:
|
||||||
|
if x > h_start:
|
||||||
|
return x
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _covers(h_start, w_time):
|
||||||
|
end = _interval_end(h_start)
|
||||||
|
return h_start <= w_time and (end is None or w_time < end)
|
||||||
|
|
||||||
|
twins = []
|
||||||
|
for c in cand_rows:
|
||||||
|
if _is_hist(c.get("record_type")) == target_hist:
|
||||||
|
continue # twins are strictly cross-type (one histogram, one waveform)
|
||||||
|
c_time = _parse(c.get("timestamp"))
|
||||||
|
if c_time is None:
|
||||||
|
continue
|
||||||
|
h_start, w_time = (t_target, c_time) if target_hist else (c_time, t_target)
|
||||||
|
if _covers(h_start, w_time):
|
||||||
|
twins.append(c)
|
||||||
|
return twins
|
||||||
|
|
||||||
|
def propagate_review_to_twins(self, event_id: str, *, window_seconds: int | None = None) -> list[str]:
|
||||||
"""
|
"""
|
||||||
Copy this event's `false_trigger`/`reviewed_real` columns onto each
|
Copy this event's `false_trigger`/`reviewed_real` columns onto each
|
||||||
of its histogram/waveform twins (see `find_twins`), so flagging one
|
of its histogram/waveform twins (see `find_twins`), so flagging one
|
||||||
twin flags both. Returns the list of twin ids updated.
|
twin flags both. Returns the list of twin ids updated.
|
||||||
|
|
||||||
|
``window_seconds`` is accepted for backward compatibility but ignored;
|
||||||
|
twin matching is now interval-based (see `find_twins`).
|
||||||
"""
|
"""
|
||||||
row = self.get_event(event_id)
|
row = self.get_event(event_id)
|
||||||
if not row:
|
if not row:
|
||||||
return []
|
return []
|
||||||
ft = 1 if row.get("false_trigger") else 0
|
ft = 1 if row.get("false_trigger") else 0
|
||||||
real = 1 if row.get("reviewed_real") else 0
|
real = 1 if row.get("reviewed_real") else 0
|
||||||
twins = self.find_twins(event_id, window_seconds=window_seconds)
|
twins = self.find_twins(event_id)
|
||||||
moved = []
|
moved = []
|
||||||
with self._connect() as conn:
|
with self._connect() as conn:
|
||||||
for tw in twins:
|
for tw in twins:
|
||||||
|
|||||||
+57
-19
@@ -1,33 +1,71 @@
|
|||||||
import datetime
|
import sqlite3
|
||||||
from sfm.database import SeismoDb
|
from sfm.database import SeismoDb
|
||||||
from minimateplus.models import Event, Timestamp
|
from minimateplus.models import Event, Timestamp
|
||||||
|
|
||||||
|
|
||||||
def _ins(db, key, serial, pvs, ts):
|
def _ins(db, key, serial, pvs, ts, record_type="Waveform"):
|
||||||
ev = Event(index=0)
|
ev = Event(index=0)
|
||||||
ev._waveform_key = bytes.fromhex(key)
|
ev._waveform_key = bytes.fromhex(key)
|
||||||
ev.timestamp = ts
|
ev.timestamp = ts
|
||||||
# peak_vector_sum comes from peak_values; simplest: insert then UPDATE pvs directly
|
|
||||||
db.insert_events([ev], serial=serial)
|
db.insert_events([ev], serial=serial)
|
||||||
row = [r for r in db.query_events(serial=serial) if r["waveform_key"] == key][0]
|
row = [r for r in db.query_events(serial=serial) if r["waveform_key"] == key][0]
|
||||||
import sqlite3
|
|
||||||
with sqlite3.connect(db.db_path) as c:
|
with sqlite3.connect(db.db_path) as c:
|
||||||
c.execute("UPDATE events SET peak_vector_sum=? WHERE id=?", (pvs, row["id"]))
|
c.execute("UPDATE events SET peak_vector_sum=?, record_type=? WHERE id=?",
|
||||||
|
(pvs, record_type, row["id"]))
|
||||||
return row["id"]
|
return row["id"]
|
||||||
|
|
||||||
|
|
||||||
def test_find_twins_matches_same_serial_pvs_near_time(tmp_path):
|
def _ts(hour, minute, second=0, day=25):
|
||||||
|
return Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=day,
|
||||||
|
hour=hour, minute=minute, second=second)
|
||||||
|
|
||||||
|
|
||||||
|
def test_histogram_and_waveform_twin_across_hours(tmp_path):
|
||||||
|
# The real UM12947 case: histogram stamped at its 7pm interval start, the
|
||||||
|
# triggered waveform 75 min later — same serial + identical PVS. The old
|
||||||
|
# ±5-min window missed this; interval matching catches it, both directions.
|
||||||
db = SeismoDb(tmp_path / "s.db")
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
base = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=5)
|
hist_pm = _ins(db, "01110001", "BE1", 0.4763, _ts(19, 31, 17), "Histogram")
|
||||||
twin = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=45)
|
wave = _ins(db, "01110002", "BE1", 0.4763, _ts(20, 46, 44), "Waveform")
|
||||||
far = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=21, minute=0, second=0)
|
_ins(db, "01110003", "BE1", 0.0100, _ts(7, 0, 0, day=26), "Histogram") # bounds the interval
|
||||||
# d needs a timestamp distinct from `twin` (UNIQUE(serial, timestamp) would
|
assert {r["id"] for r in db.find_twins(hist_pm)} == {wave}
|
||||||
# otherwise collide with b and UPSERT onto its row instead of inserting a
|
assert {r["id"] for r in db.find_twins(wave)} == {hist_pm}
|
||||||
# new one) while staying near `base` in time.
|
|
||||||
near = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=44)
|
|
||||||
a = _ins(db, "01110001", "BE1", 0.4763, base)
|
def test_same_type_not_twinned(tmp_path):
|
||||||
b = _ins(db, "01110002", "BE1", 0.4763, twin) # twin: same serial+pvs, 40s apart
|
# Two waveforms, same serial + PVS, seconds apart → NOT twins (cross-type only).
|
||||||
c = _ins(db, "01110003", "BE1", 0.4763, far) # same pvs but >window away
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
d = _ins(db, "01110004", "BE1", 0.9999, near) # near time but different pvs
|
a = _ins(db, "01110001", "BE1", 0.4763, _ts(20, 19, 5), "Waveform")
|
||||||
ids = {r["id"] for r in db.find_twins(a, window_seconds=300)}
|
_ins(db, "01110002", "BE1", 0.4763, _ts(20, 19, 45), "Waveform")
|
||||||
assert ids == {b}
|
assert db.find_twins(a) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_waveform_matches_only_the_containing_interval(tmp_path):
|
||||||
|
# Two overnight intervals with the same PVS; a waveform in the SECOND interval
|
||||||
|
# must twin with that histogram, never the first — even though PVS matches both.
|
||||||
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
|
h1 = _ins(db, "01110001", "BE1", 0.4763, _ts(19, 0, 0, day=25), "Histogram")
|
||||||
|
h2 = _ins(db, "01110002", "BE1", 0.4763, _ts(7, 0, 0, day=26), "Histogram")
|
||||||
|
w = _ins(db, "01110003", "BE1", 0.4763, _ts(8, 0, 0, day=26), "Waveform")
|
||||||
|
assert {r["id"] for r in db.find_twins(w)} == {h2}
|
||||||
|
assert w not in {r["id"] for r in db.find_twins(h1)}
|
||||||
|
|
||||||
|
|
||||||
|
def test_different_pvs_not_twinned(tmp_path):
|
||||||
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
|
h = _ins(db, "01110001", "BE1", 0.4763, _ts(19, 0, 0), "Histogram")
|
||||||
|
_ins(db, "01110002", "BE1", 0.9999, _ts(20, 0, 0), "Waveform") # different PVS
|
||||||
|
assert db.find_twins(h) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_open_ended_latest_interval(tmp_path):
|
||||||
|
# A waveform after the latest histogram (nothing bounds the interval) still twins.
|
||||||
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
|
h = _ins(db, "01110001", "BE1", 0.4763, _ts(19, 0, 0), "Histogram")
|
||||||
|
w = _ins(db, "01110002", "BE1", 0.4763, _ts(23, 30, 0), "Waveform")
|
||||||
|
assert {r["id"] for r in db.find_twins(h)} == {w}
|
||||||
|
|
||||||
|
|
||||||
|
def test_missing_fields_returns_empty(tmp_path):
|
||||||
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
|
assert db.find_twins("nonexistent-id") == []
|
||||||
|
|||||||
@@ -3,31 +3,34 @@ from sfm.database import SeismoDb
|
|||||||
from minimateplus.models import Event, Timestamp
|
from minimateplus.models import Event, Timestamp
|
||||||
|
|
||||||
|
|
||||||
def _ins(db, key, serial, pvs, ts):
|
def _ins(db, key, serial, pvs, ts, record_type="Waveform"):
|
||||||
ev = Event(index=0)
|
ev = Event(index=0)
|
||||||
ev._waveform_key = bytes.fromhex(key)
|
ev._waveform_key = bytes.fromhex(key)
|
||||||
ev.timestamp = ts
|
ev.timestamp = ts
|
||||||
# peak_vector_sum comes from peak_values; simplest: insert then UPDATE pvs directly
|
|
||||||
db.insert_events([ev], serial=serial)
|
db.insert_events([ev], serial=serial)
|
||||||
row = [r for r in db.query_events(serial=serial) if r["waveform_key"] == key][0]
|
row = [r for r in db.query_events(serial=serial) if r["waveform_key"] == key][0]
|
||||||
with sqlite3.connect(db.db_path) as c:
|
with sqlite3.connect(db.db_path) as c:
|
||||||
c.execute("UPDATE events SET peak_vector_sum=? WHERE id=?", (pvs, row["id"]))
|
c.execute("UPDATE events SET peak_vector_sum=?, record_type=? WHERE id=?",
|
||||||
|
(pvs, record_type, row["id"]))
|
||||||
return row["id"]
|
return row["id"]
|
||||||
|
|
||||||
|
|
||||||
def test_propagate_copies_flags_to_twins(tmp_path):
|
def _ts(hour, minute, second=0, day=25):
|
||||||
|
return Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=day,
|
||||||
|
hour=hour, minute=minute, second=second)
|
||||||
|
|
||||||
|
|
||||||
|
def test_propagate_copies_flags_across_hours_apart_twins(tmp_path):
|
||||||
|
# Flagging the waveform FT propagates to its histogram twin 75 min earlier
|
||||||
|
# (the interval matcher pairs them; the old ±5-min window would have missed it).
|
||||||
db = SeismoDb(tmp_path / "s.db")
|
db = SeismoDb(tmp_path / "s.db")
|
||||||
base = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=5)
|
hist = _ins(db, "01110001", "BE1", 0.4763, _ts(19, 31, 17), "Histogram")
|
||||||
twin = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=45)
|
wave = _ins(db, "01110002", "BE1", 0.4763, _ts(20, 46, 44), "Waveform") # twin, 75 min later
|
||||||
other = Timestamp(raw=b"", flag=0x10, year=2026, unknown_byte=0, month=2, day=25, hour=20, minute=19, second=44)
|
other = _ins(db, "01110003", "BE1", 0.9999, _ts(20, 20, 0), "Waveform") # different pvs
|
||||||
|
|
||||||
primary_id = _ins(db, "01110001", "BE1", 0.4763, base)
|
db.update_event_review(wave, {"false_trigger": True})
|
||||||
twin_id = _ins(db, "01110002", "BE1", 0.4763, twin) # twin: same serial+pvs, 40s apart
|
moved = db.propagate_review_to_twins(wave)
|
||||||
non_twin_id = _ins(db, "01110003", "BE1", 0.9999, other) # near time but different pvs
|
|
||||||
|
|
||||||
db.update_event_review(primary_id, {"false_trigger": True})
|
assert hist in moved
|
||||||
moved = db.propagate_review_to_twins(primary_id)
|
assert db.get_event(hist)["false_trigger"] == 1
|
||||||
|
assert db.get_event(other)["false_trigger"] == 0
|
||||||
assert twin_id in moved
|
|
||||||
assert db.get_event(twin_id)["false_trigger"] == 1
|
|
||||||
assert db.get_event(non_twin_id)["false_trigger"] == 0
|
|
||||||
|
|||||||
Reference in New Issue
Block a user