diff --git a/README.md b/README.md index 48d60ca..1482369 100644 --- a/README.md +++ b/README.md @@ -1981,8 +1981,12 @@ block with one loud sensor in it yields exactly one sensor however many are out there. **Slicing the envelope into bits never measures anything against a clock**, -and assumes as little as it can about how a bit is drawn. Nor about how long -the silences in it are: the gaps inside one message range from 200 µs on the +and assumes as little as it can about how a bit is drawn — nor about which end +of a byte goes first, both orders being tried and the checksums saying which. +(There is a fingerprint for that one: reversing the bits of a byte does not +change how many are set, so parity survives it and a sum does not. A message +read from the wrong end shows every parity holding and every checksum failing.) +Nor about how long the silences in it are: the gaps inside one message range from 200 µs on the newer sensors to 4000 µs on the oldest, so what belongs to one transmission is settled from each burst's own widest gap rather than from a figure that would have to suit every model at once. Anything that comes out longer than the @@ -2133,7 +2137,19 @@ transmitting over each other. **Bursts with clean pulse lengths and `nothing framed`.** The radio is fine and the message is from a model this does not read, or reads differently. -That line of pulse lengths is exactly what is needed to add it. +Under it comes the closest thing to a message that was found and which of its +checks held: + +``` + 61 pulses pulses 216×36 403×21 612×4 gaps 209×21 395×35 601×4 + nothing framed 8 ways of reading it tried + closest: 7 bytes at bit 4: ✓sum ✗parity ✓type ✗plausible DA 2B C4 B0 89 BF C1 +``` + +Which check fails says what kind of fault it is. A sum that holds and a parity +that does not is a different thing from neither holding, and the bytes are +printed so the arithmetic can be done by hand. That line and the pulse lengths +above it are between them everything needed to add a format. **`framed, but this model needs the same message twice`.** It was read correctly and arrived once. The two older models are only believed on a second diff --git a/bandsaunter/__init__.py b/bandsaunter/__init__.py index d4cb5a2..dfb9e58 100755 --- a/bandsaunter/__init__.py +++ b/bandsaunter/__init__.py @@ -9,7 +9,7 @@ and transcribing speech. # 2026-08-21_02 is the second build made on the 21st. The revision is padded # to two digits so versions sort as text. VERSION_DATE = "2026-09-07" -VERSION_REVISION = 3 +VERSION_REVISION = 4 __version__ = f"{VERSION_DATE}_{VERSION_REVISION:02d}" diff --git a/bandsaunter/acurite.py b/bandsaunter/acurite.py index 7d462a5..1a82323 100644 --- a/bandsaunter/acurite.py +++ b/bandsaunter/acurite.py @@ -79,7 +79,8 @@ __all__ = ["Reading", "Measure", "format_measure", "bursts", "Burst", "bits_pwm", "bits_ppm", "slicings", "pulse_train", "modulate", "baseband", "WIND_POINTS", "compass", "parity8", "crc8", "VirtualSensor", "SimulatedSensors", "default_sensors", - "survey", "Survey", "timings"] + "survey", "Survey", "timings", "near_misses", "NearMiss", + "BIT_ORDERS"] # Where every one of these sensors transmits. Nominally 433.92 MHz; the @@ -147,6 +148,10 @@ WIND_RANGE = (0.0, 260.0) # km/h; the sensor tops out below thi # length; see :func:`decode_lightning` for why. LIGHTNING_MESSAGE = 0x2F +# Every message type any decoder here recognises, for saying whether a burst +# that framed correctly is one of them. +_KNOWN_TYPES = frozenset({0x04, 0x31, 0x38, LIGHTNING_MESSAGE}) + # Where in a burst a message is allowed to sit. A real one begins after the # sync and runs to the end, so a handful of bits of sync in front of it and # almost nothing behind. See :func:`candidates`, where these do the work. @@ -271,6 +276,7 @@ MODELS = ( MODEL_NAMES = {family: name for family, name, _bytes, _r in MODELS} MESSAGE_BYTES = {family: n for family, _name, n, _r in MODELS} +MESSAGE_FAMILY = {n: family for family, _name, n, _r in MODELS} REPEATS_NEEDED = {family: repeats for family, _name, _bytes, repeats in MODELS} @@ -502,8 +508,24 @@ _DECODERS = ((7, decode_tower), (8, decode_5n1), (9, decode_lightning), # From bits to readings # --------------------------------------------------------------------------- -def _bytes_at(bits: str, at: int, count: int) -> list[int]: - return [int(bits[at + i * 8:at + (i + 1) * 8], 2) for i in range(count)] +# Which end of a byte goes down the air first. Most significant bit first is +# what the descriptions of these formats assume, and it is what nearly +# everything does, but "nearly" is the word that has cost this section three +# rounds already -- so both are tried and the checksums say which. +# +# There is a signature worth knowing for this one. Reversing the bits of a +# byte does not change how many of them are set, so parity survives it and a +# sum does not: a message read from the wrong end shows its parity holding +# and its checksum failing, every time, on every copy. +BIT_ORDERS = (False, True) + +_REVERSED = [int(f"{byte:08b}"[::-1], 2) for byte in range(256)] + + +def _bytes_at(bits: str, at: int, count: int, + reflect: bool = False) -> list[int]: + out = [int(bits[at + i * 8:at + (i + 1) * 8], 2) for i in range(count)] + return [_REVERSED[byte] for byte in out] if reflect else out def candidates(bits: str) -> list[Reading]: @@ -539,23 +561,40 @@ def candidates(bits: str) -> list[Reading]: for at in range(0, len(bits) - need + 1): if at > LEAD_BITS or len(bits) - at - need > TAIL_BITS: continue - try: - data = _bytes_at(bits, at, count) - except ValueError: - break - reading = decoder(data) - if reading is None: + if not _fits(at, need, len(bits)): continue - if not reading.measures and (at > UNREAD_LEAD_BITS - or len(bits) - at - need - > UNREAD_TAIL_BITS): - continue - reading.bits = bits[at:at + need] - reading.offset = at - found.append((at, need, reading)) + for reflect in BIT_ORDERS: + try: + data = _bytes_at(bits, at, count, reflect) + except ValueError: + break + reading = decoder(data) + if reading is None: + continue + if not reading.measures and not _fits(at, need, len(bits), + unread=True): + continue + reading.bits = bits[at:at + need] + reading.offset = at + found.append((at, need, reading)) + break # one order or the other, not both return _one_per_window(found, len(bits)) +def _fits(at: int, need: int, length: int, unread: bool = False) -> bool: + """Whether a message of this length can sit at this offset in a burst. + + A message begins after the sync and runs to the end of the burst, so a + handful of bits in front of it and almost none behind. One that cannot + be read is held to the tighter figures: it has no type field that means + anything here and no range that can be checked, so where it sits is the + only evidence there is that it is a message at all. + """ + lead = UNREAD_LEAD_BITS if unread else LEAD_BITS + tail = UNREAD_TAIL_BITS if unread else TAIL_BITS + return at <= lead and length - at - need <= tail + + def _one_per_window(found: list, length: int) -> list[Reading]: """Of two messages sharing bits, keep one: they cannot both be real. @@ -1286,6 +1325,82 @@ def timings(values, tolerance: float = 0.25) -> list[tuple[float, int]]: return [(sum(g) / len(g), len(g)) for g in groups] +@dataclass(frozen=True) +class NearMiss: + """One way of framing a burst, and which of its checks held. + + "Nothing framed" is not an answer, it is the absence of one. A message + that satisfies its checksum and fails its parity is a different fault + from one that fails both, and both are different from a burst that never + lined up on a byte boundary at all -- and the three want three different + fixes. This is what turns the first into the second. + """ + + family: str = "" + length: int = 0 # in bytes + offset: int = 0 # where in the burst it starts, in bits + data: tuple = () + checks: tuple = () # (name, passed) in the order they are run + read: bool = False # whether the decoder accepted it outright + reflect: bool = False # whether the bytes were read the other way + + @property + def score(self) -> int: + return sum(1 for _name, passed in self.checks if passed) + + @property + def hex(self) -> str: + return " ".join(f"{byte:02X}" for byte in self.data) + + def describe(self) -> str: + marks = " ".join(("\u2713" if passed else "\u2717") + name + for name, passed in self.checks) + order = ", least significant bit first" if self.reflect else "" + return (f"{self.length} bytes at bit {self.offset}{order}: {marks} " + f"{self.hex}") + + +def _checked(count: int, at: int, data: list[int], + reflect: bool = False) -> NearMiss: + """Every check that framing would have to pass, run one at a time.""" + if count in (7, 8, 9): + checks = (("sum", (sum(data[:-1]) & 0xFF) == data[-1]), + ("parity", all(parity8(byte) == 1 for byte in data[2:-1])), + ("type", (data[2] & 0x3F) in _KNOWN_TYPES)) + elif count == 5: + checks = (("sum", (sum(data[:4]) & 0xFF) == data[4]),) + else: + checks = (("crc", crc8(data[:3]) == data[3]),) + family = MESSAGE_FAMILY[count] + decoder = dict(_DECODERS)[count] + return NearMiss(family=family, length=count, offset=at, data=tuple(data), + checks=checks + (("plausible", decoder(data) is not None),), + read=decoder(data) is not None, reflect=reflect) + + +def near_misses(bits: str, most: int = 2) -> list[NearMiss]: + """The framings that came closest, best first. + + Only the positions a real message could occupy, because a window halfway + through a burst failing its checksum says nothing anybody needs. + """ + out = [] + for count, _decoder in _DECODERS: + need = count * 8 + for at in range(0, len(bits) - need + 1): + if at > LEAD_BITS or len(bits) - at - need > TAIL_BITS: + continue + for reflect in BIT_ORDERS: + try: + out.append(_checked(count, at, + _bytes_at(bits, at, count, reflect), + reflect)) + except ValueError: + break + out.sort(key=lambda miss: (miss.score, miss.length), reverse=True) + return out[:most] + + @dataclass class Survey: """What one block of samples looked like at every stage. @@ -1303,6 +1418,7 @@ class Survey: rate: float = 0.0 # what the envelope was sliced at seen: list = field(default_factory=list) # (burst, slicings, framed) readings: list = field(default_factory=list) + misses: list = field(default_factory=list) # per burst, the closest tries @property def loudest(self) -> float: @@ -1322,7 +1438,7 @@ def survey(iq: np.ndarray, sample_rate: float, offset: float = 0.0, smooth = _smoothed(envelope, rate) peak = float(np.percentile(smooth, 99.99)) if smooth.size else 0.0 quiet = float(np.percentile(smooth, 20)) if smooth.size else 0.0 - seen = [] + seen, misses = [], [] for burst in bursts(envelope, rate): tries = slicings(burst) # Without the corroboration rule, so that a message which framed @@ -1331,9 +1447,15 @@ def survey(iq: np.ndarray, sample_rate: float, offset: float = 0.0, framed = _best([r for bits in tries for r in candidates(bits)], confirm=False) seen.append((burst, tries, framed)) + # What almost worked, for a burst that came to nothing. Judged + # across every reading of it, because which reading was the right one + # is the question and not the answer. + closest = sorted((m for bits in tries for m in near_misses(bits)), + key=lambda m: (m.score, m.length), reverse=True) + misses.append(closest[:2]) return Survey(quiet=quiet, gate=_noise_gate(smooth, quiet, peak) if smooth.size else 0.0, - peak=peak, rate=rate, seen=seen, + peak=peak, rate=rate, seen=seen, misses=misses, readings=readings_from(iq, sample_rate, offset, when)) diff --git a/bandsaunter/weather.py b/bandsaunter/weather.py index e0bed29..11b5815 100644 --- a/bandsaunter/weather.py +++ b/bandsaunter/weather.py @@ -713,16 +713,21 @@ def _survey_line(console, options: WeatherOptions, samples, at: float, f"({look.loudest:.1f}x the noise) " f"{len(look.seen)} burst{'' if len(look.seen) == 1 else 's'}[/grey62]", highlight=False) - for burst, tries, framed in look.seen: + for i, (burst, tries, framed) in enumerate(look.seen): marks = " ".join(f"{v:.0f}×{n}" for v, n in timings(burst.marks)) gaps = " ".join(f"{v:.0f}×{n}" for v, n in timings(burst.spaces)) console.print(f" [cyan]{burst.pulses:3d} pulses[/cyan] " f"pulses {marks} gaps {gaps}", highlight=False) if framed is None: console.print(f" [yellow]nothing framed[/yellow] " - f"[grey62]{len(tries)} ways of reading it tried; " - f"the pulse lengths above are what to look at" + f"[grey62]{len(tries)} ways of reading it tried" f"[/grey62]", highlight=False) + # Which check failed, rather than that one did. A message whose + # sum holds and whose parity does not is a different fault from + # one where neither holds, and they want different fixes. + for miss in (look.misses[i] if i < len(look.misses) else ()): + console.print(f" [grey62]closest:[/grey62] " + f"{miss.describe()}", highlight=False) elif not look.readings: console.print(f" [yellow]{framed.describe(options.imperial)}" f"[/yellow] [grey62]framed, but this model needs " diff --git a/packaging/bandsaunter.1 b/packaging/bandsaunter.1 index 2b68cc0..a361ff1 100644 --- a/packaging/bandsaunter.1 +++ b/packaging/bandsaunter.1 @@ -1,5 +1,5 @@ .\" Generated by packaging/make-man.py -- do not edit by hand. -.TH BANDSAUNTER 1 "2026-09-07" "bandsaunter 2026-09-07_03" "User Commands" +.TH BANDSAUNTER 1 "2026-09-07" "bandsaunter 2026-09-07_04" "User Commands" .SH NAME bandsaunter \- scan, record and identify radio signals with an RTL-SDR .SH SYNOPSIS @@ -2299,9 +2299,13 @@ assumes as little as it can about how a bit is drawn. Not which of the pulse and the gap carries the bit; not whether the gap is the complement of the pulse or a fixed spacer, since a 220 microsecond pulse against a 200 microsecond spacer is the longer of the two and reads as the wrong bit; and -not which of long and short means one. The same burst is read half a dozen -ways and the checksums say which reading it was, at most one of them being -able to satisfy one. A transmitter running ten per cent fast is therefore read +not which of long and short means one; and not which end of a byte goes down +the air first. The same burst is read half a dozen ways and the checksums say +which reading it was, at most one of them being able to satisfy one. There is +a fingerprint for the last of those: reversing the bits of a byte does not +change how many are set, so parity survives it and a sum does not, and a +message read from the wrong end shows every parity holding and every checksum +failing. A transmitter running ten per cent fast is therefore read correctly and never noticed, which matters: these are unlocked and drift with the temperature, and an outdoor sensor in January is not the one that was on the fence in July. @@ -2366,8 +2370,12 @@ or three lengths with nothing in between; a smear means noise is being sliced as signal, or two sensors are transmitting over each other. .TP .B "clean pulse lengths, nothing framed" -The radio is fine and the message is from a model this does not read. The line -of pulse lengths is what is needed to add it. +The radio is fine and the message is from a model this does not read. Under it +comes the closest thing to a message that was found, which of its checks held, +and the bytes themselves in hexadecimal. Which check fails says what kind of +fault it is: a sum that holds while a parity does not is a different thing +from neither holding. That line and the pulse lengths above it are between +them everything needed to add a format. .TP .B "framed, but needs the same message twice" It was read correctly and arrived once. The two older models are believed only diff --git a/packaging/make-man.py b/packaging/make-man.py index 69864ea..6be11c8 100755 --- a/packaging/make-man.py +++ b/packaging/make-man.py @@ -1411,9 +1411,13 @@ assumes as little as it can about how a bit is drawn. Not which of the pulse and the gap carries the bit; not whether the gap is the complement of the pulse or a fixed spacer, since a 220 microsecond pulse against a 200 microsecond spacer is the longer of the two and reads as the wrong bit; and -not which of long and short means one. The same burst is read half a dozen -ways and the checksums say which reading it was, at most one of them being -able to satisfy one. A transmitter running ten per cent fast is therefore read +not which of long and short means one; and not which end of a byte goes down +the air first. The same burst is read half a dozen ways and the checksums say +which reading it was, at most one of them being able to satisfy one. There is +a fingerprint for the last of those: reversing the bits of a byte does not +change how many are set, so parity survives it and a sum does not, and a +message read from the wrong end shows every parity holding and every checksum +failing. A transmitter running ten per cent fast is therefore read correctly and never noticed, which matters: these are unlocked and drift with the temperature, and an outdoor sensor in January is not the one that was on the fence in July. @@ -1478,8 +1482,12 @@ or three lengths with nothing in between; a smear means noise is being sliced as signal, or two sensors are transmitting over each other. .TP .B "clean pulse lengths, nothing framed" -The radio is fine and the message is from a model this does not read. The line -of pulse lengths is what is needed to add it. +The radio is fine and the message is from a model this does not read. Under it +comes the closest thing to a message that was found, which of its checks held, +and the bytes themselves in hexadecimal. Which check fails says what kind of +fault it is: a sum that holds while a parity does not is a different thing +from neither holding. That line and the pulse lengths above it are between +them everything needed to add a format. .TP .B "framed, but needs the same message twice" It was read correctly and arrived once. The two older models are believed only diff --git a/tests/test_acurite.py b/tests/test_acurite.py index f445a61..f7573d0 100644 --- a/tests/test_acurite.py +++ b/tests/test_acurite.py @@ -941,3 +941,83 @@ def test_the_gate_still_keeps_noise_out_when_there_is_no_signal(): peak = float(np.percentile(smooth, 99.99)) over = (smooth > a._noise_gate(smooth, quiet, peak)).mean() assert over < 0.01, f"{over:.1%} of pure noise cleared the gate" + + +# --------------------------------------------------------------------------- +# Which end of a byte goes first +# --------------------------------------------------------------------------- + +def reversed_bytes(bits): + """The same message with each byte sent the other way round.""" + return "".join(bits[i:i + 8][::-1] for i in range(0, len(bits), 8)) + + +@pytest.mark.parametrize("order", ["as written", "least significant first"]) +def test_a_message_is_read_from_whichever_end_of_a_byte_it_arrives(order): + frame = a.tower_frame(0x1A2B, 21.5, 48, "A") + sent = frame if order == "as written" else reversed_bytes(frame) + iq = keyed_at(sent, pwm_pairs(sent)) + got = a.readings_from(iq, 250_000.0) + assert [r.sensor for r in got] == ["1A2B"], f"{order}: heard {got}" + + +def test_reading_a_byte_backwards_keeps_its_parity_and_breaks_its_sum(): + """The signature that names this fault, which is why it is worth having. + + Reversing the bits of a byte does not change how many are set, so odd + parity survives it; a checksum does not. A message read from the wrong + end therefore shows every parity holding and the sum failing, on every + copy -- which is a fingerprint rather than a guess. + """ + frame = a.tower_frame(0x1A2B, 21.5, 48, "A") + backwards = reversed_bytes(frame) + plain = [int(backwards[i:i + 8], 2) for i in range(0, len(backwards), 8)] + assert all(a.parity8(byte) == 1 for byte in plain[2:-1]) + assert (sum(plain[:-1]) & 0xFF) != plain[-1] + + +def test_only_one_of_the_two_orders_is_ever_reported(): + """Otherwise one message would corroborate itself and the rule that the + thinly checked models depend on would protect nothing.""" + frame = a.frame_609(0x5C, 4.2, 80) + found = a.candidates("0" + frame) + assert len(found) == 1 + + +# --------------------------------------------------------------------------- +# Saying which check failed, not that one did +# --------------------------------------------------------------------------- + +def test_the_closest_framing_says_which_check_held_and_which_did_not(): + frame = a.tower_frame(0x1A2B, 21.5, 48, "A") + good = a.near_misses("1111" + frame)[0] + assert good.read and good.score == len(good.checks) + assert dict(good.checks)["sum"] and dict(good.checks)["parity"] + assert good.hex.startswith("DA 2B") + assert "7 bytes at bit 4" in good.describe() + + +def test_a_broken_checksum_is_reported_as_a_broken_checksum(): + frame = list(a.tower_frame(0x1A2B, 21.5, 48, "A")) + frame[-1] = "1" if frame[-1] == "0" else "0" # the checksum byte + miss = a.near_misses("1111" + "".join(frame))[0] + assert dict(miss.checks)["parity"] is True + assert dict(miss.checks)["sum"] is False + assert not miss.read + + +def test_a_burst_of_nonsense_reports_the_checks_all_failing(): + rng = np.random.default_rng(3) + bits = "".join(rng.integers(0, 2, 60).astype(str)) + misses = a.near_misses(bits) + assert misses and all(not m.read for m in misses) + + +def test_the_hex_of_what_was_read_is_reported_so_it_can_be_worked_out_by_hand(): + """Because the alternative is asking somebody to send a photograph of a + display, and because the bytes are what any question about a format is + actually about.""" + frame = a.tower_frame(0x0C41, 3.2, 91, "B") + miss = a.near_misses("1111" + frame)[0] + assert len(miss.hex.split()) == 7 + assert all(len(byte) == 2 for byte in miss.hex.split())