From 18e57c6e5facf0fee81c5d6d16451182df10a545 Mon Sep 17 00:00:00 2001 From: bonamin Date: Wed, 30 Sep 2026 15:55:26 +0300 Subject: [PATCH] fix(mqtt): dedupe boot_report against the latest row, not any boot_count 98dd16b skipped a boot_report if ANY earlier row had the same boot_count. That is wrong: the firmware's lifetime boot counter gets reset (reflash / telemetry reset), and the data shows counts 1-4 recurring in July and again in September. With that rule a real later boot reusing a number would be dropped forever. A retained redelivery is always a copy of the device's most recent boot, so compare only against the latest row (boot_count + reset_reason). A genuine new boot always differs from it - the counter moves forward or was reset. Note: the one-off cleanup run on 2026-09-30 used the same wrong (serial, boot_count) key and deleted some genuine boot rows along with the redelivery duplicates; see the session notes / heartbeat-based reboot list. Co-Authored-By: Claude Opus 5.5 --- backend/database/pg_mqtt.py | 28 ++++++++++++++++++---------- 1 file changed, 18 insertions(+), 10 deletions(-) diff --git a/backend/database/pg_mqtt.py b/backend/database/pg_mqtt.py index 0a9a51e..d2caffc 100644 --- a/backend/database/pg_mqtt.py +++ b/backend/database/pg_mqtt.py @@ -445,22 +445,30 @@ async def insert_boot_event(device_serial: str, boot_count: int | None, crash_pc: int | None = None, crash_exc_cause: int | None = None, crash_exc_vaddr: int | None = None) -> int | None: - """Insert a boot event, unless one with the same boot_count already exists - for this device (returns None then). boot_report is published retained on - system/info, so the broker redelivers the last one every time the backend - (re)subscribes — without this check each backend restart logged the - device's last boot (crash detail included) again as a brand-new event.""" + """Insert a boot event, unless it is a repeat of the device's most recent + one (returns None then). boot_report is published retained on system/info, + so the broker redelivers the last one every time the backend (re)subscribes + — without this check each backend restart logged the device's last boot + (crash detail included) again as a brand-new event. + + Only the LATEST row is compared, never "any row with this boot_count": the + firmware's lifetime counter gets reset (reflash / telemetry reset), so the + same boot_count legitimately appears again for a later, different boot. A + genuine new boot always differs from the latest row — the counter either + moved forward or was reset to a lower number.""" async with AsyncSessionLocal() as session: if boot_count is not None: - dup = await session.execute( + latest = await session.execute( text(""" - SELECT 1 FROM device_boot_events - WHERE device_serial = :serial AND boot_count = :boot_count + SELECT boot_count, reset_reason FROM device_boot_events + WHERE device_serial = :serial + ORDER BY occurred_at DESC LIMIT 1 """), - {"serial": device_serial, "boot_count": boot_count}, + {"serial": device_serial}, ) - if dup.first() is not None: + row = latest.first() + if row is not None and row.boot_count == boot_count and row.reset_reason == reset_reason: return None result = await session.execute( text("""