feat(mqtt): store F-070 crash detail and group crashes fleet-wide

Firmware F-070 adds a fuller `crash` object (backtrace, backtrace_corrupted,
elf_sha256) and a new `pre_crash` snapshot (uptime, heap stats, optional
abort_msg) to boot_report and to telemetry.get_boot_history entries.

- Migration c9d0e1f2a3b4: device_boot_events gains crash / pre_crash JSONB,
  stored verbatim so later additive fields need no migration. The legacy
  crash_* columns are still filled; old rows and old firmware are unchanged.
  JSON is bound as CAST(CAST(:x AS TEXT) AS JSONB): a bare JSONB cast makes
  asyncpg JSON-encode the already-encoded string a second time.
- boot_report ingestion passes both objects through.
- telemetry.get_boot_history replies (control/ack) are merged into boot
  history whoever sent the command: an entry matches an existing row on
  boot_count + reset_reason + time within 30 min (boot_count alone is not
  unique - the counter gets reset), and only fills crash/pre_crash the row
  lacks; unmatched entries are boots we never saw live and are inserted at
  the device's timestamp; entries without ts are skipped.
- GET /api/mqtt/crash-groups: fault boots grouped by abort_msg with hex
  addresses stripped, else task + exception cause, else reset reason. When
  abort_msg is present, pc/exc_cause describe abort() itself and are
  ignored for grouping.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-30 16:28:40 +03:00
co-authored by Claude Opus 5.5
parent 46c5c0a846
commit 963516ece0
7 changed files with 365 additions and 25 deletions
+90
View File
@@ -0,0 +1,90 @@
"""Fleet-wide crash grouping (firmware F-070).
Groups fault boots so recurring crash types are visible across devices:
- with an abort_msg: by the message with hex addresses stripped, since the
same assert/abort reports different addresses per build;
- otherwise: by faulting task + Xtensa exception cause;
- fault resets with no coredump and no abort_msg: by reset reason alone.
When abort_msg is present, pc/exc_cause/exc_vaddr describe the abort()
mechanism (always StoreProhibited at 0x0), so they must NOT drive the grouping.
"""
import re
# Xtensa EXCCAUSE values the ESP32 actually produces.
EXC_CAUSE_NAMES = {
0: "IllegalInstruction",
2: "InstructionFetchError",
3: "LoadStoreError",
6: "IntegerDivideByZero",
9: "LoadStoreAlignment",
28: "LoadProhibited",
29: "StoreProhibited",
}
_HEX = re.compile(r"0x[0-9a-fA-F]+")
_SPACES = re.compile(r"\s+")
def normalize_abort_msg(msg: str) -> str:
return _SPACES.sub(" ", _HEX.sub("0x…", msg)).strip()
def exc_cause_name(cause) -> str:
if cause is None:
return "Unknown exception"
return EXC_CAUSE_NAMES.get(cause, f"Exception cause {cause}")
def group_key(event: dict) -> tuple[str, str, str]:
"""(key, kind, title) for one boot event."""
crash = event.get("crash") or {}
pre = event.get("pre_crash") or {}
abort_msg = pre.get("abort_msg")
if abort_msg:
norm = normalize_abort_msg(abort_msg)
return f"abort:{norm}", "abort", norm
task = crash.get("task", event.get("crash_task"))
cause = crash.get("exc_cause", event.get("crash_exc_cause"))
if task is not None or cause is not None:
title = f"{exc_cause_name(cause)} in {task or 'unknown task'}"
return f"exc:{task}:{cause}", "exception", title
reason = event.get("reset_reason") or "UNKNOWN"
return f"reason:{reason}", "reason", f"{reason.replace('_', ' ')} (no coredump)"
def group_crashes(events: list[dict]) -> list[dict]:
"""events must be newest-first (as get_crash_events returns them)."""
groups: dict[str, dict] = {}
for ev in events:
key, kind, title = group_key(ev)
g = groups.get(key)
if g is None:
g = groups[key] = {
"key": key,
"kind": kind,
"title": title,
"count": 0,
"first_at": ev["occurred_at"],
"last_at": ev["occurred_at"],
"latest": ev,
"devices": {},
}
g["count"] += 1
g["first_at"] = ev["occurred_at"] # newest-first input, so the last seen is the oldest
d = g["devices"].setdefault(ev["device_serial"], {
"device_serial": ev["device_serial"], "count": 0, "last_at": ev["occurred_at"],
})
d["count"] += 1
result = []
for g in groups.values():
devices = sorted(g["devices"].values(), key=lambda d: (-d["count"], d["device_serial"]))
result.append({**g, "devices": devices, "device_count": len(devices)})
# Most frequent first; ties broken by most recent (two stable sorts).
result.sort(key=lambda g: g["last_at"], reverse=True)
result.sort(key=lambda g: g["count"], reverse=True)
return result