feat(mqtt): store F-070 crash detail and group crashes fleet-wide
Firmware F-070 adds a fuller `crash` object (backtrace, backtrace_corrupted, elf_sha256) and a new `pre_crash` snapshot (uptime, heap stats, optional abort_msg) to boot_report and to telemetry.get_boot_history entries. - Migration c9d0e1f2a3b4: device_boot_events gains crash / pre_crash JSONB, stored verbatim so later additive fields need no migration. The legacy crash_* columns are still filled; old rows and old firmware are unchanged. JSON is bound as CAST(CAST(:x AS TEXT) AS JSONB): a bare JSONB cast makes asyncpg JSON-encode the already-encoded string a second time. - boot_report ingestion passes both objects through. - telemetry.get_boot_history replies (control/ack) are merged into boot history whoever sent the command: an entry matches an existing row on boot_count + reset_reason + time within 30 min (boot_count alone is not unique - the counter gets reset), and only fills crash/pre_crash the row lacks; unmatched entries are boots we never saw live and are inserted at the device's timestamp; entries without ts are skipped. - GET /api/mqtt/crash-groups: fault boots grouped by abort_msg with hex addresses stripped, else task + exception cause, else reset reason. When abort_msg is present, pc/exc_cause describe abort() itself and are ignored for grouping. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,90 @@
|
||||
"""Fleet-wide crash grouping (firmware F-070).
|
||||
|
||||
Groups fault boots so recurring crash types are visible across devices:
|
||||
- with an abort_msg: by the message with hex addresses stripped, since the
|
||||
same assert/abort reports different addresses per build;
|
||||
- otherwise: by faulting task + Xtensa exception cause;
|
||||
- fault resets with no coredump and no abort_msg: by reset reason alone.
|
||||
|
||||
When abort_msg is present, pc/exc_cause/exc_vaddr describe the abort()
|
||||
mechanism (always StoreProhibited at 0x0), so they must NOT drive the grouping.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
# Xtensa EXCCAUSE values the ESP32 actually produces.
|
||||
EXC_CAUSE_NAMES = {
|
||||
0: "IllegalInstruction",
|
||||
2: "InstructionFetchError",
|
||||
3: "LoadStoreError",
|
||||
6: "IntegerDivideByZero",
|
||||
9: "LoadStoreAlignment",
|
||||
28: "LoadProhibited",
|
||||
29: "StoreProhibited",
|
||||
}
|
||||
|
||||
_HEX = re.compile(r"0x[0-9a-fA-F]+")
|
||||
_SPACES = re.compile(r"\s+")
|
||||
|
||||
|
||||
def normalize_abort_msg(msg: str) -> str:
|
||||
return _SPACES.sub(" ", _HEX.sub("0x…", msg)).strip()
|
||||
|
||||
|
||||
def exc_cause_name(cause) -> str:
|
||||
if cause is None:
|
||||
return "Unknown exception"
|
||||
return EXC_CAUSE_NAMES.get(cause, f"Exception cause {cause}")
|
||||
|
||||
|
||||
def group_key(event: dict) -> tuple[str, str, str]:
|
||||
"""(key, kind, title) for one boot event."""
|
||||
crash = event.get("crash") or {}
|
||||
pre = event.get("pre_crash") or {}
|
||||
abort_msg = pre.get("abort_msg")
|
||||
if abort_msg:
|
||||
norm = normalize_abort_msg(abort_msg)
|
||||
return f"abort:{norm}", "abort", norm
|
||||
|
||||
task = crash.get("task", event.get("crash_task"))
|
||||
cause = crash.get("exc_cause", event.get("crash_exc_cause"))
|
||||
if task is not None or cause is not None:
|
||||
title = f"{exc_cause_name(cause)} in {task or 'unknown task'}"
|
||||
return f"exc:{task}:{cause}", "exception", title
|
||||
|
||||
reason = event.get("reset_reason") or "UNKNOWN"
|
||||
return f"reason:{reason}", "reason", f"{reason.replace('_', ' ')} (no coredump)"
|
||||
|
||||
|
||||
def group_crashes(events: list[dict]) -> list[dict]:
|
||||
"""events must be newest-first (as get_crash_events returns them)."""
|
||||
groups: dict[str, dict] = {}
|
||||
for ev in events:
|
||||
key, kind, title = group_key(ev)
|
||||
g = groups.get(key)
|
||||
if g is None:
|
||||
g = groups[key] = {
|
||||
"key": key,
|
||||
"kind": kind,
|
||||
"title": title,
|
||||
"count": 0,
|
||||
"first_at": ev["occurred_at"],
|
||||
"last_at": ev["occurred_at"],
|
||||
"latest": ev,
|
||||
"devices": {},
|
||||
}
|
||||
g["count"] += 1
|
||||
g["first_at"] = ev["occurred_at"] # newest-first input, so the last seen is the oldest
|
||||
d = g["devices"].setdefault(ev["device_serial"], {
|
||||
"device_serial": ev["device_serial"], "count": 0, "last_at": ev["occurred_at"],
|
||||
})
|
||||
d["count"] += 1
|
||||
|
||||
result = []
|
||||
for g in groups.values():
|
||||
devices = sorted(g["devices"].values(), key=lambda d: (-d["count"], d["device_serial"]))
|
||||
result.append({**g, "devices": devices, "device_count": len(devices)})
|
||||
# Most frequent first; ties broken by most recent (two stable sorts).
|
||||
result.sort(key=lambda g: g["last_at"], reverse=True)
|
||||
result.sort(key=lambda g: g["count"], reverse=True)
|
||||
return result
|
||||
+16
-5
@@ -163,17 +163,19 @@ async def _handle_boot_report(serial: str, payload: dict, retained: bool = False
|
||||
info event as { "type": ..., "payload": {...} }.
|
||||
"""
|
||||
data = payload.get("payload", {})
|
||||
crash = data.get("crash") or {}
|
||||
# F-070: `crash` (coredump summary, now incl. backtrace/elf_sha256) and
|
||||
# `pre_crash` (heap/uptime snapshot + optional abort_msg) are stored
|
||||
# verbatim. Older firmware omits them / sends the 4-field crash only.
|
||||
crash = data.get("crash") if isinstance(data.get("crash"), dict) else None
|
||||
pre_crash = data.get("pre_crash") if isinstance(data.get("pre_crash"), dict) else None
|
||||
await db.insert_boot_event(
|
||||
device_serial=serial,
|
||||
boot_count=data.get("boot_count"),
|
||||
reset_reason=data.get("reset_reason"),
|
||||
is_fault=bool(data.get("is_fault", False)),
|
||||
free_heap=data.get("free_heap"),
|
||||
crash_task=crash.get("task"),
|
||||
crash_pc=crash.get("pc"),
|
||||
crash_exc_cause=crash.get("exc_cause"),
|
||||
crash_exc_vaddr=crash.get("exc_vaddr"),
|
||||
crash=crash,
|
||||
pre_crash=pre_crash,
|
||||
# A live boot_report is always a new boot. A retained replay is the
|
||||
# device's last boot, recorded only if we missed it while down.
|
||||
skip_if_latest=retained,
|
||||
@@ -252,6 +254,15 @@ async def _handle_ack(serial: str, payload: dict):
|
||||
await db.insert_ping_sample(device_serial=serial, rtt_ms=rtt_ms)
|
||||
return
|
||||
|
||||
# The device's own SD boot log — merged into device_boot_events whoever
|
||||
# asked for it (Health tab "Sync from device", API reference, Control tab),
|
||||
# so crash detail for boots we never saw live isn't lost.
|
||||
if payload.get("type") == "telemetry.get_boot_history" and status == "SUCCESS":
|
||||
boots = (payload.get("data") or {}).get("boots")
|
||||
if isinstance(boots, list):
|
||||
result = await db.merge_device_boot_history(serial, boots)
|
||||
logger.info(f"Merged device boot history for {serial}: {result}")
|
||||
|
||||
pending = await db.get_pending_command(serial)
|
||||
if pending:
|
||||
cmd_status = "success" if status == "SUCCESS" else "error"
|
||||
|
||||
@@ -129,6 +129,9 @@ class BootEventEntry(BaseModel):
|
||||
crash_pc: Optional[int] = None
|
||||
crash_exc_cause: Optional[int] = None
|
||||
crash_exc_vaddr: Optional[int] = None
|
||||
# Firmware F-070 objects, verbatim — None for older firmware/rows.
|
||||
crash: Optional[Dict[str, Any]] = None
|
||||
pre_crash: Optional[Dict[str, Any]] = None
|
||||
occurred_at: str
|
||||
|
||||
|
||||
@@ -137,6 +140,29 @@ class BootEventListResponse(BaseModel):
|
||||
total: int
|
||||
|
||||
|
||||
class CrashGroupDevice(BaseModel):
|
||||
device_serial: str
|
||||
count: int
|
||||
last_at: str
|
||||
|
||||
|
||||
class CrashGroup(BaseModel):
|
||||
key: str
|
||||
kind: str # "abort" | "exception" | "reason"
|
||||
title: str
|
||||
count: int
|
||||
device_count: int
|
||||
first_at: str
|
||||
last_at: str
|
||||
latest: BootEventEntry
|
||||
devices: List[CrashGroupDevice]
|
||||
|
||||
|
||||
class CrashGroupListResponse(BaseModel):
|
||||
groups: List[CrashGroup]
|
||||
total_crashes: int
|
||||
|
||||
|
||||
class PingSampleEntry(BaseModel):
|
||||
id: int
|
||||
device_serial: str
|
||||
|
||||
@@ -9,7 +9,9 @@ from mqtt.models import (
|
||||
AlertEventListResponse, BootEventListResponse, PingSampleListResponse,
|
||||
DiagnosticsReportListResponse, LatestMetricsResponse,
|
||||
LatestDiagnosticsEntry, LatestPingEntry, DeviceReportListResponse,
|
||||
CrashGroupListResponse,
|
||||
)
|
||||
from mqtt.crash_groups import group_crashes
|
||||
from mqtt.client import mqtt_manager
|
||||
from mqtt import presence
|
||||
import database as db
|
||||
@@ -177,6 +179,17 @@ async def get_device_boot_events(
|
||||
return BootEventListResponse(events=events, total=total)
|
||||
|
||||
|
||||
@router.get("/crash-groups", response_model=CrashGroupListResponse)
|
||||
async def get_crash_groups(
|
||||
since: Optional[datetime] = Query(None, description="ISO timestamp — only crashes at/after this time"),
|
||||
until: Optional[datetime] = Query(None, description="ISO timestamp — only crashes at/before this time"),
|
||||
_user: TokenPayload = Depends(require_permission("mqtt", "view")),
|
||||
):
|
||||
"""Fault boots across the whole fleet, grouped by crash signature."""
|
||||
events = await db.get_crash_events(since=since, until=until)
|
||||
return CrashGroupListResponse(groups=group_crashes(events), total_crashes=len(events))
|
||||
|
||||
|
||||
@router.get("/ping-samples/{device_serial}", response_model=PingSampleListResponse)
|
||||
async def get_device_ping_samples(
|
||||
device_serial: str,
|
||||
|
||||
Reference in New Issue
Block a user