feat(mqtt): add device health telemetry and migrate to v2 topic spec

Two efforts that landed together because the v2 topic work extends
tables the health-telemetry effort added days earlier in the same
files/functions, making them impractical to separate cleanly:

Health/diagnostics telemetry (schema, Jul 13-17):
- New Postgres tables: device_alert_events, device_boot_events,
  device_ping_samples, device_diagnostics_reports, plus a `source`
  column on device_logs to distinguish log origins
- Query/service layer in pg_mqtt.py and database/__init__.py for
  inserting and listing this history, plus a "latest metrics" endpoint
  combining most-recent diagnostics + ping RTT per device
- mqtt/router.py gains list endpoints for alert/boot/ping/diagnostics
  history, consumed by the upcoming Health tab

MQTT v2 topic migration (Sep 21):
- Heartbeat payload flattened per vesper_mqtt_topic_spec_v2.md, adding
  rssi/free_heap/state/ok fields
- Command replies move to control/ack, device-initiated events to
  control/reports; mqtt/client.py subscribes to the new topic set and
  runs a ping_loop (wired up in main.py) for RTT sampling
- mqtt/logger.py and pg_mqtt.py updated to parse and persist the new
  payload shape alongside the legacy fields for backwards compatibility

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-21 18:24:36 +03:00
co-authored by Claude Sonnet 5
parent e5d556fee1
commit 7c533b9245
12 changed files with 1456 additions and 55 deletions
@@ -0,0 +1,73 @@
"""device diagnostics reports
Adds device_diagnostics_reports — structured storage for the firmware's
diagnostics_report MQTT event (vesper/{uid}/status/info, type="diagnostics_report"),
published every 5 minutes. Cleanly separate from device_boot_events (one row per
boot, event-driven) and heartbeats (one row every 30s, transport/liveness facts):
this table is periodic health telemetry — CPU temperature (min/max/avg over the
5-minute window), WiFi reconnect count + last disconnect reason, OTA/firmware
check state, and per-task stack high-water marks (bytes free).
stack_high_water is stored as a JSON-encoded TEXT column rather than flattened
columns — unlike the other three groups (fixed field sets), the set of
monitored tasks is open-ended on the firmware side (see project-vesper's
Telemetry::registerTaskForStackMonitoring), so a fixed column per task would
need a migration every time a task is added. TEXT (not JSONB) matches this
codebase's existing convention for JSON blobs stored via raw SQL — see
commands.command_payload / response_payload.
Revision ID: a7b8c9d0e1f2
Revises: f6a7b8c9d0e1
Create Date: 2026-07-17 00:00:00.000000
"""
from typing import Sequence, Union
import sqlalchemy as sa
from alembic import op
revision: str = "a7b8c9d0e1f2"
down_revision: Union[str, None] = "f6a7b8c9d0e1"
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
op.create_table(
"device_diagnostics_reports",
sa.Column("id", sa.BigInteger(), primary_key=True, autoincrement=True),
sa.Column("device_serial", sa.String(128), nullable=False),
# cpu_temp — omitted by firmware (all null here) if no samples were
# taken yet this window (e.g. very first report shortly after boot).
sa.Column("cpu_temp_avg", sa.Float(), nullable=True),
sa.Column("cpu_temp_min", sa.Float(), nullable=True),
sa.Column("cpu_temp_max", sa.Float(), nullable=True),
sa.Column("cpu_temp_samples", sa.Integer(), nullable=True),
# wifi_reconnects — this-boot-only lifetime count, not device lifetime.
sa.Column("wifi_reconnect_count", sa.Integer(), nullable=True),
sa.Column("wifi_last_disconnect_reason", sa.String(64), nullable=True),
sa.Column("wifi_last_disconnect_uptime_ms", sa.BigInteger(), nullable=True),
# ota
sa.Column("ota_current_version", sa.String(32), nullable=True),
sa.Column("ota_update_available", sa.Boolean(), nullable=True),
sa.Column("ota_available_version", sa.String(32), nullable=True),
sa.Column("ota_last_check_uptime_ms", sa.BigInteger(), nullable=True),
sa.Column("ota_last_error", sa.String(32), nullable=True),
# stack_high_water — open-ended task set, see module docstring.
sa.Column("stack_high_water", sa.Text(), nullable=True),
sa.Column("received_at", sa.DateTime(timezone=True), nullable=False,
server_default=sa.func.now()),
)
op.create_index(
"idx_device_diagnostics_reports_serial_received",
"device_diagnostics_reports",
["device_serial", sa.text("received_at DESC")],
)
def downgrade() -> None:
op.drop_index("idx_device_diagnostics_reports_serial_received", table_name="device_diagnostics_reports")
op.drop_table("device_diagnostics_reports")