feat(scripts): idempotent backfill of users.device_serials from devices.user_list

One-off script to populate the new device_serials array for assignments
made before the assign/unassign sync existed. Treats each device's
user_list as the source of truth and sets every user's device_serials to
exactly the matching serials, so re-running it is a no-op.

Dry run by default; --apply writes (batched, <=400 per commit).

It also reports users whose doc ID != uid field (MQTT resolves users by the
uid field), users with no uid, devices with users but no serial, and
user_list entries pointing at non-existent users. user_list entries are
accepted as DocumentReferences, "users/{id}" paths or raw doc IDs, and an
entry that matches a uid field rather than a doc ID is mapped to its doc.

Dry run against current data: 13 devices, 13 users, 11 users to update,
0 doc-ID/uid mismatches, 1 dangling user_list entry
(Cx2Va72sUzDbr1T8Ebbh on BSVSPR-26I047-STD10R-88YFJP).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-30 00:01:36 +03:00
co-authored by Claude Opus 5.5
parent 79a87e48f1
commit 85bd2f7e51
@@ -0,0 +1,149 @@
"""
Backfill `device_serials` on Firestore `users` docs from the devices' `user_list`.
`device_serials` is what the MQTT ACL uses to decide which vesper/{serial}/...
topics an app user may reach (see mqtt/app_users.py). New assignments keep it
in sync automatically; this script builds it for assignments that predate that.
The devices' user_list is treated as the source of truth: each user's
device_serials is set to exactly the serials of the devices that list them.
Running it twice is a no-op the second time.
Also reports:
- users whose document ID != their `uid` field (or who have no uid) — the
MQTT layer resolves users by the uid field, so this is informational,
but a user without a uid can never connect via the app
- devices that have users but no serial (can't be reached over MQTT)
- user_list entries that point to users that don't exist
Run from backend/ (locally or inside the backend container):
python scripts/backfill_user_device_serials.py # dry run
python scripts/backfill_user_device_serials.py --apply # write changes
"""
import argparse
import sys
from collections import defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from google.cloud.firestore_v1 import DocumentReference # noqa: E402
from shared.firebase import init_firebase, get_db # noqa: E402
from users.service import device_serial_of # noqa: E402
BATCH_SIZE = 400 # Firestore batch limit is 500 writes
def _entry_user_id(entry) -> str:
"""user_list entries are DocumentReferences, "users/{id}" paths or raw doc IDs."""
if isinstance(entry, DocumentReference):
return entry.id
if isinstance(entry, str) and entry.strip():
return entry.strip().split("/")[-1]
return ""
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--apply", action="store_true", help="write changes (default: dry run)")
args = parser.parse_args()
init_firebase()
db = get_db()
if db is None:
print("[ERROR] Firebase not initialized — check FIREBASE_SERVICE_ACCOUNT_PATH")
return 1
mode = "APPLY" if args.apply else "DRY RUN"
print(f"=== backfill device_serials ({mode}) ===\n")
# 1. devices -> {user_doc_id: {serials}}
wanted: dict[str, set[str]] = defaultdict(set)
devices_without_serial = []
device_count = 0
for doc in db.collection("devices").stream():
device_count += 1
data = doc.to_dict() or {}
user_ids = [u for u in (_entry_user_id(e) for e in (data.get("user_list") or [])) if u]
if not user_ids:
continue
serial = device_serial_of(data)
if not serial:
devices_without_serial.append(doc.id)
continue
for uid in user_ids:
wanted[uid].add(serial)
# 2. users
users = {doc.id: (doc.reference, doc.to_dict() or {}) for doc in db.collection("users").stream()}
by_uid_field = {data.get("uid"): doc_id for doc_id, (_, data) in users.items() if data.get("uid")}
# user_list entries that don't match a doc ID but do match a uid field
dangling = []
for ref_id in list(wanted):
if ref_id in users:
continue
if ref_id in by_uid_field:
wanted[by_uid_field[ref_id]].update(wanted.pop(ref_id))
else:
dangling.append((ref_id, sorted(wanted.pop(ref_id))))
id_mismatch = []
no_uid = []
changes = []
for doc_id, (ref, data) in sorted(users.items()):
uid = data.get("uid") or ""
if not uid:
no_uid.append(doc_id)
elif uid != doc_id:
id_mismatch.append((doc_id, uid))
current = data.get("device_serials")
desired = sorted(wanted.get(doc_id, set()))
if current is None and not desired:
continue # nothing assigned and field absent — leave the doc alone
if list(current or []) != desired:
changes.append((doc_id, ref, list(current or []), desired))
# 3. report
print(f"Scanned {device_count} devices, {len(users)} users.\n")
print(f"Users to update: {len(changes)}")
for doc_id, _, current, desired in changes:
print(f" users/{doc_id}: {current} -> {desired}")
print(f"\nUsers whose doc ID != uid: {len(id_mismatch)}")
for doc_id, uid in id_mismatch:
print(f" users/{doc_id} uid={uid}")
print(f"\nUsers with no uid field (cannot use app MQTT login): {len(no_uid)}")
for doc_id in no_uid:
print(f" users/{doc_id}")
print(f"\nDevices with users but no serial_number/device_id: {len(devices_without_serial)}")
for doc_id in devices_without_serial:
print(f" devices/{doc_id}")
print(f"\nuser_list entries pointing to missing users: {len(dangling)}")
for ref_id, serials in dangling:
print(f" {ref_id} (on devices {serials})")
# 4. write
if not args.apply:
print("\nDry run — nothing written. Re-run with --apply to write.")
return 0
for i in range(0, len(changes), BATCH_SIZE):
batch = db.batch()
for _, ref, _, desired in changes[i:i + BATCH_SIZE]:
batch.update(ref, {"device_serials": desired})
batch.commit()
print(f"\nWrote device_serials on {len(changes)} user docs.")
return 0
if __name__ == "__main__":
sys.exit(main())