Files
server-26/drb-c2-core/app/main.py
T
Logan Cusano a2cd2c57ca
Build & Deploy / Build & push images (push) Successful in 4m0s
Build & Deploy / Deploy to VM (push) Failing after 2m34s
Serve call audio through c2-core instead of GCS signed URLs
upload_audio() could only sign a URL when GCP_CREDENTIALS_PATH pointed at a
service-account key file. The deployed VM runs on Application Default
Credentials with no key file, so every upload silently took the fallback
branch and returned a bare gs:// URI. That broke two things at once:

  * Browsers cannot fetch a gs:// URI, so no recording was ever playable.
  * _public_url_to_gcs_uri() only matched https://storage.googleapis.com/ and
    returned None for it, so `if gcs_uri:` in the upload path was always false
    and transcription never ran. Nothing was logged, which is why this looked
    like an OpenAI credits problem rather than a storage one.

The fallback also interpolated the client-supplied filename instead of the
call_id-derived safe name, so the URI did not even name the object written.

Calls now store only the canonical gs:// location. A short-lived playback link
is minted per read as an HMAC over (call_id, expiry) keyed by SERVICE_KEY, and
audio is served from the private bucket by the new /media route. An <audio src>
cannot carry an Authorization header, so the link has to be the credential;
that router is therefore public with the check done inline, as enrollment.py
already does. Signing GCS URLs from the VM would have needed a
serviceAccountTokenCreator grant on its own service account — this avoids the
IAM change entirely and keeps the bucket private.

gcs_uri_for_call() reconstructs the object name from call_id, so recordings
made before this fix are reachable again without a data migration.

Frontend rows come straight from Firestore via onSnapshot and never see a
server-minted field, so CallRow fetches the link lazily on expand.

Also removes the last long-lived (1 year) signed URL and the log line that
printed it.
2026-08-16 16:26:41 -04:00

121 lines
5.7 KiB
Python

import asyncio
from contextlib import asynccontextmanager
from fastapi import FastAPI, Depends
from fastapi.middleware.cors import CORSMiddleware
from app.internal.logger import logger
from app.internal.mqtt_handler import mqtt_handler
from app.internal.node_sweeper import sweeper_loop
from app.internal.summarizer import summarizer_loop
from app.internal.vocabulary_learner import vocabulary_induction_loop
from app.internal.recorrelation_sweep import recorrelation_loop
from app.config import settings
from app.internal.auth import (
require_firebase_token,
require_service_or_firebase_token,
require_node_service_or_firebase_token,
)
from app.routers import nodes, systems, calls, upload, tokens, incidents, alerts, admin, trips, places, links, users
from app.routers import enrollment, media
from app.internal import dynsec
from app.internal import firestore as fstore
async def _release_orphaned_tokens():
"""Release all in-use tokens on startup — voice connections don't survive server restarts."""
def _find():
from app.internal.firestore import db
return [d for d in db.collection("bot_tokens").where("in_use", "==", True).stream()]
results = await asyncio.to_thread(_find)
for doc in results:
await fstore.doc_update("bot_tokens", doc.id, {
"in_use": False,
"assigned_node_id": None,
"assigned_at": None,
})
if results:
logger.info(f"Released {len(results)} orphaned token(s) on startup.")
@asynccontextmanager
async def lifespan(app: FastAPI):
logger.info("DRB C2 Core starting.")
await _release_orphaned_tokens()
# dynsec bootstrap + reconcile — must happen before mqtt_handler.connect()
# so that by the time the app is serving requests, c2-core's own dynsec
# client/roles exist and every already-approved node's dynsec client
# matches Firestore (see app/internal/dynsec.py "TWO-SOURCES-OF-TRUTH").
# Non-fatal by design: if the broker or MQTT_DYNSEC_ADMIN_PASS isn't
# reachable/configured yet (e.g. first-ever deploy, mosquitto still
# starting), log loudly and keep booting rather than crash-looping
# c2-core itself — mqtt_handler.connect() below has its own retry loop
# and node approval/reissue endpoints fail loudly on their own if dynsec
# calls fail later, so nothing here is silently swallowed forever.
try:
await dynsec.ensure_roles_and_c2core_grant()
await dynsec.reconcile_all()
except dynsec.DynsecError as e:
logger.error(f"dynsec bootstrap/reconcile failed — node approval/reissue will fail until this is resolved: {e}")
await mqtt_handler.connect()
sweeper_task = asyncio.create_task(sweeper_loop())
summarizer_task = asyncio.create_task(summarizer_loop())
induction_task = asyncio.create_task(vocabulary_induction_loop())
recorrelation_task = asyncio.create_task(recorrelation_loop())
yield # --- app running ---
logger.info("DRB C2 Core shutting down.")
sweeper_task.cancel()
summarizer_task.cancel()
induction_task.cancel()
recorrelation_task.cancel()
await mqtt_handler.disconnect()
app = FastAPI(title="DRB C2 Core", lifespan=lifespan)
app.add_middleware(
CORSMiddleware,
allow_origins=settings.cors_origins,
allow_methods=["*"],
allow_headers=["*"],
allow_credentials=True,
)
app.include_router(nodes.router, dependencies=[Depends(require_service_or_firebase_token)])
# systems is the one router edge nodes read directly (system_cacher.py builds
# the OP25 config from it), so its gate also accepts a per-node api_key. The
# write routes inside carry their own require_admin_token, so nodes get read
# access only.
app.include_router(systems.router, dependencies=[Depends(require_node_service_or_firebase_token)])
app.include_router(calls.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(tokens.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(incidents.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(alerts.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(trips.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(places.router, dependencies=[Depends(require_service_or_firebase_token)])
app.include_router(upload.router) # auth is per-node, handled inline
app.include_router(admin.router) # auth is per-endpoint (read: firebase, write: admin)
app.include_router(users.router) # auth: admin only
app.include_router(links.router) # auth is per-endpoint (generate: firebase, resolve: service key)
app.include_router(enrollment.router) # public; auth is the enrollment/pickup-secret tokens, checked inline
# public by necessity — an <audio src> can't send a bearer token, so the
# short-lived HMAC in the URL is the credential. Checked inline in media.py.
app.include_router(media.router)
# NOTE: there used to be an app.routers.mqtt_auth router here (an HTTP
# backend for the mosquitto-go-auth plugin). That plugin's upstream project
# is archived (no CVE patches) and was rejected for an internet-facing
# broker — see MQTT-PUBLIC-AUTH-PLAN.md. MQTT auth is now mosquitto's own
# built-in dynamic-security plugin (app/internal/dynsec.py talks to it over
# MQTT control topics, not HTTP), so there is nothing at /internal/mqtt/*
# anymore. Caddy's Caddyfile.j2 still 404s /internal/* on api.<domain> as
# defence in depth even though nothing calls it today — cheap insurance
# against a future /internal/* route being added and forgotten there.
@app.get("/health")
async def health():
return {"ok": True, "mqtt_connected": mqtt_handler.is_connected}