Files
server-26/.gitea/workflows/deploy.yml
T
Logan CusanoandClaude Sonnet 5 fe643924c7 ci: bake NEXT_PUBLIC_MAP_TILE_URL into the frontend build (#117)
The map override var was added to MapView.tsx but never passed as a build-arg, so prod still shipped the dead Carto tile URL. Point it at OSM raster tiles.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>

Claude-Session: https://claude.ai/code/session_01Tbknwttzou4s46PAykmtix
2026-09-07 18:48:42 -04:00

305 lines
13 KiB
YAML

name: Build & Deploy
on:
push:
branches: [main]
env:
# REGISTRY secret = "git.vpn.cusano.net/logan" (full image prefix)
REGISTRY: ${{ secrets.REGISTRY }}
jobs:
build:
name: Build & push images
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Log in to Gitea registry
uses: docker/login-action@v3
with:
registry: git.vpn.cusano.net
username: ${{ secrets.REGISTRY_USER }}
password: ${{ secrets.BUILD_TOKEN }}
- name: Build & push c2-core
uses: docker/build-push-action@v5
with:
context: ./drb-c2-core
push: true
build-args: |
GIT_SHA=${{ gitea.sha }}
tags: |
${{ env.REGISTRY }}/c2-core:latest
${{ env.REGISTRY }}/c2-core:${{ gitea.sha }}
- name: Build & push discord-bot
uses: docker/build-push-action@v5
with:
context: ./drb-server-discord-bot
push: true
tags: |
${{ env.REGISTRY }}/discord-bot:latest
${{ env.REGISTRY }}/discord-bot:${{ gitea.sha }}
- name: Build & push frontend
uses: docker/build-push-action@v5
with:
context: ./drb-frontend
push: true
tags: |
${{ env.REGISTRY }}/frontend:latest
${{ env.REGISTRY }}/frontend:${{ gitea.sha }}
build-args: |
NEXT_PUBLIC_C2_URL=https://api.${{ secrets.DRB_DOMAIN }}
NEXT_PUBLIC_FIREBASE_API_KEY=${{ secrets.FIREBASE_API_KEY }}
NEXT_PUBLIC_FIREBASE_AUTH_DOMAIN=${{ secrets.FIREBASE_AUTH_DOMAIN }}
NEXT_PUBLIC_FIREBASE_PROJECT_ID=${{ secrets.FIREBASE_PROJECT_ID }}
NEXT_PUBLIC_FIREBASE_STORAGE_BUCKET=${{ secrets.FIREBASE_STORAGE_BUCKET }}
NEXT_PUBLIC_FIREBASE_MESSAGING_SENDER_ID=${{ secrets.FIREBASE_MESSAGING_SENDER_ID }}
NEXT_PUBLIC_FIREBASE_APP_ID=${{ secrets.FIREBASE_APP_ID }}
NEXT_PUBLIC_FIRESTORE_DATABASE=${{ secrets.FIRESTORE_DATABASE }}
NEXT_PUBLIC_MAP_TILE_URL=https://tile.openstreetmap.org/{z}/{x}/{y}.png
deploy:
name: Deploy to VM
needs: build
runs-on: ubuntu-latest
outputs:
prev_sha: ${{ steps.deploy.outputs.prev_sha }}
rollback_status: ${{ steps.rollback.outputs.status }}
rollback_sha: ${{ steps.rollback.outputs.rolled_back_to }}
steps:
- name: Check runner outbound IP
run: curl -s ifconfig.me
- name: Write SSH key
run: |
printf '%s\n' "${{ secrets.SSH_PRIVATE_KEY }}" > /tmp/deploy_key
chmod 600 /tmp/deploy_key
ssh-keygen -l -f /tmp/deploy_key
- name: Deploy
id: deploy
run: |
set -o pipefail
OUTPUT=$(ssh -o StrictHostKeyChecking=no \
-o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \
-o ConnectTimeout=15 \
-v \
-i /tmp/deploy_key \
drb@${{ secrets.SERVER_IP }} << 'ENDSSH' | tee /dev/stderr
set -e
cd /opt/drb
# Update compose files + mosquitto config
git pull origin main
# server-26#65: capture what is actually live BEFORE switching, so
# a bad deploy has something concrete to fall back to. This reads
# from a state file rather than re-deriving it from git log,
# because a PRIOR deploy could itself have failed and already
# rolled back to something older than HEAD~1 -- the file is only
# ever written by the Health check step below, after that step
# has confirmed the tag it names actually answered /health. A
# fresh VM with no file yet falls back to :latest, same escape
# hatch as a manual `up -d` with no TAG set.
PREV_TAG=$(cat /opt/drb/.last_good_tag 2>/dev/null || echo latest)
echo "PREV_TAG=$PREV_TAG"
# Deploy THIS commit's images, not :latest. Overlapping runs are
# normal here, and with :latest whichever finishes last wins for
# both -- run 544 asserted its own SHA and found run 545's build
# already serving. compose already supports ${TAG:-latest}, so
# pinning makes each deploy deterministic and a rollback just a
# different tag. A later manual `up -d` on the VM without TAG set
# still falls back to :latest, which is the intended escape hatch.
export TAG=${{ gitea.sha }}
# Pull pre-built images and restart (no build on the VM).
#
# The retry is not defensive padding: this exact step failed fifteen
# deploys in a row (2026-08-18 to 08-20) with containerd unable to
# extract a layer -- "failed to Lchown ... no such file or directory"
# -- a corrupted entry in the snapshot store. Pruning clears the bad
# layer and the second pull succeeds. If it fails again after a
# prune that is a real problem (check the VM's disk) and should stop
# the deploy rather than be retried forever.
COMPOSE="docker compose -f docker-compose.yml -f docker-compose.prod.yml"
if ! $COMPOSE pull; then
echo "image pull failed - pruning and retrying once"
docker image prune -af
$COMPOSE pull
fi
$COMPOSE up -d --remove-orphans
docker image prune -f
ENDSSH
)
echo "$OUTPUT"
PREV_TAG=$(printf '%s\n' "$OUTPUT" | grep '^PREV_TAG=' | tail -n1 | cut -d'=' -f2)
if [ -z "$PREV_TAG" ]; then
echo "Could not determine the previous tag from deploy output - rollback target unknown."
exit 1
fi
echo "prev_sha=$PREV_TAG" >> "$GITHUB_OUTPUT"
- name: Health check
id: health
run: |
# Poll rather than sleep-once: the container has to finish starting,
# and a fixed sleep is either too short (flaky red) or wastes time on
# every deploy. A health check that cries wolf gets ignored, which is
# the failure mode this whole job exists to prevent.
BODY=""
for _ in $(seq 1 20); do
sleep 5
BODY=$(curl -fsS https://api.${{ secrets.DRB_DOMAIN }}/health) || continue
case "$BODY" in *"${{ gitea.sha }}"*) break ;; esac
done
if [ -z "$BODY" ]; then
echo "Health check failed: /health never responded"; exit 1
fi
echo "$BODY"
# Liveness alone is not enough. A deploy can report success while the
# PREVIOUS container keeps serving -- that is how production ran
# 08-18 code for two days without a single red run. Assert that the
# build which answered is the commit we just pushed.
RUNNING=$(printf '%s' "$BODY" | tr ',' '\n' | grep git_sha | cut -d'"' -f4)
if [ "$RUNNING" != "${{ gitea.sha }}" ]; then
echo "Deployed build is '$RUNNING', expected '${{ gitea.sha }}'."
echo "The container was not actually replaced."
exit 1
fi
# server-26#65: only now -- confirmed by /health, not by "up -d
# returned 0" -- record this as the rollback target for the NEXT
# deploy. A failure to write this is a bookkeeping problem, not a
# deploy problem, so it warns instead of failing the job (a hard
# failure here would trigger the Rollback step below against a
# perfectly good deploy).
ssh -o StrictHostKeyChecking=no \
-o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \
-o ConnectTimeout=15 \
-i /tmp/deploy_key \
drb@${{ secrets.SERVER_IP }} \
"echo '${{ gitea.sha }}' > /opt/drb/.last_good_tag" \
|| echo "warning: failed to persist .last_good_tag - next deploy's rollback target may be stale"
- name: Rollback on failed health check
id: rollback
if: failure()
run: |
# server-26#65 decision 3 / board minutes #62: up -d used to be the
# last word -- a build that passes tests, returns 200, and still
# corrupts incidents on live traffic would stay live for 12+ hours
# before a human noticed. This step is what makes that impossible:
# any failure above (pull, restart, or the health/SHA check) lands
# here and puts the previously-verified tag back.
PREV_TAG="${{ steps.deploy.outputs.prev_sha }}"
if [ -z "$PREV_TAG" ]; then
echo "No previous tag was captured (Deploy step itself failed before recording one) - cannot roll back automatically."
echo "status=skipped" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "Rolling back to $PREV_TAG"
ssh -o StrictHostKeyChecking=no \
-o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \
-o ConnectTimeout=15 \
-i /tmp/deploy_key \
drb@${{ secrets.SERVER_IP }} << ENDSSH
set -e
cd /opt/drb
export TAG=$PREV_TAG
COMPOSE="docker compose -f docker-compose.yml -f docker-compose.prod.yml"
if ! \$COMPOSE pull; then
echo "rollback image pull failed - pruning and retrying once"
docker image prune -af
\$COMPOSE pull
fi
\$COMPOSE up -d --remove-orphans
ENDSSH
# Re-verify exactly like the forward health check does: liveness
# alone doesn't prove the rollback took, the SHA has to match the
# tag we just switched back to.
BODY=""
for _ in $(seq 1 12); do
sleep 5
BODY=$(curl -fsS https://api.${{ secrets.DRB_DOMAIN }}/health) || continue
case "$BODY" in *"$PREV_TAG"*) break ;; esac
done
RUNNING=$(printf '%s' "$BODY" | tr ',' '\n' | grep git_sha | cut -d'"' -f4)
if [ "$RUNNING" != "$PREV_TAG" ]; then
echo "ROLLBACK FAILED: expected git_sha '$PREV_TAG', got '$RUNNING'."
echo "Production state is UNKNOWN - check the VM by hand immediately."
echo "status=failed" >> "$GITHUB_OUTPUT"
echo "rolled_back_to=$PREV_TAG" >> "$GITHUB_OUTPUT"
exit 1
fi
echo "Rolled back successfully to $PREV_TAG"
echo "status=success" >> "$GITHUB_OUTPUT"
echo "rolled_back_to=$PREV_TAG" >> "$GITHUB_OUTPUT"
notify-failure:
name: Report a failed deploy
needs: [build, deploy]
if: failure()
runs-on: ubuntu-latest
steps:
- name: Post to Discord
# A red run in Gitea is only visible to someone who opens Gitea, and
# nobody did for two days. Same shape as an AI tier dying quietly,
# which is why both now push a message out of the box instead of
# waiting to be discovered. No webhook configured => skip quietly
# rather than fail, since not every deployment will set one.
env:
WEBHOOK: ${{ secrets.DEPLOY_ALERT_WEBHOOK }}
RUN_URL: ${{ gitea.server_url }}/${{ gitea.repository }}/actions/runs/${{ gitea.run_number }}
SHA: ${{ gitea.sha }}
ROLLBACK_STATUS: ${{ needs.deploy.outputs.rollback_status }}
ROLLBACK_SHA: ${{ needs.deploy.outputs.rollback_sha }}
run: |
if [ -z "$WEBHOOK" ]; then
echo "DEPLOY_ALERT_WEBHOOK is not set - skipping notification."
exit 0
fi
python3 - <<'PY' > /tmp/payload.json
import json, os
sha = os.environ["SHA"][:8]
run_url = os.environ["RUN_URL"]
status = os.environ.get("ROLLBACK_STATUS", "")
rollback_sha = os.environ.get("ROLLBACK_SHA", "")
# server-26#65: the old text here unconditionally claimed
# "production is still running the previous build" -- true only
# when the pull/restart itself failed. It's false the moment a
# build passes the SHA check but has a live logic bug (exactly the
# class of bug the correlator instrumentation exists to catch), or
# once the deploy job's own rollback path has run. Say what
# actually happened instead.
if status == "success":
detail = "Automatic rollback to `%s` succeeded. Production is back on the previous good build." % rollback_sha[:8]
elif status == "failed":
detail = ("Automatic rollback to `%s` FAILED. Production state is UNKNOWN -- "
"check the VM by hand immediately.") % rollback_sha[:8]
elif status == "skipped":
detail = "No rollback was attempted (no previous tag captured, or build/push failed before any deploy). Check the VM by hand."
else:
detail = "Build failed before any deploy was attempted. Production is unchanged."
print(json.dumps({"content":
"**DRB deploy failed** on `%s`\n%s\n%s" % (sha, run_url, detail)}))
PY
curl -sS -X POST -H "Content-Type: application/json" \
--data @/tmp/payload.json "$WEBHOOK" || echo "notification POST failed"