name: Build & Deploy on: push: branches: [main] env: # REGISTRY secret = "git.vpn.cusano.net/logan" (full image prefix) REGISTRY: ${{ secrets.REGISTRY }} jobs: build: name: Build & push images runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - name: Log in to Gitea registry uses: docker/login-action@v3 with: registry: git.vpn.cusano.net username: ${{ secrets.REGISTRY_USER }} password: ${{ secrets.BUILD_TOKEN }} - name: Build & push c2-core uses: docker/build-push-action@v5 with: context: ./drb-c2-core push: true build-args: | GIT_SHA=${{ gitea.sha }} tags: | ${{ env.REGISTRY }}/c2-core:latest ${{ env.REGISTRY }}/c2-core:${{ gitea.sha }} - name: Build & push discord-bot uses: docker/build-push-action@v5 with: context: ./drb-server-discord-bot push: true tags: | ${{ env.REGISTRY }}/discord-bot:latest ${{ env.REGISTRY }}/discord-bot:${{ gitea.sha }} - name: Build & push frontend uses: docker/build-push-action@v5 with: context: ./drb-frontend push: true tags: | ${{ env.REGISTRY }}/frontend:latest ${{ env.REGISTRY }}/frontend:${{ gitea.sha }} build-args: | NEXT_PUBLIC_C2_URL=https://api.${{ secrets.DRB_DOMAIN }} NEXT_PUBLIC_FIREBASE_API_KEY=${{ secrets.FIREBASE_API_KEY }} NEXT_PUBLIC_FIREBASE_AUTH_DOMAIN=${{ secrets.FIREBASE_AUTH_DOMAIN }} NEXT_PUBLIC_FIREBASE_PROJECT_ID=${{ secrets.FIREBASE_PROJECT_ID }} NEXT_PUBLIC_FIREBASE_STORAGE_BUCKET=${{ secrets.FIREBASE_STORAGE_BUCKET }} NEXT_PUBLIC_FIREBASE_MESSAGING_SENDER_ID=${{ secrets.FIREBASE_MESSAGING_SENDER_ID }} NEXT_PUBLIC_FIREBASE_APP_ID=${{ secrets.FIREBASE_APP_ID }} NEXT_PUBLIC_FIRESTORE_DATABASE=${{ secrets.FIRESTORE_DATABASE }} deploy: name: Deploy to VM needs: build runs-on: ubuntu-latest outputs: prev_sha: ${{ steps.deploy.outputs.prev_sha }} rollback_status: ${{ steps.rollback.outputs.status }} rollback_sha: ${{ steps.rollback.outputs.rolled_back_to }} steps: - name: Check runner outbound IP run: curl -s ifconfig.me - name: Write SSH key run: | printf '%s\n' "${{ secrets.SSH_PRIVATE_KEY }}" > /tmp/deploy_key chmod 600 /tmp/deploy_key ssh-keygen -l -f /tmp/deploy_key - name: Deploy id: deploy run: | set -o pipefail OUTPUT=$(ssh -o StrictHostKeyChecking=no \ -o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \ -o ConnectTimeout=15 \ -v \ -i /tmp/deploy_key \ drb@${{ secrets.SERVER_IP }} << 'ENDSSH' | tee /dev/stderr set -e cd /opt/drb # Update compose files + mosquitto config git pull origin main # server-26#65: capture what is actually live BEFORE switching, so # a bad deploy has something concrete to fall back to. This reads # from a state file rather than re-deriving it from git log, # because a PRIOR deploy could itself have failed and already # rolled back to something older than HEAD~1 -- the file is only # ever written by the Health check step below, after that step # has confirmed the tag it names actually answered /health. A # fresh VM with no file yet falls back to :latest, same escape # hatch as a manual `up -d` with no TAG set. PREV_TAG=$(cat /opt/drb/.last_good_tag 2>/dev/null || echo latest) echo "PREV_TAG=$PREV_TAG" # Deploy THIS commit's images, not :latest. Overlapping runs are # normal here, and with :latest whichever finishes last wins for # both -- run 544 asserted its own SHA and found run 545's build # already serving. compose already supports ${TAG:-latest}, so # pinning makes each deploy deterministic and a rollback just a # different tag. A later manual `up -d` on the VM without TAG set # still falls back to :latest, which is the intended escape hatch. export TAG=${{ gitea.sha }} # Pull pre-built images and restart (no build on the VM). # # The retry is not defensive padding: this exact step failed fifteen # deploys in a row (2026-08-18 to 08-20) with containerd unable to # extract a layer -- "failed to Lchown ... no such file or directory" # -- a corrupted entry in the snapshot store. Pruning clears the bad # layer and the second pull succeeds. If it fails again after a # prune that is a real problem (check the VM's disk) and should stop # the deploy rather than be retried forever. COMPOSE="docker compose -f docker-compose.yml -f docker-compose.prod.yml" if ! $COMPOSE pull; then echo "image pull failed - pruning and retrying once" docker image prune -af $COMPOSE pull fi $COMPOSE up -d --remove-orphans docker image prune -f ENDSSH ) echo "$OUTPUT" PREV_TAG=$(printf '%s\n' "$OUTPUT" | grep '^PREV_TAG=' | tail -n1 | cut -d'=' -f2) if [ -z "$PREV_TAG" ]; then echo "Could not determine the previous tag from deploy output - rollback target unknown." exit 1 fi echo "prev_sha=$PREV_TAG" >> "$GITHUB_OUTPUT" - name: Health check id: health run: | # Poll rather than sleep-once: the container has to finish starting, # and a fixed sleep is either too short (flaky red) or wastes time on # every deploy. A health check that cries wolf gets ignored, which is # the failure mode this whole job exists to prevent. BODY="" for _ in $(seq 1 20); do sleep 5 BODY=$(curl -fsS https://api.${{ secrets.DRB_DOMAIN }}/health) || continue case "$BODY" in *"${{ gitea.sha }}"*) break ;; esac done if [ -z "$BODY" ]; then echo "Health check failed: /health never responded"; exit 1 fi echo "$BODY" # Liveness alone is not enough. A deploy can report success while the # PREVIOUS container keeps serving -- that is how production ran # 08-18 code for two days without a single red run. Assert that the # build which answered is the commit we just pushed. RUNNING=$(printf '%s' "$BODY" | tr ',' '\n' | grep git_sha | cut -d'"' -f4) if [ "$RUNNING" != "${{ gitea.sha }}" ]; then echo "Deployed build is '$RUNNING', expected '${{ gitea.sha }}'." echo "The container was not actually replaced." exit 1 fi # server-26#65: only now -- confirmed by /health, not by "up -d # returned 0" -- record this as the rollback target for the NEXT # deploy. A failure to write this is a bookkeeping problem, not a # deploy problem, so it warns instead of failing the job (a hard # failure here would trigger the Rollback step below against a # perfectly good deploy). ssh -o StrictHostKeyChecking=no \ -o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \ -o ConnectTimeout=15 \ -i /tmp/deploy_key \ drb@${{ secrets.SERVER_IP }} \ "echo '${{ gitea.sha }}' > /opt/drb/.last_good_tag" \ || echo "warning: failed to persist .last_good_tag - next deploy's rollback target may be stale" - name: Rollback on failed health check id: rollback if: failure() run: | # server-26#65 decision 3 / board minutes #62: up -d used to be the # last word -- a build that passes tests, returns 200, and still # corrupts incidents on live traffic would stay live for 12+ hours # before a human noticed. This step is what makes that impossible: # any failure above (pull, restart, or the health/SHA check) lands # here and puts the previously-verified tag back. PREV_TAG="${{ steps.deploy.outputs.prev_sha }}" if [ -z "$PREV_TAG" ]; then echo "No previous tag was captured (Deploy step itself failed before recording one) - cannot roll back automatically." echo "status=skipped" >> "$GITHUB_OUTPUT" exit 0 fi echo "Rolling back to $PREV_TAG" ssh -o StrictHostKeyChecking=no \ -o HostKeyAlgorithms=ssh-ed25519,rsa-sha2-256,rsa-sha2-512 \ -o ConnectTimeout=15 \ -i /tmp/deploy_key \ drb@${{ secrets.SERVER_IP }} << ENDSSH set -e cd /opt/drb export TAG=$PREV_TAG COMPOSE="docker compose -f docker-compose.yml -f docker-compose.prod.yml" if ! \$COMPOSE pull; then echo "rollback image pull failed - pruning and retrying once" docker image prune -af \$COMPOSE pull fi \$COMPOSE up -d --remove-orphans ENDSSH # Re-verify exactly like the forward health check does: liveness # alone doesn't prove the rollback took, the SHA has to match the # tag we just switched back to. BODY="" for _ in $(seq 1 12); do sleep 5 BODY=$(curl -fsS https://api.${{ secrets.DRB_DOMAIN }}/health) || continue case "$BODY" in *"$PREV_TAG"*) break ;; esac done RUNNING=$(printf '%s' "$BODY" | tr ',' '\n' | grep git_sha | cut -d'"' -f4) if [ "$RUNNING" != "$PREV_TAG" ]; then echo "ROLLBACK FAILED: expected git_sha '$PREV_TAG', got '$RUNNING'." echo "Production state is UNKNOWN - check the VM by hand immediately." echo "status=failed" >> "$GITHUB_OUTPUT" echo "rolled_back_to=$PREV_TAG" >> "$GITHUB_OUTPUT" exit 1 fi echo "Rolled back successfully to $PREV_TAG" echo "status=success" >> "$GITHUB_OUTPUT" echo "rolled_back_to=$PREV_TAG" >> "$GITHUB_OUTPUT" notify-failure: name: Report a failed deploy needs: [build, deploy] if: failure() runs-on: ubuntu-latest steps: - name: Post to Discord # A red run in Gitea is only visible to someone who opens Gitea, and # nobody did for two days. Same shape as an AI tier dying quietly, # which is why both now push a message out of the box instead of # waiting to be discovered. No webhook configured => skip quietly # rather than fail, since not every deployment will set one. env: WEBHOOK: ${{ secrets.DEPLOY_ALERT_WEBHOOK }} RUN_URL: ${{ gitea.server_url }}/${{ gitea.repository }}/actions/runs/${{ gitea.run_number }} SHA: ${{ gitea.sha }} ROLLBACK_STATUS: ${{ needs.deploy.outputs.rollback_status }} ROLLBACK_SHA: ${{ needs.deploy.outputs.rollback_sha }} run: | if [ -z "$WEBHOOK" ]; then echo "DEPLOY_ALERT_WEBHOOK is not set - skipping notification." exit 0 fi python3 - <<'PY' > /tmp/payload.json import json, os sha = os.environ["SHA"][:8] run_url = os.environ["RUN_URL"] status = os.environ.get("ROLLBACK_STATUS", "") rollback_sha = os.environ.get("ROLLBACK_SHA", "") # server-26#65: the old text here unconditionally claimed # "production is still running the previous build" -- true only # when the pull/restart itself failed. It's false the moment a # build passes the SHA check but has a live logic bug (exactly the # class of bug the correlator instrumentation exists to catch), or # once the deploy job's own rollback path has run. Say what # actually happened instead. if status == "success": detail = "Automatic rollback to `%s` succeeded. Production is back on the previous good build." % rollback_sha[:8] elif status == "failed": detail = ("Automatic rollback to `%s` FAILED. Production state is UNKNOWN -- " "check the VM by hand immediately.") % rollback_sha[:8] elif status == "skipped": detail = "No rollback was attempted (no previous tag captured, or build/push failed before any deploy). Check the VM by hand." else: detail = "Build failed before any deploy was attempted. Production is unchanged." print(json.dumps({"content": "**DRB deploy failed** on `%s`\n%s\n%s" % (sha, run_url, detail)})) PY curl -sS -X POST -H "Content-Type: application/json" \ --data @/tmp/payload.json "$WEBHOOK" || echo "notification POST failed"