Bernt Christian Egeland hai 1 mes
pai
achega
9644b933ca

+ 52 - 0
.github/workflows/deploy-cloud.yml

@@ -105,10 +105,62 @@ jobs:
             exit 1
           fi
 
+      - name: Back up production database
+        env:
+          DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }}
+        run: |
+          set -euo pipefail
+
+          DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|')
+          DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
+
+          BACKUP_DIR="${{ secrets.DATA_PATH }}/db-backups/$DB_NAME"
+          # The volume is root-owned and the runner is not root. Binding the
+          # path into a container makes Docker create it, and the chown hands
+          # it to the runner so the dump redirect below can write there.
+          docker run --rm -v "$BACKUP_DIR":/b alpine chown "$(id -u):$(id -g)" /b
+
+          # The Postgres data directory lives on this filesystem: filling it
+          # takes the database down, not just the backup.
+          MIN_GB=2
+          AVAIL_KB=$(df -Pk "$BACKUP_DIR" | awk 'NR==2 {print $4}')
+          if [ "$AVAIL_KB" -lt $((MIN_GB * 1024 * 1024)) ]; then
+            echo "::error::Only $(awk "BEGIN {printf \"%.1f\", $AVAIL_KB/1024/1024}") GB free where the database lives. Refusing to dump."
+            exit 1
+          fi
+
+          DUMP="$BACKUP_DIR/${DB_NAME}-$(date +%Y%m%d-%H%M%S)-pre-${{ steps.tag.outputs.value }}.dump"
+          echo "Dumping $DB_NAME to $DUMP"
+          docker exec torqvoice-db pg_dump -U "$DB_USER" -d "$DB_NAME" -Fc > "$DUMP"
+
+          # A dump nobody can read is worse than no dump. No backup, no deploy.
+          docker exec -i torqvoice-db pg_restore --list < "$DUMP" > /dev/null
+          echo "Verified $(du -h "$DUMP" | cut -f1) backup"
+
+          # Keep the 10 most recent pre-deploy dumps; hourly dumps in the same
+          # folder belong to the backup cron and age out there.
+          ls -t "$BACKUP_DIR"/*-pre-*.dump 2>/dev/null | tail -n +11 | xargs -r rm --
+
       - name: Deploy
         run: |
           cd $HOME/torqvoice-deploy/prod
           APP_TAG=${{ steps.tag.outputs.value }} docker compose up -d --pull always
 
+      - name: Health check
+        run: |
+          echo "Waiting for prod container to be ready..."
+          for i in $(seq 1 12); do
+            STATUS=$(docker inspect --format='{{.State.Status}}' torqvoice-app 2>/dev/null || echo "not found")
+            if [ "$STATUS" = "running" ]; then
+              echo "Deployed ${{ steps.tag.outputs.value }}."
+              exit 0
+            fi
+            echo "Status: $STATUS, retrying in 5s..."
+            sleep 5
+          done
+          echo "::error::Container failed to start"
+          docker logs torqvoice-app --tail 50
+          exit 1
+
       - name: Prune old images and build cache
         run: docker image prune -f && docker builder prune -f

+ 4 - 3
.github/workflows/deploy-staging.yml

@@ -100,7 +100,7 @@ jobs:
           DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|')
           DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
 
-          BACKUP_DIR="${{ secrets.DATA_PATH }}/db-backups/staging"
+          BACKUP_DIR="${{ secrets.DATA_PATH }}/db-backups/$DB_NAME"
           # The volume is root-owned and the runner is not root. Binding the
           # path into a container makes Docker create it, and the chown hands
           # it to the runner so the dump redirect below can write there.
@@ -126,8 +126,9 @@ jobs:
           docker exec -i torqvoice-db pg_restore --list < "$DUMP" > /dev/null
           echo "Verified $(du -h "$DUMP" | cut -f1) backup"
 
-          # Keep the 10 most recent.
-          ls -t "$BACKUP_DIR"/*.dump 2>/dev/null | tail -n +11 | xargs -r rm --
+          # Keep the 10 most recent pre-deploy dumps; hourly dumps in the same
+          # folder belong to the backup cron and age out there.
+          ls -t "$BACKUP_DIR"/*-pre-*.dump 2>/dev/null | tail -n +11 | xargs -r rm --
 
       - name: Deploy staging
         run: |

+ 123 - 0
.github/workflows/rollback-cloud.yml

@@ -0,0 +1,123 @@
+name: "Rollback: app.torqvoice.com"
+
+# Rolls production back to an earlier image, and optionally restores the
+# database from a dump taken by the deploy workflow.
+#
+# Rolling the image back is the normal case and loses nothing. Restoring the
+# database is a last resort for when the new schema is incompatible with the
+# old code: it discards EVERYTHING CUSTOMERS WROTE since the dump was taken.
+# That is why a database restore additionally requires typing
+# "restore-production" into the confirm field.
+#
+# Reuses the compose file and .env that "Deploy: app.torqvoice.com" wrote, so
+# prod must have been deployed at least once before this can run.
+
+on:
+  workflow_dispatch:
+    inputs:
+      tag:
+        description: "Version tag to roll back to (e.g. v1.2.38). 'latest' is refused."
+        required: true
+      restore_database:
+        description: "DESTRUCTIVE. Also restore the database, discarding everything written since the dump. Leave off unless the old code cannot run against the new schema."
+        type: boolean
+        default: false
+      dump:
+        description: "Dump filename to restore from (e.g. torqvoice-20260815-120000-pre-v1.2.39.dump). Only used when restore_database is on."
+        required: false
+        default: ""
+      confirm:
+        description: "Type restore-production to allow a database restore. Ignored otherwise."
+        required: false
+        default: ""
+
+jobs:
+  rollback:
+    runs-on: [self-hosted, Linux, X64, hetzner]
+
+    steps:
+      - name: Refuse floating tags
+        run: |
+          if [ "${{ inputs.tag }}" = "latest" ] || [ -z "${{ inputs.tag }}" ]; then
+            echo "::error::Roll back to a version tag (e.g. v1.2.38), never 'latest'."
+            exit 1
+          fi
+
+      - name: Show available dumps
+        run: ls -lht "${{ secrets.DATA_PATH }}/db-backups/torqvoice" 2>/dev/null | head -20 || echo "No dumps yet."
+
+      # Everything that can fail is checked before the app is stopped, so a bad
+      # input leaves production untouched and still running.
+      - name: Validate restore request
+        if: inputs.restore_database
+        env:
+          DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }}
+        run: |
+          set -euo pipefail
+
+          if [ "${{ inputs.confirm }}" != "restore-production" ]; then
+            echo "::error::Database restore requested without typing restore-production in the confirm field."
+            exit 1
+          fi
+
+          # This job may only ever drop the production database. If the secret
+          # is ever miswired, refuse rather than drop whatever it points at.
+          DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
+          if [ "$DB_NAME" != "torqvoice" ]; then
+            echo "::error::CLOUD_DATABASE_URL points at \"$DB_NAME\", expected \"torqvoice\". Refusing."
+            exit 1
+          fi
+
+          if [ -z "${{ inputs.dump }}" ]; then
+            echo "::error::restore_database is on but no dump filename was given."
+            exit 1
+          fi
+          DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}"
+          if [ ! -f "$DUMP" ]; then
+            echo "::error::Dump not found: $DUMP"
+            exit 1
+          fi
+          docker exec -i torqvoice-db pg_restore --list < "$DUMP" > /dev/null
+          echo "Dump is readable: $DUMP"
+
+      - name: Stop prod app
+        run: |
+          cd $HOME/torqvoice-deploy/prod
+          docker compose stop torqvoice-app
+
+      - name: Restore database
+        if: inputs.restore_database
+        env:
+          DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }}
+        run: |
+          set -euo pipefail
+          DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}"
+          DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|')
+          DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
+
+          echo "Restoring $DB_NAME from $(basename "$DUMP")"
+          docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "DROP DATABASE \"$DB_NAME\" WITH (FORCE);"
+          docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "CREATE DATABASE \"$DB_NAME\" OWNER \"$DB_USER\";"
+          docker exec -i torqvoice-db pg_restore -U "$DB_USER" -d "$DB_NAME" --no-owner < "$DUMP"
+          echo "Restore complete."
+
+      - name: Deploy previous tag
+        run: |
+          cd $HOME/torqvoice-deploy/prod
+          APP_TAG=${{ inputs.tag }} docker compose up -d --pull always
+
+      - name: Health check
+        run: |
+          echo "Waiting for prod container to be ready..."
+          for i in $(seq 1 12); do
+            STATUS=$(docker inspect --format='{{.State.Status}}' torqvoice-app 2>/dev/null || echo "not found")
+            if [ "$STATUS" = "running" ]; then
+              echo "Rolled back to ${{ inputs.tag }}."
+              exit 0
+            fi
+            echo "Status: $STATUS, retrying in 5s..."
+            sleep 5
+          done
+          echo "::error::Container failed to start after rollback"
+          docker logs torqvoice-app --tail 50
+          exit 1

+ 3 - 3
.github/workflows/rollback-staging.yml

@@ -31,7 +31,7 @@ jobs:
 
     steps:
       - name: Show available dumps
-        run: ls -lht "${{ secrets.DATA_PATH }}/db-backups/staging" 2>/dev/null | head -20 || echo "No dumps yet."
+        run: ls -lht "${{ secrets.DATA_PATH }}/db-backups/torqvoice-staging" 2>/dev/null | head -20 || echo "No dumps yet."
 
       # Everything that can fail is checked before the app is stopped, so a bad
       # input leaves staging untouched and still running.
@@ -63,7 +63,7 @@ jobs:
             echo "::error::restore_database is on but no dump filename was given."
             exit 1
           fi
-          DUMP="${{ secrets.DATA_PATH }}/db-backups/staging/${{ inputs.dump }}"
+          DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice-staging/${{ inputs.dump }}"
           if [ ! -f "$DUMP" ]; then
             echo "::error::Dump not found: $DUMP"
             exit 1
@@ -82,7 +82,7 @@ jobs:
           DATABASE_URL: ${{ secrets.STAGING_DATABASE_URL }}
         run: |
           set -euo pipefail
-          DUMP="${{ secrets.DATA_PATH }}/db-backups/staging/${{ inputs.dump }}"
+          DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice-staging/${{ inputs.dump }}"
           DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|')
           DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')