name: "[Prod] Rollback" # Rolls production back to an earlier image, and optionally restores the # database from a dump taken by the deploy workflow. # # Rolling the image back is the normal case and loses nothing. Restoring the # database is a last resort for when the new schema is incompatible with the # old code: it discards EVERYTHING CUSTOMERS WROTE since the dump was taken. # That is why a database restore additionally requires typing # "restore-production" into the confirm field. # # Reuses the compose file and .env that "Deploy: app.torqvoice.com" wrote, so # prod must have been deployed at least once before this can run. on: workflow_dispatch: inputs: tag: description: "Version tag to roll back to (e.g. v1.2.38). 'latest' is refused." required: true restore_database: description: "DESTRUCTIVE. Also restore the database, discarding everything written since the dump. Leave off unless the old code cannot run against the new schema." type: boolean default: false dump: description: "Dump filename to restore from (e.g. torqvoice-20260815-120000-pre-v1.2.39.dump). Only used when restore_database is on." required: false default: "" confirm: description: "Type restore-production to allow a database restore. Ignored otherwise." required: false default: "" jobs: rollback: runs-on: [self-hosted, Linux, X64, hetzner] steps: - name: Refuse floating tags run: | if [ "${{ inputs.tag }}" = "latest" ] || [ -z "${{ inputs.tag }}" ]; then echo "::error::Roll back to a version tag (e.g. v1.2.38), never 'latest'." exit 1 fi - name: Show available dumps run: ls -lht "${{ secrets.DATA_PATH }}/db-backups/torqvoice" 2>/dev/null | head -20 || echo "No dumps yet." # Everything that can fail is checked before the app is stopped, so a bad # input leaves production untouched and still running. - name: Validate restore request if: inputs.restore_database env: DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }} run: | set -euo pipefail if [ "${{ inputs.confirm }}" != "restore-production" ]; then echo "::error::Database restore requested without typing restore-production in the confirm field." exit 1 fi # This job may only ever drop the production database. If the secret # is ever miswired, refuse rather than drop whatever it points at. DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|') if [ "$DB_NAME" != "torqvoice" ]; then echo "::error::CLOUD_DATABASE_URL points at \"$DB_NAME\", expected \"torqvoice\". Refusing." exit 1 fi if [ -z "${{ inputs.dump }}" ]; then echo "::error::restore_database is on but no dump filename was given." exit 1 fi # The dump must belong to the production database: a staging dump # here would replace real customer data with test data. case "${{ inputs.dump }}" in "$DB_NAME"-[0-9]*) ;; *) echo "::error::Dump \"${{ inputs.dump }}\" is not a $DB_NAME dump. Refusing." exit 1 ;; esac DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}" if [ ! -f "$DUMP" ]; then echo "::error::Dump not found: $DUMP" exit 1 fi docker exec -i torqvoice-db pg_restore --list < "$DUMP" > /dev/null echo "Dump is readable: $DUMP" - name: Stop prod app run: | cd $HOME/torqvoice-deploy/prod docker compose stop torqvoice-app - name: Restore database if: inputs.restore_database env: DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }} run: | set -euo pipefail DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}" DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|') DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|') echo "Restoring $DB_NAME from $(basename "$DUMP")" docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "DROP DATABASE \"$DB_NAME\" WITH (FORCE);" docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "CREATE DATABASE \"$DB_NAME\" OWNER \"$DB_USER\";" docker exec -i torqvoice-db pg_restore -U "$DB_USER" -d "$DB_NAME" --no-owner < "$DUMP" echo "Restore complete." - name: Deploy previous tag run: | cd $HOME/torqvoice-deploy/prod APP_TAG=${{ inputs.tag }} docker compose up -d --pull always - name: Health check run: | echo "Waiting for prod container to be ready..." for i in $(seq 1 12); do STATUS=$(docker inspect --format='{{.State.Status}}' torqvoice-app 2>/dev/null || echo "not found") if [ "$STATUS" = "running" ]; then echo "Rolled back to ${{ inputs.tag }}." exit 0 fi echo "Status: $STATUS, retrying in 5s..." sleep 5 done echo "::error::Container failed to start after rollback" docker logs torqvoice-app --tail 50 exit 1