rollback-cloud.yml 5.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132
  1. name: "[Prod] Rollback"
  2. # Rolls production back to an earlier image, and optionally restores the
  3. # database from a dump taken by the deploy workflow.
  4. #
  5. # Rolling the image back is the normal case and loses nothing. Restoring the
  6. # database is a last resort for when the new schema is incompatible with the
  7. # old code: it discards EVERYTHING CUSTOMERS WROTE since the dump was taken.
  8. # That is why a database restore additionally requires typing
  9. # "restore-production" into the confirm field.
  10. #
  11. # Reuses the compose file and .env that "Deploy: app.torqvoice.com" wrote, so
  12. # prod must have been deployed at least once before this can run.
  13. on:
  14. workflow_dispatch:
  15. inputs:
  16. tag:
  17. description: "Version tag to roll back to (e.g. v1.2.38). 'latest' is refused."
  18. required: true
  19. restore_database:
  20. description: "DESTRUCTIVE. Also restore the database, discarding everything written since the dump. Leave off unless the old code cannot run against the new schema."
  21. type: boolean
  22. default: false
  23. dump:
  24. description: "Dump filename to restore from (e.g. torqvoice-20260815-120000-pre-v1.2.39.dump). Only used when restore_database is on."
  25. required: false
  26. default: ""
  27. confirm:
  28. description: "Type restore-production to allow a database restore. Ignored otherwise."
  29. required: false
  30. default: ""
  31. jobs:
  32. rollback:
  33. runs-on: [self-hosted, Linux, X64, hetzner]
  34. steps:
  35. - name: Refuse floating tags
  36. run: |
  37. if [ "${{ inputs.tag }}" = "latest" ] || [ -z "${{ inputs.tag }}" ]; then
  38. echo "::error::Roll back to a version tag (e.g. v1.2.38), never 'latest'."
  39. exit 1
  40. fi
  41. - name: Show available dumps
  42. run: ls -lht "${{ secrets.DATA_PATH }}/db-backups/torqvoice" 2>/dev/null | head -20 || echo "No dumps yet."
  43. # Everything that can fail is checked before the app is stopped, so a bad
  44. # input leaves production untouched and still running.
  45. - name: Validate restore request
  46. if: inputs.restore_database
  47. env:
  48. DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }}
  49. run: |
  50. set -euo pipefail
  51. if [ "${{ inputs.confirm }}" != "restore-production" ]; then
  52. echo "::error::Database restore requested without typing restore-production in the confirm field."
  53. exit 1
  54. fi
  55. # This job may only ever drop the production database. If the secret
  56. # is ever miswired, refuse rather than drop whatever it points at.
  57. DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
  58. if [ "$DB_NAME" != "torqvoice" ]; then
  59. echo "::error::CLOUD_DATABASE_URL points at \"$DB_NAME\", expected \"torqvoice\". Refusing."
  60. exit 1
  61. fi
  62. if [ -z "${{ inputs.dump }}" ]; then
  63. echo "::error::restore_database is on but no dump filename was given."
  64. exit 1
  65. fi
  66. # The dump must belong to the production database: a staging dump
  67. # here would replace real customer data with test data.
  68. case "${{ inputs.dump }}" in
  69. "$DB_NAME"-[0-9]*) ;;
  70. *)
  71. echo "::error::Dump \"${{ inputs.dump }}\" is not a $DB_NAME dump. Refusing."
  72. exit 1
  73. ;;
  74. esac
  75. DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}"
  76. if [ ! -f "$DUMP" ]; then
  77. echo "::error::Dump not found: $DUMP"
  78. exit 1
  79. fi
  80. docker exec -i torqvoice-db pg_restore --list < "$DUMP" > /dev/null
  81. echo "Dump is readable: $DUMP"
  82. - name: Stop prod app
  83. run: |
  84. cd $HOME/torqvoice-deploy/prod
  85. docker compose stop torqvoice-app
  86. - name: Restore database
  87. if: inputs.restore_database
  88. env:
  89. DATABASE_URL: ${{ secrets.CLOUD_DATABASE_URL }}
  90. run: |
  91. set -euo pipefail
  92. DUMP="${{ secrets.DATA_PATH }}/db-backups/torqvoice/${{ inputs.dump }}"
  93. DB_USER=$(echo "$DATABASE_URL" | sed -E 's|^[^:]+://([^:]+):.*|\1|')
  94. DB_NAME=$(echo "$DATABASE_URL" | sed -E 's|.*/([^/?]+)(\?.*)?$|\1|')
  95. echo "Restoring $DB_NAME from $(basename "$DUMP")"
  96. docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "DROP DATABASE \"$DB_NAME\" WITH (FORCE);"
  97. docker exec -i torqvoice-db psql -U "$DB_USER" -d postgres -c "CREATE DATABASE \"$DB_NAME\" OWNER \"$DB_USER\";"
  98. docker exec -i torqvoice-db pg_restore -U "$DB_USER" -d "$DB_NAME" --no-owner < "$DUMP"
  99. echo "Restore complete."
  100. - name: Deploy previous tag
  101. run: |
  102. cd $HOME/torqvoice-deploy/prod
  103. APP_TAG=${{ inputs.tag }} docker compose up -d --pull always
  104. - name: Health check
  105. run: |
  106. echo "Waiting for prod container to be ready..."
  107. for i in $(seq 1 12); do
  108. STATUS=$(docker inspect --format='{{.State.Status}}' torqvoice-app 2>/dev/null || echo "not found")
  109. if [ "$STATUS" = "running" ]; then
  110. echo "Rolled back to ${{ inputs.tag }}."
  111. exit 0
  112. fi
  113. echo "Status: $STATUS, retrying in 5s..."
  114. sleep 5
  115. done
  116. echo "::error::Container failed to start after rollback"
  117. docker logs torqvoice-app --tail 50
  118. exit 1