diff --git a/.github/actions/discord-notify/action.yml b/.github/actions/discord-notify/action.yml new file mode 100644 index 000000000..e9a214ffc --- /dev/null +++ b/.github/actions/discord-notify/action.yml @@ -0,0 +1,60 @@ +name: Discord deploy notification +description: Post a green (healthy) or red (failed) deploy-status embed to Discord. + +inputs: + webhook-url: + description: Discord webhook URL. When empty, the step is a no-op. + required: true + name: + description: Deploy target name shown in the title (e.g. kaapi-staging). + required: true + release: + description: Release identifier (tag or run number). + required: true + sha: + description: Full commit SHA; truncated to 7 chars for display. + required: true + run-url: + description: Link to the workflow run. + required: true + ok: + description: "'true' for a healthy deploy; anything else renders as failed." + required: true + failure-reason: + description: Optional extra line explaining a failure (e.g. CI blocked the release). + required: false + default: "" + +runs: + using: composite + steps: + - shell: bash + env: + DISCORD_WEBHOOK_URL: ${{ inputs.webhook-url }} + NAME: ${{ inputs.name }} + RELEASE: ${{ inputs.release }} + SHA: ${{ inputs.sha }} + RUN_URL: ${{ inputs.run-url }} + OK: ${{ inputs.ok }} + REASON: ${{ inputs.failure-reason }} + run: | + [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } + if [ "$OK" = "true" ]; then + TITLE="🟢 $NAME deployment healthy"; COLOR=3066993 # green + else + TITLE="🔴 $NAME deployment failed"; COLOR=15158332 # red + fi + SHA_SHORT=$(echo "$SHA" | cut -c1-7) + + fields=$(jq -n --arg release "$RELEASE" --arg sha "$SHA_SHORT" \ + '[{name:"Release",value:$release,inline:true},{name:"SHA",value:$sha,inline:true}]') + # Only a failed deploy with a stated cause gets the extra Reason line. + if [ "$OK" != "true" ] && [ -n "$REASON" ]; then + fields=$(echo "$fields" | jq --arg r "$REASON" '. + [{name:"Reason",value:$r,inline:false}]') + fi + + payload=$(jq -n --arg title "$TITLE" --argjson color "$COLOR" \ + --arg url "$RUN_URL" --argjson fields "$fields" \ + '{embeds:[{title:$title,url:$url,color:$color,fields:$fields,timestamp:(now|todate)}]}') + curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ + || echo "Discord notification failed to send" diff --git a/.github/actions/ecs-deploy/action.yml b/.github/actions/ecs-deploy/action.yml new file mode 100644 index 000000000..170185c48 --- /dev/null +++ b/.github/actions/ecs-deploy/action.yml @@ -0,0 +1,53 @@ +name: Deploy ECS service and verify rollout +description: >- + Force a new deployment on one ECS service (turning on the deployment circuit + breaker with rollback) and wait until its rollout COMPLETED or FAILED. + +inputs: + cluster: + description: ECS cluster name. + required: true + service: + description: ECS service name. + required: true + task-definition: + description: >- + Task-definition family (no revision). When set, rolls the service to the + family's latest active revision; when empty, keeps the service's current + task definition. + required: false + default: "" + poll-interval: + description: Seconds between rollout-state polls. + required: false + default: "15" + +runs: + using: composite + steps: + - shell: bash + env: + CLUSTER: ${{ inputs.cluster }} + SERVICE: ${{ inputs.service }} + FAMILY: ${{ inputs.task-definition }} + POLL_INTERVAL: ${{ inputs.poll-interval }} + run: | + args=(--cluster "$CLUSTER" --service "$SERVICE" --force-new-deployment + --deployment-configuration '{"deploymentCircuitBreaker":{"enable":true,"rollback":true},"maximumPercent":200,"minimumHealthyPercent":100}') + [ -n "$FAMILY" ] && args+=(--task-definition "$FAMILY") + + echo "[$SERVICE] forcing new deployment" + aws ecs update-service "${args[@]}" >/dev/null + + while true; do + STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ + --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ + --output text) + case "$STATE" in + COMPLETED) echo "[$SERVICE] rollout COMPLETED"; break ;; + FAILED) + echo "::error::[$SERVICE] rollout FAILED — new tasks never became healthy (rolled back by circuit breaker)" + exit 1 ;; + *) echo "[$SERVICE] rollout $STATE — waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; + esac + done diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index d094405ad..b8f2ab46d 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -46,9 +46,9 @@ jobs: uses: actions/checkout@v7 - name: Configure AWS credentials - uses: aws-actions/configure-aws-credentials@v6 # More information on this action can be found below in the 'AWS Credentials' section + uses: aws-actions/configure-aws-credentials@v6 with: - role-to-assume: arn:aws:iam::024209611402:role/github-action-role + role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} aws-region: ap-south-1 - name: Login to Amazon ECR @@ -62,6 +62,7 @@ jobs: TAG: ${{ github.ref_name }} run: | docker build \ + --build-arg GIT_SHA=${{ github.sha }} \ -t $REGISTRY/$REPOSITORY:latest \ -t $REGISTRY/$REPOSITORY:$TAG \ ./backend @@ -103,16 +104,35 @@ jobs: fi echo "Migration completed successfully" - - name: Deploy to ECS - run: | - aws ecs update-service \ - --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-cluster \ - --service ${{ vars.AWS_RESOURCE_PREFIX }}-service \ - --task-definition ${{ vars.AWS_RESOURCE_PREFIX }}-task \ - --force-new-deployment - - aws ecs update-service \ - --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-cluster \ - --service ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task \ - --task-definition ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task \ - --force-new-deployment + - name: Deploy backend service + timeout-minutes: 15 + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-service + task-definition: ${{ vars.AWS_RESOURCE_PREFIX }}-task + + - name: Deploy celery service + timeout-minutes: 15 + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task + task-definition: ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task + + notify: + needs: [verify-ci, build] + if: always() + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: ./.github/actions/discord-notify + with: + webhook-url: ${{ secrets.DISCORD_WEBHOOK_URL }} + name: kaapi-production + release: ${{ github.ref_name }} + sha: ${{ github.sha }} + run-url: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + ok: ${{ needs.build.result == 'success' }} + # Call out the specific case where the release was blocked by CI. + failure-reason: ${{ needs.verify-ci.result != 'success' && 'CI never passed on the tagged commit — release blocked.' || '' }} diff --git a/.github/workflows/deploy-staging-ecs.yml b/.github/workflows/deploy-staging-ecs.yml index fffcbb714..dc780e83a 100644 --- a/.github/workflows/deploy-staging-ecs.yml +++ b/.github/workflows/deploy-staging-ecs.yml @@ -20,10 +20,9 @@ jobs: uses: actions/checkout@v7 - name: Configure AWS credentials - # More information on this action can be found below in the 'AWS Credentials' section uses: aws-actions/configure-aws-credentials@v6 with: - role-to-assume: arn:aws:iam::024209611402:role/github-action-role + role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} aws-region: ap-south-1 - name: Login to Amazon ECR @@ -36,7 +35,7 @@ jobs: REGISTRY: ${{ steps.login-ecr.outputs.registry }} REPOSITORY: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-repo run: | - docker build -t $REGISTRY/$REPOSITORY:latest ./backend + docker build --build-arg GIT_SHA=${{ github.sha }} -t $REGISTRY/$REPOSITORY:latest ./backend docker push $REGISTRY/$REPOSITORY:latest - name: Run database migrations @@ -72,6 +71,9 @@ jobs: fi echo "Migration completed successfully" - - name: Deploy to ECS - run: | - aws ecs update-service --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster --service ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service --force-new-deployment + - name: Deploy to ECS and verify rollout + timeout-minutes: 15 + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 00ca90661..c5a60658d 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -36,14 +36,16 @@ jobs: env: INSTANCE_ID: ${{ secrets.STAGING_EC2_INSTANCE_ID }} SECRET_ID: ${{ vars.STAGING_SECRET_ID }} + DEPLOY_SHA: ${{ github.event.workflow_run.head_sha || github.sha }} run: | SECRET_ID=$(echo "$SECRET_ID" | tr -d '[:space:]') + echo "Deploying commit: $DEPLOY_SHA" DEPLOY_CMD="cd /data/kaapi-backend \ && git fetch --all \ - && git pull origin main \ + && git checkout $DEPLOY_SHA \ && SECRET_ID=$SECRET_ID sh scripts/fetch-secrets.sh \ - && docker compose -f docker-compose.staging.yml build \ + && GIT_SHA=$DEPLOY_SHA docker compose -f docker-compose.staging.yml build \ && docker compose -f docker-compose.staging.yml --profile migrate run --rm migrate \ && docker compose -f docker-compose.staging.yml up -d --wait --remove-orphans \ && docker image prune -f" @@ -96,3 +98,18 @@ jobs: --instance-id "$INSTANCE_ID" \ --query '{Status:Status,Stdout:StandardOutputContent,Stderr:StandardErrorContent}' \ --output json + + notify: + needs: [deploy] + if: ${{ always() && needs.deploy.result != 'skipped' }} + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: ./.github/actions/discord-notify + with: + webhook-url: ${{ secrets.DISCORD_WEBHOOK_URL }} + name: kaapi-staging + release: "#${{ github.run_number }}" + sha: ${{ github.event.workflow_run.head_sha || github.sha }} + run-url: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + ok: ${{ needs.deploy.result == 'success' }} diff --git a/backend/Dockerfile b/backend/Dockerfile index f34b22b36..7bc1453d4 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -44,6 +44,9 @@ COPY scripts /app/scripts COPY app /app/app COPY alembic.ini /app/alembic.ini +ARG GIT_SHA=unknown +ENV GIT_SHA=$GIT_SHA + # Expose port 80 EXPOSE 80 diff --git a/backend/app/core/config.py b/backend/app/core/config.py index c7342a567..809f872c6 100644 --- a/backend/app/core/config.py +++ b/backend/app/core/config.py @@ -46,6 +46,7 @@ class Settings(BaseSettings): PROJECT_NAME: str API_VERSION: str = "0.5.0" + GIT_SHA: str = "unknown" SENTRY_DSN: HttpUrl | None = None DISCORD_STATS_WEBHOOK_URL: HttpUrl | None = None POSTGRES_SERVER: str diff --git a/backend/app/main.py b/backend/app/main.py index fd97c03da..0ad7091b7 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -109,4 +109,5 @@ def custom_openapi(): async def health() -> dict[str, str | float]: return { "status": "ok", + "sha": settings.GIT_SHA, } diff --git a/backend/app/tests/core/test_health.py b/backend/app/tests/core/test_health.py new file mode 100644 index 000000000..83dc2c914 --- /dev/null +++ b/backend/app/tests/core/test_health.py @@ -0,0 +1,12 @@ +from fastapi.testclient import TestClient + +from app.core.config import settings + + +def test_health_reports_status_and_sha(client: TestClient) -> None: + response = client.get("/health") + + assert response.status_code == 200 + body = response.json() + assert body["status"] == "ok" + assert body["sha"] == settings.GIT_SHA diff --git a/docker-compose.staging.yml b/docker-compose.staging.yml index e7e8c149b..aa43a6f13 100644 --- a/docker-compose.staging.yml +++ b/docker-compose.staging.yml @@ -7,6 +7,8 @@ services: restart: always build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} env_file: - path: .env required: true @@ -44,6 +46,8 @@ services: image: "${DOCKER_IMAGE_BACKEND?Variable not set}:${TAG:-latest}" build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} env_file: - path: .env required: true @@ -58,6 +62,8 @@ services: restart: always build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} depends_on: backend: condition: service_healthy