From 64839a42637f479641a108f9078f0795f3fedaed Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Tue, 8 Sep 2026 22:24:05 +0530 Subject: [PATCH 1/7] feat(*): continuos deployment script updates --- .github/workflows/create-release.yml | 95 +++++++++++++++++--- .github/workflows/deploy-staging-ecs.yml | 2 +- .github/workflows/deploy-staging.yml | 105 +++++++++++++++++++++++ backend/Dockerfile | 3 + backend/app/core/config.py | 1 + backend/app/main.py | 1 + 6 files changed, 194 insertions(+), 13 deletions(-) diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index d094405ad..8df8afa82 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -62,6 +62,7 @@ jobs: TAG: ${{ github.ref_name }} run: | docker build \ + --build-arg GIT_SHA=${{ github.sha }} \ -t $REGISTRY/$REPOSITORY:latest \ -t $REGISTRY/$REPOSITORY:$TAG \ ./backend @@ -103,16 +104,86 @@ jobs: fi echo "Migration completed successfully" - - name: Deploy to ECS + - name: Deploy to ECS and verify rollout + # Bound the wait; the circuit breaker itself trips well before this. + timeout-minutes: 15 + env: + CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster + POLL_INTERVAL: "15" run: | - aws ecs update-service \ - --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-cluster \ - --service ${{ vars.AWS_RESOURCE_PREFIX }}-service \ - --task-definition ${{ vars.AWS_RESOURCE_PREFIX }}-task \ - --force-new-deployment - - aws ecs update-service \ - --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-cluster \ - --service ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task \ - --task-definition ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task \ - --force-new-deployment + # A plain update-service returns before the new tasks are healthy, and + # the old task can keep answering 200 while the new one crash-loops. + # So roll each service to its family's latest revision, then poll the + # PRIMARY deployment's rolloutState. With the deployment circuit + # breaker enabled on the service (one-time prep), a bad rollout flips + # to FAILED and auto-rolls-back โ€” which we surface here as a failure. + deploy_and_wait() { + SERVICE="$1" + FAMILY="$2" + echo "[$SERVICE] forcing new deployment on family $FAMILY" + aws ecs update-service \ + --cluster "$CLUSTER" \ + --service "$SERVICE" \ + --task-definition "$FAMILY" \ + --force-new-deployment >/dev/null + + while true; do + STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ + --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ + --output text) + case "$STATE" in + COMPLETED) + echo "[$SERVICE] rollout COMPLETED" + return 0 ;; + FAILED) + echo "::error::[$SERVICE] rollout FAILED โ€” new tasks never became healthy (rolled back by circuit breaker)" + return 1 ;; + *) + echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s" + sleep "$POLL_INTERVAL" ;; + esac + done + } + + deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-service" "${{ vars.AWS_RESOURCE_PREFIX }}-task" + deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" + + # Green "healthy" on a clean release, red "failed" otherwise (including a + # release aborted because CI never passed on the tagged commit). + notify: + needs: [verify-ci, build] + if: always() + runs-on: ubuntu-latest + steps: + - name: Notify Discord + env: + DISCORD_WEBHOOK_URL: ${{ secrets.DISCORD_WEBHOOK_URL }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + NAME: kaapi-production + RELEASE: ${{ github.ref_name }} + # true only when the build+deploy job succeeded. + OK: ${{ needs.build.result == 'success' }} + run: | + [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } + if [ "$OK" = "true" ]; then + TITLE="๐ŸŸข $NAME deployment healthy"; COLOR=3066993 # green + else + TITLE="๐Ÿ”ด $NAME deployment failed"; COLOR=15158332 # red + fi + SHA_SHORT=$(echo "${{ github.sha }}" | cut -c1-7) + payload=$(jq -n \ + --arg title "$TITLE" \ + --argjson color "$COLOR" \ + --arg release "$RELEASE" \ + --arg sha "$SHA_SHORT" \ + --arg url "$RUN_URL" \ + '{embeds: [{ + title: $title, url: $url, color: $color, + fields: [ + {name: "Release", value: $release, inline: true}, + {name: "SHA", value: $sha, inline: true} + ], + timestamp: (now | todate) + }]}') + curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ + || echo "Discord notification failed to send" diff --git a/.github/workflows/deploy-staging-ecs.yml b/.github/workflows/deploy-staging-ecs.yml index fffcbb714..3ebb43bb5 100644 --- a/.github/workflows/deploy-staging-ecs.yml +++ b/.github/workflows/deploy-staging-ecs.yml @@ -36,7 +36,7 @@ jobs: REGISTRY: ${{ steps.login-ecr.outputs.registry }} REPOSITORY: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-repo run: | - docker build -t $REGISTRY/$REPOSITORY:latest ./backend + docker build --build-arg GIT_SHA=${{ github.sha }} -t $REGISTRY/$REPOSITORY:latest ./backend docker push $REGISTRY/$REPOSITORY:latest - name: Run database migrations diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 39c788eef..ef5acaea7 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -85,3 +85,108 @@ jobs: --instance-id "$INSTANCE_ID" \ --query '{Status:Status,Stdout:StandardOutputContent,Stderr:StandardErrorContent}' \ --output json + + ecs-rehearsal: + needs: deploy + runs-on: ubuntu-latest + environment: AWS_ENV_VARS + permissions: + id-token: write + contents: read + steps: + - name: Checkout the repo + uses: actions/checkout@v7 + + - name: Configure AWS credentials + uses: aws-actions/configure-aws-credentials@v6 + with: + role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} + aws-region: ap-south-1 + + - name: Login to Amazon ECR + id: login-ecr + uses: aws-actions/amazon-ecr-login@v2 + + - name: Build and push staging image + env: + REGISTRY: ${{ steps.login-ecr.outputs.registry }} + REPOSITORY: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-repo + run: | + docker build \ + --build-arg GIT_SHA=${{ github.sha }} \ + -t $REGISTRY/$REPOSITORY:latest \ + ./backend + docker push $REGISTRY/$REPOSITORY:latest + + - name: Scale staging ECS up and verify rollout + timeout-minutes: 15 + env: + CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster + SERVICE: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service + POLL_INTERVAL: "15" + run: | + aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ + --desired-count 1 --force-new-deployment >/dev/null + echo "[$SERVICE] scaled to 1; waiting for rollout" + while true; do + STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ + --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ + --output text) + case "$STATE" in + COMPLETED) echo "[$SERVICE] rehearsal rollout COMPLETED"; break ;; + FAILED) + echo "::error::[$SERVICE] rehearsal FAILED โ€” the production ECS deploy path is broken" + exit 1 ;; + *) echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; + esac + done + + - name: Scale staging ECS back to 0 + if: always() + env: + CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster + SERVICE: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service + run: | + aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ + --desired-count 0 >/dev/null + echo "[$SERVICE] scaled back to 0" + + # Green "healthy" on a clean deploy + rehearsal, red "failed" otherwise. + # Skipped (not sent) when the deploy itself was skipped on a red CI. + notify: + needs: [deploy, ecs-rehearsal] + if: ${{ always() && needs.deploy.result != 'skipped' }} + runs-on: ubuntu-latest + steps: + - name: Notify Discord + env: + DISCORD_WEBHOOK_URL: ${{ secrets.DISCORD_WEBHOOK_URL }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + NAME: kaapi-staging + RELEASE: "#${{ github.run_number }}" + # true only when every deploy job succeeded. + OK: ${{ needs.deploy.result == 'success' && needs.ecs-rehearsal.result == 'success' }} + run: | + [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } + if [ "$OK" = "true" ]; then + TITLE="๐ŸŸข $NAME deployment healthy"; COLOR=3066993 # green + else + TITLE="๐Ÿ”ด $NAME deployment failed"; COLOR=15158332 # red + fi + SHA_SHORT=$(echo "${{ github.sha }}" | cut -c1-7) + payload=$(jq -n \ + --arg title "$TITLE" \ + --argjson color "$COLOR" \ + --arg release "$RELEASE" \ + --arg sha "$SHA_SHORT" \ + --arg url "$RUN_URL" \ + '{embeds: [{ + title: $title, url: $url, color: $color, + fields: [ + {name: "Release", value: $release, inline: true}, + {name: "SHA", value: $sha, inline: true} + ], + timestamp: (now | todate) + }]}') + curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ + || echo "Discord notification failed to send" diff --git a/backend/Dockerfile b/backend/Dockerfile index f34b22b36..7bc1453d4 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -44,6 +44,9 @@ COPY scripts /app/scripts COPY app /app/app COPY alembic.ini /app/alembic.ini +ARG GIT_SHA=unknown +ENV GIT_SHA=$GIT_SHA + # Expose port 80 EXPOSE 80 diff --git a/backend/app/core/config.py b/backend/app/core/config.py index cc105fbec..cb1f3badf 100644 --- a/backend/app/core/config.py +++ b/backend/app/core/config.py @@ -46,6 +46,7 @@ class Settings(BaseSettings): PROJECT_NAME: str API_VERSION: str = "0.5.0" + GIT_SHA: str = "unknown" SENTRY_DSN: HttpUrl | None = None DISCORD_STATS_WEBHOOK_URL: HttpUrl | None = None POSTGRES_SERVER: str diff --git a/backend/app/main.py b/backend/app/main.py index 5d2a09cf0..0c1c3f66b 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -105,4 +105,5 @@ def custom_openapi(): async def health() -> dict[str, str | float]: return { "status": "ok", + "sha": settings.GIT_SHA, } From 60185fe6519ab4f35608a06f2f5660d2aef925ab Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Tue, 8 Sep 2026 23:38:55 +0530 Subject: [PATCH 2/7] fix(*): get the role id from the github secret --- .github/workflows/create-release.yml | 12 +- .github/workflows/deploy-staging-ecs.yml | 2 +- .../glific-evals-configs-api-coverage.md | 107 ++++++++++++++++++ 3 files changed, 110 insertions(+), 11 deletions(-) create mode 100644 docs/guides/glific-evals-configs-api-coverage.md diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 8df8afa82..23ddb4495 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -46,9 +46,9 @@ jobs: uses: actions/checkout@v7 - name: Configure AWS credentials - uses: aws-actions/configure-aws-credentials@v6 # More information on this action can be found below in the 'AWS Credentials' section + uses: aws-actions/configure-aws-credentials@v6 with: - role-to-assume: arn:aws:iam::024209611402:role/github-action-role + role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} aws-region: ap-south-1 - name: Login to Amazon ECR @@ -111,12 +111,6 @@ jobs: CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster POLL_INTERVAL: "15" run: | - # A plain update-service returns before the new tasks are healthy, and - # the old task can keep answering 200 while the new one crash-loops. - # So roll each service to its family's latest revision, then poll the - # PRIMARY deployment's rolloutState. With the deployment circuit - # breaker enabled on the service (one-time prep), a bad rollout flips - # to FAILED and auto-rolls-back โ€” which we surface here as a failure. deploy_and_wait() { SERVICE="$1" FAMILY="$2" @@ -148,8 +142,6 @@ jobs: deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-service" "${{ vars.AWS_RESOURCE_PREFIX }}-task" deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" - # Green "healthy" on a clean release, red "failed" otherwise (including a - # release aborted because CI never passed on the tagged commit). notify: needs: [verify-ci, build] if: always() diff --git a/.github/workflows/deploy-staging-ecs.yml b/.github/workflows/deploy-staging-ecs.yml index 3ebb43bb5..6aacd23af 100644 --- a/.github/workflows/deploy-staging-ecs.yml +++ b/.github/workflows/deploy-staging-ecs.yml @@ -23,7 +23,7 @@ jobs: # More information on this action can be found below in the 'AWS Credentials' section uses: aws-actions/configure-aws-credentials@v6 with: - role-to-assume: arn:aws:iam::024209611402:role/github-action-role + role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} aws-region: ap-south-1 - name: Login to Amazon ECR diff --git a/docs/guides/glific-evals-configs-api-coverage.md b/docs/guides/glific-evals-configs-api-coverage.md new file mode 100644 index 000000000..89ffcaa3b --- /dev/null +++ b/docs/guides/glific-evals-configs-api-coverage.md @@ -0,0 +1,107 @@ +# Glific Evals & Configs UI โ†’ Kaapi API Coverage + +This maps the **Evals** and **Configs** flows in the Glific `AI Assistants` +prototype (`glific-evals-v13.html`) to the corresponding Kaapi backend APIs, and +marks whether each API already exists. + +**How the prototype's concepts map to Kaapi:** + +| Prototype concept | Kaapi concept | +| --- | --- | +| An "Assistant" | A **Config** (`/api/v1/configs`) | +| A saved "Version" (prompt + model + settings) | A **Config version** (`/configs/{id}/versions`) | +| A "Golden Q&A set" | An **evaluation dataset** (`/evaluations/datasets`) | +| A "Run" / evaluation | An **evaluation run** (`/api/v2/evaluations`) | + +--- + +## Configs flow + +| API | Available | +| --- | --- | +| Get Configs (list assistants) | **Yes** | +| Get Config (open an assistant) | **Yes** | +| Create Config (create assistant) | **Yes** | +| Update Config (rename / edit) | **Yes** | +| Delete Config (delete assistant) | **Yes** | +| Create Config Version (Save Version) | **Yes** | +| Get Config Versions (version dropdown) | **Yes** | +| Get Config Version (load a version) | **Yes** | +| Duplicate Config (Duplicate assistant) | **No** | +| Publish / Set-Live Config Version (Publish & go live) | **No** | + +### Endpoints & notes + +| API | Available | Kaapi endpoint / notes | +| --- | --- | --- | +| Get Configs | โœ… Yes | `GET /api/v1/configs` | +| Get Config | โœ… Yes | `GET /api/v1/configs/{config_id}` | +| Create Config | โœ… Yes | `POST /api/v1/configs` โ€” also creates version 1 in the same call | +| Update Config | โœ… Yes | `PATCH /api/v1/configs/{config_id}` | +| Delete Config | โœ… Yes | `DELETE /api/v1/configs/{config_id}` | +| Create Config Version | โœ… Yes | `POST /api/v1/configs/{config_id}/versions` โ€” matches "Save Version" (each save = a new version) | +| Get Config Versions | โœ… Yes | `GET /api/v1/configs/{config_id}/versions` | +| Get Config Version | โœ… Yes | `GET /api/v1/configs/{config_id}/versions/{version_number}` | +| Duplicate Config | โŒ No | No clone/copy endpoint. The UI "Duplicate" would need a new API (or client-side create-config from a fetched blob). | +| Publish / Set-Live Config Version | โŒ No | **No concept of a live/published/active version in Kaapi.** The `ConfigVersion` model has no `is_live`/`published`/`active` field, and there is no promote/go-live endpoint. The prototype's "Publish & go live", the LIVE badge, and "was live" states have no backing API. | + +--- + +## Evals flow + +| API | Available | +| --- | --- | +| Upload Dataset (add Golden Q&A set) | **Yes** | +| List Datasets (Manage sets) | **Yes** | +| Get Dataset (view a set) | **Yes** (partial) | +| Delete Dataset (delete a set) | **Yes** | +| Export Dataset (export set CSV) | **No** | +| Run Eval (Run evaluation) | **Yes** | +| Get Eval (in-progress / completed status) | **Yes** | +| Get Eval Results (metrics + per-question) | **Yes** | +| List Evals (History) | **Yes** | +| Export Eval Results (Export CSV) | **No** | +| Improve Prompt (What to change next) | **Yes** | +| Run-time / online evaluation (live conversation scoring) | **No** | + +### Endpoints & notes + +| API | Available | Kaapi endpoint / notes | +| --- | --- | --- | +| Upload Dataset | โœ… Yes | `POST /api/v2/evaluations/datasets` (v1 also exists). CSV columns `question`, `answer`, optional `category` โ€” matches the prototype's CSV. | +| List Datasets | โœ… Yes | `GET /api/v1/evaluations/datasets` (v1 only; no v2 variant) | +| Get Dataset | โš ๏ธ Partial | `GET /api/v1/evaluations/datasets/{dataset_id}` returns the dataset record, but there is **no per-item/questions listing route**. The prototype's "View set" question table would read rows from the stored CSV (`signed_url`), not a questions API. | +| Delete Dataset | โœ… Yes | `DELETE /api/v1/evaluations/datasets/{dataset_id}` | +| Export Dataset | โŒ No | The prototype exports the set as CSV client-side; there's no API for it (the source CSV is already retrievable via the dataset's `signed_url`). | +| Run Eval | โœ… Yes | `POST /api/v2/evaluations` โ€” body `dataset_id`, `experiment_name`, `config_id`, `config_version`. **Mismatch:** the prototype's per-run duplication (1ร— / 5ร—) does not map to the run endpoint โ€” in Kaapi `duplication_factor` is set at **dataset upload** time (`1โ€“5`), not per run. | +| Get Eval (status) | โœ… Yes | `GET /api/v1/evaluations/{evaluation_id}` โ€” `status` goes `processing โ†’ completed`/`failed`, backing the "in progress" / "completed" job banner. | +| Get Eval Results | โœ… Yes | Same `GET /api/v1/evaluations/{evaluation_id}` โ€” run-level `score` + per-row judge scores/reasoning in the `score_trace_url` trace. Covers the overall gauge + question-level table. | +| List Evals (History) | โœ… Yes | `GET /api/v1/evaluations` (`limit`/`offset`). The prototype's version/set filters and sorting would be applied client-side. | +| Export Eval Results (CSV) | โŒ No | No CSV/file export route. `GET /api/v1/evaluations/{id}` has an `export_format` param but it only accepts `row`/`grouped` and just restructures the JSON โ€” it does not produce a CSV. The prototype's "โค“ Export CSV" has no API. | +| Improve Prompt | โœ… Yes | `POST /api/v2/evaluations/{evaluation_id}/improve-prompt` (v1 also exists). Backs the "What to change next" / suggested-prompt-change panel. Async โ€” delivers to an HTTPS `callback_url`. | +| Run-time / online evaluation | โŒ No | **No API.** The entire "Run-time Evaluations" tab (continuous scoring of real conversations, rolling trend, flagged log) has no backing endpoint โ€” Kaapi only does the on-demand Golden Q&A run. | + +--- + +## Adjacent UI surfaces (outside Evals & Configs) + +Called out for completeness โ€” these prototype tabs aren't part of the Evals/Configs +flow but affect a full build: + +| API | Available | Notes | +| --- | --- | --- | +| Try It Out (run one prompt against a saved config) | โŒ No | No single-prompt sandbox route takes a `config_id`. The only config-by-reference execution is the full STS chain `POST /api/v1/llm/chain/sts`, not a text playground. `POST /responses` exists but doesn't reference a config. | +| Knowledge Base (list / add / remove files, vector store) | โ€” | KB management lives outside the config/evaluations routes; not evaluated here. The config blob only references `knowledge_base_ids`. | + +--- + +## Summary of gaps + +Everything the Evals & Configs flow needs **exists today except**: + +1. **Publish / go-live for a config version** โ€” no live/published/active concept in Kaapi at all (the biggest gap; the prototype's whole version-lifecycle UI depends on it). +2. **Duplicate config** โ€” no clone endpoint. +3. **Run-time / online evaluation** โ€” no live-traffic scoring; only on-demand runs. +4. **CSV export** of eval results and of a dataset โ€” no export API. +5. **Per-run duplication factor** โ€” Kaapi sets it at dataset-upload time, not per run. +6. **Dataset questions listing** โ€” `GET dataset` returns the record, not an items API (rows come from the stored CSV). From 3c7a560d6daa8641fb1210466b0ace2cbb9f43d1 Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Wed, 9 Sep 2026 00:02:17 +0530 Subject: [PATCH 3/7] fix(*): get the role id from the github secret --- .github/workflows/deploy-staging-ecs.yml | 1 - .github/workflows/deploy-staging.yml | 2 - .../glific-evals-configs-api-coverage.md | 107 ------------------ 3 files changed, 110 deletions(-) delete mode 100644 docs/guides/glific-evals-configs-api-coverage.md diff --git a/.github/workflows/deploy-staging-ecs.yml b/.github/workflows/deploy-staging-ecs.yml index 6aacd23af..12dbeb76d 100644 --- a/.github/workflows/deploy-staging-ecs.yml +++ b/.github/workflows/deploy-staging-ecs.yml @@ -20,7 +20,6 @@ jobs: uses: actions/checkout@v7 - name: Configure AWS credentials - # More information on this action can be found below in the 'AWS Credentials' section uses: aws-actions/configure-aws-credentials@v6 with: role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index ef5acaea7..462a459ab 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -151,8 +151,6 @@ jobs: --desired-count 0 >/dev/null echo "[$SERVICE] scaled back to 0" - # Green "healthy" on a clean deploy + rehearsal, red "failed" otherwise. - # Skipped (not sent) when the deploy itself was skipped on a red CI. notify: needs: [deploy, ecs-rehearsal] if: ${{ always() && needs.deploy.result != 'skipped' }} diff --git a/docs/guides/glific-evals-configs-api-coverage.md b/docs/guides/glific-evals-configs-api-coverage.md deleted file mode 100644 index 89ffcaa3b..000000000 --- a/docs/guides/glific-evals-configs-api-coverage.md +++ /dev/null @@ -1,107 +0,0 @@ -# Glific Evals & Configs UI โ†’ Kaapi API Coverage - -This maps the **Evals** and **Configs** flows in the Glific `AI Assistants` -prototype (`glific-evals-v13.html`) to the corresponding Kaapi backend APIs, and -marks whether each API already exists. - -**How the prototype's concepts map to Kaapi:** - -| Prototype concept | Kaapi concept | -| --- | --- | -| An "Assistant" | A **Config** (`/api/v1/configs`) | -| A saved "Version" (prompt + model + settings) | A **Config version** (`/configs/{id}/versions`) | -| A "Golden Q&A set" | An **evaluation dataset** (`/evaluations/datasets`) | -| A "Run" / evaluation | An **evaluation run** (`/api/v2/evaluations`) | - ---- - -## Configs flow - -| API | Available | -| --- | --- | -| Get Configs (list assistants) | **Yes** | -| Get Config (open an assistant) | **Yes** | -| Create Config (create assistant) | **Yes** | -| Update Config (rename / edit) | **Yes** | -| Delete Config (delete assistant) | **Yes** | -| Create Config Version (Save Version) | **Yes** | -| Get Config Versions (version dropdown) | **Yes** | -| Get Config Version (load a version) | **Yes** | -| Duplicate Config (Duplicate assistant) | **No** | -| Publish / Set-Live Config Version (Publish & go live) | **No** | - -### Endpoints & notes - -| API | Available | Kaapi endpoint / notes | -| --- | --- | --- | -| Get Configs | โœ… Yes | `GET /api/v1/configs` | -| Get Config | โœ… Yes | `GET /api/v1/configs/{config_id}` | -| Create Config | โœ… Yes | `POST /api/v1/configs` โ€” also creates version 1 in the same call | -| Update Config | โœ… Yes | `PATCH /api/v1/configs/{config_id}` | -| Delete Config | โœ… Yes | `DELETE /api/v1/configs/{config_id}` | -| Create Config Version | โœ… Yes | `POST /api/v1/configs/{config_id}/versions` โ€” matches "Save Version" (each save = a new version) | -| Get Config Versions | โœ… Yes | `GET /api/v1/configs/{config_id}/versions` | -| Get Config Version | โœ… Yes | `GET /api/v1/configs/{config_id}/versions/{version_number}` | -| Duplicate Config | โŒ No | No clone/copy endpoint. The UI "Duplicate" would need a new API (or client-side create-config from a fetched blob). | -| Publish / Set-Live Config Version | โŒ No | **No concept of a live/published/active version in Kaapi.** The `ConfigVersion` model has no `is_live`/`published`/`active` field, and there is no promote/go-live endpoint. The prototype's "Publish & go live", the LIVE badge, and "was live" states have no backing API. | - ---- - -## Evals flow - -| API | Available | -| --- | --- | -| Upload Dataset (add Golden Q&A set) | **Yes** | -| List Datasets (Manage sets) | **Yes** | -| Get Dataset (view a set) | **Yes** (partial) | -| Delete Dataset (delete a set) | **Yes** | -| Export Dataset (export set CSV) | **No** | -| Run Eval (Run evaluation) | **Yes** | -| Get Eval (in-progress / completed status) | **Yes** | -| Get Eval Results (metrics + per-question) | **Yes** | -| List Evals (History) | **Yes** | -| Export Eval Results (Export CSV) | **No** | -| Improve Prompt (What to change next) | **Yes** | -| Run-time / online evaluation (live conversation scoring) | **No** | - -### Endpoints & notes - -| API | Available | Kaapi endpoint / notes | -| --- | --- | --- | -| Upload Dataset | โœ… Yes | `POST /api/v2/evaluations/datasets` (v1 also exists). CSV columns `question`, `answer`, optional `category` โ€” matches the prototype's CSV. | -| List Datasets | โœ… Yes | `GET /api/v1/evaluations/datasets` (v1 only; no v2 variant) | -| Get Dataset | โš ๏ธ Partial | `GET /api/v1/evaluations/datasets/{dataset_id}` returns the dataset record, but there is **no per-item/questions listing route**. The prototype's "View set" question table would read rows from the stored CSV (`signed_url`), not a questions API. | -| Delete Dataset | โœ… Yes | `DELETE /api/v1/evaluations/datasets/{dataset_id}` | -| Export Dataset | โŒ No | The prototype exports the set as CSV client-side; there's no API for it (the source CSV is already retrievable via the dataset's `signed_url`). | -| Run Eval | โœ… Yes | `POST /api/v2/evaluations` โ€” body `dataset_id`, `experiment_name`, `config_id`, `config_version`. **Mismatch:** the prototype's per-run duplication (1ร— / 5ร—) does not map to the run endpoint โ€” in Kaapi `duplication_factor` is set at **dataset upload** time (`1โ€“5`), not per run. | -| Get Eval (status) | โœ… Yes | `GET /api/v1/evaluations/{evaluation_id}` โ€” `status` goes `processing โ†’ completed`/`failed`, backing the "in progress" / "completed" job banner. | -| Get Eval Results | โœ… Yes | Same `GET /api/v1/evaluations/{evaluation_id}` โ€” run-level `score` + per-row judge scores/reasoning in the `score_trace_url` trace. Covers the overall gauge + question-level table. | -| List Evals (History) | โœ… Yes | `GET /api/v1/evaluations` (`limit`/`offset`). The prototype's version/set filters and sorting would be applied client-side. | -| Export Eval Results (CSV) | โŒ No | No CSV/file export route. `GET /api/v1/evaluations/{id}` has an `export_format` param but it only accepts `row`/`grouped` and just restructures the JSON โ€” it does not produce a CSV. The prototype's "โค“ Export CSV" has no API. | -| Improve Prompt | โœ… Yes | `POST /api/v2/evaluations/{evaluation_id}/improve-prompt` (v1 also exists). Backs the "What to change next" / suggested-prompt-change panel. Async โ€” delivers to an HTTPS `callback_url`. | -| Run-time / online evaluation | โŒ No | **No API.** The entire "Run-time Evaluations" tab (continuous scoring of real conversations, rolling trend, flagged log) has no backing endpoint โ€” Kaapi only does the on-demand Golden Q&A run. | - ---- - -## Adjacent UI surfaces (outside Evals & Configs) - -Called out for completeness โ€” these prototype tabs aren't part of the Evals/Configs -flow but affect a full build: - -| API | Available | Notes | -| --- | --- | --- | -| Try It Out (run one prompt against a saved config) | โŒ No | No single-prompt sandbox route takes a `config_id`. The only config-by-reference execution is the full STS chain `POST /api/v1/llm/chain/sts`, not a text playground. `POST /responses` exists but doesn't reference a config. | -| Knowledge Base (list / add / remove files, vector store) | โ€” | KB management lives outside the config/evaluations routes; not evaluated here. The config blob only references `knowledge_base_ids`. | - ---- - -## Summary of gaps - -Everything the Evals & Configs flow needs **exists today except**: - -1. **Publish / go-live for a config version** โ€” no live/published/active concept in Kaapi at all (the biggest gap; the prototype's whole version-lifecycle UI depends on it). -2. **Duplicate config** โ€” no clone endpoint. -3. **Run-time / online evaluation** โ€” no live-traffic scoring; only on-demand runs. -4. **CSV export** of eval results and of a dataset โ€” no export API. -5. **Per-run duplication factor** โ€” Kaapi sets it at dataset-upload time, not per run. -6. **Dataset questions listing** โ€” `GET dataset` returns the record, not an items API (rows come from the stored CSV). From c4f492d22036c9352f8a232a67269b5dcc8d1d0f Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Thu, 10 Sep 2026 12:42:09 +0530 Subject: [PATCH 4/7] fix(*): few updates on the deployment staging script --- .github/workflows/deploy-staging.yml | 50 ++++++++++++++++------------ 1 file changed, 29 insertions(+), 21 deletions(-) diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 462a459ab..7d7cafd2b 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -121,35 +121,43 @@ jobs: - name: Scale staging ECS up and verify rollout timeout-minutes: 15 env: - CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster - SERVICE: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service + CLUSTER: kaapi-staging + SERVICES: "kaapi-staging-backend-celery kaapi-staging-celery-worker" POLL_INTERVAL: "15" run: | - aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ - --desired-count 1 --force-new-deployment >/dev/null - echo "[$SERVICE] scaled to 1; waiting for rollout" - while true; do - STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ - --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ - --output text) - case "$STATE" in - COMPLETED) echo "[$SERVICE] rehearsal rollout COMPLETED"; break ;; - FAILED) - echo "::error::[$SERVICE] rehearsal FAILED โ€” the production ECS deploy path is broken" - exit 1 ;; - *) echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; - esac + for SERVICE in $SERVICES; do + aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ + --desired-count 1 --force-new-deployment >/dev/null + echo "[$SERVICE] scaled to 1" + done + + for SERVICE in $SERVICES; do + echo "[$SERVICE] waiting for rollout" + while true; do + STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ + --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ + --output text) + case "$STATE" in + COMPLETED) echo "[$SERVICE] rehearsal rollout COMPLETED"; break ;; + FAILED) + echo "::error::[$SERVICE] rehearsal FAILED โ€” the production ECS deploy path is broken" + exit 1 ;; + *) echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; + esac + done done - name: Scale staging ECS back to 0 if: always() env: - CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster - SERVICE: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service + CLUSTER: kaapi-staging + SERVICES: "kaapi-staging-backend-celery kaapi-staging-celery-worker" run: | - aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ - --desired-count 0 >/dev/null - echo "[$SERVICE] scaled back to 0" + for SERVICE in $SERVICES; do + aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ + --desired-count 0 >/dev/null + echo "[$SERVICE] scaled back to 0" + done notify: needs: [deploy, ecs-rehearsal] From 188629a1ec732c6c1da03afd4762fcee81225fb8 Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:38:55 +0530 Subject: [PATCH 5/7] fix(*): made the some changes as per reviewer --- .github/actions/discord-notify/action.yml | 60 +++++++++++ .github/actions/ecs-deploy/action.yml | 55 ++++++++++ .github/workflows/create-release.yml | 93 +++++------------ .github/workflows/deploy-staging-ecs.yml | 9 +- .github/workflows/deploy-staging.yml | 122 +++------------------- backend/app/tests/core/test_health.py | 12 +++ docker-compose.staging.yml | 2 + 7 files changed, 174 insertions(+), 179 deletions(-) create mode 100644 .github/actions/discord-notify/action.yml create mode 100644 .github/actions/ecs-deploy/action.yml create mode 100644 backend/app/tests/core/test_health.py diff --git a/.github/actions/discord-notify/action.yml b/.github/actions/discord-notify/action.yml new file mode 100644 index 000000000..e9a214ffc --- /dev/null +++ b/.github/actions/discord-notify/action.yml @@ -0,0 +1,60 @@ +name: Discord deploy notification +description: Post a green (healthy) or red (failed) deploy-status embed to Discord. + +inputs: + webhook-url: + description: Discord webhook URL. When empty, the step is a no-op. + required: true + name: + description: Deploy target name shown in the title (e.g. kaapi-staging). + required: true + release: + description: Release identifier (tag or run number). + required: true + sha: + description: Full commit SHA; truncated to 7 chars for display. + required: true + run-url: + description: Link to the workflow run. + required: true + ok: + description: "'true' for a healthy deploy; anything else renders as failed." + required: true + failure-reason: + description: Optional extra line explaining a failure (e.g. CI blocked the release). + required: false + default: "" + +runs: + using: composite + steps: + - shell: bash + env: + DISCORD_WEBHOOK_URL: ${{ inputs.webhook-url }} + NAME: ${{ inputs.name }} + RELEASE: ${{ inputs.release }} + SHA: ${{ inputs.sha }} + RUN_URL: ${{ inputs.run-url }} + OK: ${{ inputs.ok }} + REASON: ${{ inputs.failure-reason }} + run: | + [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } + if [ "$OK" = "true" ]; then + TITLE="๐ŸŸข $NAME deployment healthy"; COLOR=3066993 # green + else + TITLE="๐Ÿ”ด $NAME deployment failed"; COLOR=15158332 # red + fi + SHA_SHORT=$(echo "$SHA" | cut -c1-7) + + fields=$(jq -n --arg release "$RELEASE" --arg sha "$SHA_SHORT" \ + '[{name:"Release",value:$release,inline:true},{name:"SHA",value:$sha,inline:true}]') + # Only a failed deploy with a stated cause gets the extra Reason line. + if [ "$OK" != "true" ] && [ -n "$REASON" ]; then + fields=$(echo "$fields" | jq --arg r "$REASON" '. + [{name:"Reason",value:$r,inline:false}]') + fi + + payload=$(jq -n --arg title "$TITLE" --argjson color "$COLOR" \ + --arg url "$RUN_URL" --argjson fields "$fields" \ + '{embeds:[{title:$title,url:$url,color:$color,fields:$fields,timestamp:(now|todate)}]}') + curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ + || echo "Discord notification failed to send" diff --git a/.github/actions/ecs-deploy/action.yml b/.github/actions/ecs-deploy/action.yml new file mode 100644 index 000000000..e9d83f844 --- /dev/null +++ b/.github/actions/ecs-deploy/action.yml @@ -0,0 +1,55 @@ +name: Deploy ECS service and verify rollout +description: >- + Force a new deployment on one ECS service (turning on the deployment circuit + breaker with rollback) and wait until its rollout COMPLETED or FAILED. + +inputs: + cluster: + description: ECS cluster name. + required: true + service: + description: ECS service name. + required: true + task-definition: + description: >- + Task-definition family (no revision). When set, rolls the service to the + family's latest active revision; when empty, keeps the service's current + task definition. + required: false + default: "" + poll-interval: + description: Seconds between rollout-state polls. + required: false + default: "15" + +runs: + using: composite + steps: + - shell: bash + env: + CLUSTER: ${{ inputs.cluster }} + SERVICE: ${{ inputs.service }} + FAMILY: ${{ inputs.task-definition }} + POLL_INTERVAL: ${{ inputs.poll-interval }} + run: | + # The circuit breaker is what turns a crash-looping rollout into a + # FAILED state (with rollback) instead of one that hangs until timeout. + args=(--cluster "$CLUSTER" --service "$SERVICE" --force-new-deployment + --deployment-configuration '{"deploymentCircuitBreaker":{"enable":true,"rollback":true},"maximumPercent":200,"minimumHealthyPercent":100}') + [ -n "$FAMILY" ] && args+=(--task-definition "$FAMILY") + + echo "[$SERVICE] forcing new deployment" + aws ecs update-service "${args[@]}" >/dev/null + + while true; do + STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ + --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ + --output text) + case "$STATE" in + COMPLETED) echo "[$SERVICE] rollout COMPLETED"; break ;; + FAILED) + echo "::error::[$SERVICE] rollout FAILED โ€” new tasks never became healthy (rolled back by circuit breaker)" + exit 1 ;; + *) echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; + esac + done diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 23ddb4495..b8f2ab46d 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -104,78 +104,35 @@ jobs: fi echo "Migration completed successfully" - - name: Deploy to ECS and verify rollout - # Bound the wait; the circuit breaker itself trips well before this. + - name: Deploy backend service timeout-minutes: 15 - env: - CLUSTER: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster - POLL_INTERVAL: "15" - run: | - deploy_and_wait() { - SERVICE="$1" - FAMILY="$2" - echo "[$SERVICE] forcing new deployment on family $FAMILY" - aws ecs update-service \ - --cluster "$CLUSTER" \ - --service "$SERVICE" \ - --task-definition "$FAMILY" \ - --force-new-deployment >/dev/null - - while true; do - STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ - --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ - --output text) - case "$STATE" in - COMPLETED) - echo "[$SERVICE] rollout COMPLETED" - return 0 ;; - FAILED) - echo "::error::[$SERVICE] rollout FAILED โ€” new tasks never became healthy (rolled back by circuit breaker)" - return 1 ;; - *) - echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s" - sleep "$POLL_INTERVAL" ;; - esac - done - } - - deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-service" "${{ vars.AWS_RESOURCE_PREFIX }}-task" - deploy_and_wait "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" "${{ vars.AWS_RESOURCE_PREFIX }}-celery-task" + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-service + task-definition: ${{ vars.AWS_RESOURCE_PREFIX }}-task + + - name: Deploy celery service + timeout-minutes: 15 + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task + task-definition: ${{ vars.AWS_RESOURCE_PREFIX }}-celery-task notify: needs: [verify-ci, build] if: always() runs-on: ubuntu-latest steps: - - name: Notify Discord - env: - DISCORD_WEBHOOK_URL: ${{ secrets.DISCORD_WEBHOOK_URL }} - RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - NAME: kaapi-production - RELEASE: ${{ github.ref_name }} - # true only when the build+deploy job succeeded. - OK: ${{ needs.build.result == 'success' }} - run: | - [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } - if [ "$OK" = "true" ]; then - TITLE="๐ŸŸข $NAME deployment healthy"; COLOR=3066993 # green - else - TITLE="๐Ÿ”ด $NAME deployment failed"; COLOR=15158332 # red - fi - SHA_SHORT=$(echo "${{ github.sha }}" | cut -c1-7) - payload=$(jq -n \ - --arg title "$TITLE" \ - --argjson color "$COLOR" \ - --arg release "$RELEASE" \ - --arg sha "$SHA_SHORT" \ - --arg url "$RUN_URL" \ - '{embeds: [{ - title: $title, url: $url, color: $color, - fields: [ - {name: "Release", value: $release, inline: true}, - {name: "SHA", value: $sha, inline: true} - ], - timestamp: (now | todate) - }]}') - curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ - || echo "Discord notification failed to send" + - uses: actions/checkout@v7 + - uses: ./.github/actions/discord-notify + with: + webhook-url: ${{ secrets.DISCORD_WEBHOOK_URL }} + name: kaapi-production + release: ${{ github.ref_name }} + sha: ${{ github.sha }} + run-url: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + ok: ${{ needs.build.result == 'success' }} + # Call out the specific case where the release was blocked by CI. + failure-reason: ${{ needs.verify-ci.result != 'success' && 'CI never passed on the tagged commit โ€” release blocked.' || '' }} diff --git a/.github/workflows/deploy-staging-ecs.yml b/.github/workflows/deploy-staging-ecs.yml index 12dbeb76d..dc780e83a 100644 --- a/.github/workflows/deploy-staging-ecs.yml +++ b/.github/workflows/deploy-staging-ecs.yml @@ -71,6 +71,9 @@ jobs: fi echo "Migration completed successfully" - - name: Deploy to ECS - run: | - aws ecs update-service --cluster ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster --service ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service --force-new-deployment + - name: Deploy to ECS and verify rollout + timeout-minutes: 15 + uses: ./.github/actions/ecs-deploy + with: + cluster: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-cluster + service: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-service diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 7bfa50d60..0381de2be 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -36,14 +36,16 @@ jobs: env: INSTANCE_ID: ${{ secrets.STAGING_EC2_INSTANCE_ID }} SECRET_ID: ${{ vars.STAGING_SECRET_ID }} + DEPLOY_SHA: ${{ github.event.workflow_run.head_sha || github.sha }} run: | SECRET_ID=$(echo "$SECRET_ID" | tr -d '[:space:]') + echo "Deploying commit: $DEPLOY_SHA" DEPLOY_CMD="cd /data/kaapi-backend \ && git fetch --all \ - && git pull origin main \ + && git checkout $DEPLOY_SHA \ && SECRET_ID=$SECRET_ID sh scripts/fetch-secrets.sh \ - && docker compose -f docker-compose.staging.yml build \ + && GIT_SHA=$DEPLOY_SHA docker compose -f docker-compose.staging.yml build \ && docker compose -f docker-compose.staging.yml --profile migrate run --rm migrate \ && docker compose -f docker-compose.staging.yml up -d --wait --remove-orphans \ && docker image prune -f" @@ -97,113 +99,17 @@ jobs: --query '{Status:Status,Stdout:StandardOutputContent,Stderr:StandardErrorContent}' \ --output json - ecs-rehearsal: - needs: deploy - runs-on: ubuntu-latest - environment: AWS_ENV_VARS - permissions: - id-token: write - contents: read - steps: - - name: Checkout the repo - uses: actions/checkout@v7 - - - name: Configure AWS credentials - uses: aws-actions/configure-aws-credentials@v6 - with: - role-to-assume: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} - aws-region: ap-south-1 - - - name: Login to Amazon ECR - id: login-ecr - uses: aws-actions/amazon-ecr-login@v2 - - - name: Build and push staging image - env: - REGISTRY: ${{ steps.login-ecr.outputs.registry }} - REPOSITORY: ${{ vars.AWS_RESOURCE_PREFIX }}-staging-repo - run: | - docker build \ - --build-arg GIT_SHA=${{ github.sha }} \ - -t $REGISTRY/$REPOSITORY:latest \ - ./backend - docker push $REGISTRY/$REPOSITORY:latest - - - name: Scale staging ECS up and verify rollout - timeout-minutes: 15 - env: - CLUSTER: kaapi-staging - SERVICES: "kaapi-staging-backend-celery kaapi-staging-celery-worker" - POLL_INTERVAL: "15" - run: | - for SERVICE in $SERVICES; do - aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ - --desired-count 1 --force-new-deployment >/dev/null - echo "[$SERVICE] scaled to 1" - done - - for SERVICE in $SERVICES; do - echo "[$SERVICE] waiting for rollout" - while true; do - STATE=$(aws ecs describe-services --cluster "$CLUSTER" --services "$SERVICE" \ - --query "services[0].deployments[?status=='PRIMARY'].rolloutState | [0]" \ - --output text) - case "$STATE" in - COMPLETED) echo "[$SERVICE] rehearsal rollout COMPLETED"; break ;; - FAILED) - echo "::error::[$SERVICE] rehearsal FAILED โ€” the production ECS deploy path is broken" - exit 1 ;; - *) echo "[$SERVICE] rollout $STATE โ€” waiting ${POLL_INTERVAL}s"; sleep "$POLL_INTERVAL" ;; - esac - done - done - - - name: Scale staging ECS back to 0 - if: always() - env: - CLUSTER: kaapi-staging - SERVICES: "kaapi-staging-backend-celery kaapi-staging-celery-worker" - run: | - for SERVICE in $SERVICES; do - aws ecs update-service --cluster "$CLUSTER" --service "$SERVICE" \ - --desired-count 0 >/dev/null - echo "[$SERVICE] scaled back to 0" - done - notify: - needs: [deploy, ecs-rehearsal] + needs: [deploy] if: ${{ always() && needs.deploy.result != 'skipped' }} runs-on: ubuntu-latest steps: - - name: Notify Discord - env: - DISCORD_WEBHOOK_URL: ${{ secrets.DISCORD_WEBHOOK_URL }} - RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - NAME: kaapi-staging - RELEASE: "#${{ github.run_number }}" - # true only when every deploy job succeeded. - OK: ${{ needs.deploy.result == 'success' && needs.ecs-rehearsal.result == 'success' }} - run: | - [ -z "$DISCORD_WEBHOOK_URL" ] && { echo "No webhook configured, skipping"; exit 0; } - if [ "$OK" = "true" ]; then - TITLE="๐ŸŸข $NAME deployment healthy"; COLOR=3066993 # green - else - TITLE="๐Ÿ”ด $NAME deployment failed"; COLOR=15158332 # red - fi - SHA_SHORT=$(echo "${{ github.sha }}" | cut -c1-7) - payload=$(jq -n \ - --arg title "$TITLE" \ - --argjson color "$COLOR" \ - --arg release "$RELEASE" \ - --arg sha "$SHA_SHORT" \ - --arg url "$RUN_URL" \ - '{embeds: [{ - title: $title, url: $url, color: $color, - fields: [ - {name: "Release", value: $release, inline: true}, - {name: "SHA", value: $sha, inline: true} - ], - timestamp: (now | todate) - }]}') - curl -sf -H "Content-Type: application/json" -X POST -d "$payload" "$DISCORD_WEBHOOK_URL" \ - || echo "Discord notification failed to send" + - uses: actions/checkout@v7 + - uses: ./.github/actions/discord-notify + with: + webhook-url: ${{ secrets.DISCORD_WEBHOOK_URL }} + name: kaapi-staging + release: "#${{ github.run_number }}" + sha: ${{ github.sha }} + run-url: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + ok: ${{ needs.deploy.result == 'success' }} diff --git a/backend/app/tests/core/test_health.py b/backend/app/tests/core/test_health.py new file mode 100644 index 000000000..83dc2c914 --- /dev/null +++ b/backend/app/tests/core/test_health.py @@ -0,0 +1,12 @@ +from fastapi.testclient import TestClient + +from app.core.config import settings + + +def test_health_reports_status_and_sha(client: TestClient) -> None: + response = client.get("/health") + + assert response.status_code == 200 + body = response.json() + assert body["status"] == "ok" + assert body["sha"] == settings.GIT_SHA diff --git a/docker-compose.staging.yml b/docker-compose.staging.yml index e7e8c149b..7156be249 100644 --- a/docker-compose.staging.yml +++ b/docker-compose.staging.yml @@ -7,6 +7,8 @@ services: restart: always build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} env_file: - path: .env required: true From e1f53dc162dfe2a786f1ee6f395af7f06d1e211b Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:47:11 +0530 Subject: [PATCH 6/7] fix(*): remove the comments --- .github/actions/ecs-deploy/action.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/actions/ecs-deploy/action.yml b/.github/actions/ecs-deploy/action.yml index e9d83f844..170185c48 100644 --- a/.github/actions/ecs-deploy/action.yml +++ b/.github/actions/ecs-deploy/action.yml @@ -32,8 +32,6 @@ runs: FAMILY: ${{ inputs.task-definition }} POLL_INTERVAL: ${{ inputs.poll-interval }} run: | - # The circuit breaker is what turns a crash-looping rollout into a - # FAILED state (with rollback) instead of one that hangs until timeout. args=(--cluster "$CLUSTER" --service "$SERVICE" --force-new-deployment --deployment-configuration '{"deploymentCircuitBreaker":{"enable":true,"rollback":true},"maximumPercent":200,"minimumHealthyPercent":100}') [ -n "$FAMILY" ] && args+=(--task-definition "$FAMILY") From 6213186d5bef775ae1cee3379e993e75cb763d50 Mon Sep 17 00:00:00 2001 From: Ayush8923 <80516839+Ayush8923@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:56:31 +0530 Subject: [PATCH 7/7] fix(*): comment addressed --- .github/workflows/deploy-staging.yml | 2 +- docker-compose.staging.yml | 4 ++++ 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/.github/workflows/deploy-staging.yml b/.github/workflows/deploy-staging.yml index 0381de2be..c5a60658d 100644 --- a/.github/workflows/deploy-staging.yml +++ b/.github/workflows/deploy-staging.yml @@ -110,6 +110,6 @@ jobs: webhook-url: ${{ secrets.DISCORD_WEBHOOK_URL }} name: kaapi-staging release: "#${{ github.run_number }}" - sha: ${{ github.sha }} + sha: ${{ github.event.workflow_run.head_sha || github.sha }} run-url: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} ok: ${{ needs.deploy.result == 'success' }} diff --git a/docker-compose.staging.yml b/docker-compose.staging.yml index 7156be249..aa43a6f13 100644 --- a/docker-compose.staging.yml +++ b/docker-compose.staging.yml @@ -46,6 +46,8 @@ services: image: "${DOCKER_IMAGE_BACKEND?Variable not set}:${TAG:-latest}" build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} env_file: - path: .env required: true @@ -60,6 +62,8 @@ services: restart: always build: context: ./backend + args: + GIT_SHA: ${GIT_SHA:-unknown} depends_on: backend: condition: service_healthy