From e5afedfa2f41d5925d8828b0c482ae1ba9328b95 Mon Sep 17 00:00:00 2001 From: BaoVu2k4 Date: Tue, 18 Aug 2026 09:29:25 +0700 Subject: [PATCH] Stop mislabeling the Compose box as production in read-only audits --- .../workflows/audit-production-runtime.yml | 10 +- .github/workflows/audit-qdrant-corpus.yml | 18 +- .github/workflows/build-practice-images.yml | 12 +- .github/workflows/ci.yml | 7 +- .github/workflows/deploy.yml | 113 ---------- .github/workflows/helm-chart.yml | 88 ++++---- .github/workflows/migrate-qdrant-snapshot.yml | 66 ------ .github/workflows/rollback.yml | 58 ----- ...D_PRODUCTION_MIGRATION_STATE_2026-08-17.md | 89 +++++++- .../CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md | 204 ++++++++++++++++++ docs/operations.md | 67 ++++-- infra/argocd/README.md | 36 ---- infra/argocd/applications/dev/app.yaml | 24 --- infra/argocd/applications/prod/app.yaml | 19 -- infra/argocd/applications/staging/app.yaml | 24 --- infra/helm/medical-chatbot/values-prod.yaml | 61 ------ ...-data.yaml => values-production-data.yaml} | 6 +- ...s-practice.yaml => values-production.yaml} | 6 +- 18 files changed, 414 insertions(+), 494 deletions(-) delete mode 100644 .github/workflows/deploy.yml delete mode 100644 .github/workflows/migrate-qdrant-snapshot.yml delete mode 100644 .github/workflows/rollback.yml create mode 100644 coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md delete mode 100644 infra/argocd/README.md delete mode 100644 infra/argocd/applications/dev/app.yaml delete mode 100644 infra/argocd/applications/prod/app.yaml delete mode 100644 infra/argocd/applications/staging/app.yaml delete mode 100644 infra/helm/medical-chatbot/values-prod.yaml rename infra/helm/medical-chatbot/{values-practice-data.yaml => values-production-data.yaml} (79%) rename infra/helm/medical-chatbot/{values-practice.yaml => values-production.yaml} (95%) diff --git a/.github/workflows/audit-production-runtime.yml b/.github/workflows/audit-production-runtime.yml index 1c0e703..66b0d50 100644 --- a/.github/workflows/audit-production-runtime.yml +++ b/.github/workflows/audit-production-runtime.yml @@ -1,4 +1,10 @@ -name: Audit production runtime (read-only) +name: Audit Compose rollback box (read-only) + +# `realvuxbaro.me` has run on k3s since the 2026-08-17 cutover; this workflow +# still SSHes into secrets.EC2_HOST, which is the retired Compose EC2 kept +# only as a manual DNS fallback. Useful for confirming that box is still +# healthy and on a known commit before relying on it as a fallback — it does +# NOT reflect what real production is currently running. on: workflow_dispatch: @@ -14,7 +20,7 @@ jobs: audit: runs-on: ubuntu-latest steps: - - name: Inspect production over SSH + - name: Inspect the Compose rollback box over SSH uses: appleboy/ssh-action@v1.0.3 with: host: ${{ secrets.EC2_HOST }} diff --git a/.github/workflows/audit-qdrant-corpus.yml b/.github/workflows/audit-qdrant-corpus.yml index 7ad523d..41f9903 100644 --- a/.github/workflows/audit-qdrant-corpus.yml +++ b/.github/workflows/audit-qdrant-corpus.yml @@ -1,13 +1,15 @@ -name: Audit production Qdrant corpus (read-only) +name: Audit Compose rollback Qdrant corpus (read-only) -# Answers "is the practice corpus the same corpus production serves?" with +# `realvuxbaro.me` has run on k3s since the 2026-08-17 cutover; this workflow +# still SSHes into secrets.EC2_HOST, which is the retired Compose EC2 kept +# only as a manual DNS fallback. It fingerprints that box's Qdrant with # content hashes rather than a point count, which two different corpora can -# share. The identical script runs against the k3s cluster over SSH, so the -# two fingerprints are directly comparable. +# share — the same script can be run against the k3s side (over SSH, or via +# `docker exec` on its ai-service pod) to check the fallback still matches +# real production before ever relying on it. # -# Read-only: it scrolls points and reads collection info. It changes nothing on -# production, and its path is not in deploy.yml's filters, so merging it cannot -# restart the Compose stack. +# Read-only: it scrolls points and reads collection info. It changes nothing, +# on this box or any other, and cannot trigger a rebuild of it. on: workflow_dispatch: @@ -30,7 +32,7 @@ jobs: - name: Encode fingerprint script run: echo "SCRIPT_B64=$(base64 -w0 scripts/qdrant_fingerprint.py)" >> "$GITHUB_ENV" - - name: Fingerprint production corpus over SSH + - name: Fingerprint the Compose rollback corpus over SSH uses: appleboy/ssh-action@v1.0.3 env: SCRIPT_B64: ${{ env.SCRIPT_B64 }} diff --git a/.github/workflows/build-practice-images.yml b/.github/workflows/build-practice-images.yml index b4a99a3..002a3fb 100644 --- a/.github/workflows/build-practice-images.yml +++ b/.github/workflows/build-practice-images.yml @@ -1,9 +1,11 @@ -name: Build and sync k3s practice images +name: Build and sync k3s images -# Practice-cluster only (readytochat.realvuxbaro.me, ArgoCD-managed on the -# self-hosted k3s box). Does not touch deploy.yml or the production -# EC2/Compose stack — production never pulls a GHCR image and isn't -# ArgoCD-managed at all, so this workflow has no path to affect it. +# This IS the production deploy path. `medical-chatbot-app` (ArgoCD, k3s) is +# the same release behind both realvuxbaro.me and readytochat.realvuxbaro.me +# since the 2026-08-17 cutover — there is no longer a separate "practice" +# Application this workflow avoids touching. The Compose EC2 is unaffected +# only because it has no CI/CD path left at all (deploy.yml/rollback.yml were +# removed); it is a manual DNS fallback, not a deploy target. # # ArgoCD's Applications already autosync (syncPolicy.automated) — the gap # this closes is that the image tag they deploy was a static string diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a39af83..812e621 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,8 +1,9 @@ name: CI -# Runs on every push and every pull request. `deploy.yml` triggers -# independently on push to master; until it is made to depend on this job, a -# red CI does NOT block a deploy — see docs/22-ci-cd.md. +# Runs on every push and every pull request. `build-practice-images.yml` +# (the k3s/ArgoCD production deploy path) triggers independently on push to +# master; until it is made to depend on this job, a red CI does NOT block a +# deploy — see docs/operations.md. on: push: pull_request: diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml deleted file mode 100644 index 4fe79b9..0000000 --- a/.github/workflows/deploy.yml +++ /dev/null @@ -1,113 +0,0 @@ -name: Deploy to production - -on: - push: - branches: [master] - paths: - # Everything the two Dockerfiles actually COPY into their images. - - apps/ai-service/** - - apps/web/** - - packages/** - - ingestion/data/verified/drug_entities.json - # docker-compose.prod.yml, docker-compose.observability.yml, - # Caddyfile, and the grafana/otel/prometheus/tempo configs it mounts. - - infra/docker/** - # Changes to the deploy logic itself should still be exercised. - - .github/workflows/deploy.yml - workflow_dispatch: - -jobs: - deploy: - runs-on: ubuntu-latest - steps: - - name: Deploy over SSH - uses: appleboy/ssh-action@v1.0.3 - env: - GRAFANA_ADMIN_PASSWORD: ${{ secrets.GRAFANA_ADMIN_PASSWORD }} - with: - host: ${{ secrets.EC2_HOST }} - username: ubuntu - key: ${{ secrets.EC2_SSH_KEY }} - envs: GRAFANA_ADMIN_PASSWORD - script: | - set -e - test -n "${GRAFANA_ADMIN_PASSWORD:-}" - export GRAFANA_ADMIN_PASSWORD - cd ~/app - git fetch origin master - git reset --hard origin/master - cd infra/docker - sudo -E docker compose \ - -f docker-compose.prod.yml \ - -f docker-compose.observability.yml \ - up -d --build \ - ai-service web prometheus tempo otel-collector grafana caddy - sudo docker exec docker-caddy-1 caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile - sudo docker exec docker-caddy-1 caddy reload --config /etc/caddy/Caddyfile --adapter caddyfile - sudo docker exec docker-ai-service-1 python -m migrate - sleep 10 - sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/health - sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/ready - sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null http://web:3000 - if ! condition_response=$(sudo docker run --rm --network docker_default curlimages/curl -sf \ - -X POST http://ai-service:8000/v1/rag/query \ - -H 'Content-Type: application/json' \ - --data '{"query":"Đợt gout cấp có thuốc nào được Dược thư ghi chỉ định?","subject_scope":"human","intent":"fact_lookup","conversation_id":"deploy-condition-smoke"}'); then - sudo docker logs --tail 200 docker-ai-service-1 - exit 1 - fi - if ! printf '%s' "${condition_response}" | grep -q '"decision":"answerable"'; then - printf '%s\n' "${condition_response}" - sudo docker logs --tail 200 docker-ai-service-1 - exit 1 - fi - printf '%s' "${condition_response}" | grep -q '"section_key":"chi_dinh"' - sudo docker run --rm --network docker_default curlimages/curl -sf http://prometheus:9090/-/ready - for attempt in $(seq 1 12); do - if sudo docker run --rm --network docker_default curlimages/curl -sf http://tempo:3200/ready; then - break - fi - if [ "${attempt}" -eq 12 ]; then - sudo docker logs --tail 100 docker-tempo-1 - exit 1 - fi - sleep 5 - done - sudo docker run --rm --network docker_default curlimages/curl -sf http://grafana:3000/api/health - sudo docker run --rm --network docker_default curlimages/curl -sf \ - -u "admin:${GRAFANA_ADMIN_PASSWORD}" \ - http://grafana:3000/api/datasources/uid/prometheus > /dev/null - sudo docker run --rm --network docker_default curlimages/curl -sf \ - -u "admin:${GRAFANA_ADMIN_PASSWORD}" \ - http://grafana:3000/api/datasources/uid/tempo > /dev/null - sudo docker run --rm --network docker_default curlimages/curl -sf \ - -u "admin:${GRAFANA_ADMIN_PASSWORD}" \ - http://grafana:3000/api/dashboards/uid/duocthu-observability \ - | grep -q '"uid":"duocthu-observability"' - sudo docker run --rm --network docker_default curlimages/curl -sf \ - -o /dev/null https://realvuxbaro.me/grafana/login - - correlation_id="observability-deploy-$(date +%s)" - response_headers=$(sudo docker run --rm --network docker_default curlimages/curl -sf \ - -D - -o /dev/null \ - -X POST http://ai-service:8000/v1/rag/query \ - -H 'Content-Type: application/json' \ - -H "X-Correlation-ID: ${correlation_id}" \ - --data '{"query":"Paracetamol là thuốc gì?","subject_scope":"unknown","intent":"unknown"}') - trace_id=$(printf '%s\n' "${response_headers}" | tr -d '\r' | awk -F ': ' 'tolower($1) == "x-trace-id" { print $2 }' | tail -n 1) - printf '%s' "${trace_id}" | grep -Eq '^[0-9a-f]{32}$' - - sleep 20 - sudo docker run --rm --network docker_default curlimages/curl -sfG \ - --data-urlencode 'query=duocthu_requests_total' \ - http://prometheus:9090/api/v1/query | grep -q '"__name__":"duocthu_requests_total"' - for attempt in $(seq 1 12); do - if sudo docker run --rm --network docker_default curlimages/curl -sf \ - -o /dev/null "http://tempo:3200/api/traces/${trace_id}"; then - break - fi - if [ "${attempt}" -eq 12 ]; then - exit 1 - fi - sleep 5 - done diff --git a/.github/workflows/helm-chart.yml b/.github/workflows/helm-chart.yml index 5890c7d..548be7c 100644 --- a/.github/workflows/helm-chart.yml +++ b/.github/workflows/helm-chart.yml @@ -23,45 +23,37 @@ jobs: version: v3.17.3 - name: Lint chart run: helm lint infra/helm/medical-chatbot - - name: Render default and production manifests + - name: Render defaults and check the immutable-tag guard run: | helm template default infra/helm/medical-chatbot > /tmp/default.yaml - # values-prod.yaml leaves the image tags empty on purpose: production - # must run an immutable commit SHA, supplied per deploy. Rendering it - # without one has to fail rather than fall back to a development tag, - # so assert that failure here — otherwise the guard could rot into a - # silent default and nobody would notice until a cutover. - if helm template production infra/helm/medical-chatbot \ - --values infra/helm/medical-chatbot/values-prod.yaml \ - > /tmp/untagged.yaml 2>/tmp/untagged.err; then - echo "::error::production render succeeded with no image tag; the immutable-tag guard is gone" + # The live releases carry no image tag in Git -- it is supplied per + # deploy as a commit SHA through the ArgoCD Application. Rendering + # with an empty tag must FAIL rather than fall back to the chart's + # `local` development tag, so assert the failure directly; otherwise + # the guard could rot into a silent default unnoticed. + if helm template production infra/helm/medical-chatbot --values infra/helm/medical-chatbot/values-production.yaml --set aiService.image.tag="" --set web.image.tag="" > /tmp/untagged.yaml 2>/tmp/untagged.err; then + echo "::error::render succeeded with no image tag; the immutable-tag guard is gone" exit 1 fi grep -q 'image.tag must be set to an immutable tag' /tmp/untagged.err - helm template production infra/helm/medical-chatbot \ - --values infra/helm/medical-chatbot/values-prod.yaml \ - --set aiService.image.tag="$GITHUB_SHA" \ - --set web.image.tag="$GITHUB_SHA" \ - > /tmp/production.yaml - grep -q 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"' /tmp/production.yaml - grep -q 'RERANK_ENABLED: "true"' /tmp/production.yaml - grep -q 'checksum/runtime-config:' /tmp/production.yaml - grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-ai-service:$GITHUB_SHA\"" /tmp/production.yaml - grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-web:$GITHUB_SHA\"" /tmp/production.yaml + # ...and with a tag it must resolve the GHCR package, not the local + # development image name. + helm template production infra/helm/medical-chatbot --values infra/helm/medical-chatbot/values-production.yaml --set aiService.image.repository=ghcr.io/baovu2k4/vsf-duocthu-ai-service --set web.image.repository=ghcr.io/baovu2k4/vsf-duocthu-web --set aiService.image.tag="$GITHUB_SHA" --set web.image.tag="$GITHUB_SHA" > /tmp/tagged.yaml + grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-ai-service:$GITHUB_SHA\"" /tmp/tagged.yaml + grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-web:$GITHUB_SHA\"" /tmp/tagged.yaml - # The practice cluster is only evidence for the production migration - # while it renders the same behavioural contract as production, so both - # of its releases are asserted here rather than trusted by review. - - name: Render practice manifests + # These two releases are what realvuxbaro.me actually serves, so their + # contract is asserted here rather than trusted by review. + - name: Render the live production manifests run: | helm template medical-chatbot-app infra/helm/medical-chatbot \ - --values infra/helm/medical-chatbot/values-practice.yaml \ - > /tmp/practice-app.yaml + --values infra/helm/medical-chatbot/values-production.yaml \ + > /tmp/prod-app.yaml helm template medical-chatbot-data infra/helm/medical-chatbot \ - --values infra/helm/medical-chatbot/values-practice-data.yaml \ - > /tmp/practice-data.yaml + --values infra/helm/medical-chatbot/values-production-data.yaml \ + > /tmp/prod-data.yaml # A bare `grep -q` fails the step with no indication of which # assertion broke, and `set -e` ignores a status inverted with `!`, @@ -83,35 +75,35 @@ jobs: } # Behavioural parity with the audited production runtime contract. - expect /tmp/practice-app.yaml 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"' - expect /tmp/practice-app.yaml 'ANSWER_PROVIDER: "bedrock-converse"' - expect /tmp/practice-app.yaml 'EMBEDDING_PROVIDER: "cohere-v4"' - expect /tmp/practice-app.yaml 'EMBEDDING_DIMENSIONS: "1024"' - expect /tmp/practice-app.yaml 'EVIDENCE_MINIMUM_SCORE: "0.12"' - expect /tmp/practice-app.yaml 'RERANK_ENABLED: "true"' - expect /tmp/practice-app.yaml 'AWS_REGION: "us-east-1"' - expect /tmp/practice-app.yaml 'checksum/runtime-config:' - expect /tmp/practice-app.yaml '- host: "readytochat.realvuxbaro.me"' + expect /tmp/prod-app.yaml 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"' + expect /tmp/prod-app.yaml 'ANSWER_PROVIDER: "bedrock-converse"' + expect /tmp/prod-app.yaml 'EMBEDDING_PROVIDER: "cohere-v4"' + expect /tmp/prod-app.yaml 'EMBEDDING_DIMENSIONS: "1024"' + expect /tmp/prod-app.yaml 'EVIDENCE_MINIMUM_SCORE: "0.12"' + expect /tmp/prod-app.yaml 'RERANK_ENABLED: "true"' + expect /tmp/prod-app.yaml 'AWS_REGION: "us-east-1"' + expect /tmp/prod-app.yaml 'checksum/runtime-config:' + expect /tmp/prod-app.yaml '- host: "readytochat.realvuxbaro.me"' # The production hostname now lives on this cluster, routed and with # its own certificate secret -- kept separate from the rehearsal # hostname's so one renewal failure cannot take both names offline. - expect /tmp/practice-app.yaml '- host: "realvuxbaro.me"' - expect /tmp/practice-app.yaml 'secretName: realvuxbaro-tls' - expect /tmp/practice-app.yaml 'secretName: readytochat-tls' + expect /tmp/prod-app.yaml '- host: "realvuxbaro.me"' + expect /tmp/prod-app.yaml 'secretName: realvuxbaro-tls' + expect /tmp/prod-app.yaml 'secretName: readytochat-tls' # Grafana answers on that same public hostname. Anonymous access may # be open, but never as Admin, never with the login form disabled, # and its root URL must be the name users actually arrive on. - expect /tmp/practice-app.yaml 'value: "https://realvuxbaro.me/grafana/"' - refute /tmp/practice-app.yaml 'value: "Admin"' + expect /tmp/prod-app.yaml 'value: "https://realvuxbaro.me/grafana/"' + refute /tmp/prod-app.yaml 'value: "Admin"' # grep is line-oriented, so read the value on the line after each # flag rather than trying to match the pair as one pattern. for check in "GF_AUTH_ANONYMOUS_ORG_ROLE:Viewer" "GF_AUTH_DISABLE_LOGIN_FORM:false"; do flag=${check%%:*} want=${check#*:} - got=$(grep -A1 -- "$flag" /tmp/practice-app.yaml | grep -- 'value:' | tr -d ' "' | cut -d: -f2) + got=$(grep -A1 -- "$flag" /tmp/prod-app.yaml | grep -- 'value:' | tr -d ' "' | cut -d: -f2) if [ "$got" != "$want" ]; then echo "::error::$flag rendered as '$got', expected '$want'" exit 1 @@ -123,10 +115,10 @@ jobs: # prune can never delete the corpus or the query history. Only those # two use volumeClaimTemplates — the observability PVCs are the app # release's own and are expected here. - refute /tmp/practice-app.yaml 'volumeClaimTemplates' - expect /tmp/practice-app.yaml 'medical-chatbot-data-medical-chatbot-qdrant' + refute /tmp/prod-app.yaml 'volumeClaimTemplates' + expect /tmp/prod-app.yaml 'medical-chatbot-data-medical-chatbot-qdrant' # ...and the data release must own nothing else. - refute /tmp/practice-data.yaml 'medical-chatbot-data-medical-chatbot-ai-service' - refute /tmp/practice-data.yaml 'kind: Ingress' - expect /tmp/practice-data.yaml 'volumeClaimTemplates' + refute /tmp/prod-data.yaml 'medical-chatbot-data-medical-chatbot-ai-service' + refute /tmp/prod-data.yaml 'kind: Ingress' + expect /tmp/prod-data.yaml 'volumeClaimTemplates' diff --git a/.github/workflows/migrate-qdrant-snapshot.yml b/.github/workflows/migrate-qdrant-snapshot.yml deleted file mode 100644 index 2bc0dc7..0000000 --- a/.github/workflows/migrate-qdrant-snapshot.yml +++ /dev/null @@ -1,66 +0,0 @@ -name: Migrate Qdrant snapshot to practice cluster - -# One-off, manual (workflow_dispatch only) bridge: snapshots the production -# Qdrant collection (a live, non-disruptive Qdrant operation — this is how -# the original prod migration was done, just in reverse) and relays the -# snapshot files to the isolated k3s practice EC2. Uses the SAME EC2_SSH_KEY -# deploy.yml already has (never exposed to the operator) plus a new -# PRACTICE_SSH_KEY scoped only to the practice box. Delete this workflow -# file once the one-time migration is done — it is not part of the regular -# deploy path. - -on: - workflow_dispatch: - -jobs: - migrate: - runs-on: ubuntu-latest - steps: - - name: Set up SSH key - run: | - mkdir -p ~/.ssh - printf '%s\n' "${{ secrets.EC2_SSH_KEY }}" > ~/.ssh/prod.pem - chmod 600 ~/.ssh/prod.pem - ssh-keyscan -H "${{ secrets.EC2_HOST }}" >> ~/.ssh/known_hosts 2>/dev/null - - - name: Report production Qdrant version - run: | - ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" ' - sudo docker run --rm --network docker_default curlimages/curl -sf http://qdrant:6333/ - ' - - - name: Snapshot Qdrant collections on production - run: | - ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" ' - set -e - snap() { - collection="$1"; outfile="$2" - name=$(sudo docker run --rm --network docker_default curlimages/curl -sf -X POST "http://qdrant:6333/collections/${collection}/snapshots" | grep -oP "\"name\":\"\K[^\"]+") - test -n "$name" - sudo docker run --rm --network docker_default curlimages/curl -sf "http://qdrant:6333/collections/${collection}/snapshots/${name}" > "$outfile" - } - snap duocthu_v1 /tmp/duocthu_v1.snapshot - snap duocthu_v1__manifest /tmp/duocthu_v1__manifest.snapshot - sudo chown ubuntu:ubuntu /tmp/duocthu_v1.snapshot /tmp/duocthu_v1__manifest.snapshot - ls -la /tmp/*.snapshot - ' - - - name: Pull snapshots to the runner - run: | - scp -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}:/tmp/duocthu_v1.snapshot" ./duocthu_v1.snapshot - scp -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}:/tmp/duocthu_v1__manifest.snapshot" ./duocthu_v1__manifest.snapshot - ls -la ./*.snapshot - - - name: Upload snapshots as a workflow artifact - uses: actions/upload-artifact@v4 - with: - name: qdrant-snapshots - path: | - duocthu_v1.snapshot - duocthu_v1__manifest.snapshot - retention-days: 1 - - - name: Clean up temp files on production - if: always() - run: | - ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" 'rm -f /tmp/duocthu_v1.snapshot /tmp/duocthu_v1__manifest.snapshot' || true diff --git a/.github/workflows/rollback.yml b/.github/workflows/rollback.yml deleted file mode 100644 index 7303a04..0000000 --- a/.github/workflows/rollback.yml +++ /dev/null @@ -1,58 +0,0 @@ -name: Rollback production - -# Manual escape hatch for deploy.yml. deploy.yml has NO automatic rollback: -# it runs `git reset --hard origin/master`, rebuilds and runs migrations -# BEFORE its health checks, so a deploy that fails those checks leaves the -# server on the bad commit with no automatic recovery. This workflow points -# the same reset+rebuild+health-check sequence at an earlier commit instead. -# -# Migrations are forward-only (apps/ai-service/migrate.py, no down scripts) -# but every migration so far uses IF NOT EXISTS / ADD COLUMN IF NOT EXISTS, -# so re-running them against an older commit is a no-op, not an error. A -# future non-idempotent migration would break this guarantee. - -on: - workflow_dispatch: - inputs: - target_sha: - description: "Commit SHA or tag to roll back to (e.g. the last known-good commit from a previous successful 'Deploy to production' run)" - required: true - -jobs: - rollback: - runs-on: ubuntu-latest - steps: - - name: Rollback over SSH - uses: appleboy/ssh-action@v1.0.3 - env: - GRAFANA_ADMIN_PASSWORD: ${{ secrets.GRAFANA_ADMIN_PASSWORD }} - TARGET_SHA: ${{ inputs.target_sha }} - with: - host: ${{ secrets.EC2_HOST }} - username: ubuntu - key: ${{ secrets.EC2_SSH_KEY }} - envs: GRAFANA_ADMIN_PASSWORD,TARGET_SHA - script: | - set -e - test -n "${GRAFANA_ADMIN_PASSWORD:-}" - export GRAFANA_ADMIN_PASSWORD - cd ~/app - git fetch origin - git rev-parse --verify "${TARGET_SHA}^{commit}" - git reset --hard "${TARGET_SHA}" - echo "Rolled back to $(git rev-parse HEAD) — $(git log -1 --format=%s)" - cd infra/docker - sudo -E docker compose \ - -f docker-compose.prod.yml \ - -f docker-compose.observability.yml \ - up -d --build \ - ai-service web prometheus tempo otel-collector grafana caddy - sudo docker exec docker-caddy-1 caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile - sudo docker exec docker-caddy-1 caddy reload --config /etc/caddy/Caddyfile --adapter caddyfile - sudo docker exec docker-ai-service-1 python -m migrate - sleep 10 - sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/health - sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/ready - sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null http://web:3000 - sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null https://realvuxbaro.me/grafana/login - echo "Rollback to ${TARGET_SHA} verified healthy." diff --git a/coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md b/coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md index 07d086d..c7699ec 100644 --- a/coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md +++ b/coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md @@ -461,16 +461,95 @@ certificate for that name — the SNI handshake fails with an internal error, an the certificate it does hold is apex-only. Decide whether to point `www` at k3s and add it to `tls`, or drop the record. +## Grafana exposure closed (2026-08-17, post-cutover) + +The cutover moved production onto the cluster whose Grafana ran +`GF_AUTH_ANONYMOUS_ORG_ROLE=Admin` with `GF_AUTH_DISABLE_LOGIN_FORM=true` — +defensible on a throwaway rehearsal box, not on a public production hostname. +Confirmed live before the fix: `/grafana/api/org` and `/grafana/api/datasources` +both answered HTTP 200 with no credentials, listing internal service URLs. + +Fixed at `39b1159` plus one out-of-band step: + +- Anonymous access kept, but demoted to `Viewer`, so dashboards stay open for a + demo while the datasource and dashboard write APIs are refused (verified: + `POST /api/datasources` → 403). +- `GF_AUTH_DISABLE_LOGIN_FORM` is now hardcoded `false`. Combined with + anonymous Admin it had previously left no way to sign in as a real admin. +- `ingress.host` is now `realvuxbaro.me`, with the rehearsal name moved to + `extraHosts`. Grafana builds `GF_SERVER_ROOT_URL` from `ingress.host`, so it + had continued advertising the rehearsal hostname after the cutover. +- CI now asserts the rendered role is `Viewer`, the login form is not disabled, + and the root URL is the production hostname. + +**Setting the admin password through Helm was not enough.** Grafana persists +its user table in SQLite on a PVC, so `GF_SECURITY_ADMIN_PASSWORD` did not +overwrite the existing credential: after the rollout, `admin:change-me` still +authenticated and the new password did not. It required +`grafana cli admin reset-admin-password --password-from-stdin` inside the pod. +Anyone rotating this password later must do the same — changing the Secret +alone is silently ineffective. + +Final state verified: anonymous → `/api/admin/settings` 403; admin with the new +password 200; `admin:change-me` 403; anonymous dashboard search 200; app 200. +The password lives in the gitignored `.env.k3s-practice` as +`GRAFANA_ADMIN_PASSWORD` and in the ArgoCD Application's inline values. It is +not in Git. + +## Repository now describes what is actually live + +`values-practice.yaml` / `values-practice-data.yaml` became +`values-production.yaml` / `values-production-data.yaml`, and +`global.environment` is now `production` (confirmed in the running ConfigMap). +The separate, never-deployed `values-prod.yaml` is deleted — two "prod" files +beside a "practice" file that was the real one was the worst of both. + +Done in three commits so no sync ever referenced a missing path: add the new +files (`527d3a8`), repoint both Applications and confirm Synced/Healthy, then +delete the old ones (`6468b16`). The ai-service Pod rolled cleanly on the +environment change and the site stayed up throughout. + +The immutable-tag guard moved onto the file that is actually live and now +checks both directions: an empty tag must fail the render, and a supplied SHA +must resolve the GHCR package rather than the chart's local development name. + +## Live chat driven directly against production + +Not a golden-set replay — unaccented text, a typo, missing dimensions and +pronoun follow-ups within one conversation, through the real `/api/chat`: + +| Input | Behaviour | +| --- | --- | +| `lieu paracetamol` | `clarify` — asks adult or child, offers both as quick replies; answers with a citation once told `nguoi lon` | +| `paracetamol cho tre em` → `be 3 tuoi` | `clarify` twice: age, then still demands weight. The paediatric gate requiring **both** is intact | +| `metformim co tac dung phu gi` | Resolves the typo and answers metformin's GI adverse effects with a citation | +| `amlodipin dung the nao` → `the con chong chi dinh thi sao` | Carries the referent across turns and answers the contraindication | +| `gia thuoc paracetamol ... bao nhieu tien` | Abstains cleanly: the formulary does not carry prices | + +Cosmetic only: the paediatric follow-up echoes the user's unaccented text back +inside an otherwise accented sentence ("Bé 3 tuoi nặng bao nhiêu kg?"). + +## Final verified state + +`https://realvuxbaro.me/` 200 from `44.206.194.195` with a valid certificate; +live query `answerable` in 14.3 s with a citation; Grafana anonymous read 200 +but admin 403, admin login 200; `readytochat.realvuxbaro.me` 200; +`argocd.realvuxbaro.me` 200; and the Compose host still answers the apex with +a valid certificate, so the rollback is a 60-second A-record revert. + +Temporary SSH allowances on `sg-018fc3cde8282f26d` were revoked; only the +pre-existing `103.238.70.200/32` remains on port 22. Note that this +workstation's SSH egress was `101.99.23.84`, so re-opening will be needed for +future cluster access — and the three IP-reporting services disagree, so read +the real one from the host's own `auth.log` rather than trusting any of them. + ## Migration risks currently open 1. `www.realvuxbaro.me` is broken (see above) and always was. 2. Secrets, resource limits and failure recovery still need explicit rehearsal gates. -3. Answer-content parity between environments is still unverified — decisions - and latency match, answer length does not always. -4. The cluster still serves production from values files named `values-practice*`. - Rename once the acceptance window closes and the Compose host is retired; - `values-prod.yaml` is now unused by anything live. +3. The Compose EC2 is still running as the rollback. Stop — never terminate — + once the acceptance window closes. 3. Cluster-level inspection is currently blocked from this machine: the practice security group scopes port 6443 to the operator's own outbound IP, which has changed again, so `kubectl` hangs. ArgoCD's API and the public diff --git a/coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md b/coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md new file mode 100644 index 0000000..e8e076a --- /dev/null +++ b/coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md @@ -0,0 +1,204 @@ +> **Decision update, same day, later session:** owner confirmed intent to run +> **one EC2 only** (k3s). This supersedes PR A step 2 below (reduce to +> `workflow_dispatch`) — `deploy.yml`, `rollback.yml`, and the never-applied +> `infra/argocd/applications/**` scaffold (wrong Application names, dead +> `values-prod.yaml` reference, placeholder team-repo TODOs) were deleted +> outright on branch `agent/retire-compose-cicd`, not disabled. D1/D2/D3/D4 are +> closed by deletion rather than by fixing the probe. `docs/operations.md` +> Deploy/Rollback sections rewritten to describe the actual k3s/ArgoCD path. +> Compose EC2 itself (`52.0.158.61`) is untouched pending an explicit stop/ +> terminate decision (PR D) — do not stop or terminate it without that go-ahead. +> PR B (Helm hygiene) and PR C (rollback runbook, now mostly covered in +> `docs/operations.md`) remain open. + +# Plan — make the CI/CD path safe after the k3s cutover (2026-08-18) + +Written by Claude (Opus) for execution by another agent. Every claim below was +verified on 2026-08-18 by the command shown next to it. **Re-verify before +acting** — runtime state can change after this snapshot. + +## 1. The topology changed and two workflows never noticed + +This is the root cause of everything in this plan. The 2026-08-17 cutover moved +`realvuxbaro.me` from the Compose EC2 to the k3s/ArgoCD cluster, but the +Compose-era workflows still describe and probe the old world. + +| Fact | Verified by | +| --- | --- | +| `realvuxbaro.me` → `44.206.194.195` (k3s) | `nslookup realvuxbaro.me 8.8.8.8` | +| `readytochat.realvuxbaro.me` → `44.206.194.195` — **same cluster, same release** | same | +| Compose EC2 `52.0.158.61` is off DNS; rollback target only | `values-production.yaml` `ingress.host` + DNS above | +| Compose is still at `df57e6b` | `git log df57e6b..52e8828 -- ` returns empty | +| App code at `df57e6b` == app code at `master` | the intervening commits touch only docs/CI/coordination | + +So: **the Compose rollback is currently valid**, and the window to fix this is +now, before the first post-cutover app change lands. + +`values-production.yaml` sets `ingress.host: realvuxbaro.me` and is rendered by +ArgoCD Application `medical-chatbot-app` — the same Application +`.github/scripts/sync_practice_argocd.py` repoints (`APP_NAME = +"medical-chatbot-app"`). **"Practice" and production are the same release.** + +## 2. Confirmed defects, most dangerous first + +### D1 — both Compose workflows verify the wrong machine (P0) + +`deploy.yml:88` and `rollback.yml:57` both end with: + +``` +docker run --rm --network docker_default curlimages/curl -sf \ + -o /dev/null https://realvuxbaro.me/grafana/login +``` + +That hostname now resolves to **k3s**, not the box the workflow just rebuilt. +The check passes by hitting a completely different server. + +This is worst in `rollback.yml`, whose whole purpose is to be trustworthy in an +emergency: it prints `Rollback to verified healthy` on the strength of a +probe that never touched the rolled-back box. A broken rollback would report +success. + +### D2 — one bad commit poisons production *and* the rollback (P0) + +`deploy.yml` and `build-practice-images.yml` trigger on the **same four paths**: +`apps/ai-service/**`, `apps/web/**`, `packages/**`, +`ingestion/data/verified/drug_entities.json`. + +A single push to `master` therefore rolls production forward on k3s **and** +rebuilds the Compose box that is supposed to be the known-good fallback. The +"proven rollback" only holds while Compose stays on a good commit. + +### D3 — `build-practice-images.yml` header comment is now false (P1) + +> *"Does not touch deploy.yml or the production EC2/Compose stack — production +> never pulls a GHCR image and isn't ArgoCD-managed at all, so this workflow +> has no path to affect it."* + +Production **is** ArgoCD-managed and **does** pull GHCR images. This workflow is +the production deploy pipeline. The comment invites exactly the push that breaks +production, and its name reinforces the error. + +### D4 — a dead ArgoCD manifest sits in Git (P1) + +`infra/argocd/applications/prod/app.yaml` lists +`valueFiles: [values.yaml, values-prod.yaml]`, but `values-prod.yaml` was +deleted in `6468b16`. It also declares `syncPolicy: {}` with a comment claiming +prod sync needs manual approval — the live Application is automated. Applying +this file would fail or deploy something wrong. `dev/` and `staging/` under the +same directory have not been checked and may share the defect. + +### D5 — the WIP Helm change shadows an existing variable (P2) + +Uncommitted in the main worktree. The chart **already** emits `AWS_REGION` in +the ConfigMap from `aiService.config.awsRegion` +(`templates/ai-service.yaml:15`, asserted in `helm-chart.yml`). The WIP diff +adds a second `AWS_REGION` as a container `env:` entry — and in Kubernetes an +explicit `env:` **overrides** `envFrom`, so enabling it would silently shadow +the ConfigMap. Inert today only because `aws.region` defaults to `""`. + +The static-credentials half is a genuine gap and worth keeping. + +### D6 — `helm-chart.yml` asserts invariants but never diffs (P2) + +It renders both live releases and checks a strong list (Qwen, rerank, TLS +secrets, Grafana role, `refute volumeClaimTemplates`). But nothing compares the +render against the previous commit, so a chart change that alters anything +*outside* that list reaches production silently. + +### D7 — production has no rollback workflow at all (P2) + +`rollback.yml` targets `secrets.EC2_HOST` — the Compose box. Nothing rolls back +k3s. Real production rollback today is a manual Namecheap A-record revert +(~60s TTL) or an ArgoCD revision/tag revert, neither written down. + +## 3. Execution plan + +### Ground rules + +- **Never push to `master`.** Every change goes through a PR. ArgoCD auto-syncs + `master` with `selfHeal` + `prune`; a merge touching `infra/helm/**` applies + to production with no human gate. +- `ci.yml` runs on every PR (ruff, pytest, ingestion tests, web lint+build). + `helm-chart.yml` runs on PRs touching `infra/helm/**`. Both must be green. +- Do not touch `git.vinmec.tech`, team ArgoCD/k3s, or team repos. +- Do not start, stop, or terminate any EC2 instance without an explicit go. +- Do not edit files in the other worktrees (`D:\VSF-DUOCTHU-codex-*`, + `D:\VSF-DUOCTHU-claude-gitops`); check `coordination/` for active claims and + file your own claim before starting. + +### PR A — workflow safety (D1, D2, D3, D4) + +Touches only `.github/**` and `infra/argocd/**`. ArgoCD renders +`infra/helm/medical-chatbot`, so **this PR cannot alter production manifests**. +Confirm that rather than assume it. + +1. **Fix the misdirected probes (D1).** In `deploy.yml` and `rollback.yml`, + make the Caddy/Grafana check target the box being deployed instead of a + public DNS name that now points elsewhere — e.g. resolve the hostname to the + local Caddy container so TLS and routing are still exercised. + **Verification gate:** prove the fixed check *fails* when Caddy is broken. + A probe that cannot fail is the defect being fixed, not a fix. (See the + `set -e` / `! grep` silent-pass traps already documented in + `helm-chart.yml`.) +2. **Stop `deploy.yml` firing on push (D2).** Reduce it to `workflow_dispatch` + only. Compose then stays pinned at `df57e6b` — a stable rollback rather than + one that tracks `master`. State the trade-off in the commit message: the + fallback stops drifting, but also stops receiving fixes, so it goes stale as + production moves. That is acceptable for a time-boxed acceptance window and + is the subject of PR D. +3. **Tell the truth in `build-practice-images.yml` (D3).** Replace the false + header comment. Renaming the workflow to name it as the production deploy + path is preferable — **trap:** it self-references in its own `paths:` filter + and the sync script path, so both must be updated together or the workflow + silently stops triggering. +4. **Remove or correct `infra/argocd/applications/prod/app.yaml` (D4).** Check + `dev/` and `staging/` in the same directory for the same rot. Deleting is + fine if nothing applies them; verify that first. + +### PR B — Helm hygiene (D5, D6) + +Touches `infra/helm/**`, so merging **does** reach production. Highest care. + +5. **Drop the redundant `aws.region` block from the WIP diff (D5)**; keep the + static-credentials support, still defaulted off. Guard against a nil `aws` + key so a values file that omits it cannot break the render — a template + error here means ArgoCD cannot sync production at all. +6. **Add a baseline render diff to `helm-chart.yml` (D6).** Render + `values-production.yaml` and `values-production-data.yaml` at the PR base and + at HEAD, then surface the diff in the job summary. The goal is that no chart + change ever reaches production without a human having seen exactly what it + does to the manifests. + **Verification gate:** the diff for PR B itself must be **empty** — the + static-credentials change is defaults-off and must render byte-identically. + If it is not empty, stop and explain why before merging. + +Note: `helm` is **not installed** on this workstation (`helm: command not found` +in both bash and PowerShell). Either install it or rely on the CI render — but +do not claim the render is unchanged without one of the two actually running it. + +### PR C — document the real rollback (D7) + +7. Write the production rollback runbook: revert the ArgoCD Application to the + previous image tag, and/or revert the `realvuxbaro.me` A record to + `52.0.158.61`. Note that the Namecheap edit is a **manual owner step** — a + harness permission classifier has blocked agent form input on that page + before, so the runbook must not assume an agent can do it. + +### PR D — decide the Compose lifecycle (owner call, not an agent call) + +8. Compose is a second `t3.large` running purely as a fallback. Once the + acceptance window closes, ArgoCD's own revision history covers rollback and + the instance is redundant. Surface the choice and the monthly cost; **do not + act on it without an explicit go.** + +### Out of scope + +`Feature-List-AI-Duoc-thu-V1.md`, `presentation/`, and `.claude/skills/` are +untracked and match no workflow trigger path. Committing them is inert and can +be a separate trivial commit — keep it out of PRs A–D. + +## 4. Order and why + +D1 first: an untrustworthy rollback is worse than no rollback, because it fails +silently at the moment of maximum pressure. D2 next: it is the defect that would +consume the rollback. Everything after is hardening. diff --git a/docs/operations.md b/docs/operations.md index dec1bd6..42dd9a6 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -1,32 +1,67 @@ # Vận hành, triển khai và xử lý sự cố > Loại chính: How-to -> Phạm vi: EC2 + Docker Compose hiện hành +> Phạm vi: k3s + ArgoCD (production kể từ cutover 2026-08-17) + +Production (`realvuxbaro.me`) chạy trên k3s, quản lý bởi ArgoCD Application +**`medical-chatbot-app`** (ai-service + web + observability) và +**`medical-chatbot-data`** (PostgreSQL + Qdrant, tách release để prune/self-heal +phía app không bao giờ đụng vào dữ liệu). Cả hai đặt `syncPolicy.automated` với +`selfHeal` + `prune` — **mọi merge vào `master` áp thẳng vào production, không +có cổng duyệt thủ công.** EC2 Docker Compose (`52.0.158.61`) không còn nhận +deploy tự động; xem `coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md` cho +lý do và tình trạng hiện tại của máy đó. ## Deploy -Trước deploy, ghi commit SHA, yêu cầu CI AI/ingestion/web xanh, kiểm tra secret và -Qdrant manifest tương thích, đồng thời đánh giá migration. Chạy `deploy.yml` theo -path/branch filter hoặc manual dispatch và theo dõi đến khi reconcile xong. +Hai loại thay đổi đi hai đường khác nhau: -Sau deploy: +**Thay đổi code app** (`apps/ai-service/**`, `apps/web/**`, `packages/**`, +`ingestion/data/verified/drug_entities.json`) — merge vào `master` kích hoạt +`build-practice-images.yml`: build + push image GHCR gắn tag theo commit SHA, +sau đó `.github/scripts/sync_practice_argocd.py` ghi tag mới vào Application +`medical-chatbot-app` và gọi sync. Workflow tự xác nhận +`readytochat.realvuxbaro.me` đã lên bản mới trước khi báo thành công. +`ci.yml` (ruff/pytest/lint/build) chạy độc lập trên cùng push — **CI đỏ không +tự động chặn deploy**, hai workflow không phụ thuộc nhau. -1. xác nhận SHA/image đang chạy đúng bản; -2. kiểm tra `/health` và `/ready`; -3. gửi smoke case qua web, gồm answerable có citation và abstain; -4. quan sát error rate, latency, provider failure và decision distribution; -5. ghi lại thời điểm, SHA và kết quả. +**Thay đổi chart/config** (`infra/helm/**`) — `helm-chart.yml` lint + render + +assert bất biến (Qwen, rerank, TLS, `refute volumeClaimTemplates`...) trên PR. +Merge xong, ArgoCD tự phát hiện và sync — không qua CI nào chạy trên production +thật, review ở PR là cổng chắn duy nhất. -CI và deploy độc lập về kỹ thuật; trạng thái CI đỏ không tự động chặn deploy. +Sau deploy (cả hai loại): + +1. xác nhận Application `Synced`/`Healthy` và image tag/chart revision đúng; +2. gửi smoke case qua web, gồm answerable có citation và abstain; +3. quan sát error rate, latency, provider failure và decision distribution; +4. ghi lại thời điểm, SHA/revision và kết quả. ## Rollback -Workflow `rollback.yml` nhận `target_sha`. Chọn SHA từng deploy thành công và còn -tương thích với database/corpus. Sau rollback phải xác nhận SHA, health/readiness, -smoke cases và metric qua đủ cửa sổ để thấy lỗi ban đầu biến mất. +Không có workflow rollback một-cú-bấm cho k3s hiện tại — đây là phần còn thiếu, +xem `coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md` mục PR C. -Rollback code không tự rollback Qdrant corpus hoặc database migration. Với corpus, -dùng snapshot/migration riêng; không rollback dữ liệu phá huỷ khi chưa có backup. +**Image bị lỗi (phổ biến nhất):** gọi trực tiếp ArgoCD API bằng logic của +`sync_practice_argocd.py` nhưng với `IMAGE_TAG=` — lấy SHA +đó từ lần chạy `build-practice-images.yml` thành công gần nhất trước đó +(`gh run list --workflow=build-practice-images.yml`). Không có nút bấm sẵn cho +việc này; phải chạy script hoặc gọi API thủ công. + +**Chart/config bị lỗi:** `git revert` commit gây lỗi trên `master` qua PR bình +thường; ArgoCD `selfHeal` tự áp bản revert. Muốn ngay lập tức thay vì chờ chu kỳ +poll, sync thủ công qua ArgoCD UI/CLI. + +**Sự cố nặng ở tầng cluster** (k3s tự nó hỏng, không phải lỗi ở app): trong lúc +Compose EC2 (`52.0.158.61`) còn tồn tại và chưa bị tắt, đường lui cuối cùng là +trỏ A record `realvuxbaro.me` về IP đó (TTL 60s) — **chỉ đúng khi Compose đang +chạy bản tương thích với corpus/schema hiện tại**, không phải đường lui mặc +định. Một khi Compose bị dừng/xoá theo quyết định giữ 1 EC2, đường lui này +không còn. + +Không có cơ chế nào ở trên tự rollback Qdrant corpus hay database migration. +Với corpus, dùng snapshot/migration riêng; không rollback dữ liệu phá huỷ khi +chưa có backup. ## Theo dấu request diff --git a/infra/argocd/README.md b/infra/argocd/README.md deleted file mode 100644 index 35e1576..0000000 --- a/infra/argocd/README.md +++ /dev/null @@ -1,36 +0,0 @@ -# ArgoCD (GitOps deployment) - -Deployment uses the **team's existing ArgoCD instance** (not self-hosted by -this project) rather than a custom push-based CD pipeline. See -`docs/adr/0002-argocd-gitops.md` for the rationale. - -## Flow - -1. CI (`infra/ci/github-actions/*-ci.yml`) builds and pushes a container image - per app on merge to master, then bumps that app's image tag in - `infra/helm/medical-chatbot/values-.yaml` (or a per-app values file) - and pushes that commit back to the repo. CI never runs `kubectl apply` or - `helm upgrade` directly. -2. ArgoCD (team-managed, pointed at this repo) watches `infra/argocd/applications//` - and `infra/helm/medical-chatbot/`, detects the values-file change, and - syncs the cluster to match — this is the actual deploy step, owned by - ArgoCD, not by our CI. -3. Promotion between environments (dev -> staging -> prod) is a Git operation - (merge/PR that changes the target values file or image tag for that env), - not a manual `kubectl`/`helm` command. - -## Files - -- `applications/dev/app.yaml`, `applications/staging/app.yaml`, - `applications/prod/app.yaml` — one ArgoCD `Application` CR per environment, - each pointing at this repo + the `infra/helm/medical-chatbot` chart with - that environment's values file. - -## TODO once the team's ArgoCD instance details are known - -- Fill in `spec.destination.server` (target cluster API server / context name) - in each `app.yaml` — currently a placeholder. -- Confirm which ArgoCD `project` (RBAC scoping) these Applications should - belong to, instead of the placeholder `default`. -- Confirm the repo URL placeholder in each `app.yaml` once the GitHub repo - exists (filled in as part of the initial scaffold commit/push). diff --git a/infra/argocd/applications/dev/app.yaml b/infra/argocd/applications/dev/app.yaml deleted file mode 100644 index e45e0de..0000000 --- a/infra/argocd/applications/dev/app.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: argoproj.io/v1alpha1 -kind: Application -metadata: - name: medical-chatbot-dev - namespace: argocd -spec: - project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app - source: - repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created - targetRevision: master - path: infra/helm/medical-chatbot - helm: - valueFiles: - - values.yaml - - values-dev.yaml - destination: - server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context - namespace: medical-chatbot-dev - syncPolicy: - automated: - prune: true - selfHeal: true - syncOptions: - - CreateNamespace=true diff --git a/infra/argocd/applications/prod/app.yaml b/infra/argocd/applications/prod/app.yaml deleted file mode 100644 index 944ae28..0000000 --- a/infra/argocd/applications/prod/app.yaml +++ /dev/null @@ -1,19 +0,0 @@ -apiVersion: argoproj.io/v1alpha1 -kind: Application -metadata: - name: medical-chatbot-prod - namespace: argocd -spec: - project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app - source: - repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created - targetRevision: master - path: infra/helm/medical-chatbot - helm: - valueFiles: - - values.yaml - - values-prod.yaml - destination: - server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context - namespace: medical-chatbot-prod - syncPolicy: {} # intentionally NOT automated — prod sync requires manual approval in the ArgoCD UI/CLI diff --git a/infra/argocd/applications/staging/app.yaml b/infra/argocd/applications/staging/app.yaml deleted file mode 100644 index 5eac9da..0000000 --- a/infra/argocd/applications/staging/app.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: argoproj.io/v1alpha1 -kind: Application -metadata: - name: medical-chatbot-staging - namespace: argocd -spec: - project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app - source: - repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created - targetRevision: master - path: infra/helm/medical-chatbot - helm: - valueFiles: - - values.yaml - - values-staging.yaml - destination: - server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context - namespace: medical-chatbot-staging - syncPolicy: - automated: - prune: true - selfHeal: true - syncOptions: - - CreateNamespace=true diff --git a/infra/helm/medical-chatbot/values-prod.yaml b/infra/helm/medical-chatbot/values-prod.yaml deleted file mode 100644 index 3c2424b..0000000 --- a/infra/helm/medical-chatbot/values-prod.yaml +++ /dev/null @@ -1,61 +0,0 @@ -# Production values for the eventual ArgoCD cutover of realvuxbaro.me. -# -# Not live yet: production still runs Docker Compose on its own EC2, which is -# the DNS-level rollback for the migration. This file is what the production -# ArgoCD Application will render from once the rehearsal gates pass. -# -# The behavioural settings mirror the production runtime contract audited on -# 2026-08-17 — see coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md. - -global: - environment: production - # The GHCR packages are private, same as on the practice cluster. The Secret - # must exist in the target namespace before the first sync. - imagePullSecrets: - - name: ghcr-pull-secret - -aiService: - replicaCount: 2 - image: - repository: ghcr.io/baovu2k4/vsf-duocthu-ai-service - # Deliberately empty. Production must run an immutable, verifiable image, - # so the tag is supplied per deploy as a commit SHA — through the ArgoCD - # Application's inline values, exactly as the practice cluster does. A - # `latest` here would make "which code is production running?" unanswerable - # and would break rollback, since the same tag would point at new content. - # - # `medical-chatbot.image` turns this empty value into a hard render error - # rather than a silent fallback to the chart's `local` development tag. - tag: "" - pullPolicy: Always - config: - embeddingProvider: cohere-v4 - embeddingDimensions: 1024 - evidenceMinimumScore: 0.12 - awsRegion: us-east-1 - answerProvider: bedrock-converse - answerModelId: qwen.qwen3-next-80b-a3b - rerankEnabled: true - otelSampleRatio: 0.25 - -web: - replicaCount: 2 - image: - repository: ghcr.io/baovu2k4/vsf-duocthu-web - tag: "" - pullPolicy: Always - -ingress: - enabled: true - host: realvuxbaro.me - -secret: - create: false - existingSecret: medical-chatbot-prod - -observability: - prometheus: - retention: 15d - -serviceMonitor: - enabled: false diff --git a/infra/helm/medical-chatbot/values-practice-data.yaml b/infra/helm/medical-chatbot/values-production-data.yaml similarity index 79% rename from infra/helm/medical-chatbot/values-practice-data.yaml rename to infra/helm/medical-chatbot/values-production-data.yaml index e2edf8a..cf8cbbf 100644 --- a/infra/helm/medical-chatbot/values-practice-data.yaml +++ b/infra/helm/medical-chatbot/values-production-data.yaml @@ -1,6 +1,6 @@ -# k3s practice cluster — the `medical-chatbot-data` release. +# Production — the `medical-chatbot-data` release (PostgreSQL + Qdrant). # -# Companion to values-practice.yaml. The practice cluster deliberately splits +# Companion to values-production.yaml. The practice cluster deliberately splits # the chart into two ArgoCD Applications so that PostgreSQL and Qdrant survive # every app rollout, prune and self-heal: only this release owns the # PersistentVolumeClaims, so an app-side sync failure can never delete the @@ -13,7 +13,7 @@ # image tag that CI rewrites and no secret, so it is fully tracked in Git. global: - environment: k3s-practice + environment: production aiService: enabled: false diff --git a/infra/helm/medical-chatbot/values-practice.yaml b/infra/helm/medical-chatbot/values-production.yaml similarity index 95% rename from infra/helm/medical-chatbot/values-practice.yaml rename to infra/helm/medical-chatbot/values-production.yaml index 0645864..770af21 100644 --- a/infra/helm/medical-chatbot/values-practice.yaml +++ b/infra/helm/medical-chatbot/values-production.yaml @@ -1,7 +1,7 @@ -# k3s practice cluster — the `medical-chatbot-app` release. +# Production — the `medical-chatbot-app` release serving realvuxbaro.me. # # This file is the Git source of truth for every stable, non-secret setting of -# the practice rehearsal environment. It exists because the same values used to +# this cluster. It exists because the same values used to # live only inside the ArgoCD Application's inline `spec.source.helm.values`, # where they were invisible to review, diff and rollback: the cluster could # drift from the repository without a single commit recording it. That is how @@ -25,7 +25,7 @@ # Practice is only useful as a migration rehearsal while they stay in sync. global: - environment: k3s-practice + environment: production # The GHCR packages are private; without this the Pods fail ImagePullBackOff. imagePullSecrets: - name: ghcr-pull-secret