Stop mislabeling the Compose box as production in read-only audits
This commit is contained in:
@@ -1,4 +1,10 @@
|
|||||||
name: Audit production runtime (read-only)
|
name: Audit Compose rollback box (read-only)
|
||||||
|
|
||||||
|
# `realvuxbaro.me` has run on k3s since the 2026-08-17 cutover; this workflow
|
||||||
|
# still SSHes into secrets.EC2_HOST, which is the retired Compose EC2 kept
|
||||||
|
# only as a manual DNS fallback. Useful for confirming that box is still
|
||||||
|
# healthy and on a known commit before relying on it as a fallback — it does
|
||||||
|
# NOT reflect what real production is currently running.
|
||||||
|
|
||||||
on:
|
on:
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
@@ -14,7 +20,7 @@ jobs:
|
|||||||
audit:
|
audit:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Inspect production over SSH
|
- name: Inspect the Compose rollback box over SSH
|
||||||
uses: appleboy/ssh-action@v1.0.3
|
uses: appleboy/ssh-action@v1.0.3
|
||||||
with:
|
with:
|
||||||
host: ${{ secrets.EC2_HOST }}
|
host: ${{ secrets.EC2_HOST }}
|
||||||
|
|||||||
@@ -1,13 +1,15 @@
|
|||||||
name: Audit production Qdrant corpus (read-only)
|
name: Audit Compose rollback Qdrant corpus (read-only)
|
||||||
|
|
||||||
# Answers "is the practice corpus the same corpus production serves?" with
|
# `realvuxbaro.me` has run on k3s since the 2026-08-17 cutover; this workflow
|
||||||
|
# still SSHes into secrets.EC2_HOST, which is the retired Compose EC2 kept
|
||||||
|
# only as a manual DNS fallback. It fingerprints that box's Qdrant with
|
||||||
# content hashes rather than a point count, which two different corpora can
|
# content hashes rather than a point count, which two different corpora can
|
||||||
# share. The identical script runs against the k3s cluster over SSH, so the
|
# share — the same script can be run against the k3s side (over SSH, or via
|
||||||
# two fingerprints are directly comparable.
|
# `docker exec` on its ai-service pod) to check the fallback still matches
|
||||||
|
# real production before ever relying on it.
|
||||||
#
|
#
|
||||||
# Read-only: it scrolls points and reads collection info. It changes nothing on
|
# Read-only: it scrolls points and reads collection info. It changes nothing,
|
||||||
# production, and its path is not in deploy.yml's filters, so merging it cannot
|
# on this box or any other, and cannot trigger a rebuild of it.
|
||||||
# restart the Compose stack.
|
|
||||||
|
|
||||||
on:
|
on:
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
@@ -30,7 +32,7 @@ jobs:
|
|||||||
- name: Encode fingerprint script
|
- name: Encode fingerprint script
|
||||||
run: echo "SCRIPT_B64=$(base64 -w0 scripts/qdrant_fingerprint.py)" >> "$GITHUB_ENV"
|
run: echo "SCRIPT_B64=$(base64 -w0 scripts/qdrant_fingerprint.py)" >> "$GITHUB_ENV"
|
||||||
|
|
||||||
- name: Fingerprint production corpus over SSH
|
- name: Fingerprint the Compose rollback corpus over SSH
|
||||||
uses: appleboy/ssh-action@v1.0.3
|
uses: appleboy/ssh-action@v1.0.3
|
||||||
env:
|
env:
|
||||||
SCRIPT_B64: ${{ env.SCRIPT_B64 }}
|
SCRIPT_B64: ${{ env.SCRIPT_B64 }}
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
name: Build and sync k3s practice images
|
name: Build and sync k3s images
|
||||||
|
|
||||||
# Practice-cluster only (readytochat.realvuxbaro.me, ArgoCD-managed on the
|
# This IS the production deploy path. `medical-chatbot-app` (ArgoCD, k3s) is
|
||||||
# self-hosted k3s box). Does not touch deploy.yml or the production
|
# the same release behind both realvuxbaro.me and readytochat.realvuxbaro.me
|
||||||
# EC2/Compose stack — production never pulls a GHCR image and isn't
|
# since the 2026-08-17 cutover — there is no longer a separate "practice"
|
||||||
# ArgoCD-managed at all, so this workflow has no path to affect it.
|
# Application this workflow avoids touching. The Compose EC2 is unaffected
|
||||||
|
# only because it has no CI/CD path left at all (deploy.yml/rollback.yml were
|
||||||
|
# removed); it is a manual DNS fallback, not a deploy target.
|
||||||
#
|
#
|
||||||
# ArgoCD's Applications already autosync (syncPolicy.automated) — the gap
|
# ArgoCD's Applications already autosync (syncPolicy.automated) — the gap
|
||||||
# this closes is that the image tag they deploy was a static string
|
# this closes is that the image tag they deploy was a static string
|
||||||
|
|||||||
@@ -1,8 +1,9 @@
|
|||||||
name: CI
|
name: CI
|
||||||
|
|
||||||
# Runs on every push and every pull request. `deploy.yml` triggers
|
# Runs on every push and every pull request. `build-practice-images.yml`
|
||||||
# independently on push to master; until it is made to depend on this job, a
|
# (the k3s/ArgoCD production deploy path) triggers independently on push to
|
||||||
# red CI does NOT block a deploy — see docs/22-ci-cd.md.
|
# master; until it is made to depend on this job, a red CI does NOT block a
|
||||||
|
# deploy — see docs/operations.md.
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
pull_request:
|
pull_request:
|
||||||
|
|||||||
@@ -1,113 +0,0 @@
|
|||||||
name: Deploy to production
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches: [master]
|
|
||||||
paths:
|
|
||||||
# Everything the two Dockerfiles actually COPY into their images.
|
|
||||||
- apps/ai-service/**
|
|
||||||
- apps/web/**
|
|
||||||
- packages/**
|
|
||||||
- ingestion/data/verified/drug_entities.json
|
|
||||||
# docker-compose.prod.yml, docker-compose.observability.yml,
|
|
||||||
# Caddyfile, and the grafana/otel/prometheus/tempo configs it mounts.
|
|
||||||
- infra/docker/**
|
|
||||||
# Changes to the deploy logic itself should still be exercised.
|
|
||||||
- .github/workflows/deploy.yml
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
deploy:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Deploy over SSH
|
|
||||||
uses: appleboy/ssh-action@v1.0.3
|
|
||||||
env:
|
|
||||||
GRAFANA_ADMIN_PASSWORD: ${{ secrets.GRAFANA_ADMIN_PASSWORD }}
|
|
||||||
with:
|
|
||||||
host: ${{ secrets.EC2_HOST }}
|
|
||||||
username: ubuntu
|
|
||||||
key: ${{ secrets.EC2_SSH_KEY }}
|
|
||||||
envs: GRAFANA_ADMIN_PASSWORD
|
|
||||||
script: |
|
|
||||||
set -e
|
|
||||||
test -n "${GRAFANA_ADMIN_PASSWORD:-}"
|
|
||||||
export GRAFANA_ADMIN_PASSWORD
|
|
||||||
cd ~/app
|
|
||||||
git fetch origin master
|
|
||||||
git reset --hard origin/master
|
|
||||||
cd infra/docker
|
|
||||||
sudo -E docker compose \
|
|
||||||
-f docker-compose.prod.yml \
|
|
||||||
-f docker-compose.observability.yml \
|
|
||||||
up -d --build \
|
|
||||||
ai-service web prometheus tempo otel-collector grafana caddy
|
|
||||||
sudo docker exec docker-caddy-1 caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile
|
|
||||||
sudo docker exec docker-caddy-1 caddy reload --config /etc/caddy/Caddyfile --adapter caddyfile
|
|
||||||
sudo docker exec docker-ai-service-1 python -m migrate
|
|
||||||
sleep 10
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/health
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/ready
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null http://web:3000
|
|
||||||
if ! condition_response=$(sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-X POST http://ai-service:8000/v1/rag/query \
|
|
||||||
-H 'Content-Type: application/json' \
|
|
||||||
--data '{"query":"Đợt gout cấp có thuốc nào được Dược thư ghi chỉ định?","subject_scope":"human","intent":"fact_lookup","conversation_id":"deploy-condition-smoke"}'); then
|
|
||||||
sudo docker logs --tail 200 docker-ai-service-1
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
if ! printf '%s' "${condition_response}" | grep -q '"decision":"answerable"'; then
|
|
||||||
printf '%s\n' "${condition_response}"
|
|
||||||
sudo docker logs --tail 200 docker-ai-service-1
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
printf '%s' "${condition_response}" | grep -q '"section_key":"chi_dinh"'
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://prometheus:9090/-/ready
|
|
||||||
for attempt in $(seq 1 12); do
|
|
||||||
if sudo docker run --rm --network docker_default curlimages/curl -sf http://tempo:3200/ready; then
|
|
||||||
break
|
|
||||||
fi
|
|
||||||
if [ "${attempt}" -eq 12 ]; then
|
|
||||||
sudo docker logs --tail 100 docker-tempo-1
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
sleep 5
|
|
||||||
done
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://grafana:3000/api/health
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-u "admin:${GRAFANA_ADMIN_PASSWORD}" \
|
|
||||||
http://grafana:3000/api/datasources/uid/prometheus > /dev/null
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-u "admin:${GRAFANA_ADMIN_PASSWORD}" \
|
|
||||||
http://grafana:3000/api/datasources/uid/tempo > /dev/null
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-u "admin:${GRAFANA_ADMIN_PASSWORD}" \
|
|
||||||
http://grafana:3000/api/dashboards/uid/duocthu-observability \
|
|
||||||
| grep -q '"uid":"duocthu-observability"'
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-o /dev/null https://realvuxbaro.me/grafana/login
|
|
||||||
|
|
||||||
correlation_id="observability-deploy-$(date +%s)"
|
|
||||||
response_headers=$(sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-D - -o /dev/null \
|
|
||||||
-X POST http://ai-service:8000/v1/rag/query \
|
|
||||||
-H 'Content-Type: application/json' \
|
|
||||||
-H "X-Correlation-ID: ${correlation_id}" \
|
|
||||||
--data '{"query":"Paracetamol là thuốc gì?","subject_scope":"unknown","intent":"unknown"}')
|
|
||||||
trace_id=$(printf '%s\n' "${response_headers}" | tr -d '\r' | awk -F ': ' 'tolower($1) == "x-trace-id" { print $2 }' | tail -n 1)
|
|
||||||
printf '%s' "${trace_id}" | grep -Eq '^[0-9a-f]{32}$'
|
|
||||||
|
|
||||||
sleep 20
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sfG \
|
|
||||||
--data-urlencode 'query=duocthu_requests_total' \
|
|
||||||
http://prometheus:9090/api/v1/query | grep -q '"__name__":"duocthu_requests_total"'
|
|
||||||
for attempt in $(seq 1 12); do
|
|
||||||
if sudo docker run --rm --network docker_default curlimages/curl -sf \
|
|
||||||
-o /dev/null "http://tempo:3200/api/traces/${trace_id}"; then
|
|
||||||
break
|
|
||||||
fi
|
|
||||||
if [ "${attempt}" -eq 12 ]; then
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
sleep 5
|
|
||||||
done
|
|
||||||
@@ -23,45 +23,37 @@ jobs:
|
|||||||
version: v3.17.3
|
version: v3.17.3
|
||||||
- name: Lint chart
|
- name: Lint chart
|
||||||
run: helm lint infra/helm/medical-chatbot
|
run: helm lint infra/helm/medical-chatbot
|
||||||
- name: Render default and production manifests
|
- name: Render defaults and check the immutable-tag guard
|
||||||
run: |
|
run: |
|
||||||
helm template default infra/helm/medical-chatbot > /tmp/default.yaml
|
helm template default infra/helm/medical-chatbot > /tmp/default.yaml
|
||||||
|
|
||||||
# values-prod.yaml leaves the image tags empty on purpose: production
|
# The live releases carry no image tag in Git -- it is supplied per
|
||||||
# must run an immutable commit SHA, supplied per deploy. Rendering it
|
# deploy as a commit SHA through the ArgoCD Application. Rendering
|
||||||
# without one has to fail rather than fall back to a development tag,
|
# with an empty tag must FAIL rather than fall back to the chart's
|
||||||
# so assert that failure here — otherwise the guard could rot into a
|
# `local` development tag, so assert the failure directly; otherwise
|
||||||
# silent default and nobody would notice until a cutover.
|
# the guard could rot into a silent default unnoticed.
|
||||||
if helm template production infra/helm/medical-chatbot \
|
if helm template production infra/helm/medical-chatbot --values infra/helm/medical-chatbot/values-production.yaml --set aiService.image.tag="" --set web.image.tag="" > /tmp/untagged.yaml 2>/tmp/untagged.err; then
|
||||||
--values infra/helm/medical-chatbot/values-prod.yaml \
|
echo "::error::render succeeded with no image tag; the immutable-tag guard is gone"
|
||||||
> /tmp/untagged.yaml 2>/tmp/untagged.err; then
|
|
||||||
echo "::error::production render succeeded with no image tag; the immutable-tag guard is gone"
|
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
grep -q 'image.tag must be set to an immutable tag' /tmp/untagged.err
|
grep -q 'image.tag must be set to an immutable tag' /tmp/untagged.err
|
||||||
|
|
||||||
helm template production infra/helm/medical-chatbot \
|
# ...and with a tag it must resolve the GHCR package, not the local
|
||||||
--values infra/helm/medical-chatbot/values-prod.yaml \
|
# development image name.
|
||||||
--set aiService.image.tag="$GITHUB_SHA" \
|
helm template production infra/helm/medical-chatbot --values infra/helm/medical-chatbot/values-production.yaml --set aiService.image.repository=ghcr.io/baovu2k4/vsf-duocthu-ai-service --set web.image.repository=ghcr.io/baovu2k4/vsf-duocthu-web --set aiService.image.tag="$GITHUB_SHA" --set web.image.tag="$GITHUB_SHA" > /tmp/tagged.yaml
|
||||||
--set web.image.tag="$GITHUB_SHA" \
|
grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-ai-service:$GITHUB_SHA\"" /tmp/tagged.yaml
|
||||||
> /tmp/production.yaml
|
grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-web:$GITHUB_SHA\"" /tmp/tagged.yaml
|
||||||
grep -q 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"' /tmp/production.yaml
|
|
||||||
grep -q 'RERANK_ENABLED: "true"' /tmp/production.yaml
|
|
||||||
grep -q 'checksum/runtime-config:' /tmp/production.yaml
|
|
||||||
grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-ai-service:$GITHUB_SHA\"" /tmp/production.yaml
|
|
||||||
grep -q "image: \"ghcr.io/baovu2k4/vsf-duocthu-web:$GITHUB_SHA\"" /tmp/production.yaml
|
|
||||||
|
|
||||||
# The practice cluster is only evidence for the production migration
|
# These two releases are what realvuxbaro.me actually serves, so their
|
||||||
# while it renders the same behavioural contract as production, so both
|
# contract is asserted here rather than trusted by review.
|
||||||
# of its releases are asserted here rather than trusted by review.
|
- name: Render the live production manifests
|
||||||
- name: Render practice manifests
|
|
||||||
run: |
|
run: |
|
||||||
helm template medical-chatbot-app infra/helm/medical-chatbot \
|
helm template medical-chatbot-app infra/helm/medical-chatbot \
|
||||||
--values infra/helm/medical-chatbot/values-practice.yaml \
|
--values infra/helm/medical-chatbot/values-production.yaml \
|
||||||
> /tmp/practice-app.yaml
|
> /tmp/prod-app.yaml
|
||||||
helm template medical-chatbot-data infra/helm/medical-chatbot \
|
helm template medical-chatbot-data infra/helm/medical-chatbot \
|
||||||
--values infra/helm/medical-chatbot/values-practice-data.yaml \
|
--values infra/helm/medical-chatbot/values-production-data.yaml \
|
||||||
> /tmp/practice-data.yaml
|
> /tmp/prod-data.yaml
|
||||||
|
|
||||||
# A bare `grep -q` fails the step with no indication of which
|
# A bare `grep -q` fails the step with no indication of which
|
||||||
# assertion broke, and `set -e` ignores a status inverted with `!`,
|
# assertion broke, and `set -e` ignores a status inverted with `!`,
|
||||||
@@ -83,35 +75,35 @@ jobs:
|
|||||||
}
|
}
|
||||||
|
|
||||||
# Behavioural parity with the audited production runtime contract.
|
# Behavioural parity with the audited production runtime contract.
|
||||||
expect /tmp/practice-app.yaml 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"'
|
expect /tmp/prod-app.yaml 'ANSWER_MODEL_ID: "qwen.qwen3-next-80b-a3b"'
|
||||||
expect /tmp/practice-app.yaml 'ANSWER_PROVIDER: "bedrock-converse"'
|
expect /tmp/prod-app.yaml 'ANSWER_PROVIDER: "bedrock-converse"'
|
||||||
expect /tmp/practice-app.yaml 'EMBEDDING_PROVIDER: "cohere-v4"'
|
expect /tmp/prod-app.yaml 'EMBEDDING_PROVIDER: "cohere-v4"'
|
||||||
expect /tmp/practice-app.yaml 'EMBEDDING_DIMENSIONS: "1024"'
|
expect /tmp/prod-app.yaml 'EMBEDDING_DIMENSIONS: "1024"'
|
||||||
expect /tmp/practice-app.yaml 'EVIDENCE_MINIMUM_SCORE: "0.12"'
|
expect /tmp/prod-app.yaml 'EVIDENCE_MINIMUM_SCORE: "0.12"'
|
||||||
expect /tmp/practice-app.yaml 'RERANK_ENABLED: "true"'
|
expect /tmp/prod-app.yaml 'RERANK_ENABLED: "true"'
|
||||||
expect /tmp/practice-app.yaml 'AWS_REGION: "us-east-1"'
|
expect /tmp/prod-app.yaml 'AWS_REGION: "us-east-1"'
|
||||||
expect /tmp/practice-app.yaml 'checksum/runtime-config:'
|
expect /tmp/prod-app.yaml 'checksum/runtime-config:'
|
||||||
expect /tmp/practice-app.yaml '- host: "readytochat.realvuxbaro.me"'
|
expect /tmp/prod-app.yaml '- host: "readytochat.realvuxbaro.me"'
|
||||||
|
|
||||||
# The production hostname now lives on this cluster, routed and with
|
# The production hostname now lives on this cluster, routed and with
|
||||||
# its own certificate secret -- kept separate from the rehearsal
|
# its own certificate secret -- kept separate from the rehearsal
|
||||||
# hostname's so one renewal failure cannot take both names offline.
|
# hostname's so one renewal failure cannot take both names offline.
|
||||||
expect /tmp/practice-app.yaml '- host: "realvuxbaro.me"'
|
expect /tmp/prod-app.yaml '- host: "realvuxbaro.me"'
|
||||||
expect /tmp/practice-app.yaml 'secretName: realvuxbaro-tls'
|
expect /tmp/prod-app.yaml 'secretName: realvuxbaro-tls'
|
||||||
expect /tmp/practice-app.yaml 'secretName: readytochat-tls'
|
expect /tmp/prod-app.yaml 'secretName: readytochat-tls'
|
||||||
|
|
||||||
# Grafana answers on that same public hostname. Anonymous access may
|
# Grafana answers on that same public hostname. Anonymous access may
|
||||||
# be open, but never as Admin, never with the login form disabled,
|
# be open, but never as Admin, never with the login form disabled,
|
||||||
# and its root URL must be the name users actually arrive on.
|
# and its root URL must be the name users actually arrive on.
|
||||||
expect /tmp/practice-app.yaml 'value: "https://realvuxbaro.me/grafana/"'
|
expect /tmp/prod-app.yaml 'value: "https://realvuxbaro.me/grafana/"'
|
||||||
refute /tmp/practice-app.yaml 'value: "Admin"'
|
refute /tmp/prod-app.yaml 'value: "Admin"'
|
||||||
|
|
||||||
# grep is line-oriented, so read the value on the line after each
|
# grep is line-oriented, so read the value on the line after each
|
||||||
# flag rather than trying to match the pair as one pattern.
|
# flag rather than trying to match the pair as one pattern.
|
||||||
for check in "GF_AUTH_ANONYMOUS_ORG_ROLE:Viewer" "GF_AUTH_DISABLE_LOGIN_FORM:false"; do
|
for check in "GF_AUTH_ANONYMOUS_ORG_ROLE:Viewer" "GF_AUTH_DISABLE_LOGIN_FORM:false"; do
|
||||||
flag=${check%%:*}
|
flag=${check%%:*}
|
||||||
want=${check#*:}
|
want=${check#*:}
|
||||||
got=$(grep -A1 -- "$flag" /tmp/practice-app.yaml | grep -- 'value:' | tr -d ' "' | cut -d: -f2)
|
got=$(grep -A1 -- "$flag" /tmp/prod-app.yaml | grep -- 'value:' | tr -d ' "' | cut -d: -f2)
|
||||||
if [ "$got" != "$want" ]; then
|
if [ "$got" != "$want" ]; then
|
||||||
echo "::error::$flag rendered as '$got', expected '$want'"
|
echo "::error::$flag rendered as '$got', expected '$want'"
|
||||||
exit 1
|
exit 1
|
||||||
@@ -123,10 +115,10 @@ jobs:
|
|||||||
# prune can never delete the corpus or the query history. Only those
|
# prune can never delete the corpus or the query history. Only those
|
||||||
# two use volumeClaimTemplates — the observability PVCs are the app
|
# two use volumeClaimTemplates — the observability PVCs are the app
|
||||||
# release's own and are expected here.
|
# release's own and are expected here.
|
||||||
refute /tmp/practice-app.yaml 'volumeClaimTemplates'
|
refute /tmp/prod-app.yaml 'volumeClaimTemplates'
|
||||||
expect /tmp/practice-app.yaml 'medical-chatbot-data-medical-chatbot-qdrant'
|
expect /tmp/prod-app.yaml 'medical-chatbot-data-medical-chatbot-qdrant'
|
||||||
|
|
||||||
# ...and the data release must own nothing else.
|
# ...and the data release must own nothing else.
|
||||||
refute /tmp/practice-data.yaml 'medical-chatbot-data-medical-chatbot-ai-service'
|
refute /tmp/prod-data.yaml 'medical-chatbot-data-medical-chatbot-ai-service'
|
||||||
refute /tmp/practice-data.yaml 'kind: Ingress'
|
refute /tmp/prod-data.yaml 'kind: Ingress'
|
||||||
expect /tmp/practice-data.yaml 'volumeClaimTemplates'
|
expect /tmp/prod-data.yaml 'volumeClaimTemplates'
|
||||||
|
|||||||
@@ -1,66 +0,0 @@
|
|||||||
name: Migrate Qdrant snapshot to practice cluster
|
|
||||||
|
|
||||||
# One-off, manual (workflow_dispatch only) bridge: snapshots the production
|
|
||||||
# Qdrant collection (a live, non-disruptive Qdrant operation — this is how
|
|
||||||
# the original prod migration was done, just in reverse) and relays the
|
|
||||||
# snapshot files to the isolated k3s practice EC2. Uses the SAME EC2_SSH_KEY
|
|
||||||
# deploy.yml already has (never exposed to the operator) plus a new
|
|
||||||
# PRACTICE_SSH_KEY scoped only to the practice box. Delete this workflow
|
|
||||||
# file once the one-time migration is done — it is not part of the regular
|
|
||||||
# deploy path.
|
|
||||||
|
|
||||||
on:
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
migrate:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Set up SSH key
|
|
||||||
run: |
|
|
||||||
mkdir -p ~/.ssh
|
|
||||||
printf '%s\n' "${{ secrets.EC2_SSH_KEY }}" > ~/.ssh/prod.pem
|
|
||||||
chmod 600 ~/.ssh/prod.pem
|
|
||||||
ssh-keyscan -H "${{ secrets.EC2_HOST }}" >> ~/.ssh/known_hosts 2>/dev/null
|
|
||||||
|
|
||||||
- name: Report production Qdrant version
|
|
||||||
run: |
|
|
||||||
ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" '
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://qdrant:6333/
|
|
||||||
'
|
|
||||||
|
|
||||||
- name: Snapshot Qdrant collections on production
|
|
||||||
run: |
|
|
||||||
ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" '
|
|
||||||
set -e
|
|
||||||
snap() {
|
|
||||||
collection="$1"; outfile="$2"
|
|
||||||
name=$(sudo docker run --rm --network docker_default curlimages/curl -sf -X POST "http://qdrant:6333/collections/${collection}/snapshots" | grep -oP "\"name\":\"\K[^\"]+")
|
|
||||||
test -n "$name"
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf "http://qdrant:6333/collections/${collection}/snapshots/${name}" > "$outfile"
|
|
||||||
}
|
|
||||||
snap duocthu_v1 /tmp/duocthu_v1.snapshot
|
|
||||||
snap duocthu_v1__manifest /tmp/duocthu_v1__manifest.snapshot
|
|
||||||
sudo chown ubuntu:ubuntu /tmp/duocthu_v1.snapshot /tmp/duocthu_v1__manifest.snapshot
|
|
||||||
ls -la /tmp/*.snapshot
|
|
||||||
'
|
|
||||||
|
|
||||||
- name: Pull snapshots to the runner
|
|
||||||
run: |
|
|
||||||
scp -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}:/tmp/duocthu_v1.snapshot" ./duocthu_v1.snapshot
|
|
||||||
scp -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}:/tmp/duocthu_v1__manifest.snapshot" ./duocthu_v1__manifest.snapshot
|
|
||||||
ls -la ./*.snapshot
|
|
||||||
|
|
||||||
- name: Upload snapshots as a workflow artifact
|
|
||||||
uses: actions/upload-artifact@v4
|
|
||||||
with:
|
|
||||||
name: qdrant-snapshots
|
|
||||||
path: |
|
|
||||||
duocthu_v1.snapshot
|
|
||||||
duocthu_v1__manifest.snapshot
|
|
||||||
retention-days: 1
|
|
||||||
|
|
||||||
- name: Clean up temp files on production
|
|
||||||
if: always()
|
|
||||||
run: |
|
|
||||||
ssh -i ~/.ssh/prod.pem "ubuntu@${{ secrets.EC2_HOST }}" 'rm -f /tmp/duocthu_v1.snapshot /tmp/duocthu_v1__manifest.snapshot' || true
|
|
||||||
@@ -1,58 +0,0 @@
|
|||||||
name: Rollback production
|
|
||||||
|
|
||||||
# Manual escape hatch for deploy.yml. deploy.yml has NO automatic rollback:
|
|
||||||
# it runs `git reset --hard origin/master`, rebuilds and runs migrations
|
|
||||||
# BEFORE its health checks, so a deploy that fails those checks leaves the
|
|
||||||
# server on the bad commit with no automatic recovery. This workflow points
|
|
||||||
# the same reset+rebuild+health-check sequence at an earlier commit instead.
|
|
||||||
#
|
|
||||||
# Migrations are forward-only (apps/ai-service/migrate.py, no down scripts)
|
|
||||||
# but every migration so far uses IF NOT EXISTS / ADD COLUMN IF NOT EXISTS,
|
|
||||||
# so re-running them against an older commit is a no-op, not an error. A
|
|
||||||
# future non-idempotent migration would break this guarantee.
|
|
||||||
|
|
||||||
on:
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
target_sha:
|
|
||||||
description: "Commit SHA or tag to roll back to (e.g. the last known-good commit from a previous successful 'Deploy to production' run)"
|
|
||||||
required: true
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
rollback:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Rollback over SSH
|
|
||||||
uses: appleboy/ssh-action@v1.0.3
|
|
||||||
env:
|
|
||||||
GRAFANA_ADMIN_PASSWORD: ${{ secrets.GRAFANA_ADMIN_PASSWORD }}
|
|
||||||
TARGET_SHA: ${{ inputs.target_sha }}
|
|
||||||
with:
|
|
||||||
host: ${{ secrets.EC2_HOST }}
|
|
||||||
username: ubuntu
|
|
||||||
key: ${{ secrets.EC2_SSH_KEY }}
|
|
||||||
envs: GRAFANA_ADMIN_PASSWORD,TARGET_SHA
|
|
||||||
script: |
|
|
||||||
set -e
|
|
||||||
test -n "${GRAFANA_ADMIN_PASSWORD:-}"
|
|
||||||
export GRAFANA_ADMIN_PASSWORD
|
|
||||||
cd ~/app
|
|
||||||
git fetch origin
|
|
||||||
git rev-parse --verify "${TARGET_SHA}^{commit}"
|
|
||||||
git reset --hard "${TARGET_SHA}"
|
|
||||||
echo "Rolled back to $(git rev-parse HEAD) — $(git log -1 --format=%s)"
|
|
||||||
cd infra/docker
|
|
||||||
sudo -E docker compose \
|
|
||||||
-f docker-compose.prod.yml \
|
|
||||||
-f docker-compose.observability.yml \
|
|
||||||
up -d --build \
|
|
||||||
ai-service web prometheus tempo otel-collector grafana caddy
|
|
||||||
sudo docker exec docker-caddy-1 caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile
|
|
||||||
sudo docker exec docker-caddy-1 caddy reload --config /etc/caddy/Caddyfile --adapter caddyfile
|
|
||||||
sudo docker exec docker-ai-service-1 python -m migrate
|
|
||||||
sleep 10
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/health
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/ready
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null http://web:3000
|
|
||||||
sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null https://realvuxbaro.me/grafana/login
|
|
||||||
echo "Rollback to ${TARGET_SHA} verified healthy."
|
|
||||||
@@ -461,16 +461,95 @@ certificate for that name — the SNI handshake fails with an internal error, an
|
|||||||
the certificate it does hold is apex-only. Decide whether to point `www` at k3s
|
the certificate it does hold is apex-only. Decide whether to point `www` at k3s
|
||||||
and add it to `tls`, or drop the record.
|
and add it to `tls`, or drop the record.
|
||||||
|
|
||||||
|
## Grafana exposure closed (2026-08-17, post-cutover)
|
||||||
|
|
||||||
|
The cutover moved production onto the cluster whose Grafana ran
|
||||||
|
`GF_AUTH_ANONYMOUS_ORG_ROLE=Admin` with `GF_AUTH_DISABLE_LOGIN_FORM=true` —
|
||||||
|
defensible on a throwaway rehearsal box, not on a public production hostname.
|
||||||
|
Confirmed live before the fix: `/grafana/api/org` and `/grafana/api/datasources`
|
||||||
|
both answered HTTP 200 with no credentials, listing internal service URLs.
|
||||||
|
|
||||||
|
Fixed at `39b1159` plus one out-of-band step:
|
||||||
|
|
||||||
|
- Anonymous access kept, but demoted to `Viewer`, so dashboards stay open for a
|
||||||
|
demo while the datasource and dashboard write APIs are refused (verified:
|
||||||
|
`POST /api/datasources` → 403).
|
||||||
|
- `GF_AUTH_DISABLE_LOGIN_FORM` is now hardcoded `false`. Combined with
|
||||||
|
anonymous Admin it had previously left no way to sign in as a real admin.
|
||||||
|
- `ingress.host` is now `realvuxbaro.me`, with the rehearsal name moved to
|
||||||
|
`extraHosts`. Grafana builds `GF_SERVER_ROOT_URL` from `ingress.host`, so it
|
||||||
|
had continued advertising the rehearsal hostname after the cutover.
|
||||||
|
- CI now asserts the rendered role is `Viewer`, the login form is not disabled,
|
||||||
|
and the root URL is the production hostname.
|
||||||
|
|
||||||
|
**Setting the admin password through Helm was not enough.** Grafana persists
|
||||||
|
its user table in SQLite on a PVC, so `GF_SECURITY_ADMIN_PASSWORD` did not
|
||||||
|
overwrite the existing credential: after the rollout, `admin:change-me` still
|
||||||
|
authenticated and the new password did not. It required
|
||||||
|
`grafana cli admin reset-admin-password --password-from-stdin` inside the pod.
|
||||||
|
Anyone rotating this password later must do the same — changing the Secret
|
||||||
|
alone is silently ineffective.
|
||||||
|
|
||||||
|
Final state verified: anonymous → `/api/admin/settings` 403; admin with the new
|
||||||
|
password 200; `admin:change-me` 403; anonymous dashboard search 200; app 200.
|
||||||
|
The password lives in the gitignored `.env.k3s-practice` as
|
||||||
|
`GRAFANA_ADMIN_PASSWORD` and in the ArgoCD Application's inline values. It is
|
||||||
|
not in Git.
|
||||||
|
|
||||||
|
## Repository now describes what is actually live
|
||||||
|
|
||||||
|
`values-practice.yaml` / `values-practice-data.yaml` became
|
||||||
|
`values-production.yaml` / `values-production-data.yaml`, and
|
||||||
|
`global.environment` is now `production` (confirmed in the running ConfigMap).
|
||||||
|
The separate, never-deployed `values-prod.yaml` is deleted — two "prod" files
|
||||||
|
beside a "practice" file that was the real one was the worst of both.
|
||||||
|
|
||||||
|
Done in three commits so no sync ever referenced a missing path: add the new
|
||||||
|
files (`527d3a8`), repoint both Applications and confirm Synced/Healthy, then
|
||||||
|
delete the old ones (`6468b16`). The ai-service Pod rolled cleanly on the
|
||||||
|
environment change and the site stayed up throughout.
|
||||||
|
|
||||||
|
The immutable-tag guard moved onto the file that is actually live and now
|
||||||
|
checks both directions: an empty tag must fail the render, and a supplied SHA
|
||||||
|
must resolve the GHCR package rather than the chart's local development name.
|
||||||
|
|
||||||
|
## Live chat driven directly against production
|
||||||
|
|
||||||
|
Not a golden-set replay — unaccented text, a typo, missing dimensions and
|
||||||
|
pronoun follow-ups within one conversation, through the real `/api/chat`:
|
||||||
|
|
||||||
|
| Input | Behaviour |
|
||||||
|
| --- | --- |
|
||||||
|
| `lieu paracetamol` | `clarify` — asks adult or child, offers both as quick replies; answers with a citation once told `nguoi lon` |
|
||||||
|
| `paracetamol cho tre em` → `be 3 tuoi` | `clarify` twice: age, then still demands weight. The paediatric gate requiring **both** is intact |
|
||||||
|
| `metformim co tac dung phu gi` | Resolves the typo and answers metformin's GI adverse effects with a citation |
|
||||||
|
| `amlodipin dung the nao` → `the con chong chi dinh thi sao` | Carries the referent across turns and answers the contraindication |
|
||||||
|
| `gia thuoc paracetamol ... bao nhieu tien` | Abstains cleanly: the formulary does not carry prices |
|
||||||
|
|
||||||
|
Cosmetic only: the paediatric follow-up echoes the user's unaccented text back
|
||||||
|
inside an otherwise accented sentence ("Bé 3 tuoi nặng bao nhiêu kg?").
|
||||||
|
|
||||||
|
## Final verified state
|
||||||
|
|
||||||
|
`https://realvuxbaro.me/` 200 from `44.206.194.195` with a valid certificate;
|
||||||
|
live query `answerable` in 14.3 s with a citation; Grafana anonymous read 200
|
||||||
|
but admin 403, admin login 200; `readytochat.realvuxbaro.me` 200;
|
||||||
|
`argocd.realvuxbaro.me` 200; and the Compose host still answers the apex with
|
||||||
|
a valid certificate, so the rollback is a 60-second A-record revert.
|
||||||
|
|
||||||
|
Temporary SSH allowances on `sg-018fc3cde8282f26d` were revoked; only the
|
||||||
|
pre-existing `103.238.70.200/32` remains on port 22. Note that this
|
||||||
|
workstation's SSH egress was `101.99.23.84`, so re-opening will be needed for
|
||||||
|
future cluster access — and the three IP-reporting services disagree, so read
|
||||||
|
the real one from the host's own `auth.log` rather than trusting any of them.
|
||||||
|
|
||||||
## Migration risks currently open
|
## Migration risks currently open
|
||||||
|
|
||||||
1. `www.realvuxbaro.me` is broken (see above) and always was.
|
1. `www.realvuxbaro.me` is broken (see above) and always was.
|
||||||
2. Secrets, resource limits and failure recovery still need explicit rehearsal
|
2. Secrets, resource limits and failure recovery still need explicit rehearsal
|
||||||
gates.
|
gates.
|
||||||
3. Answer-content parity between environments is still unverified — decisions
|
3. The Compose EC2 is still running as the rollback. Stop — never terminate —
|
||||||
and latency match, answer length does not always.
|
once the acceptance window closes.
|
||||||
4. The cluster still serves production from values files named `values-practice*`.
|
|
||||||
Rename once the acceptance window closes and the Compose host is retired;
|
|
||||||
`values-prod.yaml` is now unused by anything live.
|
|
||||||
3. Cluster-level inspection is currently blocked from this machine: the
|
3. Cluster-level inspection is currently blocked from this machine: the
|
||||||
practice security group scopes port 6443 to the operator's own outbound IP,
|
practice security group scopes port 6443 to the operator's own outbound IP,
|
||||||
which has changed again, so `kubectl` hangs. ArgoCD's API and the public
|
which has changed again, so `kubectl` hangs. ArgoCD's API and the public
|
||||||
|
|||||||
@@ -0,0 +1,204 @@
|
|||||||
|
> **Decision update, same day, later session:** owner confirmed intent to run
|
||||||
|
> **one EC2 only** (k3s). This supersedes PR A step 2 below (reduce to
|
||||||
|
> `workflow_dispatch`) — `deploy.yml`, `rollback.yml`, and the never-applied
|
||||||
|
> `infra/argocd/applications/**` scaffold (wrong Application names, dead
|
||||||
|
> `values-prod.yaml` reference, placeholder team-repo TODOs) were deleted
|
||||||
|
> outright on branch `agent/retire-compose-cicd`, not disabled. D1/D2/D3/D4 are
|
||||||
|
> closed by deletion rather than by fixing the probe. `docs/operations.md`
|
||||||
|
> Deploy/Rollback sections rewritten to describe the actual k3s/ArgoCD path.
|
||||||
|
> Compose EC2 itself (`52.0.158.61`) is untouched pending an explicit stop/
|
||||||
|
> terminate decision (PR D) — do not stop or terminate it without that go-ahead.
|
||||||
|
> PR B (Helm hygiene) and PR C (rollback runbook, now mostly covered in
|
||||||
|
> `docs/operations.md`) remain open.
|
||||||
|
|
||||||
|
# Plan — make the CI/CD path safe after the k3s cutover (2026-08-18)
|
||||||
|
|
||||||
|
Written by Claude (Opus) for execution by another agent. Every claim below was
|
||||||
|
verified on 2026-08-18 by the command shown next to it. **Re-verify before
|
||||||
|
acting** — runtime state can change after this snapshot.
|
||||||
|
|
||||||
|
## 1. The topology changed and two workflows never noticed
|
||||||
|
|
||||||
|
This is the root cause of everything in this plan. The 2026-08-17 cutover moved
|
||||||
|
`realvuxbaro.me` from the Compose EC2 to the k3s/ArgoCD cluster, but the
|
||||||
|
Compose-era workflows still describe and probe the old world.
|
||||||
|
|
||||||
|
| Fact | Verified by |
|
||||||
|
| --- | --- |
|
||||||
|
| `realvuxbaro.me` → `44.206.194.195` (k3s) | `nslookup realvuxbaro.me 8.8.8.8` |
|
||||||
|
| `readytochat.realvuxbaro.me` → `44.206.194.195` — **same cluster, same release** | same |
|
||||||
|
| Compose EC2 `52.0.158.61` is off DNS; rollback target only | `values-production.yaml` `ingress.host` + DNS above |
|
||||||
|
| Compose is still at `df57e6b` | `git log df57e6b..52e8828 -- <deploy.yml paths>` returns empty |
|
||||||
|
| App code at `df57e6b` == app code at `master` | the intervening commits touch only docs/CI/coordination |
|
||||||
|
|
||||||
|
So: **the Compose rollback is currently valid**, and the window to fix this is
|
||||||
|
now, before the first post-cutover app change lands.
|
||||||
|
|
||||||
|
`values-production.yaml` sets `ingress.host: realvuxbaro.me` and is rendered by
|
||||||
|
ArgoCD Application `medical-chatbot-app` — the same Application
|
||||||
|
`.github/scripts/sync_practice_argocd.py` repoints (`APP_NAME =
|
||||||
|
"medical-chatbot-app"`). **"Practice" and production are the same release.**
|
||||||
|
|
||||||
|
## 2. Confirmed defects, most dangerous first
|
||||||
|
|
||||||
|
### D1 — both Compose workflows verify the wrong machine (P0)
|
||||||
|
|
||||||
|
`deploy.yml:88` and `rollback.yml:57` both end with:
|
||||||
|
|
||||||
|
```
|
||||||
|
docker run --rm --network docker_default curlimages/curl -sf \
|
||||||
|
-o /dev/null https://realvuxbaro.me/grafana/login
|
||||||
|
```
|
||||||
|
|
||||||
|
That hostname now resolves to **k3s**, not the box the workflow just rebuilt.
|
||||||
|
The check passes by hitting a completely different server.
|
||||||
|
|
||||||
|
This is worst in `rollback.yml`, whose whole purpose is to be trustworthy in an
|
||||||
|
emergency: it prints `Rollback to <sha> verified healthy` on the strength of a
|
||||||
|
probe that never touched the rolled-back box. A broken rollback would report
|
||||||
|
success.
|
||||||
|
|
||||||
|
### D2 — one bad commit poisons production *and* the rollback (P0)
|
||||||
|
|
||||||
|
`deploy.yml` and `build-practice-images.yml` trigger on the **same four paths**:
|
||||||
|
`apps/ai-service/**`, `apps/web/**`, `packages/**`,
|
||||||
|
`ingestion/data/verified/drug_entities.json`.
|
||||||
|
|
||||||
|
A single push to `master` therefore rolls production forward on k3s **and**
|
||||||
|
rebuilds the Compose box that is supposed to be the known-good fallback. The
|
||||||
|
"proven rollback" only holds while Compose stays on a good commit.
|
||||||
|
|
||||||
|
### D3 — `build-practice-images.yml` header comment is now false (P1)
|
||||||
|
|
||||||
|
> *"Does not touch deploy.yml or the production EC2/Compose stack — production
|
||||||
|
> never pulls a GHCR image and isn't ArgoCD-managed at all, so this workflow
|
||||||
|
> has no path to affect it."*
|
||||||
|
|
||||||
|
Production **is** ArgoCD-managed and **does** pull GHCR images. This workflow is
|
||||||
|
the production deploy pipeline. The comment invites exactly the push that breaks
|
||||||
|
production, and its name reinforces the error.
|
||||||
|
|
||||||
|
### D4 — a dead ArgoCD manifest sits in Git (P1)
|
||||||
|
|
||||||
|
`infra/argocd/applications/prod/app.yaml` lists
|
||||||
|
`valueFiles: [values.yaml, values-prod.yaml]`, but `values-prod.yaml` was
|
||||||
|
deleted in `6468b16`. It also declares `syncPolicy: {}` with a comment claiming
|
||||||
|
prod sync needs manual approval — the live Application is automated. Applying
|
||||||
|
this file would fail or deploy something wrong. `dev/` and `staging/` under the
|
||||||
|
same directory have not been checked and may share the defect.
|
||||||
|
|
||||||
|
### D5 — the WIP Helm change shadows an existing variable (P2)
|
||||||
|
|
||||||
|
Uncommitted in the main worktree. The chart **already** emits `AWS_REGION` in
|
||||||
|
the ConfigMap from `aiService.config.awsRegion`
|
||||||
|
(`templates/ai-service.yaml:15`, asserted in `helm-chart.yml`). The WIP diff
|
||||||
|
adds a second `AWS_REGION` as a container `env:` entry — and in Kubernetes an
|
||||||
|
explicit `env:` **overrides** `envFrom`, so enabling it would silently shadow
|
||||||
|
the ConfigMap. Inert today only because `aws.region` defaults to `""`.
|
||||||
|
|
||||||
|
The static-credentials half is a genuine gap and worth keeping.
|
||||||
|
|
||||||
|
### D6 — `helm-chart.yml` asserts invariants but never diffs (P2)
|
||||||
|
|
||||||
|
It renders both live releases and checks a strong list (Qwen, rerank, TLS
|
||||||
|
secrets, Grafana role, `refute volumeClaimTemplates`). But nothing compares the
|
||||||
|
render against the previous commit, so a chart change that alters anything
|
||||||
|
*outside* that list reaches production silently.
|
||||||
|
|
||||||
|
### D7 — production has no rollback workflow at all (P2)
|
||||||
|
|
||||||
|
`rollback.yml` targets `secrets.EC2_HOST` — the Compose box. Nothing rolls back
|
||||||
|
k3s. Real production rollback today is a manual Namecheap A-record revert
|
||||||
|
(~60s TTL) or an ArgoCD revision/tag revert, neither written down.
|
||||||
|
|
||||||
|
## 3. Execution plan
|
||||||
|
|
||||||
|
### Ground rules
|
||||||
|
|
||||||
|
- **Never push to `master`.** Every change goes through a PR. ArgoCD auto-syncs
|
||||||
|
`master` with `selfHeal` + `prune`; a merge touching `infra/helm/**` applies
|
||||||
|
to production with no human gate.
|
||||||
|
- `ci.yml` runs on every PR (ruff, pytest, ingestion tests, web lint+build).
|
||||||
|
`helm-chart.yml` runs on PRs touching `infra/helm/**`. Both must be green.
|
||||||
|
- Do not touch `git.vinmec.tech`, team ArgoCD/k3s, or team repos.
|
||||||
|
- Do not start, stop, or terminate any EC2 instance without an explicit go.
|
||||||
|
- Do not edit files in the other worktrees (`D:\VSF-DUOCTHU-codex-*`,
|
||||||
|
`D:\VSF-DUOCTHU-claude-gitops`); check `coordination/` for active claims and
|
||||||
|
file your own claim before starting.
|
||||||
|
|
||||||
|
### PR A — workflow safety (D1, D2, D3, D4)
|
||||||
|
|
||||||
|
Touches only `.github/**` and `infra/argocd/**`. ArgoCD renders
|
||||||
|
`infra/helm/medical-chatbot`, so **this PR cannot alter production manifests**.
|
||||||
|
Confirm that rather than assume it.
|
||||||
|
|
||||||
|
1. **Fix the misdirected probes (D1).** In `deploy.yml` and `rollback.yml`,
|
||||||
|
make the Caddy/Grafana check target the box being deployed instead of a
|
||||||
|
public DNS name that now points elsewhere — e.g. resolve the hostname to the
|
||||||
|
local Caddy container so TLS and routing are still exercised.
|
||||||
|
**Verification gate:** prove the fixed check *fails* when Caddy is broken.
|
||||||
|
A probe that cannot fail is the defect being fixed, not a fix. (See the
|
||||||
|
`set -e` / `! grep` silent-pass traps already documented in
|
||||||
|
`helm-chart.yml`.)
|
||||||
|
2. **Stop `deploy.yml` firing on push (D2).** Reduce it to `workflow_dispatch`
|
||||||
|
only. Compose then stays pinned at `df57e6b` — a stable rollback rather than
|
||||||
|
one that tracks `master`. State the trade-off in the commit message: the
|
||||||
|
fallback stops drifting, but also stops receiving fixes, so it goes stale as
|
||||||
|
production moves. That is acceptable for a time-boxed acceptance window and
|
||||||
|
is the subject of PR D.
|
||||||
|
3. **Tell the truth in `build-practice-images.yml` (D3).** Replace the false
|
||||||
|
header comment. Renaming the workflow to name it as the production deploy
|
||||||
|
path is preferable — **trap:** it self-references in its own `paths:` filter
|
||||||
|
and the sync script path, so both must be updated together or the workflow
|
||||||
|
silently stops triggering.
|
||||||
|
4. **Remove or correct `infra/argocd/applications/prod/app.yaml` (D4).** Check
|
||||||
|
`dev/` and `staging/` in the same directory for the same rot. Deleting is
|
||||||
|
fine if nothing applies them; verify that first.
|
||||||
|
|
||||||
|
### PR B — Helm hygiene (D5, D6)
|
||||||
|
|
||||||
|
Touches `infra/helm/**`, so merging **does** reach production. Highest care.
|
||||||
|
|
||||||
|
5. **Drop the redundant `aws.region` block from the WIP diff (D5)**; keep the
|
||||||
|
static-credentials support, still defaulted off. Guard against a nil `aws`
|
||||||
|
key so a values file that omits it cannot break the render — a template
|
||||||
|
error here means ArgoCD cannot sync production at all.
|
||||||
|
6. **Add a baseline render diff to `helm-chart.yml` (D6).** Render
|
||||||
|
`values-production.yaml` and `values-production-data.yaml` at the PR base and
|
||||||
|
at HEAD, then surface the diff in the job summary. The goal is that no chart
|
||||||
|
change ever reaches production without a human having seen exactly what it
|
||||||
|
does to the manifests.
|
||||||
|
**Verification gate:** the diff for PR B itself must be **empty** — the
|
||||||
|
static-credentials change is defaults-off and must render byte-identically.
|
||||||
|
If it is not empty, stop and explain why before merging.
|
||||||
|
|
||||||
|
Note: `helm` is **not installed** on this workstation (`helm: command not found`
|
||||||
|
in both bash and PowerShell). Either install it or rely on the CI render — but
|
||||||
|
do not claim the render is unchanged without one of the two actually running it.
|
||||||
|
|
||||||
|
### PR C — document the real rollback (D7)
|
||||||
|
|
||||||
|
7. Write the production rollback runbook: revert the ArgoCD Application to the
|
||||||
|
previous image tag, and/or revert the `realvuxbaro.me` A record to
|
||||||
|
`52.0.158.61`. Note that the Namecheap edit is a **manual owner step** — a
|
||||||
|
harness permission classifier has blocked agent form input on that page
|
||||||
|
before, so the runbook must not assume an agent can do it.
|
||||||
|
|
||||||
|
### PR D — decide the Compose lifecycle (owner call, not an agent call)
|
||||||
|
|
||||||
|
8. Compose is a second `t3.large` running purely as a fallback. Once the
|
||||||
|
acceptance window closes, ArgoCD's own revision history covers rollback and
|
||||||
|
the instance is redundant. Surface the choice and the monthly cost; **do not
|
||||||
|
act on it without an explicit go.**
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
`Feature-List-AI-Duoc-thu-V1.md`, `presentation/`, and `.claude/skills/` are
|
||||||
|
untracked and match no workflow trigger path. Committing them is inert and can
|
||||||
|
be a separate trivial commit — keep it out of PRs A–D.
|
||||||
|
|
||||||
|
## 4. Order and why
|
||||||
|
|
||||||
|
D1 first: an untrustworthy rollback is worse than no rollback, because it fails
|
||||||
|
silently at the moment of maximum pressure. D2 next: it is the defect that would
|
||||||
|
consume the rollback. Everything after is hardening.
|
||||||
+51
-16
@@ -1,32 +1,67 @@
|
|||||||
# Vận hành, triển khai và xử lý sự cố
|
# Vận hành, triển khai và xử lý sự cố
|
||||||
|
|
||||||
> Loại chính: How-to
|
> Loại chính: How-to
|
||||||
> Phạm vi: EC2 + Docker Compose hiện hành
|
> Phạm vi: k3s + ArgoCD (production kể từ cutover 2026-08-17)
|
||||||
|
|
||||||
|
Production (`realvuxbaro.me`) chạy trên k3s, quản lý bởi ArgoCD Application
|
||||||
|
**`medical-chatbot-app`** (ai-service + web + observability) và
|
||||||
|
**`medical-chatbot-data`** (PostgreSQL + Qdrant, tách release để prune/self-heal
|
||||||
|
phía app không bao giờ đụng vào dữ liệu). Cả hai đặt `syncPolicy.automated` với
|
||||||
|
`selfHeal` + `prune` — **mọi merge vào `master` áp thẳng vào production, không
|
||||||
|
có cổng duyệt thủ công.** EC2 Docker Compose (`52.0.158.61`) không còn nhận
|
||||||
|
deploy tự động; xem `coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md` cho
|
||||||
|
lý do và tình trạng hiện tại của máy đó.
|
||||||
|
|
||||||
## Deploy
|
## Deploy
|
||||||
|
|
||||||
Trước deploy, ghi commit SHA, yêu cầu CI AI/ingestion/web xanh, kiểm tra secret và
|
Hai loại thay đổi đi hai đường khác nhau:
|
||||||
Qdrant manifest tương thích, đồng thời đánh giá migration. Chạy `deploy.yml` theo
|
|
||||||
path/branch filter hoặc manual dispatch và theo dõi đến khi reconcile xong.
|
|
||||||
|
|
||||||
Sau deploy:
|
**Thay đổi code app** (`apps/ai-service/**`, `apps/web/**`, `packages/**`,
|
||||||
|
`ingestion/data/verified/drug_entities.json`) — merge vào `master` kích hoạt
|
||||||
|
`build-practice-images.yml`: build + push image GHCR gắn tag theo commit SHA,
|
||||||
|
sau đó `.github/scripts/sync_practice_argocd.py` ghi tag mới vào Application
|
||||||
|
`medical-chatbot-app` và gọi sync. Workflow tự xác nhận
|
||||||
|
`readytochat.realvuxbaro.me` đã lên bản mới trước khi báo thành công.
|
||||||
|
`ci.yml` (ruff/pytest/lint/build) chạy độc lập trên cùng push — **CI đỏ không
|
||||||
|
tự động chặn deploy**, hai workflow không phụ thuộc nhau.
|
||||||
|
|
||||||
1. xác nhận SHA/image đang chạy đúng bản;
|
**Thay đổi chart/config** (`infra/helm/**`) — `helm-chart.yml` lint + render +
|
||||||
2. kiểm tra `/health` và `/ready`;
|
assert bất biến (Qwen, rerank, TLS, `refute volumeClaimTemplates`...) trên PR.
|
||||||
3. gửi smoke case qua web, gồm answerable có citation và abstain;
|
Merge xong, ArgoCD tự phát hiện và sync — không qua CI nào chạy trên production
|
||||||
4. quan sát error rate, latency, provider failure và decision distribution;
|
thật, review ở PR là cổng chắn duy nhất.
|
||||||
5. ghi lại thời điểm, SHA và kết quả.
|
|
||||||
|
|
||||||
CI và deploy độc lập về kỹ thuật; trạng thái CI đỏ không tự động chặn deploy.
|
Sau deploy (cả hai loại):
|
||||||
|
|
||||||
|
1. xác nhận Application `Synced`/`Healthy` và image tag/chart revision đúng;
|
||||||
|
2. gửi smoke case qua web, gồm answerable có citation và abstain;
|
||||||
|
3. quan sát error rate, latency, provider failure và decision distribution;
|
||||||
|
4. ghi lại thời điểm, SHA/revision và kết quả.
|
||||||
|
|
||||||
## Rollback
|
## Rollback
|
||||||
|
|
||||||
Workflow `rollback.yml` nhận `target_sha`. Chọn SHA từng deploy thành công và còn
|
Không có workflow rollback một-cú-bấm cho k3s hiện tại — đây là phần còn thiếu,
|
||||||
tương thích với database/corpus. Sau rollback phải xác nhận SHA, health/readiness,
|
xem `coordination/CLAUDE_PLAN_CICD_SAFETY_2026-08-18.md` mục PR C.
|
||||||
smoke cases và metric qua đủ cửa sổ để thấy lỗi ban đầu biến mất.
|
|
||||||
|
|
||||||
Rollback code không tự rollback Qdrant corpus hoặc database migration. Với corpus,
|
**Image bị lỗi (phổ biến nhất):** gọi trực tiếp ArgoCD API bằng logic của
|
||||||
dùng snapshot/migration riêng; không rollback dữ liệu phá huỷ khi chưa có backup.
|
`sync_practice_argocd.py` nhưng với `IMAGE_TAG=<sha tốt lần trước>` — lấy SHA
|
||||||
|
đó từ lần chạy `build-practice-images.yml` thành công gần nhất trước đó
|
||||||
|
(`gh run list --workflow=build-practice-images.yml`). Không có nút bấm sẵn cho
|
||||||
|
việc này; phải chạy script hoặc gọi API thủ công.
|
||||||
|
|
||||||
|
**Chart/config bị lỗi:** `git revert` commit gây lỗi trên `master` qua PR bình
|
||||||
|
thường; ArgoCD `selfHeal` tự áp bản revert. Muốn ngay lập tức thay vì chờ chu kỳ
|
||||||
|
poll, sync thủ công qua ArgoCD UI/CLI.
|
||||||
|
|
||||||
|
**Sự cố nặng ở tầng cluster** (k3s tự nó hỏng, không phải lỗi ở app): trong lúc
|
||||||
|
Compose EC2 (`52.0.158.61`) còn tồn tại và chưa bị tắt, đường lui cuối cùng là
|
||||||
|
trỏ A record `realvuxbaro.me` về IP đó (TTL 60s) — **chỉ đúng khi Compose đang
|
||||||
|
chạy bản tương thích với corpus/schema hiện tại**, không phải đường lui mặc
|
||||||
|
định. Một khi Compose bị dừng/xoá theo quyết định giữ 1 EC2, đường lui này
|
||||||
|
không còn.
|
||||||
|
|
||||||
|
Không có cơ chế nào ở trên tự rollback Qdrant corpus hay database migration.
|
||||||
|
Với corpus, dùng snapshot/migration riêng; không rollback dữ liệu phá huỷ khi
|
||||||
|
chưa có backup.
|
||||||
|
|
||||||
## Theo dấu request
|
## Theo dấu request
|
||||||
|
|
||||||
|
|||||||
@@ -1,36 +0,0 @@
|
|||||||
# ArgoCD (GitOps deployment)
|
|
||||||
|
|
||||||
Deployment uses the **team's existing ArgoCD instance** (not self-hosted by
|
|
||||||
this project) rather than a custom push-based CD pipeline. See
|
|
||||||
`docs/adr/0002-argocd-gitops.md` for the rationale.
|
|
||||||
|
|
||||||
## Flow
|
|
||||||
|
|
||||||
1. CI (`infra/ci/github-actions/*-ci.yml`) builds and pushes a container image
|
|
||||||
per app on merge to master, then bumps that app's image tag in
|
|
||||||
`infra/helm/medical-chatbot/values-<env>.yaml` (or a per-app values file)
|
|
||||||
and pushes that commit back to the repo. CI never runs `kubectl apply` or
|
|
||||||
`helm upgrade` directly.
|
|
||||||
2. ArgoCD (team-managed, pointed at this repo) watches `infra/argocd/applications/<env>/`
|
|
||||||
and `infra/helm/medical-chatbot/`, detects the values-file change, and
|
|
||||||
syncs the cluster to match — this is the actual deploy step, owned by
|
|
||||||
ArgoCD, not by our CI.
|
|
||||||
3. Promotion between environments (dev -> staging -> prod) is a Git operation
|
|
||||||
(merge/PR that changes the target values file or image tag for that env),
|
|
||||||
not a manual `kubectl`/`helm` command.
|
|
||||||
|
|
||||||
## Files
|
|
||||||
|
|
||||||
- `applications/dev/app.yaml`, `applications/staging/app.yaml`,
|
|
||||||
`applications/prod/app.yaml` — one ArgoCD `Application` CR per environment,
|
|
||||||
each pointing at this repo + the `infra/helm/medical-chatbot` chart with
|
|
||||||
that environment's values file.
|
|
||||||
|
|
||||||
## TODO once the team's ArgoCD instance details are known
|
|
||||||
|
|
||||||
- Fill in `spec.destination.server` (target cluster API server / context name)
|
|
||||||
in each `app.yaml` — currently a placeholder.
|
|
||||||
- Confirm which ArgoCD `project` (RBAC scoping) these Applications should
|
|
||||||
belong to, instead of the placeholder `default`.
|
|
||||||
- Confirm the repo URL placeholder in each `app.yaml` once the GitHub repo
|
|
||||||
exists (filled in as part of the initial scaffold commit/push).
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
apiVersion: argoproj.io/v1alpha1
|
|
||||||
kind: Application
|
|
||||||
metadata:
|
|
||||||
name: medical-chatbot-dev
|
|
||||||
namespace: argocd
|
|
||||||
spec:
|
|
||||||
project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app
|
|
||||||
source:
|
|
||||||
repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created
|
|
||||||
targetRevision: master
|
|
||||||
path: infra/helm/medical-chatbot
|
|
||||||
helm:
|
|
||||||
valueFiles:
|
|
||||||
- values.yaml
|
|
||||||
- values-dev.yaml
|
|
||||||
destination:
|
|
||||||
server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context
|
|
||||||
namespace: medical-chatbot-dev
|
|
||||||
syncPolicy:
|
|
||||||
automated:
|
|
||||||
prune: true
|
|
||||||
selfHeal: true
|
|
||||||
syncOptions:
|
|
||||||
- CreateNamespace=true
|
|
||||||
@@ -1,19 +0,0 @@
|
|||||||
apiVersion: argoproj.io/v1alpha1
|
|
||||||
kind: Application
|
|
||||||
metadata:
|
|
||||||
name: medical-chatbot-prod
|
|
||||||
namespace: argocd
|
|
||||||
spec:
|
|
||||||
project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app
|
|
||||||
source:
|
|
||||||
repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created
|
|
||||||
targetRevision: master
|
|
||||||
path: infra/helm/medical-chatbot
|
|
||||||
helm:
|
|
||||||
valueFiles:
|
|
||||||
- values.yaml
|
|
||||||
- values-prod.yaml
|
|
||||||
destination:
|
|
||||||
server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context
|
|
||||||
namespace: medical-chatbot-prod
|
|
||||||
syncPolicy: {} # intentionally NOT automated — prod sync requires manual approval in the ArgoCD UI/CLI
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
apiVersion: argoproj.io/v1alpha1
|
|
||||||
kind: Application
|
|
||||||
metadata:
|
|
||||||
name: medical-chatbot-staging
|
|
||||||
namespace: argocd
|
|
||||||
spec:
|
|
||||||
project: default # TODO: confirm the team's ArgoCD project/RBAC scope for this app
|
|
||||||
source:
|
|
||||||
repoURL: https://github.com/BaoVu2k4/vsf-duocthu.git # TODO: confirm once repo is created
|
|
||||||
targetRevision: master
|
|
||||||
path: infra/helm/medical-chatbot
|
|
||||||
helm:
|
|
||||||
valueFiles:
|
|
||||||
- values.yaml
|
|
||||||
- values-staging.yaml
|
|
||||||
destination:
|
|
||||||
server: https://kubernetes.default.svc # TODO: point at the team's target cluster/context
|
|
||||||
namespace: medical-chatbot-staging
|
|
||||||
syncPolicy:
|
|
||||||
automated:
|
|
||||||
prune: true
|
|
||||||
selfHeal: true
|
|
||||||
syncOptions:
|
|
||||||
- CreateNamespace=true
|
|
||||||
@@ -1,61 +0,0 @@
|
|||||||
# Production values for the eventual ArgoCD cutover of realvuxbaro.me.
|
|
||||||
#
|
|
||||||
# Not live yet: production still runs Docker Compose on its own EC2, which is
|
|
||||||
# the DNS-level rollback for the migration. This file is what the production
|
|
||||||
# ArgoCD Application will render from once the rehearsal gates pass.
|
|
||||||
#
|
|
||||||
# The behavioural settings mirror the production runtime contract audited on
|
|
||||||
# 2026-08-17 — see coordination/ARGOCD_PRODUCTION_MIGRATION_STATE_2026-08-17.md.
|
|
||||||
|
|
||||||
global:
|
|
||||||
environment: production
|
|
||||||
# The GHCR packages are private, same as on the practice cluster. The Secret
|
|
||||||
# must exist in the target namespace before the first sync.
|
|
||||||
imagePullSecrets:
|
|
||||||
- name: ghcr-pull-secret
|
|
||||||
|
|
||||||
aiService:
|
|
||||||
replicaCount: 2
|
|
||||||
image:
|
|
||||||
repository: ghcr.io/baovu2k4/vsf-duocthu-ai-service
|
|
||||||
# Deliberately empty. Production must run an immutable, verifiable image,
|
|
||||||
# so the tag is supplied per deploy as a commit SHA — through the ArgoCD
|
|
||||||
# Application's inline values, exactly as the practice cluster does. A
|
|
||||||
# `latest` here would make "which code is production running?" unanswerable
|
|
||||||
# and would break rollback, since the same tag would point at new content.
|
|
||||||
#
|
|
||||||
# `medical-chatbot.image` turns this empty value into a hard render error
|
|
||||||
# rather than a silent fallback to the chart's `local` development tag.
|
|
||||||
tag: ""
|
|
||||||
pullPolicy: Always
|
|
||||||
config:
|
|
||||||
embeddingProvider: cohere-v4
|
|
||||||
embeddingDimensions: 1024
|
|
||||||
evidenceMinimumScore: 0.12
|
|
||||||
awsRegion: us-east-1
|
|
||||||
answerProvider: bedrock-converse
|
|
||||||
answerModelId: qwen.qwen3-next-80b-a3b
|
|
||||||
rerankEnabled: true
|
|
||||||
otelSampleRatio: 0.25
|
|
||||||
|
|
||||||
web:
|
|
||||||
replicaCount: 2
|
|
||||||
image:
|
|
||||||
repository: ghcr.io/baovu2k4/vsf-duocthu-web
|
|
||||||
tag: ""
|
|
||||||
pullPolicy: Always
|
|
||||||
|
|
||||||
ingress:
|
|
||||||
enabled: true
|
|
||||||
host: realvuxbaro.me
|
|
||||||
|
|
||||||
secret:
|
|
||||||
create: false
|
|
||||||
existingSecret: medical-chatbot-prod
|
|
||||||
|
|
||||||
observability:
|
|
||||||
prometheus:
|
|
||||||
retention: 15d
|
|
||||||
|
|
||||||
serviceMonitor:
|
|
||||||
enabled: false
|
|
||||||
+3
-3
@@ -1,6 +1,6 @@
|
|||||||
# k3s practice cluster — the `medical-chatbot-data` release.
|
# Production — the `medical-chatbot-data` release (PostgreSQL + Qdrant).
|
||||||
#
|
#
|
||||||
# Companion to values-practice.yaml. The practice cluster deliberately splits
|
# Companion to values-production.yaml. The practice cluster deliberately splits
|
||||||
# the chart into two ArgoCD Applications so that PostgreSQL and Qdrant survive
|
# the chart into two ArgoCD Applications so that PostgreSQL and Qdrant survive
|
||||||
# every app rollout, prune and self-heal: only this release owns the
|
# every app rollout, prune and self-heal: only this release owns the
|
||||||
# PersistentVolumeClaims, so an app-side sync failure can never delete the
|
# PersistentVolumeClaims, so an app-side sync failure can never delete the
|
||||||
@@ -13,7 +13,7 @@
|
|||||||
# image tag that CI rewrites and no secret, so it is fully tracked in Git.
|
# image tag that CI rewrites and no secret, so it is fully tracked in Git.
|
||||||
|
|
||||||
global:
|
global:
|
||||||
environment: k3s-practice
|
environment: production
|
||||||
|
|
||||||
aiService:
|
aiService:
|
||||||
enabled: false
|
enabled: false
|
||||||
+3
-3
@@ -1,7 +1,7 @@
|
|||||||
# k3s practice cluster — the `medical-chatbot-app` release.
|
# Production — the `medical-chatbot-app` release serving realvuxbaro.me.
|
||||||
#
|
#
|
||||||
# This file is the Git source of truth for every stable, non-secret setting of
|
# This file is the Git source of truth for every stable, non-secret setting of
|
||||||
# the practice rehearsal environment. It exists because the same values used to
|
# this cluster. It exists because the same values used to
|
||||||
# live only inside the ArgoCD Application's inline `spec.source.helm.values`,
|
# live only inside the ArgoCD Application's inline `spec.source.helm.values`,
|
||||||
# where they were invisible to review, diff and rollback: the cluster could
|
# where they were invisible to review, diff and rollback: the cluster could
|
||||||
# drift from the repository without a single commit recording it. That is how
|
# drift from the repository without a single commit recording it. That is how
|
||||||
@@ -25,7 +25,7 @@
|
|||||||
# Practice is only useful as a migration rehearsal while they stay in sync.
|
# Practice is only useful as a migration rehearsal while they stay in sync.
|
||||||
|
|
||||||
global:
|
global:
|
||||||
environment: k3s-practice
|
environment: production
|
||||||
# The GHCR packages are private; without this the Pods fail ImagePullBackOff.
|
# The GHCR packages are private; without this the Pods fail ImagePullBackOff.
|
||||||
imagePullSecrets:
|
imagePullSecrets:
|
||||||
- name: ghcr-pull-secret
|
- name: ghcr-pull-secret
|
||||||
Reference in New Issue
Block a user