Verify exact production traces and record rollout

This commit is contained in:
2026-08-11 11:29:59 +07:00
parent 6b8f7584ed
commit 59e6ad2d0d
46 changed files with 2795 additions and 290 deletions
+53 -2
View File
@@ -3,6 +3,7 @@ name: Deploy to production
on:
push:
branches: [master]
workflow_dispatch:
jobs:
deploy:
@@ -10,18 +11,68 @@ jobs:
steps:
- name: Deploy over SSH
uses: appleboy/ssh-action@v1.0.3
env:
GRAFANA_ADMIN_PASSWORD: ${{ secrets.GRAFANA_ADMIN_PASSWORD }}
with:
host: ${{ secrets.EC2_HOST }}
username: ubuntu
key: ${{ secrets.EC2_SSH_KEY }}
envs: GRAFANA_ADMIN_PASSWORD
script: |
set -e
test -n "${GRAFANA_ADMIN_PASSWORD:-}"
export GRAFANA_ADMIN_PASSWORD
cd ~/app
git fetch origin master
git reset --hard origin/master
cd infra/docker
sudo docker compose -f docker-compose.prod.yml up -d --build ai-service web
sudo -E docker compose \
-f docker-compose.prod.yml \
-f docker-compose.observability.yml \
up -d --build \
ai-service web prometheus tempo otel-collector grafana
sudo docker exec docker-ai-service-1 python -m migrate
sleep 5
sleep 10
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/health
sudo docker run --rm --network docker_default curlimages/curl -sf http://ai-service:8000/ready
sudo docker run --rm --network docker_default curlimages/curl -sf -o /dev/null http://web:3000
sudo docker run --rm --network docker_default curlimages/curl -sf http://prometheus:9090/-/ready
for attempt in $(seq 1 12); do
if sudo docker run --rm --network docker_default curlimages/curl -sf http://tempo:3200/ready; then
break
fi
if [ "${attempt}" -eq 12 ]; then
sudo docker logs --tail 100 docker-tempo-1
exit 1
fi
sleep 5
done
sudo docker run --rm --network docker_default curlimages/curl -sf http://grafana:3000/api/health
sudo docker run --rm --network docker_default curlimages/curl -sf \
-u "admin:${GRAFANA_ADMIN_PASSWORD}" \
http://grafana:3000/api/datasources/uid/prometheus > /dev/null
correlation_id="observability-deploy-$(date +%s)"
response_headers=$(sudo docker run --rm --network docker_default curlimages/curl -sf \
-D - -o /dev/null \
-X POST http://ai-service:8000/v1/rag/query \
-H 'Content-Type: application/json' \
-H "X-Correlation-ID: ${correlation_id}" \
--data '{"query":"Paracetamol là thuốc gì?","subject_scope":"unknown","intent":"unknown"}')
trace_id=$(printf '%s\n' "${response_headers}" | tr -d '\r' | awk -F ': ' 'tolower($1) == "x-trace-id" { print $2 }' | tail -n 1)
printf '%s' "${trace_id}" | grep -Eq '^[0-9a-f]{32}$'
sleep 20
sudo docker run --rm --network docker_default curlimages/curl -sfG \
--data-urlencode 'query=duocthu_requests_total' \
http://prometheus:9090/api/v1/query | grep -q '"__name__":"duocthu_requests_total"'
for attempt in $(seq 1 12); do
if sudo docker run --rm --network docker_default curlimages/curl -sf \
-o /dev/null "http://tempo:3200/api/traces/${trace_id}"; then
break
fi
if [ "${attempt}" -eq 12 ]; then
exit 1
fi
sleep 5
done
+81 -2
View File
@@ -30,8 +30,9 @@ today.
> | `apps/web/` | Done — chat UI with citation/evidence panel |
> | `apps/api-gateway`, `auth-service`, `user-service`, `chat-service` | **Not built** — `README.md` + `package.json` only |
> | `apps/mobile/` | **Not built** — reserved |
> | `infra/docker/` | Done — this is what production actually runs |
> | `infra/k8s`, `helm`, `terraform`, `argocd` | **Not built yet** — empty scaffold. Still the target (ADR 0002), not abandoned: the plan is the team's self-hosted Gitea + ArgoCD; the current EC2/Compose setup is an interim stopgap |
> | `infra/docker/` | Done — production runs Compose, including the Prometheus/Grafana/Tempo observability overlay |
> | `infra/helm/medical-chatbot/` | Built and validated as an offline migration kit; **not deployed** to Docker Desktop, k3s or ArgoCD |
> | `infra/k8s`, `terraform`, `argocd` | **Not built yet** — still the target (ADR 0002), not abandoned: the plan is the team's self-hosted Gitea + ArgoCD; the current EC2/Compose setup is an interim stopgap |
>
> Because the gateway and auth services do not exist, `apps/web` talks
> **directly** to `apps/ai-service`; there is no authentication layer. See
@@ -96,6 +97,84 @@ Tests: `cd apps/ai-service && python -m pytest -q` — 230 pass. `test_api.py` a
with `--ignore` when the stack is down. `apps/web` has **no test setup at all**,
so a green suite says nothing about the frontend — drive it in a browser.
## Observability: Prometheus, Grafana and Tempo
The observability stack is provisioned in the repository and has been deployed
to the production EC2 instance since 2026-08-11.
- **Prometheus** scrapes `/metrics` from `ai-service`. It records request rate
and latency, latency for each RAG stage, routing decisions and reasons,
provider failures, trace-write failures and the existing domain counters.
- **Grafana** is the user interface for dashboards and metric queries. Its
datasource and the Dược Thư dashboard are provisioned automatically.
- **Tempo** stores OpenTelemetry traces. A trace contains the receive,
understanding, routing, retrieval, rerank/evidence, generation,
grounding/entailment, persistence and response stages. Correlation and trace
IDs follow the request from the Next.js BFF into FastAPI.
- **OpenTelemetry Collector** receives spans from `ai-service` and exports them
to Tempo. Grafana exemplars link aggregate latency metrics to an individual
Tempo trace.
For answer lineage, use the three views together:
1. The web citation/evidence panel shows which source chunks, pages and exact
evidence text were selected for the answer.
2. **Grafana -> Explore -> Tempo** shows which pipeline stages ran, their
nesting and timing, the final decision/reason, provider failures and the
persisted trace ID.
3. PostgreSQL table `rag_retrieval_trace` is the durable audit record. It stores
the query, resolved drug, decision/reason, selected citations/evidence,
correlation ID and OpenTelemetry trace ID, so a returned `trace_id` can be
joined to its Tempo trace.
This is provenance and execution tracing, not model chain-of-thought logging.
Full prompts/responses, hidden reasoning, every rejected retrieval candidate
and every ranking score are deliberately not stored today. If deeper debugging
is needed, add bounded audit fields rather than putting sensitive prompt or
patient content into metric labels or span names.
Start the local stack from the repository root:
```powershell
docker compose -f infra\docker\docker-compose.yml up -d prometheus tempo otel-collector grafana
```
Local endpoints:
| Service | Address | Use |
|---|---|---|
| Grafana | `http://localhost:3002` | Dashboards and Explore |
| Prometheus | `http://localhost:9090` | Raw targets, PromQL and metrics |
| Tempo | `http://localhost:3200` | Trace backend; normally queried through Grafana |
| ai-service metrics | `http://localhost:8079/metrics` | Raw OpenMetrics output when ai-service runs on port 8079 |
For the existing EC2 Compose deployment, the optional overlay is
`infra/docker/docker-compose.observability.yml`. It leaves
`docker-compose.prod.yml` unchanged. A deployment, when explicitly approved,
uses both files:
```powershell
docker compose `
-f infra/docker/docker-compose.prod.yml `
-f infra/docker/docker-compose.observability.yml `
up -d
```
Only Grafana is mapped to the EC2 host (`3002:3000`) by the production overlay;
Prometheus and Tempo stay on the internal Compose network. The EC2 security
group does not expose port 3002 publicly. View Grafana through an SSH tunnel:
```powershell
ssh -L 3002:127.0.0.1:3002 <ssh-user>@52.0.158.61
```
Keep that session open and visit `http://localhost:3002`. Set
`GRAFANA_ADMIN_USER` and `GRAFANA_ADMIN_PASSWORD` in the production environment
before deployment; do not use the fallback password in production. Prometheus
metrics are available in **Grafana -> Explore -> Prometheus**. To investigate a
slow request, open the request-latency panel, follow its exemplar/trace link, or
paste the returned `X-Trace-ID` into **Explore -> Tempo**.
## Production
Live at [realvuxbaro.me](https://realvuxbaro.me): a single EC2 `t3.large`
+3
View File
@@ -26,6 +26,9 @@ RUN pip install --no-cache-dir \
"qdrant-client>=1.7,<2" \
"uvicorn[standard]>=0.30,<1" \
"prometheus-client>=0.20,<1" \
"opentelemetry-api>=1.27,<2" \
"opentelemetry-sdk>=1.27,<2" \
"opentelemetry-exporter-otlp-proto-http>=1.27,<2" \
"anthropic>=0.112,<1" \
"boto3"
+12 -4
View File
@@ -18,6 +18,8 @@ class RetrievalTrace:
reason: str
resolved_drug_id: str | None
citations: tuple[dict[str, Any], ...]
correlation_id: str | None = None
otel_trace_id: str | None = None
created_at: datetime | None = None
@@ -55,6 +57,8 @@ class PostgresTraceRepository:
reason: str,
resolved_drug_id: str | None,
citations: tuple[dict[str, Any], ...],
correlation_id: str | None = None,
otel_trace_id: str | None = None,
) -> str:
import psycopg
@@ -64,12 +68,14 @@ class PostgresTraceRepository:
"""
INSERT INTO rag_retrieval_trace (
trace_id, query_text, subject_scope, query_intent,
decision, reason, resolved_drug_id, citations
) VALUES (%s, %s, %s, %s, %s, %s, %s, %s::jsonb)
decision, reason, resolved_drug_id, citations,
correlation_id, otel_trace_id
) VALUES (%s, %s, %s, %s, %s, %s, %s, %s::jsonb, %s, %s)
""",
(
trace_id, query, subject_scope, intent, decision, reason,
resolved_drug_id, json.dumps(citations, ensure_ascii=False),
correlation_id, otel_trace_id,
),
)
return trace_id
@@ -81,7 +87,8 @@ class PostgresTraceRepository:
row = connection.execute(
"""
SELECT trace_id::text, query_text, subject_scope, query_intent,
decision, reason, resolved_drug_id, citations, created_at
decision, reason, resolved_drug_id, citations,
correlation_id, otel_trace_id, created_at
FROM rag_retrieval_trace WHERE trace_id = %s
""",
(trace_id,),
@@ -91,7 +98,8 @@ class PostgresTraceRepository:
return RetrievalTrace(
trace_id=row[0], query=row[1], subject_scope=row[2], intent=row[3],
decision=row[4], reason=row[5], resolved_drug_id=row[6],
citations=tuple(row[7]), created_at=row[8],
citations=tuple(row[7]), correlation_id=row[8], otel_trace_id=row[9],
created_at=row[10],
)
+189 -32
View File
@@ -1,11 +1,9 @@
"""Exports the domain's counters; the only module that names prometheus_client.
"""Prometheus export for bounded domain and request-path telemetry.
`rag/metrics.py` defines what is counted and why. This decides how it leaves
the process, and is imported lazily so the service runs — and the suite passes
— with no metrics stack installed.
Counter names carry a `duocthu_` prefix and a `_total` suffix because that is
what Prometheus expects of a counter; the dashboard queries them by name.
The domain owns metric names in :mod:`rag.metrics`; this adapter owns label
vocabularies, buckets and OpenMetrics exposition. Unknown label values are
collapsed to ``other`` so a query, drug id, exception message or raw URL can
never accidentally create an unbounded time series.
"""
from __future__ import annotations
@@ -14,44 +12,192 @@ from typing import Any
from rag.metrics import (
ABSTENTION,
ANSWER_EXTRACTIVE,
CLARIFY_ASKED,
DECISION,
FOLLOWUP_INHERITED,
GENERATION_REJECTED,
GENERATION_SERVED,
LOOP_REFINED,
LOOP_REPAIRED,
LOOP_ROUNDS,
PROVIDER_FAILURE,
REQUEST_DURATION,
REQUESTS,
RETRIEVAL_ROUTE,
STAGE_DURATION,
TRACE_WRITE_FAILED,
)
_LABELS: dict[str, tuple[str, ...]] = {
_COUNTER_LABELS: dict[str, tuple[str, ...]] = {
ABSTENTION: ("reason",),
GENERATION_REJECTED: ("reason",),
RETRIEVAL_ROUTE: ("route",),
GENERATION_SERVED: (),
ANSWER_EXTRACTIVE: (),
CLARIFY_ASKED: ("reason",),
DECISION: ("decision", "reason"),
FOLLOWUP_INHERITED: (),
GENERATION_REJECTED: ("reason",),
GENERATION_SERVED: (),
LOOP_REFINED: (),
LOOP_REPAIRED: (),
LOOP_ROUNDS: (),
PROVIDER_FAILURE: ("provider", "operation", "reason"),
REQUESTS: ("method", "route", "status"),
RETRIEVAL_ROUTE: ("route",),
TRACE_WRITE_FAILED: (),
}
_HISTOGRAM_LABELS: dict[str, tuple[str, ...]] = {
REQUEST_DURATION: ("method", "route", "status"),
STAGE_DURATION: ("stage", "outcome"),
}
_HELP = {
ABSTENTION: "Answers refused, by the reason retrieval gave.",
GENERATION_REJECTED: (
"Generations discarded before reaching the caller. `reason=\"ungrounded_number\"` "
"counts answers that stated a figure absent from the cited source; "
"`reason=\"uncited_claim\"` counts claims with no valid citation at all; "
"`reason=\"unsupported_claim\"` counts claims the entailment pass judged "
"not actually stated by the block they cite."
),
GENERATION_SERVED: "Generations that passed grounding verification and were served.",
ABSTENTION: "Answers refused, by bounded domain reason.",
ANSWER_EXTRACTIVE: "Answers served as verbatim source text.",
RETRIEVAL_ROUTE: "Retrievals by route: section filter, or similarity fallback.",
CLARIFY_ASKED: "Clarifying questions returned instead of guessing.",
DECISION: "Final request decisions by bounded domain reason.",
FOLLOWUP_INHERITED: "Follow-up turns that inherited prior context.",
GENERATION_REJECTED: "Generated answers discarded by a safety or availability gate.",
GENERATION_SERVED: "Generated answers that passed grounding and entailment.",
LOOP_REFINED: "Conversational retrieval loops that refined a query.",
LOOP_REPAIRED: "Conversational retrieval loops that repaired an answer.",
LOOP_ROUNDS: "Retrieval loop rounds completed.",
PROVIDER_FAILURE: "Failures from bounded external-provider operations.",
REQUESTS: "HTTP requests by route template and status class.",
REQUEST_DURATION: "End-to-end HTTP request latency in seconds.",
RETRIEVAL_ROUTE: "Retrievals by bounded route.",
STAGE_DURATION: "RAG stage latency in seconds.",
TRACE_WRITE_FAILED: "Final trace rows that could not be persisted to PostgreSQL.",
}
_REQUEST_BUCKETS = (0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 20, 40, 60)
_STAGE_BUCKETS = (0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 20, 40)
_ALLOWED: dict[str, frozenset[str]] = {
"decision": frozenset({"answerable", "clarify", "verify_pdf", "abstain", "error"}),
"method": frozenset({"GET", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"}),
"operation": frozenset(
{"understand", "embed", "retrieve", "rerank", "generate", "entailment", "persist"}
),
"outcome": frozenset({"ok", "error", "cancelled"}),
"provider": frozenset(
{"bedrock_converse", "bedrock_claude", "bedrock_cohere", "qdrant", "postgres"}
),
"route": frozenset(
{
"/health",
"/ready",
"/metrics",
"/v1/rag/query",
"/v1/rag/suggest",
"section",
"overview",
"similarity",
"indication",
"other",
}
),
"stage": frozenset(
{
"receive",
"context",
"understanding",
"routing",
"retrieval",
"rerank",
"evidence",
"generation",
"grounding",
"entailment",
"persistence",
"response",
}
),
"status": frozenset({"1xx", "2xx", "3xx", "4xx", "5xx"}),
}
_REASONS = frozenset(
{
"clarify_loop_exhausted",
"drug_not_in_formulary",
"drug_not_resolved",
"drug_resolution_ambiguous",
"drug_resolution_invalid_state",
"evidence_insufficient",
"error",
"generation_unavailable",
"grounded_evidence_available",
"incomplete_answer",
"insufficient_retrieval_score",
"invalid_citation",
"malformed_output",
"missing_attribute",
"missing_indication",
"missing_pediatric_age_or_weight",
"missing_population",
"missing_printed_page_provenance",
"missing_provenance",
"missing_query_or_drug",
"needs_more_info",
"no_drug",
"no_indication",
"no_indication_match",
"no_interaction_evidence",
"out_of_scope",
"out_of_scope_non_human",
"parent_hydration_failed",
"provider_unavailable",
"query_embedding_unavailable",
"query_intent_unknown",
"request_budget_exhausted",
"smalltalk",
"subject_scope_unknown",
"timeout",
"uncited_claim",
"ungrounded_number",
"unsupported_claim",
"visual_verification_required",
}
)
def _bounded(label: str, value: str) -> str:
allowed = _REASONS if label == "reason" else _ALLOWED.get(label)
if allowed is None:
return value
return value if value in allowed else "other"
def _exemplar() -> dict[str, str] | None:
# Lazy import avoids making OpenTelemetry a hard requirement for metrics.
try:
from rag.telemetry import current_trace_id
trace_id = current_trace_id()
except ImportError:
trace_id = None
return {"trace_id": trace_id} if trace_id else None
class PrometheusMetrics:
"""Domain `Metrics` backed by a Prometheus registry."""
"""Domain ``Metrics`` backed by an isolated Prometheus registry."""
def __init__(self, registry: Any | None = None) -> None:
from prometheus_client import CollectorRegistry, Counter
from prometheus_client import CollectorRegistry, Counter, Histogram
self._registry = registry or CollectorRegistry()
self._counters = {
name: Counter(name, _HELP[name], labels, registry=self._registry)
for name, labels in _LABELS.items()
for name, labels in _COUNTER_LABELS.items()
}
self._histograms = {
name: Histogram(
name,
_HELP[name],
labels,
buckets=_REQUEST_BUCKETS if name == REQUEST_DURATION else _STAGE_BUCKETS,
registry=self._registry,
)
for name, labels in _HISTOGRAM_LABELS.items()
}
@property
@@ -60,17 +206,28 @@ class PrometheusMetrics:
def increment(self, name: str, **labels: str) -> None:
counter = self._counters.get(name)
if counter is None:
expected = _COUNTER_LABELS.get(name)
if counter is None or expected is None or set(labels) != set(expected):
return
# An unexpected label would raise at scrape time, far from its cause.
# Metrics must not be able to break a clinical answer, so a mismatch
# drops the sample rather than the request.
expected = set(_LABELS[name])
if set(labels) != expected:
safe = {key: _bounded(key, labels[key]) for key in expected}
child = counter.labels(**safe) if safe else counter
child.inc(exemplar=_exemplar())
def observe(self, name: str, value: float, **labels: str) -> None:
histogram = self._histograms.get(name)
expected = _HISTOGRAM_LABELS.get(name)
if histogram is None or expected is None or set(labels) != set(expected):
return
(counter.labels(**labels) if labels else counter).inc()
safe = {key: _bounded(key, labels[key]) for key in expected}
child = histogram.labels(**safe) if safe else histogram
child.observe(value, exemplar=_exemplar())
def render(self) -> tuple[bytes, str]:
from prometheus_client import CONTENT_TYPE_LATEST, generate_latest
# OpenMetrics exposition preserves histogram exemplars. Grafana uses
# their trace_id label to jump from a slow aggregate bucket to Tempo.
from prometheus_client.openmetrics.exposition import (
CONTENT_TYPE_LATEST,
generate_latest,
)
return generate_latest(self._registry), CONTENT_TYPE_LATEST
+34 -10
View File
@@ -6,14 +6,22 @@ from adapters.embedding import BedrockCohereQueryEmbedder
from adapters.postgres import PostgresConversationStore, PostgresTraceRepository
from adapters.qdrant import QdrantParentStore, QdrantRetriever
from config import Settings
from rag.agent import RagAgent
from rag.answer import GroundedAnswerService
from rag.artifacts import load_aliases
from rag.manifest import MANIFEST_POINT_ID, check_manifest, manifest_collection
from rag.instrumentation import (
InstrumentedEmbedder,
InstrumentedGenerator,
InstrumentedGroundedAnswerService,
InstrumentedQueryUnderstander,
InstrumentedRagAgent,
InstrumentedReranker,
InstrumentedRetrievalService,
)
from rag.metrics import NullMetrics
from rag.routing import CatalogDrugResolver, QueryRoutingService
from rag.sections import SectionResolver
from rag.service import EvidencePolicy, RetrievalService
from rag.service import EvidencePolicy
from rag.telemetry import configure_telemetry
from rag.understanding import LlmQueryUnderstander
# How many aliases to show per candidate drug (F-04 bounds *which* drugs are
@@ -126,6 +134,8 @@ def _verify_corpus_manifest(client, collection: str, embedder, settings: Setting
def build_runtime(settings: Settings):
metrics = _build_metrics(settings)
effective_metrics = metrics or NullMetrics()
configure_telemetry(settings, effective_metrics)
if settings.embedding_provider == "disabled":
return None, None, PostgresTraceRepository(settings.postgres_dsn), metrics
if settings.embedding_provider != "cohere-v4":
@@ -140,8 +150,11 @@ def build_runtime(settings: Settings):
api_key=settings.qdrant_api_key,
timeout=30,
)
embedder = BedrockCohereQueryEmbedder(
embedder = InstrumentedEmbedder(
BedrockCohereQueryEmbedder(
settings.embedding_dimensions, region=settings.aws_region
),
effective_metrics,
)
# F-05: a collection built with one model and queried with another
# returns hits and raises nothing — the results are just meaningless,
@@ -155,17 +168,25 @@ def build_runtime(settings: Settings):
# `.resolve()` is no longer on the live query path; `RagAgent` resolves
# drug identity through `LlmQueryUnderstander` against the same catalog.
resolver = CatalogDrugResolver(aliases)
retrieval = RetrievalService(
reranker = _build_reranker(settings)
retrieval = InstrumentedRetrievalService(
QdrantRetriever(client, settings.qdrant_collection, embedder),
QdrantParentStore(client, settings.qdrant_collection),
EvidencePolicy(minimum_score=settings.evidence_minimum_score),
section_resolver=section_resolver,
reranker=_build_reranker(settings),
reranker=(
InstrumentedReranker(reranker, effective_metrics)
if reranker is not None
else None
),
metrics=effective_metrics,
)
routing = QueryRoutingService(retrieval, resolver)
generator = _build_generator(settings)
answers = GroundedAnswerService(
routing, generator=generator, metrics=metrics or NullMetrics()
if generator is not None:
generator = InstrumentedGenerator(generator, effective_metrics)
answers = InstrumentedGroundedAnswerService(
routing, generator=generator, metrics=effective_metrics
)
trace_writer = PostgresTraceRepository(settings.postgres_dsn)
if generator is None:
@@ -175,13 +196,16 @@ def build_runtime(settings: Settings):
# capability to offer. Answer-only (retrieval-verified, no
# generation) still works through `answers` directly.
return answers, None, trace_writer, metrics
agent = RagAgent(
understander=LlmQueryUnderstander(generator, _catalog_names(aliases), resolver),
agent = InstrumentedRagAgent(
understander=InstrumentedQueryUnderstander(
LlmQueryUnderstander(generator, _catalog_names(aliases), resolver)
),
retrieval=retrieval,
answers=answers,
autocomplete=resolver,
max_wall_clock_ms=settings.max_wall_clock_ms,
max_llm_calls_per_turn=settings.max_llm_calls_per_turn,
store=PostgresConversationStore(settings.postgres_dsn),
metrics=effective_metrics,
)
return answers, agent, trace_writer, metrics
+8
View File
@@ -25,6 +25,7 @@ class Settings(BaseSettings):
model_config = SettingsConfigDict(env_file=".env", extra="ignore")
app_name: str = "vsf-duoc-thu-ai-service"
environment: str = "local"
qdrant_url: str = "http://localhost:6333"
qdrant_collection: str = "duocthu_v1"
qdrant_api_key: str | None = None
@@ -50,6 +51,13 @@ class Settings(BaseSettings):
# never uses it.
rerank_enabled: bool = False
metrics_enabled: bool = True
# OpenTelemetry is opt-in so the existing EC2 Compose deployment keeps
# answering when no collector is present. Docker/Kubernetes observability
# profiles enable it and point OTLP/HTTP at their local collector Service.
otel_enabled: bool = False
otel_service_name: str = "ai-service"
otel_exporter_otlp_endpoint: str = "http://localhost:4318/v1/traces"
otel_sample_ratio: float = Field(default=1.0, ge=0.0, le=1.0)
entities_path: Path = _default_entities_path()
# F-08: a per-turn budget across RagAgent's sequential Bedrock calls
# (understand, generate, one entailment check on the live agent path).
+64 -1
View File
@@ -1,13 +1,22 @@
from __future__ import annotations
from time import perf_counter
from typing import Any
from fastapi import FastAPI, Response
from fastapi import FastAPI, Request, Response
from adapters.postgres import PostgresTraceRepository
from bootstrap import build_runtime
from config import Settings, get_settings
from rag.answer import GroundedAnswerService
from rag.metrics import REQUEST_DURATION, REQUESTS, NullMetrics
from rag.telemetry import (
annotate_current_span,
configure_telemetry,
correlation_context,
current_trace_id,
request_span,
)
from routers.rag import router as rag_router
@@ -20,16 +29,65 @@ def create_app(
metrics: Any | None = None,
) -> FastAPI:
configured = settings or get_settings()
effective_metrics = metrics or NullMetrics()
configure_telemetry(configured, effective_metrics)
app = FastAPI(title=configured.app_name, version="0.1.0")
app.state.answer_service = answer_service
app.state.conversational = conversational
app.state.trace_writer = trace_writer
app.state.metrics = metrics
@app.middleware("http")
async def correlate_and_trace(request: Request, call_next):
route = _route_label(request.url.path)
method = request.method.upper()
started = perf_counter()
status_code = 500
with correlation_context(request.headers.get("x-correlation-id")) as correlation_id:
with request_span(method, route, request.headers) as span:
try:
response = await call_next(request)
status_code = response.status_code
response.headers["X-Correlation-ID"] = correlation_id
trace_id = current_trace_id()
if trace_id:
response.headers["X-Trace-ID"] = trace_id
return response
finally:
status = f"{status_code // 100}xx"
elapsed = perf_counter() - started
effective_metrics.increment(
REQUESTS, method=method, route=route, status=status
)
effective_metrics.observe(
REQUEST_DURATION,
elapsed,
method=method,
route=route,
status=status,
)
if span is not None:
span.set_attribute("http.response.status_code", status_code)
annotate_current_span(
**{
"duocthu.http.status_class": status,
"duocthu.duration_ms": elapsed * 1000,
}
)
@app.get("/health")
def health() -> dict[str, str]:
return {"status": "ok"}
@app.get("/ready")
def ready() -> Response:
# Runtime construction already verifies the Qdrant corpus manifest and
# fails startup on mismatch. PostgreSQL trace/history are intentionally
# fail-open and therefore must not make readiness flap.
if app.state.answer_service is None and configured.embedding_provider != "disabled":
return Response(status_code=503)
return Response(content='{"status":"ready"}', media_type="application/json")
@app.get("/metrics")
def prometheus_metrics() -> Response:
exporter = getattr(app.state, "metrics", None)
@@ -44,6 +102,11 @@ def create_app(
return app
def _route_label(path: str) -> str:
known = {"/health", "/ready", "/metrics", "/v1/rag/query", "/v1/rag/suggest"}
return path if path in known else "other"
_settings = get_settings()
_answer_service, _conversational, _trace_writer, _metrics = build_runtime(_settings)
app = create_app(
+5 -8
View File
@@ -1,6 +1,6 @@
from pathlib import Path
from adapters.postgres import PostgresConversationStore, PostgresTraceRepository
from adapters.postgres import PostgresTraceRepository
from config import get_settings
@@ -8,13 +8,10 @@ def main() -> None:
dsn = get_settings().postgres_dsn
migrations_dir = Path(__file__).parent / "migrations"
trace_migration = migrations_dir / "001_rag_retrieval_trace.sql"
PostgresTraceRepository(dsn).migrate(trace_migration)
print(f"Applied {trace_migration.name}")
conversation_migration = migrations_dir / "002_rag_conversation_turn.sql"
PostgresConversationStore(dsn).migrate(conversation_migration)
print(f"Applied {conversation_migration.name}")
repository = PostgresTraceRepository(dsn)
for migration in sorted(migrations_dir.glob("*.sql")):
repository.migrate(migration)
print(f"Applied {migration.name}")
if __name__ == "__main__":
@@ -0,0 +1,11 @@
ALTER TABLE rag_retrieval_trace
ADD COLUMN IF NOT EXISTS correlation_id text,
ADD COLUMN IF NOT EXISTS otel_trace_id varchar(32);
CREATE INDEX IF NOT EXISTS rag_retrieval_trace_correlation_idx
ON rag_retrieval_trace (correlation_id)
WHERE correlation_id IS NOT NULL;
CREATE INDEX IF NOT EXISTS rag_retrieval_trace_otel_idx
ON rag_retrieval_trace (otel_trace_id)
WHERE otel_trace_id IS NOT NULL;
+5
View File
@@ -18,6 +18,11 @@ test = ["pytest>=7.4,<9"]
# without a cloud generator. Neither is a precondition for a grounded answer.
metrics = ["prometheus-client>=0.20,<1"]
generation = ["anthropic>=0.112,<1"]
observability = [
"opentelemetry-api>=1.27,<2",
"opentelemetry-sdk>=1.27,<2",
"opentelemetry-exporter-otlp-proto-http>=1.27,<2",
]
[build-system]
requires = ["setuptools>=68"]
+18
View File
@@ -99,12 +99,30 @@ class AnswerPlan:
needs_warning: bool = False
# A fixed, non-LLM string. `docs/architecture.md`'s guardrail section
# specifies the disclaimer at several layers, and the web banner
# (`packages/ui/src/DisclaimerBanner.tsx`) was the only one in place: the
# `disclaimer` field declared in `packages/shared-types/src/dto/chat.ts` was
# never populated, so any consumer other than this one web UI received medical
# content with nothing attached. Keeping it out of the prompt is deliberate —
# a disclaimer the model writes is one the model can also reword, shorten or
# omit, and it would then have to be verified like any other generated claim.
DISCLAIMER = (
"Nội dung được trích từ Dược thư Quốc gia Việt Nam 2018, phục vụ tra cứu "
"chuyên môn và không thay thế chỉ định của bác sĩ hoặc dược sĩ lâm sàng."
)
@dataclass(frozen=True)
class GroundedAnswer:
result: RetrievalResult
answer: str | None
citations: tuple[Citation, ...] = ()
generated: bool = False
# Carried on the payload rather than added by the caller, so no response
# path can be built that omits it — including abstains and clarifications,
# which are also medical content in the sense that matters here.
disclaimer: str = DISCLAIMER
# Set when the model decided the turn is under-specified and asked back
# (e.g. a dose question with no age/weight). The answer field carries the
# question; the caller renders it as a clarification, not a final answer.
+3
View File
@@ -41,6 +41,8 @@ from __future__ import annotations
import re
from dataclasses import dataclass
from .telemetry import traced_stage
# A digit run with internal separators kept: "500", "7,5", "1.000".
# Ranges ("4 - 6 giờ") yield two tokens, and each is checked on its own.
_NUMBER = re.compile(r"\d+(?:[.,]\d+)*")
@@ -127,6 +129,7 @@ def split_claims(answer: str, evidence_count: int) -> tuple[Claim, ...]:
return tuple(claims)
@traced_stage("grounding")
def verify(answer: str, evidence_texts: tuple[str, ...]) -> GroundingReport:
"""Whether `answer` states only figures and sources traceable to the
specific evidence block(s) cited immediately after each claim.
+240
View File
@@ -0,0 +1,240 @@
"""Non-invasive instrumentation wrappers for the live RAG object graph.
Claude owns ``rag/agent.py`` and ``rag/answer.py`` in the shared worktree.
Subclasses here add spans at their stable method boundaries without changing
those files or duplicating their domain decisions.
"""
from __future__ import annotations
from typing import Any
from .agent import RagAgent
from .answer import GroundedAnswerService
from .metrics import CLARIFY_ASKED, Metrics, PROVIDER_FAILURE, RETRIEVAL_ROUTE
from .service import RetrievalService
from .telemetry import (
annotate_current_span,
current_stage,
dependency_span,
stage,
)
def _provider_name(delegate: Any) -> str:
name = delegate.__class__.__name__.casefold()
if "cohere" in name:
return "bedrock_cohere"
if "converse" in name:
return "bedrock_converse"
if "claude" in name:
return "bedrock_claude"
return "other"
def _failure_reason(exc: BaseException) -> str:
name = exc.__class__.__name__.casefold()
if "timeout" in name:
return "timeout"
if "budget" in name:
return "request_budget_exhausted"
if "unavailable" in name or "connection" in name:
return "provider_unavailable"
return "error"
class InstrumentedGenerator:
def __init__(self, delegate: Any, metrics: Metrics) -> None:
self._delegate = delegate
self._metrics = metrics
self._provider = _provider_name(delegate)
@property
def model_id(self) -> str:
return self._delegate.model_id
def generate(self, system: str, user: str, schema: dict) -> str:
operation = {
"understanding": "understand",
"generation": "generate",
"entailment": "entailment",
}.get(current_stage(), "generate")
with dependency_span(self._provider, operation):
try:
return self._delegate.generate(system, user, schema)
except Exception as exc:
self._metrics.increment(
PROVIDER_FAILURE,
provider=self._provider,
operation=operation,
reason=_failure_reason(exc),
)
raise
class InstrumentedEmbedder:
def __init__(self, delegate: Any, metrics: Metrics) -> None:
self._delegate = delegate
self._metrics = metrics
@property
def dimensions(self) -> int:
return self._delegate.dimensions
@property
def model_id(self) -> str:
return self._delegate.model_id
def embed_query(self, text: str) -> list[float]:
with dependency_span("bedrock_cohere", "embed"):
try:
return self._delegate.embed_query(text)
except Exception as exc:
self._metrics.increment(
PROVIDER_FAILURE,
provider="bedrock_cohere",
operation="embed",
reason=_failure_reason(exc),
)
raise
class InstrumentedReranker:
def __init__(self, delegate: Any, metrics: Metrics) -> None:
self._delegate = delegate
self._metrics = metrics
def rerank(
self, query: str, documents: list[str], top_n: int | None = None
) -> list[int]:
with dependency_span("bedrock_cohere", "rerank"):
try:
return self._delegate.rerank(query, documents, top_n=top_n)
except Exception as exc:
self._metrics.increment(
PROVIDER_FAILURE,
provider="bedrock_cohere",
operation="rerank",
reason=_failure_reason(exc),
)
raise
class InstrumentedRetrievalService(RetrievalService):
def __init__(self, *args, metrics: Metrics, **kwargs) -> None:
super().__init__(*args, **kwargs)
self._observability_metrics = metrics
def retrieve_framed(self, *args, **kwargs):
with stage("retrieval"):
section_key = args[1] if len(args) > 1 else kwargs.get("section_key")
is_overview = args[3] if len(args) > 3 else kwargs.get("is_overview", False)
route = "section" if section_key else (
"overview" if is_overview else "similarity"
)
self._observability_metrics.increment(RETRIEVAL_ROUTE, route=route)
try:
result = super().retrieve_framed(*args, **kwargs)
except Exception as exc:
self._record_retrieval_failure(exc)
raise
self._annotate_result(result)
return result
def retrieve(self, *args, **kwargs):
with stage("retrieval"):
self._observability_metrics.increment(RETRIEVAL_ROUTE, route="other")
try:
result = super().retrieve(*args, **kwargs)
except Exception as exc:
self._record_retrieval_failure(exc)
raise
self._annotate_result(result)
return result
def retrieve_by_indication(self, *args, **kwargs):
with stage("retrieval"):
self._observability_metrics.increment(RETRIEVAL_ROUTE, route="indication")
try:
result = super().retrieve_by_indication(*args, **kwargs)
except Exception as exc:
self._record_retrieval_failure(exc)
raise
self._annotate_result(result)
return result
def _record_retrieval_failure(self, exc: BaseException) -> None:
self._observability_metrics.increment(
PROVIDER_FAILURE,
provider="qdrant",
operation="retrieve",
reason=_failure_reason(exc),
)
@staticmethod
def _annotate_result(result) -> None:
annotate_current_span(
**{
"duocthu.decision": result.decision.value,
"duocthu.reason": result.reason,
"duocthu.evidence_count": len(result.evidence),
}
)
def _rerank(self, query, hits):
with stage("rerank", configured=self._reranker is not None):
return super()._rerank(query, hits)
def _decide(self, evidence, is_drug_overview=False):
with stage("evidence", evidence_count=len(evidence)):
return super()._decide(evidence, is_drug_overview=is_drug_overview)
class InstrumentedGroundedAnswerService(GroundedAnswerService):
def _generate(self, *args, **kwargs):
with stage("generation"):
return super()._generate(*args, **kwargs)
def _verify_entailment(self, *args, **kwargs):
with stage("entailment"):
return super()._verify_entailment(*args, **kwargs)
class InstrumentedQueryUnderstander:
def __init__(self, delegate: Any) -> None:
self._delegate = delegate
def understand(self, *args, **kwargs):
with stage("understanding"):
frame = self._delegate.understand(*args, **kwargs)
annotate_current_span(
**{
"duocthu.turn_type": frame.turn_type,
"duocthu.needs_clarify": frame.needs_clarify,
"duocthu.system_error": frame.system_error,
}
)
return frame
class InstrumentedRagAgent(RagAgent):
def __init__(self, *args, metrics: Metrics, **kwargs) -> None:
super().__init__(*args, **kwargs)
self._observability_metrics = metrics
def handle(self, *args, **kwargs):
reply = super().handle(*args, **kwargs)
if reply.decision == "clarify":
self._observability_metrics.increment(CLARIFY_ASKED, reason=reply.reason)
return reply
def _get_history(self, *args, **kwargs):
with stage("context"):
return super()._get_history(*args, **kwargs)
def _route(self, *args, **kwargs):
with stage("routing"):
return super()._route(*args, **kwargs)
def _remember(self, *args, **kwargs):
with stage("persistence"):
return super()._remember(*args, **kwargs)
+35
View File
@@ -18,18 +18,28 @@ from typing import Protocol
class Metrics(Protocol):
def increment(self, name: str, **labels: str) -> None: ...
def observe(self, name: str, value: float, **labels: str) -> None: ...
class NullMetrics:
def increment(self, name: str, **labels: str) -> None: # noqa: ARG002
# Deliberately inert: the default when no metrics stack is configured.
return None
def observe(
self, name: str, value: float, **labels: str # noqa: ARG002
) -> None:
return None
class InMemoryMetrics:
"""Reference implementation of the contract; also what tests assert on."""
def __init__(self) -> None:
self.counts: dict[tuple[str, tuple[tuple[str, str], ...]], int] = {}
self.observations: dict[
tuple[str, tuple[tuple[str, str], ...]], list[float]
] = {}
def increment(self, name: str, **labels: str) -> None:
key = (name, tuple(sorted(labels.items())))
@@ -40,6 +50,22 @@ class InMemoryMetrics:
return self.counts.get((name, tuple(sorted(labels.items()))), 0)
return sum(count for (n, _), count in self.counts.items() if n == name)
def observe(self, name: str, value: float, **labels: str) -> None:
key = (name, tuple(sorted(labels.items())))
self.observations.setdefault(key, []).append(value)
def observed(self, name: str, **labels: str) -> tuple[float, ...]:
if labels:
return tuple(
self.observations.get((name, tuple(sorted(labels.items()))), [])
)
return tuple(
value
for (metric_name, _), values in self.observations.items()
if metric_name == name
for value in values
)
RETRIEVAL_ROUTE = "duocthu_retrieval_route_total"
ABSTENTION = "duocthu_abstention_total"
@@ -60,3 +86,12 @@ FOLLOWUP_INHERITED = "duocthu_followup_inherited_total"
# degradation actually happens, since a silent fail-open with no counter is
# indistinguishable from tracing quietly working.
TRACE_WRITE_FAILED = "duocthu_trace_write_failed_total"
# HTTP and request-pipeline telemetry. Every label is a bounded vocabulary:
# route is a route template (never a raw path), status is a class (2xx/4xx/5xx),
# and stage/provider/reason values are normalized by the Prometheus adapter.
REQUESTS = "duocthu_requests_total"
REQUEST_DURATION = "duocthu_request_duration_seconds"
STAGE_DURATION = "duocthu_stage_duration_seconds"
DECISION = "duocthu_decision_total"
PROVIDER_FAILURE = "duocthu_provider_failure_total"
+217
View File
@@ -0,0 +1,217 @@
"""Request correlation, OpenTelemetry spans and stage timing.
This module deliberately has a no-op default. The RAG domain remains runnable
without the OpenTelemetry packages or collector, while a configured runtime
gets one trace context shared by FastAPI's async middleware and its sync
thread-pool endpoint.
"""
from __future__ import annotations
import re
import uuid
from contextlib import contextmanager
from contextvars import ContextVar
from functools import wraps
from time import perf_counter
from typing import Any, Iterator, Mapping
from .metrics import Metrics, NullMetrics, STAGE_DURATION
_CORRELATION_ID: ContextVar[str | None] = ContextVar(
"duocthu_correlation_id", default=None
)
_STAGE: ContextVar[str | None] = ContextVar("duocthu_stage", default=None)
_SAFE_CORRELATION = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$")
_metrics: Metrics = NullMetrics()
try:
from opentelemetry import trace
_tracer = trace.get_tracer("duocthu.ai-service")
except ImportError: # pragma: no cover - exercised only in minimal installs
trace = None
_tracer = None
def configure_telemetry(settings: Any, metrics: Metrics | None = None) -> bool:
"""Configure one OTLP tracer provider; return whether tracing is active.
Metrics are configured independently and continue to work when tracing is
disabled or optional OpenTelemetry packages are absent.
"""
global _metrics, _tracer
_metrics = metrics or NullMetrics()
if not getattr(settings, "otel_enabled", False):
return False
try:
from opentelemetry import trace as otel_trace
from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter
from opentelemetry.sdk.resources import Resource
from opentelemetry.sdk.trace import TracerProvider
from opentelemetry.sdk.trace.export import BatchSpanProcessor
from opentelemetry.sdk.trace.sampling import ParentBased, TraceIdRatioBased
except ImportError:
return False
current = otel_trace.get_tracer_provider()
# OpenTelemetry only permits setting the global provider once. Reuse a
# provider installed by a host/test instead of replacing it and warning.
if isinstance(current, TracerProvider):
_tracer = current.get_tracer("duocthu.ai-service")
return True
provider = TracerProvider(
resource=Resource.create(
{
"service.name": getattr(settings, "otel_service_name", "ai-service"),
"service.version": "0.1.0",
"deployment.environment": getattr(settings, "environment", "local"),
}
),
sampler=ParentBased(
TraceIdRatioBased(float(getattr(settings, "otel_sample_ratio", 1.0)))
),
)
exporter = OTLPSpanExporter(endpoint=settings.otel_exporter_otlp_endpoint)
provider.add_span_processor(BatchSpanProcessor(exporter))
otel_trace.set_tracer_provider(provider)
_tracer = provider.get_tracer("duocthu.ai-service")
return True
def normalize_correlation_id(candidate: str | None) -> str:
value = (candidate or "").strip()
return value if _SAFE_CORRELATION.fullmatch(value) else str(uuid.uuid4())
@contextmanager
def correlation_context(candidate: str | None) -> Iterator[str]:
correlation_id = normalize_correlation_id(candidate)
token = _CORRELATION_ID.set(correlation_id)
try:
yield correlation_id
finally:
_CORRELATION_ID.reset(token)
def current_correlation_id() -> str:
value = _CORRELATION_ID.get()
return value or str(uuid.uuid4())
def current_trace_id() -> str | None:
if trace is None:
return None
context = trace.get_current_span().get_span_context()
if not context.is_valid:
return None
return f"{context.trace_id:032x}"
def current_stage() -> str | None:
return _STAGE.get()
@contextmanager
def request_span(
method: str, route: str, headers: Mapping[str, str]
) -> Iterator[Any]:
"""Start a server span and extract an incoming W3C ``traceparent``."""
if _tracer is None:
yield None
return
try:
from opentelemetry import propagate
from opentelemetry.trace import SpanKind
except ImportError: # pragma: no cover - minimal install fallback
yield None
return
parent = propagate.extract(headers)
with _tracer.start_as_current_span(
f"{method} {route}",
context=parent,
kind=SpanKind.SERVER,
attributes={
"http.request.method": method,
"http.route": route,
"duocthu.correlation_id": current_correlation_id(),
},
) as span:
yield span
@contextmanager
def dependency_span(provider: str, operation: str) -> Iterator[Any]:
"""Trace one external dependency call without creating another histogram."""
if _tracer is None:
yield None
return
with _tracer.start_as_current_span(
f"provider.{provider}.{operation}",
attributes={
"duocthu.provider": provider,
"duocthu.operation": operation,
},
) as span:
yield span
@contextmanager
def stage(name: str, **attributes: Any) -> Iterator[Any]:
"""Create a child span and observe a low-cardinality stage histogram."""
started = perf_counter()
outcome = "ok"
span = None
token = _STAGE.set(name)
try:
if _tracer is None:
yield None
else:
safe_attributes = {
key: value
for key, value in attributes.items()
if value is not None and isinstance(value, (str, bool, int, float))
}
safe_attributes["duocthu.stage"] = name
with _tracer.start_as_current_span(
f"rag.stage.{name}", attributes=safe_attributes
) as span:
yield span
except BaseException as exc:
outcome = "cancelled" if exc.__class__.__name__ == "CancelledError" else "error"
if span is not None:
span.set_attribute("duocthu.outcome", outcome)
raise
finally:
elapsed = perf_counter() - started
if span is not None:
span.set_attribute("duocthu.outcome", outcome)
span.set_attribute("duocthu.duration_ms", elapsed * 1000)
_metrics.observe(STAGE_DURATION, elapsed, stage=name, outcome=outcome)
_STAGE.reset(token)
def annotate_current_span(**attributes: Any) -> None:
if trace is None:
return
span = trace.get_current_span()
if not span.is_recording():
return
for key, value in attributes.items():
if value is not None and isinstance(value, (str, bool, int, float)):
span.set_attribute(key, value)
def traced_stage(name: str):
"""Decorator form for pure functions that should remain otherwise untouched."""
def decorate(function):
@wraps(function)
def wrapped(*args, **kwargs):
with stage(name):
return function(*args, **kwargs)
return wrapped
return decorate
+33 -5
View File
@@ -7,9 +7,15 @@ from fastapi import APIRouter, Depends, HTTPException, Request
from pydantic import BaseModel, Field
from rag.answer import GroundedAnswerService
from rag.metrics import TRACE_WRITE_FAILED, Metrics, NullMetrics
from rag.metrics import DECISION, TRACE_WRITE_FAILED, Metrics, NullMetrics
from rag.models import QueryIntent, SubjectScope
from rag.policy import resolve_subject_scope
from rag.telemetry import (
annotate_current_span,
current_correlation_id,
current_trace_id,
stage,
)
class TraceWriter(Protocol):
@@ -61,6 +67,8 @@ class AnswerPlanResponse(BaseModel):
class RagQueryResponse(BaseModel):
trace_id: str
correlation_id: str
otel_trace_id: str | None = None
decision: str
reason: str
answer: str | None
@@ -166,7 +174,10 @@ def query_rag(
traces: Annotated[TraceWriter, Depends(_trace_writer)],
metrics: Annotated[Metrics, Depends(_metrics)],
) -> RagQueryResponse:
with stage("receive"):
agent = getattr(request.app.state, "conversational", None)
subject_scope = resolve_subject_scope(payload.query, payload.subject_scope)
intent = payload.intent
# `payload.subject_scope`/`payload.intent` are what the CALLER claims —
# logged below for audit, but the RagAgent path does not take them as an
@@ -175,9 +186,6 @@ def query_rag(
# own LLM understanding call, and does not gate on intent at all (this
# product is for doctors/pharmacists; a client label must not be, and
# here structurally cannot be, the safety decision — F-02).
subject_scope = resolve_subject_scope(payload.query, payload.subject_scope)
intent = payload.intent
if agent is not None:
# The live path (F-03): one LLM call understands the turn (drug
# identity against the real catalog, turn type, population/weight),
@@ -199,6 +207,7 @@ def query_rag(
# No generator configured (ANSWER_PROVIDER=disabled): there is no LLM
# to understand a turn with, so this is retrieval-only, single-turn,
# unchanged from before F-03.
with stage("routing"):
grounded = answers.answer(payload.query, subject_scope, intent)
if grounded.clarification is not None:
decision, reason = "clarify", "needs_more_info"
@@ -229,7 +238,10 @@ def query_rag(
# with whether the answer was safe. `trace_id` degrades to a local,
# unpersisted uuid — still a valid response field, just not one `GET
# /v1/rag/trace/{id}` (if it existed) could later look up.
correlation_id = current_correlation_id()
otel_trace_id = current_trace_id()
try:
with stage("persistence"):
trace_id = traces.save(
query=payload.query,
# The resolved (server-derived) values, not the caller's claim —
@@ -241,12 +253,27 @@ def query_rag(
reason=reason,
resolved_drug_id=resolved_drug_id,
citations=tuple(item.model_dump() for item in citations),
correlation_id=correlation_id,
otel_trace_id=otel_trace_id,
)
except Exception:
metrics.increment(TRACE_WRITE_FAILED)
trace_id = str(uuid.uuid4())
return RagQueryResponse(
metrics.increment(DECISION, decision=decision, reason=reason)
annotate_current_span(
**{
"duocthu.decision": decision,
"duocthu.reason": reason,
"duocthu.citation_count": len(citations),
"duocthu.generated": generated,
"duocthu.persisted_trace_id": trace_id,
}
)
with stage("response"):
response = RagQueryResponse(
trace_id=trace_id,
correlation_id=correlation_id,
otel_trace_id=otel_trace_id,
decision=decision,
reason=reason,
answer=answer,
@@ -258,3 +285,4 @@ def query_rag(
answer_mode=answer_mode,
answer_plan=answer_plan,
)
return response
+61
View File
@@ -1,5 +1,6 @@
from fastapi.testclient import TestClient
from adapters.prometheus import PrometheusMetrics
from config import Settings
from main import create_app
from rag.agent import AgentReply
@@ -177,3 +178,63 @@ def test_a_trace_write_failure_does_not_turn_a_good_answer_into_a_500():
assert body["answer"] == "Liều 500 mg [1]."
assert body["trace_id"] # a locally-generated fallback id, still present
assert metrics.total(TRACE_WRITE_FAILED) == 1
class _AnyRouting:
def retrieve(self, query, subject_scope, intent):
return RetrievalResult(EvidenceDecision.ABSTAIN, "drug_not_resolved")
def test_correlation_id_round_trips_through_headers_body_and_trace_row():
traces = MemoryTraceWriter()
app = create_app(
settings=Settings(embedding_provider="disabled"),
answer_service=GroundedAnswerService(_AnyRouting()),
trace_writer=traces,
)
response = TestClient(app).post(
"/v1/rag/query",
headers={"X-Correlation-ID": "req-test-1"},
json={
"query": "Paracetamol dose?",
"subject_scope": "human",
"intent": "fact_lookup",
},
)
assert response.status_code == 200
assert response.headers["x-correlation-id"] == "req-test-1"
assert response.json()["correlation_id"] == "req-test-1"
assert traces.rows[0]["correlation_id"] == "req-test-1"
assert "otel_trace_id" in traces.rows[0]
def test_metrics_endpoint_exposes_request_decision_and_stage_histograms():
metrics = PrometheusMetrics()
app = create_app(
settings=Settings(embedding_provider="disabled"),
answer_service=GroundedAnswerService(_AnyRouting()),
trace_writer=MemoryTraceWriter(),
metrics=metrics,
)
client = TestClient(app)
response = client.post(
"/v1/rag/query",
json={
"query": "Paracetamol dose?",
"subject_scope": "human",
"intent": "fact_lookup",
},
)
assert response.status_code == 200
scrape = client.get("/metrics")
assert scrape.status_code == 200
body = scrape.text
assert 'duocthu_requests_total{method="POST",route="/v1/rag/query",status="2xx"}' in body
assert 'duocthu_decision_total{decision="abstain",reason="drug_not_resolved"}' in body
assert 'duocthu_stage_duration_seconds_count{outcome="ok",stage="receive"}' in body
assert 'duocthu_stage_duration_seconds_count{outcome="ok",stage="persistence"}' in body
assert 'duocthu_stage_duration_seconds_count{outcome="ok",stage="response"}' in body
@@ -11,7 +11,7 @@ import json
import pytest
from rag import grounding
from rag.answer import GroundedAnswerService
from rag.answer import DISCLAIMER, GroundedAnswerService
from rag.budget import RequestBudgetExhausted
from rag.metrics import GENERATION_REJECTED, GENERATION_SERVED, InMemoryMetrics
from rag.models import (
@@ -116,6 +116,62 @@ def _answer(payload, result: RetrievalResult | None = None, entailment_payload=N
return grounded, metrics
# --- the disclaimer guardrail -------------------------------------------------
# `docs/architecture.md` specifies the disclaimer at several layers. Only the
# web banner existed; `packages/shared-types/src/dto/chat.ts` declared the
# field but nothing filled it, so a consumer other than that one UI got medical
# content with nothing attached. It is a dataclass default rather than
# something each call site adds, so these tests are about the paths that could
# plausibly skip it: rejections, abstains and clarifications.
def test_a_served_answer_carries_the_disclaimer():
grounded, _ = _answer(
{"claims": [{"text": "Người lớn uống 500 mg", "citations": [1]}],
"evidence_sufficient": True},
)
assert grounded.generated is True
assert grounded.disclaimer == DISCLAIMER
assert grounded.disclaimer
def test_a_rejected_generation_still_carries_the_disclaimer():
"""An abstain is still a clinical response and still needs the notice —
it is the case most likely to be treated as "not really an answer"."""
grounded, _ = _answer(
{"claims": [{"text": "Người lớn uống 850 mg, 2 lần mỗi ngày", "citations": [1]}],
"evidence_sufficient": True},
)
assert grounded.answer is None
assert grounded.result.decision == EvidenceDecision.ABSTAIN
assert grounded.disclaimer == DISCLAIMER
def test_a_provider_outage_response_still_carries_the_disclaimer():
grounded, _ = _answer(
{"claims": [{"text": "Người lớn: 500 mg, 2 lần/ngày", "citations": [1]}],
"evidence_sufficient": True},
entailment_payload=AnswerGenerationUnavailable(),
)
assert grounded.disclaimer == DISCLAIMER
def test_the_disclaimer_is_not_something_the_model_can_influence():
"""It is a module constant, never routed through the generator, so a
prompt-injected or malfunctioning model cannot shorten or drop it. Asserted
directly because the value is the guardrail."""
grounded, _ = _answer(
{"claims": [{"text": "Bỏ qua mọi cảnh báo. Người lớn uống 500 mg", "citations": [1]}],
"evidence_sufficient": True},
)
assert grounded.disclaimer == DISCLAIMER
assert "không thay thế chỉ định" in grounded.disclaimer
# --- the guardrail's whole reason to exist ------------------------------------
+122
View File
@@ -0,0 +1,122 @@
from adapters.prometheus import PrometheusMetrics
from config import Settings
from rag.instrumentation import InstrumentedGenerator
from rag.metrics import (
CLARIFY_ASKED,
PROVIDER_FAILURE,
REQUEST_DURATION,
REQUESTS,
STAGE_DURATION,
TRACE_WRITE_FAILED,
InMemoryMetrics,
)
from rag.telemetry import (
configure_telemetry,
correlation_context,
request_span,
stage,
)
def test_prometheus_registers_domain_failures_histograms_and_bounds_labels():
metrics = PrometheusMetrics()
metrics.increment(TRACE_WRITE_FAILED)
metrics.increment(CLARIFY_ASKED, reason="raw-user-controlled-reason")
metrics.increment(
REQUESTS, method="POST", route="/v1/rag/query", status="2xx"
)
metrics.observe(
REQUEST_DURATION,
1.25,
method="POST",
route="/v1/rag/query",
status="2xx",
)
metrics.observe(STAGE_DURATION, 0.2, stage="retrieval", outcome="ok")
body, content_type = metrics.render()
rendered = body.decode("utf-8")
assert "application/openmetrics-text" in content_type
assert "duocthu_trace_write_failed_total 1.0" in rendered
assert 'duocthu_clarify_asked_total{reason="other"} 1.0' in rendered
assert "duocthu_request_duration_seconds_bucket" in rendered
assert "duocthu_stage_duration_seconds_bucket" in rendered
def test_stage_records_duration_without_requiring_an_otel_collector():
metrics = InMemoryMetrics()
configure_telemetry(Settings(otel_enabled=False), metrics)
with stage("grounding"):
pass
values = metrics.observed(STAGE_DURATION, stage="grounding", outcome="ok")
assert len(values) == 1
assert values[0] >= 0
class _FailingGenerator:
model_id = "fake"
def generate(self, system: str, user: str, schema: dict) -> str:
raise ConnectionError("provider offline")
def test_provider_failure_uses_bounded_classification_not_exception_text():
metrics = InMemoryMetrics()
generator = InstrumentedGenerator(_FailingGenerator(), metrics)
try:
generator.generate("system", "user", {})
except ConnectionError:
pass
else: # pragma: no cover - assertion guard
raise AssertionError("expected provider failure")
assert metrics.total(
PROVIDER_FAILURE,
provider="other",
operation="generate",
reason="provider_unavailable",
) == 1
def test_otel_server_and_stage_spans_share_trace_and_emit_metric_exemplar():
otel_trace = pytest.importorskip("opentelemetry.trace")
sdk_trace = pytest.importorskip("opentelemetry.sdk.trace")
exporter_module = pytest.importorskip(
"opentelemetry.sdk.trace.export.in_memory_span_exporter"
)
export_module = pytest.importorskip("opentelemetry.sdk.trace.export")
exporter = exporter_module.InMemorySpanExporter()
provider = sdk_trace.TracerProvider()
provider.add_span_processor(export_module.SimpleSpanProcessor(exporter))
otel_trace.set_tracer_provider(provider)
metrics = PrometheusMetrics()
configure_telemetry(
Settings(
otel_enabled=True,
otel_exporter_otlp_endpoint="http://127.0.0.1:4318/v1/traces",
),
metrics,
)
with correlation_context("req-span-test"):
with request_span("POST", "/v1/rag/query", {}):
with stage("retrieval"):
pass
provider.force_flush()
spans = exporter.get_finished_spans()
assert {item.name for item in spans} == {
"POST /v1/rag/query",
"rag.stage.retrieval",
}
assert len({item.context.trace_id for item in spans}) == 1
assert spans[-1].attributes["duocthu.correlation_id"] == "req-span-test"
rendered = metrics.render()[0].decode("utf-8")
assert "duocthu_stage_duration_seconds_bucket" in rendered
assert 'trace_id="' in rendered
import pytest
+26 -8
View File
@@ -1,4 +1,5 @@
import { NextResponse } from "next/server";
import { randomUUID } from "node:crypto";
import type { AnswerBlock, AnswerPlan, Citation, SendMessageResponse } from "@duoc-thu/shared-types";
export const runtime = "nodejs";
@@ -211,7 +212,13 @@ export async function POST(request: Request) {
return NextResponse.json({ error: "conversation_id_too_long" }, { status: 400 });
}
const correlationId = `req-${Date.now()}-${Math.random().toString(36).slice(2, 7)}`;
const incomingCorrelationId = request.headers.get("x-correlation-id")?.trim();
const correlationId =
incomingCorrelationId && /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(incomingCorrelationId)
? incomingCorrelationId
: randomUUID();
let responseCorrelationId = correlationId;
let responseTraceId: string | null = null;
let rag: RagResponse;
try {
@@ -219,13 +226,19 @@ export async function POST(request: Request) {
? API_GATEWAY_URL
: `${API_GATEWAY_URL}/v1/rag/query`;
const upstream = await fetch(targetUrl, {
method: "POST",
headers: {
const upstreamHeaders: Record<string, string> = {
"Content-Type": "application/json",
"X-Correlation-ID": correlationId,
"X-Client-Version": "1.0.0",
},
};
const traceparent = request.headers.get("traceparent");
const tracestate = request.headers.get("tracestate");
if (traceparent) upstreamHeaders.traceparent = traceparent;
if (tracestate) upstreamHeaders.tracestate = tracestate;
const upstream = await fetch(targetUrl, {
method: "POST",
headers: upstreamHeaders,
body: JSON.stringify({
query: content,
subject_scope: "human",
@@ -238,6 +251,8 @@ export async function POST(request: Request) {
// prevents the BFF itself from keeping an orphaned HTTP request open.
signal: request.signal,
});
responseCorrelationId = upstream.headers.get("x-correlation-id") ?? correlationId;
responseTraceId = upstream.headers.get("x-trace-id");
if (!upstream.ok) {
rag = {
trace_id: `fallback-${Date.now()}`,
@@ -296,12 +311,15 @@ export async function POST(request: Request) {
answerPlan: rag.answer_plan ? toAnswerPlan(rag.answer_plan) : undefined,
};
const responseHeaders = new Headers({
"X-Correlation-ID": responseCorrelationId,
});
if (responseTraceId) responseHeaders.set("X-Trace-ID", responseTraceId);
return NextResponse.json(
{ message, sessionId: conversationId ?? undefined } satisfies SendMessageResponse,
{
headers: {
"X-Correlation-ID": correlationId,
},
headers: responseHeaders,
}
);
}
+142
View File
@@ -0,0 +1,142 @@
import { NextResponse } from "next/server";
import type { NextRequest } from "next/server";
/**
* Rate limiting for the public API surface.
*
* `/api/chat` is reachable by anyone on the internet, takes no credentials,
* and spends AWS Bedrock credit on every call (understanding + generation +
* entailment, several model calls per turn) against a small personal budget.
* The architecture assigns rate limiting to `api-gateway`, which is not built
* yet, so until it exists this is the only place the limit can live.
*
* It runs here rather than inside the route handlers because the edge
* middleware rejects an abusive request before any handler work — and because
* it covers every current and future `/api/*` route by default rather than
* one endpoint at a time.
*
* Known limitations, stated rather than hidden:
* - Counters are per process and in memory. Production runs a single `web`
* container, so this is a real limit today; the moment that scales to more
* than one replica each replica gets its own allowance, and this needs to
* move to Redis (already reserved for exactly this in `docs/architecture.md`)
* or to the gateway.
* - It is keyed by client IP, so it throttles a shared NAT as one caller. That
* is the correct trade for a cost guard with no authentication; per-user
* limits need auth, which does not exist yet.
* - It is a cost and abuse guard, not a security control. It does not
* authenticate anyone and must not be described as if it does.
*/
interface Bucket {
hits: number[];
}
interface Rule {
windowMs: number;
max: number;
}
// Chat is the expensive path: several Bedrock calls per request, and a single
// turn was measured taking up to ~45s of model time. Autocomplete is a local
// catalog lookup with no model call, so it can be far more generous without
// costing anything.
const RULES: Array<{ prefix: string; rules: Rule[] }> = [
{
prefix: "/api/chat",
rules: [
{ windowMs: 60_000, max: 12 },
{ windowMs: 3_600_000, max: 120 },
],
},
{
prefix: "/api/suggest",
rules: [{ windowMs: 60_000, max: 120 }],
},
];
const buckets = new Map<string, Bucket>();
const LONGEST_WINDOW_MS = 3_600_000;
let lastSweep = 0;
/** Drop entries no rule can still be counting, so the map cannot grow without bound. */
function sweep(now: number) {
if (now - lastSweep < 60_000) return;
lastSweep = now;
for (const [key, bucket] of buckets) {
const live = bucket.hits.filter((t) => now - t < LONGEST_WINDOW_MS);
if (live.length === 0) buckets.delete(key);
else bucket.hits = live;
}
}
/**
* Caddy sits in front and sets `X-Forwarded-For`; the left-most entry is the
* original client. Falling back to a shared key rather than to "unlimited"
* matters: an unknown IP must not become a way to opt out of the limit.
*/
function clientKey(request: NextRequest): string {
const forwarded = request.headers.get("x-forwarded-for");
if (forwarded) {
const first = forwarded.split(",")[0]?.trim();
if (first) return first;
}
return request.headers.get("x-real-ip")?.trim() || "unknown";
}
function matchRules(pathname: string) {
return RULES.find((entry) => pathname.startsWith(entry.prefix))?.rules;
}
export function middleware(request: NextRequest) {
const rules = matchRules(request.nextUrl.pathname);
if (!rules) return NextResponse.next();
const now = Date.now();
sweep(now);
const key = `${clientKey(request)}:${request.nextUrl.pathname}`;
const bucket = buckets.get(key) ?? { hits: [] };
bucket.hits = bucket.hits.filter((t) => now - t < LONGEST_WINDOW_MS);
for (const rule of rules) {
const inWindow = bucket.hits.filter((t) => now - t < rule.windowMs);
if (inWindow.length >= rule.max) {
const oldest = Math.min(...inWindow);
const retryAfterSec = Math.max(1, Math.ceil((rule.windowMs - (now - oldest)) / 1000));
// Record nothing for a rejected request: a client hammering the endpoint
// should not keep pushing its own window forward and lock itself out for
// longer than the rule says.
buckets.set(key, bucket);
return NextResponse.json(
{
error: "rate_limited",
message:
"Bạn đang gửi quá nhiều yêu cầu trong thời gian ngắn. Vui lòng đợi một lát rồi thử lại.",
},
{
status: 429,
headers: {
"Retry-After": String(retryAfterSec),
"X-RateLimit-Limit": String(rule.max),
"X-RateLimit-Remaining": "0",
},
}
);
}
}
bucket.hits.push(now);
buckets.set(key, bucket);
const tightest = rules[0];
const used = bucket.hits.filter((t) => now - t < tightest.windowMs).length;
const response = NextResponse.next();
response.headers.set("X-RateLimit-Limit", String(tightest.max));
response.headers.set("X-RateLimit-Remaining", String(Math.max(0, tightest.max - used)));
return response;
}
export const config = {
matcher: ["/api/:path*"],
};
+38
View File
@@ -53,6 +53,44 @@ hand plus direct `/api/chat` probes — measured, not inferred from docs.
evidence blocks unreachable from the prose, and provenance is a hard
guardrail.
## Second batch — guardrail gaps (same day)
A guardrail review against `docs/architecture.md` found two things the design
specifies that were not in the code. Both were implemented around the files
currently carrying uncommitted changes, so nothing in that set was touched.
- **Rate limiting — new `apps/web/middleware.ts`.** `/api/chat` is public,
takes no credentials and spends Bedrock credit per call against a small
personal AWS budget; the architecture assigns this to `api-gateway`, which
is not built. 12/min and 120/hour for `/api/chat`, 120/min for
`/api/suggest` (a local catalog lookup, no model call), keyed on the
left-most `X-Forwarded-For` entry that Caddy sets, returning 429 with
`Retry-After`. Counters are per process and in memory: correct for the
single `web` container in production today, and the point at which that
scales past one replica is the point this has to move to Redis or to the
gateway. It is a cost/abuse guard, not authentication.
- **Disclaimer — `rag/answer.py` only, wire-up still pending.**
`GroundedAnswer` now carries `disclaimer: str = DISCLAIMER` as a dataclass
default, so no response path can be constructed without it, including
abstains and clarifications. Deliberately a module constant and never sent
through the generator: a model-written disclaimer can be reworded or
dropped, and would then need verifying like any other generated claim.
### Wire-up left for whoever next owns `routers/rag.py`
`routers/rag.py` has uncommitted changes in this worktree, so the last step is
left undone rather than edited around someone else's work. Two small changes
complete it:
1. Add `disclaimer: str` to `RagQueryResponse` and pass
`grounded.disclaimer` through when the response is built.
2. `packages/shared-types/src/dto/chat.ts` already declares
`disclaimer?: string`, so the BFF only needs to copy it onto the message it
returns — no type change required.
Until step 1 lands, the guarantee exists in the domain object but is not yet
visible to an API consumer.
## Files claimed
`apps/ai-service/rag/answer.py`, `apps/ai-service/rag/agent.py`,
@@ -0,0 +1,62 @@
# Optional overlay for the existing EC2 production Compose topology:
# docker compose -f docker-compose.prod.yml -f docker-compose.observability.yml up -d
# The base production file remains unchanged and does not require this stack.
services:
ai-service:
environment:
OTEL_ENABLED: "true"
OTEL_SERVICE_NAME: ai-service
OTEL_EXPORTER_OTLP_ENDPOINT: http://otel-collector:4318/v1/traces
ENVIRONMENT: compose
depends_on:
- otel-collector
prometheus:
image: prom/prometheus:v3.3.0
command:
- --config.file=/etc/prometheus/prometheus.yml
- --storage.tsdb.path=/prometheus
- --enable-feature=exemplar-storage
volumes:
- ./prometheus/prometheus-compose.yml:/etc/prometheus/prometheus.yml:ro
- prometheus-data:/prometheus
restart: unless-stopped
tempo:
image: grafana/tempo:2.7.2
command: ["-config.file=/etc/tempo/tempo.yml"]
volumes:
- ./tempo/tempo.yml:/etc/tempo/tempo.yml:ro
- tempo-data:/var/tempo
restart: unless-stopped
otel-collector:
image: otel/opentelemetry-collector-contrib:0.123.0
command: ["--config=/etc/otelcol/config.yml"]
volumes:
- ./otel/collector.yml:/etc/otelcol/config.yml:ro
depends_on:
- tempo
restart: unless-stopped
grafana:
image: grafana/grafana:11.5.2
ports:
- "3002:3000"
environment:
GF_SECURITY_ADMIN_USER: ${GRAFANA_ADMIN_USER:-admin}
GF_SECURITY_ADMIN_PASSWORD: ${GRAFANA_ADMIN_PASSWORD:-change-me}
GF_AUTH_ANONYMOUS_ENABLED: "false"
volumes:
- ./grafana/provisioning:/etc/grafana/provisioning:ro
- ./grafana/dashboards:/var/lib/grafana/dashboards:ro
- grafana-data:/var/lib/grafana
depends_on:
- prometheus
- tempo
restart: unless-stopped
volumes:
prometheus-data:
tempo-data:
grafana-data:
+29 -2
View File
@@ -35,7 +35,11 @@ services:
# when ai-service moves into this compose file, change the target to
# `ai-service:8000` and drop the extra_hosts entry.
prometheus:
image: prom/prometheus:latest
image: prom/prometheus:v3.3.0
command:
- --config.file=/etc/prometheus/prometheus.yml
- --storage.tsdb.path=/prometheus
- --enable-feature=exemplar-storage
ports:
- "9090:9090"
volumes:
@@ -45,7 +49,7 @@ services:
- "host.docker.internal:host-gateway"
grafana:
image: grafana/grafana:latest
image: grafana/grafana:11.5.2
ports:
- "3002:3000"
environment:
@@ -61,6 +65,28 @@ services:
- grafana-data:/var/lib/grafana
depends_on:
- prometheus
- tempo
tempo:
image: grafana/tempo:2.7.2
command: ["-config.file=/etc/tempo/tempo.yml"]
ports:
- "3200:3200"
volumes:
- ./tempo/tempo.yml:/etc/tempo/tempo.yml:ro
- tempo-data:/var/tempo
otel-collector:
image: otel/opentelemetry-collector-contrib:0.123.0
command: ["--config=/etc/otelcol/config.yml"]
ports:
- "4317:4317"
- "4318:4318"
- "13133:13133"
volumes:
- ./otel/collector.yml:/etc/otelcol/config.yml:ro
depends_on:
- tempo
# ai-service:
# build: ../../apps/ai-service
@@ -100,3 +126,4 @@ volumes:
redis-data:
prometheus-data:
grafana-data:
tempo-data:
@@ -1,215 +1,97 @@
{
"uid": "duocthu-grounding",
"title": "Dược thư — Grounding & Retrieval",
"tags": ["duocthu", "rag"],
"uid": "duocthu-observability",
"title": "Dược Thư — Request path observability",
"tags": ["duocthu", "rag", "opentelemetry"],
"timezone": "browser",
"schemaVersion": 39,
"version": 1,
"version": 2,
"refresh": "10s",
"time": { "from": "now-1h", "to": "now" },
"panels": [
{
"id": 1,
"type": "stat",
"title": "Số lần LLM bịa số và bị chặn",
"description": "Generations discarded because they stated a figure that does not appear character-for-character in the cited source. This is the measured form of the claim that the answer layer cannot invent a dose. Non-zero is not a failure — it is the guardrail doing its job.",
"gridPos": { "h": 6, "w": 6, "x": 0, "y": 0 },
"targets": [
{
"refId": "A",
"expr": "sum(duocthu_generation_rejected_total{reason=\"ungrounded_number\"})",
"legendFormat": "blocked"
}
],
"fieldConfig": {
"defaults": {
"unit": "short",
"decimals": 0,
"color": { "mode": "thresholds" },
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "green", "value": null },
{ "color": "orange", "value": 1 }
]
}
},
"overrides": []
},
"options": {
"graphMode": "area",
"textMode": "value",
"colorMode": "value",
"reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false }
}
"title": "Request rate",
"gridPos": { "h": 5, "w": 6, "x": 0, "y": 0 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "sum(rate(duocthu_requests_total{route=\"/v1/rag/query\"}[5m]))", "legendFormat": "requests/s" }],
"fieldConfig": { "defaults": { "unit": "reqps", "decimals": 2 }, "overrides": [] },
"options": { "graphMode": "area", "colorMode": "value", "reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false } }
},
{
"id": 2,
"type": "stat",
"title": "Tỷ lệ câu trả lời có kiểm chứng",
"description": "Share of served answers that were LLM-generated and passed grounding verification. The remainder are served as verbatim source text — safe, just less readable.",
"gridPos": { "h": 6, "w": 6, "x": 6, "y": 0 },
"targets": [
{
"refId": "A",
"expr": "sum(duocthu_generation_served_total) / clamp_min(sum(duocthu_generation_served_total) + sum(duocthu_answer_extractive_total), 1)",
"legendFormat": "verified"
}
],
"fieldConfig": {
"defaults": {
"unit": "percentunit",
"decimals": 1,
"min": 0,
"max": 1,
"color": { "mode": "thresholds" },
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "red", "value": null },
{ "color": "orange", "value": 0.5 },
{ "color": "green", "value": 0.8 }
]
}
},
"overrides": []
},
"options": {
"graphMode": "area",
"textMode": "value",
"colorMode": "value",
"reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false }
}
"title": "Request p95",
"description": "The latency histogram carries Tempo trace exemplars. Click an exemplar in the latency panel below to inspect the complete request path.",
"gridPos": { "h": 5, "w": 6, "x": 6, "y": 0 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "histogram_quantile(0.95, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p95" }],
"fieldConfig": { "defaults": { "unit": "s", "decimals": 2 }, "overrides": [] },
"options": { "graphMode": "area", "colorMode": "value", "reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false } }
},
{
"id": 3,
"type": "stat",
"title": "Tỷ lệ từ chối trả lời",
"description": "Share of requests the system declined. A medical reference tool is expected to abstain — symptom questions, invented drug names and out-of-scope asks all land here by design.",
"gridPos": { "h": 6, "w": 6, "x": 12, "y": 0 },
"targets": [
{
"refId": "A",
"expr": "sum(duocthu_abstention_total) / clamp_min(sum(duocthu_abstention_total) + sum(duocthu_generation_served_total) + sum(duocthu_answer_extractive_total), 1)",
"legendFormat": "abstained"
}
],
"fieldConfig": {
"defaults": {
"unit": "percentunit",
"decimals": 1,
"min": 0,
"max": 1,
"color": { "mode": "fixed", "fixedColor": "blue" }
},
"overrides": []
},
"options": {
"graphMode": "area",
"textMode": "value",
"colorMode": "value",
"reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false }
}
"title": "Answerable ratio",
"gridPos": { "h": 5, "w": 6, "x": 12, "y": 0 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "sum(rate(duocthu_decision_total{decision=\"answerable\"}[5m])) / clamp_min(sum(rate(duocthu_decision_total[5m])), 0.000001)", "legendFormat": "answerable" }],
"fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1, "decimals": 1 }, "overrides": [] },
"options": { "graphMode": "area", "colorMode": "value", "reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false } }
},
{
"id": 4,
"type": "stat",
"title": "Câu đi đúng đường section route",
"description": "Retrievals resolved by section filter — the route measured at 16/16 on human-written questions. The remainder fall back to similarity, measured at hit@1 0.544.",
"gridPos": { "h": 6, "w": 6, "x": 18, "y": 0 },
"targets": [
{
"refId": "A",
"expr": "sum(duocthu_retrieval_route_total{route=\"section\"}) / clamp_min(sum(duocthu_retrieval_route_total), 1)",
"legendFormat": "section"
}
],
"fieldConfig": {
"defaults": {
"unit": "percentunit",
"decimals": 1,
"min": 0,
"max": 1,
"color": { "mode": "thresholds" },
"thresholds": {
"mode": "absolute",
"steps": [
{ "color": "red", "value": null },
{ "color": "orange", "value": 0.6 },
{ "color": "green", "value": 0.85 }
]
}
},
"overrides": []
},
"options": {
"graphMode": "area",
"textMode": "value",
"colorMode": "value",
"reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false }
}
"title": "Trace-write failures",
"gridPos": { "h": 5, "w": 6, "x": 18, "y": 0 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "sum(increase(duocthu_trace_write_failed_total[1h]))", "legendFormat": "fail-open writes" }],
"fieldConfig": { "defaults": { "unit": "short", "decimals": 0 }, "overrides": [] },
"options": { "graphMode": "area", "colorMode": "value", "reduceOptions": { "calcs": ["lastNotNull"], "fields": "", "values": false } }
},
{
"id": 5,
"type": "timeseries",
"title": "Vì sao một bản sinh bị loại",
"description": "Every reason a generation was discarded before reaching a clinician. `ungrounded_number` is a fabrication caught; `provider_unavailable` is an outage; `evidence_insufficient` is the model correctly declining.",
"gridPos": { "h": 9, "w": 12, "x": 0, "y": 6 },
"title": "End-to-end latency — click exemplar for Tempo trace",
"gridPos": { "h": 9, "w": 12, "x": 0, "y": 5 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [
{
"refId": "A",
"expr": "sum by (reason) (rate(duocthu_generation_rejected_total[5m]))",
"legendFormat": "{{reason}}"
}
{ "refId": "A", "expr": "histogram_quantile(0.50, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p50" },
{ "refId": "B", "expr": "histogram_quantile(0.95, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p95" },
{ "refId": "C", "expr": "histogram_quantile(0.99, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p99" }
],
"fieldConfig": {
"defaults": {
"unit": "reqps",
"custom": {
"drawStyle": "line",
"lineWidth": 2,
"fillOpacity": 12,
"showPoints": "never",
"stacking": { "mode": "normal", "group": "A" }
}
},
"overrides": []
},
"options": {
"legend": { "displayMode": "table", "placement": "right", "calcs": ["sum"] },
"tooltip": { "mode": "multi", "sort": "desc" }
}
"fieldConfig": { "defaults": { "unit": "s", "custom": { "drawStyle": "line", "lineWidth": 2, "fillOpacity": 10, "showPoints": "never" } }, "overrides": [] },
"options": { "legend": { "displayMode": "table", "placement": "bottom", "calcs": ["lastNotNull"] }, "tooltip": { "mode": "multi", "sort": "desc" } }
},
{
"id": 6,
"type": "timeseries",
"title": "Vì sao hệ thống từ chối trả lời",
"description": "Abstentions by the reason retrieval gave. `drug_not_resolved` dominating means most refusals are questions that never named a drug in the formulary — symptom questions and invented names.",
"gridPos": { "h": 9, "w": 12, "x": 12, "y": 6 },
"targets": [
"title": "Stage p95 latency",
"gridPos": { "h": 9, "w": 12, "x": 12, "y": 5 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "histogram_quantile(0.95, sum by (stage, le) (rate(duocthu_stage_duration_seconds_bucket[5m])))", "legendFormat": "{{stage}}" }],
"fieldConfig": { "defaults": { "unit": "s", "custom": { "drawStyle": "line", "lineWidth": 2, "fillOpacity": 8, "showPoints": "never" } }, "overrides": [] },
"options": { "legend": { "displayMode": "table", "placement": "right", "calcs": ["lastNotNull"] }, "tooltip": { "mode": "multi", "sort": "desc" } }
},
{
"refId": "A",
"expr": "sum by (reason) (rate(duocthu_abstention_total[5m]))",
"legendFormat": "{{reason}}"
}
],
"fieldConfig": {
"defaults": {
"unit": "reqps",
"custom": {
"drawStyle": "line",
"lineWidth": 2,
"fillOpacity": 12,
"showPoints": "never",
"stacking": { "mode": "normal", "group": "A" }
}
"id": 7,
"type": "timeseries",
"title": "Decision and reason rate",
"gridPos": { "h": 9, "w": 12, "x": 0, "y": 14 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "sum by (decision, reason) (rate(duocthu_decision_total[5m]))", "legendFormat": "{{decision}} · {{reason}}" }],
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "line", "lineWidth": 2, "fillOpacity": 12, "showPoints": "never", "stacking": { "mode": "normal", "group": "A" } } }, "overrides": [] },
"options": { "legend": { "displayMode": "table", "placement": "right", "calcs": ["sum"] }, "tooltip": { "mode": "multi", "sort": "desc" } }
},
"overrides": []
},
"options": {
"legend": { "displayMode": "table", "placement": "right", "calcs": ["sum"] },
"tooltip": { "mode": "multi", "sort": "desc" }
}
{
"id": 8,
"type": "timeseries",
"title": "Provider failures",
"gridPos": { "h": 9, "w": 12, "x": 12, "y": 14 },
"datasource": { "type": "prometheus", "uid": "prometheus" },
"targets": [{ "refId": "A", "expr": "sum by (provider, operation, reason) (rate(duocthu_provider_failure_total[5m]))", "legendFormat": "{{provider}} · {{operation}} · {{reason}}" }],
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "bars", "lineWidth": 1, "fillOpacity": 35, "showPoints": "never" } }, "overrides": [] },
"options": { "legend": { "displayMode": "table", "placement": "right", "calcs": ["sum"] }, "tooltip": { "mode": "multi", "sort": "desc" } }
}
]
}
@@ -2,8 +2,33 @@ apiVersion: 1
datasources:
- name: Prometheus
uid: prometheus
type: prometheus
access: proxy
url: http://prometheus:9090
isDefault: true
editable: false
jsonData:
httpMethod: POST
exemplarTraceIdDestinations:
- datasourceUid: tempo
name: trace_id
- name: Tempo
uid: tempo
type: tempo
access: proxy
url: http://tempo:3200
editable: false
jsonData:
nodeGraph:
enabled: true
serviceMap:
datasourceUid: prometheus
tracesToMetrics:
datasourceUid: prometheus
spanStartTimeShift: -2m
spanEndTimeShift: 2m
tags:
- key: service.name
value: service
+34
View File
@@ -0,0 +1,34 @@
receivers:
otlp:
protocols:
grpc:
endpoint: 0.0.0.0:4317
http:
endpoint: 0.0.0.0:4318
processors:
memory_limiter:
check_interval: 1s
limit_mib: 256
spike_limit_mib: 64
batch:
timeout: 2s
send_batch_size: 512
exporters:
otlp/tempo:
endpoint: tempo:4317
tls:
insecure: true
extensions:
health_check:
endpoint: 0.0.0.0:13133
service:
extensions: [health_check]
pipelines:
traces:
receivers: [otlp]
processors: [memory_limiter, batch]
exporters: [otlp/tempo]
@@ -0,0 +1,16 @@
global:
scrape_interval: 15s
evaluation_interval: 15s
storage:
exemplars:
max_exemplars: 100000
scrape_configs:
- job_name: ai-service
metrics_path: /metrics
static_configs:
- targets: ["ai-service:8000"]
labels:
service: ai-service
env: compose
+4
View File
@@ -2,6 +2,10 @@ global:
scrape_interval: 15s
evaluation_interval: 15s
storage:
exemplars:
max_exemplars: 100000
scrape_configs:
- job_name: ai-service
metrics_path: /metrics
+26
View File
@@ -0,0 +1,26 @@
server:
http_listen_port: 3200
distributor:
receivers:
otlp:
protocols:
grpc:
endpoint: 0.0.0.0:4317
http:
endpoint: 0.0.0.0:4318
ingester:
max_block_duration: 5m
compactor:
compaction:
block_retention: 24h
storage:
trace:
backend: local
wal:
path: /var/tempo/wal
local:
path: /var/tempo/blocks
+2 -2
View File
@@ -2,5 +2,5 @@ apiVersion: v2
name: medical-chatbot
description: Umbrella Helm chart for the Duoc Thu RAG medical chatbot platform
type: application
version: 0.0.0
appVersion: "0.0.0"
version: 0.1.0
appVersion: "0.1.0"
@@ -0,0 +1,34 @@
{{- define "medical-chatbot.name" -}}
{{- default .Chart.Name .Values.nameOverride | trunc 63 | trimSuffix "-" -}}
{{- end -}}
{{- define "medical-chatbot.fullname" -}}
{{- if .Values.fullnameOverride -}}
{{- .Values.fullnameOverride | trunc 63 | trimSuffix "-" -}}
{{- else -}}
{{- printf "%s-%s" .Release.Name (include "medical-chatbot.name" .) | trunc 63 | trimSuffix "-" -}}
{{- end -}}
{{- end -}}
{{- define "medical-chatbot.labels" -}}
app.kubernetes.io/name: {{ include "medical-chatbot.name" . }}
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/managed-by: {{ .Release.Service }}
helm.sh/chart: {{ printf "%s-%s" .Chart.Name .Chart.Version | quote }}
{{- end -}}
{{- define "medical-chatbot.serviceAccountName" -}}
{{- if .Values.serviceAccount.create -}}
{{- default (include "medical-chatbot.fullname" .) .Values.serviceAccount.name -}}
{{- else -}}
{{- default "default" .Values.serviceAccount.name -}}
{{- end -}}
{{- end -}}
{{- define "medical-chatbot.secretName" -}}
{{- if .Values.secret.create -}}
{{- printf "%s-runtime" (include "medical-chatbot.fullname" .) -}}
{{- else -}}
{{- required "secret.existingSecret is required when secret.create=false" .Values.secret.existingSecret -}}
{{- end -}}
{{- end -}}
@@ -0,0 +1,107 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "medical-chatbot.fullname" . }}-ai-service
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
data:
ENVIRONMENT: {{ .Values.global.environment | quote }}
QDRANT_URL: {{ default (printf "http://%s-qdrant:6333" (include "medical-chatbot.fullname" .)) .Values.qdrant.url | quote }}
QDRANT_COLLECTION: {{ .Values.aiService.config.qdrantCollection | quote }}
EMBEDDING_PROVIDER: {{ .Values.aiService.config.embeddingProvider | quote }}
ANSWER_PROVIDER: {{ .Values.aiService.config.answerProvider | quote }}
ANSWER_MODEL_ID: {{ .Values.aiService.config.answerModelId | quote }}
METRICS_ENABLED: {{ .Values.aiService.config.metricsEnabled | quote }}
OTEL_ENABLED: {{ and .Values.observability.enabled .Values.aiService.config.otelEnabled | quote }}
OTEL_SERVICE_NAME: ai-service
OTEL_EXPORTER_OTLP_ENDPOINT: {{ printf "http://%s-otel-collector:4318/v1/traces" (include "medical-chatbot.fullname" .) | quote }}
OTEL_SAMPLE_RATIO: {{ .Values.aiService.config.otelSampleRatio | quote }}
MAX_WALL_CLOCK_MS: {{ .Values.aiService.config.maxWallClockMs | quote }}
MAX_LLM_CALLS_PER_TURN: {{ .Values.aiService.config.maxLlmCallsPerTurn | quote }}
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-ai-service
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: ai-service
spec:
replicas: {{ .Values.aiService.replicaCount }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: ai-service
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: ai-service
annotations:
prometheus.io/scrape: "true"
prometheus.io/path: /metrics
prometheus.io/port: "8000"
spec:
serviceAccountName: {{ include "medical-chatbot.serviceAccountName" . }}
imagePullSecrets:
{{- toYaml .Values.global.imagePullSecrets | nindent 8 }}
{{- if .Values.aiService.migration.enabled }}
initContainers:
- name: migrate
image: "{{ .Values.aiService.image.repository }}:{{ .Values.aiService.image.tag }}"
imagePullPolicy: {{ .Values.aiService.image.pullPolicy }}
command: ["python", "migrate.py"]
envFrom:
- configMapRef: { name: {{ include "medical-chatbot.fullname" . }}-ai-service }
env:
- name: POSTGRES_DSN
valueFrom:
secretKeyRef:
name: {{ include "medical-chatbot.secretName" . }}
key: postgres-dsn
{{- end }}
containers:
- name: ai-service
image: "{{ .Values.aiService.image.repository }}:{{ .Values.aiService.image.tag }}"
imagePullPolicy: {{ .Values.aiService.image.pullPolicy }}
ports:
- { name: http, containerPort: 8000 }
envFrom:
- configMapRef: { name: {{ include "medical-chatbot.fullname" . }}-ai-service }
env:
- name: POSTGRES_DSN
valueFrom:
secretKeyRef:
name: {{ include "medical-chatbot.secretName" . }}
key: postgres-dsn
readinessProbe:
httpGet: { path: /ready, port: http }
initialDelaySeconds: 5
periodSeconds: 10
livenessProbe:
httpGet: { path: /health, port: http }
initialDelaySeconds: 15
periodSeconds: 20
startupProbe:
httpGet: { path: /health, port: http }
failureThreshold: 30
periodSeconds: 5
resources:
{{- toYaml .Values.aiService.resources | nindent 12 }}
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-ai-service
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: ai-service
spec:
type: {{ .Values.aiService.service.type }}
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: ai-service
ports:
- name: http
port: {{ .Values.aiService.service.port }}
targetPort: http
@@ -0,0 +1,131 @@
{{- if .Values.postgres.enabled }}
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-postgres
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: postgres
spec:
clusterIP: None
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: postgres
ports:
- { name: postgres, port: 5432, targetPort: postgres }
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: {{ include "medical-chatbot.fullname" . }}-postgres
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
serviceName: {{ include "medical-chatbot.fullname" . }}-postgres
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: postgres
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: postgres
spec:
containers:
- name: postgres
image: {{ .Values.postgres.image }}
ports:
- { name: postgres, containerPort: 5432 }
env:
- { name: POSTGRES_USER, value: duoc_thu }
- { name: POSTGRES_DB, value: duoc_thu }
- name: POSTGRES_PASSWORD
valueFrom:
secretKeyRef:
name: {{ include "medical-chatbot.secretName" . }}
key: postgres-password
readinessProbe:
exec: { command: ["pg_isready", "-U", "duoc_thu", "-d", "duoc_thu"] }
initialDelaySeconds: 5
periodSeconds: 10
livenessProbe:
exec: { command: ["pg_isready", "-U", "duoc_thu", "-d", "duoc_thu"] }
initialDelaySeconds: 20
periodSeconds: 20
resources:
{{- toYaml .Values.postgres.resources | nindent 12 }}
volumeMounts:
- { name: data, mountPath: /var/lib/postgresql/data }
volumeClaimTemplates:
- metadata: { name: data }
spec:
accessModes: [ReadWriteOnce]
resources:
requests: { storage: {{ .Values.postgres.storage }} }
{{- end }}
{{- if and .Values.postgres.enabled .Values.qdrant.enabled }}
---
{{- end }}
{{- if .Values.qdrant.enabled }}
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-qdrant
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: qdrant
spec:
clusterIP: None
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: qdrant
ports:
- { name: http, port: 6333, targetPort: http }
- { name: grpc, port: 6334, targetPort: grpc }
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: {{ include "medical-chatbot.fullname" . }}-qdrant
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
serviceName: {{ include "medical-chatbot.fullname" . }}-qdrant
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: qdrant
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: qdrant
spec:
containers:
- name: qdrant
image: {{ .Values.qdrant.image }}
ports:
- { name: http, containerPort: 6333 }
- { name: grpc, containerPort: 6334 }
readinessProbe:
httpGet: { path: /readyz, port: http }
initialDelaySeconds: 5
periodSeconds: 10
livenessProbe:
httpGet: { path: /healthz, port: http }
initialDelaySeconds: 20
periodSeconds: 20
resources:
{{- toYaml .Values.qdrant.resources | nindent 12 }}
volumeMounts:
- { name: data, mountPath: /qdrant/storage }
volumeClaimTemplates:
- metadata: { name: data }
spec:
accessModes: [ReadWriteOnce]
resources:
requests: { storage: {{ .Values.qdrant.storage }} }
{{- end }}
@@ -0,0 +1,32 @@
{{- if .Values.ingress.enabled }}
apiVersion: networking.k8s.io/v1
kind: Ingress
metadata:
name: {{ include "medical-chatbot.fullname" . }}
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
annotations:
{{- toYaml .Values.ingress.annotations | nindent 4 }}
spec:
ingressClassName: {{ .Values.ingress.className }}
{{- with .Values.ingress.tls }}
tls:
{{- toYaml . | nindent 4 }}
{{- end }}
rules:
- host: {{ .Values.ingress.host | quote }}
http:
paths:
- path: /v1/rag
pathType: Prefix
backend:
service:
name: {{ include "medical-chatbot.fullname" . }}-ai-service
port: { name: http }
- path: /
pathType: Prefix
backend:
service:
name: {{ include "medical-chatbot.fullname" . }}-web
port: { name: http }
{{- end }}
@@ -0,0 +1,170 @@
{{- if .Values.observability.enabled }}
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "medical-chatbot.fullname" . }}-prometheus-config
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
data:
prometheus.yml: |
global:
scrape_interval: 15s
evaluation_interval: 15s
storage:
exemplars:
max_exemplars: 100000
scrape_configs:
- job_name: ai-service
metrics_path: /metrics
static_configs:
- targets: [{{ printf "%s-ai-service:%v" (include "medical-chatbot.fullname" .) .Values.aiService.service.port | quote }}]
labels:
service: ai-service
env: {{ .Values.global.environment | quote }}
---
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "medical-chatbot.fullname" . }}-tempo-config
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
data:
tempo.yml: |
server:
http_listen_port: 3200
distributor:
receivers:
otlp:
protocols:
grpc: { endpoint: 0.0.0.0:4317 }
http: { endpoint: 0.0.0.0:4318 }
ingester:
max_block_duration: 5m
compactor:
compaction:
block_retention: {{ .Values.observability.tempo.retention }}
storage:
trace:
backend: local
wal: { path: /var/tempo/wal }
local: { path: /var/tempo/blocks }
---
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "medical-chatbot.fullname" . }}-otel-config
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
data:
collector.yml: |
receivers:
otlp:
protocols:
grpc: { endpoint: 0.0.0.0:4317 }
http: { endpoint: 0.0.0.0:4318 }
processors:
memory_limiter: { check_interval: 1s, limit_mib: 256, spike_limit_mib: 64 }
batch: { timeout: 2s, send_batch_size: 512 }
exporters:
otlp/tempo:
endpoint: {{ include "medical-chatbot.fullname" . }}-tempo:4317
tls: { insecure: true }
extensions:
health_check: { endpoint: 0.0.0.0:13133 }
service:
extensions: [health_check]
pipelines:
traces:
receivers: [otlp]
processors: [memory_limiter, batch]
exporters: [otlp/tempo]
---
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "medical-chatbot.fullname" . }}-grafana-provisioning
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
data:
datasources.yml: |
apiVersion: 1
datasources:
- name: Prometheus
uid: prometheus
type: prometheus
access: proxy
url: http://{{ include "medical-chatbot.fullname" . }}-prometheus:9090
isDefault: true
editable: false
jsonData:
httpMethod: POST
exemplarTraceIdDestinations:
- { datasourceUid: tempo, name: trace_id }
- name: Tempo
uid: tempo
type: tempo
access: proxy
url: http://{{ include "medical-chatbot.fullname" . }}-tempo:3200
editable: false
jsonData:
nodeGraph: { enabled: true }
serviceMap: { datasourceUid: prometheus }
tracesToMetrics:
datasourceUid: prometheus
spanStartTimeShift: -2m
spanEndTimeShift: 2m
tags:
- { key: service.name, value: service }
dashboards.yml: |
apiVersion: 1
providers:
- name: duocthu
folder: Dược Thư
type: file
disableDeletion: false
updateIntervalSeconds: 30
options: { path: /var/lib/grafana/dashboards }
dashboard.json: |
{
"uid": "duocthu-observability",
"title": "Dược Thư — Request path observability",
"tags": ["duocthu", "rag", "opentelemetry"],
"schemaVersion": 39,
"refresh": "10s",
"time": {"from": "now-1h", "to": "now"},
"panels": [
{
"id": 1, "type": "timeseries", "title": "Request p50/p95 — exemplars open Tempo",
"gridPos": {"h": 9, "w": 12, "x": 0, "y": 0},
"datasource": {"type": "prometheus", "uid": "prometheus"},
"targets": [
{"refId": "A", "expr": "histogram_quantile(0.50, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p50"},
{"refId": "B", "expr": "histogram_quantile(0.95, sum by (le) (rate(duocthu_request_duration_seconds_bucket{route=\"/v1/rag/query\"}[5m])))", "legendFormat": "p95"}
],
"fieldConfig": {"defaults": {"unit": "s"}, "overrides": []}
},
{
"id": 2, "type": "timeseries", "title": "Stage p95 latency",
"gridPos": {"h": 9, "w": 12, "x": 12, "y": 0},
"datasource": {"type": "prometheus", "uid": "prometheus"},
"targets": [{"refId": "A", "expr": "histogram_quantile(0.95, sum by (stage, le) (rate(duocthu_stage_duration_seconds_bucket[5m])))", "legendFormat": "{{`{{stage}}`}}"}],
"fieldConfig": {"defaults": {"unit": "s"}, "overrides": []}
},
{
"id": 3, "type": "timeseries", "title": "Decision / reason",
"gridPos": {"h": 9, "w": 12, "x": 0, "y": 9},
"datasource": {"type": "prometheus", "uid": "prometheus"},
"targets": [{"refId": "A", "expr": "sum by (decision, reason) (rate(duocthu_decision_total[5m]))", "legendFormat": "{{`{{decision}}`}} · {{`{{reason}}`}}"}]
},
{
"id": 4, "type": "timeseries", "title": "Provider and trace-write failures",
"gridPos": {"h": 9, "w": 12, "x": 12, "y": 9},
"datasource": {"type": "prometheus", "uid": "prometheus"},
"targets": [
{"refId": "A", "expr": "sum by (provider, operation, reason) (rate(duocthu_provider_failure_total[5m]))", "legendFormat": "{{`{{provider}}`}} · {{`{{operation}}`}} · {{`{{reason}}`}}"},
{"refId": "B", "expr": "rate(duocthu_trace_write_failed_total[5m])", "legendFormat": "trace write"}
]
}
]
}
{{- end }}
@@ -0,0 +1,292 @@
{{- if .Values.observability.enabled }}
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: {{ include "medical-chatbot.fullname" . }}-prometheus
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
accessModes: [ReadWriteOnce]
resources:
requests: { storage: {{ .Values.observability.prometheus.storage }} }
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-prometheus
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: prometheus
spec:
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: prometheus
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: prometheus
spec:
securityContext: { fsGroup: 65534 }
containers:
- name: prometheus
image: {{ .Values.observability.prometheus.image }}
args:
- --config.file=/etc/prometheus/prometheus.yml
- --storage.tsdb.path=/prometheus
- --storage.tsdb.retention.time={{ .Values.observability.prometheus.retention }}
- --enable-feature=exemplar-storage
ports:
- { name: http, containerPort: 9090 }
readinessProbe:
httpGet: { path: /-/ready, port: http }
livenessProbe:
httpGet: { path: /-/healthy, port: http }
initialDelaySeconds: 15
resources:
{{- toYaml .Values.observability.prometheus.resources | nindent 12 }}
volumeMounts:
- { name: config, mountPath: /etc/prometheus }
- { name: data, mountPath: /prometheus }
volumes:
- name: config
configMap: { name: {{ include "medical-chatbot.fullname" . }}-prometheus-config }
- name: data
persistentVolumeClaim: { claimName: {{ include "medical-chatbot.fullname" . }}-prometheus }
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-prometheus
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: prometheus
ports:
- { name: http, port: 9090, targetPort: http }
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: {{ include "medical-chatbot.fullname" . }}-tempo
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
accessModes: [ReadWriteOnce]
resources:
requests: { storage: {{ .Values.observability.tempo.storage }} }
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-tempo
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: tempo
spec:
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: tempo
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: tempo
spec:
securityContext: { fsGroup: 10001 }
containers:
- name: tempo
image: {{ .Values.observability.tempo.image }}
args: ["-config.file=/etc/tempo/tempo.yml"]
ports:
- { name: http, containerPort: 3200 }
- { name: otlp-grpc, containerPort: 4317 }
readinessProbe:
httpGet: { path: /ready, port: http }
initialDelaySeconds: 5
livenessProbe:
httpGet: { path: /ready, port: http }
initialDelaySeconds: 20
resources:
{{- toYaml .Values.observability.tempo.resources | nindent 12 }}
volumeMounts:
- { name: config, mountPath: /etc/tempo }
- { name: data, mountPath: /var/tempo }
volumes:
- name: config
configMap: { name: {{ include "medical-chatbot.fullname" . }}-tempo-config }
- name: data
persistentVolumeClaim: { claimName: {{ include "medical-chatbot.fullname" . }}-tempo }
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-tempo
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: tempo
ports:
- { name: http, port: 3200, targetPort: http }
- { name: otlp-grpc, port: 4317, targetPort: otlp-grpc }
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-otel-collector
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: otel-collector
spec:
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: otel-collector
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: otel-collector
spec:
containers:
- name: otel-collector
image: {{ .Values.observability.collector.image }}
args: ["--config=/etc/otelcol/collector.yml"]
ports:
- { name: otlp-grpc, containerPort: 4317 }
- { name: otlp-http, containerPort: 4318 }
- { name: health, containerPort: 13133 }
readinessProbe:
httpGet: { path: /, port: health }
livenessProbe:
httpGet: { path: /, port: health }
initialDelaySeconds: 10
resources:
{{- toYaml .Values.observability.collector.resources | nindent 12 }}
volumeMounts:
- { name: config, mountPath: /etc/otelcol }
volumes:
- name: config
configMap: { name: {{ include "medical-chatbot.fullname" . }}-otel-config }
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-otel-collector
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: otel-collector
ports:
- { name: otlp-grpc, port: 4317, targetPort: otlp-grpc }
- { name: otlp-http, port: 4318, targetPort: otlp-http }
- { name: health, port: 13133, targetPort: health }
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: {{ include "medical-chatbot.fullname" . }}-grafana
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
accessModes: [ReadWriteOnce]
resources:
requests: { storage: {{ .Values.observability.grafana.storage }} }
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-grafana
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: grafana
spec:
replicas: 1
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: grafana
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: grafana
spec:
securityContext: { fsGroup: 472 }
containers:
- name: grafana
image: {{ .Values.observability.grafana.image }}
ports:
- { name: http, containerPort: 3000 }
env:
- { name: GF_SECURITY_ADMIN_USER, value: admin }
- name: GF_SECURITY_ADMIN_PASSWORD
valueFrom:
secretKeyRef:
name: {{ include "medical-chatbot.secretName" . }}
key: grafana-admin-password
- name: GF_AUTH_ANONYMOUS_ENABLED
value: {{ .Values.observability.grafana.anonymousAdmin | quote }}
- name: GF_AUTH_ANONYMOUS_ORG_ROLE
value: Admin
- name: GF_AUTH_DISABLE_LOGIN_FORM
value: {{ .Values.observability.grafana.anonymousAdmin | quote }}
readinessProbe:
httpGet: { path: /api/health, port: http }
initialDelaySeconds: 10
livenessProbe:
httpGet: { path: /api/health, port: http }
initialDelaySeconds: 30
resources:
{{- toYaml .Values.observability.grafana.resources | nindent 12 }}
volumeMounts:
- { name: datasource, mountPath: /etc/grafana/provisioning/datasources }
- { name: dashboard-provider, mountPath: /etc/grafana/provisioning/dashboards }
- { name: dashboards, mountPath: /var/lib/grafana/dashboards }
- { name: data, mountPath: /var/lib/grafana }
volumes:
- name: datasource
configMap:
name: {{ include "medical-chatbot.fullname" . }}-grafana-provisioning
items: [{ key: datasources.yml, path: datasources.yml }]
- name: dashboard-provider
configMap:
name: {{ include "medical-chatbot.fullname" . }}-grafana-provisioning
items: [{ key: dashboards.yml, path: dashboards.yml }]
- name: dashboards
configMap:
name: {{ include "medical-chatbot.fullname" . }}-grafana-provisioning
items: [{ key: dashboard.json, path: dashboard.json }]
- name: data
persistentVolumeClaim: { claimName: {{ include "medical-chatbot.fullname" . }}-grafana }
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-grafana
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
spec:
type: {{ .Values.observability.grafana.service.type }}
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: grafana
ports:
- name: http
port: {{ .Values.observability.grafana.service.port }}
targetPort: http
{{- if and (eq .Values.observability.grafana.service.type "NodePort") .Values.observability.grafana.service.nodePort }}
nodePort: {{ .Values.observability.grafana.service.nodePort }}
{{- end }}
{{- end }}
@@ -0,0 +1,13 @@
{{- if .Values.secret.create }}
apiVersion: v1
kind: Secret
metadata:
name: {{ include "medical-chatbot.secretName" . }}
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
type: Opaque
stringData:
postgres-password: {{ .Values.secret.postgresPassword | quote }}
postgres-dsn: {{ printf "postgresql://duoc_thu:%s@%s-postgres:5432/duoc_thu" .Values.secret.postgresPassword (include "medical-chatbot.fullname" .) | quote }}
grafana-admin-password: {{ .Values.secret.grafanaAdminPassword | quote }}
{{- end }}
@@ -0,0 +1,10 @@
{{- if .Values.serviceAccount.create }}
apiVersion: v1
kind: ServiceAccount
metadata:
name: {{ include "medical-chatbot.serviceAccountName" . }}
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
annotations:
{{- toYaml .Values.serviceAccount.annotations | nindent 4 }}
{{- end }}
@@ -0,0 +1,18 @@
{{- if .Values.serviceMonitor.enabled }}
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: {{ include "medical-chatbot.fullname" . }}-ai-service
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
{{- toYaml .Values.serviceMonitor.additionalLabels | nindent 4 }}
spec:
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: ai-service
endpoints:
- port: http
path: /metrics
interval: {{ .Values.serviceMonitor.interval }}
{{- end }}
@@ -0,0 +1,60 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "medical-chatbot.fullname" . }}-web
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: web
spec:
replicas: {{ .Values.web.replicaCount }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: web
template:
metadata:
labels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: web
spec:
imagePullSecrets:
{{- toYaml .Values.global.imagePullSecrets | nindent 8 }}
containers:
- name: web
image: "{{ .Values.web.image.repository }}:{{ .Values.web.image.tag }}"
imagePullPolicy: {{ .Values.web.image.pullPolicy }}
env:
- name: AI_SERVICE_URL
value: {{ printf "http://%s-ai-service:%v" (include "medical-chatbot.fullname" .) .Values.aiService.service.port | quote }}
ports:
- { name: http, containerPort: 3000 }
readinessProbe:
httpGet: { path: /, port: http }
initialDelaySeconds: 5
periodSeconds: 10
livenessProbe:
httpGet: { path: /, port: http }
initialDelaySeconds: 15
periodSeconds: 20
resources:
{{- toYaml .Values.web.resources | nindent 12 }}
---
apiVersion: v1
kind: Service
metadata:
name: {{ include "medical-chatbot.fullname" . }}-web
labels:
{{- include "medical-chatbot.labels" . | nindent 4 }}
app.kubernetes.io/component: web
spec:
type: {{ .Values.web.service.type }}
selector:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: web
ports:
- name: http
port: {{ .Values.web.service.port }}
targetPort: http
{{- if and (eq .Values.web.service.type "NodePort") .Values.web.service.nodePort }}
nodePort: {{ .Values.web.service.nodePort }}
{{- end }}
+17 -1
View File
@@ -1 +1,17 @@
# dev environment overrides (TBD, Phase 6)
global:
environment: docker-desktop
web:
service:
type: NodePort
nodePort: 30080
observability:
grafana:
anonymousAdmin: true
service:
type: NodePort
nodePort: 30082
serviceMonitor:
enabled: false
+31 -1
View File
@@ -1 +1,31 @@
# prod environment overrides (TBD, Phase 6)
global:
environment: production
aiService:
replicaCount: 2
image:
tag: latest
config:
embeddingProvider: cohere-v4
answerProvider: bedrock-converse
otelSampleRatio: 0.25
web:
replicaCount: 2
image:
tag: latest
ingress:
enabled: true
host: realvuxbaro.me
secret:
create: false
existingSecret: medical-chatbot-prod
observability:
prometheus:
retention: 15d
serviceMonitor:
enabled: false
+16 -1
View File
@@ -1 +1,16 @@
# staging environment overrides (TBD, Phase 6)
global:
environment: staging
aiService:
config:
embeddingProvider: cohere-v4
answerProvider: bedrock-converse
otelSampleRatio: 1.0
ingress:
enabled: true
host: staging.duocthu.example.com
secret:
create: false
existingSecret: medical-chatbot-staging
+119 -2
View File
@@ -1,2 +1,119 @@
# Base values — filled in during Phase 6. Overridden per-environment by
# values-dev.yaml / values-staging.yaml / values-prod.yaml.
nameOverride: ""
fullnameOverride: ""
global:
environment: local
imagePullSecrets: []
serviceAccount:
create: true
name: ""
annotations: {}
secret:
create: true
existingSecret: ""
postgresPassword: duoc_thu
grafanaAdminPassword: change-me
aiService:
replicaCount: 1
image:
repository: duocthu-ai-service
tag: local
pullPolicy: IfNotPresent
service:
type: ClusterIP
port: 8000
config:
embeddingProvider: disabled
answerProvider: disabled
answerModelId: deepseek.v3.2
qdrantCollection: duocthu_v1
metricsEnabled: true
otelEnabled: true
otelSampleRatio: 1.0
maxWallClockMs: 40000
maxLlmCallsPerTurn: 8
migration:
enabled: true
resources:
requests: { cpu: 100m, memory: 256Mi }
limits: { cpu: "1", memory: 1Gi }
web:
replicaCount: 1
image:
repository: duocthu-web
tag: local
pullPolicy: IfNotPresent
service:
type: ClusterIP
port: 3000
nodePort: null
resources:
requests: { cpu: 50m, memory: 128Mi }
limits: { cpu: 500m, memory: 512Mi }
ingress:
enabled: false
className: nginx
annotations: {}
host: duocthu.local
tls: []
postgres:
enabled: true
image: postgres:16-alpine
storage: 5Gi
resources:
requests: { cpu: 50m, memory: 128Mi }
limits: { cpu: 500m, memory: 512Mi }
qdrant:
enabled: true
# Set when using an external/managed Qdrant; empty uses the bundled Service.
url: ""
image: qdrant/qdrant:v1.13.4
storage: 10Gi
resources:
requests: { cpu: 100m, memory: 256Mi }
limits: { cpu: "1", memory: 1Gi }
observability:
enabled: true
prometheus:
image: prom/prometheus:v3.3.0
retention: 7d
storage: 5Gi
resources:
requests: { cpu: 100m, memory: 256Mi }
limits: { cpu: 500m, memory: 1Gi }
tempo:
image: grafana/tempo:2.7.2
retention: 24h
storage: 5Gi
resources:
requests: { cpu: 100m, memory: 256Mi }
limits: { cpu: 500m, memory: 1Gi }
collector:
image: otel/opentelemetry-collector-contrib:0.123.0
resources:
requests: { cpu: 50m, memory: 128Mi }
limits: { cpu: 500m, memory: 512Mi }
grafana:
image: grafana/grafana:11.5.2
service:
type: ClusterIP
port: 3000
nodePort: null
storage: 2Gi
anonymousAdmin: false
resources:
requests: { cpu: 50m, memory: 128Mi }
limits: { cpu: 500m, memory: 512Mi }
serviceMonitor:
enabled: false
interval: 15s
additionalLabels: {}