Fix ai-service Dockerfile: bake in drug_entities.json, override its path

This commit is contained in:
2026-08-10 10:35:13 +07:00
parent a4b8e1c4db
commit 60b4397032
51 changed files with 4302 additions and 2087 deletions
+207 -63
View File
@@ -5,12 +5,17 @@ import re
from dataclasses import dataclass, replace
from . import grounding, metrics as metric_names
from .budget import RequestBudget, RequestBudgetExhausted
from .metrics import Metrics, NullMetrics
from .models import EvidenceDecision, QueryIntent, RetrievalResult, SubjectScope
from .ports import AnswerGenerationUnavailable, AnswerGenerator
from .prompt import build_entailment_request, build_request, build_sufficiency_request
from .routing import QueryRoutingService
# See `_verify_entailment`'s docstring for the measured trade-off behind
# widening this from 2 to 3.
_ENTAILMENT_MAX_ATTEMPTS = 3
@dataclass(frozen=True)
class Citation:
@@ -22,6 +27,10 @@ class Citation:
bbox: tuple[float, float, float, float] | None = None
source_crop: str | None = None
attachment: str | None = None
# The exact retrieved text this citation stands for — the same string
# handed to the generator/entailment checks, so the UI can show precisely
# what was retrieved rather than a fabricated summary of it.
evidence_text: str = ""
@dataclass(frozen=True)
@@ -34,12 +43,35 @@ class GroundedAnswer:
# (e.g. a dose question with no age/weight). The answer field carries the
# question; the caller renders it as a clarification, not a final answer.
clarification: str | None = None
# Short suggested replies for `clarification`, e.g. ("Người lớn", "Trẻ
# em") — only populated when the sufficiency check judged the question
# to have a few natural discrete answers, never invented client-side.
quick_replies: tuple[str, ...] = ()
@dataclass(frozen=True)
class _GenOutcome:
answer: str | None = None
clarification: str | None = None
# The specific reason a rejection happened — the exact string already
# used for the GENERATION_REJECTED metric, propagated here so
# `answer_from_result` can put it in the API response's `reason` field
# instead of a generic catch-all. `None` when `answer`/`clarification`
# is set (nothing was rejected).
reject_reason: str | None = None
@dataclass(frozen=True)
class _RawAttempt:
"""One raw `_attempt_generation` call, before any metric is charged —
lets `_generate` retry the noisy `insufficient` case without
double-counting a rejection metric across both attempts."""
answer: str | None = None
clarification: str | None = None
insufficient: bool = False
outage: bool = False
budget_exhausted: bool = False
malformed: bool = False
class GroundedAnswerService:
@@ -57,10 +89,13 @@ class GroundedAnswerService:
pass confirming each cited claim's *content* — not just its numbers —
is actually stated by that block). If a configured generation fails
any check, or the provider itself is unreachable, or its output is
malformed, the turn **abstains** (`reason="generation_unavailable"`
or the specific `grounding.verify` reason) rather than silently
degrading to a raw source dump — this product is a real LLM chatbot,
and a citation-stapled paragraph of book text is not an acceptable
malformed, the turn **abstains** with the specific reason that failed
it (`provider_unavailable`, `malformed_output`, `evidence_insufficient`,
a `grounding.verify` reason, or `unsupported_claim`; falls back to
the generic `generation_unavailable` only if none of those was set)
rather than silently degrading to a raw source dump — this product is
a real LLM chatbot, and a citation-stapled paragraph of book text is
not an acceptable
stand-in for an answer the model was supposed to produce.
"""
@@ -94,12 +129,27 @@ class GroundedAnswerService:
return self.answer_from_result(query, result)
def answer_from_result(
self, query: str, result: RetrievalResult
self, query: str, result: RetrievalResult, list_mode: bool = False,
budget: RequestBudget | None = None,
) -> GroundedAnswer:
"""Everything after retrieval — grounding, sufficiency, generation,
citations. Split out so the new understanding-driven orchestrator
(`rag/agent.py`) reuses the safe answer path without going through the
old `QueryRoutingService` text resolution."""
old `QueryRoutingService` text resolution.
`list_mode=True`: the evidence is several DIFFERENT drugs' own
sections (symptom_to_drug), not alternative phrasings of one drug's
answer — the sufficiency clarify ("which kind of headache?") that's
right for a single dose question doesn't fit a reverse lookup, whose
whole point is to show what the formulary has and let the clinician
narrow it themselves; skipped here the same way a bare-name intro
already skips it.
`budget` (F-08): threaded through to every LLM call this method
makes (sufficiency, generate, up to 2 entailment). `None` (the
default) means unbounded, unchanged from before F-08 — only
`RagAgent` constructs a real budget today.
"""
if result.decision == EvidenceDecision.ABSTAIN:
self._metrics.increment(metric_names.ABSTENTION, reason=result.reason)
return GroundedAnswer(result, None)
@@ -135,11 +185,22 @@ class GroundedAnswerService:
# Reasoning step BEFORE answering: if the turn is under-specified (a dose
# with several bands and no age/weight/condition), ask instead of dumping.
# A separate focused call is more reliable than folding it into generation.
clarify_q = self._check_sufficiency(query, evidence_texts, result.is_drug_overview)
if clarify_q is not None:
return GroundedAnswer(result, clarify_q, (), clarification=clarify_q)
sufficiency = (
None if list_mode
else self._check_sufficiency(
query, evidence_texts, result.is_drug_overview, budget=budget
)
)
if sufficiency is not None:
clarify_q, quick_replies = sufficiency
return GroundedAnswer(
result, clarify_q, (), clarification=clarify_q, quick_replies=quick_replies
)
outcome = self._generate(query, evidence_texts, intro=result.is_drug_overview)
outcome = self._generate(
query, evidence_texts, intro=result.is_drug_overview, list_mode=list_mode,
budget=budget,
)
if outcome.clarification is not None:
# The model judged the turn under-specified (a dose with no
# age/weight/renal-function/indication…) and asked back instead of
@@ -163,14 +224,23 @@ class GroundedAnswerService:
# source dump is not an acceptable stand-in for a failed
# generation, so this abstains instead of silently degrading to
# one.
self._metrics.increment(
metric_names.ABSTENTION, reason="generation_unavailable"
)
# The specific check that failed (provider_unavailable,
# malformed_output, evidence_insufficient, ungrounded_number,
# uncited_claim, unsupported_claim, request_budget_exhausted) —
# found live 2026-08-07: every one of these used to collapse into
# the same generic "generation_unavailable" by the time it
# reached the API response/trace, so a real, diagnosable cause
# (e.g. a genuine provider outage) was indistinguishable from
# ordinary entailment noise without reading server-side metrics
# by hand. `outcome.reject_reason` already carries the granular
# value the metric above uses — just propagate it.
reason = outcome.reject_reason or "generation_unavailable"
self._metrics.increment(metric_names.ABSTENTION, reason=reason)
return GroundedAnswer(
replace(
result,
decision=EvidenceDecision.ABSTAIN,
reason="generation_unavailable",
reason=reason,
),
None,
)
@@ -182,67 +252,107 @@ class GroundedAnswerService:
self._metrics.increment(metric_names.GENERATION_SERVED)
return GroundedAnswer(result, outcome.answer, citations, generated=True)
def _generate(
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False
) -> "_GenOutcome":
"""A verified generation, a clarifying question, or empty to fall back."""
if self._generator is None or not evidence_texts:
return _GenOutcome()
request = build_request(query, evidence_texts, intro=intro)
def _attempt_generation(
self, request: "GenerationRequest", budget: RequestBudget | None
) -> "_RawAttempt":
"""One raw generation call, parsed but not yet metric-counted or
verified — the caller decides whether to retry before charging a
metric to any particular reason."""
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except RequestBudgetExhausted:
return _RawAttempt(budget_exhausted=True)
except AnswerGenerationUnavailable:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
)
return _GenOutcome()
return _RawAttempt(outage=True)
try:
payload = json.loads(raw)
answer = payload["answer"]
sufficient = payload["evidence_sufficient"]
except (ValueError, TypeError, KeyError):
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="malformed_output"
)
return _GenOutcome()
return _RawAttempt(malformed=True)
# The model asked for a missing detail (age/weight/renal function/
# indication…) instead of listing every band. A clarify is not a grounded
# claim, so it skips the number check — it states no dose.
clarify = payload.get("clarifying_question") if isinstance(payload, dict) else None
if isinstance(clarify, str) and clarify.strip():
return _GenOutcome(clarification=clarify.strip())
return _RawAttempt(clarification=clarify.strip())
if not isinstance(answer, str) or not isinstance(sufficient, bool):
return _RawAttempt(malformed=True)
if not sufficient:
return _RawAttempt(insufficient=True)
return _RawAttempt(answer=answer)
def _generate(
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False,
list_mode: bool = False, budget: RequestBudget | None = None,
) -> "_GenOutcome":
"""A verified generation, a clarifying question, or empty to fall back."""
if self._generator is None or not evidence_texts:
return _GenOutcome()
request = build_request(query, evidence_texts, intro=intro, list_mode=list_mode)
attempt = self._attempt_generation(request, budget)
if attempt.insufficient:
# Empirically noisy (found live 2026-08-07, reproduced 3/3 on a
# fresh retry): the model's own evidence_sufficient=false
# self-assessment sometimes flips to a correct, fully grounded
# answer when asked again with the IDENTICAL evidence — the same
# one-retry pattern `_verify_entailment` already uses below for
# its own noisy judge call. Only the terminal "insufficient AND
# no clarifying question" case retries; a legitimate ask-for-
# more-detail clarify is untouched.
attempt = self._attempt_generation(request, budget)
if attempt.budget_exhausted:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="request_budget_exhausted"
)
return _GenOutcome(reject_reason="request_budget_exhausted")
if attempt.outage:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
)
return _GenOutcome(reject_reason="provider_unavailable")
if attempt.malformed:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="malformed_output"
)
return _GenOutcome()
if not sufficient:
# The model says the evidence does not answer the question. Showing
# the retrieved section verbatim lets the clinician judge that.
return _GenOutcome(reject_reason="malformed_output")
if attempt.clarification is not None:
return _GenOutcome(clarification=attempt.clarification)
if attempt.insufficient:
# The model says the evidence does not answer the question, on
# both attempts. Showing the retrieved section verbatim lets the
# clinician judge that.
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="evidence_insufficient"
)
return _GenOutcome()
return _GenOutcome(reject_reason="evidence_insufficient")
answer = attempt.answer
report = grounding.verify(answer, evidence_texts)
if not report.grounded:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason=report.reason
)
return _GenOutcome()
return _GenOutcome(reject_reason=report.reason)
if not self._verify_entailment(answer, evidence_texts):
if not self._verify_entailment(answer, evidence_texts, budget=budget):
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="unsupported_claim"
)
return _GenOutcome()
return _GenOutcome(reject_reason="unsupported_claim")
return _GenOutcome(answer=answer)
def _verify_entailment(self, answer: str, evidence_texts: tuple[str, ...]) -> bool:
def _verify_entailment(
self, answer: str, evidence_texts: tuple[str, ...],
budget: RequestBudget | None = None,
) -> bool:
"""A second, adversarial LLM pass over an answer that already passed
`grounding.verify`.
@@ -254,14 +364,22 @@ class GroundedAnswerService:
is checked against only the evidence block(s) it names, by a model
told to compare wording, not to reason about medicine.
Fails closed on an outage or malformed output. A single rejection is
NOT: live probing (2026-08-06) found the judge call itself is noisy
— the identical claim/evidence pair, called three times, came back
entailed twice and rejected once, discarding a correct, well-cited
interaction answer. So a reject triggers one same-claim retry, and
only a second, agreeing reject discards the generation; a single
provider outage/malformed response still fails closed immediately
(that failure mode is reliable, not noisy — no retry needed there).
Fails closed on an outage or malformed output — that failure mode is
reliable, not noisy, so it stops immediately rather than spending
retries on it. A single rejection is NOT reliable: live probing
(2026-08-06) found the judge call itself is noisy — the identical
claim/evidence pair, called three times, came back entailed twice
and rejected once, discarding a correct, well-cited interaction
answer. Up to `_ENTAILMENT_MAX_ATTEMPTS` same-claim calls run;
accept on the first `True`, discard only if every attempt agrees
reject. Widened from 2 to 3 attempts 2026-08-07 after a live
adversarial sample (50 real questions) measured this specific check
as roughly half of all false abstentions on genuinely answerable
questions. Trade-off, stated plainly: this raises the bar a
genuinely fabricated claim must now clear too (it survives if ANY
one of 3 noisy calls wrongly accepts it, not just 1 of 2) — accepted
because the probed noise is symmetric and the entailment prompt
itself is unchanged, not because the risk is zero.
An answer with no claim text at all (nothing between or after its
citation markers) is vacuously fine — nothing to verify, no call.
"""
@@ -274,18 +392,23 @@ class GroundedAnswerService:
return True
request = build_entailment_request(claims)
first = self._run_entailment_check(request)
if first is None:
return False
if first:
return True
second = self._run_entailment_check(request)
return bool(second)
for _ in range(_ENTAILMENT_MAX_ATTEMPTS):
verdict = self._run_entailment_check(request, budget=budget)
if verdict is None:
return False
if verdict:
return True
return False
def _run_entailment_check(self, request) -> bool | None:
"""One entailment call. `None` = outage/malformed (fails closed by the
caller without a retry); `True`/`False` = the judge's verdict."""
def _run_entailment_check(
self, request, budget: RequestBudget | None = None
) -> bool | None:
"""One entailment call. `None` = outage/malformed/budget-exhausted
(fails closed by the caller without a retry); `True`/`False` = the
judge's verdict."""
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except AnswerGenerationUnavailable:
return None
@@ -300,17 +423,31 @@ class GroundedAnswerService:
return entailed and not unsupported
def _check_sufficiency(
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False
) -> str | None:
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False,
budget: RequestBudget | None = None,
) -> tuple[str, tuple[str, ...]] | None:
"""A focused reasoning call: is the turn specific enough to answer, or
must we ask? Returns a clarifying question, or None to proceed.
must we ask? Returns (clarifying_question, quick_replies), or None to
proceed. `quick_replies` is often empty — only populated when the
model judged the missing detail has a few natural discrete answers
(e.g. "Người lớn"/"Trẻ em"), never invented here.
Skipped without a model, for a bare-name intro (not a dose), or for a
single evidence block (nothing to disambiguate)."""
single evidence block (nothing to disambiguate).
Fails OPEN on outage/budget-exhaustion (returns None, proceeds to
generate) — deliberately different from every other call in this
file, which fail closed. This is a reasoning heuristic, not a safety
check; grounding + entailment remain the real gate on whatever gets
generated next, so skipping this one costs UX quality (a dose
question that should have asked for age/weight might not), not
safety."""
if self._generator is None or intro or len(evidence_texts) < 2:
return None
request = build_sufficiency_request(query, evidence_texts)
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except AnswerGenerationUnavailable:
return None
@@ -321,7 +458,13 @@ class GroundedAnswerService:
if isinstance(payload, dict) and payload.get("sufficient") is False:
question = payload.get("clarifying_question")
if isinstance(question, str) and question.strip():
return question.strip()
raw_replies = payload.get("quick_replies")
replies = tuple(
reply.strip()
for reply in raw_replies
if isinstance(reply, str) and reply.strip()
) if isinstance(raw_replies, list) else ()
return question.strip(), replies
return None
@staticmethod
@@ -362,5 +505,6 @@ class GroundedAnswerService:
# real crop path wins; otherwise the block id plus the
# structured page/bbox fields is enough to render later.
attachment=source.source_crop or source.block_id,
evidence_text=evidence.text,
)))
return citations