Fix ai-service Dockerfile: bake in drug_entities.json, override its path
This commit is contained in:
+207
-63
@@ -5,12 +5,17 @@ import re
|
||||
from dataclasses import dataclass, replace
|
||||
|
||||
from . import grounding, metrics as metric_names
|
||||
from .budget import RequestBudget, RequestBudgetExhausted
|
||||
from .metrics import Metrics, NullMetrics
|
||||
from .models import EvidenceDecision, QueryIntent, RetrievalResult, SubjectScope
|
||||
from .ports import AnswerGenerationUnavailable, AnswerGenerator
|
||||
from .prompt import build_entailment_request, build_request, build_sufficiency_request
|
||||
from .routing import QueryRoutingService
|
||||
|
||||
# See `_verify_entailment`'s docstring for the measured trade-off behind
|
||||
# widening this from 2 to 3.
|
||||
_ENTAILMENT_MAX_ATTEMPTS = 3
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Citation:
|
||||
@@ -22,6 +27,10 @@ class Citation:
|
||||
bbox: tuple[float, float, float, float] | None = None
|
||||
source_crop: str | None = None
|
||||
attachment: str | None = None
|
||||
# The exact retrieved text this citation stands for — the same string
|
||||
# handed to the generator/entailment checks, so the UI can show precisely
|
||||
# what was retrieved rather than a fabricated summary of it.
|
||||
evidence_text: str = ""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -34,12 +43,35 @@ class GroundedAnswer:
|
||||
# (e.g. a dose question with no age/weight). The answer field carries the
|
||||
# question; the caller renders it as a clarification, not a final answer.
|
||||
clarification: str | None = None
|
||||
# Short suggested replies for `clarification`, e.g. ("Người lớn", "Trẻ
|
||||
# em") — only populated when the sufficiency check judged the question
|
||||
# to have a few natural discrete answers, never invented client-side.
|
||||
quick_replies: tuple[str, ...] = ()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _GenOutcome:
|
||||
answer: str | None = None
|
||||
clarification: str | None = None
|
||||
# The specific reason a rejection happened — the exact string already
|
||||
# used for the GENERATION_REJECTED metric, propagated here so
|
||||
# `answer_from_result` can put it in the API response's `reason` field
|
||||
# instead of a generic catch-all. `None` when `answer`/`clarification`
|
||||
# is set (nothing was rejected).
|
||||
reject_reason: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _RawAttempt:
|
||||
"""One raw `_attempt_generation` call, before any metric is charged —
|
||||
lets `_generate` retry the noisy `insufficient` case without
|
||||
double-counting a rejection metric across both attempts."""
|
||||
answer: str | None = None
|
||||
clarification: str | None = None
|
||||
insufficient: bool = False
|
||||
outage: bool = False
|
||||
budget_exhausted: bool = False
|
||||
malformed: bool = False
|
||||
|
||||
|
||||
class GroundedAnswerService:
|
||||
@@ -57,10 +89,13 @@ class GroundedAnswerService:
|
||||
pass confirming each cited claim's *content* — not just its numbers —
|
||||
is actually stated by that block). If a configured generation fails
|
||||
any check, or the provider itself is unreachable, or its output is
|
||||
malformed, the turn **abstains** (`reason="generation_unavailable"`
|
||||
or the specific `grounding.verify` reason) rather than silently
|
||||
degrading to a raw source dump — this product is a real LLM chatbot,
|
||||
and a citation-stapled paragraph of book text is not an acceptable
|
||||
malformed, the turn **abstains** with the specific reason that failed
|
||||
it (`provider_unavailable`, `malformed_output`, `evidence_insufficient`,
|
||||
a `grounding.verify` reason, or `unsupported_claim`; falls back to
|
||||
the generic `generation_unavailable` only if none of those was set)
|
||||
rather than silently degrading to a raw source dump — this product is
|
||||
a real LLM chatbot, and a citation-stapled paragraph of book text is
|
||||
not an acceptable
|
||||
stand-in for an answer the model was supposed to produce.
|
||||
"""
|
||||
|
||||
@@ -94,12 +129,27 @@ class GroundedAnswerService:
|
||||
return self.answer_from_result(query, result)
|
||||
|
||||
def answer_from_result(
|
||||
self, query: str, result: RetrievalResult
|
||||
self, query: str, result: RetrievalResult, list_mode: bool = False,
|
||||
budget: RequestBudget | None = None,
|
||||
) -> GroundedAnswer:
|
||||
"""Everything after retrieval — grounding, sufficiency, generation,
|
||||
citations. Split out so the new understanding-driven orchestrator
|
||||
(`rag/agent.py`) reuses the safe answer path without going through the
|
||||
old `QueryRoutingService` text resolution."""
|
||||
old `QueryRoutingService` text resolution.
|
||||
|
||||
`list_mode=True`: the evidence is several DIFFERENT drugs' own
|
||||
sections (symptom_to_drug), not alternative phrasings of one drug's
|
||||
answer — the sufficiency clarify ("which kind of headache?") that's
|
||||
right for a single dose question doesn't fit a reverse lookup, whose
|
||||
whole point is to show what the formulary has and let the clinician
|
||||
narrow it themselves; skipped here the same way a bare-name intro
|
||||
already skips it.
|
||||
|
||||
`budget` (F-08): threaded through to every LLM call this method
|
||||
makes (sufficiency, generate, up to 2 entailment). `None` (the
|
||||
default) means unbounded, unchanged from before F-08 — only
|
||||
`RagAgent` constructs a real budget today.
|
||||
"""
|
||||
if result.decision == EvidenceDecision.ABSTAIN:
|
||||
self._metrics.increment(metric_names.ABSTENTION, reason=result.reason)
|
||||
return GroundedAnswer(result, None)
|
||||
@@ -135,11 +185,22 @@ class GroundedAnswerService:
|
||||
# Reasoning step BEFORE answering: if the turn is under-specified (a dose
|
||||
# with several bands and no age/weight/condition), ask instead of dumping.
|
||||
# A separate focused call is more reliable than folding it into generation.
|
||||
clarify_q = self._check_sufficiency(query, evidence_texts, result.is_drug_overview)
|
||||
if clarify_q is not None:
|
||||
return GroundedAnswer(result, clarify_q, (), clarification=clarify_q)
|
||||
sufficiency = (
|
||||
None if list_mode
|
||||
else self._check_sufficiency(
|
||||
query, evidence_texts, result.is_drug_overview, budget=budget
|
||||
)
|
||||
)
|
||||
if sufficiency is not None:
|
||||
clarify_q, quick_replies = sufficiency
|
||||
return GroundedAnswer(
|
||||
result, clarify_q, (), clarification=clarify_q, quick_replies=quick_replies
|
||||
)
|
||||
|
||||
outcome = self._generate(query, evidence_texts, intro=result.is_drug_overview)
|
||||
outcome = self._generate(
|
||||
query, evidence_texts, intro=result.is_drug_overview, list_mode=list_mode,
|
||||
budget=budget,
|
||||
)
|
||||
if outcome.clarification is not None:
|
||||
# The model judged the turn under-specified (a dose with no
|
||||
# age/weight/renal-function/indication…) and asked back instead of
|
||||
@@ -163,14 +224,23 @@ class GroundedAnswerService:
|
||||
# source dump is not an acceptable stand-in for a failed
|
||||
# generation, so this abstains instead of silently degrading to
|
||||
# one.
|
||||
self._metrics.increment(
|
||||
metric_names.ABSTENTION, reason="generation_unavailable"
|
||||
)
|
||||
# The specific check that failed (provider_unavailable,
|
||||
# malformed_output, evidence_insufficient, ungrounded_number,
|
||||
# uncited_claim, unsupported_claim, request_budget_exhausted) —
|
||||
# found live 2026-08-07: every one of these used to collapse into
|
||||
# the same generic "generation_unavailable" by the time it
|
||||
# reached the API response/trace, so a real, diagnosable cause
|
||||
# (e.g. a genuine provider outage) was indistinguishable from
|
||||
# ordinary entailment noise without reading server-side metrics
|
||||
# by hand. `outcome.reject_reason` already carries the granular
|
||||
# value the metric above uses — just propagate it.
|
||||
reason = outcome.reject_reason or "generation_unavailable"
|
||||
self._metrics.increment(metric_names.ABSTENTION, reason=reason)
|
||||
return GroundedAnswer(
|
||||
replace(
|
||||
result,
|
||||
decision=EvidenceDecision.ABSTAIN,
|
||||
reason="generation_unavailable",
|
||||
reason=reason,
|
||||
),
|
||||
None,
|
||||
)
|
||||
@@ -182,67 +252,107 @@ class GroundedAnswerService:
|
||||
self._metrics.increment(metric_names.GENERATION_SERVED)
|
||||
return GroundedAnswer(result, outcome.answer, citations, generated=True)
|
||||
|
||||
def _generate(
|
||||
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False
|
||||
) -> "_GenOutcome":
|
||||
"""A verified generation, a clarifying question, or empty to fall back."""
|
||||
if self._generator is None or not evidence_texts:
|
||||
return _GenOutcome()
|
||||
|
||||
request = build_request(query, evidence_texts, intro=intro)
|
||||
def _attempt_generation(
|
||||
self, request: "GenerationRequest", budget: RequestBudget | None
|
||||
) -> "_RawAttempt":
|
||||
"""One raw generation call, parsed but not yet metric-counted or
|
||||
verified — the caller decides whether to retry before charging a
|
||||
metric to any particular reason."""
|
||||
try:
|
||||
if budget is not None:
|
||||
budget.require()
|
||||
raw = self._generator.generate(request.system, request.user, request.schema)
|
||||
except RequestBudgetExhausted:
|
||||
return _RawAttempt(budget_exhausted=True)
|
||||
except AnswerGenerationUnavailable:
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
|
||||
)
|
||||
return _GenOutcome()
|
||||
return _RawAttempt(outage=True)
|
||||
|
||||
try:
|
||||
payload = json.loads(raw)
|
||||
answer = payload["answer"]
|
||||
sufficient = payload["evidence_sufficient"]
|
||||
except (ValueError, TypeError, KeyError):
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="malformed_output"
|
||||
)
|
||||
return _GenOutcome()
|
||||
return _RawAttempt(malformed=True)
|
||||
|
||||
# The model asked for a missing detail (age/weight/renal function/
|
||||
# indication…) instead of listing every band. A clarify is not a grounded
|
||||
# claim, so it skips the number check — it states no dose.
|
||||
clarify = payload.get("clarifying_question") if isinstance(payload, dict) else None
|
||||
if isinstance(clarify, str) and clarify.strip():
|
||||
return _GenOutcome(clarification=clarify.strip())
|
||||
return _RawAttempt(clarification=clarify.strip())
|
||||
|
||||
if not isinstance(answer, str) or not isinstance(sufficient, bool):
|
||||
return _RawAttempt(malformed=True)
|
||||
if not sufficient:
|
||||
return _RawAttempt(insufficient=True)
|
||||
return _RawAttempt(answer=answer)
|
||||
|
||||
def _generate(
|
||||
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False,
|
||||
list_mode: bool = False, budget: RequestBudget | None = None,
|
||||
) -> "_GenOutcome":
|
||||
"""A verified generation, a clarifying question, or empty to fall back."""
|
||||
if self._generator is None or not evidence_texts:
|
||||
return _GenOutcome()
|
||||
|
||||
request = build_request(query, evidence_texts, intro=intro, list_mode=list_mode)
|
||||
attempt = self._attempt_generation(request, budget)
|
||||
if attempt.insufficient:
|
||||
# Empirically noisy (found live 2026-08-07, reproduced 3/3 on a
|
||||
# fresh retry): the model's own evidence_sufficient=false
|
||||
# self-assessment sometimes flips to a correct, fully grounded
|
||||
# answer when asked again with the IDENTICAL evidence — the same
|
||||
# one-retry pattern `_verify_entailment` already uses below for
|
||||
# its own noisy judge call. Only the terminal "insufficient AND
|
||||
# no clarifying question" case retries; a legitimate ask-for-
|
||||
# more-detail clarify is untouched.
|
||||
attempt = self._attempt_generation(request, budget)
|
||||
|
||||
if attempt.budget_exhausted:
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="request_budget_exhausted"
|
||||
)
|
||||
return _GenOutcome(reject_reason="request_budget_exhausted")
|
||||
if attempt.outage:
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
|
||||
)
|
||||
return _GenOutcome(reject_reason="provider_unavailable")
|
||||
if attempt.malformed:
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="malformed_output"
|
||||
)
|
||||
return _GenOutcome()
|
||||
if not sufficient:
|
||||
# The model says the evidence does not answer the question. Showing
|
||||
# the retrieved section verbatim lets the clinician judge that.
|
||||
return _GenOutcome(reject_reason="malformed_output")
|
||||
if attempt.clarification is not None:
|
||||
return _GenOutcome(clarification=attempt.clarification)
|
||||
if attempt.insufficient:
|
||||
# The model says the evidence does not answer the question, on
|
||||
# both attempts. Showing the retrieved section verbatim lets the
|
||||
# clinician judge that.
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="evidence_insufficient"
|
||||
)
|
||||
return _GenOutcome()
|
||||
return _GenOutcome(reject_reason="evidence_insufficient")
|
||||
|
||||
answer = attempt.answer
|
||||
report = grounding.verify(answer, evidence_texts)
|
||||
if not report.grounded:
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason=report.reason
|
||||
)
|
||||
return _GenOutcome()
|
||||
return _GenOutcome(reject_reason=report.reason)
|
||||
|
||||
if not self._verify_entailment(answer, evidence_texts):
|
||||
if not self._verify_entailment(answer, evidence_texts, budget=budget):
|
||||
self._metrics.increment(
|
||||
metric_names.GENERATION_REJECTED, reason="unsupported_claim"
|
||||
)
|
||||
return _GenOutcome()
|
||||
return _GenOutcome(reject_reason="unsupported_claim")
|
||||
return _GenOutcome(answer=answer)
|
||||
|
||||
def _verify_entailment(self, answer: str, evidence_texts: tuple[str, ...]) -> bool:
|
||||
def _verify_entailment(
|
||||
self, answer: str, evidence_texts: tuple[str, ...],
|
||||
budget: RequestBudget | None = None,
|
||||
) -> bool:
|
||||
"""A second, adversarial LLM pass over an answer that already passed
|
||||
`grounding.verify`.
|
||||
|
||||
@@ -254,14 +364,22 @@ class GroundedAnswerService:
|
||||
is checked against only the evidence block(s) it names, by a model
|
||||
told to compare wording, not to reason about medicine.
|
||||
|
||||
Fails closed on an outage or malformed output. A single rejection is
|
||||
NOT: live probing (2026-08-06) found the judge call itself is noisy
|
||||
— the identical claim/evidence pair, called three times, came back
|
||||
entailed twice and rejected once, discarding a correct, well-cited
|
||||
interaction answer. So a reject triggers one same-claim retry, and
|
||||
only a second, agreeing reject discards the generation; a single
|
||||
provider outage/malformed response still fails closed immediately
|
||||
(that failure mode is reliable, not noisy — no retry needed there).
|
||||
Fails closed on an outage or malformed output — that failure mode is
|
||||
reliable, not noisy, so it stops immediately rather than spending
|
||||
retries on it. A single rejection is NOT reliable: live probing
|
||||
(2026-08-06) found the judge call itself is noisy — the identical
|
||||
claim/evidence pair, called three times, came back entailed twice
|
||||
and rejected once, discarding a correct, well-cited interaction
|
||||
answer. Up to `_ENTAILMENT_MAX_ATTEMPTS` same-claim calls run;
|
||||
accept on the first `True`, discard only if every attempt agrees
|
||||
reject. Widened from 2 to 3 attempts 2026-08-07 after a live
|
||||
adversarial sample (50 real questions) measured this specific check
|
||||
as roughly half of all false abstentions on genuinely answerable
|
||||
questions. Trade-off, stated plainly: this raises the bar a
|
||||
genuinely fabricated claim must now clear too (it survives if ANY
|
||||
one of 3 noisy calls wrongly accepts it, not just 1 of 2) — accepted
|
||||
because the probed noise is symmetric and the entailment prompt
|
||||
itself is unchanged, not because the risk is zero.
|
||||
An answer with no claim text at all (nothing between or after its
|
||||
citation markers) is vacuously fine — nothing to verify, no call.
|
||||
"""
|
||||
@@ -274,18 +392,23 @@ class GroundedAnswerService:
|
||||
return True
|
||||
|
||||
request = build_entailment_request(claims)
|
||||
first = self._run_entailment_check(request)
|
||||
if first is None:
|
||||
return False
|
||||
if first:
|
||||
return True
|
||||
second = self._run_entailment_check(request)
|
||||
return bool(second)
|
||||
for _ in range(_ENTAILMENT_MAX_ATTEMPTS):
|
||||
verdict = self._run_entailment_check(request, budget=budget)
|
||||
if verdict is None:
|
||||
return False
|
||||
if verdict:
|
||||
return True
|
||||
return False
|
||||
|
||||
def _run_entailment_check(self, request) -> bool | None:
|
||||
"""One entailment call. `None` = outage/malformed (fails closed by the
|
||||
caller without a retry); `True`/`False` = the judge's verdict."""
|
||||
def _run_entailment_check(
|
||||
self, request, budget: RequestBudget | None = None
|
||||
) -> bool | None:
|
||||
"""One entailment call. `None` = outage/malformed/budget-exhausted
|
||||
(fails closed by the caller without a retry); `True`/`False` = the
|
||||
judge's verdict."""
|
||||
try:
|
||||
if budget is not None:
|
||||
budget.require()
|
||||
raw = self._generator.generate(request.system, request.user, request.schema)
|
||||
except AnswerGenerationUnavailable:
|
||||
return None
|
||||
@@ -300,17 +423,31 @@ class GroundedAnswerService:
|
||||
return entailed and not unsupported
|
||||
|
||||
def _check_sufficiency(
|
||||
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False
|
||||
) -> str | None:
|
||||
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False,
|
||||
budget: RequestBudget | None = None,
|
||||
) -> tuple[str, tuple[str, ...]] | None:
|
||||
"""A focused reasoning call: is the turn specific enough to answer, or
|
||||
must we ask? Returns a clarifying question, or None to proceed.
|
||||
must we ask? Returns (clarifying_question, quick_replies), or None to
|
||||
proceed. `quick_replies` is often empty — only populated when the
|
||||
model judged the missing detail has a few natural discrete answers
|
||||
(e.g. "Người lớn"/"Trẻ em"), never invented here.
|
||||
|
||||
Skipped without a model, for a bare-name intro (not a dose), or for a
|
||||
single evidence block (nothing to disambiguate)."""
|
||||
single evidence block (nothing to disambiguate).
|
||||
|
||||
Fails OPEN on outage/budget-exhaustion (returns None, proceeds to
|
||||
generate) — deliberately different from every other call in this
|
||||
file, which fail closed. This is a reasoning heuristic, not a safety
|
||||
check; grounding + entailment remain the real gate on whatever gets
|
||||
generated next, so skipping this one costs UX quality (a dose
|
||||
question that should have asked for age/weight might not), not
|
||||
safety."""
|
||||
if self._generator is None or intro or len(evidence_texts) < 2:
|
||||
return None
|
||||
request = build_sufficiency_request(query, evidence_texts)
|
||||
try:
|
||||
if budget is not None:
|
||||
budget.require()
|
||||
raw = self._generator.generate(request.system, request.user, request.schema)
|
||||
except AnswerGenerationUnavailable:
|
||||
return None
|
||||
@@ -321,7 +458,13 @@ class GroundedAnswerService:
|
||||
if isinstance(payload, dict) and payload.get("sufficient") is False:
|
||||
question = payload.get("clarifying_question")
|
||||
if isinstance(question, str) and question.strip():
|
||||
return question.strip()
|
||||
raw_replies = payload.get("quick_replies")
|
||||
replies = tuple(
|
||||
reply.strip()
|
||||
for reply in raw_replies
|
||||
if isinstance(reply, str) and reply.strip()
|
||||
) if isinstance(raw_replies, list) else ()
|
||||
return question.strip(), replies
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
@@ -362,5 +505,6 @@ class GroundedAnswerService:
|
||||
# real crop path wins; otherwise the block id plus the
|
||||
# structured page/bbox fields is enough to render later.
|
||||
attachment=source.source_crop or source.block_id,
|
||||
evidence_text=evidence.text,
|
||||
)))
|
||||
return citations
|
||||
|
||||
Reference in New Issue
Block a user