Files
duocthu/apps/ai-service/rag/answer.py
T
2026-08-14 11:57:58 +07:00

1195 lines
52 KiB
Python

from __future__ import annotations
import json
import logging
import re
import unicodedata
from dataclasses import dataclass, replace
from . import grounding, metrics as metric_names
from .budget import RequestBudget, RequestBudgetExhausted
from .metrics import Metrics, NullMetrics
from .models import EvidenceDecision, QueryIntent, RetrievalResult, SubjectScope
from .ports import AnswerGenerationUnavailable, AnswerGenerator
from .prompt import (
GenerationRequest,
build_entailment_request,
build_request,
build_sufficiency_request,
)
from .routing import QueryRoutingService
# Repeating the same temperature-0 prompt against the same model is a
# correlated retry, not an independent vote. One fail-closed semantic pass is
# kept after structured-claim parsing and deterministic grounding.
#
# This is expressed as a single call rather than a `range(1)` loop on
# purpose: the loop this replaced returned on its first iteration on every
# path, so raising the constant looked like it added retries while silently
# doing nothing. If a future change genuinely wants more passes, they must
# be different prompts (or a different judge) to be independent evidence —
# see this function's own reasoning above.
_QUICK_REPLY_MAX_ITEMS = 18 # one per monograph section (see rag/sections.py SECTION_ORDER)
_QUICK_REPLY_MAX_CHARS = 40
logger = logging.getLogger(__name__)
def _sanitize_quick_replies(value) -> tuple[str, ...]:
"""Validate LLM-proposed chips without hard-coding their content."""
if not isinstance(value, list):
return ()
replies: list[str] = []
seen: set[str] = set()
for item in value:
if not isinstance(item, str):
continue
reply = " ".join(item.split())
key = reply.casefold()
if not reply or len(reply) > _QUICK_REPLY_MAX_CHARS or key in seen:
continue
seen.add(key)
replies.append(reply)
if len(replies) >= _QUICK_REPLY_MAX_ITEMS:
break
return tuple(replies)
@dataclass(frozen=True)
class Citation:
chunk_id: str
printed_page_start: int
printed_page_end: int
physical_page: int
block_id: str | None = None
bbox: tuple[float, float, float, float] | None = None
source_crop: str | None = None
attachment: str | None = None
# The exact retrieved text this citation stands for — the same string
# handed to the generator/entailment checks, so the UI can show precisely
# what was retrieved rather than a fabricated summary of it.
evidence_text: str = ""
drug_id: str | None = None
drug_name: str | None = None
section_key: str | None = None
section_title: str | None = None
source_document: str = "Dược thư Quốc gia Việt Nam 2018"
@dataclass(frozen=True)
class AnswerClaim:
"""One user-visible fact and the exact retrieved chunks supporting it."""
text: str
source_ids: tuple[str, ...] = ()
@dataclass(frozen=True)
class AnswerBlock:
"""Semantic presentation unit produced by the answer service, not the UI."""
title: str
kind: str
claims: tuple[AnswerClaim, ...]
@dataclass(frozen=True)
class AnswerPlan:
"""Compact, non-reasoning presentation plan decided before generation."""
verbosity: str
layout: str
reasoning_mode: str
show_heading: bool = False
needs_warning: bool = False
# A fixed, non-LLM string. `docs/architecture.md`'s guardrail section
# specifies the disclaimer at several layers, and the web banner
# (`packages/ui/src/DisclaimerBanner.tsx`) was the only one in place: the
# `disclaimer` field declared in `packages/shared-types/src/dto/chat.ts` was
# never populated, so any consumer other than this one web UI received medical
# content with nothing attached. Keeping it out of the prompt is deliberate —
# a disclaimer the model writes is one the model can also reword, shorten or
# omit, and it would then have to be verified like any other generated claim.
DISCLAIMER = (
"Nội dung được trích từ Dược thư Quốc gia Việt Nam 2018, phục vụ tra cứu "
"chuyên môn và không thay thế chỉ định của bác sĩ hoặc dược sĩ lâm sàng."
)
@dataclass(frozen=True)
class GroundedAnswer:
result: RetrievalResult
answer: str | None
citations: tuple[Citation, ...] = ()
generated: bool = False
# Carried on the payload rather than added by the caller, so no response
# path can be built that omits it — including abstains and clarifications,
# which are also medical content in the sense that matters here.
disclaimer: str = DISCLAIMER
# Set when the model decided the turn is under-specified and asked back
# (e.g. a dose question with no age/weight). The answer field carries the
# question; the caller renders it as a clarification, not a final answer.
clarification: str | None = None
# Short suggested replies for `clarification`, e.g. ("Người lớn", "Trẻ
# em") — only populated when the sufficiency check judged the question
# to have a few natural discrete answers, never invented client-side.
quick_replies: tuple[str, ...] = ()
blocks: tuple[AnswerBlock, ...] = ()
answer_mode: str = "concise"
plan: AnswerPlan | None = None
@dataclass(frozen=True)
class _CheckNotRun:
"""The entailment judge could not be consulted at all.
Distinct from a negative verdict on purpose. Fail-closed behaviour is
identical either way — the answer is still discarded — but the reason
code should not report an unsupported claim when the check never ran.
`docs/current-rag-pipeline-audit.md` §4's failure taxonomy keeps
availability and content failures separate for this reason. Observed
live 2026-08-11: a request that ran out of wall-clock budget mid-
verification reached the user as "bước đối chiếu chưa xác nhận được câu
trả lời khớp với nguồn", which describes the answer rather than the
timeout that actually occurred.
`reason` is deliberately one of the codes that already exist and are
already mapped in `apps/web/app/api/chat/route.ts`; a code with no entry
there falls back to wording that reads as "no data in the formulary",
which would misdescribe these cases.
"""
reason: str
@dataclass(frozen=True)
class _GenOutcome:
answer: str | None = None
clarification: str | None = None
quick_replies: tuple[str, ...] = ()
# The specific reason a rejection happened — the exact string already
# used for the GENERATION_REJECTED metric, propagated here so
# `answer_from_result` can put it in the API response's `reason` field
# instead of a generic catch-all. `None` when `answer`/`clarification`
# is set (nothing was rejected).
reject_reason: str | None = None
claims: tuple[tuple[str, tuple[int, ...]], ...] = ()
@dataclass(frozen=True)
class _RawAttempt:
"""One raw `_attempt_generation` call, before any metric is charged —
lets `_generate` retry the noisy `insufficient` case without
double-counting a rejection metric across both attempts."""
answer: str | None = None
# The model's own claim/citation boundaries (2026-08-10 structured-claims
# change) — `_verify_entailment` checks these directly instead of
# re-deriving claim boundaries from `answer` via regex. `answer` is still
# populated (assembled from these by `_assemble_answer`) because it's
# the exact string `grounding.verify` and the API response need — one
# format, not two representations that could drift apart.
claims: tuple[tuple[str, tuple[int, ...]], ...] = ()
clarification: str | None = None
quick_replies: tuple[str, ...] = ()
insufficient: bool = False
outage: bool = False
budget_exhausted: bool = False
malformed: bool = False
unsupported_drug: bool = False
@dataclass(frozen=True)
class _VerificationOutcome:
supported: bool
complete: bool
missing: tuple[str, ...] = ()
# Display-only label override, 2026-08-13: these 3 of 684 catalog drug_ids are
# missing the letter for "Đ"/"đ" entirely (ingestion's slug generator drops it
# instead of mapping it to "d" like every other Vietnamese diacritic), so
# `drug_id.replace("_", " ").upper()` can never reconstruct the accented name
# and the fold-based dedup below always mismatches for them. Does not touch
# drug_id or any stored data — only the label text shown in generated answers.
_DRUG_LABEL_OVERRIDES: dict[str, str] = {
"giai_oc_to_uon_van_hap_phu_vac_xin_uon_van_hap_phu":
"GIẢI ĐỘC TỐ UỐN VÁN HẤP PHỤ (VẮC XIN UỐN VÁN HẤP PHỤ)",
"khang_oc_to_bach_hau": "KHÁNG ĐỘC TỐ BẠCH HẦU",
"thuoc_uong_bu_nuoc_va_ien_giai": "THUỐC UỐNG BÙ NƯỚC VÀ ĐIỆN GIẢI",
}
def _fold_diacritics(text: str) -> str:
"""Accent-insensitive fold. Vietnamese "Đ"/"đ" is not a combining-mark
decomposition under NFKD (unlike every other Vietnamese diacritic), so it
survives the strip below unless mapped explicitly first."""
text = text.replace("Đ", "D").replace("đ", "d")
stripped = unicodedata.normalize("NFKD", text)
return "".join(ch for ch in stripped if not unicodedata.combining(ch)).casefold()
def _parse_claims(
raw_claims: list, *, include_drug_label: bool = False
) -> tuple[tuple[str, tuple[int, ...]], ...] | None:
"""Validate the model's `claims` array into `(text, citation_indices)`
pairs, or `None` on any malformed entry — same fail-closed contract as
every other shape check in `_attempt_generation`."""
claims: list[tuple[str, tuple[int, ...]]] = []
for item in raw_claims:
if not isinstance(item, dict):
return None
text = item.get("text")
citations = item.get("citations")
if not isinstance(text, str) or not text.strip():
return None
if not isinstance(citations, list) or not all(
isinstance(c, int) and not isinstance(c, bool) for c in citations
):
return None
cleaned = text.strip()
if include_drug_label:
drug_id = item.get("drug_id")
if not isinstance(drug_id, str) or not drug_id.strip():
return None
label = _DRUG_LABEL_OVERRIDES.get(drug_id) or drug_id.replace("_", " ").upper()
if _fold_diacritics(label) not in _fold_diacritics(cleaned):
cleaned = f"{label}: {cleaned}"
claims.append((cleaned, tuple(citations)))
return tuple(claims)
def _candidate_claims_are_valid(
raw_claims: list,
candidate_drug_ids: tuple[str, ...],
evidence_drug_ids: tuple[str | None, ...],
) -> bool:
"""Deterministic generated-candidate subset and citation binding check."""
if not candidate_drug_ids:
return True
allowed = set(candidate_drug_ids)
for item in raw_claims:
if not isinstance(item, dict):
return False
drug_id = item.get("drug_id")
citations = item.get("citations")
if drug_id not in allowed or not isinstance(citations, list) or not citations:
return False
for citation in citations:
if (
not isinstance(citation, int)
or isinstance(citation, bool)
or not 1 <= citation <= len(evidence_drug_ids)
or evidence_drug_ids[citation - 1] != drug_id
):
return False
return True
def _prompt_evidence_texts(evidence: tuple) -> tuple[str, ...]:
"""Evidence shown to generation/entailment with trusted source metadata.
Found live 2026-08-10: a drug interaction section routinely refers to the
drug it belongs to by pharmacological class rather than by name (e.g.
warfarin's own "tương tác thuốc" section says "thuốc kháng vitamin K",
never "warfarin" — normal in a monograph, since the reader already knows
which drug's chapter they're in). Read as an isolated chunk with no
section header, that self-reference is lost, and the entailment judge
was measured flip-flopping ~50/50 on a claim naming the drug directly
against evidence that never does (10 identical calls: 5 entailed, 5 not
— see the commit this function was added in for the full trace). Labeling
each block with its own `chunk_id` drug prefix (`ingestion/.../chunker.py`
always writes `{drug_id}__{section}__{n}`, so this is not a per-drug
special case) restores that anchor without asking the judge to reason
about pharmacology — it only has to match a name already handed to it.
The same anchor is needed for a single monograph: its interaction or
contraindication prose can use only the pharmacological class while the
answer correctly names the drug from metadata. Label every block so that
naming that source drug is not mistaken for an invented clinical fact.
"""
drug_ids = [
item.drug_id or item.matched_doc_id.split("__", 1)[0]
for item in evidence
]
return tuple(
(
f"(drug_id={drug_id}; thuốc={item.drug_name or drug_id.replace('_', ' ')}; "
f"mục={item.section_title or item.section_key or 'không rõ'}) {item.text}"
)
for drug_id, item in zip(drug_ids, evidence, strict=True)
)
def _assemble_answer(claims: tuple[tuple[str, tuple[int, ...]], ...]) -> str:
"""The exact display string `grounding.verify` and the API response use
— built from the model's own claim/citation structure, not written by
the model as free text. Format matches what the product already shows
(`text [n]` / `text [n][m]`) so the frontend needs no changes and
`grounding.verify`'s existing `[n]`-marker parsing applies unmodified."""
parts = []
for text, citations in claims:
marker = "".join(f"[{c}]" for c in citations)
parts.append(f"{text} {marker}".rstrip() if marker else text)
return " ".join(parts)
_SECTION_PRESENTATION = {
"adr": ("Tác dụng không mong muốn", "fact_list"),
"tac_dung_khong_mong_muon": ("Tác dụng không mong muốn", "fact_list"),
"tuong_tac_thuoc": ("Tương tác thuốc", "warning"),
"tuong_tac": ("Tương tác thuốc", "warning"),
"chong_chi_dinh": ("Chống chỉ định", "warning"),
"than_trong": ("Thận trọng", "warning"),
"canh_bao": ("Cảnh báo", "warning"),
"lieu_luong_va_cach_dung": ("Liều lượng và cách dùng", "dosage"),
"lieu_dung": ("Liều lượng và cách dùng", "dosage"),
"cach_dung": ("Cách dùng", "dosage"),
"bao_quan": ("Bảo quản", "fact_list"),
"do_on_dinh_va_bao_quan": ("Bảo quản", "fact_list"),
"chi_dinh": ("Chỉ định", "fact_list"),
"duoc_luc_hoc": ("Dược lực học", "fact_list"),
"duoc_dong_hoc": ("Dược động học", "fact_list"),
"qua_lieu_va_xu_tri": ("Quá liều và xử trí", "warning"),
}
def _section_key(chunk_id: str) -> str:
parts = chunk_id.split("__")
return parts[1].casefold() if len(parts) > 1 else ""
def _presentation_for_sources(source_ids: tuple[str, ...]) -> tuple[str, str]:
section = _section_key(source_ids[0]) if source_ids else ""
if section in _SECTION_PRESENTATION:
return _SECTION_PRESENTATION[section]
if section:
return section.replace("_", " ").strip().capitalize(), "fact_list"
return "Trả lời", "fact_list"
def _build_blocks(
claims: tuple[tuple[str, tuple[int, ...]], ...],
indexed: list[tuple[str, Citation]],
) -> tuple[AnswerBlock, ...]:
"""Map verified claims to semantic blocks and stable source ids."""
groups: list[tuple[str, str, list[AnswerClaim]]] = []
for text, citation_indices in claims:
source_ids = tuple(
indexed[index - 1][1].chunk_id
for index in citation_indices
if 1 <= index <= len(indexed)
)
claim = AnswerClaim(text=text, source_ids=tuple(dict.fromkeys(source_ids)))
title, kind = _presentation_for_sources(claim.source_ids)
if groups and groups[-1][0] == title and groups[-1][1] == kind:
groups[-1][2].append(claim)
else:
groups.append((title, kind, [claim]))
return tuple(
AnswerBlock(title=title, kind=kind, claims=tuple(items))
for title, kind, items in groups
)
def _answer_mode(claim_count: int) -> str:
if claim_count <= 2:
return "concise"
if claim_count <= 6:
return "normal"
return "detailed"
_BROAD_QUERY_CUES = (
"đầy đủ", "tất cả", "toàn bộ", "tổng hợp", "so sánh", "đối chiếu",
)
_WARNING_SECTIONS = frozenset({
"chong_chi_dinh", "than_trong", "qua_lieu_va_xu_tri",
})
_DOSAGE_SECTIONS = frozenset({"lieu_luong_va_cach_dung", "lieu_dung", "cach_dung"})
_LIST_SECTIONS = frozenset({
"chi_dinh", "chong_chi_dinh", "than_trong", "tac_dung_khong_mong_muon",
"tuong_tac_thuoc", "do_on_dinh_va_bao_quan", "qua_lieu_va_xu_tri",
})
def _plan_answer(query: str, result: RetrievalResult, list_mode: bool) -> AnswerPlan:
sections = tuple(dict.fromkeys(
_section_key(item.matched_doc_id) for item in result.evidence
if _section_key(item.matched_doc_id)
))
drugs = tuple(dict.fromkeys(
item.matched_doc_id.split("__", 1)[0] for item in result.evidence
))
multi_source = len(sections) > 1 or len(drugs) > 1 or list_mode
broad = any(cue in query.casefold() for cue in _BROAD_QUERY_CUES)
verbosity = "detailed" if broad or multi_source else (
"concise" if len(result.evidence) == 1 else "normal"
)
if any(section in _DOSAGE_SECTIONS for section in sections):
layout = "dosage"
elif multi_source or any(section in _LIST_SECTIONS for section in sections):
layout = "bullet_list"
else:
layout = "prose"
return AnswerPlan(
verbosity=verbosity,
layout=layout,
reasoning_mode="synthesis" if multi_source else "direct_lookup",
show_heading=multi_source or verbosity == "detailed",
needs_warning=any(section in _WARNING_SECTIONS for section in sections),
)
def _normalise_for_coverage(text: str) -> str:
folded = unicodedata.normalize("NFKC", text).casefold()
return " ".join(re.sub(r"[^\w]+", " ", folded, flags=re.UNICODE).split())
_MISSING_META_WORDS = {
"không", "chưa", "thiếu", "nêu", "đề", "cập", "đến", "yêu", "cầu",
"ghi", "nhận", "thông", "tin", "trong", "câu", "trả", "lời", "về",
"của", "cho", "", "hoặc", "các", "một", "những",
}
def _quote_supports_missing_description(description: str, quote: str) -> bool:
"""The cited quote must actually contain the fact described as missing."""
description_tokens = {
token
for token in _normalise_for_coverage(description).split()
if len(token) > 1 and token not in _MISSING_META_WORDS
}
quote_tokens = set(_normalise_for_coverage(quote).split())
if not description_tokens:
return False
description_numbers = {token for token in description_tokens if any(c.isdigit() for c in token)}
if not description_numbers.issubset(quote_tokens):
return False
overlap = len(description_tokens & quote_tokens) / len(description_tokens)
return overlap >= 0.5
def _missing_is_already_explicit(
missing: tuple[str, ...],
claims: tuple[tuple[str, tuple[int, ...]], ...],
) -> bool:
"""Reject a judge contradiction when every quoted 'missing' fact is present.
The semantic judge occasionally reports that an exact condition is absent
while quoting that condition verbatim from a claim that already contains it.
This narrow check only resolves that self-contradiction; unquoted or
paraphrased missing facts still fail closed and enter the repair path.
"""
answer = _normalise_for_coverage("\n".join(text for text, _ in claims))
if not missing:
return False
for item in missing:
quoted = [
value for value in re.findall(r"['\"]([^'\"]{8,})['\"]", item)
if _normalise_for_coverage(value)
]
if not quoted or not all(_normalise_for_coverage(value) in answer for value in quoted):
return False
return True
class GroundedAnswerService:
"""Retrieval decides what is true; generation only decides how it reads.
Two operating modes, not to be confused with each other:
- **No generator configured** (`generator=None`, e.g. `ANSWER_PROVIDER=
disabled`) is retrieval-only mode, a deliberate and fully supported
way to run this service. It quotes the retrieved source verbatim.
- **A generator IS configured.** Its output replaces the extractive text
only if it clears two independent checks: `grounding.verify` (every
figure and citation traces to the specific evidence block it cites,
and every claim carries one) and `_verify_entailment` (a second LLM
pass confirming each cited claim's *content* — not just its numbers —
is actually stated by that block). If a configured generation fails
any check, or the provider itself is unreachable, or its output is
malformed, the turn **abstains** with the specific reason that failed
it (`provider_unavailable`, `malformed_output`, `evidence_insufficient`,
a `grounding.verify` reason, or `unsupported_claim`; falls back to
the generic `generation_unavailable` only if none of those was set)
rather than silently degrading to a raw source dump — this product is
a real LLM chatbot, and a citation-stapled paragraph of book text is
not an acceptable
stand-in for an answer the model was supposed to produce.
"""
def __init__(
self,
routing: QueryRoutingService,
generator: AnswerGenerator | None = None,
metrics: Metrics | None = None,
) -> None:
self._routing = routing
self._generator = generator
self._metrics = metrics or NullMetrics()
def answer(
self,
query: str,
subject_scope: SubjectScope,
intent: QueryIntent,
drug_id: str | None = None,
) -> GroundedAnswer:
# When the caller already resolved the drug (e.g. the conversational
# layer, incl. an inherited follow-up), retrieve for it directly instead
# of re-resolving from the turn text — re-resolution from a rewritten
# turn is what abstained good follow-ups as "ambiguous".
if drug_id is not None:
result = self._routing.retrieve_for_drug(
query, drug_id, subject_scope, intent
)
else:
result = self._routing.retrieve(query, subject_scope, intent)
return self.answer_from_result(query, result)
def answer_from_result(
self, query: str, result: RetrievalResult, list_mode: bool = False,
patient_specific: bool = False,
candidate_drug_ids: tuple[str, ...] = (),
budget: RequestBudget | None = None,
prechecked: bool = False,
) -> GroundedAnswer:
"""Everything after retrieval — grounding, sufficiency, generation,
citations. Split out so the new understanding-driven orchestrator
(`rag/agent.py`) reuses the safe answer path without going through the
old `QueryRoutingService` text resolution.
`list_mode=True`: the evidence is several DIFFERENT drugs' own
sections (symptom_to_drug), not alternative phrasings of one drug's
answer — the sufficiency clarify ("which kind of headache?") that's
right for a single dose question doesn't fit a reverse lookup, whose
whole point is to show what the formulary has and let the clinician
narrow it themselves; skipped here the same way a bare-name intro
already skips it.
`budget` (F-08): threaded through to every LLM call this method
makes (sufficiency, generate, one entailment verification). `None` (the
default) means unbounded, unchanged from before F-08 — only
`RagAgent` constructs a real budget today.
`prechecked=True` means the structured agent has already enforced its
required input fields. The generation contract still has its own
`evidence_sufficient`/clarifying-question gate for ambiguity visible
only after retrieval, so the separate same-model sufficiency call is
redundant on that path. The legacy direct-answer path keeps it.
"""
if result.decision == EvidenceDecision.ABSTAIN:
self._metrics.increment(metric_names.ABSTENTION, reason=result.reason)
return GroundedAnswer(result, None)
indexed = self._indexed_citations(result)
if indexed is None:
return GroundedAnswer(
replace(
result,
decision=EvidenceDecision.ABSTAIN,
reason="missing_printed_page_provenance",
evidence=(),
),
None,
)
all_citations = tuple(citation for _, citation in indexed)
if result.decision == EvidenceDecision.VERIFY_PDF:
# Never generated over. A quarantined table or formula is exactly
# the evidence whose numbers were not reliably reconstructed, so
# rephrasing it is the one case where fluency could invent a dose.
return GroundedAnswer(
result,
"Nguồn có bảng hoặc công thức cần đối chiếu trực tiếp với ảnh PDF; "
"không tự động trích số liệu.",
all_citations,
)
evidence_texts = tuple(item.text for item in result.evidence)
prompt_evidence_texts = _prompt_evidence_texts(result.evidence)
plan = _plan_answer(query, result, list_mode)
# Reasoning step BEFORE answering: if the turn is under-specified (a dose
# with several bands and no age/weight/condition), ask instead of dumping.
# A separate focused call is more reliable than folding it into generation.
sufficiency = (
None if list_mode or prechecked
else self._check_sufficiency(
query, evidence_texts, result.is_drug_overview, budget=budget
)
)
if sufficiency is not None:
clarify_q, quick_replies = sufficiency
return GroundedAnswer(
result, clarify_q, (), clarification=clarify_q, quick_replies=quick_replies
)
outcome = self._generate(
query, evidence_texts, prompt_evidence_texts,
intro=result.is_drug_overview,
list_mode=list_mode,
patient_specific=patient_specific,
candidate_drug_ids=(candidate_drug_ids if list_mode else ()),
evidence_drug_ids=tuple(
item.drug_id or item.matched_doc_id.split("__", 1)[0]
for item in result.evidence
),
budget=budget,
plan=plan,
)
if outcome.clarification is not None:
if patient_specific:
# A patient-list clarification can smuggle an uncited negative
# corpus claim ("Dược thư không nêu tương tác...") through the
# branch that deliberately skips grounding because ordinary
# input questions contain no clinical assertion. Fail closed;
# the structured candidate statuses carry missing-evidence
# state without inventing a medical conclusion.
self._metrics.increment(
metric_names.ABSTENTION, reason="evidence_insufficient"
)
return GroundedAnswer(
replace(
result,
decision=EvidenceDecision.ABSTAIN,
reason="evidence_insufficient",
),
None,
)
# The model judged the turn under-specified (a dose with no
# age/weight/renal-function/indication…) and asked back instead of
# listing every band. Return the question, not the whole section.
return GroundedAnswer(
result,
outcome.clarification,
(),
clarification=outcome.clarification,
quick_replies=outcome.quick_replies,
)
if outcome.answer is None:
if self._generator is None:
# No generator configured at all — retrieval-only mode. A
# deliberate operating mode (e.g. ANSWER_PROVIDER=disabled),
# not a failure, so the source is quoted verbatim.
self._metrics.increment(metric_names.ANSWER_EXTRACTIVE)
claims = tuple(
(text, (index,))
for index, text in enumerate(evidence_texts, start=1)
)
blocks = _build_blocks(claims, indexed)
return GroundedAnswer(
result,
"\n".join(text for text, _ in claims),
all_citations,
blocks=blocks,
answer_mode=plan.verbosity,
plan=plan,
)
# A generator WAS configured and this specific generation did not
# clear the safety checks (provider outage, malformed output, an
# ungrounded/uncited/unsupported claim). This product is a real
# LLM chatbot, not the retired offline-extractive build — a raw
# source dump is not an acceptable stand-in for a failed
# generation, so this abstains instead of silently degrading to
# one.
# The specific check that failed (provider_unavailable,
# malformed_output, evidence_insufficient, ungrounded_number,
# uncited_claim, unsupported_claim, request_budget_exhausted) —
# found live 2026-08-07: every one of these used to collapse into
# the same generic "generation_unavailable" by the time it
# reached the API response/trace, so a real, diagnosable cause
# (e.g. a genuine provider outage) was indistinguishable from
# ordinary entailment noise without reading server-side metrics
# by hand. `outcome.reject_reason` already carries the granular
# value the metric above uses — just propagate it.
reason = outcome.reject_reason or "generation_unavailable"
self._metrics.increment(metric_names.ABSTENTION, reason=reason)
return GroundedAnswer(
replace(
result,
decision=EvidenceDecision.ABSTAIN,
reason=reason,
),
None,
)
# Show only the sources the answer actually cited, not every chunk that
# was retrieved — a paragraph that cites [4] must not drag 13 citation
# chips onto the screen. Falls back to all when the text cites nothing.
cited_indices = tuple(dict.fromkeys(
index
for _, claim_indices in outcome.claims
for index in claim_indices
if 1 <= index <= len(indexed)
))
citations = tuple(indexed[index - 1][1] for index in cited_indices)
blocks = _build_blocks(outcome.claims, indexed)
clean_answer = "\n".join(text for text, _ in outcome.claims)
self._metrics.increment(metric_names.GENERATION_SERVED)
return GroundedAnswer(
result,
clean_answer,
citations,
generated=True,
blocks=blocks,
answer_mode=plan.verbosity,
plan=plan,
)
def _attempt_generation(
self,
request: "GenerationRequest",
budget: RequestBudget | None,
*,
candidate_drug_ids: tuple[str, ...] = (),
evidence_drug_ids: tuple[str | None, ...] = (),
) -> "_RawAttempt":
"""One raw generation call, parsed but not yet metric-counted or
verified — the caller decides whether to retry before charging a
metric to any particular reason."""
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except RequestBudgetExhausted:
return _RawAttempt(budget_exhausted=True)
except AnswerGenerationUnavailable:
return _RawAttempt(outage=True)
try:
payload = json.loads(raw)
raw_claims = payload["claims"]
sufficient = payload["evidence_sufficient"]
except (ValueError, TypeError, KeyError):
return _RawAttempt(malformed=True)
# The model asked for a missing detail (age/weight/renal function/
# indication…) instead of listing every band. A clarify is not a grounded
# claim, so it skips the number check — it states no dose.
clarify = payload.get("clarifying_question") if isinstance(payload, dict) else None
if isinstance(clarify, str) and clarify.strip():
return _RawAttempt(
clarification=clarify.strip(),
quick_replies=_sanitize_quick_replies(payload.get("quick_replies")),
)
if not isinstance(raw_claims, list) or not isinstance(sufficient, bool):
return _RawAttempt(malformed=True)
if not sufficient:
return _RawAttempt(insufficient=True)
if not _candidate_claims_are_valid(
raw_claims, candidate_drug_ids, evidence_drug_ids
):
return _RawAttempt(unsupported_drug=True)
claims = _parse_claims(
raw_claims, include_drug_label=bool(candidate_drug_ids)
)
if claims is None:
return _RawAttempt(malformed=True)
return _RawAttempt(answer=_assemble_answer(claims), claims=claims)
def _generate(
self,
query: str,
evidence_texts: tuple[str, ...],
prompt_evidence_texts: tuple[str, ...] | None = None,
*,
intro: bool = False,
list_mode: bool = False,
patient_specific: bool = False,
candidate_drug_ids: tuple[str, ...] = (),
evidence_drug_ids: tuple[str | None, ...] = (),
budget: RequestBudget | None = None,
plan: AnswerPlan | None = None,
) -> "_GenOutcome":
"""A verified generation, a clarifying question, or empty to fall back."""
if self._generator is None or not evidence_texts:
return _GenOutcome()
shown_evidence = prompt_evidence_texts or evidence_texts
plan = plan or AnswerPlan("normal", "prose", "direct_lookup")
request = build_request(
query,
shown_evidence,
intro=intro,
list_mode=list_mode,
answer_mode=plan.verbosity,
layout=plan.layout,
reasoning_mode=plan.reasoning_mode,
show_heading=plan.show_heading,
needs_warning=plan.needs_warning,
patient_specific=patient_specific,
candidate_drug_ids=candidate_drug_ids,
)
attempt = self._attempt_generation(
request,
budget,
candidate_drug_ids=candidate_drug_ids,
evidence_drug_ids=evidence_drug_ids,
)
if attempt.insufficient:
# Empirically noisy (found live 2026-08-07, reproduced 3/3 on a
# fresh retry): the model's own evidence_sufficient=false
# self-assessment sometimes flips to a correct, fully grounded
# answer when asked again with the IDENTICAL evidence — the same
# one-retry pattern `_verify_entailment` already uses below for
# its own noisy judge call. Only the terminal "insufficient AND
# no clarifying question" case retries; a legitimate ask-for-
# more-detail clarify is untouched.
attempt = self._attempt_generation(
request,
budget,
candidate_drug_ids=candidate_drug_ids,
evidence_drug_ids=evidence_drug_ids,
)
if attempt.budget_exhausted:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="request_budget_exhausted"
)
return _GenOutcome(reject_reason="request_budget_exhausted")
if attempt.outage:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
)
return _GenOutcome(reject_reason="provider_unavailable")
if attempt.malformed:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="malformed_output"
)
return _GenOutcome(reject_reason="malformed_output")
if attempt.unsupported_drug:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="unsupported_drug"
)
return _GenOutcome(reject_reason="unsupported_drug")
if attempt.clarification is not None:
return _GenOutcome(
clarification=attempt.clarification,
quick_replies=attempt.quick_replies,
)
if attempt.insufficient:
# The model says the evidence does not answer the question, on
# both attempts. Showing the retrieved section verbatim lets the
# clinician judge that.
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="evidence_insufficient"
)
return _GenOutcome(reject_reason="evidence_insufficient")
answer = attempt.answer
report = grounding.verify(answer, evidence_texts)
if not report.grounded:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason=report.reason
)
return _GenOutcome(reject_reason=report.reason)
verification = self._verify_entailment(
query, attempt.claims, shown_evidence, budget=budget
)
if isinstance(verification, _CheckNotRun):
# The judge never ran. Still fail closed, but report why: a
# timeout or outage recorded as "unsupported claim" would sit in
# the content-failure bucket and be hard to spot in metrics.
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason=verification.reason
)
return _GenOutcome(reject_reason=verification.reason)
if not verification.supported:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="unsupported_claim"
)
return _GenOutcome(reject_reason="unsupported_claim")
if not verification.complete:
missing = "; ".join(verification.missing) or "dữ kiện liên quan trong nguồn"
logger.warning(
"answer completeness repair: query=%r missing=%r claims=%r",
query,
verification.missing,
attempt.claims,
)
repair_request = GenerationRequest(
system=request.system,
user=(
f"{request.user}\n\nBẢN TRƯỚC ĐÃ BỊ LOẠI VÌ THIẾU: {missing}. "
"Hãy tạo lại từ đầu, giữ đúng phạm vi câu hỏi nhưng bổ sung các "
"nhãn, điều kiện, con số và mục liên quan bị thiếu."
),
schema=request.schema,
)
repaired = self._attempt_generation(
repair_request,
budget,
candidate_drug_ids=candidate_drug_ids,
evidence_drug_ids=evidence_drug_ids,
)
# The repair roughly doubles a turn's model calls, so it is the
# most likely place to run out of wall-clock budget. Observed
# live 2026-08-11 (Isosorbid dinitrat dosage, 40.3s against a 40s
# budget): running out here fell through to `incomplete_answer`,
# which describes the answer as missing source information rather
# than reporting that the repair did not finish. Report the
# availability failure as itself, as the first attempt above
# already does.
if repaired.budget_exhausted:
self._metrics.increment(
metric_names.GENERATION_REJECTED,
reason="request_budget_exhausted",
)
return _GenOutcome(reject_reason="request_budget_exhausted")
if repaired.outage:
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="provider_unavailable"
)
return _GenOutcome(reject_reason="provider_unavailable")
if repaired.answer is not None:
repaired_report = grounding.verify(repaired.answer, evidence_texts)
repaired_verification = self._verify_entailment(
query, repaired.claims, shown_evidence, budget=budget
)
logger.warning(
"answer completeness repair result: claims=%r verdict=%r",
repaired.claims,
repaired_verification,
)
if isinstance(repaired_verification, _CheckNotRun):
self._metrics.increment(
metric_names.GENERATION_REJECTED,
reason=repaired_verification.reason,
)
return _GenOutcome(reject_reason=repaired_verification.reason)
if (
repaired_report.grounded
and repaired_verification.supported
and repaired_verification.complete
):
return _GenOutcome(answer=repaired.answer, claims=repaired.claims)
self._metrics.increment(
metric_names.GENERATION_REJECTED, reason="incomplete_answer"
)
return _GenOutcome(reject_reason="incomplete_answer")
return _GenOutcome(answer=answer, claims=attempt.claims)
def _verify_entailment(
self,
query: str,
structured_claims: tuple[tuple[str, tuple[int, ...]], ...],
evidence_texts: tuple[str, ...],
budget: RequestBudget | None = None,
) -> _VerificationOutcome | _CheckNotRun:
"""A second, adversarial LLM pass over an answer that already passed
`grounding.verify`.
The regex check above only binds numbers and citation indices — it
has no notion of meaning, so "Metformin chữa ung thư [1]" citing an
evidence block about "điều trị đái tháo đường" sails through it
untouched: right drug, right citation shape, fabricated indication.
This call is what catches that: each substantive, validly-cited claim
is checked against only the evidence block(s) it names, by a model
told to compare wording, not to reason about medicine.
Fails closed on an outage, malformed output, budget exhaustion, or a
negative verdict. The same deterministic model and temperature are
used for every judge call, so repeating the identical prompt is not
independent evidence: it adds correlated latency and can amplify a
false acceptance. One bounded adversarial pass is the safer default;
judge quality is measured with an eval set instead of manufactured
by retrying the same request.
Reads the model's OWN claim/citation boundaries (2026-08-10
structured-claims change) instead of re-deriving them from the
assembled text via regex — `structured_claims` is exactly what the
model returned, already validated by `_parse_claims`. A claim with
no in-range citation, or no real content, has nothing to check
against and is skipped, same as before.
"""
claims = [
(
text,
"\n".join(
evidence_texts[i - 1] for i in citations
if 1 <= i <= len(evidence_texts)
),
)
for text, citations in structured_claims
if any(1 <= i <= len(evidence_texts) for i in citations)
and grounding.has_content(text)
]
if not claims:
return _VerificationOutcome(supported=True, complete=True)
request = build_entailment_request(query, claims, evidence_texts)
verdict = self._run_entailment_check(
request, evidence_texts=evidence_texts, budget=budget
)
if isinstance(verdict, _CheckNotRun):
return verdict
if (
verdict.supported
and not verdict.complete
and _missing_is_already_explicit(verdict.missing, structured_claims)
):
return _VerificationOutcome(supported=True, complete=True)
return verdict
def _run_entailment_check(
self,
request,
evidence_texts: tuple[str, ...],
budget: RequestBudget | None = None,
) -> _VerificationOutcome | _CheckNotRun:
"""One entailment call.
Returns `_CheckNotRun` when the judge could not be consulted at all
(budget exhausted, provider outage, or a reply this code cannot
parse) and a `_VerificationOutcome` when it ran and reached a
verdict. Both make the caller fail closed; they differ only in the
reason reported, which used to collapse into `unsupported_claim`
for all of them.
"""
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except RequestBudgetExhausted:
# Must be caught before its parent below — see `budget.py`.
return _CheckNotRun("request_budget_exhausted")
except AnswerGenerationUnavailable:
return _CheckNotRun("provider_unavailable")
try:
payload = json.loads(raw)
entailed = payload["entailed"]
unsupported = payload["unsupported"]
missing_evidence = payload.get("missing_evidence", [])
except (ValueError, TypeError, KeyError):
return _CheckNotRun("malformed_output")
if (
not isinstance(entailed, bool)
or not isinstance(unsupported, list)
or not isinstance(missing_evidence, list)
):
return _CheckNotRun("malformed_output")
if not entailed or unsupported:
return _VerificationOutcome(supported=False, complete=False)
# A completeness objection is itself a factual claim about the raw
# evidence. Require the judge to point to an exact source quote and
# validate it locally. This prevents false objections such as
# "missing humidity" when the storage section never mentions humidity,
# which previously discarded a fully grounded answer after two extra
# model calls. Entailment still fails closed; only ungrounded
# *completeness objections* are ignored.
normalised_evidence = tuple(
_normalise_for_coverage(text) for text in evidence_texts
)
grounded_missing: list[str] = []
for item in missing_evidence:
if not isinstance(item, dict):
return _CheckNotRun("malformed_output")
description = item.get("description")
evidence_quote = item.get("evidence_quote")
if not isinstance(description, str) or not isinstance(evidence_quote, str):
return _CheckNotRun("malformed_output")
description = description.strip()
quote = _normalise_for_coverage(evidence_quote)
if (
description
and len(quote) >= 4
and any(quote in evidence for evidence in normalised_evidence)
and _quote_supports_missing_description(description, evidence_quote)
):
grounded_missing.append(description)
return _VerificationOutcome(
supported=True,
complete=not grounded_missing,
missing=tuple(grounded_missing),
)
def _check_sufficiency(
self, query: str, evidence_texts: tuple[str, ...], intro: bool = False,
budget: RequestBudget | None = None,
) -> tuple[str, tuple[str, ...]] | None:
"""A focused reasoning call: is the turn specific enough to answer, or
must we ask? Returns (clarifying_question, quick_replies), or None to
proceed. `quick_replies` is often empty — only populated when the
model judged the missing detail has a few natural discrete answers
(e.g. "Người lớn"/"Trẻ em"), never invented here.
Skipped without a model, for a bare-name intro (not a dose), or for a
single evidence block (nothing to disambiguate).
Fails OPEN on outage/budget-exhaustion (returns None, proceeds to
generate) — deliberately different from every other call in this
file, which fail closed. This is a reasoning heuristic, not a safety
check; grounding + entailment remain the real gate on whatever gets
generated next, so skipping this one costs UX quality (a dose
question that should have asked for age/weight might not), not
safety."""
if self._generator is None or intro or len(evidence_texts) < 2:
return None
request = build_sufficiency_request(query, evidence_texts)
try:
if budget is not None:
budget.require()
raw = self._generator.generate(request.system, request.user, request.schema)
except AnswerGenerationUnavailable:
return None
try:
payload = json.loads(raw)
except (ValueError, TypeError):
return None
if isinstance(payload, dict) and payload.get("sufficient") is False:
question = payload.get("clarifying_question")
if isinstance(question, str) and question.strip():
replies = _sanitize_quick_replies(payload.get("quick_replies"))
return question.strip(), replies
return None
@staticmethod
def _cited_only(
indexed: list[tuple[int, Citation]], answer_text: str
) -> tuple[Citation, ...]:
"""Keep citations whose 1-based evidence marker [n] appears in the text."""
used = {int(m) for m in re.findall(r"\[(\d+)\]", answer_text)}
return tuple(citation for index, citation in indexed if index in used)
@staticmethod
def _indexed_citations(
result: RetrievalResult,
) -> list[tuple[int, Citation]] | None:
"""Citations tagged with the 1-based evidence index the prompt gives them,
so the response can show only the ones the answer cited."""
citations: list[tuple[int, Citation]] = []
for index, evidence in enumerate(result.evidence, start=1):
if not evidence.source_refs:
return None
for source in evidence.source_refs:
printed_range = source.printed_page_range
if printed_range is not None:
start, end = printed_range
elif source.printed_page is not None:
start = end = source.printed_page
else:
return None
citations.append((index, Citation(
chunk_id=evidence.matched_doc_id,
printed_page_start=int(start),
printed_page_end=int(end),
physical_page=source.physical_page,
block_id=source.block_id,
bbox=source.bbox,
source_crop=source.source_crop,
# Backward-compatible compact attachment identifier. A
# real crop path wins; otherwise the block id plus the
# structured page/bbox fields is enough to render later.
attachment=source.source_crop or source.block_id,
evidence_text=evidence.text,
drug_id=evidence.drug_id,
drug_name=evidence.drug_name,
section_key=evidence.section_key,
section_title=evidence.section_title,
source_document=evidence.source_document,
)))
return citations