354 lines
17 KiB
Python
354 lines
17 KiB
Python
"""LLM-driven query understanding — the front-end of the RAG chatbot.
|
|
|
|
The old drug-first path resolved a drug with a fuzzy `SequenceMatcher` and routed
|
|
sections with a Vietnamese phrase table. Both are brittle string heuristics: they
|
|
false-matched a made-up name (``aspirinol`` -> aspirin), tied on a correctly
|
|
spelled English INN (``amoxicillin``), and mistook a common word (``uống``) for a
|
|
drug. This module replaces that with the model the system already has.
|
|
|
|
Division of labour, and why it is safe for a formulary:
|
|
|
|
- The **LLM** does the messy human-language part — which token is a drug, is this
|
|
an interaction / a symptom lookup / a weight-based dose, what section is asked,
|
|
what population/weight. It is good at exactly the fuzziness the heuristics were
|
|
bad at.
|
|
- The **catalog** stays the authority on drug *identity*. The model may only pick
|
|
``drug_id`` values from a list *bounded before the model ever runs* — a
|
|
deterministic alias/fuzzy pass over the turn and history (`CandidateSource`)
|
|
decides which real drugs are even plausible candidates, and only those are
|
|
shown. This closes a gap the catalog-whitelist alone did not (F-04, Codex
|
|
2026-08-06 review): validating that an output id is *some* real drug_id does
|
|
not prove it is the *one the user's text actually named* — an LLM could
|
|
satisfy that whitelist while mapping an unrelated or invented name to any of
|
|
the other 683 real drugs. Bounding candidates first removes that degree of
|
|
freedom: the model can still read ``amoxicillin`` as ``amoxicilin`` (a fuzzy
|
|
match puts it in the candidate set) but cannot map ``aspirinol`` to aspirin,
|
|
because nothing about ``aspirinol`` fuzzy-matches anything and the candidate
|
|
set the model is shown is empty or contains unrelated drugs, not aspirin.
|
|
- Nothing here answers the medical question. It only produces a structured frame;
|
|
retrieval + ``grounding.verify`` remain the load-bearing safety layer downstream.
|
|
|
|
`rag/` imports no SDK: the LLM is injected as a ``JsonLlm`` protocol (satisfied by
|
|
`adapters.bedrock_converse.BedrockConverseAnswerGenerator`), and a deterministic
|
|
stub runs the whole path offline in tests.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from dataclasses import dataclass, field
|
|
from typing import Protocol, Sequence
|
|
|
|
# The 19 monograph section keys, kept here as the closed vocabulary the model may
|
|
# use for `attribute`. Adding a new section is one entry, not a code change.
|
|
SECTION_KEYS = (
|
|
"ten_chung_quoc_te",
|
|
"ten_thuong_mai",
|
|
"ma_atc",
|
|
"loai_thuoc",
|
|
"dang_thuoc_va_ham_luong",
|
|
"duoc_ly_va_co_che_tac_dung",
|
|
"chi_dinh",
|
|
"chong_chi_dinh",
|
|
"than_trong",
|
|
"thoi_ky_mang_thai",
|
|
"thoi_ky_cho_con_bu",
|
|
"tac_dung_khong_mong_muon",
|
|
"huong_dan_xu_tri_adr",
|
|
"lieu_luong_va_cach_dung",
|
|
"tuong_tac_thuoc",
|
|
"qua_lieu_va_xu_tri",
|
|
"do_on_dinh_va_bao_quan",
|
|
"tuong_ky",
|
|
"thong_tin_quy_che",
|
|
)
|
|
|
|
# Short glosses shown to the model alongside each key. Found live 2026-08-06
|
|
# (golden e2e set): a bare key list gives the model nothing to disambiguate
|
|
# "thận trọng" from "chống chỉ định" — 9/9 live calls for "X cần thận trọng
|
|
# gì?" picked chong_chi_dinh, silently answering from the wrong section
|
|
# (and, downstream, dropping the specific safety content the precautions
|
|
# section actually has, e.g. metformin's lactic acidosis warning). The two
|
|
# are genuinely adjacent concepts in Vietnamese medical text; a bare slug
|
|
# name is not enough to tell a model which one a question means.
|
|
SECTION_KEY_HINTS: dict[str, str] = {
|
|
"ten_chung_quoc_te": "tên chung quốc tế/INN",
|
|
"ten_thuong_mai": "tên thương mại/biệt dược",
|
|
"ma_atc": "mã ATC",
|
|
"loai_thuoc": "phân loại thuốc",
|
|
"dang_thuoc_va_ham_luong": "dạng bào chế và hàm lượng",
|
|
"duoc_ly_va_co_che_tac_dung": "dược lý, cơ chế tác dụng",
|
|
"chi_dinh": "chỉ định — bệnh/triệu chứng thuốc dùng để điều trị",
|
|
"chong_chi_dinh": (
|
|
"CHỐNG CHỈ ĐỊNH — trường hợp TUYỆT ĐỐI KHÔNG được dùng thuốc này"
|
|
),
|
|
"than_trong": (
|
|
"THẬN TRỌNG — KHÁC chống chỉ định: vẫn dùng được nhưng cần cảnh "
|
|
"giác/theo dõi/chỉnh liều (ví dụ nguy cơ nhiễm toan lactic của "
|
|
"metformin, độc tính thận/tai của gentamicin). Câu hỏi có chữ "
|
|
"\"thận trọng\", \"cẩn thận\", \"lưu ý gì\", \"cần chú ý\" → key này, "
|
|
"KHÔNG PHẢI chong_chi_dinh."
|
|
),
|
|
"thoi_ky_mang_thai": "dùng khi mang thai",
|
|
"thoi_ky_cho_con_bu": "dùng khi cho con bú",
|
|
"tac_dung_khong_mong_muon": "tác dụng phụ/ADR",
|
|
"huong_dan_xu_tri_adr": "cách xử trí khi gặp ADR",
|
|
"lieu_luong_va_cach_dung": "liều dùng và cách dùng",
|
|
"tuong_tac_thuoc": "tương tác với thuốc khác",
|
|
"qua_lieu_va_xu_tri": "quá liều và cách xử trí",
|
|
"do_on_dinh_va_bao_quan": "độ ổn định, bảo quản",
|
|
"tuong_ky": "tương kỵ (không pha/trộn được với gì)",
|
|
"thong_tin_quy_che": "thông tin quy chế/pháp lý",
|
|
}
|
|
|
|
# What kind of turn this is — the router branches on it. Deliberately explicit so a
|
|
# symptom lookup is never silently treated as a failed drug lookup, and a two-drug
|
|
# interaction never collapses to an "ambiguous drug" abstain.
|
|
TURN_TYPES = (
|
|
"drug_attribute", # one drug, one/more sections ("liều paracetamol")
|
|
"drug_overview", # a bare drug name, wants the monograph
|
|
"interaction", # 2+ drugs, asks about combining them
|
|
"symptom_to_drug", # a symptom/indication, wants candidate drugs
|
|
"dosing_calc", # a dose that needs weight/age arithmetic
|
|
"smalltalk", # greeting / meta, not a medical query
|
|
"out_of_scope", # not answerable from the Part-2 monographs
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class QueryFrame:
|
|
"""The structured reading of one user turn. No medical content, only intent."""
|
|
|
|
turn_type: str
|
|
drugs: tuple[str, ...] = () # canonical drug_ids, from the catalog only
|
|
unknown_drugs: tuple[str, ...] = () # mentioned, not in the catalog
|
|
attribute: str | None = None # a SECTION_KEYS value, or None
|
|
population: str | None = None # e.g. "tre_em", "nguoi_lon", "suy_than"
|
|
weight_kg: float | None = None
|
|
age_text: str | None = None
|
|
indication: str | None = None # symptom/disease, for symptom_to_drug
|
|
needs_clarify: bool = False
|
|
clarify_reason: str | None = None
|
|
raw: dict = field(default_factory=dict, compare=False)
|
|
|
|
|
|
# The JSON contract the model must fill. Stated in the prompt (Converse has no
|
|
# server-side schema) and validated on the way back.
|
|
FRAME_SCHEMA = {
|
|
"turn_type": "one of: " + " | ".join(TURN_TYPES),
|
|
"drugs": ["drug_id exactly as it appears in the provided catalog list"],
|
|
"unknown_drugs": ["a drug name the user mentioned that is NOT in the catalog"],
|
|
"attribute": "one of the section keys provided, or null",
|
|
"population": "tre_em | tre_so_sinh | nguoi_lon | nguoi_cao_tuoi | phu_nu_co_thai | phu_nu_cho_con_bu | suy_than | suy_gan | null",
|
|
"weight_kg": (
|
|
"number if a body weight is given, else null. Vietnamese casual speech "
|
|
"states weight as a bare number of 'cân' or 'ký' with no unit word "
|
|
"('bé 30 cân', 'nặng 30 ký') — both mean kilograms; read the number as "
|
|
"weight_kg the same as if 'kg' had been written."
|
|
),
|
|
"age_text": "the age exactly as stated (e.g. '3 tuổi', '5 tháng'), else null",
|
|
"indication": "the symptom or disease if turn_type is symptom_to_drug, else null",
|
|
"needs_clarify": "true only if the turn cannot be acted on without more info",
|
|
"clarify_reason": "short Vietnamese question to ask, or null",
|
|
}
|
|
|
|
_SYSTEM = """Bạn là bộ HIỂU CÂU HỎI cho một chatbot tra cứu Dược thư Quốc gia Việt Nam.
|
|
Nhiệm vụ: đọc câu hỏi (tiếng Việt, có thể sai chính tả, viết tắt, nhiều lượt) và
|
|
XUẤT RA một JSON mô tả ý định. TUYỆT ĐỐI KHÔNG trả lời câu hỏi y khoa, không nêu liều.
|
|
|
|
Quy tắc bắt buộc:
|
|
- Trường "drugs" CHỈ được chứa các drug_id có trong DANH SÁCH THUỐC được cung cấp.
|
|
Nếu người dùng nhắc một thuốc KHÔNG có trong danh sách (kể cả tên bịa như
|
|
"aspirinol"), đưa tên đó vào "unknown_drugs", KHÔNG được gán sang thuốc gần giống.
|
|
- Sai chính tả một thuốc CÓ trong danh sách thì sửa về đúng drug_id của nó
|
|
(ví dụ "amoxicillin" -> "amoxicilin", "metfomin" -> "metformin").
|
|
- Nếu câu nhắc 2 thuốc trở lên và hỏi về dùng chung/tương tác -> turn_type="interaction".
|
|
- Nếu là triệu chứng/bệnh cần gợi ý thuốc (không nêu tên thuốc) -> "symptom_to_drug",
|
|
điền "indication".
|
|
- Nếu hỏi liều cần cân nặng/tuổi -> "dosing_calc", điền weight_kg/age_text nếu có.
|
|
Nói cân nặng kiểu thường ngày ("bé 30 cân", "nặng 30 ký", chỉ 1 số + "cân"/"ký"
|
|
không kèm đơn vị khác) NGHĨA LÀ 30 kg -> điền weight_kg=30, không bỏ trống.
|
|
- Lượt nối tiếp ("còn liều thì sao", "nó dùng cho trẻ em?") -> dùng LỊCH SỬ để biết
|
|
thuốc đang nói tới và điền vào "drugs".
|
|
- Chào hỏi/vu vơ -> "smalltalk". Ngoài phạm vi chuyên luận thuốc -> "out_of_scope"."""
|
|
|
|
|
|
class JsonLlm(Protocol):
|
|
"""A model that returns a single JSON object as text. Satisfied by the
|
|
existing Bedrock Converse generator, so this adds no SDK to `rag/`."""
|
|
|
|
def generate(self, system: str, user: str, schema: dict) -> str: ...
|
|
|
|
|
|
class QueryUnderstander(Protocol):
|
|
def understand(
|
|
self, turn: str, history: Sequence[str] = ()
|
|
) -> QueryFrame: ...
|
|
|
|
|
|
class CandidateSource(Protocol):
|
|
"""Deterministic, no-LLM drug-name matching — what bounds the model's
|
|
choice before it ever runs (F-04). Satisfied by `routing.CatalogDrugResolver`;
|
|
kept as a protocol (not an import of it) so this module stays decoupled
|
|
from the fuzzy-matching implementation, only its shape.
|
|
"""
|
|
|
|
def resolve(self, query: str): ...
|
|
def suggest(
|
|
self, query: str, k: int = 3, min_score: float = 0.5
|
|
) -> list[tuple[str, float]]: ...
|
|
|
|
|
|
class LlmQueryUnderstander:
|
|
"""Turns a raw user turn into a `QueryFrame` with one LLM call.
|
|
|
|
`catalog` maps drug_id -> a human name (used only to label whichever
|
|
candidates get shown). `resolver` is what actually decides which real
|
|
drugs are plausible for this turn, deterministically, before the model
|
|
runs at all: every drug_id an exact-alias or fuzzy match finds anywhere
|
|
in the turn or the raw history text. The model then picks only among
|
|
those — never the full ~684-drug catalog — so it structurally cannot
|
|
map an invented or unrelated name to some other real drug_id it merely
|
|
happens to also list correctly (F-04). This also directly answers a
|
|
separate 2026-08-06 review finding: sending the full catalog on every
|
|
turn is unbounded token cost; a per-turn candidate shortlist is both
|
|
safer and cheaper.
|
|
"""
|
|
|
|
def __init__(self, llm: JsonLlm, catalog: dict[str, str], resolver: CandidateSource) -> None:
|
|
self._llm = llm
|
|
self._catalog = catalog
|
|
self._resolver = resolver
|
|
|
|
def _candidate_ids(self, turn: str, history: Sequence[str]) -> set[str]:
|
|
"""Every drug_id a deterministic pass finds plausible in the turn or
|
|
the raw history text. Deliberately generous — an exact alias match
|
|
plus a fuzzy `suggest` well below the resolver's own auto-answer
|
|
threshold — because the job here is only to rule out drugs nothing
|
|
in the conversation plausibly refers to, not to pick the right one;
|
|
that disambiguation is still the model's job, within this bound.
|
|
"""
|
|
ids: set[str] = set()
|
|
for line in (turn, *history):
|
|
if not line.strip():
|
|
continue
|
|
resolution = self._resolver.resolve(line)
|
|
if resolution.status == "resolved" and resolution.drug_id:
|
|
ids.add(resolution.drug_id)
|
|
elif resolution.status == "ambiguous":
|
|
ids.update(resolution.candidate_drug_ids)
|
|
for drug_id, _score in self._resolver.suggest(line, k=5, min_score=0.55):
|
|
ids.add(drug_id)
|
|
return ids
|
|
|
|
def understand(self, turn: str, history: Sequence[str] = ()) -> QueryFrame:
|
|
shown = {
|
|
drug_id: self._catalog[drug_id]
|
|
for drug_id in self._candidate_ids(turn, history)
|
|
if drug_id in self._catalog
|
|
}
|
|
catalog_block = (
|
|
"\n".join(f"{drug_id}\t{name}" for drug_id, name in sorted(shown.items()))
|
|
if shown
|
|
else "(không có thuốc nào trong Dược thư khớp với lượt này hoặc lịch sử gần đây)"
|
|
)
|
|
history_block = (
|
|
"LỊCH SỬ HỘI THOẠI (cũ -> mới):\n" + "\n".join(history)
|
|
if history else "LỊCH SỬ HỘI THOẠI: (chưa có)"
|
|
)
|
|
user = (
|
|
f"DANH SÁCH THUỐC ỨNG VIÊN cho lượt này (drug_id\\ttên) — CHỈ được chọn "
|
|
f"drug_id từ đây, đây KHÔNG phải toàn bộ Dược thư, chỉ là các thuốc khớp "
|
|
f"với chữ trong lượt/lịch sử:\n"
|
|
f"{catalog_block}\n\n"
|
|
"CÁC SECTION KEY hợp lệ cho 'attribute' (key: ý nghĩa):\n"
|
|
+ "\n".join(f"{key}: {SECTION_KEY_HINTS[key]}" for key in SECTION_KEYS)
|
|
+ "\n\n"
|
|
f"{history_block}\n\n"
|
|
f"CÂU HỎI HIỆN TẠI: {turn}"
|
|
)
|
|
raw_text = self._llm.generate(_SYSTEM, user, FRAME_SCHEMA)
|
|
return self._parse(raw_text, shown)
|
|
|
|
@staticmethod
|
|
def _resolve_id(value: str, shown: dict[str, str]) -> str | None:
|
|
if value in shown:
|
|
return value
|
|
# `drug_id` is shown with underscores ("paracetamol_acetaminophen")
|
|
# but its own canonical display name (bootstrap's `_catalog_names`)
|
|
# is the same string with spaces — found live 2026-08-06: the two
|
|
# look near-identical in the "drug_id\tname" table, and the model
|
|
# sometimes echoes the spaced display form instead of the id. This
|
|
# is a deterministic, lossless formatting difference (not a fuzzy
|
|
# match — one specific known substitution), so it's tolerated here
|
|
# rather than dropping a correctly-identified drug to unknown.
|
|
spaced = value.strip().casefold()
|
|
for drug_id in shown:
|
|
if drug_id.replace("_", " ").casefold() == spaced:
|
|
return drug_id
|
|
return None
|
|
|
|
def _parse(self, raw_text: str, shown: dict[str, str]) -> QueryFrame:
|
|
try:
|
|
data = json.loads(raw_text)
|
|
except (json.JSONDecodeError, TypeError):
|
|
# Fail closed to a clarify rather than to a wrong reading.
|
|
return QueryFrame(
|
|
turn_type="out_of_scope",
|
|
needs_clarify=True,
|
|
clarify_reason="Xin lỗi, tôi chưa hiểu rõ câu hỏi. Anh/chị hỏi lại giúp nhé?",
|
|
)
|
|
resolved = [
|
|
(d, self._resolve_id(d, shown)) for d in _as_list(data.get("drugs"))
|
|
]
|
|
drugs = tuple(dict.fromkeys(rid for _, rid in resolved if rid is not None))
|
|
# A drug the model named but that resolves to no id among the shown
|
|
# candidates (exact or underscore/space form) is unknown, not a
|
|
# silent drop and not a fuzzy substitution to an unrelated drug.
|
|
unknown = tuple(
|
|
d for d, rid in resolved if rid is None
|
|
) + tuple(_as_list(data.get("unknown_drugs")))
|
|
attribute = data.get("attribute")
|
|
if attribute not in SECTION_KEYS:
|
|
attribute = None
|
|
turn_type = data.get("turn_type")
|
|
if turn_type not in TURN_TYPES:
|
|
turn_type = "drug_attribute" if drugs else "out_of_scope"
|
|
return QueryFrame(
|
|
turn_type=turn_type,
|
|
drugs=drugs,
|
|
unknown_drugs=tuple(dict.fromkeys(unknown)),
|
|
attribute=attribute,
|
|
population=_clean_str(data.get("population")),
|
|
weight_kg=_clean_float(data.get("weight_kg")),
|
|
age_text=_clean_str(data.get("age_text")),
|
|
indication=_clean_str(data.get("indication")),
|
|
needs_clarify=bool(data.get("needs_clarify")),
|
|
clarify_reason=_clean_str(data.get("clarify_reason")),
|
|
raw=data if isinstance(data, dict) else {},
|
|
)
|
|
|
|
|
|
def _as_list(value) -> list[str]:
|
|
if isinstance(value, str):
|
|
return [value] if value.strip() else []
|
|
if isinstance(value, list):
|
|
return [str(v).strip() for v in value if str(v).strip()]
|
|
return []
|
|
|
|
|
|
def _clean_str(value) -> str | None:
|
|
if isinstance(value, str) and value.strip() and value.strip().lower() != "null":
|
|
return value.strip()
|
|
return None
|
|
|
|
|
|
def _clean_float(value) -> float | None:
|
|
if isinstance(value, (int, float)):
|
|
return float(value)
|
|
if isinstance(value, str):
|
|
try:
|
|
return float(value.replace(",", ".").split()[0])
|
|
except (ValueError, IndexError):
|
|
return None
|
|
return None
|