Wire the guarded conversational RAG answer layer end-to-end
This commit is contained in:
@@ -0,0 +1,76 @@
|
||||
"""A section must be served in the order it was written.
|
||||
|
||||
Found 2026-08-05 by reading a real answer in the UI rather than a test:
|
||||
`liều dùng paracetamol` opened mid-sentence on `5 - 12 tuổi:` and buried
|
||||
`Liều lượng: Người lớn:` seven hundred words down. Qdrant scrolls in point-id
|
||||
order and point ids are `uuid5(chunk_id)`, so PARACETAMOL's five dosing parts
|
||||
came back **3, 4, 1, 2, 0**.
|
||||
|
||||
This is a clinical defect, not a cosmetic one: a reader who stops partway
|
||||
through stops in the middle of a different population's dose.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from adapters.qdrant import QdrantRetriever
|
||||
|
||||
|
||||
class _ScrambledClient:
|
||||
"""Returns parts out of order, the way a real scroll did."""
|
||||
|
||||
def __init__(self, part_indices: list[int], include_index: bool = True) -> None:
|
||||
self._payloads = [
|
||||
{
|
||||
"chunk_id": f"paracetamol__lieu__{index}",
|
||||
"drug_id": "paracetamol",
|
||||
"section_key": "lieu_luong_va_cach_dung",
|
||||
"chunk_kind": "prose",
|
||||
"text": f"part {index}",
|
||||
"source_refs": [{"physical_page": 1120, "precision": "page"}],
|
||||
**({"part_index": index} if include_index else {}),
|
||||
}
|
||||
for index in part_indices
|
||||
]
|
||||
|
||||
def scroll(self, **kwargs):
|
||||
points = [type("P", (), {"payload": payload})() for payload in self._payloads]
|
||||
return points, None
|
||||
|
||||
|
||||
class _Embedder:
|
||||
dimensions = 4
|
||||
|
||||
def embed_query(self, text: str) -> list[float]: # never used by this route
|
||||
raise AssertionError("find_by_section must not embed anything")
|
||||
|
||||
|
||||
def _hits(part_indices: list[int], include_index: bool = True) -> list[str]:
|
||||
retriever = QdrantRetriever(
|
||||
_ScrambledClient(part_indices, include_index), "duocthu_v1", _Embedder()
|
||||
)
|
||||
return [
|
||||
hit.document.text
|
||||
for hit in retriever.find_by_section("paracetamol", "lieu_luong_va_cach_dung")
|
||||
]
|
||||
|
||||
|
||||
def test_the_exact_scramble_observed_against_the_real_collection():
|
||||
assert _hits([3, 4, 1, 2, 0]) == [
|
||||
"part 0",
|
||||
"part 1",
|
||||
"part 2",
|
||||
"part 3",
|
||||
"part 4",
|
||||
]
|
||||
|
||||
|
||||
def test_an_already_ordered_section_is_left_alone():
|
||||
assert _hits([0, 1, 2, 3]) == ["part 0", "part 1", "part 2", "part 3"]
|
||||
|
||||
|
||||
def test_a_part_missing_its_index_is_kept_and_sorted_last():
|
||||
"""Dropping it would silently shorten a dose list, which is the one
|
||||
outcome worse than showing it out of order."""
|
||||
texts = _hits([1, 0], include_index=False)
|
||||
|
||||
assert len(texts) == 2
|
||||
assert set(texts) == {"part 0", "part 1"}
|
||||
Reference in New Issue
Block a user