Files

217 lines
9.5 KiB
Python

from __future__ import annotations
from qdrant_client import QdrantClient
from adapters.embedding import BedrockCohereQueryEmbedder
from adapters.postgres import PostgresConversationStore, PostgresTraceRepository
from adapters.qdrant import QdrantParentStore, QdrantRetriever
from config import Settings
from rag.artifacts import load_aliases
from rag.manifest import MANIFEST_POINT_ID, check_manifest, manifest_collection
from rag.instrumentation import (
InstrumentedEmbedder,
InstrumentedGenerator,
InstrumentedGroundedAnswerService,
InstrumentedQueryUnderstander,
InstrumentedRagAgent,
InstrumentedReranker,
InstrumentedRetrievalService,
)
from rag.metrics import NullMetrics
from rag.routing import CatalogDrugResolver, QueryRoutingService
from rag.sections import SectionResolver
from rag.service import EvidencePolicy
from rag.telemetry import configure_telemetry
from rag.understanding import LlmQueryUnderstander
# How many aliases to show per candidate drug (F-04 bounds *which* drugs are
# shown at all, per turn; this bounds how many names each shown one gets).
_CATALOG_ALIASES_PER_DRUG = 3
def _catalog_names(aliases: dict[str, set[str]]) -> dict[str, str]:
"""The name(s) shown to the LLM for each drug_id.
Found live 2026-08-06: picking the first 3 aliases *alphabetically* could
drop the drug's own canonical/INN name entirely — paracetamol has 191
aliases (mostly trade names), and the alphabetically-first 3 were
"0Frezefev, ABAB, Ace kid 80", none recognizable. Mid-conversation, once
the current turn no longer restates the drug name in raw text, the model
has to re-derive it from history + this catalog line alone — with no
recognizable name shown, it read the earlier "paracetamol" mention as an
unknown drug and answered "not found in the formulary" for a drug that
plainly is. Fixed by always showing the drug_id's own name form first
(guaranteed present, always recognizable — it's the exact string a user
who names a drug is most likely to type), then filling remaining slots
with short, ALL-CAPS-preferring aliases (the book's own heading
convention, so usually the generic name, not a dosage-suffixed brand
like "Ace kid 80").
"""
result: dict[str, str] = {}
for drug_id, names in aliases.items():
canonical = drug_id.replace("_", " ")
ranked = sorted(names, key=lambda n: (not n.isupper(), len(n), n))
extra = [n for n in ranked if n.strip().casefold() != canonical]
shown = [canonical, *extra][:_CATALOG_ALIASES_PER_DRUG]
result[drug_id] = ", ".join(dict.fromkeys(shown))
return result
def _build_metrics(settings: Settings):
"""A Prometheus exporter, or None when the package or the flag is absent.
Missing `prometheus_client` degrades to no metrics rather than to a service
that will not start: observability is not a precondition for answering.
"""
if not settings.metrics_enabled:
return None
try:
from adapters.prometheus import PrometheusMetrics
return PrometheusMetrics()
except ImportError:
return None
def _build_generator(settings: Settings):
if settings.answer_provider == "disabled":
return None
if settings.answer_provider == "stub":
from adapters.bedrock_claude import StubAnswerGenerator
return StubAnswerGenerator(
"Câu trả lời mẫu, không gọi nhà cung cấp nào. [1]"
)
if settings.answer_provider == "bedrock-claude":
from adapters.bedrock_claude import BedrockClaudeAnswerGenerator
return BedrockClaudeAnswerGenerator(
region=settings.aws_region, model_id=settings.answer_model_id
)
if settings.answer_provider == "bedrock-converse":
from adapters.bedrock_converse import BedrockConverseAnswerGenerator
return BedrockConverseAnswerGenerator(
region=settings.aws_region, model_id=settings.answer_model_id
)
raise ValueError(
"Unknown ANSWER_PROVIDER. Supported values: disabled (default), "
"stub (local, no cloud), bedrock-claude, bedrock-converse"
)
def _build_reranker(settings: Settings):
"""A Cohere reranker, or None when disabled. Only used on the similarity /
overview fallback; the section route never reranks."""
if not settings.rerank_enabled:
return None
from adapters.bedrock_converse import BedrockCohereReranker
return BedrockCohereReranker(region=settings.aws_region)
def _verify_corpus_manifest(client, collection: str, embedder, settings: Settings) -> None:
"""Read the sidecar manifest point the ingestion loader writes
(`ingestion/ingestion/load/manifest.py`) and refuse to start on a
mismatch. Raises `rag.manifest.ManifestMismatch` (a `RuntimeError`),
which crashes startup — a deliberate refusal, not an oversight: this
runs at process start (`main.py` calls `build_runtime` at import time),
so a mismatch here means the process never comes up and never serves a
query from a corpus it wasn't verified against.
"""
sidecar = manifest_collection(collection)
payload = None
# This qdrant-client version has no `collection_exists`, and
# `get_collection` (singular) is a known parse-bug risk in this
# environment — list collections and check membership instead.
existing = {col.name for col in client.get_collections().collections}
if sidecar in existing:
points = client.retrieve(sidecar, [MANIFEST_POINT_ID], with_payload=True)
if points:
payload = points[0].payload
check_manifest(payload, collection, embedder.model_id, settings.embedding_dimensions)
def build_runtime(settings: Settings):
metrics = _build_metrics(settings)
effective_metrics = metrics or NullMetrics()
configure_telemetry(settings, effective_metrics)
if settings.embedding_provider == "disabled":
return None, None, PostgresTraceRepository(settings.postgres_dsn), metrics, None
if settings.embedding_provider != "cohere-v4":
raise ValueError(
"No production query embedder is configured. Supported values: "
"EMBEDDING_PROVIDER=cohere-v4 (semantic query embedding) or "
"disabled. The old local/section-only stubs were removed."
)
client = QdrantClient(
url=settings.qdrant_url,
api_key=settings.qdrant_api_key,
timeout=30,
)
embedder = InstrumentedEmbedder(
BedrockCohereQueryEmbedder(
settings.embedding_dimensions, region=settings.aws_region
),
effective_metrics,
)
# F-05: a collection built with one model and queried with another
# returns hits and raises nothing — the results are just meaningless,
# with no error to notice. Refuse to start rather than search with
# vectors this collection was not built from.
_verify_corpus_manifest(client, settings.qdrant_collection, embedder, settings)
section_resolver = SectionResolver()
aliases = load_aliases(settings.entities_path)
# Kept only for deterministic input-time autocomplete (`RagAgent.complete`)
# and to satisfy `GroundedAnswerService`'s constructor — its fuzzy
# `.resolve()` is no longer on the live query path; `RagAgent` resolves
# drug identity through `LlmQueryUnderstander` against the same catalog.
resolver = CatalogDrugResolver(aliases)
reranker = _build_reranker(settings)
# Returned on its own below (Feature-List #4/#23's section-list and
# verbatim-section-text endpoints) — both are plain payload-filtered
# Qdrant reads with no LLM/generation involved, so they read straight
# from this adapter rather than through `RetrievalService`/`RagAgent`.
qdrant_retriever = QdrantRetriever(client, settings.qdrant_collection, embedder)
retrieval = InstrumentedRetrievalService(
qdrant_retriever,
QdrantParentStore(client, settings.qdrant_collection),
EvidencePolicy(minimum_score=settings.evidence_minimum_score),
section_resolver=section_resolver,
reranker=(
InstrumentedReranker(reranker, effective_metrics)
if reranker is not None
else None
),
metrics=effective_metrics,
)
routing = QueryRoutingService(retrieval, resolver)
generator = _build_generator(settings)
if generator is not None:
generator = InstrumentedGenerator(generator, effective_metrics)
answers = InstrumentedGroundedAnswerService(
routing, generator=generator, metrics=effective_metrics
)
trace_writer = PostgresTraceRepository(settings.postgres_dsn)
if generator is None:
# The new front end understands a turn with the same LLM call that
# answers it — with no generator configured there is no query
# understanding either, so there is no conversational/agent
# capability to offer. Answer-only (retrieval-verified, no
# generation) still works through `answers` directly.
return answers, None, trace_writer, metrics, qdrant_retriever
agent = InstrumentedRagAgent(
understander=InstrumentedQueryUnderstander(
LlmQueryUnderstander(generator, _catalog_names(aliases), resolver)
),
retrieval=retrieval,
answers=answers,
autocomplete=resolver,
max_wall_clock_ms=settings.max_wall_clock_ms,
max_llm_calls_per_turn=settings.max_llm_calls_per_turn,
store=PostgresConversationStore(settings.postgres_dsn),
metrics=effective_metrics,
)
return answers, agent, trace_writer, metrics, qdrant_retriever