Files
duocthu/apps/ai-service/config.py
T

90 lines
4.1 KiB
Python

from __future__ import annotations
from functools import lru_cache
from pathlib import Path
from pydantic import Field
from pydantic_settings import BaseSettings, SettingsConfigDict
def _default_entities_path() -> Path:
"""2 parents up from apps/ai-service/config.py in a full repo checkout.
A container image that flattens apps/ai-service/ into its own root
(found live 2026-08-10: the deploy image does exactly this) doesn't have
that depth — ENTITIES_PATH env override is for it; this fallback just
keeps the class from crashing at import time when it's shallower.
"""
here = Path(__file__).resolve()
return (
here.parents[2] if len(here.parents) > 2 else here.parent
) / "ingestion/data/verified/drug_entities.json"
class Settings(BaseSettings):
model_config = SettingsConfigDict(env_file=".env", extra="ignore")
app_name: str = "vsf-duoc-thu-ai-service"
environment: str = "local"
qdrant_url: str = "http://localhost:6333"
qdrant_collection: str = "duocthu_v1"
qdrant_api_key: str | None = None
postgres_dsn: str = Field(
default="postgresql://duoc_thu:duoc_thu@localhost:5432/duoc_thu",
repr=False,
)
# `cohere-v4` = semantic query embedding in the corpus's own space (requires
# live Bedrock). `disabled` skips retrieval entirely. The old local-hash /
# section-only stub embedders were removed in the 2026-08-06 rebuild.
embedding_provider: str = "cohere-v4"
embedding_dimensions: int = 1024
evidence_minimum_score: float = 0.12
aws_region: str = "us-east-1"
# Generation is off unless asked for. `stub` runs the whole answer path —
# prompt, schema parsing, grounding check, fallback — with no cloud call.
# Live options: `bedrock-converse` (DeepSeek/Qwen/GLM/Nova via the Converse
# API) or `bedrock-claude` (Anthropic via the Messages path).
answer_provider: str = "disabled"
answer_model_id: str = "deepseek.v3.2"
# Optional cross-encoder rerank on the similarity fallback (needs live
# Bedrock invoke on the rerank model). Off by default; the section route
# never uses it.
rerank_enabled: bool = False
metrics_enabled: bool = True
# Optional bearer token for `GET /metrics`. Empty by default so the
# current Compose scrape and local development keep working: the endpoint
# is not reachable from the internet today (Caddy proxies only `web`, and
# ai-service publishes no host port). Set it wherever the service is
# exposed through an Ingress, which the Helm chart now allows — metrics
# carry query volumes, provider failure counts and abstain reasons.
metrics_token: str = ""
# OpenTelemetry is opt-in so the existing EC2 Compose deployment keeps
# answering when no collector is present. Docker/Kubernetes observability
# profiles enable it and point OTLP/HTTP at their local collector Service.
otel_enabled: bool = False
otel_service_name: str = "ai-service"
otel_exporter_otlp_endpoint: str = "http://localhost:4318/v1/traces"
otel_sample_ratio: float = Field(default=1.0, ge=0.0, le=1.0)
# Langfuse ingests the SAME spans over OTLP as a second exporter, so it
# adds an LLM-shaped view (per-trace answers, eval scores) without
# replacing Tempo or touching any instrumentation. All three must be set
# for the exporter to be added; keys are secrets and live only in .env /
# the cluster Secret, never in values.yaml.
langfuse_base_url: str = ""
langfuse_public_key: str = ""
langfuse_secret_key: str = ""
entities_path: Path = _default_entities_path()
# F-08: a per-turn budget across RagAgent's sequential Bedrock calls
# (understand, generate, one entailment check on the live agent path).
# Defaults sized with headroom above what a normal turn measures live
# (~8-9s before the bounded-provider change) — see rag/agent.py's
# MAX_WALL_CLOCK_MS/
# MAX_LLM_CALLS_PER_TURN for the full rationale.
max_wall_clock_ms: int = 40_000
max_llm_calls_per_turn: int = 8
@lru_cache
def get_settings() -> Settings:
return Settings()