Fingerprint the Qdrant corpus by content, not by count

This commit is contained in:
2026-08-17 14:31:56 +07:00
parent c778f7e9cc
commit d1c9933b9e
8 changed files with 391 additions and 43 deletions
+141
View File
@@ -0,0 +1,141 @@
"""Content fingerprint of a Qdrant collection, for comparing two environments.
Point count and collection status say nothing about whether two collections
hold the *same corpus*: the same 15,100 points could carry different text, be
embedded by a different model, or be a stale re-ingest. This walks every point
and reduces it to hashes that only match when the content matches.
Payload and vectors are hashed separately on purpose. If the payload hash
matches but the vector hash does not, the same source text was embedded
differently — a different embedding model or dimension — which is exactly the
failure a migration can introduce silently.
Per-point digests are sorted before the final hash, so scroll order cannot
change the result. Vectors are rounded before hashing because float formatting
is not guaranteed identical across versions; 6 decimals is far finer than any
meaningful embedding difference.
Reads QDRANT_URL and QDRANT_COLLECTION from the environment, so it runs
unchanged inside the Compose ai-service container and the k3s ai-service pod.
Read-only: it issues nothing but scroll and collection-info requests.
"""
from __future__ import annotations
import hashlib
import json
import os
import sys
import urllib.request
BATCH = 256
VECTOR_PRECISION = 6
def post(url: str, body: dict) -> dict:
req = urllib.request.Request(
url,
data=json.dumps(body).encode(),
method="POST",
headers={"Content-Type": "application/json"},
)
with urllib.request.urlopen(req, timeout=120) as resp:
return json.load(resp)
def get(url: str) -> dict:
with urllib.request.urlopen(url, timeout=60) as resp:
return json.load(resp)
def canonical(value) -> str:
return json.dumps(value, sort_keys=True, ensure_ascii=True, separators=(",", ":"))
def main() -> int:
base = os.environ["QDRANT_URL"].rstrip("/")
collection = os.environ.get("QDRANT_COLLECTION", "duocthu_v1")
info = get(f"{base}/collections/{collection}")["result"]
vectors_config = info.get("config", {}).get("params", {}).get("vectors", {})
payload_digests: list[str] = []
vector_digests: list[str] = []
drug_ids: set[str] = set()
section_keys: set[str] = set()
pages: list[int] = []
missing_vectors = 0
offset = None
seen = 0
while True:
body = {"limit": BATCH, "with_payload": True, "with_vector": True}
if offset is not None:
body["offset"] = offset
result = post(f"{base}/collections/{collection}/points/scroll", body)["result"]
points = result.get("points", [])
if not points:
break
for point in points:
pid = str(point.get("id"))
payload = point.get("payload") or {}
payload_digests.append(
hashlib.sha256((pid + "|" + canonical(payload)).encode()).hexdigest()
)
vector = point.get("vector")
if isinstance(vector, dict): # named vectors
vector = canonical(
{k: [round(float(x), VECTOR_PRECISION) for x in v] for k, v in vector.items()}
)
elif isinstance(vector, list):
vector = canonical([round(float(x), VECTOR_PRECISION) for x in vector])
else:
missing_vectors += 1
vector = "null"
vector_digests.append(hashlib.sha256((pid + "|" + vector).encode()).hexdigest())
if payload.get("drug_id"):
drug_ids.add(str(payload["drug_id"]))
if payload.get("section_key"):
section_keys.add(str(payload["section_key"]))
for key in ("printed_page_start", "printed_page_end"):
value = payload.get(key)
if isinstance(value, int):
pages.append(value)
seen += len(points)
print(f"...scrolled {seen}", file=sys.stderr, flush=True)
offset = result.get("next_page_offset")
if offset is None:
break
payload_digests.sort()
vector_digests.sort()
print(
json.dumps(
{
"collection": collection,
"points_scrolled": seen,
"points_count_reported": info.get("points_count"),
"status": info.get("status"),
"vectors_config": vectors_config,
"payload_hash": hashlib.sha256("".join(payload_digests).encode()).hexdigest(),
"vector_hash": hashlib.sha256("".join(vector_digests).encode()).hexdigest(),
"missing_vectors": missing_vectors,
"distinct_drug_ids": len(drug_ids),
"distinct_section_keys": len(section_keys),
"printed_page_min": min(pages) if pages else None,
"printed_page_max": max(pages) if pages else None,
},
sort_keys=True,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())