Fix every real lint finding and drop degenerate splice fragments

This commit is contained in:
2026-08-01 13:51:38 +07:00
parent 967b917001
commit 834d9e51b0
69 changed files with 10119 additions and 39 deletions
@@ -0,0 +1,185 @@
{
"note": "2D (stacked-fraction) formula regions, every one confirmed by rendering the page and reading it. bbox is the fraction bar itself; numerator and denominator sit above and below it.",
"verified_on": "2026-08-01",
"method": "residual-ink fraction_bar_candidate, then visual inspection of all 23 candidates",
"regions": [
{
"physical_page": 43,
"bar_bbox": [
133.44,
308.16,
222.72,
308.16
]
},
{
"physical_page": 92,
"bar_bbox": [
430.08,
603.36,
465.6,
603.36
]
},
{
"physical_page": 92,
"bar_bbox": [
362.88,
704.64,
386.4,
704.64
]
},
{
"physical_page": 92,
"bar_bbox": [
400.32,
704.64,
422.88,
704.64
]
},
{
"physical_page": 92,
"bar_bbox": [
456.96,
704.64,
480.0,
704.64
]
},
{
"physical_page": 92,
"bar_bbox": [
505.44,
704.64,
522.72,
704.64
]
},
{
"physical_page": 202,
"bar_bbox": [
371.52,
151.68,
489.6,
151.68
]
},
{
"physical_page": 325,
"bar_bbox": [
445.92,
562.08,
544.32,
562.56
]
},
{
"physical_page": 325,
"bar_bbox": [
439.68,
623.52,
560.16,
623.52
]
},
{
"physical_page": 349,
"bar_bbox": [
384.48,
336.0,
491.52,
336.48
]
},
{
"physical_page": 1042,
"bar_bbox": [
97.92,
492.0,
286.56,
492.0
]
},
{
"physical_page": 1043,
"bar_bbox": [
94.56,
536.16,
140.16,
536.16
]
},
{
"physical_page": 1043,
"bar_bbox": [
145.92,
536.16,
244.8,
536.16
]
},
{
"physical_page": 1132,
"bar_bbox": [
63.36,
498.72,
239.52,
498.72
]
},
{
"physical_page": 1402,
"bar_bbox": [
120.96,
711.84,
201.6,
711.84
]
},
{
"physical_page": 1402,
"bar_bbox": [
180.48,
770.4,
205.92,
770.4
]
},
{
"physical_page": 147,
"bar_bbox": [
307.9,
672.0,
428.5,
672.0
],
"source_prints_no_bar": true,
"note": "ADENOSIN infusion-rate formula. The source page prints three plain lines with no fraction bar at all, so no geometric detector can find it — confirmed by rendering the region and reading it. Left in prose it reads as a multiplication chain. Quarantined on the strength of the reading, and flagged for human confirmation of the intended division."
}
],
"rejected": [
{
"physical_page": 4,
"reason": "decorative underlines on the Ministry decision page"
},
{
"physical_page": 63,
"reason": "ruled box around a treatment-protocol paragraph"
},
{
"physical_page": 845,
"reason": "table header cell border"
},
{
"physical_page": 878,
"reason": "table header cell border"
},
{
"physical_page": 1667,
"reason": "rule above the colophon on the last page"
}
],
"recall_limit": "The fraction-bar signal cannot find a fraction the source never typeset. ADENOSIN (physical page 147) is one confirmed case, found only because a prose-leak gate matched its text. The true number of bar-less formulas in the book is UNMEASURED."
}
@@ -0,0 +1,671 @@
{
"note": "Text that exists in the PDF only as vector outlines. No extractor returns it (PyMuPDF, pdfplumber and opendataloader-pdf all omit it). Every 'text' value below is a transcription read off the rendered page, not extracted data.",
"transcribed_on": "2026-08-01",
"method": "ingestion.extract.detect_outlined_text located the runs; each run was rendered at 210-300 dpi and read directly",
"confidence": "Full-line runs are read with high confidence. Single-glyph runs are Vietnamese diacritic characters dropped out of an otherwise-extracted line; the glyph identity is legible but these should still be spot-checked by a human before the corpus is treated as complete.",
"runs": [
{
"physical_page": 714,
"bbox": [
470.32,
37.64,
521.85,
44.57
],
"path_items": 337,
"text": "Gatifloxacin",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
35.31,
75.76,
286.72,
84.41
],
"path_items": 1638,
"text": "Nghiên cứu trên động vật, gatifloxacin gây ngộ độc cho thai.",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
35.77,
87.91,
287.05,
96.56
],
"path_items": 1714,
"text": "Gatifloxacin chỉ sử dụng cho phụ nữ có thai khi lợi ích vượt trội so",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.67,
127.07,
551.33,
137.58
],
"path_items": 1765,
"text": "Thuốc kháng acid (antacid): Gatifloxacin bị giảm hấp thu khi sử",
"extracted_line_it_belongs_to": "Do chưa biết thuốc có phân bố vào sữa mẹ khi dùng trên người hay ",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
35.51,
139.65,
287.22,
150.17
],
"path_items": 1831,
"text": "không, cần thận trọng khi sử dụng gatifloxacin cho phụ nữ đang",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.71,
151.2,
550.66,
161.71
],
"path_items": 1787,
"text": "cần dùng gatifloxacin ít nhất 4 giờ trước khi dùng các antacid này.",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.44,
177.2,
455.19,
185.84
],
"path_items": 1126,
"text": "học có ý nghĩa lâm sàng với gatifloxacin.",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.66,
223.59,
397.93,
234.1
],
"path_items": 792,
"text": "giảm hấp thu gatifloxacin.",
"extracted_line_it_belongs_to": "Mắt: Chứng sưng viêm mi mắt, xuất huyết kết mạc, rát kết mạc, ",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.66,
247.72,
551.32,
258.23
],
"path_items": 1731,
"text": "giữa warfarin và gatifloxacin, nhưng do một số quinolon có khả",
"extracted_line_it_belongs_to": "khô mắt, phù, rát, viêm giác mạc, giảm thị lực, kích ứng kết mạc.",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.71,
321.98,
363.88,
330.62
],
"path_items": 480,
"text": "của gatifloxacin.",
"extracted_line_it_belongs_to": "Thần kinh: Căng thẳng, kích động, lo lắng, mất ngủ, hoa mắt, giấc ",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
35.78,
441.28,
287.38,
451.79
],
"path_items": 1778,
"text": "Cần ngừng gatifloxacin trong các trường hợp: Bắt đầu có các biểu",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
299.71,
452.82,
461.52,
463.34
],
"path_items": 1115,
"text": "Gatifloxacin dùng với các thuốc làm thay đ",
"extracted_line_it_belongs_to": "hiện ban da hoặc bất kỳ dấu hiệu nào của phản ứng quá mẫn, có ",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
461.95,
452.82,
466.05,
461.42
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "i nồng độ glucose máu ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
35.72,
503.9,
81.99,
512.55
],
"path_items": 373,
"text": "gatifloxacin.",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
313.73,
506.23,
317.83,
514.82
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
542.15,
506.23,
546.25,
514.82
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
450.53,
520.35,
455.28,
526.89
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "định trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
299.71,
532.42,
304.45,
538.95
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": " nhiệt độ 2 - 8 oC. Dung dịch pha loãng này (trừ pha trong natri ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
385.77,
542.42,
389.87,
551.01
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
36.09,
543.49,
287.58,
554.0
],
"path_items": 1809,
"text": "Ghi chú: Đối với gatifloxacin dạng viên và dạng tiêm, nhà sản xuất",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
513.35,
544.48,
518.1,
551.01
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
522.12,
554.48,
526.22,
563.08
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "-25 đến -10 oC, sau khi đưa ra khỏi tủ lạnh sâu, tiếp tục n định ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
426.75,
568.61,
431.5,
575.14
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
525.62,
568.61,
530.37,
575.14
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
299.47,
619.81,
551.17,
630.32
],
"path_items": 1692,
"text": "Vì có rất ít các thông tin về tương ky của gatifloxacin, nên không",
"extracted_line_it_belongs_to": "Tiêm truyền tĩnh mạch dưới dạng dung dịch 2 mg/ml trong 60 phút.",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
105.83,
628.89,
109.93,
637.14
],
"path_items": 39,
"text": "ỗ",
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
299.48,
631.87,
551.03,
642.39
],
"path_items": 1739,
"text": "thêm bất kỳ một thuốc nào khác vào dịch truyền gatifloxacin hoặc",
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
"single_glyph": false
},
{
"physical_page": 714,
"bbox": [
391.91,
685.48,
396.01,
693.72
],
"path_items": 39,
"text": "ỗ",
"extracted_line_it_belongs_to": "triệu chứng và điều trị h trợ, bao gồm: Gây nôn và rửa dạ dày để ",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
223.28,
713.6,
227.39,
722.19
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
167.4,
715.66,
172.14,
722.19
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
"single_glyph": true
},
{
"physical_page": 714,
"bbox": [
299.72,
786.64,
551.3,
797.16
],
"path_items": 1687,
"text": "Gatifloxacin thuộc Danh mục nguyên liệu và thuốc thành phẩm",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 736,
"bbox": [
448.56,
88.17,
453.3,
94.7
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "không màu, đóng kín tránh ánh sáng điều kiện lạnh 2 - 8 oC; ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
266.53,
98.19,
270.63,
106.79
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
282.68,
100.26,
287.43,
106.79
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
129.96,
122.5,
209.16,
133.01
],
"path_items": 580,
"text": "nh tổn thương dây th",
"extracted_line_it_belongs_to": "vào vùng cơ mông để trá",
"single_glyph": false
},
{
"physical_page": 736,
"bbox": [
172.81,
221.76,
177.56,
228.3
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "Liều thường dùng của GMDCUV người lớn và trẻ em để dự ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
91.42,
280.45,
95.52,
289.05
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "tiêm các liều b sung với các khoảng cách là 4 tuần.",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
157.23,
332.32,
161.98,
338.85
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "lại. Liều thông thường HTCUV người lớn và trẻ em để dự phòng ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
192.28,
369.96,
197.03,
376.5
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "chậm trễ trong bắt đầu tiêm phòng hoặc người có thể trọng quá ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
399.06,
424.38,
403.8,
430.91
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "huyết thanh của người trư ng thành khỏe mạnh đã được tạo miễn ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
187.41,
513.01,
192.16,
519.54
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "GMDCUV hoặc HTCUV không ảnh hư ng tới đáp ứng miễn dịch ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
280.17,
625.95,
284.91,
632.48
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "miễn dịch đối với một vài loại vắc xin virus sống (vắc xin virus s i ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
282.99,
699.18,
287.09,
707.78
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "dịch hoặc huyết thanh ngựa thì nên dùng thêm một liều vắc xin b ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
419.76,
726.48,
424.51,
733.01
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
442.01,
726.48,
446.75,
733.01
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
305.51,
737.0,
309.61,
745.6
],
"path_items": 43,
"text": "ổ",
"extracted_line_it_belongs_to": "Do các chế phẩm có chứa globulin miễn dịch không có biểu hiện ",
"single_glyph": true
},
{
"physical_page": 736,
"bbox": [
62.37,
751.45,
67.12,
757.98
],
"path_items": 45,
"text": "ở",
"extracted_line_it_belongs_to": "ảnh hư ng tới các đáp ứng miễn dịch của vắc xin uống virus bại ",
"single_glyph": true
},
{
"physical_page": 1373,
"bbox": [
43.81,
98.58,
295.66,
109.1
],
"path_items": 1458,
"text": "Nếu phối hợp với flutamid ở giai đoạn T2b - T4 (B2 - C), điều trị",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 1444,
"bbox": [
35.72,
136.22,
287.35,
146.74
],
"path_items": 1731,
"text": "Trimovax (Sanofi Pasteur): Một liều vắc xin chứa virus sống giảm",
"extracted_line_it_belongs_to": null,
"single_glyph": false
},
{
"physical_page": 1445,
"bbox": [
308.28,
114.39,
559.87,
123.03
],
"path_items": 1440,
"text": "(Typhoid, inactivated, whole cell), J07AP03 (Typhoid, purified",
"extracted_line_it_belongs_to": "thể xảy ra 5 ngày sau khi tiêm: Sốt (có thể dự phòng bằng các loại ",
"single_glyph": false
},
{
"physical_page": 1445,
"bbox": [
115.46,
646.46,
208.24,
655.11
],
"path_items": 660,
"text": "Haemophilus influenzae",
"extracted_line_it_belongs_to": "khác như vắc xin ",
"single_glyph": false
}
]
}
+23
View File
@@ -0,0 +1,23 @@
from .chunker import chunk_all, chunk_monograph, chunk_section, estimate_tokens
from .io import read_monographs_jsonl, write_chunks_jsonl
from .models import (
CHUNK_KIND_BLOCK_DESCRIPTOR,
CHUNK_KIND_PROSE,
SCHEMA_VERSION,
Chunk,
ChunkAttachment,
)
__all__ = [
"Chunk",
"ChunkAttachment",
"SCHEMA_VERSION",
"CHUNK_KIND_PROSE",
"CHUNK_KIND_BLOCK_DESCRIPTOR",
"chunk_all",
"chunk_monograph",
"chunk_section",
"estimate_tokens",
"read_monographs_jsonl",
"write_chunks_jsonl",
]
+215
View File
@@ -0,0 +1,215 @@
"""Section -> chunk logic (pure; no filesystem, no embedding client).
ADR 0004: chunk unit is `(drug_id, section_key)`. A section under the token
ceiling becomes one chunk verbatim. Only the long-tail sections above it are
sub-chunked, with a sentence-boundary-aware sliding window.
"""
from __future__ import annotations
import re
from typing import Dict, Iterable, Iterator, List, Sequence
from ..segment.models import Monograph, SectionSpan, TableBlock
from ..tables.classify import SHAPE_FORMULA_2D, SHAPE_SIMPLE
from .models import (
CHUNK_KIND_BLOCK_DESCRIPTOR,
CHUNK_KIND_PROSE,
Chunk,
ChunkAttachment,
)
from .sentences import split_sentences
CEILING_TOKENS = 800
TARGET_TOKENS = 650
OVERLAP_TOKENS = 65
# Physical -> printed page. Empirically constant across every tested
# milestone page (extract/page_map.py, ADR 0003); the descriptor quotes the
# printed number because that is what a reader holding the book looks for.
PRINTED_PAGE_OFFSET = 1
KIND_TABLE = "table"
KIND_FORMULA = "formula"
# A header row is only safe to embed when it is genuinely a row of labels.
# Measured on the corpus: 42 of 124 simple-table headers (34%) contain a
# digit, and AMIODARON's (physical page 183) is
# "Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)" — a dose,
# inside what pdfplumber called a header, from an extraction never verified by
# eye. A label carrying no digit cannot be mistaken for a dose; a long cell is
# content rather than a label.
_DIGIT = re.compile(r"\d")
HEADER_CELL_MAX_CHARS = 40
def _is_label_row(cells: Sequence[str]) -> bool:
kept = [c for c in cells if c and c.strip()]
if not kept:
return False
return all(
not _DIGIT.search(cell) and len(cell.strip()) <= HEADER_CELL_MAX_CHARS
for cell in kept
)
def estimate_tokens(text: str) -> int:
"""ADR 0004's chars/4 estimate — an estimate, not a tokenizer count."""
return len(text) // 4
def _pack(sentences: List[str]) -> List[List[str]]:
"""Greedily pack sentences up to TARGET_TOKENS, overlapping by OVERLAP_TOKENS.
A single sentence longer than the target becomes its own part rather than
being cut mid-sentence — the caller flags it instead of splitting it.
"""
parts: List[List[str]] = []
current: List[str] = []
current_tokens = 0
for sentence in sentences:
tokens = estimate_tokens(sentence)
if current and current_tokens + tokens > TARGET_TOKENS:
parts.append(current)
overlap: List[str] = []
acc = 0
for prev in reversed(current):
overlap.insert(0, prev)
acc += estimate_tokens(prev)
if acc >= OVERLAP_TOKENS:
break
current = list(overlap)
current_tokens = sum(estimate_tokens(s) for s in current)
current.append(sentence)
current_tokens += tokens
if current:
parts.append(current)
return parts
def _block_kind(block: TableBlock) -> str:
return KIND_FORMULA if block.shape == SHAPE_FORMULA_2D else KIND_TABLE
def _attachment(block: TableBlock, header_row: List[str]) -> ChunkAttachment:
return ChunkAttachment(
block_id=block.table_id,
kind=_block_kind(block),
shape=block.shape,
physical_page=block.physical_page,
bbox=list(block.bbox),
quarantined=block.quarantined,
# Only a simple table's first row can be a row of plain labels, and
# only when it actually reads like one. A multi-level or merged header
# is the shape whose extraction is least trustworthy, so it
# contributes nothing rather than something wrong.
header_row=(list(header_row)
if block.shape == SHAPE_SIMPLE and _is_label_row(header_row)
else []),
)
def _blocks_by_section(monograph: Monograph) -> Dict[str, List[TableBlock]]:
grouped: Dict[str, List[TableBlock]] = {}
for block in monograph.tables:
if block.section_key:
grouped.setdefault(block.section_key, []).append(block)
return grouped
def describe_block(monograph: Monograph, section: SectionSpan,
attachment: ChunkAttachment) -> str:
"""Retrieval text for a block, built only from metadata.
No cell value ever appears here. A header row is a row of labels;
linearising it cannot invent a numeric relationship, which is exactly what
linearising a body row does.
"""
noun = "công thức" if attachment.kind == KIND_FORMULA else "bảng"
printed = attachment.physical_page + PRINTED_PAGE_OFFSET
text = (f"{monograph.drug_name}{section.display_name}{noun}, "
f"trang {printed}.")
if attachment.header_row:
columns = " | ".join(c.replace("\n", " ").strip()
for c in attachment.header_row if c and c.strip())
if columns:
text += f" Cột: {columns}."
text += (" Nội dung chỉ tra cứu được trên ảnh trang gốc, "
"không trích dẫn được dưới dạng văn bản.")
return text
def chunk_section(monograph: Monograph, section: SectionSpan,
blocks: Sequence[TableBlock] = (),
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
header_rows = header_rows or {}
attachments = [_attachment(b, header_rows.get(b.table_id, [])) for b in blocks]
quarantined = any(a.quarantined for a in attachments)
def build(body: str, part_index: int, part_count: int) -> Chunk:
tokens = estimate_tokens(body)
return Chunk(
chunk_id=f"{monograph.drug_id}__{section.key}__{part_index}",
drug_id=monograph.drug_id,
drug_name=monograph.drug_name,
section_key=section.key,
section_display_name=section.display_name,
text=body,
heading_physical_page=section.heading.physical_page,
source_page_range=list(monograph.source_page_range),
atc_codes=list(monograph.atc_codes),
part_index=part_index,
part_count=part_count,
est_tokens=tokens,
oversized=tokens > CEILING_TOKENS,
chunk_kind=CHUNK_KIND_PROSE,
attachments=list(attachments),
has_quarantined_content=quarantined,
)
text = section.text.strip()
prose: List[Chunk] = []
if text:
if estimate_tokens(text) <= CEILING_TOKENS:
prose = [build(text, 0, 1)]
else:
parts = _pack(split_sentences(text))
bodies = [b for b in ("".join(p).strip() for p in parts) if b]
prose = [build(b, i, len(bodies)) for i, b in enumerate(bodies)]
descriptors = []
for attachment in attachments:
body = describe_block(monograph, section, attachment)
descriptors.append(Chunk(
chunk_id=f"{monograph.drug_id}__{section.key}__block__{attachment.block_id}",
drug_id=monograph.drug_id,
drug_name=monograph.drug_name,
section_key=section.key,
section_display_name=section.display_name,
text=body,
heading_physical_page=section.heading.physical_page,
source_page_range=list(monograph.source_page_range),
atc_codes=list(monograph.atc_codes),
est_tokens=estimate_tokens(body),
chunk_kind=CHUNK_KIND_BLOCK_DESCRIPTOR,
attachments=[attachment],
has_quarantined_content=attachment.quarantined,
))
return prose + descriptors
def chunk_monograph(monograph: Monograph,
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
grouped = _blocks_by_section(monograph)
chunks: List[Chunk] = []
for section in monograph.sections.values():
chunks.extend(chunk_section(monograph, section,
grouped.get(section.key, ()), header_rows))
return chunks
def chunk_all(monographs: Iterable[Monograph],
header_rows: Dict[str, List[str]] | None = None) -> Iterator[Chunk]:
for monograph in monographs:
yield from chunk_monograph(monograph, header_rows)
+71
View File
@@ -0,0 +1,71 @@
"""Filesystem boundary for the chunk stage — kept out of the pure logic."""
from __future__ import annotations
import json
from dataclasses import asdict
from pathlib import Path
from typing import Iterable, Iterator
from ..segment.models import Heading, Monograph, SectionSpan, TableBlock
from .models import SCHEMA_VERSION, Chunk
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
with path.open(encoding="utf-8") as fh:
for line in fh:
line = line.strip()
if not line:
continue
raw = json.loads(line)
sections = {}
for key, s in raw.get("sections", {}).items():
h = s["heading"]
sections[key] = SectionSpan(
key=s["key"],
display_name=s["display_name"],
heading=Heading(
text=h["text"],
physical_page=h["physical_page"],
y0=h["y0"],
is_monograph_title=h["is_monograph_title"],
section_key=h.get("section_key"),
),
text=s["text"],
)
yield Monograph(
drug_id=raw["drug_id"],
drug_name=raw["drug_name"],
source_page_range=raw["source_page_range"],
sections=sections,
atc_codes=raw.get("atc_codes", []),
atc_stated_absent=raw.get("atc_stated_absent", False),
tables=[
TableBlock(
table_id=t["table_id"],
shape=t["shape"],
physical_page=t["physical_page"],
bbox=t["bbox"],
section_key=t.get("section_key"),
text=t.get("text", ""),
quarantined=t.get("quarantined", False),
table_part_id=t.get("table_part_id"),
continuation_group=t.get("continuation_group"),
source_span_ids=t.get("source_span_ids", []),
)
for t in raw.get("tables", [])
],
)
def write_chunks_jsonl(chunks: Iterable[Chunk], path: Path) -> int:
path.parent.mkdir(parents=True, exist_ok=True)
count = 0
with path.open("w", encoding="utf-8") as fh:
for chunk in chunks:
# ADR 0005 flagged the absence of a version and ADR 0006 made it
# necessary: the record now has two chunk kinds and an attachment
# list, so a consumer must be able to tell which shape it has.
record = {"schema_version": SCHEMA_VERSION, **asdict(chunk)}
fh.write(json.dumps(record, ensure_ascii=False) + "\n")
count += 1
return count
+66
View File
@@ -0,0 +1,66 @@
"""Chunk record — the unit handed to embedding/indexing.
Provenance fields follow ADR 0004 and CLAUDE.md's provenance rule: a chunk
must carry enough to trace it back to a monograph, a section, and the page
its section heading was found on.
ADR 0006 adds attachments. `segment/` lifts tables and 2D formulas out of
section prose because linearising them is actively wrong — AMPICILIN VÀ
SULBACTAM's Cockcroft-Gault fraction read as `Clcr (ml/phút) = 72 x
creatinin huyết thanh`, a division presented as a multiplication in a
renal-dosing section. Without a reference back, a chunk of that section is
grammatical, complete-looking prose with the dosing table silently absent.
Measured: 127 of 167 lifted blocks (76%) came out of `liều lượng và cách
dùng`.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import List
SCHEMA_VERSION = 2
CHUNK_KIND_PROSE = "prose"
CHUNK_KIND_BLOCK_DESCRIPTOR = "block_descriptor"
@dataclass(frozen=True)
class ChunkAttachment:
"""A table or formula that was lifted out of this chunk's section.
`bbox` + `physical_page` are what let the answer layer render the source
crop, which for a quarantined block is the only faithful answer available.
"""
block_id: str
kind: str # "table" | "formula"
shape: str
physical_page: int
bbox: List[float]
quarantined: bool = True
# First row of a `simple_table`, used to make the block findable. Comes
# from pdfplumber and has NOT been verified by eye — the 180 real tables'
# shapes are rule-derived. Retrieval bait, never an answer.
header_row: List[str] = field(default_factory=list)
@dataclass(frozen=True)
class Chunk:
chunk_id: str
drug_id: str
drug_name: str
section_key: str
section_display_name: str
text: str
heading_physical_page: int
source_page_range: List[int]
atc_codes: List[str] = field(default_factory=list)
part_index: int = 0
part_count: int = 1
est_tokens: int = 0
oversized: bool = False
chunk_kind: str = CHUNK_KIND_PROSE
attachments: List[ChunkAttachment] = field(default_factory=list)
# Derivable from `attachments`, stored anyway: the defect this schema
# exists to prevent is a consumer not knowing what it was not told.
has_quarantined_content: bool = False
+102
View File
@@ -0,0 +1,102 @@
"""Vietnamese sentence-boundary splitting for medical formulary text.
ADR 0004 requires splitting at sentence boundaries rather than a blind
character window: `segment/assembler.py` joins body lines at PDF visual
line-wrap points, so a character window can land mid-sentence — and outlier
item 17 measured adult/child dosing sentences ("Người lớn"/"Trẻ em") on
1,121 of ~1,400 monograph pages, where a mid-sentence cut is a
patient-safety defect rather than a cosmetic one.
Boundary rule: `.`, `;`, `:`, `?` or `!` followed by whitespace and an
opening character (uppercase letter or digit), minus the exclusions below.
"""
from __future__ import annotations
import re
from typing import List
_TERMINATORS = ".;:?!"
# Tokens that end in '.' but do not end a sentence.
_ABBREVIATIONS = frozenset({
"v.v", "vv", "tr", "tp", "ts", "bs", "gs", "pgs", "ths", "dr", "st",
"no", "nxb", "cs", "kg", "mg", "ml", "mcg", "gr", "hb", "tm", "tb",
})
_OPENS_SENTENCE = re.compile(r"[A-ZÀ-Ỹ0-9(\-]")
_TRAILING_TOKEN = re.compile(r"([\wÀ-ỹ.]+)\.$")
def _is_abbreviation(left: str) -> bool:
m = _TRAILING_TOKEN.search(left.rstrip())
if not m:
return False
token = m.group(1).rstrip(".").lower()
if token in _ABBREVIATIONS:
return True
# single letter -> an initial ("P." in a name), not a sentence end
return len(token) == 1 and token.isalpha()
def _is_decimal_or_numbering(text: str, i: int) -> bool:
"""A period/comma sitting between digits, or a list numbering like '1. '."""
if text[i] != ".":
return False
prev_ch = text[i - 1] if i > 0 else ""
next_ch = text[i + 1] if i + 1 < len(text) else ""
if prev_ch.isdigit() and next_ch.isdigit():
return True
# "1." / "12." starting a numbered list item: digits preceded by start/newline
j = i - 1
while j >= 0 and text[j].isdigit():
j -= 1
if j < i - 1 and (j < 0 or text[j] in "\n \t("):
return True
return False
def split_sentences(text: str) -> List[str]:
"""Split into sentence-ish units, preserving all characters.
Concatenating the result (without added separators) reproduces the input
exactly — no character is dropped, which the coverage ledger depends on.
"""
if not text:
return []
out: List[str] = []
start = 0
i = 0
n = len(text)
while i < n:
ch = text[i]
if ch not in _TERMINATORS:
i += 1
continue
if _is_decimal_or_numbering(text, i):
i += 1
continue
j = i + 1
if j < n and text[j] in "\")]”’":
j += 1
ws_start = j
while j < n and text[j].isspace():
j += 1
if j == ws_start or j >= n:
i += 1
continue
if not _OPENS_SENTENCE.match(text[j]):
i += 1
continue
if ch == "." and _is_abbreviation(text[start:i + 1]):
i += 1
continue
out.append(text[start:j])
start = j
i = j
if start < n:
out.append(text[start:])
return out
+516
View File
@@ -0,0 +1,516 @@
"""CLI entry point: `python -m ingestion.cli <subcommand>`.
`run` is the real Phase 1 pipeline (extract -> segment -> write). `validate`,
`visual-diff`, and `scaffold-golden` are Phase 1.4-1.7 work — declared here
now (per the approved plan's CLI contract) but not yet implemented; they
raise `NotImplementedError` explicitly rather than silently no-op-ing.
"""
from __future__ import annotations
import argparse
import json
import sys
from collections import Counter
from pathlib import Path
import fitz
from .extract import (
extract_spans,
load_transcribed_runs,
merge_outlined_runs,
index_formula_regions_by_page,
load_formula_regions,
scan_glyph_order,
scan_reading_order,
)
from .chunk import (
CHUNK_KIND_PROSE,
chunk_all,
read_monographs_jsonl,
write_chunks_jsonl,
)
from .segment import DuplicateDrugIdError, assemble, write_monographs_jsonl
from .tables import (
detect_table_regions,
index_by_page,
read_regions_json,
write_regions_json,
)
from .validation import (
FRACTION_BAR_CANDIDATE,
corpus_size,
evaluate,
evaluate_chunks,
read_chunks,
read_monographs,
UNCLASSIFIED,
compute_recall_precision,
parse_back_index,
)
from .validation import scan_document as scan_residual_ink
def _extracted_and_repaired_spans(doc, verbose: bool = False):
"""The span stream, with vector-outlined text put back into it.
Outlier-catalog item 24: 51 runs of type exist only as vector paths, so
extraction alone leaves holes mid-sentence ("Độ ổn định" -> "Độ n định").
Every command that builds monographs must repair the stream the same way,
or the ledger and the output describe different pipelines.
"""
spans = list(extract_spans(doc))
if verbose:
print(f"extracted {len(spans)} spans")
runs = load_transcribed_runs()
if not runs:
return spans
spans = merge_outlined_runs(spans, runs, doc=doc)
if verbose:
chars = sum(len(r.text) for r in runs)
print(f"merged {len(runs)} transcribed vector-outlined runs "
f"({chars} characters) back into the stream")
return spans
def _region_index(tables_arg, verbose: bool = False):
"""Every region whose spans must be lifted out of prose, keyed by page.
`run` and `coverage` must build this the same way — when `coverage` loaded
only tables while `run` also loaded formulas, the ledger described a
pipeline that was not the one producing the output.
"""
index = {}
regions_path = Path(tables_arg) if tables_arg else None
if regions_path and regions_path.exists():
index = index_by_page(read_regions_json(regions_path))
if verbose:
real = sum(len(v) for v in index.values())
print(f"loaded {real} table regions on {len(index)} pages "
f"from {regions_path}")
elif regions_path and verbose:
print(f"note: no table region map at {regions_path} — table text will "
f"stay in section prose (run 'detect-tables' to produce one)")
formulas = load_formula_regions()
for page, page_formulas in index_formula_regions_by_page(formulas).items():
index.setdefault(page, []).extend(page_formulas)
if formulas and verbose:
print(f"loaded {len(formulas)} verified 2D formula regions on "
f"{len({f.physical_page for f in formulas})} pages")
return index or None
def _cmd_run(args: argparse.Namespace) -> int:
pdf_path = Path(args.pdf)
if not pdf_path.exists():
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
return 1
doc = fitz.open(pdf_path)
print(f"opened {pdf_path} ({doc.page_count} pages)")
glyph_issues = scan_glyph_order(doc)
reading_issues = scan_reading_order(doc)
total_defects = len(glyph_issues) + len(reading_issues)
if total_defects:
print(
f"glyph/reading-order sanity gate: {len(glyph_issues)} within-span + "
f"{len(reading_issues)} cross-fragment issue(s) found "
f"(see docs/pdf-parsing-outlier-catalog.md item 9 for known cases; "
f"formula-region issues are expected there, not auto-corrected)."
)
spans = _extracted_and_repaired_spans(doc, verbose=True)
table_index = _region_index(args.tables, verbose=True)
try:
monographs = list(assemble(spans, table_index=table_index))
except DuplicateDrugIdError as e:
print(f"error: {e}", file=sys.stderr)
return 1
table_blocks = sum(len(m.tables) for m in monographs)
quarantined = sum(1 for m in monographs for t in m.tables if t.quarantined)
if table_index:
print(f"lifted {table_blocks} table blocks out of section prose "
f"({quarantined} quarantined)")
out_path = Path(args.out)
out_path.parent.mkdir(parents=True, exist_ok=True)
count = write_monographs_jsonl(monographs, out_path)
print(f"wrote {count} monographs to {out_path}")
return 0
def _cmd_validate(args: argparse.Namespace) -> int:
pdf_path = Path(args.pdf)
if not pdf_path.exists():
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
return 1
doc = fitz.open(pdf_path)
spans = list(extract_spans(doc))
try:
monographs = list(assemble(spans))
except DuplicateDrugIdError as e:
print(f"error: {e}", file=sys.stderr)
return 1
ground_truth = parse_back_index(doc)
result = compute_recall_precision(monographs, ground_truth)
print(f"detected monographs: {result.total_detected}")
print(f"ground-truth entries: {result.total_ground_truth}")
print(f"recall: {result.recall:.1%} ({result.matched_count}/{result.total_ground_truth})")
print(f"precision: {result.precision:.1%}")
if result.unmatched_ground_truth:
print(f"\nunmatched ground-truth entries (first 20 of {len(result.unmatched_ground_truth)}):")
for entry in result.unmatched_ground_truth[:20]:
print(f" {entry.name}, {entry.printed_page}")
if result.unmatched_detected:
print(f"\nunmatched detected monographs (first 20 of {len(result.unmatched_detected)}):")
for m in result.unmatched_detected[:20]:
print(f" {m.drug_name} (physical page {m.source_page_range[0]})")
return 0
def _cmd_detect_tables(args: argparse.Namespace) -> int:
pdf_path = Path(args.pdf)
if not pdf_path.exists():
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
return 1
regions = list(detect_table_regions(pdf_path))
out_path = Path(args.out)
count = write_regions_json(regions, out_path)
shapes = Counter(r.shape for r in regions)
real = sum(1 for r in regions if r.is_real_table)
print(f"detected {count} candidate regions on "
f"{len({r.physical_page for r in regions})} pages")
for shape, n in shapes.most_common():
print(f" {shape:32} {n:5}")
print(f"real tables: {real} not tables: {count - real}")
print(f"wrote {out_path}")
return 0
def _cmd_coverage(args: argparse.Namespace) -> int:
"""Span-level coverage ledger: where did every span end up?
Characters cannot be balanced directly — normalization joins, substitutes
and drops them — so each span is assigned a state and characters are
aggregated from those states.
"""
pdf_path = Path(args.pdf)
if not pdf_path.exists():
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
return 1
doc = fitz.open(pdf_path)
spans = _extracted_and_repaired_spans(doc)
table_index = _region_index(args.tables)
ledger: list = []
list(assemble(spans, table_index=table_index, ledger=ledger))
header, rows = ledger[0], ledger[1:]
by_state = Counter(r["state"] for r in rows)
chars = Counter()
for r in rows:
chars[r["state"]] += r["chars"]
total_spans = len(rows)
total_chars = sum(chars.values())
print(f"SCOPE: {pdf_path} — all {doc.page_count} pages")
print(f"raw chars before span merge: {header['raw_chars_before_merge']:,}")
print(f"spans after merge: {total_spans:,} chars: {total_chars:,}")
print()
print(f"{'state':<24}{'spans':>10}{'% spans':>10}{'chars':>14}{'% chars':>10}")
print("-" * 68)
for state, n in by_state.most_common():
print(f"{state:<24}{n:>10,}{n/total_spans*100:>9.1f}%"
f"{chars[state]:>14,}{chars[state]/total_chars*100:>9.1f}%")
unassigned = [r for r in rows if r["state"] == "unassigned"]
print()
print(f"UNASSIGNED: {len(unassigned):,} spans, "
f"{sum(r['chars'] for r in unassigned):,} chars")
if unassigned:
pages = Counter(r["physical_page"] for r in unassigned)
print(f" on {len(pages)} pages; worst: {pages.most_common(10)}")
print(" first 10 examples:")
for r in unassigned[:10]:
print(f" p{r['physical_page']} {r['bbox']} {r['text']!r}")
if args.out:
out_path = Path(args.out)
out_path.parent.mkdir(parents=True, exist_ok=True)
out_path.write_text(json.dumps(ledger, ensure_ascii=False), encoding="utf-8")
print(f"wrote full ledger to {out_path}")
return 0
def _cmd_residual_ink(args: argparse.Namespace) -> int:
"""Ask the page, not a detector: what ink did the text layer never emit?
Gate: `unclassified` must reach 0 — every surviving region has to be
named, not silently tolerated.
"""
pdf_path = Path(args.pdf)
if not pdf_path.exists():
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
return 1
doc = fitz.open(pdf_path)
tables_by_page = None
regions_path = Path(args.tables) if args.tables else None
if regions_path and regions_path.exists():
tables_by_page = index_by_page(read_regions_json(regions_path))
print(f"loaded table regions on {len(tables_by_page)} pages from {regions_path}")
pages = range(doc.page_count) if args.pages is None else _parse_pages(args.pages)
pages = list(pages)
findings = list(scan_residual_ink(doc, tables_by_page, pages))
kinds = Counter(kind for _, kind in findings)
print(f"SCOPE: {pdf_path}{len(pages)} of {doc.page_count} pages")
print(f"residual regions: {len(findings)}")
for kind, n in kinds.most_common():
print(f" {kind:<26}{n:>7}")
print(f"\nGATE unclassified = {kinds[UNCLASSIFIED]} (target 0)")
flagged = [(r, k) for r, k in findings
if k in (FRACTION_BAR_CANDIDATE, UNCLASSIFIED)]
print(f"needs eyes on it: {len(flagged)} region(s) on "
f"{len({r.physical_page for r, _ in flagged})} pages")
for region, kind in flagged[:20]:
print(f" p{region.physical_page:<5} {kind:<24} "
f"w={region.width_pt:6.1f} h={region.height_pt:5.1f} "
f"bbox={region.bbox}")
if args.out:
out_path = Path(args.out)
out_path.parent.mkdir(parents=True, exist_ok=True)
payload = [
{"physical_page": r.physical_page, "bbox": list(r.bbox),
"ink_px": r.ink_px, "width_pt": round(r.width_pt, 2),
"height_pt": round(r.height_pt, 2), "kind": k}
for r, k in findings
]
out_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
print(f"wrote {len(payload)} regions to {out_path}")
return 0
def _parse_pages(spec: str):
"""Parse '202', '200-210' or '202,1042' into physical page numbers."""
pages = []
for part in spec.split(","):
if "-" in part:
start, end = part.split("-", 1)
pages.extend(range(int(start), int(end) + 1))
else:
pages.append(int(part))
return pages
def _cmd_chunk(args: argparse.Namespace) -> int:
"""Build retrieval chunks from the segmented monographs."""
monographs_path = Path(args.monographs)
if not monographs_path.exists():
print(f"error: no monographs at {monographs_path} — run 'run' first",
file=sys.stderr)
return 1
header_rows = {}
regions_path = Path(args.tables) if args.tables else None
if regions_path and regions_path.exists():
header_rows = {r.table_id: r.first_row
for r in read_regions_json(regions_path)}
monographs = list(read_monographs_jsonl(monographs_path))
chunks = list(chunk_all(monographs, header_rows))
kinds = Counter(c.chunk_kind for c in chunks)
with_attachments = sum(1 for c in chunks
if c.chunk_kind == CHUNK_KIND_PROSE and c.attachments)
oversized = sum(1 for c in chunks if c.oversized)
tokens = sum(c.est_tokens for c in chunks)
print(f"SCOPE: {monographs_path}{len(monographs)} monographs")
print(f"chunks: {len(chunks)}")
for kind, n in kinds.most_common():
print(f" {kind:<22}{n:>7}")
print(f"prose chunks carrying a lifted block: {with_attachments}")
print(f"oversized (over the {800}-token ceiling): {oversized}")
print(f"estimated tokens (chars/4, an estimate): {tokens:,}")
out_path = Path(args.out)
written = write_chunks_jsonl(chunks, out_path)
print(f"wrote {written} chunks to {out_path}")
return 0
def _cmd_chunk_ready(args: argparse.Namespace) -> int:
"""Every gate that must hold before the corpus may be chunked.
Chunking turns text into embeddings, where a defect stops being
inspectable — so each invariant is printed with its own count and its own
target rather than folded into one verdict.
"""
monographs_path = Path(args.monographs)
if not monographs_path.exists():
print(f"error: no monographs at {monographs_path} — run 'run' first",
file=sys.stderr)
return 1
monographs = read_monographs(monographs_path)
runs = load_transcribed_runs()
gates = evaluate(monographs, [
{"physical_page": r.physical_page, "text": r.text} for r in runs
])
size = corpus_size(monographs)
print(f"SCOPE: {monographs_path}{size['monographs']} monographs, "
f"{size['sections']} sections, {size['section_chars']:,} characters")
print(f" {size['quarantined_blocks']} quarantined table/formula blocks "
f"(excluded from prose, citable only with their source crop)")
print()
print(f"{'gate':<34}{'count':>8}{'target':>8} result")
print("-" * 62)
for gate in gates:
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
f"{'PASS' if gate.passed else 'FAIL'}"
+ (f" {gate.detail}" if gate.detail else ""))
chunks_path = Path(args.chunks)
if chunks_path.exists():
chunk_gates = evaluate_chunks(monographs, read_chunks(chunks_path))
print()
print(f"ADR 0006 — chunk references ({chunks_path}):")
for gate in chunk_gates:
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
f"{'PASS' if gate.passed else 'FAIL'}"
+ (f" {gate.detail}" if gate.detail else ""))
gates = gates + chunk_gates
else:
print()
print(f"note: no chunks at {chunks_path} — ADR 0006 gates not run "
f"(run 'chunk' to produce them)")
failed = [g for g in gates if not g.passed]
print()
if failed:
print(f"NOT READY TO CHUNK — {len(failed)} gate(s) failing: "
+ ", ".join(g.name for g in failed))
return 1
print("READY TO CHUNK — every gate above met its target.")
print("Not proven by these gates: content accuracy against the source "
"(no whole-document human-reviewed ground truth exists), table "
"row/column reconstruction, and recall for borderless tables and "
"bar-less formulas.")
return 0
def _cmd_not_implemented(name: str):
def _cmd(_args: argparse.Namespace) -> int:
raise NotImplementedError(
f"'{name}' is planned (see the approved segmentation/eval plan) but not yet built."
)
return _cmd
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="python -m ingestion.cli")
sub = parser.add_subparsers(dest="command", required=True)
p_run = sub.add_parser("run", help="Extract + segment the PDF into monographs.jsonl")
p_run.add_argument("--pdf", required=True, help="Path to the source PDF")
p_run.add_argument(
"--out", default="data/processed/monographs.jsonl",
help="Output JSONL path (default: data/processed/monographs.jsonl)",
)
p_run.add_argument(
"--tables", default="data/processed/table_regions.json",
help="Table region map from 'detect-tables'. When present, table text "
"is lifted out of section prose (default: "
"data/processed/table_regions.json)",
)
p_run.set_defaults(func=_cmd_run)
p_validate = sub.add_parser("validate", help="Whole-book recall/precision vs. back-of-book index")
p_validate.add_argument("--pdf", required=True)
p_validate.set_defaults(func=_cmd_validate)
p_tables = sub.add_parser(
"detect-tables",
help="Locate + classify table regions (slow; result is cached and reused)",
)
p_tables.add_argument("--pdf", required=True)
p_tables.add_argument(
"--out", default="data/processed/table_regions.json",
help="Output region-map path (default: data/processed/table_regions.json)",
)
p_tables.set_defaults(func=_cmd_detect_tables)
p_cov = sub.add_parser(
"coverage", help="Span-level coverage ledger — where every span ended up")
p_cov.add_argument("--pdf", required=True)
p_cov.add_argument("--tables", default="data/processed/table_regions.json")
p_cov.add_argument("--out", default="data/processed/coverage_ledger.json")
p_cov.set_defaults(func=_cmd_coverage)
p_residual = sub.add_parser(
"residual-ink",
help="Ink on the page that no extracted span accounts for "
"(no ground truth needed; gate: unclassified = 0)",
)
p_residual.add_argument("--pdf", required=True)
p_residual.add_argument("--tables", default="data/processed/table_regions.json")
p_residual.add_argument(
"--pages", default=None,
help="Limit to pages, e.g. '202' or '200-210' or '202,1042' "
"(default: every page)",
)
p_residual.add_argument("--out", default="data/processed/residual_ink.json")
p_residual.set_defaults(func=_cmd_residual_ink)
p_ready = sub.add_parser(
"chunk-ready",
help="Named gates that must all hold before chunking (garbage-in guard)")
p_ready.add_argument(
"--monographs", default="data/processed/monographs.jsonl")
p_ready.add_argument("--chunks", default="data/processed/chunks.jsonl")
p_ready.set_defaults(func=_cmd_chunk_ready)
p_chunk = sub.add_parser(
"chunk", help="Build retrieval chunks (ADR 0004/0005/0006)")
p_chunk.add_argument("--monographs", default="data/processed/monographs.jsonl")
p_chunk.add_argument("--tables", default="data/processed/table_regions.json")
p_chunk.add_argument("--out", default="data/processed/chunks.jsonl")
p_chunk.set_defaults(func=_cmd_chunk)
p_visual = sub.add_parser("visual-diff", help="Render a page with detected boundaries overlaid")
p_visual.set_defaults(func=_cmd_not_implemented("visual-diff"))
p_scaffold = sub.add_parser("scaffold-golden", help="Draft golden-set entries for human review")
p_scaffold.set_defaults(func=_cmd_not_implemented("scaffold-golden"))
return parser
def main(argv=None) -> int:
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
sys.stderr.reconfigure(encoding="utf-8")
parser = build_parser()
args = parser.parse_args(argv)
return args.func(args)
if __name__ == "__main__":
raise SystemExit(main())
+36
View File
@@ -0,0 +1,36 @@
from .glyph_order import (
GlyphOrderIssue,
ReadingOrderIssue,
find_reading_order_issues,
is_reversed_order,
scan_glyph_order,
scan_reading_order,
)
from .formulas import index_formula_regions_by_page, load_formula_regions
from .models import Span
from .outlined_text import (
OutlinedTextRun,
detect_outlined_text,
load_transcribed_runs,
)
from .repair import merge_outlined_runs
from .page_map import build_page_map
from .spans import extract_spans
__all__ = [
"Span",
"OutlinedTextRun",
"load_formula_regions",
"index_formula_regions_by_page",
"detect_outlined_text",
"load_transcribed_runs",
"merge_outlined_runs",
"build_page_map",
"extract_spans",
"GlyphOrderIssue",
"is_reversed_order",
"scan_glyph_order",
"ReadingOrderIssue",
"find_reading_order_issues",
"scan_reading_order",
]
+74
View File
@@ -0,0 +1,74 @@
"""2D (stacked-fraction) formula regions, loaded from a verified list.
Why a curated file and not a detector: the residual-ink check produces
*candidates* — thin ink bars that no extracted span accounts for — and its
measured precision on this book is 16 of 23, **69.6%**. The seven misses are
decorative underlines on the Ministry decision page, ruled boxes and table
borders. A 70%-precise rule must not be allowed to quarantine content on its
own, so every candidate was rendered and read, and only the confirmed ones
are listed in `data/verified/formula_regions_2d.json`.
The stored bbox is the fraction bar itself. The numerator sits above it and
the denominator below, so the bar is grown vertically here to cover the whole
formula. The growth factor is deliberately generous: over-capturing a line of
neighbouring prose into a quarantined block is recoverable, leaving half a
formula in the prose is not.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Dict, List
from ..tables.classify import SHAPE_FORMULA_2D
from ..tables.models import TableRegion
# One line of body type on this book measures ~10.5pt; a fraction spans the
# numerator line, the bar and the denominator line.
FORMULA_BAND_HEIGHT_PT = 13.0
# Wide on purpose. The bar is often narrower than the numerator above it, and
# a numerator span can carry leading spaces that push its box's centre well to
# the left of the bar: at 4pt of margin, AMPICILIN VÀ SULBACTAM's numerator
# 'Thể trọng (kg)' stayed behind in the prose while the rest of the fraction
# was lifted. A fraction sits alone on its lines, so taking most of the column
# width costs at worst a neighbouring line inside a quarantined block.
FORMULA_SIDE_MARGIN_PT = 95.0
DEFAULT_VERIFIED_PATH = (
Path(__file__).resolve().parents[2] / "data" / "verified"
/ "formula_regions_2d.json"
)
def load_formula_regions(path: Path | None = None) -> List[TableRegion]:
"""Read the verified 2D-formula regions as page regions to divert."""
source = path or DEFAULT_VERIFIED_PATH
if not source.exists():
return []
payload = json.loads(source.read_text(encoding="utf-8"))
regions = []
for index, entry in enumerate(payload["regions"]):
x0, y0, x1, y1 = entry["bar_bbox"]
regions.append(
TableRegion(
table_id=f"p{entry['physical_page']}_f{index}",
physical_page=entry["physical_page"],
bbox=(x0 - FORMULA_SIDE_MARGIN_PT, y0 - FORMULA_BAND_HEIGHT_PT,
x1 + FORMULA_SIDE_MARGIN_PT, y1 + FORMULA_BAND_HEIGHT_PT),
n_rows=2,
n_cols=1,
shape=SHAPE_FORMULA_2D,
)
)
return regions
def index_formula_regions_by_page(
regions: List[TableRegion],
) -> Dict[int, List[TableRegion]]:
index: Dict[int, List[TableRegion]] = {}
for region in regions:
index.setdefault(region.physical_page, []).append(region)
return index
+178
View File
@@ -0,0 +1,178 @@
"""Mandatory glyph/reading-order sanity check.
Two distinct defect shapes were confirmed by testing this module against the
real PDF (not assumed from the ADR description alone):
1. **Within-span glyph reversal** (`scan_glyph_order` / `GlyphOrderIssue`):
physical page 1373 (0-indexed) contains a span whose characters are
positioned in strictly decreasing x-origin order, producing scrambled
text (e.g. "= tịx 8 y..." instead of "y 8 xịt ="). Matches ADR 0003's
original description.
2. **Cross-span row misordering within one PyMuPDF block**
(`scan_reading_order` / `ReadingOrderIssue`) — a genuinely different,
previously undocumented shape found while testing this module end-to-end:
physical page 714 has a visual text row split into multiple PyMuPDF line
objects, within a single `block`, that are emitted out of left-to-right
order relative to each other (each individual span's own characters are
fine, but the fragments interleave incorrectly), e.g. the row "...bảo
quản nhiệt độ..." is emitted as fragments "quản ", "", "đ tệih", "n " in
that (wrong) order. Concatenating characters in raw extraction order
produces garbled text; re-sorting the *same* characters within one visual
row by x-origin recovers the correct reading order exactly. This means
ADR 0003's "exactly 1 occurrence in the whole book" claim was based on a
narrower (within-span-only) check and undercounted the real defect
population — corrected here, see docs/pdf-parsing-outlier-catalog.md
item 9 update.
**Getting the row-grouping key right took three iterations, each caught by
running against the real book rather than trusting the first result (per
CLAUDE.md's no-fabrication rule) — recorded here since the failure modes
generalize to any from-scratch "reconstruct visual rows from raw
coordinates" approach:**
- v1 (group by rounded y only): 1113 "issues", almost all false positives.
- v2 (group by (`_column_for_x` tag, rounded y), using the same ±20pt
tolerance `extract/spans.py` uses for informational span tagging): dropped
to 32, but a real false-positive class remained — kerning jitter (e.g.
"mefloquin"'s 'l'/'o' origins differ by only 0.095pt, well inside normal
font kerning) was treated as a reversal with no decrease tolerance, and
the ±20pt column tolerance creates an *overlapping* accepted x-range for
"left" (24-319) and "right" (288-582) — a right-column paragraph
starting near x=299 was misclassified "left" and merged with an unrelated
left-column line sharing the same y.
- v3 (this version — group by (PyMuPDF's own `block` index, rounded y)):
the real fix. Two paragraphs from genuinely different columns (e.g. page
1104: one block starting at x=299.4, another at x=35.4, both at y=70.4)
turned out to sit in **different PyMuPDF blocks**, while page 714's 3
genuinely-misordered fragments sit in the **same block** (block 20) split
across multiple `line` entries. Block identity — PyMuPDF's own layout
analysis, already validated in ADR 0003 to respect this document's
two-column structure — is a reliable discriminator that no fixed
x-coordinate threshold can be, since real paragraph start positions vary
enough to overlap any hand-picked column boundary. A minimum-decrease
threshold (`_MIN_DECREASE_PT`, well above observed kerning jitter <0.3pt
and well below observed real defects >2pt) still guards against sub-pixel
jitter within a block/row. The header band (running page number + drug
name, two unrelated boilerplate fields sharing a y-coordinate — stripped
before chunking regardless, outlier-catalog item 13) is excluded outright.
Both checks are cheap (seconds per full-book pass) and must run over 100% of
pages, not sampled, per ADR 0003's standing rigor bar.
"""
from __future__ import annotations
from collections import defaultdict
from dataclasses import dataclass
from typing import Dict, List, Sequence, Tuple
import fitz
_ROW_Y_PRECISION = 1 # decimal places; same-baseline chars share y to <0.01pt in practice
_HEADER_BAND_Y = 60.0 # page number + running drug name live here; boilerplate, stripped separately
_MIN_DECREASE_PT = 1.0 # observed kerning jitter <0.3pt; observed real defects >2pt — safely between
@dataclass(frozen=True)
class GlyphOrderIssue:
physical_page: int
span_bbox: Tuple[float, float, float, float]
original_text: str
corrected_text: str
@dataclass(frozen=True)
class ReadingOrderIssue:
physical_page: int
block_index: int
row_y: float
extracted_text: str
corrected_text: str
def is_reversed_order(x_origins: Sequence[float]) -> bool:
"""True if every consecutive pair strictly decreases in x — the exact
shape of the confirmed within-span defect. A normal LTR span's
x-origins strictly increase; requiring *every* pair to decrease (not
just "not sorted") avoids false-triggering on ordinary spans.
"""
if len(x_origins) < 2:
return False
# strict=False on purpose: this is the adjacent-pair idiom, so the two
# sequences differ in length by one by construction.
return all(b < a for a, b in zip(x_origins, x_origins[1:], strict=False))
def scan_glyph_order(doc: fitz.Document) -> List[GlyphOrderIssue]:
issues: List[GlyphOrderIssue] = []
for pno in range(doc.page_count):
for block in doc[pno].get_text("rawdict").get("blocks", []):
for line in block.get("lines", []):
for span in line.get("spans", []):
chars = span.get("chars", [])
if not chars:
continue
x_origins = [c["origin"][0] for c in chars]
if is_reversed_order(x_origins):
issues.append(GlyphOrderIssue(
physical_page=pno,
span_bbox=tuple(span["bbox"]),
original_text="".join(c["c"] for c in chars),
corrected_text="".join(c["c"] for c in reversed(chars)),
))
return issues
def _has_significant_backward_jump(xs: Sequence[float], min_decrease: float) -> bool:
return any(b < a - min_decrease for a, b in zip(xs, xs[1:], strict=False))
def find_reading_order_issues(
chars_by_row: Dict[Tuple[int, float], List[Tuple[float, str]]],
min_decrease: float = _MIN_DECREASE_PT,
) -> List[ReadingOrderIssue]:
"""Pure logic, unit-testable without a real PDF: given characters already
grouped by (block_index, row_y) in raw extraction order, flag a row only
when it contains a backward x-jump larger than `min_decrease` — ordinary
font kerning produces sub-0.3pt jitter (see module docstring's v2
entry), so a plain "resorting changes the text" check without this
threshold is not reliable; it self-corrupts already-correct text.
Grouping by block index (not a hand-picked x-coordinate column
boundary) is what the caller must guarantee — see module docstring's
v1/v2/v3 history for why a coordinate-based row reconstruction alone is
not safe.
"""
issues = []
for (block_index, y), chars in chars_by_row.items():
if len(chars) < 2:
continue
xs = [x for x, _ in chars]
if not _has_significant_backward_jump(xs, min_decrease):
continue
extracted = "".join(c for _, c in chars)
corrected = "".join(c for _, c in sorted(chars, key=lambda t: t[0]))
if extracted != corrected:
issues.append(ReadingOrderIssue(
physical_page=-1, block_index=block_index, row_y=y,
extracted_text=extracted, corrected_text=corrected,
))
return issues
def scan_reading_order(doc: fitz.Document) -> List[ReadingOrderIssue]:
issues: List[ReadingOrderIssue] = []
for pno in range(doc.page_count):
rows: Dict[Tuple[int, float], List[Tuple[float, str]]] = defaultdict(list)
for block_index, block in enumerate(doc[pno].get_text("rawdict").get("blocks", [])):
for line in block.get("lines", []):
for span in line.get("spans", []):
for c in span.get("chars", []):
x, y = c["origin"]
if y < _HEADER_BAND_Y:
continue
rows[(block_index, round(y, _ROW_Y_PRECISION))].append((x, c["c"]))
for issue in find_reading_order_issues(rows):
issues.append(ReadingOrderIssue(
physical_page=pno, block_index=issue.block_index, row_y=issue.row_y,
extracted_text=issue.extracted_text, corrected_text=issue.corrected_text,
))
return issues
+28
View File
@@ -0,0 +1,28 @@
"""Persists the extracted span stream so re-segmentation doesn't require
re-running PyMuPDF over the whole PDF every time.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Iterable, Iterator
from .models import Span
def write_spans_jsonl(spans: Iterable[Span], path: Path) -> int:
count = 0
with open(path, "w", encoding="utf-8") as f:
for span in spans:
f.write(json.dumps(span.__dict__, ensure_ascii=False) + "\n")
count += 1
return count
def read_spans_jsonl(path: Path) -> Iterator[Span]:
with open(path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
yield Span(**json.loads(line))
+43
View File
@@ -0,0 +1,43 @@
"""Data model for text spans extracted from the source PDF.
A Span is one PyMuPDF text span (a run of characters sharing one font/size),
tagged with page and column position. This is the sole unit `segment/`
consumes — it never touches PyMuPDF or fitz.Document directly (see ADR 0003
and docs/pdf-parsing-outlier-catalog.md for why: PyMuPDF is the validated
sole general-text extractor for this document).
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class Span:
physical_page: int
printed_page: Optional[int]
column: str # "left" | "right" | "full_width" | "unknown"
block: int
line: int
span_index: int
x0: float
y0: float
x1: float
y1: float
text: str
font: str
size: float
@property
def bold(self) -> bool:
return "Bold" in self.font
@property
def span_id(self) -> str:
"""Stable identifier for one source span.
Built from PyMuPDF's own page/block/line/span indices, so the same PDF
always yields the same id — a counter would renumber whenever anything
upstream changed, which makes downstream provenance unverifiable.
"""
return f"p{self.physical_page}_b{self.block}_l{self.line}_s{self.span_index}"
@@ -0,0 +1,108 @@
"""Text that was drawn as vector outlines instead of text operators.
Physical page 714 prints 17 lines of ordinary Gatifloxacin prose that no text
extractor returns: `page.get_text()` omits them, `page.search_for()` finds
nothing, `pdfplumber` and `opendataloader-pdf` omit them too. They are not
text at all in the file — each line is a filled path of ~1,600-1,800 items,
shaped exactly like one line of type and filled with the body-text colour.
Nothing that asks a text layer can see this, which is why it survived every
earlier check in this project. It was found by masking extracted spans over a
rendered page and looking at the ink that was left.
Detection is deliberately shape-based, not content-based: a filled drawing
with hundreds of path items whose box is the height of one line and at least
30pt wide. Recovery cannot be automatic — the glyphs carry no character
codes — so these regions are reported for transcription, never guessed at.
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Iterator, List, Tuple
import fitz
# Measured whole-document. Full outlined lines carry 1,126-1,831 path items;
# single outlined glyphs carry 39-45. Ordinary decoration (the running-header
# rule, cell borders) carries 1-2, so 30 separates them cleanly. Lowering the
# threshold from 200 to 30 was checked before it was applied: it adds 29 runs
# and no new page, all on pages 714 and 736, which were already affected.
MIN_PATH_ITEMS = 30
MIN_RUN_WIDTH_PT = 2.0
RUN_HEIGHT_RANGE_PT = (3.0, 20.0)
@dataclass(frozen=True)
class OutlinedTextRun:
"""A run of type that exists only as vector paths — a line, or one glyph.
`text` stays empty unless a human (or a rendered-page reading) fills it
in: the paths carry no character codes, so any text here is a
transcription and must be recorded as one.
"""
physical_page: int
bbox: Tuple[float, float, float, float]
path_items: int
text: str = ""
@property
def is_transcribed(self) -> bool:
return bool(self.text)
def _is_outlined_run(drawing: dict, page_width: float) -> bool:
rect = drawing["rect"]
low, high = RUN_HEIGHT_RANGE_PT
return (
drawing["type"] == "f"
and len(drawing["items"]) >= MIN_PATH_ITEMS
and low <= rect.height <= high
and rect.width >= MIN_RUN_WIDTH_PT
and rect.x0 >= 0
and rect.x1 <= page_width + 1
)
def detect_outlined_text(
doc: "fitz.Document", pages: Iterable[int] | None = None,
) -> Iterator[OutlinedTextRun]:
"""Yield every run of vector-outlined type in the document."""
page_numbers = range(doc.page_count) if pages is None else pages
for number in page_numbers:
page = doc[number]
for drawing in page.get_drawings():
if not _is_outlined_run(drawing, page.rect.x1):
continue
rect = drawing["rect"]
yield OutlinedTextRun(
physical_page=number,
bbox=(round(rect.x0, 2), round(rect.y0, 2),
round(rect.x1, 2), round(rect.y1, 2)),
path_items=len(drawing["items"]),
)
DEFAULT_TRANSCRIPTIONS_PATH = (
Path(__file__).resolve().parents[2] / "data" / "verified"
/ "outlined_text_transcriptions.json"
)
def load_transcribed_runs(path: "Path | None" = None) -> List[OutlinedTextRun]:
"""Read the transcribed runs. Every `text` here was read off a rendering."""
source = path or DEFAULT_TRANSCRIPTIONS_PATH
if not source.exists():
return []
payload = json.loads(source.read_text(encoding="utf-8"))
return [
OutlinedTextRun(
physical_page=run["physical_page"],
bbox=tuple(run["bbox"]),
path_items=run["path_items"],
text=run["text"],
)
for run in payload["runs"]
]
+70
View File
@@ -0,0 +1,70 @@
"""Maps physical (0-indexed) page numbers to the book's own printed folio
number, by reading the isolated numeric token in each page's header band.
Required because ADR 0003's page-range rules ("monographs run printed pages
99-1496") are meaningless without a real per-page mapping — verified rather
than assumed to be a constant offset, since front matter in some books uses
roman numerals or restarts numbering. In this book the mapping is empirically
a constant (physical + 1) across the entire 1668 pages (verified against the
milestone pages: physical 36->printed 37, physical 98->printed 99, physical
100->printed 101 "Abacavir", physical 1496->printed 1497), but this module
still reads the real folio per page rather than hard-coding that constant, so
a future edition/scan with different numbering does not silently mis-map.
Confirmed real false-conflict case (physical page 1243, "RIBOFLAVIN (Vitamin
B2)" monograph, found via a whole-book `cli validate` run and confirmed by
rendering the page to an image): the monograph's own title sits high enough
on the page that its "2" subscript (size 5.83) falls inside the header band
alongside the real folio "1244" (size 10.0), producing two conflicting
digit-only candidates and silently dropping the printed page — and with it
the entire monograph, since every span on the page then fails the
printed-page-range check. A genuine folio is set in the header's own running
font size, never a subscript's reduced size, so preferring the
largest-font-size candidate(s) resolves this without weakening the
"never guess on a real conflict" rule for pages with, e.g., two same-size
candidates (still returns None).
"""
from __future__ import annotations
import re
from typing import Dict, List, Optional, Tuple
import fitz
_FOLIO_RE = re.compile(r"^\d{1,4}$")
HEADER_BAND_Y = 60.0 # printed folio always appears in the top header band
def build_page_map(doc: fitz.Document) -> Dict[int, Optional[int]]:
"""Returns {physical_page: printed_page_or_None}. None means no folio
was recoverable (blank/separator pages, title pages) — a valid state,
not an error.
"""
return {pno: _read_folio(doc[pno]) for pno in range(doc.page_count)}
def _read_folio(page: fitz.Page) -> Optional[int]:
candidates = [] # (text, size) pairs
for block in page.get_text("dict").get("blocks", []):
for line in block.get("lines", []):
for span in line.get("spans", []):
text = span["text"].strip()
if span["bbox"][1] < HEADER_BAND_Y and _FOLIO_RE.match(text):
candidates.append((text, span["size"]))
return pick_folio(candidates)
def pick_folio(candidates: List[Tuple[str, float]]) -> Optional[int]:
"""Pure decision logic, given the header-band digit-only (text, size)
candidates already collected from a page: which one is the real folio.
"""
if not candidates:
return None # blank/separator page: unrecoverable, never guess.
max_size = max(size for _, size in candidates)
largest = {text for text, size in candidates if size == max_size}
if len(largest) == 1:
return int(largest.pop())
# still conflicting even after dropping smaller-font stray digits
# (e.g. subscripts): genuinely ambiguous, never guess.
return None
+241
View File
@@ -0,0 +1,241 @@
"""Put transcribed vector-outlined text back into the span stream.
Outlier-catalog item 24: 51 runs of type on 5 pages exist only as filled
vector paths, so no extractor emits a span for them. They were transcribed by
reading rendered crops (`data/verified/outlined_text_transcriptions.json`).
This module is what makes that transcription part of the corpus rather than a
note beside it.
Placement is geometric, not textual. A run that vertically overlaps an
existing visual line is a character (or fragment) dropped out of *that* line
and is spliced into it in x order — this is the common case and the damaging
one, because a missing diacritic turns "Độ ổn định" into "Độ n định" and
still reads as ordinary prose. A run that overlaps no line is a whole missing
line and is inserted at a line boundary, ordered by column then y, so the
reading order the rest of the pipeline depends on is preserved.
Synthetic spans are marked by `SYNTHETIC_LINE_BASE` in their line index, so
their provenance ids stay distinguishable from real extracted spans forever.
"""
from __future__ import annotations
from dataclasses import replace
from typing import Iterable, List, Sequence, Tuple
from .models import Span
from .outlined_text import OutlinedTextRun
from .page_map import HEADER_BAND_Y
from .spans import classify_column
# Line indices at or above this never come from PyMuPDF — a real page has
# nothing close to this many lines in a block.
SYNTHETIC_LINE_BASE = 100_000
SYNTHETIC_FONT = "TimesNewRomanPSMT"
SYNTHETIC_SIZE = 9.5
# A run counts as belonging to an existing line when their vertical extents
# overlap by at least this fraction of the run's height.
LINE_OVERLAP_RATIO = 0.5
def _vertical_overlap(a: Tuple[float, float], b: Tuple[float, float]) -> float:
return max(0.0, min(a[1], b[1]) - max(a[0], b[0]))
def _line_groups(spans: Sequence[Span]) -> List[Tuple[int, int, List[Span]]]:
"""Consecutive spans sharing a visual line, with their index range.
Mirrors `normalize.group_visual_lines`' notion of a line so that a span
spliced here lands in the same group there.
"""
groups: List[Tuple[int, int, List[Span]]] = []
for index, span in enumerate(spans):
key = (span.physical_page, span.block, span.line)
if groups and (groups[-1][2][0].physical_page, groups[-1][2][0].block,
groups[-1][2][0].line) == key:
start, _, members = groups[-1]
members.append(span)
groups[-1] = (start, index, members)
else:
groups.append((index, index, [span]))
return groups
def _synthetic_span(run: OutlinedTextRun, template: Span | None,
line_index: int) -> Span:
x0, y0, x1, y1 = run.bbox
column = classify_column(run.bbox)
if y0 < HEADER_BAND_Y:
# Part of the running header, which is a full-width band. Tagging it
# as such lets the one existing boilerplate rule strip it, instead of
# this module deciding separately what boilerplate is.
column = "full_width"
if template is not None:
return replace(
template,
column=template.column if y0 >= HEADER_BAND_Y else column,
line=line_index,
span_index=0,
x0=x0, y0=y0, x1=x1, y1=y1,
text=run.text,
font=SYNTHETIC_FONT,
)
return Span(
physical_page=run.physical_page,
printed_page=None,
column=column,
block=0,
line=line_index,
span_index=0,
x0=x0, y0=y0, x1=x1, y1=y1,
text=run.text,
font=SYNTHETIC_FONT,
size=SYNTHETIC_SIZE,
)
def char_boxes(doc, page_number: int) -> List[Tuple[str, Tuple[float, ...]]]:
"""Per-character boxes for one page, in extraction order.
Needed because a dropped glyph usually sits *inside* an extracted span,
not between two of them: on physical page 714 the span
`'Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu'` runs from x=35.5 to
x=223.0 and the missing '' belongs at x=167. Splicing at span boundaries
put it at the end and produced 'tuở ổi'. Character geometry is the only
thing that says where the hole actually is.
"""
boxes = []
for block in doc[page_number].get_text("rawdict")["blocks"]:
for line in block.get("lines", []):
for span in line["spans"]:
for char in span["chars"]:
boxes.append((char["c"], char["bbox"]))
return boxes
def _split_offset(span: Span, run: OutlinedTextRun,
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> int | None:
"""Character offset inside `span.text` where the run's glyph belongs."""
inside = [
box for box in boxes
if span.x0 - 0.5 <= box[1][0] and box[1][2] <= span.x1 + 0.5
and span.y0 - 1.0 <= box[1][1] and box[1][3] <= span.y1 + 1.0
]
if len(inside) != len(span.text):
return None
for offset, (_, bbox) in enumerate(inside):
if bbox[0] >= run.bbox[2] - 0.5:
return offset
return None
def _splice_into_line(spans: List[Span], group, run: OutlinedTextRun,
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> None:
start, end, members = group
anchor = members[0]
synthetic = replace(
anchor,
span_index=SYNTHETIC_LINE_BASE,
x0=run.bbox[0], y0=run.bbox[1], x1=run.bbox[2], y1=run.bbox[3],
text=run.text,
font=SYNTHETIC_FONT,
)
for offset, member in enumerate(members):
if not (member.x0 <= run.bbox[0] and run.bbox[2] <= member.x1):
continue
split_at = _split_offset(member, run, boxes)
if split_at is None or split_at == 0:
continue
index = start + offset
left = replace(member, text=member.text[:split_at], x1=run.bbox[0])
right = replace(member, text=member.text[split_at:],
span_index=member.span_index + SYNTHETIC_LINE_BASE,
x0=run.bbox[2])
# A split at the very end of a span leaves a fragment holding nothing
# but a space. Dropping it costs no text — `join_visual_line` decides
# spacing from the horizontal gap, not from a span's own padding.
pieces = [p for p in (left, synthetic, right) if p.text.strip()]
spans[index:index + 1] = pieces
return
position = end + 1
for offset, member in enumerate(members):
if run.bbox[0] < member.x0:
position = start + offset
break
spans.insert(position, synthetic)
def _insert_as_new_line(spans: List[Span], run: OutlinedTextRun,
line_index: int) -> None:
column = classify_column(run.bbox)
if run.bbox[1] < HEADER_BAND_Y:
column = "full_width"
position = len(spans)
template = None
for start, _, members in _line_groups(spans):
first = members[0]
if first.physical_page < run.physical_page:
template = first
continue
if first.physical_page > run.physical_page:
position = start
break
if first.column == column:
template = first
if first.y0 > run.bbox[1]:
position = start
break
elif template is not None and first.column != column and position == len(spans):
# first line of the next column on this page — the run belongs
# before it if we never found a lower line in its own column
position = start
spans.insert(position, _synthetic_span(run, template, line_index))
def merge_outlined_runs(
spans: Iterable[Span], runs: Sequence[OutlinedTextRun], doc=None,
) -> List[Span]:
"""Return the span stream with every transcribed run put back in place.
`doc` enables character-accurate splicing of a glyph that fell out of the
middle of an extracted span. Without it the run can only be placed at a
span boundary, which is wrong for exactly the case that matters most.
"""
merged = list(spans)
ordered = sorted(runs, key=lambda r: (r.physical_page, r.bbox[1], r.bbox[0]))
boxes_cache: dict = {}
for offset, run in enumerate(ordered):
if not run.text:
continue
run_extent = (run.bbox[1], run.bbox[3])
run_column = classify_column(run.bbox)
target = None
for group in _line_groups(merged):
first = group[2][0]
if first.physical_page != run.physical_page:
continue
# Column, not just height: this book sets two columns, so a
# right-column run sits at the same y as an unrelated left-column
# line. Without this, page 714's "…làm thay đ" was spliced onto
# the left column and "…ổi nồng độ glucose máu" stayed broken.
if first.column != run_column:
continue
line_extent = (min(s.y0 for s in group[2]),
max(s.y1 for s in group[2]))
overlap = _vertical_overlap(run_extent, line_extent)
height = max(run.bbox[3] - run.bbox[1], 0.1)
if overlap / height >= LINE_OVERLAP_RATIO:
target = group
break
if target is not None:
if doc is not None and run.physical_page not in boxes_cache:
boxes_cache[run.physical_page] = char_boxes(doc, run.physical_page)
_splice_into_line(merged, target, run,
boxes_cache.get(run.physical_page, ()))
else:
_insert_as_new_line(merged, run, SYNTHETIC_LINE_BASE + offset)
return merged
+107
View File
@@ -0,0 +1,107 @@
"""Continuous cross-page span extraction.
Per ADR 0003: the pipeline must consume text as one continuous cross-page
stream, never per-page silos, so that multi-line headings and paragraphs
spanning a page/column break can be handled correctly downstream. This
module's only job is to yield that stream in reading order; it does not
decide what is a heading or a monograph boundary (that's `segment/`'s job).
Column tagging uses the bounding-box ranges confirmed by inspection in ADR
0003 (left column x~44-299, right column x~308-562, page width ~595).
An earlier version of this module trusted PyMuPDF's own raw block order to
already sequence left-then-right correctly, validated only against one
example page during ADR 0003. Confirmed wrong via a whole-document
character-diff against an independent parser (opendataloader-pdf) plus
visual page reads: on 12 of 1398 monograph-range pages (e.g. physical page
1100, the OXYBUTYNIN/OXYMETAZOLIN boundary), PyMuPDF's raw block order
emits the *right* column before the *left* column. Left uncorrected, this
silently corrupts monograph data at a column-reversed page's drug boundary
— the wrong column's section content (e.g. "Chống chỉ định") gets appended
to whichever monograph is still open when it's encountered, overwriting
that monograph's real section and leaving the next monograph missing it.
Fixed by explicitly sorting blocks (full_width header band first, then
left column, then right column, each by y-position) instead of trusting
raw order — full_width blocks are confirmed to be page-header material
only in this document (real monograph titles and section headings sit
within one column's x-range), so this ordering matches the book's actual
two-column-with-running-header layout.
"""
from __future__ import annotations
from typing import Iterator
import fitz
from .models import Span
from .page_map import build_page_map
_LEFT_COLUMN_X = (44.0, 299.0)
_RIGHT_COLUMN_X = (308.0, 562.0)
_COLUMN_TOLERANCE = 20.0
_FULL_WIDTH_MIN = 400.0
_COLUMN_SORT_RANK = {"full_width": 0, "left": 1, "right": 2, "unknown": 3}
def extract_spans(doc: fitz.Document) -> Iterator[Span]:
page_map = build_page_map(doc)
for pno in range(doc.page_count):
printed = page_map[pno]
page_dict = doc[pno].get_text("dict")
blocks = _sort_blocks_reading_order(page_dict.get("blocks", []))
for block_idx, block in enumerate(blocks):
column = classify_column(block.get("bbox"))
for line_idx, line in enumerate(block.get("lines", [])):
for span_idx, span in enumerate(line.get("spans", [])):
text = span["text"]
if not text.strip():
continue
x0, y0, x1, y1 = span["bbox"]
yield Span(
physical_page=pno,
printed_page=printed,
column=column,
block=block_idx,
line=line_idx,
span_index=span_idx,
x0=x0, y0=y0, x1=x1, y1=y1,
text=text,
font=span["font"],
size=span["size"],
)
def _sort_blocks_reading_order(blocks: list) -> list:
"""Full_width header band first, then left column, then right column,
each by y-position — see module docstring for the confirmed real bug
this replaces (trusting PyMuPDF's raw block order).
"""
return sorted(
blocks,
key=lambda b: (_COLUMN_SORT_RANK[classify_column(b.get("bbox"))], b.get("bbox", (0, 0, 0, 0))[1]),
)
def classify_column(bbox) -> str:
"""Which of the book's two columns a box sits in (or the header band)."""
if bbox is None:
return "unknown"
x0, _, x1, _ = bbox
if (x1 - x0) >= _FULL_WIDTH_MIN:
return "full_width"
mid = (x0 + x1) / 2
# Exact containment before tolerance. The two tolerance bands overlap
# between x=288 and x=319, and testing left first put anything in that
# strip in the left column — invisible for a full-width block, wrong for a
# narrow one. A single 4pt glyph at x=315 on physical page 714 was
# classified left, so the 'ổ' missing from "Độ ổn định" could not be
# matched to its own line and the corruption survived the repair.
if _LEFT_COLUMN_X[0] <= mid <= _LEFT_COLUMN_X[1]:
return "left"
if _RIGHT_COLUMN_X[0] <= mid <= _RIGHT_COLUMN_X[1]:
return "right"
if _LEFT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _LEFT_COLUMN_X[1] + _COLUMN_TOLERANCE:
return "left"
if _RIGHT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _RIGHT_COLUMN_X[1] + _COLUMN_TOLERANCE:
return "right"
return "unknown"
+18
View File
@@ -0,0 +1,18 @@
"""Text normalization shared by the pipeline and any validation script.
Kept as its own stage so the rules live in exactly one place (CLAUDE.md's DRY
rule): `segment/` applies them when assembling section text, and audits
import the same functions rather than re-implementing them.
"""
from .glyphs import PUA_SUBSTITUTIONS, find_unmapped_pua, substitute_pua
from .text_flow import SPACE_GAP_PT, group_visual_lines, join_spans, join_visual_line
__all__ = [
"PUA_SUBSTITUTIONS",
"SPACE_GAP_PT",
"find_unmapped_pua",
"substitute_pua",
"group_visual_lines",
"join_spans",
"join_visual_line",
]
+47
View File
@@ -0,0 +1,47 @@
"""Private-use-area glyph substitution.
The source PDF sets several symbols in the `SymbolTiger` / `Symbol` fonts,
which PyMuPDF faithfully returns as raw Unicode private-use-area codepoints.
Left untranslated they reach embeddings as junk — and 74 of the 86
occurrences in this corpus are the comparison operators inside dosing
sentences, where losing the operator changes clinical meaning ("liều ≤ 100
mg" is not "liều 100 mg").
Every entry below was located in the source PDF, rendered to an image, and
read visually — none inferred from surrounding context. Counts and the page
each was confirmed on are recorded in docs/progress-log.md.
"""
from __future__ import annotations
PUA_LO, PUA_HI = 0xE000, 0xF8FF
# codepoint -> replacement, with the page the glyph was visually confirmed on
PUA_SUBSTITUTIONS: dict[str, str] = {
"": "", # p.141 "trẻ em ≥ 10 tuổi"
"": "", # p.169 "liều ≤ 100 mg"
"": "α", # p.334 "Streptococcus α tan huyết"
"": "", # p.1027 "HCO₃⁻ + H⁺ → H₂CO₃"
"": "®", # p.891 "Plasma Lyte® 56/5%"
"": "", # p.957 "alpha₁-acid glycoprotein"
"": "", # p.1033 "rhodanese ↓" (catalysis arrow)
"": "γ", # p.1352 "interferon - γ"
}
_TABLE = str.maketrans(PUA_SUBSTITUTIONS)
def substitute_pua(text: str) -> str:
return text.translate(_TABLE)
def find_unmapped_pua(text: str) -> list[str]:
"""PUA codepoints with no verified replacement.
Returned rather than silently passed through: an unmapped glyph means the
corpus contains a symbol nobody has visually confirmed yet, which must be
surfaced instead of embedded as junk.
"""
return sorted({
ch for ch in text
if PUA_LO <= ord(ch) <= PUA_HI and ch not in PUA_SUBSTITUTIONS
})
@@ -0,0 +1,83 @@
"""Rejoin PDF spans into flowing text.
`segment/assembler.py` originally appended one line per *span*, so any visual
line that the PDF split into several spans (an italic run, a subscript, a
symbol-font glyph) became several "lines". Measured on the whole corpus that
produced 99,501 mid-sentence line breaks across 71.8% of sections and 11,612
sub-4-character fragment lines — e.g. `"cytochrom P\n450\ngây"`,
`"(\nfeline immunodeficiency virus\n)"`, `"Cl\ncr\n< 50 ml/"`.
Text alone cannot tell a mid-word span split from a genuine line wrap, so the
join is driven by geometry instead: PyMuPDF's own `(block, line)` indices say
which spans share a visual line, and the horizontal gap says whether a space
belongs between them.
"""
from __future__ import annotations
from typing import Iterable, List, Sequence, Tuple
from ..extract.models import Span
# Horizontal gap (pt) above which two spans on one visual line are separated
# by a real space. Kerning noise between adjacent glyph runs sits well under
# 1pt; a space at this book's 9.5-10pt body size is ≈2.4pt.
SPACE_GAP_PT = 1.0
_SENTENCE_END = ".;:!?"
def _line_key(span: Span) -> Tuple[int, int, int]:
return (span.physical_page, span.block, span.line)
def group_visual_lines(spans: Sequence[Span]) -> List[List[Span]]:
"""Group consecutive spans that share a visual line, preserving order."""
lines: List[List[Span]] = []
for span in spans:
if lines and _line_key(lines[-1][0]) == _line_key(span):
lines[-1].append(span)
else:
lines.append([span])
return lines
def join_visual_line(spans: Sequence[Span]) -> str:
"""Concatenate one visual line, inserting a space only where one exists."""
out = ""
previous: Span | None = None
for span in spans:
text = span.text
if previous is not None:
gap = span.x0 - previous.x1
needs_space = (
gap >= SPACE_GAP_PT
and not out.endswith(" ")
and not text.startswith(" ")
)
if needs_space:
out += " "
out += text
previous = span
return out.strip()
def join_spans(spans: Iterable[Span]) -> str:
"""Rejoin spans into flowing text.
A visual line that does not end a sentence is treated as a soft wrap and
joined to the next line with a space; a line ending in sentence
punctuation keeps its newline, which preserves paragraph and list
structure for display and citation.
"""
lines = [join_visual_line(group) for group in group_visual_lines(list(spans))]
lines = [line for line in lines if line]
if not lines:
return ""
out = lines[0]
for line in lines[1:]:
if out.rstrip().endswith(tuple(_SENTENCE_END)):
out += "\n" + line
else:
out += " " + line
return out
+32
View File
@@ -0,0 +1,32 @@
from .assembler import DuplicateDrugIdError, assemble
from .atc import ATCResult, extract_atc_codes, is_stated_absent, normalize_atc_candidate
from .detector import detect_monograph_titles, detect_section_headings
from .io import read_monographs_jsonl, write_monographs_jsonl
from .merge import merge_multiline_headings
from .models import Heading, Monograph, SectionSpan
from .units import normalize_unit_token, validate_unit_tokens
from .vocab import SECTION_DEFS, SectionDef, is_part_divider, match_section, normalize_heading_text
__all__ = [
"Heading",
"SectionSpan",
"Monograph",
"assemble",
"DuplicateDrugIdError",
"detect_monograph_titles",
"detect_section_headings",
"merge_multiline_headings",
"write_monographs_jsonl",
"read_monographs_jsonl",
"ATCResult",
"extract_atc_codes",
"is_stated_absent",
"normalize_atc_candidate",
"normalize_unit_token",
"validate_unit_tokens",
"SectionDef",
"SECTION_DEFS",
"match_section",
"is_part_divider",
"normalize_heading_text",
]
+485
View File
@@ -0,0 +1,485 @@
"""Assembles a raw span stream into ordered Monograph records.
Three simple passes, each independently easy to reason about — avoids a
single tangled state machine (SRP: classify, then merge titles, then build):
1. Classify each span in reading order as a title candidate, a section
heading, or body text.
2. Coalesce consecutive title-candidate spans into single merged Heading
events via `merge.merge_multiline_headings` (handles both the multi-line
wrap and same-line font-size-split cases — see merge.py).
3. Walk the resulting flat event stream once, building Monograph records.
Handles the confirmed real "qualifier line" case (outlier-catalog item 18):
a monograph title can legitimately repeat (e.g. two distinct "SALBUTAMOL"
monographs, "Dùng trong hô hấp" vs "Dùng trong sản khoa") disambiguated by a
bold, parenthesized, non-all-caps line directly beneath the title. That
qualifier is folded into `drug_id` so two legitimate entries don't collide;
a genuine duplicate `drug_id` (no qualifier, same name) raises rather than
silently overwriting, since the one apparent duplicate found during ADR
0003's investigation (GONADOTROPIN) turned out to be a detector artifact,
not real — a real second collision should be surfaced, not hidden.
"""
from __future__ import annotations
import re
import unicodedata
from dataclasses import dataclass
from typing import Iterator, List, Optional, Union
from ..extract.models import Span
from ..extract.page_map import HEADER_BAND_Y
from ..normalize import join_spans, substitute_pua
from ..tables.classify import QUARANTINE_SHAPES
from .atc import extract_atc_codes
from .detector import in_monograph_range, is_monograph_title_candidate
from .merge import merge_multiline_headings, merge_same_line_bold_fragments
from .models import (
PART_PROSE,
PART_TABLE,
Heading,
Monograph,
SectionPart,
SectionSpan,
TableBlock,
)
from .vocab import (
SectionDef,
is_part_divider,
match_section,
match_section_with_inline_value,
)
_QUALIFIER_RE = re.compile(r"^\(.+\)$")
def _is_page_boilerplate(span: Span) -> bool:
"""Confirmed real (outlier-catalog item 13, measured via a whole-book
`assemble()` run): the running header ("DTQGVN 2" + page number +
current monograph name, e.g. physical page 1008's "DTQGVN 2" / "1009" /
"Morphin sulfat") was falling through every classification branch below
into plain body text, since it matches no section heading and isn't a
real all-caps title — silently splicing itself into the *middle* of
whatever section happens to be open when a physical page turns (1,374
of 11,409 sections / 671 of 682 monographs affected). It's reliably
identifiable independent of its (non-vocabulary) text: always the
full-page-width block in the header band, same signal `page_map.py`
already uses to read the folio.
"""
return span.column == "full_width" and span.y0 < HEADER_BAND_Y
class DuplicateDrugIdError(ValueError):
pass
@dataclass(frozen=True)
class _SectionEvent:
section_def: SectionDef
span: Span
inline_value: Optional[str] = None
@dataclass(frozen=True)
class _TextEvent:
span: Span
_Event = Union[Heading, _SectionEvent, _TextEvent] # Heading == a title event
def _slugify(text: str) -> str:
normalized = unicodedata.normalize("NFKD", text)
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
return re.sub(r"[^a-z0-9]+", "_", ascii_text.lower()).strip("_")
def _is_body_line_that_reads_like_a_label(span: Span, items: List) -> bool:
"""A plain line that repeats a section name, sitting under a heading.
Confirmed real and clinically material: FLUOROURACIL (physical page 681)
prints `Thời kỳ mang thai` / `Chống chỉ định.` and `Thời kỳ cho con bú` /
`Chống chỉ định.`, verified by rendering the page. The body line matches
the section vocabulary, so it was read as a heading — leaving both
pregnancy and lactation sections empty and dropping the statement that
fluorouracil is contraindicated in both.
The book never prints an empty section, so a *non-bold* label immediately
after a heading is that heading's body. Boldness still cannot be required
in general (outlier item 20: `Mã ATC: N06AA09.` is a plain span), which is
why this is narrowed to the directly-under-a-heading position.
"""
if span.bold:
return False
return bool(items) and isinstance(items[-1], _SectionEvent)
def _classify(spans: List[Span]) -> List[Union[Span, _SectionEvent, _TextEvent]]:
"""Pass 1: tag each span. Title candidates are left as raw Span objects
(pass 2 groups + merges them); everything else becomes a typed event.
Section matching does NOT require `span.bold` — confirmed real (outlier
item 20): AMITRIPTYLIN's "Mã ATC: N06AA09." is a single **plain, non-bold**
span (Abacavir's equivalent is bold "Mã ATC: " + a separate plain value
span), inconsistent across the book's ~700 individually-authored
monographs (the book's own foreword notes "biên soạn bởi nhiều tác giả").
Matching by exact vocabulary text (not styling) is the reliable signal,
same lesson as "don't gate on font size" (ADR 0003 item 10) applied to
boldness instead.
"""
items: List[Union[Span, _SectionEvent, _TextEvent]] = []
for span in spans:
if not span.text.strip():
continue
if _is_page_boilerplate(span):
continue
if is_monograph_title_candidate(span):
items.append(span)
continue
section_def = match_section(span.text)
if section_def is not None and not _is_body_line_that_reads_like_a_label(
span, items
):
items.append(_SectionEvent(section_def, span))
continue
if section_def is not None:
items.append(_TextEvent(span))
continue
inline = match_section_with_inline_value(span.text)
if inline is not None:
items.append(_SectionEvent(inline[0], span, inline_value=inline[1]))
else:
items.append(_TextEvent(span))
return items
def _coalesce_titles(items: List[Union[Span, _SectionEvent, _TextEvent]]) -> List[_Event]:
"""Pass 2: merge consecutive raw title-candidate Span runs into single
Heading events, preserving the order of everything else.
"""
events: List[_Event] = []
run: List[Span] = []
def flush_run():
if run:
events.extend(merge_multiline_headings(list(run)))
run.clear()
for item in items:
if isinstance(item, Span):
run.append(item)
else:
flush_run()
events.append(item)
flush_run()
return events
def _is_qualifier_line(span: Span) -> bool:
text = span.text.strip()
return span.bold and not text.isupper() and bool(_QUALIFIER_RE.match(text))
_ANCHOR_LOOKAHEAD = 6
_ANCHOR_SECTION_KEY = "ten_chung_quoc_te"
def _has_anchor_ahead(events: List[_Event], title_index: int) -> bool:
"""Every real monograph documents "Tên chung quốc tế" as its very first
section (the book's own template, item 2 — see vocab.py docstring).
Loosening this to "any known section" was tried and reverted: it let
a real, different false positive through (outlier item 21) — individual
statin names ("SIMVASTATIN", "LOVASTATIN", ...) are bold+all-caps+short
sub-headings *inside* the class-level "CÁC CHẤT ỨC CHẾ HMG-CoA
REDUCTASE" monograph, each immediately followed by their own "Liều
lượng và cách dùng" sub-section but NOT by "Tên chung quốc tế" (that
section belongs only to the parent class monograph) — the loose
"any section" check couldn't tell this apart from a real monograph
start, but the strict "Tên chung quốc tế specifically" check correctly
rejects it, since the specific book-documented template guarantees this
exact section is always first for genuine top-level monographs.
Still correctly rejects the other confirmed false positive (outlier
item 19: "HSV"/"CMV" table column headers), which aren't followed by
ANY recognized section, let alone this specific one.
"""
for j in range(title_index + 1, min(title_index + 1 + _ANCHOR_LOOKAHEAD, len(events))):
event = events[j]
if isinstance(event, Heading) and event.is_monograph_title:
return False
if isinstance(event, _SectionEvent) and event.section_def.key == _ANCHOR_SECTION_KEY:
return True
return False
def _filter_false_positive_titles(events: List[_Event]) -> List[_Event]:
"""Pass 2.5: drop title-shaped candidates that aren't followed by any
recognized section anchor before the next title candidate.
"""
return [
event for i, event in enumerate(events)
if not (isinstance(event, Heading) and event.is_monograph_title)
or _has_anchor_ahead(events, i)
]
def _region_for(table_index, span: Span):
"""The table region a span sits in, if any."""
if not table_index:
return None
for region in table_index.get(span.physical_page, ()):
if region.contains(span.x0, span.y0, span.x1, span.y1):
return region
return None
SPAN_STATE_TEXT = "normalized_text"
SPAN_STATE_TABLE = "table"
SPAN_STATE_QUARANTINED = "quarantined"
SPAN_STATE_BOILERPLATE = "boilerplate_excluded"
SPAN_STATE_HEADING = "heading"
SPAN_STATE_OUT_OF_SCOPE = "out_of_scope"
SPAN_STATE_UNASSIGNED = "unassigned"
# Deliberately dropped, not missed: the book's own part-divider titles
# ("CÁC CHUYÊN LUẬN THUỐC" etc.) are structure, not content. Reporting them
# as `unassigned` would make a clean acceptance target of unassigned == 0
# impossible to state honestly.
SPAN_STATE_STRUCTURAL = "structural_excluded"
def assemble(spans: List[Span], table_index=None, ledger: Optional[list] = None) -> Iterator[Monograph]:
"""Assemble monographs from spans.
`table_index` maps a physical page to the table regions on it (see
`tables.index_by_page`). When supplied, spans falling inside a region are
diverted into `Monograph.tables` instead of section prose — measured
reason: physical page 109's dosage-form table was otherwise concatenated
cell by cell into a section body. Omitting it keeps the previous
behaviour, so callers without a region map still work.
"""
raw_chars = sum(len(s.text) for s in spans)
spans = merge_same_line_bold_fragments(spans)
events = _filter_false_positive_titles(_coalesce_titles(_classify(spans)))
# Span-level coverage ledger. Character counts alone cannot balance here
# (normalization joins, substitutes and drops characters), so every span
# is given a state first and characters are aggregated from that.
states: dict = {}
if ledger is not None:
for s in spans:
if not s.text.strip():
states[id(s)] = "whitespace_only"
elif _is_page_boilerplate(s):
states[id(s)] = SPAN_STATE_BOILERPLATE
elif is_part_divider(s.text):
states[id(s)] = SPAN_STATE_STRUCTURAL
elif not in_monograph_range(s):
states[id(s)] = SPAN_STATE_OUT_OF_SCOPE
elif is_monograph_title_candidate(s):
# title spans are merged into a Heading event and lose their
# link back to the source span, so they are accounted for here
# using the same predicate the classifier uses
states[id(s)] = SPAN_STATE_HEADING
else:
states[id(s)] = SPAN_STATE_UNASSIGNED
def mark(span: Span, state: str):
if ledger is not None:
states[id(span)] = state
seen_ids: set = set()
current: Optional[Monograph] = None
current_section_key: Optional[str] = None
runs: List[tuple] = [] # ordered [(region_or_None, [spans])]
inline_prefix: str = ""
awaiting_qualifier = False
def append_span(span: Span, region):
"""Keep spans in reading order, starting a new run whenever the
prose/table context changes — this is what preserves the real
prose -> table -> prose sequence inside one section."""
key = region.table_id if region is not None else None
if runs and runs[-1][0] == key:
runs[-1][1].append(span)
else:
runs.append((key, [span], region))
def build_parts() -> List[SectionPart]:
parts: List[SectionPart] = []
for entry in runs:
key, collected = entry[0], entry[1]
region = entry[2] if len(entry) > 2 else None
if not collected:
continue
text = substitute_pua(join_spans(collected))
if not text.strip():
continue
pages = [s_.physical_page for s_ in collected]
xs0 = min(s_.x0 for s_ in collected); ys0 = min(s_.y0 for s_ in collected)
xs1 = max(s_.x1 for s_ in collected); ys1 = max(s_.y1 for s_ in collected)
ids = [s_.span_id for s_ in collected]
if key is None:
parts.append(SectionPart(
kind=PART_PROSE, text=text, physical_page=min(pages),
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
))
else:
parts.append(SectionPart(
kind=PART_TABLE, text=text, physical_page=min(pages),
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
table_id=key,
# deterministic: derived from the first source span, so the
# same PDF always produces the same id. A counter suffix
# would merely hide a duplicate rather than identify it.
table_part_id=f"{key}@{ids[0]}",
continuation_group=key,
shape=region.shape if region is not None else None,
quarantined=(region.shape in QUARANTINE_SHAPES)
if region is not None else False,
))
if inline_prefix:
head = SectionPart(
kind=PART_PROSE, text=inline_prefix,
physical_page=parts[0].physical_page if parts else 0,
bbox=parts[0].bbox if parts else [0.0, 0.0, 0.0, 0.0],
)
parts.insert(0, head)
return parts
def close_current_section():
nonlocal runs, inline_prefix
if current is not None and current_section_key is None and runs:
# spans seen after the title but before any section heading
current.preamble.extend(build_parts())
if current is not None and current_section_key is not None:
existing = current.sections[current_section_key]
addition = build_parts()
# A section heading can legitimately appear twice inside one
# monograph (measured: 33 monographs, 38 occurrences — e.g.
# CEFAMANDOL's "Liều lượng và cách dùng" resumes on physical page
# 339 after a renal-dosing table). Replacing the SectionSpan here
# silently destroyed everything captured before the repeat, so
# the parts are concatenated instead. The first heading stays the
# section's provenance anchor.
combined = list(existing.parts) + addition
current.sections[current_section_key] = SectionSpan(
key=existing.key, display_name=existing.display_name,
heading=existing.heading,
# `text` is prose only. Table parts stay in `parts` with their
# own provenance and quarantine flag, so anything reading
# `.text` (the chunker included) cannot pick up linearised
# cells by accident — the ordering is preserved in `parts`.
text="\n".join(
p_.text for p_ in combined
if p_.kind == PART_PROSE and not p_.quarantined and p_.text
).strip(),
parts=combined,
)
for part in addition:
if part.kind == PART_TABLE:
current.tables.append(TableBlock(
table_id=part.table_id, shape=part.shape or "",
physical_page=part.physical_page, bbox=list(part.bbox),
section_key=current_section_key, text=part.text,
quarantined=part.quarantined,
table_part_id=part.table_part_id,
continuation_group=part.continuation_group,
source_span_ids=list(part.source_span_ids),
))
runs = []
inline_prefix = ""
def finalize(monograph: Monograph) -> Monograph:
# Duplicate check happens here, not at title-detection time: the
# qualifier line (if any) is only known a few events later, so
# checking at open-time would false-positive on the legitimate
# SALBUTAMOL case (outlier item 18) before the qualifier resolves.
if monograph.drug_id in seen_ids:
raise DuplicateDrugIdError(
f"duplicate drug_id '{monograph.drug_id}' (title '{monograph.drug_name}', "
f"physical page {monograph.source_page_range[0]}) — check for a qualifier "
f"line (outlier item 18) before assuming this is a real collision"
)
seen_ids.add(monograph.drug_id)
if "ma_atc" in monograph.sections:
result = extract_atc_codes(monograph.sections["ma_atc"].text)
monograph.atc_codes = result.codes
monograph.atc_stated_absent = result.stated_absent
return monograph
for event in events:
if isinstance(event, Heading) and event.is_monograph_title:
close_current_section()
if current is not None:
yield finalize(current)
current = Monograph(
drug_id=_slugify(event.text), drug_name=event.text,
source_page_range=[event.physical_page, event.physical_page],
)
current_section_key = None
awaiting_qualifier = True
for src in getattr(event, "source_spans", ()) or ():
mark(src, SPAN_STATE_HEADING)
continue
if current is None:
continue # front matter / general chapters before the first monograph
if isinstance(event, _SectionEvent):
close_current_section()
current_section_key = event.section_def.key
if event.section_def.key not in current.sections:
current.sections[event.section_def.key] = SectionSpan(
key=event.section_def.key,
display_name=event.section_def.display_name,
heading=Heading(
text=event.section_def.display_name,
physical_page=event.span.physical_page, y0=event.span.y0,
is_monograph_title=False, section_key=event.section_def.key,
),
text="",
)
if event.inline_value:
inline_prefix = event.inline_value
awaiting_qualifier = False
mark(event.span, SPAN_STATE_HEADING)
current.source_page_range[1] = max(current.source_page_range[1], event.span.physical_page)
continue
# _TextEvent
span = event.span
if awaiting_qualifier and _is_qualifier_line(span):
text = span.text.strip()
current.drug_id = f"{current.drug_id}_{_slugify(text)}"
current.drug_name = f"{current.drug_name} {text}"
awaiting_qualifier = False
mark(span, SPAN_STATE_HEADING)
continue
awaiting_qualifier = False
if not in_monograph_range(span):
continue
current.source_page_range[1] = max(current.source_page_range[1], span.physical_page)
region = _region_for(table_index, span)
append_span(span, region)
if region is not None:
mark(span, SPAN_STATE_QUARANTINED
if region.shape in QUARANTINE_SHAPES else SPAN_STATE_TABLE)
else:
mark(span, SPAN_STATE_TEXT)
if ledger is not None:
ledger.append({"raw_chars_before_merge": raw_chars})
for s_obj in spans:
ledger.append({
"state": states[id(s_obj)],
"physical_page": s_obj.physical_page,
"bbox": [s_obj.x0, s_obj.y0, s_obj.x1, s_obj.y1],
"chars": len(s_obj.text),
"text": s_obj.text[:60],
})
close_current_section()
if current is not None:
yield finalize(current)
+135
View File
@@ -0,0 +1,135 @@
"""ATC-code extraction and normalization.
Confirmed real text-extraction noise (outlier-catalog item 12c), found while
investigating why 22/680 monographs appeared to have zero ATC codes — two
distinct causes, both extraction noise rather than missing content:
- **Stray internal whitespace** splitting one code into two tokens, e.g.
"L01X X02" (should be "L01XX02"), "J04A C01" (should be "J04AC01").
- **Digit/letter confusion**: a literal "0" rendered/typeset as "O", e.g.
"NO3AX12" (should be "N03AX12").
A third, genuinely different outcome: the source text explicitly states
"Mã ATC: Chưa có." / "Không có." — a valid "no ATC assigned yet" data state,
not an error, and must never be conflated with a parse failure.
Two more real defects found via a real whole-book `assemble()` run (not
assumed, measured against actual monograph text):
- **Trailing sentence punctuation**: "Mã ATC: J05AF06." — Abacavir's real
field text ends the sentence with a period that isn't part of the code;
an earlier version without this fix silently produced zero codes for
every single-code monograph ending in ".".
- **Per-code parenthetical annotations in multi-ATC monographs**: INSULIN's
real field lists all 20 codes each with a species/type note, e.g. "A10AB01
(người); A10AB02 (bò); A10AB03 (lợn); ..." — without stripping the
trailing "(...)" before length-checking, only 2 of 20 codes survived (the
two that happened to have a line-wrap fall between the code and its
parenthetical, accidentally isolating the bare code) — a striking example
of why this needs whole-corpus validation, not a single clean example.
A fourth defect, found via the same method (12 vaccine monographs -
VẮC XIN SỞI among them - appeared zero-ATC-and-not-stated-absent): the
segment split ran *before* parenthetical annotations were stripped, so an
annotation containing its own comma broke the split, e.g. "Mã ATC: J07BD01
(Measles, live attenuated)." split on "," into "J07BD01 (Measles" and
" live attenuated)." — neither a recoverable code shape. INSULIN's
Vietnamese annotations ("người", "", "lợn") never contain a comma, so
this only surfaced with vaccines' English annotations. Fixed by stripping
*all* parenthetical groups from the whole field text before splitting,
not just a trailing one per already-split segment.
A fifth defect, same method (15 more monographs, e.g. ALCURONIUM CLORID,
AMLODIPIN): some real monographs render the bold section label as "Mã ATC"
with no colon, and the colon belongs to the *value* span instead, e.g.
bold "Mã ATC" + plain ": M03AA01." (Abacavir's equivalent is bold "Mã ATC:
" + plain "J05AF06.", colon on the label side). The heading still matches
correctly (`vocab.normalize_heading_text` already strips a trailing
colon from either side), but the captured field text keeps the leading
": " from the value span, making the stripped candidate 8 characters
(":M03AA01") instead of 7 — silently failing the length check. Fixed by
taking only the text after the last ":" per segment before normalizing —
a strict generalization of the leading-colon strip (see the sixth defect
below) that still normalizes a plain "N03AX12" unchanged (no colon to
split on).
A sixth defect, same method (7 monographs with multiple salt/ester forms,
e.g. ARGININ, ENALAPRIL, VASOPRESSIN, the INTERFERON and gonadotropin
entries): each form is its own "Name: CODE" line rather than a bare code,
e.g. "Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01" — the whole
segment (including the name) was compared against the 7-character code
shape and rejected. Solved by trying the text after the last colon first:
"Arginin glutamat: A05BA01" -> "A05BA01".
A seventh defect, same method (1 monograph, the class-level "CÁC CHẤT ỨC
CHẾ HMG-CoA REDUCTASE"): its "Mã ATC" field lists every statin the
*opposite* way round, code first — "C10A A01: Simvastatin\nC10A A02:
Lovastatin\n..." — so "take the text after the colon" extracts the drug
name, not the code. Since a real drug name essentially never happens to
match the strict 7-character ATC shape, trying the after-colon part first
and falling back to the before-colon part costs nothing for the "Name:
CODE" case (defect six) while recovering this reversed "CODE: Name" case
too, without needing to special-case either monograph.
This module returns all three ATC-presence outcomes distinctly (found /
recovered-from-noise / stated-absent), never collapsed into one boolean,
per the outlier catalog's explicit guidance.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from typing import List, Optional
_ATC_PATTERN = re.compile(r"^[A-Z]\d{2}[A-Z]{2}\d{2}$")
_DIGIT_POSITIONS = (1, 2, 5, 6) # 0-indexed positions that must be digits
_ABSENT_MARKERS = ("chưa có", "không có")
_SEGMENT_SPLIT_RE = re.compile(r"[,;\n]")
_PAREN_RE = re.compile(r"\([^()]*\)")
@dataclass(frozen=True)
class ATCResult:
codes: List[str] = field(default_factory=list)
stated_absent: bool = False
def is_stated_absent(field_text: str) -> bool:
normalized = field_text.strip().lower()
return any(marker in normalized for marker in _ABSENT_MARKERS)
def normalize_atc_candidate(raw: str) -> Optional[str]:
"""Tries each side of the last ":" (whole string if there is none) as
the code, after-colon first since "Name: CODE" is the far more common
real shape ("CODE: Name" is confirmed real too, but rare) — returns the
first side that normalizes to a valid ATC shape. Normalizing strips
trailing sentence punctuation and internal whitespace (fixes the
split-token case), then fixes O/0 confusion only at the code's known
digit positions (never touches the letter positions, so a genuine "X" in
"L01XX02" is left alone). Parenthetical annotations must already be
stripped by the caller — see `extract_atc_codes`.
"""
parts = raw.rsplit(":", 1)
candidates = [parts[-1]] if len(parts) == 1 else [parts[1], parts[0]]
for part in candidates:
stripped = re.sub(r"\s+", "", part.upper()).rstrip(".,;")
if len(stripped) != 7:
continue
chars = list(stripped)
for i in _DIGIT_POSITIONS:
if chars[i] == "O":
chars[i] = "0"
candidate = "".join(chars)
if _ATC_PATTERN.match(candidate):
return candidate
return None
def extract_atc_codes(field_text: str) -> ATCResult:
if is_stated_absent(field_text):
return ATCResult(codes=[], stated_absent=True)
without_annotations = _PAREN_RE.sub("", field_text)
codes = []
for segment in _SEGMENT_SPLIT_RE.split(without_annotations):
candidate = normalize_atc_candidate(segment)
if candidate:
codes.append(candidate)
return ATCResult(codes=codes, stated_absent=False)
+89
View File
@@ -0,0 +1,89 @@
"""Monograph and section boundary detection.
Validated signal (ADR 0003): monograph titles are bold + all-caps + short
line length, scoped to printed pages 99-1496 — font **size** is explicitly
NOT part of the rule (a size>=9.8 threshold silently dropped ~15% of real
monographs). Section headings are bold spans cross-checked against the
known (open/extensible) vocabulary in `vocab.py`, no all-caps requirement
(most section headings, e.g. "Chỉ định", are not all-caps).
Known false positive, explicitly excluded rather than tuned around (outlier
item 12d): "CÁC CHUYÊN LUẬN THUỐC" and other part-divider titles sit exactly
at the printed-page-99 boundary and are bold + all-caps + short, identical
in shape to a real monograph title.
"All-caps" itself is not 100% reliable either (confirmed real, outlier item
21): the class-level monograph "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE" embeds
the mixed-case abbreviation "CoA" (Coenzyme A) — a strict `text.isupper()`
check silently dropped this entire monograph. `_is_mostly_upper` tolerates
a small number of lowercase letters (a strict superset of `isupper()`, so
no previously-valid case is excluded) rather than requiring zero.
"""
from __future__ import annotations
from typing import Iterator, List
from ..extract.models import Span
from .merge import merge_multiline_headings
from .models import Heading
from .vocab import is_part_divider, match_section
MONOGRAPH_PRINTED_PAGE_START = 99
MONOGRAPH_PRINTED_PAGE_END = 1496
_MIN_TITLE_LEN = 3
_MAX_TITLE_LEN = 60
_MAX_LOWERCASE_RATIO = 0.10 # HMG-CoA: 1/27 = 3.7% (real title) vs "Mã ATC:": 1/5 = 20% (real
# section label, correctly rejected) — a ratio, not an absolute count, is what separates a
# long title with one embedded mixed-case abbreviation from a short label with a normal
# lowercase diacritic (found via a real regression: an earlier absolute-count version of
# this check let "Mã ATC:" through as a false title candidate).
def _is_mostly_upper(text: str) -> bool:
letters = [c for c in text if c.isalpha()]
if not letters:
return False
lowercase_ratio = sum(1 for c in letters if c.islower()) / len(letters)
return lowercase_ratio <= _MAX_LOWERCASE_RATIO
def in_monograph_range(span: Span) -> bool:
return (
span.printed_page is not None
and MONOGRAPH_PRINTED_PAGE_START <= span.printed_page <= MONOGRAPH_PRINTED_PAGE_END
)
def is_monograph_title_candidate(span: Span) -> bool:
text = span.text.strip()
if not (span.bold and _is_mostly_upper(text)):
return False
if not (_MIN_TITLE_LEN <= len(text) <= _MAX_TITLE_LEN):
return False
if not in_monograph_range(span):
return False
if is_part_divider(text):
return False
return True
def detect_monograph_titles(spans: List[Span]) -> Iterator[Heading]:
"""`spans` must be in reading order (as `extract_spans` yields them)."""
candidates = [s for s in spans if is_monograph_title_candidate(s)]
yield from merge_multiline_headings(candidates)
def detect_section_headings(spans: List[Span]) -> Iterator[Heading]:
for span in spans:
if not span.bold or not in_monograph_range(span):
continue
section_def = match_section(span.text)
if section_def is None:
continue
yield Heading(
text=section_def.display_name,
physical_page=span.physical_page,
y0=span.y0,
is_monograph_title=False,
section_key=section_def.key,
)
+133
View File
@@ -0,0 +1,133 @@
"""Pure I/O boundary for Monograph records — kept separate from detection/
assembly logic so those stay testable without disk (Clean Architecture).
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Iterable, Iterator
from .models import Heading, Monograph, SectionPart, SectionSpan, TableBlock
def _heading_to_dict(h: Heading) -> dict:
return {
"text": h.text, "physical_page": h.physical_page, "y0": h.y0,
"is_monograph_title": h.is_monograph_title, "section_key": h.section_key,
}
def _heading_from_dict(d: dict) -> Heading:
return Heading(**d)
def _monograph_to_dict(m: Monograph) -> dict:
return {
"drug_id": m.drug_id,
"drug_name": m.drug_name,
"source_page_range": m.source_page_range,
"atc_codes": m.atc_codes,
"atc_stated_absent": m.atc_stated_absent,
"sections": {
key: {
"key": s.key, "display_name": s.display_name,
"heading": _heading_to_dict(s.heading), "text": s.text,
"parts": [
{
"kind": p.kind, "text": p.text,
"physical_page": p.physical_page, "bbox": p.bbox,
"source_span_ids": p.source_span_ids,
"table_id": p.table_id, "table_part_id": p.table_part_id,
"continuation_group": p.continuation_group,
"shape": p.shape, "quarantined": p.quarantined,
}
for p in s.parts
],
}
for key, s in m.sections.items()
},
"preamble": [
{
"kind": p.kind, "text": p.text, "physical_page": p.physical_page,
"bbox": p.bbox, "source_span_ids": p.source_span_ids,
"quarantined": p.quarantined,
}
for p in m.preamble
],
"tables": [
{
"table_id": t.table_id, "shape": t.shape,
"table_part_id": t.table_part_id,
"continuation_group": t.continuation_group,
"source_span_ids": t.source_span_ids,
"physical_page": t.physical_page, "bbox": t.bbox,
"section_key": t.section_key, "text": t.text,
"quarantined": t.quarantined,
}
for t in m.tables
],
}
def _monograph_from_dict(d: dict) -> Monograph:
sections = {
key: SectionSpan(
key=s["key"], display_name=s["display_name"],
heading=_heading_from_dict(s["heading"]), text=s["text"],
parts=[
SectionPart(
kind=p["kind"], text=p["text"],
physical_page=p["physical_page"], bbox=p["bbox"],
source_span_ids=p.get("source_span_ids", []),
table_id=p.get("table_id"), table_part_id=p.get("table_part_id"),
continuation_group=p.get("continuation_group"),
shape=p.get("shape"), quarantined=p.get("quarantined", False),
)
for p in s.get("parts", [])
],
)
for key, s in d["sections"].items()
}
return Monograph(
drug_id=d["drug_id"], drug_name=d["drug_name"],
source_page_range=d["source_page_range"], sections=sections,
atc_codes=d.get("atc_codes", []), atc_stated_absent=d.get("atc_stated_absent", False),
preamble=[
SectionPart(
kind=p["kind"], text=p["text"], physical_page=p["physical_page"],
bbox=p["bbox"], source_span_ids=p.get("source_span_ids", []),
quarantined=p.get("quarantined", False),
)
for p in d.get("preamble", [])
],
tables=[
TableBlock(
table_id=t["table_id"], shape=t["shape"],
physical_page=t["physical_page"], bbox=t["bbox"],
section_key=t.get("section_key"), text=t["text"],
quarantined=t.get("quarantined", False),
table_part_id=t.get("table_part_id"),
continuation_group=t.get("continuation_group"),
source_span_ids=t.get("source_span_ids", []),
)
for t in d.get("tables", [])
],
)
def write_monographs_jsonl(monographs: Iterable[Monograph], path: Path) -> int:
count = 0
with open(path, "w", encoding="utf-8") as f:
for m in monographs:
f.write(json.dumps(_monograph_to_dict(m), ensure_ascii=False) + "\n")
count += 1
return count
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
with open(path, "r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line:
continue
yield _monograph_from_dict(json.loads(line))
+133
View File
@@ -0,0 +1,133 @@
"""Multi-line monograph-title merging, and same-line bold-run reassembly.
Two distinct real fragmentation shapes were confirmed, both requiring merge:
1. **Multi-line wrap** (ADR 0003's original finding, dominant cause of its
recall gap and of the GONADOTROPIN false-collision, outlier item 11):
long titles wrap across 2+ physical lines, e.g. physical page 1371 has
"THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG" (y0=664.46) immediately followed by
"GONADOTROPIN" (y0=676.24) — a ~11.8pt line-height step, same page.
2. **Same-line font-size split, found by visually inspecting a real page**
(physical page 113, rendered to an image and read directly — not
inferred from coordinates alone): "ACICLOVIR" is split into two spans,
"ACIC" (size 10.0) and "LOVIR" (size 9.5), touching with a ~0.5pt y0
difference and near-zero x-gap. An earlier version of this module
required exact font-size equality to merge, which correctly handled
case 1 (GONADOTROPIN: both fragments size 9.5) but silently missed case
2 — the same "font size is not reliable" lesson from ADR 0003 applies
*within* a single title's fragments, not just across different
monographs. Fixed by dropping the size-equality requirement; the y-gap
+ same-page check alone is sufficient (a real next-monograph title is
always much farther down the page/on a different page, given a full
monograph's worth of section content in between).
The join character between merged fragments must differ by case: case 1
needs a space (distinct words across a real line wrap); case 2 needs no
space (mid-word split, "ACIC" + "LOVIR" = "ACICLOVIR", not "ACIC LOVIR").
Distinguished by the y0 gap: small (<= `_SAME_LINE_Y_TOLERANCE`) means same
visual line -> concatenate directly; larger means a real new line -> join
with a space.
Candidate spans passed in here are already filtered by the caller (bold +
all-caps + short + in the monograph page range) — this module only decides
which *consecutive* candidates belong to the same title and how to join them.
A third, unrelated fragmentation shape was confirmed via a whole-book
`cli validate` run against the back-of-book index (5 real monographs -
GUAIFENESIN, MEPHENESIN, NATRI THIOSULFAT, RAMIPRIL, TENOXICAM - silently
dropped): PyMuPDF splits some bold section-heading runs into several spans
around diacritic characters even though the text is a single, visually
unbroken line in the rendered page (confirmed by rendering physical page
759 to an image and reading it directly — "Tên chung quốc tế" looks
completely normal to a human reader; the fragmentation exists only in
PyMuPDF's span boundaries, not the document). Confirmed page 759's actual
spans: "Tên chung qu" (y0=157.614), "" (y0=157.33), "c t" (y0=157.614),
"ế" (y0=157.33), ": " (y0=157.614) — all within `_SAME_LINE_Y_TOLERANCE`,
so `merge_same_line_bold_fragments` (applied to *all* bold spans, not just
title candidates, before section-vocabulary matching) reassembles them the
same way case 2 above reassembles "ACIC" + "LOVIR".
"""
from __future__ import annotations
import dataclasses
from typing import Iterator, List
from ..extract.models import Span
from .models import Heading
_MAX_LINE_GAP_PT = 20.0 # comfortably above the confirmed ~11.8pt wrap case
_SAME_LINE_Y_TOLERANCE = 3.0 # comfortably above the confirmed ~0.5pt same-line split
def merge_same_line_bold_fragments(spans: List[Span]) -> List[Span]:
"""Reassemble consecutive bold spans that PyMuPDF split mid-line (same
page, same visual line) back into one span, so downstream section-vocab
matching sees the real text instead of a diacritic-boundary fragment.
Non-bold spans and spans on different lines pass through unchanged.
Provenance (page/column/block/line/span_index/y-position/font/size) is
kept from the first fragment; only `text` and `x1` are updated, so the
merged span still traces back to its exact source region.
"""
merged: List[Span] = []
buffer: List[Span] = []
def flush():
if not buffer:
return
if len(buffer) == 1:
merged.append(buffer[0])
else:
merged.append(dataclasses.replace(
buffer[0], text="".join(s.text for s in buffer), x1=buffer[-1].x1,
))
for span in spans:
same_line_bold_run = (
buffer and span.bold and buffer[-1].bold
and span.physical_page == buffer[-1].physical_page
and abs(span.y0 - buffer[-1].y0) <= _SAME_LINE_Y_TOLERANCE
)
if same_line_bold_run:
buffer.append(span)
else:
flush()
buffer = [span]
flush()
return merged
def _same_title_run(prev: Span, curr: Span) -> bool:
gap = curr.y0 - prev.y0
return curr.physical_page == prev.physical_page and 0 <= gap <= _MAX_LINE_GAP_PT
def merge_multiline_headings(candidates: List[Span]) -> Iterator[Heading]:
"""`candidates` must already be in reading order (as extract_spans
yields them) and pre-filtered to heading candidates only.
"""
buffer: List[Span] = []
for span in candidates:
if buffer and _same_title_run(buffer[-1], span):
buffer.append(span)
else:
if buffer:
yield _flush(buffer)
buffer = [span]
if buffer:
yield _flush(buffer)
def _flush(buffer: List[Span]) -> Heading:
parts = [buffer[0].text.strip()]
for prev, curr in zip(buffer, buffer[1:], strict=False):
same_line = abs(curr.y0 - prev.y0) <= _SAME_LINE_Y_TOLERANCE
parts.append("" if same_line else " ")
parts.append(curr.text.strip())
first = buffer[0]
return Heading(
text="".join(parts),
physical_page=first.physical_page,
y0=first.y0,
is_monograph_title=True,
)
+95
View File
@@ -0,0 +1,95 @@
"""Data model for segmented output, matching docs/architecture.md's contract:
{drug_id, drug_name, source_page_range, sections: {...}}
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Dict, List, Optional
@dataclass(frozen=True)
class Heading:
text: str
physical_page: int
y0: float
is_monograph_title: bool
section_key: Optional[str] = None
PART_PROSE = "prose"
PART_TABLE = "table"
@dataclass(frozen=True)
class SectionPart:
"""One contiguous run of a section, in reading order.
A section is not uniformly prose: a dosing section routinely reads
prose -> table -> prose. Flattening that to a single string loses both the
ordering and the ability to say which part a sentence came from, so the
parts are kept in sequence with their own provenance.
"""
kind: str
text: str
physical_page: int
bbox: List[float]
source_span_ids: List[str] = field(default_factory=list)
table_id: Optional[str] = None
table_part_id: Optional[str] = None
continuation_group: Optional[str] = None
shape: Optional[str] = None
quarantined: bool = False
@dataclass(frozen=True)
class SectionSpan:
key: str
display_name: str
heading: Heading
text: str
parts: List[SectionPart] = field(default_factory=list)
@property
def prose_text(self) -> str:
"""Only the parts safe to read as prose — excludes quarantined ones."""
return "\n".join(
p.text for p in self.parts
if p.kind == PART_PROSE and not p.quarantined and p.text
).strip()
@dataclass(frozen=True)
class TableBlock:
"""Text lifted out of a table region, kept beside the prose instead of
inside it.
`quarantined` marks content whose flattened text is actively misleading
(a 2D lookup grid means nothing without its row and column headers) —
such a block must not be embedded or cited as if it were prose.
"""
table_id: str
shape: str
physical_page: int
bbox: List[float]
section_key: Optional[str]
text: str
quarantined: bool = False
table_part_id: Optional[str] = None
continuation_group: Optional[str] = None
source_span_ids: List[str] = field(default_factory=list)
@dataclass
class Monograph:
drug_id: str
drug_name: str
source_page_range: List[int]
sections: Dict[str, SectionSpan] = field(default_factory=dict)
atc_codes: List[str] = field(default_factory=list)
atc_stated_absent: bool = False
tables: List[TableBlock] = field(default_factory=list)
# Text between the monograph title and its first section heading. Real and
# clinically important — e.g. ARTEMETHER (physical page 210) opens with the
# regulatory notice that single-agent artemisinin products were withdrawn
# to limit resistance. It belongs to no section, so it was being dropped.
preamble: List[SectionPart] = field(default_factory=list)
+44
View File
@@ -0,0 +1,44 @@
"""Dosing-unit token validation (mg/mcg/mmol/g/ml).
Unlike `atc.py`'s whitespace-split defect (confirmed with real examples,
outlier-catalog item 12c), a targeted regex scan of the full monograph page
range (99-1496 printed) for the analogous unit-token pattern (a unit like
"mg" split into "m g" by a stray internal space) found **zero occurrences**
— this is NOT a confirmed defect in this corpus. This module exists as a
defensive check by analogy, per the project's explicit dosing-safety
requirement: a silent mg/mcg confusion is a 1000x dosing error, and the
book's own "Người lớn"/"Trẻ em" dosing-population split appears on the
majority of monograph pages (outlier-catalog item 17), so the cost of an
undetected unit-token corruption is high enough to check for even without a
confirmed prior occurrence — but callers must not describe what this module
guards against as "a confirmed real defect," only as a validated absence
plus a standing defensive gate.
"""
from __future__ import annotations
import re
from typing import Optional
_KNOWN_UNITS = ("mg", "mcg", "mmol", "microgam", "g", "ml", "iu", "đvqt")
_UNIT_PATTERN = re.compile(
"^(" + "|".join(re.escape(u) for u in _KNOWN_UNITS) + ")$", re.IGNORECASE
)
def normalize_unit_token(raw: str) -> Optional[str]:
"""Strips internal whitespace (defends against a stray-space split, the
same failure class as the confirmed ATC whitespace-split defect) and
validates against the known dosing-unit vocabulary. Returns the
lowercase canonical unit string, or None if unrecognized.
"""
stripped = re.sub(r"\s+", "", raw).lower()
return stripped if _UNIT_PATTERN.match(stripped) else None
def validate_unit_tokens(tokens: list) -> "list[str]":
"""Returns the subset of `tokens` that fail normalization — callers use
this to flag a dosing section for manual review, not to silently drop
or auto-correct (unlike ATC codes, there is no confirmed-safe recovery
rule here since no real corruption pattern has been observed yet).
"""
return [t for t in tokens if normalize_unit_token(t) is None]
+157
View File
@@ -0,0 +1,157 @@
"""Section-name taxonomy for drug monographs.
Canonical list transcribed directly from the book's own documented template
(physical page 38, printed page 39, "HƯỚNG DẪN SỬ DỤNG DƯỢC THƯ QUỐC GIA
VIỆT NAM") and cross-checked against real bold headings in the Abacavir/
Acarbose monographs (physical pages 100-102). The book documents 19 fields
per monograph, of which #1 ("Tên chuyên luận thuốc") is the monograph title
itself (handled by `detector.detect_monograph_titles`, not a section) —
leaving 18 documented sections. `ten_thuong_mai` ("Tên thương mại") is a
19th, real, but *undocumented* field confirmed present in real monographs
(outlier-catalog item 12) — open/closed taxonomy: add new entries here as
they're found, never change the matching logic in detector.py.
"""
from __future__ import annotations
import re
import unicodedata
from dataclasses import dataclass
from typing import Dict, Optional, Tuple
@dataclass(frozen=True)
class SectionDef:
key: str
display_name: str
# Real spelling variants observed in the book itself. The source is not
# typographically consistent: it prints "qui chế" 469 times against the
# documented "quy chế", and carries assorted typos ("Mã ACT", "sử trí",
# "Chống chỉ đinh"). Whitespace and look-alike-character differences are
# NOT listed here — `_lookup_key` folds those away for every entry at
# once, so this stays a list of genuinely different wordings.
aliases: Tuple[str, ...] = ()
@property
def labels(self) -> Tuple[str, ...]:
return (self.display_name,) + self.aliases
SECTION_DEFS = [
SectionDef("ten_chung_quoc_te", "Tên chung quốc tế", ("Ten chung quốc tế",)),
SectionDef("ma_atc", "Mã ATC", ("Mã ACT",)),
SectionDef("loai_thuoc", "Loại thuốc", ("Loại thuôc", "Lọai thuốc", "Phân loại thuốc")),
SectionDef("dang_thuoc_va_ham_luong", "Dạng thuốc và hàm lượng",
("Dạng dùng và hàm lượng",)),
SectionDef("duoc_ly_va_co_che_tac_dung", "Dược lý và cơ chế tác dụng",
("Dược lí và cơ chế tác dụng", "Dược lý học và cơ chế tác dụng")),
SectionDef("chi_dinh", "Chỉ định"),
SectionDef("chong_chi_dinh", "Chống chỉ định", ("Chống chỉ đinh",)),
SectionDef("than_trong", "Thận trọng"),
SectionDef("thoi_ky_mang_thai", "Thời kỳ mang thai", ("Thời kì mang thai",)),
SectionDef("thoi_ky_cho_con_bu", "Thời kỳ cho con bú", ("Thời kì cho con bú",)),
SectionDef("tac_dung_khong_mong_muon", "Tác dụng không mong muốn (ADR)",
("Tác dụng không mong muốn", "Tác dụng không mong muốn ADR")),
SectionDef("huong_dan_xu_tri_adr", "Hướng dẫn cách xử trí ADR",
("Hướng dẫn xử trí ADR", "Hướng dẫn cách sử trí ADR",
"Hướng dẫn cách xử trí các ADR")),
SectionDef("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
("Liều lượng cách dùng", "Liều lượng, cách dùng",
"Liều dùng và cách dùng", "Liều lượng và cách sử dụng")),
SectionDef("tuong_tac_thuoc", "Tương tác thuốc"),
SectionDef("do_on_dinh_va_bao_quan", "Độ ổn định và bảo quản"),
SectionDef("tuong_ky", "Tương kỵ"),
SectionDef("qua_lieu_va_xu_tri", "Quá liều và xử trí",
("Quá liều và cách xử trí", "Quá liều và xử lý",
"Quá liều cấp tính và xử trí")),
SectionDef("thong_tin_quy_che", "Thông tin quy chế",
("Thông tin qui chế", "Thông tin về qui chế", "Thông tin và quy chế")),
SectionDef("ten_thuong_mai", "Tên thương mại"),
]
# Near-miss strings deliberately NOT treated as section headings, recorded so
# a later reader does not "helpfully" add them: "Thể trọng" is body weight,
# not "Thận trọng" (caution); "Tác dụng không mong muốn của opioid" is a
# drug-specific sub-heading inside a section, not the section itself.
REJECTED_NEAR_MISSES = frozenset({"Thể trọng", "Tác dụng không mong muốn của opioid"})
# Part/section-divider titles (from the book's own table of contents) that
# are bold + all-caps + short, exactly like a monograph title, but are NOT
# drug monographs — confirmed false positive, outlier-catalog item 12d.
PART_DIVIDER_TITLES = {
"CÁC CHUYÊN LUẬN CHUNG",
"CÁC CHUYÊN LUẬN THUỐC",
"CÁC PHỤ LỤC",
}
_TRAILING_PUNCT_RE = re.compile(r"[:.\s]+$")
_WHITESPACE_RE = re.compile(r"\s+")
_ALL_WHITESPACE_RE = re.compile(r"\s")
# Look-alike characters the typesetting mixes with their correct forms:
# U+00D0 LATIN CAPITAL LETTER ETH is used where U+0110 LATIN CAPITAL LETTER D
# WITH STROKE belongs ("Ðộ ổn định" vs "Độ ổn định"), and NFC does not unify
# them because they are genuinely distinct codepoints that merely look alike.
_CONFUSABLES = str.maketrans({"Ð": "Đ", "ð": "đ"})
def normalize_heading_text(text: str) -> str:
"""Strip trailing colon/period/whitespace and collapse internal runs so
"Tên chung quốc tế:" and "Tên chung quốc tế" (both observed verbatim in
real monographs) render the same. Preserves single spaces — this is the
display form, not the lookup form.
"""
normalized = _TRAILING_PUNCT_RE.sub("", text.strip())
return _WHITESPACE_RE.sub(" ", normalized)
def _lookup_key(text: str) -> str:
"""Fold away the differences that are typesetting noise, not wording.
The source splits and joins headings inconsistently — "Chỉđịnh",
"H ướng dẫn cách xử trí ADR", "Tác dụng khôngmong muốn (ADR)" and
"Độổn định và bảo quản" all appear — so whitespace is removed entirely
rather than enumerated as aliases. Case and look-alike characters are
folded for the same reason.
"""
folded = unicodedata.normalize("NFC", normalize_heading_text(text))
folded = folded.translate(_CONFUSABLES)
return _ALL_WHITESPACE_RE.sub("", folded).lower()
_LOOKUP: Dict[str, SectionDef] = {
_lookup_key(label): d for d in SECTION_DEFS for label in d.labels
}
def match_section(text: str) -> Optional[SectionDef]:
return _LOOKUP.get(_lookup_key(text))
# Sorted longest-label-first so a prefix check never matches a shorter
# label that happens to also be a prefix of a longer one (none currently
# collide, but this is a cheap, permanent safety property to keep).
_PREFIX_CANDIDATES: list = sorted(
((label, d) for d in SECTION_DEFS for label in d.labels),
key=lambda pair: -len(pair[0]),
)
def match_section_with_inline_value(text: str) -> Optional[Tuple[SectionDef, str]]:
"""Handles a real, confirmed structural variant (outlier item 20):
some monographs render a section heading and its value as ONE
non-bold, non-separated span, e.g. AMITRIPTYLIN's "Mã ATC: N06AA09."
(Abacavir's equivalent is bold "Mã ATC: " + separate plain "J05AF06.").
Returns (matched section, remaining value text) or None.
"""
stripped = text.strip()
for label, section_def in _PREFIX_CANDIDATES:
if stripped[: len(label)].lower() != label.lower():
continue
remainder = stripped[len(label):].lstrip()
if remainder.startswith(":"):
return section_def, remainder[1:].strip()
return None
def is_part_divider(text: str) -> bool:
return normalize_heading_text(text).upper() in PART_DIVIDER_TITLES
+37
View File
@@ -0,0 +1,37 @@
"""Table stage: detect tabular regions so their text stops leaking into prose.
Scope note: this stage locates and classifies table *regions*. Reconstructing
correct rows and columns is deliberately not attempted here — see ADR 0003
and outlier-catalog items 5-7 for why that is a separate, harder problem.
"""
from .classify import (
QUARANTINE_SHAPES,
SHAPE_CROSS_PAGE,
SHAPE_FORMULA_2D,
SHAPE_GRID_2D,
SHAPE_MULTI_HEADER,
SHAPE_SINGLE_COLUMN_BOXED,
SHAPE_NOT_TABLE_FULL_PAGE,
SHAPE_SIMPLE,
classify_shape,
)
from .detect import detect_table_regions
from .io import index_by_page, read_regions_json, write_regions_json
from .models import TableRegion
__all__ = [
"QUARANTINE_SHAPES",
"SHAPE_CROSS_PAGE",
"SHAPE_FORMULA_2D",
"SHAPE_GRID_2D",
"SHAPE_MULTI_HEADER",
"SHAPE_SINGLE_COLUMN_BOXED",
"SHAPE_NOT_TABLE_FULL_PAGE",
"SHAPE_SIMPLE",
"TableRegion",
"classify_shape",
"detect_table_regions",
"index_by_page",
"read_regions_json",
"write_regions_json",
]
+92
View File
@@ -0,0 +1,92 @@
"""Shape classification for detected table regions.
Whole-corpus measurement: of 200 regions `pdfplumber.find_tables()` reports,
22 are not tables at all (17 cover a whole page — e.g. the copyright page —
and 5 are single-column text blocks such as the epilepsy classification
list). Routing every region through one generic reconstructor would treat
those 22 as tables, so shape is decided first and handling follows from it.
Open/closed: adding a shape means adding a rule here, not editing callers.
"""
from __future__ import annotations
from typing import List
PAGE_WIDTH, PAGE_HEIGHT = 595.3, 841.9
FULL_PAGE_AREA_RATIO = 0.75
SHAPE_SIMPLE = "simple_table"
SHAPE_MULTI_HEADER = "multi_level_or_merged_header"
SHAPE_CROSS_PAGE = "cross_page_continuation"
SHAPE_GRID_2D = "grid_2d_numeric"
SHAPE_NOT_TABLE_FULL_PAGE = "not_a_table_full_page"
SHAPE_SINGLE_COLUMN_BOXED = "single_column_boxed_list"
# Not produced by table detection — a stacked fraction is not a table — but it
# is the same kind of object as far as assembly is concerned: a rectangle whose
# spans must be lifted out of prose rather than run together. Measured on
# NETILMICIN (physical page 1042) and AMPICILIN VÀ SULBACTAM (202): linearised,
# the numerator lands before the '=' and the division reads as multiplication.
SHAPE_FORMULA_2D = "formula_2d"
# Shapes whose flattened text must not be embedded or cited as prose until a
# real row/column reconstruction exists. Any multi-column table loses its
# cell semantics when linearised — a 2D lookup grid most severely (its values
# are meaningless without both headers, outlier-catalog item 7), but a plain
# dosing table is no safer to quote once its columns are run together.
# `single_column_boxed_list` is excluded deliberately: one column linearises
# correctly, so it reads as ordinary text (physical page 55's "Bảng 2").
QUARANTINE_SHAPES = frozenset({
SHAPE_GRID_2D,
SHAPE_SIMPLE,
SHAPE_MULTI_HEADER,
SHAPE_CROSS_PAGE,
SHAPE_FORMULA_2D,
})
def _area_ratio(bbox) -> float:
x0, y0, x1, y1 = bbox
return abs((x1 - x0) * (y1 - y0)) / (PAGE_WIDTH * PAGE_HEIGHT)
def classify_shape(
bbox,
n_rows: int,
n_cols: int,
first_row: List[str],
starts_near_top: bool,
all_cells_numeric: bool,
) -> str:
if _area_ratio(bbox) >= FULL_PAGE_AREA_RATIO:
return SHAPE_NOT_TABLE_FULL_PAGE
# Only a single *column* is degenerate. A single ROW with several columns
# is the opposite of degenerate — it is the orphaned continuation row of
# a table broken across a page (outlier-catalog item 5), the case where
# losing the content is most damaging because a row without its header
# cannot be interpreted. Verified visually: physical pages 62 and 72 are
# exactly this (1x3, with cell rules visible), and an earlier version of
# this rule discarded both as "not a table".
# One column inside a ruled box. Structurally not a row/column table, but
# the book may still number it as one — physical page 55 is captioned
# "Bảng 2: Phân loại quốc tế các cơn động kinh (1989)" and is a nested
# numbered list drawn inside a frame. Named for what it is rather than
# "not a table": single-column content linearises correctly and must stay
# in the text, unlike a real 2D table.
if n_cols <= 1:
return SHAPE_SINGLE_COLUMN_BOXED
if n_rows <= 1:
return SHAPE_CROSS_PAGE
if all_cells_numeric and n_cols >= 4:
return SHAPE_GRID_2D
cells = [(c or "").strip() for c in first_row]
textual = sum(
1 for c in cells
if c and not c.replace(",", "").replace(".", "").replace("-", "").isdigit()
)
if starts_near_top and textual <= 1:
return SHAPE_CROSS_PAGE
if cells and any(not c for c in cells) and textual >= 1:
return SHAPE_MULTI_HEADER
return SHAPE_SIMPLE
+71
View File
@@ -0,0 +1,71 @@
"""Table region detection.
`pdfplumber` is used here and nowhere else in the pipeline: ADR 0003 records
that its general text extraction scrambles reading order on this document,
so it is kept strictly to table geometry, where it is the only tool that
works. PyMuPDF remains the sole text extractor.
Detection is slow (≈17 minutes over the 1668-page book), so the result is
written once to a region map and reused — see `io.py`. The detection itself
lives here, in the pipeline, rather than in a throwaway script.
"""
from __future__ import annotations
from pathlib import Path
from typing import Iterator, List
import pdfplumber
from .classify import classify_shape
from .models import TableRegion
TOP_BAND_Y = 120.0
def _all_numeric(data) -> bool:
values = [(c or "").strip() for row in data for c in row]
values = [v for v in values if v]
if not values:
return False
return all(
v.replace(",", "").replace(".", "").replace("-", "").isdigit()
for v in values
)
def detect_table_regions(pdf_path: Path) -> Iterator[TableRegion]:
with pdfplumber.open(pdf_path) as pdf:
for page_number, page in enumerate(pdf.pages):
try:
found = page.find_tables()
except Exception:
continue
for index, table in enumerate(found):
data = table.extract() or []
first_row: List[str] = [
(c or "").strip() for c in (data[0] if data else [])
]
n_rows = len(data)
n_cols = max((len(r) for r in data), default=0)
bbox = tuple(round(v, 1) for v in table.bbox)
yield TableRegion(
table_id=f"p{page_number}_t{index}",
physical_page=page_number,
bbox=bbox,
n_rows=n_rows,
n_cols=n_cols,
shape=classify_shape(
bbox=bbox,
n_rows=n_rows,
n_cols=n_cols,
first_row=first_row,
starts_near_top=bbox[1] < TOP_BAND_Y,
all_cells_numeric=_all_numeric(data),
),
first_row=first_row[:8],
)
# pdfplumber caches every parsed object per page; without this the
# 1668-page book grows the process past 6 GB and the run dies on
# a paging-file error rather than finishing.
page.flush_cache()
page.get_textmap.cache_clear()
+42
View File
@@ -0,0 +1,42 @@
"""Filesystem boundary for the table stage."""
from __future__ import annotations
import json
from dataclasses import asdict
from pathlib import Path
from typing import Dict, Iterable, List
from .models import TableRegion
def write_regions_json(regions: Iterable[TableRegion], path: Path) -> int:
path.parent.mkdir(parents=True, exist_ok=True)
rows = [asdict(r) for r in regions]
path.write_text(json.dumps(rows, ensure_ascii=False, indent=1), encoding="utf-8")
return len(rows)
def read_regions_json(path: Path) -> List[TableRegion]:
rows = json.loads(path.read_text(encoding="utf-8"))
return [
TableRegion(
table_id=r["table_id"],
physical_page=r["physical_page"],
bbox=tuple(r["bbox"]),
n_rows=r["n_rows"],
n_cols=r["n_cols"],
shape=r["shape"],
first_row=r.get("first_row", []),
)
for r in rows
]
def index_by_page(regions: Iterable[TableRegion]) -> Dict[int, List[TableRegion]]:
"""Group real table regions by page for O(1) lookup during assembly."""
index: Dict[int, List[TableRegion]] = {}
for region in regions:
if not region.is_real_table:
continue
index.setdefault(region.physical_page, []).append(region)
return index
+39
View File
@@ -0,0 +1,39 @@
"""Table region model.
A region is a rectangle on one page that holds tabular content. It is
deliberately separate from the table's *contents*: the pipeline's first
obligation is to stop tabular text leaking into prose (measured: page 109's
dosage-form table was being concatenated cell-by-cell into a section body),
which needs only the geometry. Reconstructing rows and columns correctly is
a later, harder step.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import List, Tuple
@dataclass(frozen=True)
class TableRegion:
table_id: str
physical_page: int
bbox: Tuple[float, float, float, float]
n_rows: int
n_cols: int
shape: str
first_row: List[str] = field(default_factory=list)
@property
def is_real_table(self) -> bool:
return not self.shape.startswith("not_a_table")
def contains(self, x0: float, y0: float, x1: float, y1: float) -> bool:
"""True when a span's box lies (mostly) inside this region.
Uses the span's centre rather than full containment: PyMuPDF span
boxes and pdfplumber table boxes come from different engines and
disagree by a point or two at the edges.
"""
cx, cy = (x0 + x1) / 2, (y0 + y1) / 2
left, top, right, bottom = self.bbox
return left <= cx <= right and top <= cy <= bottom
@@ -0,0 +1,51 @@
from .back_index import GroundTruthEntry, parse_back_index
from .metrics import RecallPrecisionResult, compute_recall_precision
from .readiness import (
Gate,
corpus_size,
evaluate,
evaluate_chunks,
read_chunks,
read_monographs,
)
from .residual_ink import (
ANTIALIAS_SPECK,
FRACTION_BAR_CANDIDATE,
HEADER_BAND_FRAGMENT,
HEADER_RULE,
RULE_FRAGMENT,
TABLE_FRAME,
TEXT_AS_VECTOR_OUTLINE,
UNCLASSIFIED,
PageContext,
ResidualRegion,
classify,
scan_document,
scan_page,
)
__all__ = [
"GroundTruthEntry",
"parse_back_index",
"RecallPrecisionResult",
"compute_recall_precision",
"Gate",
"evaluate",
"evaluate_chunks",
"read_chunks",
"corpus_size",
"read_monographs",
"PageContext",
"ResidualRegion",
"classify",
"scan_page",
"scan_document",
"HEADER_RULE",
"TABLE_FRAME",
"TEXT_AS_VECTOR_OUTLINE",
"FRACTION_BAR_CANDIDATE",
"HEADER_BAND_FRAGMENT",
"RULE_FRAGMENT",
"ANTIALIAS_SPECK",
"UNCLASSIFIED",
]
@@ -0,0 +1,52 @@
"""Parses the book's own "Mục lục tra cứu" (back-of-book index) into
page-verified ground truth — per ADR 0003, this is the correct validation
source (exact page numbers per generic name), not the front-matter drug list
(no page numbers).
Real format confirmed by reading physical pages 1530+ directly:
- Genuine generic-name entries: "Abacavir, 101" (name, comma, printed page).
- Brand-name cross-references: "Ziagen - Abacavir, 101" / "ABAB -
Paracetamol, 1118" (brand " - " generic, page) — skipped for ground
truth, per ADR 0003.
- Section-letter headers ("A", "B", ...) and running header/footer
boilerplate lines don't match the entry pattern and are naturally
ignored, not specially cased.
Known limitation, inherited from the already-validated ADR 0003 approach
(not newly introduced here): a handful of genuine compound-name entries in
the book use " - " *within* the generic name itself (e.g. "Carbidopa -
levodopa"), which this parser's cross-reference exclusion will also skip —
the same trade-off the original 91.7%-recall validation already made
successfully, not re-litigated here.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import List
import fitz
BACK_INDEX_START_PHYSICAL = 1530 # printed 1531 — first page of real entries ("A" section)
_ENTRY_RE = re.compile(r"^(.+?),\s*(\d+)\s*$")
_CROSS_REF_MARKER = " - "
@dataclass(frozen=True)
class GroundTruthEntry:
name: str
printed_page: int
def parse_back_index(doc: fitz.Document, start_physical_page: int = BACK_INDEX_START_PHYSICAL) -> List[GroundTruthEntry]:
entries: List[GroundTruthEntry] = []
for pno in range(start_physical_page, doc.page_count):
for line in doc[pno].get_text().split("\n"):
line = line.strip()
if not line or _CROSS_REF_MARKER in line:
continue
match = _ENTRY_RE.match(line)
if match:
entries.append(GroundTruthEntry(name=match.group(1).strip(), printed_page=int(match.group(2))))
return entries
+107
View File
@@ -0,0 +1,107 @@
"""Monograph-boundary recall/precision against the back-of-book index.
Per ADR 0003, only recall was ever measured before (91.7%, 665/725) — this
module adds precision (never measured previously) alongside recall, per the
approved eval-framework plan.
Page comparison: `GroundTruthEntry.printed_page` is a *printed* page number;
`Monograph.source_page_range` is *physical*. The physical->printed offset
was empirically confirmed constant (+1) across every tested milestone page
in Phase 1.1 (`extract/page_map.py`) — reused here rather than re-derived.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import List
from ..segment.models import Monograph
from .back_index import GroundTruthEntry
PRINTED_PAGE_OFFSET = 1
PAGE_TOLERANCE = 2
@dataclass(frozen=True)
class RecallPrecisionResult:
recall: float
precision: float
matched_count: int
total_ground_truth: int
total_detected: int
unmatched_ground_truth: List[GroundTruthEntry]
unmatched_detected: List[Monograph]
_WHITESPACE_RE = re.compile(r"\s+")
def _normalize_name(name: str) -> str:
# collapse-whitespace: confirmed real case — "ALVERIN CITRAT" (double
# space, likely a genuine PDF-rendering artifact) failed to match
# ground truth's "Alverin citrat" under plain strip+upper, found via a
# real `cli validate` run (4 of 12 unmatched-detected monographs had
# this exact shape: ALVERIN CITRAT, OXYMETAZOLIN HYDROCLORID,
# TERBUTALIN SULFAT, TIOTROPIUM BROMID).
return _WHITESPACE_RE.sub(" ", name.strip()).upper()
def _monograph_start_printed_page(monograph: Monograph) -> int:
return monograph.source_page_range[0] + PRINTED_PAGE_OFFSET
def _names_match(entry_name: str, drug_name: str) -> bool:
a, b = _normalize_name(entry_name), _normalize_name(drug_name)
return a in b or b in a
def _names_match_exactly(entry_name: str, drug_name: str) -> bool:
return _normalize_name(entry_name) == _normalize_name(drug_name)
def _find_match(entry: GroundTruthEntry, monographs: List[Monograph]):
# Exact match first, substring fallback only if no exact match exists:
# confirmed real case, "Isosorbid" and "Isosorbid dinitrat" are two
# distinct real monographs a page apart. A substring-only search finds
# "Isosorbid" for BOTH ground-truth entries (it's a substring of
# "Isosorbid dinitrat" too) and, being first in page order, wins via
# `next()` for both — leaving the real "Isosorbid dinitrat" monograph
# spuriously unmatched. Same shape confirmed for "Ampicilin" /
# "Ampicilin và sulbactam". Trying each entry's exact match across all
# monographs before falling back to substring resolves both without
# needing order-dependent tie-breaking.
in_tolerance = [
m for m in monographs
if abs(_monograph_start_printed_page(m) - entry.printed_page) <= PAGE_TOLERANCE
]
return next(
(m for m in in_tolerance if _names_match_exactly(entry.name, m.drug_name)),
next((m for m in in_tolerance if _names_match(entry.name, m.drug_name)), None),
)
def compute_recall_precision(
monographs: List[Monograph], ground_truth: List[GroundTruthEntry],
) -> RecallPrecisionResult:
matched_gt = []
unmatched_gt = []
matched_detected_ids: set = set()
for entry in ground_truth:
match = _find_match(entry, monographs)
if match is not None:
matched_gt.append(entry)
matched_detected_ids.add(match.drug_id)
else:
unmatched_gt.append(entry)
unmatched_detected = [m for m in monographs if m.drug_id not in matched_detected_ids]
return RecallPrecisionResult(
recall=len(matched_gt) / len(ground_truth) if ground_truth else 0.0,
precision=len(matched_detected_ids) / len(monographs) if monographs else 0.0,
matched_count=len(matched_gt),
total_ground_truth=len(ground_truth),
total_detected=len(monographs),
unmatched_ground_truth=unmatched_gt,
unmatched_detected=unmatched_detected,
)
+202
View File
@@ -0,0 +1,202 @@
"""Named gates that must hold before the corpus is chunked.
Chunking bakes whatever it is given into embeddings, where defects stop being
inspectable. So the question this module answers is not "did the pipeline
run" but "is the text going in actually the text on the page". Each gate is
reported on its own line with its own number and its own target — a single
pass/fail would hide exactly the problems that took a whole session to find.
Every gate here is computed from the artefacts, never remembered from an
earlier run: quoting a number from before a code change is the specific
mistake this project keeps catching.
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, Iterable, List, Sequence
PUA_RANGE = (0xE000, 0xF8FF)
REPLACEMENT_CHAR = ""
# Strings that were confirmed by eye to be corruption, each traced to a
# dropped vector-outlined glyph (outlier-catalog item 24). They are checked
# literally: if one reappears, the repair regressed.
KNOWN_CORRUPTIONS = (
"Độ n định",
"≥ 1 tu i",
"tại ch :",
)
# Fragments of 2D formulas that must never sit in prose, where the missing
# fraction bar turns a division into a multiplication.
FORMULA_FRAGMENTS = (
"Thể trọng (kg)",
"(140 - tuổi) x cân nặng",
"x (140 - số tuổi)",
"Giá trị Clcr của bệnh nhân",
"218 x P x",
"× trọng lượng cơ thể (kg)",
"Cân nặng (kg) x liều",
)
@dataclass(frozen=True)
class Gate:
name: str
count: int
target: int = 0
detail: str = ""
@property
def passed(self) -> bool:
return self.count == self.target
def _section_texts(monograph: dict) -> Iterable[str]:
for section in (monograph.get("sections") or {}).values():
yield section.get("text") or ""
def _count_pua(text: str) -> int:
return sum(1 for ch in text if PUA_RANGE[0] <= ord(ch) <= PUA_RANGE[1])
def evaluate(monographs: Sequence[dict],
transcribed_runs: Sequence[dict] = ()) -> List[Gate]:
"""Compute every readiness gate over the whole corpus."""
pua = replacement = empty = no_provenance = 0
corruptions: Dict[str, int] = {c: 0 for c in KNOWN_CORRUPTIONS}
formula_leaks: Dict[str, int] = {f: 0 for f in FORMULA_FRAGMENTS}
unflagged_blocks = 0
ids: Dict[str, int] = {}
no_page_range = 0
corpus = []
for monograph in monographs:
ids[monograph["drug_id"]] = ids.get(monograph["drug_id"], 0) + 1
if not monograph.get("source_page_range"):
no_page_range += 1
for section in (monograph.get("sections") or {}).values():
text = section.get("text") or ""
corpus.append(text)
if not text.strip():
empty += 1
if not section.get("parts"):
no_provenance += 1
pua += _count_pua(text)
replacement += text.count(REPLACEMENT_CHAR)
for phrase in KNOWN_CORRUPTIONS:
corruptions[phrase] += text.count(phrase)
for phrase in FORMULA_FRAGMENTS:
formula_leaks[phrase] += text.count(phrase)
for block in monograph.get("tables") or []:
if not block.get("quarantined"):
unflagged_blocks += 1
joined = "\n".join(corpus)
unmerged = [
run for run in transcribed_runs
if len(run["text"].strip()) > 2 and run["text"].strip() not in joined
]
return [
Gate("outlined_run_not_merged", len(unmerged),
detail="; ".join(f"p{r['physical_page']} {r['text'][:40]!r}"
for r in unmerged[:5])),
Gate("known_corruption_string", sum(corruptions.values()),
detail=", ".join(f"{k!r}={v}" for k, v in corruptions.items() if v)),
Gate("formula_fragment_in_prose", sum(formula_leaks.values()),
detail=", ".join(f"{k!r}={v}" for k, v in formula_leaks.items() if v)),
Gate("pua_char", pua),
Gate("replacement_char_ufffd", replacement),
Gate("empty_section", empty),
Gate("section_without_provenance", no_provenance),
Gate("unflagged_quarantine_block", unflagged_blocks),
Gate("duplicate_drug_id", sum(1 for n in ids.values() if n > 1)),
Gate("monograph_without_page_range", no_page_range),
]
def corpus_size(monographs: Sequence[dict]) -> Dict[str, int]:
"""Informational, not a gate: how much text chunking would consume."""
sections = [t for m in monographs for t in _section_texts(m)]
return {
"monographs": len(monographs),
"sections": len(sections),
"section_chars": sum(len(t) for t in sections),
"quarantined_blocks": sum(len(m.get("tables") or []) for m in monographs),
}
def read_monographs(path: Path) -> List[dict]:
with path.open(encoding="utf-8") as handle:
return [json.loads(line) for line in handle if line.strip()]
def evaluate_chunks(monographs: Sequence[dict],
chunks: Sequence[dict]) -> List[Gate]:
"""ADR 0006 gates: a chunk must never hide that a block was lifted.
The failure being guarded against is silent, not visible: a chunk of
AMPICILIN VÀ SULBACTAM's dosing section is grammatical, complete-looking
prose with the renal-dosing table absent and nothing marking the absence.
Measured: 127 of 167 lifted blocks came out of `liều lượng và cách dùng`.
"""
blocks_by_section: Dict[tuple, list] = {}
block_ids: Dict[str, str] = {}
block_texts: Dict[str, str] = {}
for monograph in monographs:
for block in monograph.get("tables") or []:
key = (monograph["drug_id"], block.get("section_key"))
blocks_by_section.setdefault(key, []).append(block)
block_ids[block["table_id"]] = monograph["drug_id"]
if block.get("text"):
block_texts[block["table_id"]] = block["text"]
referenced: Dict[tuple, set] = {}
unknown_id = missing_provenance = leaked = 0
descriptors = 0
descriptor_without_attachment = 0
for chunk in chunks:
attachments = chunk.get("attachments") or []
if chunk.get("chunk_kind") == "block_descriptor":
descriptors += 1
if not attachments:
descriptor_without_attachment += 1
key = (chunk["drug_id"], chunk["section_key"])
for attachment in attachments:
referenced.setdefault(key, set()).add(attachment["block_id"])
if block_ids.get(attachment["block_id"]) != chunk["drug_id"]:
unknown_id += 1
if attachment.get("physical_page") is None or not attachment.get("bbox"):
missing_provenance += 1
body = chunk.get("text") or ""
for attachment in attachments:
source = block_texts.get(attachment["block_id"], "")
probe = source.strip()[:60]
if len(probe) > 20 and probe in body:
leaked += 1
unreferenced = 0
for key, blocks in blocks_by_section.items():
seen = referenced.get(key, set())
unreferenced += sum(1 for b in blocks if b["table_id"] not in seen)
total_blocks = sum(len(v) for v in blocks_by_section.values())
return [
Gate("section_block_without_chunk_reference", unreferenced),
Gate("attachment_block_id_unknown", unknown_id),
Gate("attachment_without_page_or_bbox", missing_provenance),
Gate("block_text_leaked_into_chunk_text", leaked),
Gate("descriptor_chunk_without_attachment", descriptor_without_attachment),
Gate("descriptor_count_vs_block_count", descriptors, target=total_blocks,
detail=f"{descriptors} descriptors for {total_blocks} blocks"),
]
def read_chunks(path: Path) -> List[dict]:
with path.open(encoding="utf-8") as handle:
return [json.loads(line) for line in handle if line.strip()]
@@ -0,0 +1,245 @@
"""Residual-ink coverage check: what is on the page that the text layer never emitted.
Every other check in this project asks a detector whether it found something.
This one asks the page. It renders each page, whites out every pixel covered
by a span the extractor actually produced, and reports the ink that survives.
Whatever survives is content the text layer cannot account for — vector
rules, fraction bars, figures.
Why it earns its place: the two confirmed 2D-formula corruptions
(NETILMICIN physical page 1042, AMPICILIN VÀ SULBACTAM physical page 202)
are invisible to both table detectors in this repo — `pdfplumber` reports 0
regions on those pages and so does `opendataloader-pdf`. The two tools share
a blind spot because both need ruling lines. Pixels do not share it: the
fraction bar is ink, so it survives the mask and gets reported.
The check needs no ground truth and no sampling — measured at 0.06 s/page,
so all 1668 pages run in under two minutes.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Callable, Dict, Iterable, Iterator, List, Sequence, Tuple
import fitz
import numpy as np
from scipy import ndimage
from ..extract.outlined_text import OutlinedTextRun, detect_outlined_text
from ..tables.models import TableRegion
RENDER_DPI = 150
INK_THRESHOLD = 200
# Measured, not guessed: at 1.0pt the mask eats the fraction bar itself —
# page 1042's bar survives as 9.1pt of its true 188.6pt. At 0.5pt the full
# bar survives, and a 10-page prose sample produced the same region count as
# 1.0pt (11 regions), i.e. the looser padding adds no noise.
MASK_PAD_PT = 0.5
THIN_HEIGHT_PT = 3.0
HEADER_BAND_PT = 60.0
RULE_MIN_WIDTH_PT = 400.0
BAR_MIN_WIDTH_PT = 10.0
# A glyph outline can extend a fraction past the filled path's own box, so the
# overlap test is given room: without it, three ink fragments on page 714 sit
# just outside their line's box and read as unexplained text.
OUTLINE_TOLERANCE_PT = 2.0
# Smaller than any mark a real glyph leaves. Measured against the 1,054
# components of confirmed outlined text on the five affected pages: the
# smallest is well above this, so the rule cannot swallow real text.
SPECK_EXTENT_PT = 2.0
HEADER_RULE = "header_rule"
TABLE_FRAME = "table_frame"
TEXT_AS_VECTOR_OUTLINE = "text_as_vector_outline"
FRACTION_BAR_CANDIDATE = "fraction_bar_candidate"
HEADER_BAND_FRAGMENT = "header_band_fragment"
RULE_FRAGMENT = "rule_fragment"
ANTIALIAS_SPECK = "antialias_speck"
UNCLASSIFIED = "unclassified"
@dataclass(frozen=True)
class ResidualRegion:
"""Ink left on a page after masking every extracted span.
Provenance is the point: `physical_page` + `bbox` locate the region in the
source PDF exactly, so any verdict about it can be re-checked by eye.
"""
physical_page: int
bbox: Tuple[float, float, float, float]
ink_px: int
@property
def width_pt(self) -> float:
return self.bbox[2] - self.bbox[0]
@property
def height_pt(self) -> float:
return self.bbox[3] - self.bbox[1]
@dataclass(frozen=True)
class PageContext:
"""What else is known to be on the page, for naming residual ink.
Carried as one object so a new kind of context is a new field here rather
than a new positional argument on every predicate.
"""
tables: Sequence[TableRegion] = ()
outlined_runs: Sequence[OutlinedTextRun] = ()
def _overlaps(bbox, other) -> bool:
return not (bbox[2] < other[0] or bbox[0] > other[2]
or bbox[3] < other[1] or bbox[1] > other[3])
def _is_header_rule(region: ResidualRegion, _context: PageContext) -> bool:
return (
region.height_pt <= THIN_HEIGHT_PT
and region.bbox[1] < HEADER_BAND_PT
and region.width_pt >= RULE_MIN_WIDTH_PT
)
def _is_table_frame(region: ResidualRegion, context: PageContext) -> bool:
return any(table.contains(*region.bbox) for table in context.tables)
def _is_outlined_text(region: ResidualRegion, context: PageContext) -> bool:
pad = OUTLINE_TOLERANCE_PT
return any(
_overlaps(region.bbox,
(line.bbox[0] - pad, line.bbox[1] - pad,
line.bbox[2] + pad, line.bbox[3] + pad))
for line in context.outlined_runs
)
def _is_fraction_bar(region: ResidualRegion, _context: PageContext) -> bool:
return region.height_pt <= THIN_HEIGHT_PT and region.width_pt >= BAR_MIN_WIDTH_PT
def _is_header_band_fragment(region: ResidualRegion, _context: PageContext) -> bool:
"""Leftovers of the running-header rule, chopped up by the text over it.
Confirmed by eye on physical page 382: a 31.7 x 9.6pt L-shape that is the
header rule meeting a vertical tick, split into its own component because
the header text's mask cut the rule either side of it.
"""
return region.bbox[3] <= HEADER_BAND_PT
def _is_rule_fragment(region: ResidualRegion, _context: PageContext) -> bool:
return min(region.width_pt, region.height_pt) <= THIN_HEIGHT_PT
def _is_speck(region: ResidualRegion, _context: PageContext) -> bool:
return (region.width_pt < SPECK_EXTENT_PT
and region.height_pt < SPECK_EXTENT_PT)
# Open/closed: a new residual kind is a new entry here, not an edit to the
# existing predicates. Order matters — first match wins. Outlined text is
# tested before the fraction-bar shape rule, which its underline-like
# fragments would otherwise satisfy; the header band is tested before it too,
# because the header text's mask cuts the running rule into short pieces that
# are bar-shaped (8 of them on physical page 382 alone).
_RULES: List[Tuple[str, Callable[[ResidualRegion, PageContext], bool]]] = [
(HEADER_RULE, _is_header_rule),
(TABLE_FRAME, _is_table_frame),
(TEXT_AS_VECTOR_OUTLINE, _is_outlined_text),
(HEADER_BAND_FRAGMENT, _is_header_band_fragment),
(FRACTION_BAR_CANDIDATE, _is_fraction_bar),
(ANTIALIAS_SPECK, _is_speck),
(RULE_FRAGMENT, _is_rule_fragment),
]
def classify(region: ResidualRegion, context: PageContext | None = None) -> str:
"""Name what a residual region is. Pure — no PDF, no rendering."""
context = context or PageContext()
for kind, predicate in _RULES:
if predicate(region, context):
return kind
return UNCLASSIFIED
def _ink_boxes(mask: "np.ndarray") -> Iterator[Tuple[int, int, int, int]]:
"""One box per connected blob of surviving ink.
Two cheaper splits were tried first and both misreport real pages. Cutting
into horizontal bands only merges a table in the left column with one in
the right column, so the merged box's centre lands in the gutter, matches
no table region, and physical page 209's ADR table is reported as
unaccounted-for ink. Adding a column-run split then cuts a single table
grid into its individual rules, because masking the text leaves the rules
standing with empty gaps between them. A table grid is one connected
object and a fraction bar is another, so connectivity is the property that
actually separates them.
"""
labelled, _ = ndimage.label(mask, structure=np.ones((3, 3), dtype=bool))
for top_bottom, left_right in ndimage.find_objects(labelled) or []:
yield left_right.start, top_bottom.start, left_right.stop - 1, top_bottom.stop - 1
def scan_page(page: "fitz.Page", dpi: int = RENDER_DPI) -> List[ResidualRegion]:
"""Render one page, mask its extracted spans, return the surviving ink."""
scale = dpi / 72.0
pixmap = page.get_pixmap(dpi=dpi, colorspace=fitz.csGRAY)
image = np.frombuffer(pixmap.samples, dtype=np.uint8).reshape(
pixmap.height, pixmap.width
).copy()
for block in page.get_text("dict")["blocks"]:
for line in block.get("lines", []):
for span in line["spans"]:
x0, y0, x1, y1 = span["bbox"]
top = max(0, int((y0 - MASK_PAD_PT) * scale))
bottom = min(pixmap.height, int((y1 + MASK_PAD_PT) * scale) + 1)
left = max(0, int((x0 - MASK_PAD_PT) * scale))
right = min(pixmap.width, int((x1 + MASK_PAD_PT) * scale) + 1)
image[top:bottom, left:right] = 255
mask = image < INK_THRESHOLD
return [
ResidualRegion(
physical_page=page.number,
bbox=(
round(left / scale, 2),
round(top / scale, 2),
round(right / scale, 2),
round(bottom / scale, 2),
),
ink_px=int(mask[top:bottom + 1, left:right + 1].sum()),
)
for left, top, right, bottom in _ink_boxes(mask)
]
def scan_document(
doc: "fitz.Document",
tables_by_page: Dict[int, List[TableRegion]] | None = None,
pages: Iterable[int] | None = None,
) -> Iterator[Tuple[ResidualRegion, str]]:
"""Yield every residual region in the document with its classification."""
tables_by_page = tables_by_page or {}
page_numbers = list(range(doc.page_count) if pages is None else pages)
outlines: Dict[int, List[OutlinedTextRun]] = {}
for line in detect_outlined_text(doc, page_numbers):
outlines.setdefault(line.physical_page, []).append(line)
for number in page_numbers:
context = PageContext(
tables=tables_by_page.get(number, ()),
outlined_runs=outlines.get(number, ()),
)
for region in scan_page(doc[number]):
yield region, classify(region, context)
+7 -1
View File
@@ -3,8 +3,14 @@ name = "ingestion"
version = "0.0.0"
description = "Offline batch pipeline: PDF -> monographs -> chunks -> embeddings -> Qdrant"
requires-python = ">=3.11"
dependencies = []
dependencies = ["pymupdf>=1.24", "numpy>=1.26", "scipy>=1.11"]
[project.optional-dependencies]
dev = ["pytest>=7.4"]
[build-system]
requires = ["setuptools>=68"]
build-backend = "setuptools.build_meta"
[tool.setuptools.packages.find]
include = ["ingestion*"]
+168
View File
@@ -0,0 +1,168 @@
import json
from pathlib import Path
from ingestion.chunk import (
CHUNK_KIND_BLOCK_DESCRIPTOR,
CHUNK_KIND_PROSE,
SCHEMA_VERSION,
chunk_monograph,
chunk_section,
write_chunks_jsonl,
)
from ingestion.chunk.chunker import _is_label_row, describe_block
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
from ingestion.tables import SHAPE_FORMULA_2D, SHAPE_MULTI_HEADER, SHAPE_SIMPLE
def _section(key, display, text, page=202):
return SectionSpan(
key=key, display_name=display,
heading=Heading(text=display, physical_page=page, y0=100.0,
is_monograph_title=False, section_key=key),
text=text,
)
def _monograph(sections, tables=()):
return Monograph(
drug_id="ampicilin_va_sulbactam",
drug_name="AMPICILIN VÀ SULBACTAM",
source_page_range=[200, 203],
sections={s.key: s for s in sections},
atc_codes=["J01CR01"],
tables=list(tables),
)
def _block(block_id="p202_t0", shape=SHAPE_SIMPLE, section_key="lieu_luong_va_cach_dung"):
return TableBlock(
table_id=block_id, shape=shape, physical_page=202,
bbox=[299.0, 189.6, 552.4, 300.5], section_key=section_key,
text="Độ thanh thải creatinin Nửa đời Liều 1,5 - 3,0 g",
quarantined=True,
)
def test_a_section_whose_table_was_lifted_says_so():
"""The defect this exists to prevent is silent, not visible.
Without the reference, this chunk is grammatical, complete-looking prose
with the renal-dosing table absent and nothing marking the absence.
"""
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
monograph = _monograph([section], [_block()])
chunks = chunk_monograph(monograph)
prose = [c for c in chunks if c.chunk_kind == CHUNK_KIND_PROSE]
assert len(prose) == 1
assert prose[0].has_quarantined_content is True
assert [a.block_id for a in prose[0].attachments] == ["p202_t0"]
assert prose[0].attachments[0].physical_page == 202
assert prose[0].attachments[0].bbox
def test_a_lifted_block_gets_its_own_retrievable_descriptor():
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
monograph = _monograph([section], [_block()])
descriptors = [c for c in chunk_monograph(monograph)
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
assert len(descriptors) == 1
assert "AMPICILIN VÀ SULBACTAM" in descriptors[0].text
assert "Liều lượng và cách dùng" in descriptors[0].text
# printed page, which is what a reader holding the book looks for
assert "trang 203" in descriptors[0].text
def test_no_cell_value_ever_reaches_the_descriptor_text():
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
block = _block()
monograph = _monograph([section], [block])
descriptors = [c for c in chunk_monograph(monograph, {"p202_t0": []})
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
assert "1,5 - 3,0 g" not in descriptors[0].text
def test_a_header_row_carrying_a_number_is_refused():
"""AMIODARON, physical page 183 — a real case, caught by a gate.
pdfplumber reported the first row as
"Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)", i.e. a
dose inside what it called a header, from an extraction never verified by
eye. Measured: 42 of 124 simple-table headers (34%) contain a digit.
"""
assert _is_label_row(["Các Statin", "Khởi đầu", "Liều duy trì"]) is True
assert _is_label_row(["Liều 720 mg/ngày (0,5 mg/phút)"]) is False
assert _is_label_row(["x" * 45]) is False
assert _is_label_row([]) is False
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
monograph = _monograph([section], [_block()])
chunks = chunk_monograph(
monograph, {"p202_t0": ["Liều 720 mg/ngày (0,5 mg/phút)"]})
descriptor = next(c for c in chunks
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
assert "720" not in descriptor.text
assert descriptor.attachments[0].header_row == []
def test_only_a_simple_table_contributes_a_header():
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
header = {"p202_t0": ["Nhóm", "Liều"]}
for shape, expected in ((SHAPE_SIMPLE, ["Nhóm", "Liều"]),
(SHAPE_MULTI_HEADER, [])):
monograph = _monograph([section], [_block(shape=shape)])
descriptor = next(c for c in chunk_monograph(monograph, header)
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
assert descriptor.attachments[0].header_row == expected
def test_a_formula_block_is_described_as_a_formula():
section = _section("than_trong", "Thận trọng", "Prose.")
block = _block(block_id="p1042_f0", shape=SHAPE_FORMULA_2D,
section_key="than_trong")
monograph = _monograph([section], [block])
descriptor = next(c for c in chunk_monograph(monograph)
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
assert "công thức" in descriptor.text
assert "bảng" not in descriptor.text
def test_attachments_do_not_change_the_prose_text():
"""The condition under which this feature was accepted at all."""
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
with_block = chunk_section(_monograph([section], [_block()]), section,
[_block()])
without = chunk_section(_monograph([section]), section)
prose_with = [c for c in with_block if c.chunk_kind == CHUNK_KIND_PROSE]
assert [c.text for c in prose_with] == [c.text for c in without]
assert [c.chunk_id for c in prose_with] == [c.chunk_id for c in without]
def test_a_section_with_no_text_but_a_block_still_yields_the_descriptor():
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "")
chunks = chunk_monograph(_monograph([section], [_block()]))
assert [c.chunk_kind for c in chunks] == [CHUNK_KIND_BLOCK_DESCRIPTOR]
def test_written_chunks_declare_their_schema_version(tmp_path: Path):
section = _section("chi_dinh", "Chỉ định", "Nhiễm khuẩn.")
chunks = chunk_monograph(_monograph([section]))
out = tmp_path / "chunks.jsonl"
assert write_chunks_jsonl(chunks, out) == 1
record = json.loads(out.read_text(encoding="utf-8").splitlines()[0])
assert record["schema_version"] == SCHEMA_VERSION
assert record["chunk_kind"] == CHUNK_KIND_PROSE
assert record["has_quarantined_content"] is False
def test_describe_block_names_the_page_even_with_no_header():
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
monograph = _monograph([section], [_block()])
chunks = chunk_monograph(monograph)
attachment = next(c for c in chunks
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR).attachments[0]
text = describe_block(monograph, section, attachment)
assert "trang 203" in text
assert "không trích dẫn được dưới dạng văn bản" in text
+49
View File
@@ -0,0 +1,49 @@
import pytest
from ingestion.cli import build_parser
def test_run_subcommand_parses_required_pdf_arg():
parser = build_parser()
args = parser.parse_args(["run", "--pdf", "some.pdf"])
assert args.command == "run"
assert args.pdf == "some.pdf"
assert args.out == "data/processed/monographs.jsonl"
def test_run_subcommand_accepts_custom_out():
parser = build_parser()
args = parser.parse_args(["run", "--pdf", "a.pdf", "--out", "b.jsonl"])
assert args.out == "b.jsonl"
def test_run_requires_pdf_arg():
parser = build_parser()
with pytest.raises(SystemExit):
parser.parse_args(["run"])
@pytest.mark.parametrize("command", ["visual-diff", "scaffold-golden"])
def test_not_yet_implemented_commands_raise_explicitly(command):
parser = build_parser()
args = parser.parse_args([command])
with pytest.raises(NotImplementedError):
args.func(args)
def test_run_reports_missing_pdf_file(tmp_path, capsys):
parser = build_parser()
missing = tmp_path / "does_not_exist.pdf"
args = parser.parse_args(["run", "--pdf", str(missing)])
exit_code = args.func(args)
assert exit_code == 1
assert "not found" in capsys.readouterr().err
def test_validate_reports_missing_pdf_file(tmp_path, capsys):
parser = build_parser()
missing = tmp_path / "does_not_exist.pdf"
args = parser.parse_args(["validate", "--pdf", str(missing)])
exit_code = args.func(args)
assert exit_code == 1
assert "not found" in capsys.readouterr().err
+75
View File
@@ -0,0 +1,75 @@
import json
from pathlib import Path
from ingestion.extract.formulas import (
FORMULA_BAND_HEIGHT_PT,
FORMULA_SIDE_MARGIN_PT,
load_formula_regions,
)
from ingestion.tables import QUARANTINE_SHAPES, SHAPE_FORMULA_2D
VERIFIED = (Path(__file__).resolve().parents[1] / "data" / "verified"
/ "formula_regions_2d.json")
TRANSCRIPTIONS = (Path(__file__).resolve().parents[1] / "data" / "verified"
/ "outlined_text_transcriptions.json")
def test_a_2d_formula_is_always_quarantined():
# linearised, "a / b" reads as "a x b" — a dosing error, not a cosmetic one
assert SHAPE_FORMULA_2D in QUARANTINE_SHAPES
def test_verified_formula_regions_load_with_the_confirmed_pages():
regions = load_formula_regions()
assert {r.physical_page for r in regions} == {
43, 92, 147, 202, 325, 349, 1042, 1043, 1132, 1402,
}
assert all(r.shape == SHAPE_FORMULA_2D for r in regions)
def test_the_region_covers_numerator_and_denominator_not_just_the_bar():
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
bar = next(r for r in payload["regions"] if r["physical_page"] == 1042)
region = next(r for r in load_formula_regions() if r.physical_page == 1042)
x0, y0, x1, y1 = bar["bar_bbox"]
assert region.bbox[1] == y0 - FORMULA_BAND_HEIGHT_PT
assert region.bbox[3] == y1 + FORMULA_BAND_HEIGHT_PT
assert region.bbox[0] == x0 - FORMULA_SIDE_MARGIN_PT
def test_the_barless_adenosin_formula_is_recorded_as_a_recall_limit():
"""The source prints no bar, so no geometric detector can find it.
Recorded so a later reader does not mistake the fraction-bar scan for
complete formula coverage — how many bar-less formulas the book contains
has never been measured.
"""
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
barless = [r for r in payload["regions"] if r.get("source_prints_no_bar")]
assert [r["physical_page"] for r in barless] == [147]
assert "UNMEASURED" in payload["recall_limit"]
def test_outlined_text_transcriptions_cover_every_detected_run():
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
runs = payload["runs"]
assert len(runs) == 51
assert all(r["text"] for r in runs), "a run with no transcription is data loss"
pages = {}
for run in runs:
pages[run["physical_page"]] = pages.get(run["physical_page"], 0) + 1
assert pages == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}
def test_single_glyph_transcriptions_name_the_line_they_were_dropped_from():
"""The subtlest form of the defect: one character missing mid-sentence.
"Độ ổn định" extracts as "Độ n định" and reads as ordinary text, so
nothing downstream can notice. Keeping the owning line in the record is
what makes the repair checkable.
"""
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
singles = [r for r in payload["runs"] if r["single_glyph"]]
assert len(singles) == 29
with_context = [r for r in singles if r["extracted_line_it_belongs_to"]]
assert with_context, "no dropped glyph could be tied back to its line"
@@ -0,0 +1,79 @@
from ingestion.extract.glyph_order import find_reading_order_issues, is_reversed_order
def test_normal_ltr_span_not_flagged():
# ordinary increasing x-origins, as any normal left-to-right span has
assert not is_reversed_order([264.7, 269.4, 271.6, 276.3, 278.5])
def test_confirmed_page_1373_defect_shape_is_flagged():
# exact x-origins read via get_text("rawdict") from physical page 1373's
# affected span (" tịx 4 =" reversed) — see docs/pdf-parsing-outlier-catalog.md item 9
x_origins = [66.32, 64.17, 61.53, 58.89, 54.14, 51.98, 47.23]
assert is_reversed_order(x_origins)
def test_single_char_span_not_flagged():
assert not is_reversed_order([100.0])
def test_empty_span_not_flagged():
assert not is_reversed_order([])
def test_tied_x_origins_not_flagged_as_reversed():
# equal x-origins (e.g. stacked/overlapping glyphs) are not "decreasing"
assert not is_reversed_order([100.0, 100.0, 100.0])
def test_correctly_ordered_row_not_flagged():
row = {(20, 550.9): [(518.0, "n"), (525.2, "h"), (532.6, "i"), (536.8, "e")]}
assert find_reading_order_issues(row) == []
def test_confirmed_page_714_row_misorder_is_flagged():
# reproduces the real page-714 finding: within one PyMuPDF block (20),
# 4 line fragments are emitted out of x-order ("quản ", " ộ", "đ tệih",
# "n " concatenated) that reconstruct correctly ("...nhiệt độ") when
# re-sorted by x-origin — see outlier catalog item 9.
row = {
(20, 550.9): [
(518.06, "n"), (525.20, " "),
(546.59, " "), (553.71, ""),
(541.84, "đ"), (539.45, " "), (536.81, "t"), (532.59, ""), (529.95, "i"), (525.20, "h"),
]
}
issues = find_reading_order_issues(row)
assert len(issues) == 1
assert issues[0].extracted_text != issues[0].corrected_text
def test_different_blocks_at_same_y_not_merged():
# regression test for a real false positive: two DIFFERENT paragraphs in
# different PyMuPDF blocks (a right-column paragraph starting at x=299.4
# and a left-column paragraph starting at x=35.4, page 1104) coincide at
# the same y — grouping by block index (not a hand-picked x-coordinate
# column boundary) is what keeps them from being merged into one "row".
# This is the caller's responsibility (scan_reading_order groups by real
# PyMuPDF block index); find_reading_order_issues just trusts its input
# is already correctly grouped, which these two dict entries demonstrate.
row_block_1 = {(1, 70.4): [(299.39, "m"), (306.78, "ô")]}
row_block_4 = {(4, 70.4): [(35.43, "d"), (40.18, "e")]}
assert find_reading_order_issues(row_block_1) == []
assert find_reading_order_issues(row_block_4) == []
def test_kerning_jitter_not_flagged_as_reading_order_defect():
# regression test for a real false positive found by running against the
# actual PDF: "mefloquin" ('l' at x=491.566, 'o' at x=491.471 — a
# 0.095pt kerning-driven dip) was previously "corrected" into the wrong
# word "mefolquin". A row-level check with no decrease tolerance treats
# ordinary kerning as a defect and corrupts already-correct text.
row = {
(5, 449.7): [
(474.865, "m"), (482.161, "e"), (486.284, "f"),
(491.566, "l"), (491.471, "o"), (496.126, "q"),
(500.781, "u"), (505.436, "i"), (507.982, "n"),
]
}
assert find_reading_order_issues(row) == []
+27
View File
@@ -0,0 +1,27 @@
from ingestion.extract.page_map import pick_folio
def test_single_candidate_is_the_folio():
assert pick_folio([("101", 10.0)]) == 101
def test_no_candidates_is_unrecoverable():
assert pick_folio([]) is None
def test_confirmed_riboflavin_subscript_conflict_resolved_by_size():
# exact (text, size) pairs read from physical page 1243's header band:
# the real folio "1244" (size 10.0, matching the rest of the running
# header) and the "2" subscript from "Vitamin B2" (size 5.83), which
# happens to fall in the same y<60 header band because the RIBOFLAVIN
# title sits high on the page — see module docstring. Silently dropped
# the whole monograph before this fix, confirmed via a whole-book
# `cli validate` run and by rendering the page to an image.
candidates = [("1244", 10.0), ("2", 5.83)]
assert pick_folio(candidates) == 1244
def test_genuine_same_size_conflict_still_returns_none():
# two same-size digit-only candidates: real ambiguity, must not guess
candidates = [("101", 10.0), ("205", 10.0)]
assert pick_folio(candidates) is None
+67
View File
@@ -0,0 +1,67 @@
from ingestion.extract.spans import classify_column, _sort_blocks_reading_order
def _block(x0, y0, x1, y1):
return {"bbox": (x0, y0, x1, y1)}
def testclassify_column_left():
assert classify_column((35.0, 100.0, 280.0, 120.0)) == "left"
def testclassify_column_right():
assert classify_column((299.0, 100.0, 553.0, 120.0)) == "right"
def testclassify_column_full_width_header():
assert classify_column((35.0, 34.0, 552.0, 48.0)) == "full_width"
def testclassify_column_none_bbox_is_unknown():
assert classify_column(None) == "unknown"
def test_confirmed_real_oxymetazolin_page_reversed_order_is_corrected():
# exact bboxes from physical page 1100 (the OXYBUTYNIN/OXYMETAZOLIN
# boundary — see spans.py module docstring): PyMuPDF's raw block order
# is [header, right x7, left x8], right column before left. An earlier
# version of this module trusted that raw order, silently attributing
# OXYMETAZOLIN's "Chống chỉ định" (right column) to the still-open
# OXYBUTYNIN monograph. Confirmed via a whole-book cli validate run,
# a whole-document cross-tool character-diff, and rendering the page.
raw_order = [
_block(34.96, 34.39, 552.10, 47.72), # 0: full_width header
_block(299.39, 60.46, 553.72, 121.96), # 1: right
_block(299.39, 124.33, 553.72, 368.94), # 2: right
_block(299.39, 371.31, 553.72, 408.39), # 3: right
_block(35.43, 60.77, 289.77, 330.56), # 4: left (Xử trí: ...)
_block(35.43, 379.21, 231.23, 391.87), # 5: left (Tên chung quốc tế)
]
sorted_blocks = _sort_blocks_reading_order(raw_order)
columns_in_order = [classify_column(b["bbox"]) for b in sorted_blocks]
assert columns_in_order == ["full_width", "left", "left", "right", "right", "right"]
def test_already_correct_order_is_left_unchanged_in_content():
blocks = [
_block(35.0, 60.0, 280.0, 100.0), # left
_block(35.0, 110.0, 280.0, 150.0), # left, further down
_block(299.0, 60.0, 553.0, 100.0), # right
]
sorted_blocks = _sort_blocks_reading_order(blocks)
assert sorted_blocks == blocks
def test_a_narrow_box_between_the_columns_belongs_to_the_right_column():
"""The two tolerance bands overlap between x=288 and x=319.
Testing left first put everything in that strip in the left column. It is
invisible for a full-width block and wrong for a narrow one: a single 4pt
glyph at x=315 on physical page 714 was classified left, so the ''
missing from "Độ ổn định" could not be matched to its own line and the
corruption survived the repair.
"""
assert classify_column((313.7, 506.2, 317.8, 514.8)) == "right"
assert classify_column((35.4, 500.0, 289.7, 510.0)) == "left"
# a box that lands in neither range still resolves by tolerance
assert classify_column((300.0, 500.0, 305.0, 510.0)) == "left"
+95
View File
@@ -0,0 +1,95 @@
from ingestion.extract.models import Span
from ingestion.normalize import (
PUA_SUBSTITUTIONS,
find_unmapped_pua,
group_visual_lines,
join_spans,
substitute_pua,
)
def _span(text, *, page=100, block=0, line=0, index=0, x0=50.0, x1=None, y0=100.0):
return Span(
physical_page=page, printed_page=page + 1, column="left",
block=block, line=line, span_index=index,
x0=x0, y0=y0, x1=(x0 + len(text) * 4.5) if x1 is None else x1, y1=y0 + 10,
text=text, font="Tiger", size=9.5,
)
def test_pua_map_covers_every_codepoint_confirmed_in_the_corpus():
# all 8 were located in the source PDF, rendered, and read visually —
# see docs/progress-log.md for the page each was confirmed on
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "α"
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "®"
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "γ"
def test_comparison_operators_in_real_dosing_sentences_are_restored():
# the clinically dangerous case: without this, "liều ≤ 100 mg" reaches
# embeddings as "liều  100 mg" and the operator is lost
assert substitute_pua("trẻ em  10 tuổi") == "trẻ em ≥ 10 tuổi"
assert substitute_pua("liều  100 mg") == "liều ≤ 100 mg"
def test_unmapped_pua_is_reported_not_silently_passed_through():
assert find_unmapped_pua("liều  100 mg") == []
assert find_unmapped_pua("bất ngờ  đây") == [""]
def test_subscript_span_rejoins_without_a_spurious_space():
# real corpus case: "cytochrom P450" arrived as "cytochrom P\n450\ngây"
spans = [
_span("cytochrom P", x0=50.0, x1=100.0),
_span("450", x0=100.2, x1=110.0),
_span(" gây chuyển hóa.", x0=110.1, x1=180.0),
]
assert join_spans(spans) == "cytochrom P450 gây chuyển hóa."
def test_italic_run_inside_parentheses_rejoins_on_one_line():
# real corpus case: "(\nfeline immunodeficiency virus\n)"
spans = [
_span("(", x0=50.0, x1=53.0),
_span("feline immunodeficiency virus", x0=53.1, x1=180.0),
_span(")", x0=180.1, x1=183.0),
]
assert join_spans(spans) == "(feline immunodeficiency virus)"
def test_wrap_without_sentence_end_is_joined_with_a_space():
spans = [
_span("không nhai. Nếu", line=0, y0=100.0),
_span("uống viên thuốc", line=1, y0=112.0),
]
assert join_spans(spans) == "không nhai. Nếu uống viên thuốc"
def test_sentence_end_keeps_the_line_break():
spans = [
_span("Liều người lớn: 10 mg.", line=0, y0=100.0),
_span("Trẻ em: 5 mg.", line=1, y0=112.0),
]
assert join_spans(spans) == "Liều người lớn: 10 mg.\nTrẻ em: 5 mg."
def test_wide_gap_on_one_line_still_yields_a_space():
spans = [
_span("Người bệnh", x0=50.0, x1=100.0),
_span("100 kg", x0=104.0, x1=130.0),
]
assert join_spans(spans) == "Người bệnh 100 kg"
def test_visual_lines_group_by_pymupdf_block_and_line_indices():
spans = [
_span("a", block=0, line=0), _span("b", block=0, line=0),
_span("c", block=0, line=1),
_span("d", block=1, line=0),
]
assert [len(g) for g in group_visual_lines(spans)] == [2, 1, 1]
+332
View File
@@ -0,0 +1,332 @@
import pytest
from ingestion.extract.models import Span
from ingestion.segment.assembler import DuplicateDrugIdError, assemble
def _span(text, page, y0, bold=True, size=9.5, printed=None, column="left"):
return Span(
physical_page=page, printed_page=printed if printed is not None else page + 1,
column=column, block=0, line=0, span_index=0,
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
text=text, font=("TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"), size=size,
)
def test_basic_single_monograph_with_sections_and_body():
spans = [
_span("ABACAVIR", 100, 60.0),
_span("Tên chung quốc tế:", 100, 80.0),
_span("Abacavir (Acyclovir-like).", 100, 92.0, bold=False),
_span("Mã ATC:", 100, 104.0),
_span("J05AF06", 100, 116.0, bold=False),
_span("Chỉ định", 101, 60.0),
_span("Điều trị nhiễm HIV.", 101, 72.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
m = monographs[0]
assert m.drug_id == "abacavir"
assert m.drug_name == "ABACAVIR"
assert m.source_page_range == [100, 101]
assert m.sections["ten_chung_quoc_te"].text == "Abacavir (Acyclovir-like)."
assert m.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
assert m.atc_codes == ["J05AF06"]
assert m.atc_stated_absent is False
def test_non_bold_combined_heading_value_span_confirmed_real_amitriptylin_case():
# AMITRIPTYLIN's real "Mã ATC:" heading is a single non-bold span
# combining label and value ("Mã ATC: N06AA09."), unlike Abacavir's
# bold-label + separate-value spans — see outlier item 20.
spans = [
_span("AMITRIPTYLIN", 184, 60.0),
_span("Tên chung quốc tế: ", 184, 85.0),
_span("Amitriptyline.", 184, 85.2, bold=False),
_span("Mã ATC: N06AA09.", 184, 100.0, bold=False),
_span("Loại thuốc:", 184, 115.0),
_span("Thuốc chống trầm cảm.", 184, 115.2, bold=False),
]
m = list(assemble(spans))[0]
assert m.sections["ma_atc"].text == "N06AA09."
assert m.atc_codes == ["N06AA09"]
def test_atc_stated_absent_propagates():
spans = [
_span("ADIPIODON", 100, 60.0),
_span("Tên chung quốc tế:", 100, 72.0),
_span("Adipiodon.", 100, 84.0, bold=False),
_span("Mã ATC:", 100, 96.0),
_span("Chưa có.", 100, 108.0, bold=False),
]
m = list(assemble(spans))[0]
assert m.atc_codes == []
assert m.atc_stated_absent is True
def test_qualifier_line_disambiguates_same_name_monographs():
# reproduces the confirmed real SALBUTAMOL case (outlier item 18):
# same base title, disambiguated by a bold non-caps parenthesized line.
spans = [
_span("SALBUTAMOL", 1261, 60.0),
_span("(Dùng trong hô hấp)", 1261, 72.0),
_span("Tên chung quốc tế:", 1261, 84.0),
_span("Salbutamol.", 1261, 96.0, bold=False),
_span("Chỉ định", 1261, 108.0),
_span("Điều trị hen.", 1261, 120.0, bold=False),
_span("SALBUTAMOL", 1263, 60.0),
_span("(Dùng trong sản khoa)", 1263, 72.0),
_span("Tên chung quốc tế:", 1263, 84.0),
_span("Salbutamol.", 1263, 96.0, bold=False),
_span("Chỉ định", 1263, 108.0),
_span("Điều trị dọa sinh non.", 1263, 120.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 2
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
assert monographs[0].drug_name == "SALBUTAMOL (Dùng trong hô hấp)"
assert monographs[1].drug_id == "salbutamol_dung_trong_san_khoa"
assert monographs[0].sections["chi_dinh"].text == "Điều trị hen."
assert monographs[1].sections["chi_dinh"].text == "Điều trị dọa sinh non."
def test_genuine_duplicate_drug_id_raises():
spans = [
_span("FOOBARDRUG", 200, 60.0),
_span("Tên chung quốc tế:", 200, 72.0),
_span("Foobardrug.", 200, 84.0, bold=False),
_span("Chỉ định", 200, 96.0),
_span("A.", 200, 108.0, bold=False),
_span("FOOBARDRUG", 300, 60.0),
_span("Tên chung quốc tế:", 300, 72.0),
_span("Foobardrug.", 300, 84.0, bold=False),
_span("Chỉ định", 300, 96.0),
_span("B.", 300, 108.0, bold=False),
]
with pytest.raises(DuplicateDrugIdError):
list(assemble(spans))
def test_gonadotropin_wrap_does_not_falsely_trigger_duplicate_check():
# regression: the multi-line wrap must merge BEFORE the duplicate check
# runs, so this is never treated as two separate "GONADOTROPIN" titles
spans = [
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
_span("GONADOTROPIN", 1371, 676.2354736328125),
_span("Tên chung quốc tế:", 1371, 690.0),
_span("Gonadorelin.", 1371, 700.0, bold=False),
_span("Chỉ định", 1371, 712.0),
_span("X.", 1371, 724.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_name == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
def test_front_matter_before_first_monograph_is_ignored():
spans = [
_span("Some front matter heading", 5, 60.0, bold=False, printed=6),
_span("random body text", 5, 72.0, bold=False, printed=6),
_span("ABACAVIR", 100, 60.0),
_span("Tên chung quốc tế:", 100, 80.0),
_span("Abacavir.", 100, 92.0, bold=False),
_span("Chỉ định", 100, 104.0),
_span("X.", 100, 116.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_id == "abacavir"
def test_empty_spans_yields_nothing():
assert list(assemble([])) == []
def test_table_header_false_positive_not_treated_as_monograph():
# reproduces the confirmed real "HSV"/"CMV" table-column-header case
# (outlier item 19, physical page 698, inside the Foscarnet natri
# monograph's dosing table) — bold+all-caps+short, identical shape to a
# real title, but never followed by "Tên chung quốc tế" before the next
# real title. Must not be treated as a monograph boundary.
spans = [
_span("FOSCARNET NATRI", 690, 60.0),
_span("Tên chung quốc tế:", 690, 80.0),
_span("Foscarnet.", 690, 92.0, bold=False),
_span("Chỉ định", 690, 104.0),
_span("Điều trị CMV.", 690, 116.0, bold=False),
_span("HSV", 698, 523.0),
_span("HSV", 698, 523.0),
_span("CMV", 698, 523.0),
_span("CMV", 698, 523.0),
_span("40 mg/kg cách nhau 12 giờ", 698, 540.0, bold=False),
_span("ARTEMETHER", 700, 60.0),
_span("Tên chung quốc tế:", 700, 80.0),
_span("Artemether.", 700, 92.0, bold=False),
]
monographs = list(assemble(spans))
assert [m.drug_id for m in monographs] == ["foscarnet_natri", "artemether"]
# the table row's numbers/labels stay attached to Foscarnet's Chỉ định
# section body (dropped from a dedicated section, which is fine — no
# false monograph boundary is what matters here)
assert "hsv" not in monographs[0].drug_id
assert "cmv" not in monographs[0].drug_id
def test_real_title_immediately_followed_by_anchor_is_kept():
spans = [
_span("ABACAVIR", 100, 60.0),
_span("Tên chung quốc tế:", 100, 80.0),
_span("Abacavir.", 100, 92.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_id == "abacavir"
def test_real_title_with_qualifier_before_anchor_is_still_kept():
# the anchor lookahead must tolerate one intervening qualifier-line
# event (the SALBUTAMOL case), not just immediate adjacency
spans = [
_span("SALBUTAMOL", 1261, 60.0),
_span("(Dùng trong hô hấp)", 1261, 72.0),
_span("Tên chung quốc tế:", 1261, 84.0),
_span("Salbutamol.", 1261, 96.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
def test_class_level_monograph_sub_heading_not_treated_as_own_monograph():
# reproduces the confirmed real case (outlier item 21): "SIMVASTATIN" is
# a bold+all-caps+short sub-heading *inside* the class-level "CÁC CHẤT
# ỨC CHẾ HMG-CoA REDUCTASE" monograph, immediately followed by its own
# "Liều lượng và cách dùng" but NOT by "Tên chung quốc tế" (that section
# belongs only to the parent). Must stay folded into the parent, not
# become its own monograph.
spans = [
_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 60.0, printed=285),
_span("Tên chung quốc tế:", 284, 72.0, printed=285),
_span("Simvastatin, Lovastatin.", 284, 84.0, bold=False, printed=285),
_span("Chỉ định", 284, 96.0, printed=285),
_span("Tăng lipid huyết.", 284, 108.0, bold=False, printed=285),
_span("SIMVASTATIN", 285, 60.0, printed=286),
_span("Liều lượng và cách dùng", 285, 72.0, printed=286),
_span("Uống 10 - 20 mg mỗi tối.", 285, 84.0, bold=False, printed=286),
_span("LOVASTATIN", 285, 96.0, printed=286),
_span("Liều lượng và cách dùng", 285, 108.0, printed=286),
_span("Uống 20 mg mỗi ngày.", 285, 120.0, bold=False, printed=286),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_name == "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"
# the sub-headings' own dosing text stays attached to the parent
# monograph's content rather than vanishing or becoming new monographs
assert "Uống 20 mg mỗi ngày." in monographs[0].sections["lieu_luong_va_cach_dung"].text
def test_running_header_boilerplate_stripped_from_mid_section_body_confirmed_real_morphin_case():
# exact confirmed real case: physical page 1008's running header
# ("DTQGVN 2" / "1009" / "Morphin sulfat", all column="full_width",
# y0~34, well inside the header band) falls squarely in the middle of
# MORPHIN SULFAT's "Liều lượng và cách dùng" section, which spans the
# page 1007->1008 boundary — see outlier-catalog item 13 / assembler.py
# module docstring. Whole-corpus measured: 1,374/11,409 sections (12.0%)
# affected before this fix, 671/682 monographs (98.4%) had at least one.
spans = [
_span("MORPHIN SULFAT", 1007, 60.0),
_span("Tên chung quốc tế:", 1007, 80.0),
_span("Morphini sulfas.", 1007, 92.0, bold=False),
_span("Liều lượng và cách dùng", 1007, 700.0),
_span("Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu", 1007, 785.4, bold=False),
_span("DTQGVN 2", 1008, 34.6, bold=False, column="full_width"),
_span("1009", 1008, 34.6, bold=False, column="full_width"),
_span("Morphin sulfat", 1008, 34.4, bold=False, column="full_width"),
_span("uống viên thuốc giải phóng chậm thì không được nghiền.", 1008, 60.8, bold=False),
]
m = list(assemble(spans))[0]
section_text = m.sections["lieu_luong_va_cach_dung"].text
assert "DTQGVN" not in section_text
assert "1009" not in section_text
# the two body spans are one sentence broken by a page boundary: "Nếu"
# does not end a sentence, so normalize/text_flow rejoins them with a
# space rather than preserving the PDF's visual wrap as a hard newline
assert section_text == (
"Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu "
"uống viên thuốc giải phóng chậm thì không được nghiền."
)
def test_last_real_monograph_in_book_still_kept_near_end_of_input():
# anchor lookahead must not require a "next title" to exist — the very
# last monograph in the book has no following title at all
spans = [
_span("ZOLPIDEM", 1494, 60.0),
_span("Tên chung quốc tế:", 1494, 80.0),
_span("Zolpidem.", 1494, 92.0, bold=False),
]
monographs = list(assemble(spans))
assert len(monographs) == 1
assert monographs[0].drug_id == "zolpidem"
def test_repeated_section_heading_appends_instead_of_overwriting():
# measured real case: 33 monographs repeat a section heading (38
# occurrences). CEFAMANDOL's "Liều lượng và cách dùng" resumes on
# physical page 339 after a renal-dosing table; the old code replaced the
# SectionSpan, destroying everything captured before the repeat — for
# CEFAMANDOL that left the dosing section holding only the table.
spans = [
_span("CEFAMANDOL", 338, 60.0),
_span("Tên chung quốc tế", 338, 80.0),
_span("Cefamandolum.", 338, 92.0, bold=False),
_span("Liều lượng và cách dùng", 338, 400.0),
_span("Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần.", 338, 412.0, bold=False),
_span("Liều lượng và cách dùng", 339, 200.0),
_span("Suy thận: giảm liều theo độ thanh thải creatinin.", 339, 212.0, bold=False),
]
m = list(assemble(spans))[0]
text = m.sections["lieu_luong_va_cach_dung"].text
assert "Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần." in text
assert "Suy thận: giảm liều theo độ thanh thải creatinin." in text
# the first heading stays the provenance anchor
assert m.sections["lieu_luong_va_cach_dung"].heading.physical_page == 338
def test_a_plain_label_line_under_a_heading_is_body_not_a_new_section():
"""FLUOROURACIL, physical page 681 — verified by rendering the page.
The book prints "Thời kỳ mang thai" / "Chống chỉ định." and "Thời kỳ cho
con bú" / "Chống chỉ định.". The body line matches the section vocabulary,
so it was read as a heading and both sections came out empty — dropping
the statement that fluorouracil is contraindicated in pregnancy and while
breastfeeding.
"""
spans = [
_span("FLUOROURACIL", 681, 60.0),
_span("Tên chung quốc tế", 681, 80.0),
_span("Fluorouracilum.", 681, 92.0, bold=False),
_span("Chống chỉ định", 681, 110.0),
_span("Suy tủy nặng.", 681, 122.0, bold=False),
_span("Thời kỳ mang thai", 681, 140.0),
_span("Chống chỉ định.", 681, 152.0, bold=False),
_span("Thời kỳ cho con bú", 681, 170.0),
_span("Chống chỉ định.", 681, 182.0, bold=False),
]
monograph = list(assemble(spans))[0]
assert monograph.sections["thoi_ky_mang_thai"].text == "Chống chỉ định."
assert monograph.sections["thoi_ky_cho_con_bu"].text == "Chống chỉ định."
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
def test_a_bold_label_line_still_opens_its_section():
spans = [
_span("FLUOROURACIL", 681, 60.0),
_span("Tên chung quốc tế", 681, 80.0),
_span("Fluorouracilum.", 681, 92.0, bold=False),
_span("Chống chỉ định", 681, 110.0),
_span("Suy tủy nặng.", 681, 122.0, bold=False),
]
monograph = list(assemble(spans))[0]
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
+144
View File
@@ -0,0 +1,144 @@
from ingestion.segment.atc import extract_atc_codes, is_stated_absent, normalize_atc_candidate
def test_stray_whitespace_split_j04a_c01_recovered():
assert normalize_atc_candidate("J04A C01") == "J04AC01"
def test_stray_whitespace_split_n05b_a06_recovered():
assert normalize_atc_candidate("N05B A06") == "N05BA06"
def test_stray_whitespace_split_l01x_x02_recovered():
assert normalize_atc_candidate("L01X X02") == "L01XX02"
def test_digit_letter_confusion_no3ax12_recovered():
assert normalize_atc_candidate("NO3AX12") == "N03AX12"
def test_digit_letter_confusion_jo1dc07_recovered():
assert normalize_atc_candidate("JO1DC07") == "J01DC07"
def test_clean_code_passes_through():
assert normalize_atc_candidate("N03AX12") == "N03AX12"
def test_garbage_not_recovered():
assert normalize_atc_candidate("NOT AN ATC CODE") is None
assert normalize_atc_candidate("") is None
def test_stated_absent_chua_co():
assert is_stated_absent("Mã ATC: Chưa có.") is True
def test_stated_absent_khong_co():
assert is_stated_absent("Không có.") is True
def test_stated_present_not_flagged_absent():
assert is_stated_absent("N03AX12") is False
def test_extract_single_code():
result = extract_atc_codes("N03AX12")
assert result.codes == ["N03AX12"]
assert result.stated_absent is False
def test_extract_multi_code_insulin_style():
result = extract_atc_codes("A10AB01, A10AC01, A10AD01")
assert result.codes == ["A10AB01", "A10AC01", "A10AD01"]
def test_extract_multi_code_with_noise_mixed_in():
# one clean code, one noisy code recovered, matching the real corpus
# pattern where a monograph has some clean and some noisy ATC entries
result = extract_atc_codes("N03AX12, J04A C01")
assert result.codes == ["N03AX12", "J04AC01"]
def test_extract_stated_absent_returns_no_codes():
result = extract_atc_codes("Mã ATC: Chưa có.")
assert result.codes == []
assert result.stated_absent is True
def test_trailing_period_recovered_confirmed_real_abacavir_case():
# real field text is "J05AF06." — a sentence-ending period, not part of
# the code; an earlier version silently produced zero codes here.
assert normalize_atc_candidate("J05AF06.") == "J05AF06"
result = extract_atc_codes("J05AF06.")
assert result.codes == ["J05AF06"]
def test_species_annotation_stripped_confirmed_real_insulin_case():
# annotation-stripping is extract_atc_codes's job (must run before the
# comma/semicolon split, see below) — normalize_atc_candidate itself
# only normalizes an already-isolated code token.
result = extract_atc_codes("A10AB01 (người); A10AB02 (bò)")
assert result.codes == ["A10AB01", "A10AB02"]
def test_leading_colon_from_value_span_stripped_confirmed_real_alcuronium_case():
# real field text for ALCURONIUM CLORID (physical page 152): the bold
# label span is "Mã ATC" with no colon, and the plain value span is
# ": M03AA01." — the colon belongs to the value side here, not the
# label side (Abacavir's equivalent has it on the label side instead:
# "Mã ATC: " + "J05AF06."). See atc.py module docstring, defect 5.
assert normalize_atc_candidate(": M03AA01.") == "M03AA01"
result = extract_atc_codes(": M03AA01.")
assert result.codes == ["M03AA01"]
def test_name_prefixed_code_stripped_confirmed_real_arginin_case():
# real field text for ARGININ (physical page 204): two salt forms, each
# its own "Name: CODE" line, not a bare code — see atc.py module
# docstring, defect 6.
assert normalize_atc_candidate("Arginin glutamat: A05BA01") == "A05BA01"
result = extract_atc_codes("Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01")
assert result.codes == ["A05BA01", "B05XB01"]
def test_plain_code_with_no_colon_still_normalizes():
assert normalize_atc_candidate("N03AX12") == "N03AX12"
def test_reversed_code_first_shape_confirmed_real_hmg_coa_case():
# real field text for CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE (physical page
# 284): each statin is "CODE: Name", the opposite order from the
# "Name: CODE" shape above — see atc.py module docstring, defect 7.
# "C10A A01" also has the already-fixed stray-whitespace split.
assert normalize_atc_candidate("C10A A01: Simvastatin") == "C10AA01"
result = extract_atc_codes("C10A A01: Simvastatin\nC10A A02: Lovastatin")
assert result.codes == ["C10AA01", "C10AA02"]
def test_annotation_containing_a_comma_does_not_break_the_split_confirmed_vaccine_case():
# real field text for VẮC XIN SỞI (physical page 1437): the English
# annotation "(Measles, live attenuated)" contains its own comma. An
# earlier version split on "," *before* stripping the annotation,
# breaking "J07BD01 (Measles, live attenuated)." into two unrecoverable
# fragments and silently returning zero codes — see atc.py module
# docstring, defect 4.
result = extract_atc_codes("J07BD01 (Measles, live attenuated).")
assert result.codes == ["J07BD01"]
def test_extract_all_20_insulin_codes_from_real_field_text():
# exact real field text for INSULIN (physical page 809) — see atc.py
# module docstring; confirms the fix recovers all 20, not just 2.
field_text = (
"A10AB01 (người); A10AB02 (bò); A10AB03 (lợn);\n"
"A10AB04 (lispro); A10AB05 (aspart); A10AB06 (glulisin);\n"
"A10AC01 (người); A10AC02 (bò); A10AC03 (lợn); A10AC04\n"
"(lispro); A10AD01 (người), A10AD02 (bò), A10AD03 (lợn),\n"
"A10AD04 (lispro), A10AE01 (người); A10AE02 (bò); A10AE03\n"
"(lợn); A10AE04 (glargin); A10AE05 (detemir), A10AF01 (người)."
)
result = extract_atc_codes(field_text)
assert len(result.codes) == 20
assert "A10AB01" in result.codes
assert "A10AF01" in result.codes
+88
View File
@@ -0,0 +1,88 @@
from ingestion.extract.models import Span
from ingestion.segment.detector import detect_monograph_titles, detect_section_headings
def _span(text, physical_page, printed_page, y0=100.0, bold=True, size=10.0):
font = "TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"
return Span(
physical_page=physical_page, printed_page=printed_page, column="left",
block=0, line=0, span_index=0,
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
text=text, font=font, size=size,
)
def test_confirmed_part_divider_excluded_at_page_99_boundary():
# "CÁC CHUYÊN LUẬN THUỐC" at physical page 98 / printed 99 — bold,
# all-caps, short: identical shape to a real title, must be excluded.
spans = [_span("CÁC CHUYÊN LUẬN THUỐC", 98, 99), _span("ABACAVIR", 100, 101)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["ABACAVIR"]
def test_monograph_title_outside_page_range_excluded():
# bold all-caps short text in front matter (e.g. an org name) must not
# be picked up — scoping to printed 99-1496 is required, not optional.
spans = [_span("BỘ Y TẾ", 2, 3), _span("ABACAVIR", 100, 101)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["ABACAVIR"]
def test_gonadotropin_wrap_detected_as_one_title():
spans = [
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 1372, y0=664.4554443359375),
_span("GONADOTROPIN", 1371, 1372, y0=676.2354736328125),
]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"]
def test_non_bold_all_caps_text_not_a_title_candidate():
spans = [_span("NOT BOLD BUT CAPS", 100, 101, bold=False)]
assert list(detect_monograph_titles(spans)) == []
def test_lowercase_bold_text_not_a_title_candidate():
spans = [_span("Abacavir", 100, 101)]
assert list(detect_monograph_titles(spans)) == []
def test_short_section_label_with_normal_diacritic_not_a_title_candidate():
# regression: an earlier absolute-count (not ratio) version of the
# mixed-case tolerance let "Mã ATC:" through as a false title candidate
# — its single lowercase diacritic ('ã') is normal Vietnamese
# orthography, not a HMG-CoA-style embedded abbreviation. A ratio
# threshold correctly rejects this short label (1/5 = 20% lowercase)
# while still accepting the long HMG-CoA title (1/27 = 3.7%).
spans = [_span("Mã ATC:", 100, 101)]
assert list(detect_monograph_titles(spans)) == []
def test_confirmed_hmg_coa_mixed_case_title_still_detected():
# "CoA" (Coenzyme A) is a real mixed-case abbreviation embedded in an
# otherwise all-caps title — outlier item 21. A strict isupper() check
# silently dropped this entire class-level monograph from the corpus.
spans = [_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 285)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"]
def test_section_heading_matched_with_and_without_trailing_colon():
spans = [
_span("Tên chung quốc tế:", 100, 101, bold=True, size=9.5),
_span("Chỉ định", 100, 101, bold=True, size=9.5),
]
headings = list(detect_section_headings(spans))
assert [h.section_key for h in headings] == ["ten_chung_quoc_te", "chi_dinh"]
def test_unknown_bold_text_not_matched_as_section():
# e.g. "Cách dùng:" — a real sub-heading within "Liều lượng và cách
# dùng" that is NOT one of the known top-level section names.
spans = [_span("Cách dùng:", 100, 101, bold=True, size=9.5)]
assert list(detect_section_headings(spans)) == []
def test_section_heading_outside_monograph_range_excluded():
spans = [_span("Chỉ định", 5, 6, bold=True, size=9.5)]
assert list(detect_section_headings(spans)) == []
+72
View File
@@ -0,0 +1,72 @@
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
from ingestion.segment.models import Heading, Monograph, SectionSpan
def test_round_trip_preserves_all_fields(tmp_path):
heading = Heading(text="Chỉ định", physical_page=100, y0=80.0, is_monograph_title=False, section_key="chi_dinh")
section = SectionSpan(key="chi_dinh", display_name="Chỉ định", heading=heading, text="Điều trị nhiễm HIV.")
monograph = Monograph(
drug_id="abacavir", drug_name="ABACAVIR", source_page_range=[100, 101],
sections={"chi_dinh": section}, atc_codes=["J05AF06"], atc_stated_absent=False,
)
path = tmp_path / "monographs.jsonl"
count = write_monographs_jsonl([monograph], path)
assert count == 1
result = list(read_monographs_jsonl(path))
assert len(result) == 1
r = result[0]
assert r.drug_id == "abacavir"
assert r.drug_name == "ABACAVIR"
assert r.source_page_range == [100, 101]
assert r.atc_codes == ["J05AF06"]
assert r.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
assert r.sections["chi_dinh"].heading.section_key == "chi_dinh"
def test_multiple_monographs_round_trip(tmp_path):
m1 = Monograph(drug_id="a", drug_name="A", source_page_range=[1, 2])
m2 = Monograph(drug_id="b", drug_name="B", source_page_range=[3, 4])
path = tmp_path / "monographs.jsonl"
write_monographs_jsonl([m1, m2], path)
result = list(read_monographs_jsonl(path))
assert [r.drug_id for r in result] == ["a", "b"]
def test_empty_write_produces_empty_file(tmp_path):
path = tmp_path / "monographs.jsonl"
count = write_monographs_jsonl([], path)
assert count == 0
assert list(read_monographs_jsonl(path)) == []
def test_table_blocks_survive_a_write_read_round_trip(tmp_path):
# the lifted table blocks were being computed in memory and then dropped
# at the file boundary — 148 blocks existed in the run summary but the
# JSONL had no "tables" key at all
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
heading = Heading(text="Liều lượng và cách dùng", physical_page=339, y0=200.0,
is_monograph_title=False, section_key="lieu_luong_va_cach_dung")
m = Monograph(
drug_id="cefamandol", drug_name="CEFAMANDOL", source_page_range=[338, 340],
sections={"lieu_luong_va_cach_dung": SectionSpan(
key="lieu_luong_va_cach_dung", display_name="Liều lượng và cách dùng",
heading=heading, text="Cách dùng ...")},
tables=[TableBlock(
table_id="p339_t0", shape="simple_table", physical_page=339,
bbox=[40.0, 380.0, 400.0, 620.0],
section_key="lieu_luong_va_cach_dung",
text="80 - 50 750 mg - 2 g, 6 giờ/lần.", quarantined=True)],
)
path = tmp_path / "m.jsonl"
write_monographs_jsonl([m], path)
back = list(read_monographs_jsonl(path))[0]
assert len(back.tables) == 1
t = back.tables[0]
assert t.table_id == "p339_t0"
assert t.physical_page == 339
assert t.bbox == [40.0, 380.0, 400.0, 620.0]
assert t.quarantined is True
assert "750 mg - 2 g" in t.text
+134
View File
@@ -0,0 +1,134 @@
from ingestion.extract.models import Span
from ingestion.segment.merge import merge_multiline_headings, merge_same_line_bold_fragments
def _span(text, page, y0, size=9.5, font="TimesNewRomanPS-BoldMT"):
return Span(
physical_page=page, printed_page=page + 1, column="right",
block=0, line=0, span_index=0,
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
text=text, font=font, size=size,
)
def test_confirmed_gonadotropin_wrap_merges_into_one_heading():
# exact bboxes from physical page 1371 (0-indexed) — see module docstring
candidates = [
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
_span("GONADOTROPIN", 1371, 676.2354736328125),
]
headings = list(merge_multiline_headings(candidates))
assert len(headings) == 1
assert headings[0].text == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
def test_unrelated_single_line_titles_on_different_pages_not_merged():
candidates = [
_span("GONADOTROPIN", 755, 200.0),
_span("HYDROCORTISON", 900, 300.0),
]
headings = list(merge_multiline_headings(candidates))
assert len(headings) == 2
assert [h.text for h in headings] == ["GONADOTROPIN", "HYDROCORTISON"]
def test_large_y_gap_on_same_page_not_merged():
# two genuinely separate single-line titles far apart on the same page
# (e.g. two short monographs stacked in one column) must not merge
candidates = [
_span("ATENOLOL", 219, 100.0),
_span("ATRACURIUM BESYLAT", 219, 500.0),
]
headings = list(merge_multiline_headings(candidates))
assert len(headings) == 2
def test_confirmed_aciclovir_same_line_split_merges_without_space():
# exact bboxes from physical page 113 (0-indexed), found by rendering the
# page to an image and reading it directly: "ACIC" (size 10.0) and
# "LOVIR" (size 9.5) are one word split into two spans on the same
# visual line — different font size, ~0.5pt y0 gap, near-zero x-gap.
# Must merge WITHOUT a space ("ACICLOVIR", not "ACIC LOVIR") — see
# module docstring.
candidates = [
_span("ACIC", 113, 515.1914672851562, size=10.0),
_span("LOVIR", 113, 515.7044677734375, size=9.5),
]
headings = list(merge_multiline_headings(candidates))
assert len(headings) == 1
assert headings[0].text == "ACICLOVIR"
def test_wrap_and_same_line_split_use_different_join_characters():
# a genuine line-wrap (large y-gap) still joins with a space even when
# font size differs, since size is no longer part of the merge decision
candidates = [
_span("FIRST LINE", 100, 200.0, size=10.0),
_span("SECOND LINE", 100, 212.0, size=9.5),
]
headings = list(merge_multiline_headings(candidates))
assert len(headings) == 1
assert headings[0].text == "FIRST LINE SECOND LINE"
def test_single_candidate_yields_one_heading():
headings = list(merge_multiline_headings([_span("ABACAVIR", 100, 60.29)]))
assert len(headings) == 1
assert headings[0].text == "ABACAVIR"
def test_empty_input_yields_nothing():
assert list(merge_multiline_headings([])) == []
def test_confirmed_ten_chung_quoc_te_diacritic_split_reassembles():
# exact fragments + y0 from physical page 759's "GUAIFENESIN" monograph,
# found via a whole-book `cli validate` run (the monograph was silently
# dropped because "Tên chung quốc tế" never matched the section
# vocabulary) and confirmed by rendering the page to an image: to a
# human reader the line looks completely normal, but PyMuPDF splits it
# into 5 spans around the diacritic characters — see module docstring.
fragments = [
_span("Tên chung qu", 759, 157.614),
_span("", 759, 157.33),
_span("c t", 759, 157.614),
_span("ế", 759, 157.33),
_span(": ", 759, 157.614),
]
merged = merge_same_line_bold_fragments(fragments)
assert len(merged) == 1
assert merged[0].text == "Tên chung quốc tế: "
def test_non_bold_spans_pass_through_unmerged():
fragments = [
_span("Guaifenesin", 759, 157.24, font="TimesNewRomanPSMT"),
_span(".", 759, 157.24, font="TimesNewRomanPSMT"),
]
merged = merge_same_line_bold_fragments(fragments)
assert len(merged) == 2
def test_bold_spans_on_different_lines_not_merged():
fragments = [_span("Chỉ định", 100, 200.0), _span("Chống chỉ định", 100, 220.0)]
merged = merge_same_line_bold_fragments(fragments)
assert len(merged) == 2
def test_merged_span_keeps_provenance_of_first_fragment():
fragments = [_span("Tên chung qu", 759, 157.614), _span("", 759, 157.33)]
merged = merge_same_line_bold_fragments(fragments)
assert merged[0].physical_page == 759
assert merged[0].printed_page == 760
assert merged[0].x0 == fragments[0].x0
assert merged[0].x1 == fragments[-1].x1
def test_single_bold_span_passes_through_unchanged():
fragments = [_span("ABACAVIR", 100, 60.29)]
merged = merge_same_line_bold_fragments(fragments)
assert merged == fragments
def test_empty_input_to_same_line_merge_yields_nothing():
assert merge_same_line_bold_fragments([]) == []
+108
View File
@@ -0,0 +1,108 @@
from ingestion.extract.models import Span
from ingestion.segment import assemble
from ingestion.tables import SHAPE_GRID_2D, SHAPE_SIMPLE, TableRegion, index_by_page
def _span(text, page, y0, *, bold=False, x0=50.0, block=0, line=0, column="left"):
return Span(
physical_page=page, printed_page=page + 1, column=column,
block=block, line=line, span_index=0,
x0=x0, y0=y0, x1=x0 + len(text) * 4.5, y1=y0 + 10,
text=text, font="Tiger-Bold" if bold else "Tiger", size=9.5,
)
def _monograph_spans(extra):
return [
_span("PARACETAMOL", 109, 60.0, bold=True),
_span("Tên chung quốc tế", 109, 80.0, bold=True),
_span("Paracetamolum.", 109, 92.0),
_span("Dạng thuốc và hàm lượng", 109, 200.0, bold=True),
] + extra
def test_table_spans_are_lifted_out_of_section_prose():
# real measured case: physical page 109's dosage-form table was being
# concatenated cell by cell into the section body
# ('Viên nén' + '1' + '1 - 4' + '8 - 12' + 'Viên nang tác' ...)
spans = _monograph_spans([
_span("Thuốc dùng đường uống.", 109, 220.0),
_span("Viên nén", 109, 400.0, block=5),
_span("1", 109, 400.0, block=5, x0=200.0),
_span("1 - 4", 109, 400.0, block=5, x0=260.0),
_span("Sau khi uống hấp thu nhanh.", 109, 600.0, block=9),
])
# the region must cover the table's first column too — it starts at the
# left margin, same x as body prose
region = TableRegion("p109_t0", 109, (40.0, 380.0, 400.0, 460.0), 3, 3, SHAPE_SIMPLE)
m = list(assemble(spans, table_index=index_by_page([region])))[0]
body = m.sections["dang_thuoc_va_ham_luong"].text
assert "Viên nén" not in body
assert "1 - 4" not in body
assert "Thuốc dùng đường uống." in body
assert "Sau khi uống hấp thu nhanh." in body
assert len(m.tables) == 1
block = m.tables[0]
assert block.table_id == "p109_t0"
assert "Viên nén" in block.text and "1 - 4" in block.text
assert block.section_key == "dang_thuoc_va_ham_luong"
assert block.physical_page == 109
# every multi-column table is quarantined until a real row/column
# reconstruction exists — its linearised text is not safe to cite as prose
assert block.quarantined is True
def test_without_a_region_map_behaviour_is_unchanged():
spans = _monograph_spans([
_span("Thuốc dùng đường uống.", 109, 220.0),
_span("Viên nén", 109, 400.0, block=5),
])
m = list(assemble(spans))[0]
assert m.tables == []
assert "Viên nén" in m.sections["dang_thuoc_va_ham_luong"].text
def test_2d_grid_block_is_quarantined():
# a 2D lookup grid's flattened text is meaningless without row/column
# headers (outlier item 7) — it must be marked, not silently embedded
spans = _monograph_spans([_span("0,52", 109, 400.0, block=5, x0=200.0)])
region = TableRegion("p109_t1", 109, (150.0, 380.0, 400.0, 460.0), 6, 5, SHAPE_GRID_2D)
m = list(assemble(spans, table_index=index_by_page([region])))[0]
assert len(m.tables) == 1
assert m.tables[0].quarantined is True
def test_non_table_regions_are_never_lifted():
# the 17 full-page false positives must not swallow a whole page of prose
spans = _monograph_spans([_span("Thuốc dùng đường uống.", 109, 220.0)])
region = TableRegion("p109_t0", 109, (0.0, 0.0, 595.3, 836.2), 1, 2,
"not_a_table_full_page")
m = list(assemble(spans, table_index=index_by_page([region])))[0]
assert m.tables == []
assert "Thuốc dùng đường uống." in m.sections["dang_thuoc_va_ham_luong"].text
def test_table_block_ids_stay_unique_when_a_section_resumes():
# a region flushed twice (section closes, then resumes) must not emit two
# blocks with the same table_id — provenance ids have to be unique
spans = [
_span("CEFAMANDOL", 339, 60.0, bold=True),
_span("Tên chung quốc tế", 339, 80.0, bold=True),
_span("Cefamandolum.", 339, 92.0),
_span("Liều lượng và cách dùng", 339, 200.0, bold=True),
_span("80 - 50", 339, 400.0, block=5),
_span("Liều lượng và cách dùng", 339, 500.0, bold=True),
_span("< 25 - 10", 339, 600.0, block=9),
]
region = TableRegion("p339_t0", 339, (40.0, 380.0, 400.0, 620.0), 5, 2, SHAPE_SIMPLE)
m = list(assemble(spans, table_index=index_by_page([region])))[0]
# table_id is deterministic per REGION, so two parts of one table share
# it on purpose; table_part_id is the unique key, derived from the first
# source span rather than a counter (a counter would renumber whenever
# anything upstream shifted, hiding rather than identifying a duplicate)
assert len({t.table_part_id for t in m.tables}) == len(m.tables)
assert {t.continuation_group for t in m.tables} == {"p339_t0"}
assert all(t.table_part_id.startswith("p339_t0@") for t in m.tables)
assert all(t.quarantined for t in m.tables)
+30
View File
@@ -0,0 +1,30 @@
from ingestion.segment.units import normalize_unit_token, validate_unit_tokens
def test_clean_unit_passes_through():
assert normalize_unit_token("mg") == "mg"
assert normalize_unit_token("mcg") == "mcg"
assert normalize_unit_token("mmol") == "mmol"
def test_stray_whitespace_split_recovered_by_analogy_to_atc():
assert normalize_unit_token("m g") == "mg"
assert normalize_unit_token("m cg") == "mcg"
def test_case_insensitive():
assert normalize_unit_token("MG") == "mg"
def test_unknown_token_not_recovered():
assert normalize_unit_token("xyz") is None
assert normalize_unit_token("") is None
def test_validate_unit_tokens_flags_only_bad_ones():
bad = validate_unit_tokens(["mg", "mcg", "xyz", "ml"])
assert bad == ["xyz"]
def test_validate_unit_tokens_empty_when_all_valid():
assert validate_unit_tokens(["mg", "mcg", "mmol"]) == []
+83
View File
@@ -0,0 +1,83 @@
from ingestion.segment.vocab import match_section, match_section_with_inline_value
def test_exact_label_match_with_trailing_colon():
d = match_section("Tên chung quốc tế:")
assert d is not None and d.key == "ten_chung_quoc_te"
def test_exact_label_match_without_trailing_colon():
d = match_section("Chỉ định")
assert d is not None and d.key == "chi_dinh"
def test_inline_value_combined_span_confirmed_real_amitriptylin_case():
# AMITRIPTYLIN's real "Mã ATC:" field is one non-bold span combining
# label and value: "Mã ATC: N06AA09." — see outlier item 20.
result = match_section_with_inline_value("Mã ATC: N06AA09.")
assert result is not None
section_def, value = result
assert section_def.key == "ma_atc"
assert value == "N06AA09."
def test_inline_value_not_matched_when_no_colon_follows():
assert match_section_with_inline_value("Mã ATC something else entirely") is None
def test_inline_value_does_not_confuse_plain_body_text():
assert match_section_with_inline_value("Bệnh nhân cần theo dõi chặt chẽ.") is None
def test_exact_match_takes_priority_over_prefix_for_label_only_span():
d = match_section("Mã ATC:")
assert d is not None and d.key == "ma_atc"
def test_real_spelling_variants_found_in_the_book_all_match():
# measured whole-corpus: 42 distinct near-miss heading strings, 542
# occurrences, none of which matched before aliases were added. The
# heaviest is "Thông tin qui chế" (469x) — the book prints "qui" where
# its own documented template says "quy", which cost 586 of 682
# monographs their thong_tin_quy_che section entirely.
from ingestion.segment.vocab import match_section
cases = {
"Thông tin qui chế": "thong_tin_quy_che",
"Thông tin về qui chế": "thong_tin_quy_che",
"Thông tin và quy chế": "thong_tin_quy_che",
"Mã ACT": "ma_atc",
"Chống chỉ đinh": "chong_chi_dinh",
"Thời kì mang thai": "thoi_ky_mang_thai",
"Thời kì cho con bú": "thoi_ky_cho_con_bu",
"Dược lí và cơ chế tác dụng": "duoc_ly_va_co_che_tac_dung",
"Hướng dẫn cách sử trí ADR": "huong_dan_xu_tri_adr",
"Quá liều và xử lý": "qua_lieu_va_xu_tri",
"Lọai thuốc": "loai_thuoc",
}
for text, expected_key in cases.items():
matched = match_section(text)
assert matched is not None, f"{text!r} should match a section"
assert matched.key == expected_key
def test_typesetting_noise_is_folded_without_needing_an_alias_each():
# missing/extra spaces and the Ð/Đ look-alike are handled by the lookup
# key, not enumerated per-variant
from ingestion.segment.vocab import match_section
assert match_section("Chỉđịnh").key == "chi_dinh"
assert match_section("Chống chỉđịnh").key == "chong_chi_dinh"
assert match_section("Độổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
assert match_section("Ðộ ổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
assert match_section("H ướng dẫn cách xử trí ADR").key == "huong_dan_xu_tri_adr"
assert match_section("Tư ơng kỵ").key == "tuong_ky"
assert match_section("Tác dụng khôngmong muốn (ADR)").key == "tac_dung_khong_mong_muon"
assert match_section("Thận trọng.").key == "than_trong"
def test_near_misses_that_are_not_sections_stay_unmatched():
# "Thể trọng" is body weight, not "Thận trọng" (caution) — a 0.84
# similarity that must NOT become an alias; the opioid string is a
# drug-specific sub-heading inside a section, not the section itself
from ingestion.segment.vocab import match_section
assert match_section("Thể trọng") is None
assert match_section("Tác dụng không mong muốn của opioid") is None
+107
View File
@@ -0,0 +1,107 @@
from ingestion.segment.models import Monograph
from ingestion.validation.back_index import GroundTruthEntry
from ingestion.validation.metrics import compute_recall_precision
def _mono(drug_id, drug_name, start_physical):
return Monograph(drug_id=drug_id, drug_name=drug_name, source_page_range=[start_physical, start_physical + 1])
def test_perfect_match_recall_and_precision_are_one():
monographs = [_mono("abacavir", "ABACAVIR", 100)]
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
assert result.precision == 1.0
assert result.matched_count == 1
def test_missed_ground_truth_entry_lowers_recall_not_precision():
monographs = [_mono("abacavir", "ABACAVIR", 100)]
ground_truth = [
GroundTruthEntry(name="Abacavir", printed_page=101),
GroundTruthEntry(name="Acarbose", printed_page=103),
]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 0.5
assert result.precision == 1.0
assert len(result.unmatched_ground_truth) == 1
assert result.unmatched_ground_truth[0].name == "Acarbose"
def test_spurious_detected_monograph_lowers_precision_not_recall():
monographs = [
_mono("abacavir", "ABACAVIR", 100),
_mono("cac_chuyen_luan_thuoc", "CÁC CHUYÊN LUẬN THUỐC", 98),
]
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
assert result.precision == 0.5
assert len(result.unmatched_detected) == 1
def test_page_tolerance_allows_small_offset():
monographs = [_mono("abacavir", "ABACAVIR", 100)]
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=103)] # +2 tolerance
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
def test_page_beyond_tolerance_does_not_match():
monographs = [_mono("abacavir", "ABACAVIR", 100)]
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=110)]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 0.0
def test_qualifier_suffixed_name_still_matches_base_ground_truth_name():
# SALBUTAMOL (Dùng trong hô hấp) should still match a ground-truth
# entry that just says "Salbutamol"
monographs = [_mono("salbutamol_dung_trong_ho_hap", "SALBUTAMOL (Dùng trong hô hấp)", 1261)]
ground_truth = [GroundTruthEntry(name="Salbutamol", printed_page=1262)]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
def test_empty_ground_truth_gives_zero_recall_not_error():
result = compute_recall_precision([_mono("a", "A", 1)], [])
assert result.recall == 0.0
def test_empty_monographs_gives_zero_precision_not_error():
result = compute_recall_precision([], [GroundTruthEntry(name="A", printed_page=1)])
assert result.precision == 0.0
assert result.recall == 0.0
def test_exact_match_preferred_over_substring_steal_confirmed_real_case():
# Confirmed real case from a whole-book `cli validate` run: "ISOSORBID"
# and "ISOSORBID DINITRAT" are two distinct, correctly-segmented
# monographs a page apart. A pure substring match lets the shorter name
# "steal" both ground-truth entries (it's a substring of the longer one
# too) via `next()`'s order-dependent first match, leaving the real
# "ISOSORBID DINITRAT" monograph spuriously unmatched even though an
# exact match for it exists.
monographs = [
_mono("isosorbid", "ISOSORBID", 844),
_mono("isosorbid_dinitrat", "ISOSORBID DINITRAT", 845),
]
ground_truth = [
GroundTruthEntry(name="Isosorbid", printed_page=845),
GroundTruthEntry(name="Isosorbid dinitrat", printed_page=846),
]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
assert result.precision == 1.0
assert len(result.unmatched_detected) == 0
def test_double_space_in_detected_name_still_matches_confirmed_real_case():
# confirmed real case from a whole-book `cli validate` run: "ALVERIN
# CITRAT" (double space) failed to match ground truth's single-spaced
# "Alverin citrat" under plain strip+upper comparison.
monographs = [_mono("alverin_citrat", "ALVERIN CITRAT", 171)]
ground_truth = [GroundTruthEntry(name="Alverin citrat", printed_page=172)]
result = compute_recall_precision(monographs, ground_truth)
assert result.recall == 1.0
@@ -0,0 +1,124 @@
from pathlib import Path
import pytest
from ingestion.extract import OutlinedTextRun
from ingestion.tables import TableRegion
from ingestion.validation import (
FRACTION_BAR_CANDIDATE,
HEADER_RULE,
RULE_FRAGMENT,
TABLE_FRAME,
TEXT_AS_VECTOR_OUTLINE,
UNCLASSIFIED,
PageContext,
ResidualRegion,
classify,
scan_page,
)
from ingestion.validation.residual_ink import FRACTION_BAR_CANDIDATE as BAR
PDF_PATH = Path(__file__).resolve().parents[1] / "data" / "raw" / (
"duoc-thu-quoc-gia-viet-nam-2018.pdf"
)
needs_pdf = pytest.mark.skipif(not PDF_PATH.exists(), reason="source PDF not present")
def _region(x0, y0, x1, y1, page=100, ink=500):
return ResidualRegion(physical_page=page, bbox=(x0, y0, x1, y1), ink_px=ink)
def test_running_header_rule_is_named_not_left_unclassified():
# measured on real pages: a ~516pt wide, 0pt tall rule at y≈48-52 appears
# on essentially every page of the book
assert classify(_region(36.0, 48.5, 552.0, 48.5)) == HEADER_RULE
def test_a_thin_bar_below_the_header_band_is_a_fraction_bar_candidate():
# NETILMICIN, physical page 1042: the Cockcroft-Gault fraction bar
assert classify(_region(97.9, 492.0, 286.5, 492.0)) == FRACTION_BAR_CANDIDATE
def test_ink_inside_a_known_table_region_is_a_table_frame_not_a_formula():
table = TableRegion(
table_id="p202_t0", physical_page=202, bbox=(299.0, 189.6, 552.4, 300.5),
n_rows=4, n_cols=3, shape="simple_table",
)
region = _region(299.0, 189.6, 552.4, 300.5, page=202)
assert classify(region, PageContext(tables=[table])) == TABLE_FRAME
# ...and the same geometry with no table map degrades to "look at it",
# never to a silent pass
assert classify(region) == UNCLASSIFIED
def test_a_wide_rule_outside_the_header_band_is_not_treated_as_a_header_rule():
assert classify(_region(36.0, 700.0, 552.0, 700.0)) == FRACTION_BAR_CANDIDATE
def test_a_tall_block_of_unaccounted_ink_stays_unclassified():
# a figure or an image of text must never be silently absorbed by a rule
assert classify(_region(100.0, 300.0, 400.0, 500.0)) == UNCLASSIFIED
def test_hairline_shorter_than_the_minimum_bar_width_is_a_rule_fragment():
# too short to be a fraction bar, too thin to be anything but a rule
assert classify(_region(100.0, 300.0, 105.0, 300.0)) == RULE_FRAGMENT
@needs_pdf
@pytest.mark.parametrize(
"page,expected_bar_width_pt",
[
(1042, 188.6), # NETILMICIN — Cockcroft-Gault
(202, 118.1), # AMPICILIN VÀ SULBACTAM — Cockcroft-Gault
],
)
def test_confirmed_2d_formula_bars_survive_the_span_mask(page, expected_bar_width_pt):
"""Regression fixture for the two visually confirmed corrupted formulas.
Both pages are reported as having zero tables by `pdfplumber` and zero by
`opendataloader-pdf`; the bar is only findable as ink. If the mask padding
is ever loosened again the bar disappears (at 1.0pt page 1042's bar
shrinks from 188.6pt to 9.1pt) — this test is what catches that.
"""
import fitz
doc = fitz.open(PDF_PATH)
bars = [
r for r in scan_page(doc[page])
if classify(r) == BAR and r.bbox[1] > 60.0
]
assert bars, f"no fraction-bar candidate found on physical page {page}"
assert max(b.width_pt for b in bars) == pytest.approx(expected_bar_width_pt, abs=1.0)
def test_vector_outlined_text_is_named_rather_than_left_unclassified():
# physical page 714 prints 17 lines of Gatifloxacin prose as filled paths;
# no text extractor returns them, so the gate must name the defect
line = OutlinedTextRun(
physical_page=714, bbox=(35.3, 75.8, 286.7, 84.4), path_items=1638,
)
region = _region(35.5, 76.0, 120.0, 84.0, page=714)
context = PageContext(outlined_runs=[line])
assert classify(region, context) == TEXT_AS_VECTOR_OUTLINE
# an untranscribed line must never be mistaken for recovered content
assert not line.is_transcribed
@needs_pdf
def test_outlined_text_lines_are_found_on_exactly_the_five_known_pages():
"""Whole-document regression: 51 outlined runs on 5 pages.
Cross-checked two ways at the time of writing — the drawing-shape scan
below, and independently by counting glyph-shaped leftovers in the
residual-ink mask, which found the same five pages.
"""
import fitz
from ingestion.extract import detect_outlined_text
lines = list(detect_outlined_text(fitz.open(PDF_PATH)))
by_page = {}
for line in lines:
by_page[line.physical_page] = by_page.get(line.physical_page, 0) + 1
assert by_page == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}