Fix every real lint finding and drop degenerate splice fragments
This commit is contained in:
@@ -0,0 +1,185 @@
|
||||
{
|
||||
"note": "2D (stacked-fraction) formula regions, every one confirmed by rendering the page and reading it. bbox is the fraction bar itself; numerator and denominator sit above and below it.",
|
||||
"verified_on": "2026-08-01",
|
||||
"method": "residual-ink fraction_bar_candidate, then visual inspection of all 23 candidates",
|
||||
"regions": [
|
||||
{
|
||||
"physical_page": 43,
|
||||
"bar_bbox": [
|
||||
133.44,
|
||||
308.16,
|
||||
222.72,
|
||||
308.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
430.08,
|
||||
603.36,
|
||||
465.6,
|
||||
603.36
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
362.88,
|
||||
704.64,
|
||||
386.4,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
400.32,
|
||||
704.64,
|
||||
422.88,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
456.96,
|
||||
704.64,
|
||||
480.0,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
505.44,
|
||||
704.64,
|
||||
522.72,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 202,
|
||||
"bar_bbox": [
|
||||
371.52,
|
||||
151.68,
|
||||
489.6,
|
||||
151.68
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 325,
|
||||
"bar_bbox": [
|
||||
445.92,
|
||||
562.08,
|
||||
544.32,
|
||||
562.56
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 325,
|
||||
"bar_bbox": [
|
||||
439.68,
|
||||
623.52,
|
||||
560.16,
|
||||
623.52
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 349,
|
||||
"bar_bbox": [
|
||||
384.48,
|
||||
336.0,
|
||||
491.52,
|
||||
336.48
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1042,
|
||||
"bar_bbox": [
|
||||
97.92,
|
||||
492.0,
|
||||
286.56,
|
||||
492.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1043,
|
||||
"bar_bbox": [
|
||||
94.56,
|
||||
536.16,
|
||||
140.16,
|
||||
536.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1043,
|
||||
"bar_bbox": [
|
||||
145.92,
|
||||
536.16,
|
||||
244.8,
|
||||
536.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1132,
|
||||
"bar_bbox": [
|
||||
63.36,
|
||||
498.72,
|
||||
239.52,
|
||||
498.72
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1402,
|
||||
"bar_bbox": [
|
||||
120.96,
|
||||
711.84,
|
||||
201.6,
|
||||
711.84
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1402,
|
||||
"bar_bbox": [
|
||||
180.48,
|
||||
770.4,
|
||||
205.92,
|
||||
770.4
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 147,
|
||||
"bar_bbox": [
|
||||
307.9,
|
||||
672.0,
|
||||
428.5,
|
||||
672.0
|
||||
],
|
||||
"source_prints_no_bar": true,
|
||||
"note": "ADENOSIN infusion-rate formula. The source page prints three plain lines with no fraction bar at all, so no geometric detector can find it — confirmed by rendering the region and reading it. Left in prose it reads as a multiplication chain. Quarantined on the strength of the reading, and flagged for human confirmation of the intended division."
|
||||
}
|
||||
],
|
||||
"rejected": [
|
||||
{
|
||||
"physical_page": 4,
|
||||
"reason": "decorative underlines on the Ministry decision page"
|
||||
},
|
||||
{
|
||||
"physical_page": 63,
|
||||
"reason": "ruled box around a treatment-protocol paragraph"
|
||||
},
|
||||
{
|
||||
"physical_page": 845,
|
||||
"reason": "table header cell border"
|
||||
},
|
||||
{
|
||||
"physical_page": 878,
|
||||
"reason": "table header cell border"
|
||||
},
|
||||
{
|
||||
"physical_page": 1667,
|
||||
"reason": "rule above the colophon on the last page"
|
||||
}
|
||||
],
|
||||
"recall_limit": "The fraction-bar signal cannot find a fraction the source never typeset. ADENOSIN (physical page 147) is one confirmed case, found only because a prose-leak gate matched its text. The true number of bar-less formulas in the book is UNMEASURED."
|
||||
}
|
||||
@@ -0,0 +1,671 @@
|
||||
{
|
||||
"note": "Text that exists in the PDF only as vector outlines. No extractor returns it (PyMuPDF, pdfplumber and opendataloader-pdf all omit it). Every 'text' value below is a transcription read off the rendered page, not extracted data.",
|
||||
"transcribed_on": "2026-08-01",
|
||||
"method": "ingestion.extract.detect_outlined_text located the runs; each run was rendered at 210-300 dpi and read directly",
|
||||
"confidence": "Full-line runs are read with high confidence. Single-glyph runs are Vietnamese diacritic characters dropped out of an otherwise-extracted line; the glyph identity is legible but these should still be spot-checked by a human before the corpus is treated as complete.",
|
||||
"runs": [
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
470.32,
|
||||
37.64,
|
||||
521.85,
|
||||
44.57
|
||||
],
|
||||
"path_items": 337,
|
||||
"text": "Gatifloxacin",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.31,
|
||||
75.76,
|
||||
286.72,
|
||||
84.41
|
||||
],
|
||||
"path_items": 1638,
|
||||
"text": "Nghiên cứu trên động vật, gatifloxacin gây ngộ độc cho thai.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.77,
|
||||
87.91,
|
||||
287.05,
|
||||
96.56
|
||||
],
|
||||
"path_items": 1714,
|
||||
"text": "Gatifloxacin chỉ sử dụng cho phụ nữ có thai khi lợi ích vượt trội so",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.67,
|
||||
127.07,
|
||||
551.33,
|
||||
137.58
|
||||
],
|
||||
"path_items": 1765,
|
||||
"text": "Thuốc kháng acid (antacid): Gatifloxacin bị giảm hấp thu khi sử",
|
||||
"extracted_line_it_belongs_to": "Do chưa biết thuốc có phân bố vào sữa mẹ khi dùng trên người hay ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.51,
|
||||
139.65,
|
||||
287.22,
|
||||
150.17
|
||||
],
|
||||
"path_items": 1831,
|
||||
"text": "không, cần thận trọng khi sử dụng gatifloxacin cho phụ nữ đang",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
151.2,
|
||||
550.66,
|
||||
161.71
|
||||
],
|
||||
"path_items": 1787,
|
||||
"text": "cần dùng gatifloxacin ít nhất 4 giờ trước khi dùng các antacid này.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.44,
|
||||
177.2,
|
||||
455.19,
|
||||
185.84
|
||||
],
|
||||
"path_items": 1126,
|
||||
"text": "học có ý nghĩa lâm sàng với gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.66,
|
||||
223.59,
|
||||
397.93,
|
||||
234.1
|
||||
],
|
||||
"path_items": 792,
|
||||
"text": "giảm hấp thu gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": "Mắt: Chứng sưng viêm mi mắt, xuất huyết kết mạc, rát kết mạc, ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.66,
|
||||
247.72,
|
||||
551.32,
|
||||
258.23
|
||||
],
|
||||
"path_items": 1731,
|
||||
"text": "giữa warfarin và gatifloxacin, nhưng do một số quinolon có khả",
|
||||
"extracted_line_it_belongs_to": "khô mắt, phù, rát, viêm giác mạc, giảm thị lực, kích ứng kết mạc.",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
321.98,
|
||||
363.88,
|
||||
330.62
|
||||
],
|
||||
"path_items": 480,
|
||||
"text": "của gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": "Thần kinh: Căng thẳng, kích động, lo lắng, mất ngủ, hoa mắt, giấc ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.78,
|
||||
441.28,
|
||||
287.38,
|
||||
451.79
|
||||
],
|
||||
"path_items": 1778,
|
||||
"text": "Cần ngừng gatifloxacin trong các trường hợp: Bắt đầu có các biểu",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
452.82,
|
||||
461.52,
|
||||
463.34
|
||||
],
|
||||
"path_items": 1115,
|
||||
"text": "Gatifloxacin dùng với các thuốc làm thay đ",
|
||||
"extracted_line_it_belongs_to": "hiện ban da hoặc bất kỳ dấu hiệu nào của phản ứng quá mẫn, có ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
461.95,
|
||||
452.82,
|
||||
466.05,
|
||||
461.42
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "i nồng độ glucose máu ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.72,
|
||||
503.9,
|
||||
81.99,
|
||||
512.55
|
||||
],
|
||||
"path_items": 373,
|
||||
"text": "gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
313.73,
|
||||
506.23,
|
||||
317.83,
|
||||
514.82
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
542.15,
|
||||
506.23,
|
||||
546.25,
|
||||
514.82
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
450.53,
|
||||
520.35,
|
||||
455.28,
|
||||
526.89
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "định trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
532.42,
|
||||
304.45,
|
||||
538.95
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": " nhiệt độ 2 - 8 oC. Dung dịch pha loãng này (trừ pha trong natri ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
385.77,
|
||||
542.42,
|
||||
389.87,
|
||||
551.01
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
36.09,
|
||||
543.49,
|
||||
287.58,
|
||||
554.0
|
||||
],
|
||||
"path_items": 1809,
|
||||
"text": "Ghi chú: Đối với gatifloxacin dạng viên và dạng tiêm, nhà sản xuất",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
513.35,
|
||||
544.48,
|
||||
518.1,
|
||||
551.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
522.12,
|
||||
554.48,
|
||||
526.22,
|
||||
563.08
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "-25 đến -10 oC, sau khi đưa ra khỏi tủ lạnh sâu, tiếp tục n định ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
426.75,
|
||||
568.61,
|
||||
431.5,
|
||||
575.14
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
525.62,
|
||||
568.61,
|
||||
530.37,
|
||||
575.14
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.47,
|
||||
619.81,
|
||||
551.17,
|
||||
630.32
|
||||
],
|
||||
"path_items": 1692,
|
||||
"text": "Vì có rất ít các thông tin về tương ky của gatifloxacin, nên không",
|
||||
"extracted_line_it_belongs_to": "Tiêm truyền tĩnh mạch dưới dạng dung dịch 2 mg/ml trong 60 phút.",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
105.83,
|
||||
628.89,
|
||||
109.93,
|
||||
637.14
|
||||
],
|
||||
"path_items": 39,
|
||||
"text": "ỗ",
|
||||
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.48,
|
||||
631.87,
|
||||
551.03,
|
||||
642.39
|
||||
],
|
||||
"path_items": 1739,
|
||||
"text": "thêm bất kỳ một thuốc nào khác vào dịch truyền gatifloxacin hoặc",
|
||||
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
391.91,
|
||||
685.48,
|
||||
396.01,
|
||||
693.72
|
||||
],
|
||||
"path_items": 39,
|
||||
"text": "ỗ",
|
||||
"extracted_line_it_belongs_to": "triệu chứng và điều trị h trợ, bao gồm: Gây nôn và rửa dạ dày để ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
223.28,
|
||||
713.6,
|
||||
227.39,
|
||||
722.19
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
167.4,
|
||||
715.66,
|
||||
172.14,
|
||||
722.19
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.72,
|
||||
786.64,
|
||||
551.3,
|
||||
797.16
|
||||
],
|
||||
"path_items": 1687,
|
||||
"text": "Gatifloxacin thuộc Danh mục nguyên liệu và thuốc thành phẩm",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
448.56,
|
||||
88.17,
|
||||
453.3,
|
||||
94.7
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "không màu, đóng kín tránh ánh sáng điều kiện lạnh 2 - 8 oC; ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
266.53,
|
||||
98.19,
|
||||
270.63,
|
||||
106.79
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
282.68,
|
||||
100.26,
|
||||
287.43,
|
||||
106.79
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
129.96,
|
||||
122.5,
|
||||
209.16,
|
||||
133.01
|
||||
],
|
||||
"path_items": 580,
|
||||
"text": "nh tổn thương dây th",
|
||||
"extracted_line_it_belongs_to": "vào vùng cơ mông để trá",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
172.81,
|
||||
221.76,
|
||||
177.56,
|
||||
228.3
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "Liều thường dùng của GMDCUV người lớn và trẻ em để dự ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
91.42,
|
||||
280.45,
|
||||
95.52,
|
||||
289.05
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "tiêm các liều b sung với các khoảng cách là 4 tuần.",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
157.23,
|
||||
332.32,
|
||||
161.98,
|
||||
338.85
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "lại. Liều thông thường HTCUV người lớn và trẻ em để dự phòng ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
192.28,
|
||||
369.96,
|
||||
197.03,
|
||||
376.5
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "chậm trễ trong bắt đầu tiêm phòng hoặc người có thể trọng quá ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
399.06,
|
||||
424.38,
|
||||
403.8,
|
||||
430.91
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "huyết thanh của người trư ng thành khỏe mạnh đã được tạo miễn ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
187.41,
|
||||
513.01,
|
||||
192.16,
|
||||
519.54
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "GMDCUV hoặc HTCUV không ảnh hư ng tới đáp ứng miễn dịch ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
280.17,
|
||||
625.95,
|
||||
284.91,
|
||||
632.48
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "miễn dịch đối với một vài loại vắc xin virus sống (vắc xin virus s i ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
282.99,
|
||||
699.18,
|
||||
287.09,
|
||||
707.78
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "dịch hoặc huyết thanh ngựa thì nên dùng thêm một liều vắc xin b ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
419.76,
|
||||
726.48,
|
||||
424.51,
|
||||
733.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
442.01,
|
||||
726.48,
|
||||
446.75,
|
||||
733.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
305.51,
|
||||
737.0,
|
||||
309.61,
|
||||
745.6
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Do các chế phẩm có chứa globulin miễn dịch không có biểu hiện ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
62.37,
|
||||
751.45,
|
||||
67.12,
|
||||
757.98
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "ảnh hư ng tới các đáp ứng miễn dịch của vắc xin uống virus bại ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 1373,
|
||||
"bbox": [
|
||||
43.81,
|
||||
98.58,
|
||||
295.66,
|
||||
109.1
|
||||
],
|
||||
"path_items": 1458,
|
||||
"text": "Nếu phối hợp với flutamid ở giai đoạn T2b - T4 (B2 - C), điều trị",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1444,
|
||||
"bbox": [
|
||||
35.72,
|
||||
136.22,
|
||||
287.35,
|
||||
146.74
|
||||
],
|
||||
"path_items": 1731,
|
||||
"text": "Trimovax (Sanofi Pasteur): Một liều vắc xin chứa virus sống giảm",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1445,
|
||||
"bbox": [
|
||||
308.28,
|
||||
114.39,
|
||||
559.87,
|
||||
123.03
|
||||
],
|
||||
"path_items": 1440,
|
||||
"text": "(Typhoid, inactivated, whole cell), J07AP03 (Typhoid, purified",
|
||||
"extracted_line_it_belongs_to": "thể xảy ra 5 ngày sau khi tiêm: Sốt (có thể dự phòng bằng các loại ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1445,
|
||||
"bbox": [
|
||||
115.46,
|
||||
646.46,
|
||||
208.24,
|
||||
655.11
|
||||
],
|
||||
"path_items": 660,
|
||||
"text": "Haemophilus influenzae",
|
||||
"extracted_line_it_belongs_to": "khác như vắc xin ",
|
||||
"single_glyph": false
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
from .chunker import chunk_all, chunk_monograph, chunk_section, estimate_tokens
|
||||
from .io import read_monographs_jsonl, write_chunks_jsonl
|
||||
from .models import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
SCHEMA_VERSION,
|
||||
Chunk,
|
||||
ChunkAttachment,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"Chunk",
|
||||
"ChunkAttachment",
|
||||
"SCHEMA_VERSION",
|
||||
"CHUNK_KIND_PROSE",
|
||||
"CHUNK_KIND_BLOCK_DESCRIPTOR",
|
||||
"chunk_all",
|
||||
"chunk_monograph",
|
||||
"chunk_section",
|
||||
"estimate_tokens",
|
||||
"read_monographs_jsonl",
|
||||
"write_chunks_jsonl",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Section -> chunk logic (pure; no filesystem, no embedding client).
|
||||
|
||||
ADR 0004: chunk unit is `(drug_id, section_key)`. A section under the token
|
||||
ceiling becomes one chunk verbatim. Only the long-tail sections above it are
|
||||
sub-chunked, with a sentence-boundary-aware sliding window.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, Iterable, Iterator, List, Sequence
|
||||
|
||||
from ..segment.models import Monograph, SectionSpan, TableBlock
|
||||
from ..tables.classify import SHAPE_FORMULA_2D, SHAPE_SIMPLE
|
||||
from .models import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
Chunk,
|
||||
ChunkAttachment,
|
||||
)
|
||||
from .sentences import split_sentences
|
||||
|
||||
CEILING_TOKENS = 800
|
||||
TARGET_TOKENS = 650
|
||||
OVERLAP_TOKENS = 65
|
||||
|
||||
# Physical -> printed page. Empirically constant across every tested
|
||||
# milestone page (extract/page_map.py, ADR 0003); the descriptor quotes the
|
||||
# printed number because that is what a reader holding the book looks for.
|
||||
PRINTED_PAGE_OFFSET = 1
|
||||
|
||||
KIND_TABLE = "table"
|
||||
KIND_FORMULA = "formula"
|
||||
|
||||
# A header row is only safe to embed when it is genuinely a row of labels.
|
||||
# Measured on the corpus: 42 of 124 simple-table headers (34%) contain a
|
||||
# digit, and AMIODARON's (physical page 183) is
|
||||
# "Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)" — a dose,
|
||||
# inside what pdfplumber called a header, from an extraction never verified by
|
||||
# eye. A label carrying no digit cannot be mistaken for a dose; a long cell is
|
||||
# content rather than a label.
|
||||
_DIGIT = re.compile(r"\d")
|
||||
HEADER_CELL_MAX_CHARS = 40
|
||||
|
||||
|
||||
def _is_label_row(cells: Sequence[str]) -> bool:
|
||||
kept = [c for c in cells if c and c.strip()]
|
||||
if not kept:
|
||||
return False
|
||||
return all(
|
||||
not _DIGIT.search(cell) and len(cell.strip()) <= HEADER_CELL_MAX_CHARS
|
||||
for cell in kept
|
||||
)
|
||||
|
||||
|
||||
def estimate_tokens(text: str) -> int:
|
||||
"""ADR 0004's chars/4 estimate — an estimate, not a tokenizer count."""
|
||||
return len(text) // 4
|
||||
|
||||
|
||||
def _pack(sentences: List[str]) -> List[List[str]]:
|
||||
"""Greedily pack sentences up to TARGET_TOKENS, overlapping by OVERLAP_TOKENS.
|
||||
|
||||
A single sentence longer than the target becomes its own part rather than
|
||||
being cut mid-sentence — the caller flags it instead of splitting it.
|
||||
"""
|
||||
parts: List[List[str]] = []
|
||||
current: List[str] = []
|
||||
current_tokens = 0
|
||||
|
||||
for sentence in sentences:
|
||||
tokens = estimate_tokens(sentence)
|
||||
if current and current_tokens + tokens > TARGET_TOKENS:
|
||||
parts.append(current)
|
||||
overlap: List[str] = []
|
||||
acc = 0
|
||||
for prev in reversed(current):
|
||||
overlap.insert(0, prev)
|
||||
acc += estimate_tokens(prev)
|
||||
if acc >= OVERLAP_TOKENS:
|
||||
break
|
||||
current = list(overlap)
|
||||
current_tokens = sum(estimate_tokens(s) for s in current)
|
||||
current.append(sentence)
|
||||
current_tokens += tokens
|
||||
|
||||
if current:
|
||||
parts.append(current)
|
||||
return parts
|
||||
|
||||
|
||||
def _block_kind(block: TableBlock) -> str:
|
||||
return KIND_FORMULA if block.shape == SHAPE_FORMULA_2D else KIND_TABLE
|
||||
|
||||
|
||||
def _attachment(block: TableBlock, header_row: List[str]) -> ChunkAttachment:
|
||||
return ChunkAttachment(
|
||||
block_id=block.table_id,
|
||||
kind=_block_kind(block),
|
||||
shape=block.shape,
|
||||
physical_page=block.physical_page,
|
||||
bbox=list(block.bbox),
|
||||
quarantined=block.quarantined,
|
||||
# Only a simple table's first row can be a row of plain labels, and
|
||||
# only when it actually reads like one. A multi-level or merged header
|
||||
# is the shape whose extraction is least trustworthy, so it
|
||||
# contributes nothing rather than something wrong.
|
||||
header_row=(list(header_row)
|
||||
if block.shape == SHAPE_SIMPLE and _is_label_row(header_row)
|
||||
else []),
|
||||
)
|
||||
|
||||
|
||||
def _blocks_by_section(monograph: Monograph) -> Dict[str, List[TableBlock]]:
|
||||
grouped: Dict[str, List[TableBlock]] = {}
|
||||
for block in monograph.tables:
|
||||
if block.section_key:
|
||||
grouped.setdefault(block.section_key, []).append(block)
|
||||
return grouped
|
||||
|
||||
|
||||
def describe_block(monograph: Monograph, section: SectionSpan,
|
||||
attachment: ChunkAttachment) -> str:
|
||||
"""Retrieval text for a block, built only from metadata.
|
||||
|
||||
No cell value ever appears here. A header row is a row of labels;
|
||||
linearising it cannot invent a numeric relationship, which is exactly what
|
||||
linearising a body row does.
|
||||
"""
|
||||
noun = "công thức" if attachment.kind == KIND_FORMULA else "bảng"
|
||||
printed = attachment.physical_page + PRINTED_PAGE_OFFSET
|
||||
text = (f"{monograph.drug_name} — {section.display_name} — {noun}, "
|
||||
f"trang {printed}.")
|
||||
if attachment.header_row:
|
||||
columns = " | ".join(c.replace("\n", " ").strip()
|
||||
for c in attachment.header_row if c and c.strip())
|
||||
if columns:
|
||||
text += f" Cột: {columns}."
|
||||
text += (" Nội dung chỉ tra cứu được trên ảnh trang gốc, "
|
||||
"không trích dẫn được dưới dạng văn bản.")
|
||||
return text
|
||||
|
||||
|
||||
def chunk_section(monograph: Monograph, section: SectionSpan,
|
||||
blocks: Sequence[TableBlock] = (),
|
||||
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
|
||||
header_rows = header_rows or {}
|
||||
attachments = [_attachment(b, header_rows.get(b.table_id, [])) for b in blocks]
|
||||
quarantined = any(a.quarantined for a in attachments)
|
||||
|
||||
def build(body: str, part_index: int, part_count: int) -> Chunk:
|
||||
tokens = estimate_tokens(body)
|
||||
return Chunk(
|
||||
chunk_id=f"{monograph.drug_id}__{section.key}__{part_index}",
|
||||
drug_id=monograph.drug_id,
|
||||
drug_name=monograph.drug_name,
|
||||
section_key=section.key,
|
||||
section_display_name=section.display_name,
|
||||
text=body,
|
||||
heading_physical_page=section.heading.physical_page,
|
||||
source_page_range=list(monograph.source_page_range),
|
||||
atc_codes=list(monograph.atc_codes),
|
||||
part_index=part_index,
|
||||
part_count=part_count,
|
||||
est_tokens=tokens,
|
||||
oversized=tokens > CEILING_TOKENS,
|
||||
chunk_kind=CHUNK_KIND_PROSE,
|
||||
attachments=list(attachments),
|
||||
has_quarantined_content=quarantined,
|
||||
)
|
||||
|
||||
text = section.text.strip()
|
||||
prose: List[Chunk] = []
|
||||
if text:
|
||||
if estimate_tokens(text) <= CEILING_TOKENS:
|
||||
prose = [build(text, 0, 1)]
|
||||
else:
|
||||
parts = _pack(split_sentences(text))
|
||||
bodies = [b for b in ("".join(p).strip() for p in parts) if b]
|
||||
prose = [build(b, i, len(bodies)) for i, b in enumerate(bodies)]
|
||||
|
||||
descriptors = []
|
||||
for attachment in attachments:
|
||||
body = describe_block(monograph, section, attachment)
|
||||
descriptors.append(Chunk(
|
||||
chunk_id=f"{monograph.drug_id}__{section.key}__block__{attachment.block_id}",
|
||||
drug_id=monograph.drug_id,
|
||||
drug_name=monograph.drug_name,
|
||||
section_key=section.key,
|
||||
section_display_name=section.display_name,
|
||||
text=body,
|
||||
heading_physical_page=section.heading.physical_page,
|
||||
source_page_range=list(monograph.source_page_range),
|
||||
atc_codes=list(monograph.atc_codes),
|
||||
est_tokens=estimate_tokens(body),
|
||||
chunk_kind=CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
attachments=[attachment],
|
||||
has_quarantined_content=attachment.quarantined,
|
||||
))
|
||||
return prose + descriptors
|
||||
|
||||
|
||||
def chunk_monograph(monograph: Monograph,
|
||||
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
|
||||
grouped = _blocks_by_section(monograph)
|
||||
chunks: List[Chunk] = []
|
||||
for section in monograph.sections.values():
|
||||
chunks.extend(chunk_section(monograph, section,
|
||||
grouped.get(section.key, ()), header_rows))
|
||||
return chunks
|
||||
|
||||
|
||||
def chunk_all(monographs: Iterable[Monograph],
|
||||
header_rows: Dict[str, List[str]] | None = None) -> Iterator[Chunk]:
|
||||
for monograph in monographs:
|
||||
yield from chunk_monograph(monograph, header_rows)
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Filesystem boundary for the chunk stage — kept out of the pure logic."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from ..segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from .models import SCHEMA_VERSION, Chunk
|
||||
|
||||
|
||||
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
|
||||
with path.open(encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
raw = json.loads(line)
|
||||
sections = {}
|
||||
for key, s in raw.get("sections", {}).items():
|
||||
h = s["heading"]
|
||||
sections[key] = SectionSpan(
|
||||
key=s["key"],
|
||||
display_name=s["display_name"],
|
||||
heading=Heading(
|
||||
text=h["text"],
|
||||
physical_page=h["physical_page"],
|
||||
y0=h["y0"],
|
||||
is_monograph_title=h["is_monograph_title"],
|
||||
section_key=h.get("section_key"),
|
||||
),
|
||||
text=s["text"],
|
||||
)
|
||||
yield Monograph(
|
||||
drug_id=raw["drug_id"],
|
||||
drug_name=raw["drug_name"],
|
||||
source_page_range=raw["source_page_range"],
|
||||
sections=sections,
|
||||
atc_codes=raw.get("atc_codes", []),
|
||||
atc_stated_absent=raw.get("atc_stated_absent", False),
|
||||
tables=[
|
||||
TableBlock(
|
||||
table_id=t["table_id"],
|
||||
shape=t["shape"],
|
||||
physical_page=t["physical_page"],
|
||||
bbox=t["bbox"],
|
||||
section_key=t.get("section_key"),
|
||||
text=t.get("text", ""),
|
||||
quarantined=t.get("quarantined", False),
|
||||
table_part_id=t.get("table_part_id"),
|
||||
continuation_group=t.get("continuation_group"),
|
||||
source_span_ids=t.get("source_span_ids", []),
|
||||
)
|
||||
for t in raw.get("tables", [])
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def write_chunks_jsonl(chunks: Iterable[Chunk], path: Path) -> int:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
count = 0
|
||||
with path.open("w", encoding="utf-8") as fh:
|
||||
for chunk in chunks:
|
||||
# ADR 0005 flagged the absence of a version and ADR 0006 made it
|
||||
# necessary: the record now has two chunk kinds and an attachment
|
||||
# list, so a consumer must be able to tell which shape it has.
|
||||
record = {"schema_version": SCHEMA_VERSION, **asdict(chunk)}
|
||||
fh.write(json.dumps(record, ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Chunk record — the unit handed to embedding/indexing.
|
||||
|
||||
Provenance fields follow ADR 0004 and CLAUDE.md's provenance rule: a chunk
|
||||
must carry enough to trace it back to a monograph, a section, and the page
|
||||
its section heading was found on.
|
||||
|
||||
ADR 0006 adds attachments. `segment/` lifts tables and 2D formulas out of
|
||||
section prose because linearising them is actively wrong — AMPICILIN VÀ
|
||||
SULBACTAM's Cockcroft-Gault fraction read as `Clcr (ml/phút) = 72 x
|
||||
creatinin huyết thanh`, a division presented as a multiplication in a
|
||||
renal-dosing section. Without a reference back, a chunk of that section is
|
||||
grammatical, complete-looking prose with the dosing table silently absent.
|
||||
Measured: 127 of 167 lifted blocks (76%) came out of `liều lượng và cách
|
||||
dùng`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List
|
||||
|
||||
SCHEMA_VERSION = 2
|
||||
|
||||
CHUNK_KIND_PROSE = "prose"
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR = "block_descriptor"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ChunkAttachment:
|
||||
"""A table or formula that was lifted out of this chunk's section.
|
||||
|
||||
`bbox` + `physical_page` are what let the answer layer render the source
|
||||
crop, which for a quarantined block is the only faithful answer available.
|
||||
"""
|
||||
|
||||
block_id: str
|
||||
kind: str # "table" | "formula"
|
||||
shape: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
quarantined: bool = True
|
||||
# First row of a `simple_table`, used to make the block findable. Comes
|
||||
# from pdfplumber and has NOT been verified by eye — the 180 real tables'
|
||||
# shapes are rule-derived. Retrieval bait, never an answer.
|
||||
header_row: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Chunk:
|
||||
chunk_id: str
|
||||
drug_id: str
|
||||
drug_name: str
|
||||
section_key: str
|
||||
section_display_name: str
|
||||
text: str
|
||||
heading_physical_page: int
|
||||
source_page_range: List[int]
|
||||
atc_codes: List[str] = field(default_factory=list)
|
||||
part_index: int = 0
|
||||
part_count: int = 1
|
||||
est_tokens: int = 0
|
||||
oversized: bool = False
|
||||
chunk_kind: str = CHUNK_KIND_PROSE
|
||||
attachments: List[ChunkAttachment] = field(default_factory=list)
|
||||
# Derivable from `attachments`, stored anyway: the defect this schema
|
||||
# exists to prevent is a consumer not knowing what it was not told.
|
||||
has_quarantined_content: bool = False
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Vietnamese sentence-boundary splitting for medical formulary text.
|
||||
|
||||
ADR 0004 requires splitting at sentence boundaries rather than a blind
|
||||
character window: `segment/assembler.py` joins body lines at PDF visual
|
||||
line-wrap points, so a character window can land mid-sentence — and outlier
|
||||
item 17 measured adult/child dosing sentences ("Người lớn"/"Trẻ em") on
|
||||
1,121 of ~1,400 monograph pages, where a mid-sentence cut is a
|
||||
patient-safety defect rather than a cosmetic one.
|
||||
|
||||
Boundary rule: `.`, `;`, `:`, `?` or `!` followed by whitespace and an
|
||||
opening character (uppercase letter or digit), minus the exclusions below.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List
|
||||
|
||||
_TERMINATORS = ".;:?!"
|
||||
|
||||
# Tokens that end in '.' but do not end a sentence.
|
||||
_ABBREVIATIONS = frozenset({
|
||||
"v.v", "vv", "tr", "tp", "ts", "bs", "gs", "pgs", "ths", "dr", "st",
|
||||
"no", "nxb", "cs", "kg", "mg", "ml", "mcg", "gr", "hb", "tm", "tb",
|
||||
})
|
||||
|
||||
_OPENS_SENTENCE = re.compile(r"[A-ZÀ-Ỹ0-9(\-–]")
|
||||
_TRAILING_TOKEN = re.compile(r"([\wÀ-ỹ.]+)\.$")
|
||||
|
||||
|
||||
def _is_abbreviation(left: str) -> bool:
|
||||
m = _TRAILING_TOKEN.search(left.rstrip())
|
||||
if not m:
|
||||
return False
|
||||
token = m.group(1).rstrip(".").lower()
|
||||
if token in _ABBREVIATIONS:
|
||||
return True
|
||||
# single letter -> an initial ("P." in a name), not a sentence end
|
||||
return len(token) == 1 and token.isalpha()
|
||||
|
||||
|
||||
def _is_decimal_or_numbering(text: str, i: int) -> bool:
|
||||
"""A period/comma sitting between digits, or a list numbering like '1. '."""
|
||||
if text[i] != ".":
|
||||
return False
|
||||
prev_ch = text[i - 1] if i > 0 else ""
|
||||
next_ch = text[i + 1] if i + 1 < len(text) else ""
|
||||
if prev_ch.isdigit() and next_ch.isdigit():
|
||||
return True
|
||||
# "1." / "12." starting a numbered list item: digits preceded by start/newline
|
||||
j = i - 1
|
||||
while j >= 0 and text[j].isdigit():
|
||||
j -= 1
|
||||
if j < i - 1 and (j < 0 or text[j] in "\n \t("):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def split_sentences(text: str) -> List[str]:
|
||||
"""Split into sentence-ish units, preserving all characters.
|
||||
|
||||
Concatenating the result (without added separators) reproduces the input
|
||||
exactly — no character is dropped, which the coverage ledger depends on.
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
|
||||
out: List[str] = []
|
||||
start = 0
|
||||
i = 0
|
||||
n = len(text)
|
||||
while i < n:
|
||||
ch = text[i]
|
||||
if ch not in _TERMINATORS:
|
||||
i += 1
|
||||
continue
|
||||
if _is_decimal_or_numbering(text, i):
|
||||
i += 1
|
||||
continue
|
||||
|
||||
j = i + 1
|
||||
if j < n and text[j] in "\")]”’":
|
||||
j += 1
|
||||
ws_start = j
|
||||
while j < n and text[j].isspace():
|
||||
j += 1
|
||||
if j == ws_start or j >= n:
|
||||
i += 1
|
||||
continue
|
||||
if not _OPENS_SENTENCE.match(text[j]):
|
||||
i += 1
|
||||
continue
|
||||
if ch == "." and _is_abbreviation(text[start:i + 1]):
|
||||
i += 1
|
||||
continue
|
||||
|
||||
out.append(text[start:j])
|
||||
start = j
|
||||
i = j
|
||||
|
||||
if start < n:
|
||||
out.append(text[start:])
|
||||
return out
|
||||
@@ -0,0 +1,516 @@
|
||||
"""CLI entry point: `python -m ingestion.cli <subcommand>`.
|
||||
|
||||
`run` is the real Phase 1 pipeline (extract -> segment -> write). `validate`,
|
||||
`visual-diff`, and `scaffold-golden` are Phase 1.4-1.7 work — declared here
|
||||
now (per the approved plan's CLI contract) but not yet implemented; they
|
||||
raise `NotImplementedError` explicitly rather than silently no-op-ing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
from .extract import (
|
||||
extract_spans,
|
||||
load_transcribed_runs,
|
||||
merge_outlined_runs,
|
||||
index_formula_regions_by_page,
|
||||
load_formula_regions,
|
||||
scan_glyph_order,
|
||||
scan_reading_order,
|
||||
)
|
||||
from .chunk import (
|
||||
CHUNK_KIND_PROSE,
|
||||
chunk_all,
|
||||
read_monographs_jsonl,
|
||||
write_chunks_jsonl,
|
||||
)
|
||||
from .segment import DuplicateDrugIdError, assemble, write_monographs_jsonl
|
||||
from .tables import (
|
||||
detect_table_regions,
|
||||
index_by_page,
|
||||
read_regions_json,
|
||||
write_regions_json,
|
||||
)
|
||||
from .validation import (
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
corpus_size,
|
||||
evaluate,
|
||||
evaluate_chunks,
|
||||
read_chunks,
|
||||
read_monographs,
|
||||
UNCLASSIFIED,
|
||||
compute_recall_precision,
|
||||
parse_back_index,
|
||||
)
|
||||
from .validation import scan_document as scan_residual_ink
|
||||
|
||||
|
||||
def _extracted_and_repaired_spans(doc, verbose: bool = False):
|
||||
"""The span stream, with vector-outlined text put back into it.
|
||||
|
||||
Outlier-catalog item 24: 51 runs of type exist only as vector paths, so
|
||||
extraction alone leaves holes mid-sentence ("Độ ổn định" -> "Độ n định").
|
||||
Every command that builds monographs must repair the stream the same way,
|
||||
or the ledger and the output describe different pipelines.
|
||||
"""
|
||||
spans = list(extract_spans(doc))
|
||||
if verbose:
|
||||
print(f"extracted {len(spans)} spans")
|
||||
runs = load_transcribed_runs()
|
||||
if not runs:
|
||||
return spans
|
||||
spans = merge_outlined_runs(spans, runs, doc=doc)
|
||||
if verbose:
|
||||
chars = sum(len(r.text) for r in runs)
|
||||
print(f"merged {len(runs)} transcribed vector-outlined runs "
|
||||
f"({chars} characters) back into the stream")
|
||||
return spans
|
||||
|
||||
|
||||
def _region_index(tables_arg, verbose: bool = False):
|
||||
"""Every region whose spans must be lifted out of prose, keyed by page.
|
||||
|
||||
`run` and `coverage` must build this the same way — when `coverage` loaded
|
||||
only tables while `run` also loaded formulas, the ledger described a
|
||||
pipeline that was not the one producing the output.
|
||||
"""
|
||||
index = {}
|
||||
regions_path = Path(tables_arg) if tables_arg else None
|
||||
if regions_path and regions_path.exists():
|
||||
index = index_by_page(read_regions_json(regions_path))
|
||||
if verbose:
|
||||
real = sum(len(v) for v in index.values())
|
||||
print(f"loaded {real} table regions on {len(index)} pages "
|
||||
f"from {regions_path}")
|
||||
elif regions_path and verbose:
|
||||
print(f"note: no table region map at {regions_path} — table text will "
|
||||
f"stay in section prose (run 'detect-tables' to produce one)")
|
||||
|
||||
formulas = load_formula_regions()
|
||||
for page, page_formulas in index_formula_regions_by_page(formulas).items():
|
||||
index.setdefault(page, []).extend(page_formulas)
|
||||
if formulas and verbose:
|
||||
print(f"loaded {len(formulas)} verified 2D formula regions on "
|
||||
f"{len({f.physical_page for f in formulas})} pages")
|
||||
return index or None
|
||||
|
||||
|
||||
def _cmd_run(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
print(f"opened {pdf_path} ({doc.page_count} pages)")
|
||||
|
||||
glyph_issues = scan_glyph_order(doc)
|
||||
reading_issues = scan_reading_order(doc)
|
||||
total_defects = len(glyph_issues) + len(reading_issues)
|
||||
if total_defects:
|
||||
print(
|
||||
f"glyph/reading-order sanity gate: {len(glyph_issues)} within-span + "
|
||||
f"{len(reading_issues)} cross-fragment issue(s) found "
|
||||
f"(see docs/pdf-parsing-outlier-catalog.md item 9 for known cases; "
|
||||
f"formula-region issues are expected there, not auto-corrected)."
|
||||
)
|
||||
|
||||
spans = _extracted_and_repaired_spans(doc, verbose=True)
|
||||
|
||||
table_index = _region_index(args.tables, verbose=True)
|
||||
|
||||
try:
|
||||
monographs = list(assemble(spans, table_index=table_index))
|
||||
except DuplicateDrugIdError as e:
|
||||
print(f"error: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
table_blocks = sum(len(m.tables) for m in monographs)
|
||||
quarantined = sum(1 for m in monographs for t in m.tables if t.quarantined)
|
||||
if table_index:
|
||||
print(f"lifted {table_blocks} table blocks out of section prose "
|
||||
f"({quarantined} quarantined)")
|
||||
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
count = write_monographs_jsonl(monographs, out_path)
|
||||
print(f"wrote {count} monographs to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_validate(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
spans = list(extract_spans(doc))
|
||||
try:
|
||||
monographs = list(assemble(spans))
|
||||
except DuplicateDrugIdError as e:
|
||||
print(f"error: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
ground_truth = parse_back_index(doc)
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
|
||||
print(f"detected monographs: {result.total_detected}")
|
||||
print(f"ground-truth entries: {result.total_ground_truth}")
|
||||
print(f"recall: {result.recall:.1%} ({result.matched_count}/{result.total_ground_truth})")
|
||||
print(f"precision: {result.precision:.1%}")
|
||||
if result.unmatched_ground_truth:
|
||||
print(f"\nunmatched ground-truth entries (first 20 of {len(result.unmatched_ground_truth)}):")
|
||||
for entry in result.unmatched_ground_truth[:20]:
|
||||
print(f" {entry.name}, {entry.printed_page}")
|
||||
if result.unmatched_detected:
|
||||
print(f"\nunmatched detected monographs (first 20 of {len(result.unmatched_detected)}):")
|
||||
for m in result.unmatched_detected[:20]:
|
||||
print(f" {m.drug_name} (physical page {m.source_page_range[0]})")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_detect_tables(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
regions = list(detect_table_regions(pdf_path))
|
||||
out_path = Path(args.out)
|
||||
count = write_regions_json(regions, out_path)
|
||||
|
||||
shapes = Counter(r.shape for r in regions)
|
||||
real = sum(1 for r in regions if r.is_real_table)
|
||||
print(f"detected {count} candidate regions on "
|
||||
f"{len({r.physical_page for r in regions})} pages")
|
||||
for shape, n in shapes.most_common():
|
||||
print(f" {shape:32} {n:5}")
|
||||
print(f"real tables: {real} not tables: {count - real}")
|
||||
print(f"wrote {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_coverage(args: argparse.Namespace) -> int:
|
||||
"""Span-level coverage ledger: where did every span end up?
|
||||
|
||||
Characters cannot be balanced directly — normalization joins, substitutes
|
||||
and drops them — so each span is assigned a state and characters are
|
||||
aggregated from those states.
|
||||
"""
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
spans = _extracted_and_repaired_spans(doc)
|
||||
|
||||
table_index = _region_index(args.tables)
|
||||
|
||||
ledger: list = []
|
||||
list(assemble(spans, table_index=table_index, ledger=ledger))
|
||||
|
||||
header, rows = ledger[0], ledger[1:]
|
||||
by_state = Counter(r["state"] for r in rows)
|
||||
chars = Counter()
|
||||
for r in rows:
|
||||
chars[r["state"]] += r["chars"]
|
||||
total_spans = len(rows)
|
||||
total_chars = sum(chars.values())
|
||||
|
||||
print(f"SCOPE: {pdf_path} — all {doc.page_count} pages")
|
||||
print(f"raw chars before span merge: {header['raw_chars_before_merge']:,}")
|
||||
print(f"spans after merge: {total_spans:,} chars: {total_chars:,}")
|
||||
print()
|
||||
print(f"{'state':<24}{'spans':>10}{'% spans':>10}{'chars':>14}{'% chars':>10}")
|
||||
print("-" * 68)
|
||||
for state, n in by_state.most_common():
|
||||
print(f"{state:<24}{n:>10,}{n/total_spans*100:>9.1f}%"
|
||||
f"{chars[state]:>14,}{chars[state]/total_chars*100:>9.1f}%")
|
||||
|
||||
unassigned = [r for r in rows if r["state"] == "unassigned"]
|
||||
print()
|
||||
print(f"UNASSIGNED: {len(unassigned):,} spans, "
|
||||
f"{sum(r['chars'] for r in unassigned):,} chars")
|
||||
if unassigned:
|
||||
pages = Counter(r["physical_page"] for r in unassigned)
|
||||
print(f" on {len(pages)} pages; worst: {pages.most_common(10)}")
|
||||
print(" first 10 examples:")
|
||||
for r in unassigned[:10]:
|
||||
print(f" p{r['physical_page']} {r['bbox']} {r['text']!r}")
|
||||
|
||||
if args.out:
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(json.dumps(ledger, ensure_ascii=False), encoding="utf-8")
|
||||
print(f"wrote full ledger to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_residual_ink(args: argparse.Namespace) -> int:
|
||||
"""Ask the page, not a detector: what ink did the text layer never emit?
|
||||
|
||||
Gate: `unclassified` must reach 0 — every surviving region has to be
|
||||
named, not silently tolerated.
|
||||
"""
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
tables_by_page = None
|
||||
regions_path = Path(args.tables) if args.tables else None
|
||||
if regions_path and regions_path.exists():
|
||||
tables_by_page = index_by_page(read_regions_json(regions_path))
|
||||
print(f"loaded table regions on {len(tables_by_page)} pages from {regions_path}")
|
||||
|
||||
pages = range(doc.page_count) if args.pages is None else _parse_pages(args.pages)
|
||||
pages = list(pages)
|
||||
findings = list(scan_residual_ink(doc, tables_by_page, pages))
|
||||
|
||||
kinds = Counter(kind for _, kind in findings)
|
||||
print(f"SCOPE: {pdf_path} — {len(pages)} of {doc.page_count} pages")
|
||||
print(f"residual regions: {len(findings)}")
|
||||
for kind, n in kinds.most_common():
|
||||
print(f" {kind:<26}{n:>7}")
|
||||
print(f"\nGATE unclassified = {kinds[UNCLASSIFIED]} (target 0)")
|
||||
|
||||
flagged = [(r, k) for r, k in findings
|
||||
if k in (FRACTION_BAR_CANDIDATE, UNCLASSIFIED)]
|
||||
print(f"needs eyes on it: {len(flagged)} region(s) on "
|
||||
f"{len({r.physical_page for r, _ in flagged})} pages")
|
||||
for region, kind in flagged[:20]:
|
||||
print(f" p{region.physical_page:<5} {kind:<24} "
|
||||
f"w={region.width_pt:6.1f} h={region.height_pt:5.1f} "
|
||||
f"bbox={region.bbox}")
|
||||
|
||||
if args.out:
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
payload = [
|
||||
{"physical_page": r.physical_page, "bbox": list(r.bbox),
|
||||
"ink_px": r.ink_px, "width_pt": round(r.width_pt, 2),
|
||||
"height_pt": round(r.height_pt, 2), "kind": k}
|
||||
for r, k in findings
|
||||
]
|
||||
out_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
||||
print(f"wrote {len(payload)} regions to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _parse_pages(spec: str):
|
||||
"""Parse '202', '200-210' or '202,1042' into physical page numbers."""
|
||||
pages = []
|
||||
for part in spec.split(","):
|
||||
if "-" in part:
|
||||
start, end = part.split("-", 1)
|
||||
pages.extend(range(int(start), int(end) + 1))
|
||||
else:
|
||||
pages.append(int(part))
|
||||
return pages
|
||||
|
||||
|
||||
def _cmd_chunk(args: argparse.Namespace) -> int:
|
||||
"""Build retrieval chunks from the segmented monographs."""
|
||||
monographs_path = Path(args.monographs)
|
||||
if not monographs_path.exists():
|
||||
print(f"error: no monographs at {monographs_path} — run 'run' first",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
|
||||
header_rows = {}
|
||||
regions_path = Path(args.tables) if args.tables else None
|
||||
if regions_path and regions_path.exists():
|
||||
header_rows = {r.table_id: r.first_row
|
||||
for r in read_regions_json(regions_path)}
|
||||
|
||||
monographs = list(read_monographs_jsonl(monographs_path))
|
||||
chunks = list(chunk_all(monographs, header_rows))
|
||||
|
||||
kinds = Counter(c.chunk_kind for c in chunks)
|
||||
with_attachments = sum(1 for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_PROSE and c.attachments)
|
||||
oversized = sum(1 for c in chunks if c.oversized)
|
||||
tokens = sum(c.est_tokens for c in chunks)
|
||||
|
||||
print(f"SCOPE: {monographs_path} — {len(monographs)} monographs")
|
||||
print(f"chunks: {len(chunks)}")
|
||||
for kind, n in kinds.most_common():
|
||||
print(f" {kind:<22}{n:>7}")
|
||||
print(f"prose chunks carrying a lifted block: {with_attachments}")
|
||||
print(f"oversized (over the {800}-token ceiling): {oversized}")
|
||||
print(f"estimated tokens (chars/4, an estimate): {tokens:,}")
|
||||
|
||||
out_path = Path(args.out)
|
||||
written = write_chunks_jsonl(chunks, out_path)
|
||||
print(f"wrote {written} chunks to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_chunk_ready(args: argparse.Namespace) -> int:
|
||||
"""Every gate that must hold before the corpus may be chunked.
|
||||
|
||||
Chunking turns text into embeddings, where a defect stops being
|
||||
inspectable — so each invariant is printed with its own count and its own
|
||||
target rather than folded into one verdict.
|
||||
"""
|
||||
monographs_path = Path(args.monographs)
|
||||
if not monographs_path.exists():
|
||||
print(f"error: no monographs at {monographs_path} — run 'run' first",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
|
||||
monographs = read_monographs(monographs_path)
|
||||
runs = load_transcribed_runs()
|
||||
gates = evaluate(monographs, [
|
||||
{"physical_page": r.physical_page, "text": r.text} for r in runs
|
||||
])
|
||||
|
||||
size = corpus_size(monographs)
|
||||
print(f"SCOPE: {monographs_path} — {size['monographs']} monographs, "
|
||||
f"{size['sections']} sections, {size['section_chars']:,} characters")
|
||||
print(f" {size['quarantined_blocks']} quarantined table/formula blocks "
|
||||
f"(excluded from prose, citable only with their source crop)")
|
||||
print()
|
||||
print(f"{'gate':<34}{'count':>8}{'target':>8} result")
|
||||
print("-" * 62)
|
||||
for gate in gates:
|
||||
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
|
||||
f"{'PASS' if gate.passed else 'FAIL'}"
|
||||
+ (f" {gate.detail}" if gate.detail else ""))
|
||||
|
||||
chunks_path = Path(args.chunks)
|
||||
if chunks_path.exists():
|
||||
chunk_gates = evaluate_chunks(monographs, read_chunks(chunks_path))
|
||||
print()
|
||||
print(f"ADR 0006 — chunk references ({chunks_path}):")
|
||||
for gate in chunk_gates:
|
||||
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
|
||||
f"{'PASS' if gate.passed else 'FAIL'}"
|
||||
+ (f" {gate.detail}" if gate.detail else ""))
|
||||
gates = gates + chunk_gates
|
||||
else:
|
||||
print()
|
||||
print(f"note: no chunks at {chunks_path} — ADR 0006 gates not run "
|
||||
f"(run 'chunk' to produce them)")
|
||||
|
||||
failed = [g for g in gates if not g.passed]
|
||||
print()
|
||||
if failed:
|
||||
print(f"NOT READY TO CHUNK — {len(failed)} gate(s) failing: "
|
||||
+ ", ".join(g.name for g in failed))
|
||||
return 1
|
||||
print("READY TO CHUNK — every gate above met its target.")
|
||||
print("Not proven by these gates: content accuracy against the source "
|
||||
"(no whole-document human-reviewed ground truth exists), table "
|
||||
"row/column reconstruction, and recall for borderless tables and "
|
||||
"bar-less formulas.")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_not_implemented(name: str):
|
||||
def _cmd(_args: argparse.Namespace) -> int:
|
||||
raise NotImplementedError(
|
||||
f"'{name}' is planned (see the approved segmentation/eval plan) but not yet built."
|
||||
)
|
||||
return _cmd
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(prog="python -m ingestion.cli")
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
|
||||
p_run = sub.add_parser("run", help="Extract + segment the PDF into monographs.jsonl")
|
||||
p_run.add_argument("--pdf", required=True, help="Path to the source PDF")
|
||||
p_run.add_argument(
|
||||
"--out", default="data/processed/monographs.jsonl",
|
||||
help="Output JSONL path (default: data/processed/monographs.jsonl)",
|
||||
)
|
||||
p_run.add_argument(
|
||||
"--tables", default="data/processed/table_regions.json",
|
||||
help="Table region map from 'detect-tables'. When present, table text "
|
||||
"is lifted out of section prose (default: "
|
||||
"data/processed/table_regions.json)",
|
||||
)
|
||||
p_run.set_defaults(func=_cmd_run)
|
||||
|
||||
p_validate = sub.add_parser("validate", help="Whole-book recall/precision vs. back-of-book index")
|
||||
p_validate.add_argument("--pdf", required=True)
|
||||
p_validate.set_defaults(func=_cmd_validate)
|
||||
|
||||
p_tables = sub.add_parser(
|
||||
"detect-tables",
|
||||
help="Locate + classify table regions (slow; result is cached and reused)",
|
||||
)
|
||||
p_tables.add_argument("--pdf", required=True)
|
||||
p_tables.add_argument(
|
||||
"--out", default="data/processed/table_regions.json",
|
||||
help="Output region-map path (default: data/processed/table_regions.json)",
|
||||
)
|
||||
p_tables.set_defaults(func=_cmd_detect_tables)
|
||||
|
||||
p_cov = sub.add_parser(
|
||||
"coverage", help="Span-level coverage ledger — where every span ended up")
|
||||
p_cov.add_argument("--pdf", required=True)
|
||||
p_cov.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_cov.add_argument("--out", default="data/processed/coverage_ledger.json")
|
||||
p_cov.set_defaults(func=_cmd_coverage)
|
||||
|
||||
p_residual = sub.add_parser(
|
||||
"residual-ink",
|
||||
help="Ink on the page that no extracted span accounts for "
|
||||
"(no ground truth needed; gate: unclassified = 0)",
|
||||
)
|
||||
p_residual.add_argument("--pdf", required=True)
|
||||
p_residual.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_residual.add_argument(
|
||||
"--pages", default=None,
|
||||
help="Limit to pages, e.g. '202' or '200-210' or '202,1042' "
|
||||
"(default: every page)",
|
||||
)
|
||||
p_residual.add_argument("--out", default="data/processed/residual_ink.json")
|
||||
p_residual.set_defaults(func=_cmd_residual_ink)
|
||||
|
||||
p_ready = sub.add_parser(
|
||||
"chunk-ready",
|
||||
help="Named gates that must all hold before chunking (garbage-in guard)")
|
||||
p_ready.add_argument(
|
||||
"--monographs", default="data/processed/monographs.jsonl")
|
||||
p_ready.add_argument("--chunks", default="data/processed/chunks.jsonl")
|
||||
p_ready.set_defaults(func=_cmd_chunk_ready)
|
||||
|
||||
p_chunk = sub.add_parser(
|
||||
"chunk", help="Build retrieval chunks (ADR 0004/0005/0006)")
|
||||
p_chunk.add_argument("--monographs", default="data/processed/monographs.jsonl")
|
||||
p_chunk.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_chunk.add_argument("--out", default="data/processed/chunks.jsonl")
|
||||
p_chunk.set_defaults(func=_cmd_chunk)
|
||||
|
||||
p_visual = sub.add_parser("visual-diff", help="Render a page with detected boundaries overlaid")
|
||||
p_visual.set_defaults(func=_cmd_not_implemented("visual-diff"))
|
||||
|
||||
p_scaffold = sub.add_parser("scaffold-golden", help="Draft golden-set entries for human review")
|
||||
p_scaffold.set_defaults(func=_cmd_not_implemented("scaffold-golden"))
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv=None) -> int:
|
||||
if hasattr(sys.stdout, "reconfigure"):
|
||||
sys.stdout.reconfigure(encoding="utf-8")
|
||||
sys.stderr.reconfigure(encoding="utf-8")
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(argv)
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,36 @@
|
||||
from .glyph_order import (
|
||||
GlyphOrderIssue,
|
||||
ReadingOrderIssue,
|
||||
find_reading_order_issues,
|
||||
is_reversed_order,
|
||||
scan_glyph_order,
|
||||
scan_reading_order,
|
||||
)
|
||||
from .formulas import index_formula_regions_by_page, load_formula_regions
|
||||
from .models import Span
|
||||
from .outlined_text import (
|
||||
OutlinedTextRun,
|
||||
detect_outlined_text,
|
||||
load_transcribed_runs,
|
||||
)
|
||||
from .repair import merge_outlined_runs
|
||||
from .page_map import build_page_map
|
||||
from .spans import extract_spans
|
||||
|
||||
__all__ = [
|
||||
"Span",
|
||||
"OutlinedTextRun",
|
||||
"load_formula_regions",
|
||||
"index_formula_regions_by_page",
|
||||
"detect_outlined_text",
|
||||
"load_transcribed_runs",
|
||||
"merge_outlined_runs",
|
||||
"build_page_map",
|
||||
"extract_spans",
|
||||
"GlyphOrderIssue",
|
||||
"is_reversed_order",
|
||||
"scan_glyph_order",
|
||||
"ReadingOrderIssue",
|
||||
"find_reading_order_issues",
|
||||
"scan_reading_order",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
"""2D (stacked-fraction) formula regions, loaded from a verified list.
|
||||
|
||||
Why a curated file and not a detector: the residual-ink check produces
|
||||
*candidates* — thin ink bars that no extracted span accounts for — and its
|
||||
measured precision on this book is 16 of 23, **69.6%**. The seven misses are
|
||||
decorative underlines on the Ministry decision page, ruled boxes and table
|
||||
borders. A 70%-precise rule must not be allowed to quarantine content on its
|
||||
own, so every candidate was rendered and read, and only the confirmed ones
|
||||
are listed in `data/verified/formula_regions_2d.json`.
|
||||
|
||||
The stored bbox is the fraction bar itself. The numerator sits above it and
|
||||
the denominator below, so the bar is grown vertically here to cover the whole
|
||||
formula. The growth factor is deliberately generous: over-capturing a line of
|
||||
neighbouring prose into a quarantined block is recoverable, leaving half a
|
||||
formula in the prose is not.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
|
||||
from ..tables.classify import SHAPE_FORMULA_2D
|
||||
from ..tables.models import TableRegion
|
||||
|
||||
# One line of body type on this book measures ~10.5pt; a fraction spans the
|
||||
# numerator line, the bar and the denominator line.
|
||||
FORMULA_BAND_HEIGHT_PT = 13.0
|
||||
|
||||
# Wide on purpose. The bar is often narrower than the numerator above it, and
|
||||
# a numerator span can carry leading spaces that push its box's centre well to
|
||||
# the left of the bar: at 4pt of margin, AMPICILIN VÀ SULBACTAM's numerator
|
||||
# 'Thể trọng (kg)' stayed behind in the prose while the rest of the fraction
|
||||
# was lifted. A fraction sits alone on its lines, so taking most of the column
|
||||
# width costs at worst a neighbouring line inside a quarantined block.
|
||||
FORMULA_SIDE_MARGIN_PT = 95.0
|
||||
|
||||
DEFAULT_VERIFIED_PATH = (
|
||||
Path(__file__).resolve().parents[2] / "data" / "verified"
|
||||
/ "formula_regions_2d.json"
|
||||
)
|
||||
|
||||
|
||||
def load_formula_regions(path: Path | None = None) -> List[TableRegion]:
|
||||
"""Read the verified 2D-formula regions as page regions to divert."""
|
||||
source = path or DEFAULT_VERIFIED_PATH
|
||||
if not source.exists():
|
||||
return []
|
||||
|
||||
payload = json.loads(source.read_text(encoding="utf-8"))
|
||||
regions = []
|
||||
for index, entry in enumerate(payload["regions"]):
|
||||
x0, y0, x1, y1 = entry["bar_bbox"]
|
||||
regions.append(
|
||||
TableRegion(
|
||||
table_id=f"p{entry['physical_page']}_f{index}",
|
||||
physical_page=entry["physical_page"],
|
||||
bbox=(x0 - FORMULA_SIDE_MARGIN_PT, y0 - FORMULA_BAND_HEIGHT_PT,
|
||||
x1 + FORMULA_SIDE_MARGIN_PT, y1 + FORMULA_BAND_HEIGHT_PT),
|
||||
n_rows=2,
|
||||
n_cols=1,
|
||||
shape=SHAPE_FORMULA_2D,
|
||||
)
|
||||
)
|
||||
return regions
|
||||
|
||||
|
||||
def index_formula_regions_by_page(
|
||||
regions: List[TableRegion],
|
||||
) -> Dict[int, List[TableRegion]]:
|
||||
index: Dict[int, List[TableRegion]] = {}
|
||||
for region in regions:
|
||||
index.setdefault(region.physical_page, []).append(region)
|
||||
return index
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Mandatory glyph/reading-order sanity check.
|
||||
|
||||
Two distinct defect shapes were confirmed by testing this module against the
|
||||
real PDF (not assumed from the ADR description alone):
|
||||
|
||||
1. **Within-span glyph reversal** (`scan_glyph_order` / `GlyphOrderIssue`):
|
||||
physical page 1373 (0-indexed) contains a span whose characters are
|
||||
positioned in strictly decreasing x-origin order, producing scrambled
|
||||
text (e.g. "= tịx 8 y..." instead of "y 8 xịt ="). Matches ADR 0003's
|
||||
original description.
|
||||
|
||||
2. **Cross-span row misordering within one PyMuPDF block**
|
||||
(`scan_reading_order` / `ReadingOrderIssue`) — a genuinely different,
|
||||
previously undocumented shape found while testing this module end-to-end:
|
||||
physical page 714 has a visual text row split into multiple PyMuPDF line
|
||||
objects, within a single `block`, that are emitted out of left-to-right
|
||||
order relative to each other (each individual span's own characters are
|
||||
fine, but the fragments interleave incorrectly), e.g. the row "...bảo
|
||||
quản nhiệt độ..." is emitted as fragments "quản ", " ộ", "đ tệih", "n " in
|
||||
that (wrong) order. Concatenating characters in raw extraction order
|
||||
produces garbled text; re-sorting the *same* characters within one visual
|
||||
row by x-origin recovers the correct reading order exactly. This means
|
||||
ADR 0003's "exactly 1 occurrence in the whole book" claim was based on a
|
||||
narrower (within-span-only) check and undercounted the real defect
|
||||
population — corrected here, see docs/pdf-parsing-outlier-catalog.md
|
||||
item 9 update.
|
||||
|
||||
**Getting the row-grouping key right took three iterations, each caught by
|
||||
running against the real book rather than trusting the first result (per
|
||||
CLAUDE.md's no-fabrication rule) — recorded here since the failure modes
|
||||
generalize to any from-scratch "reconstruct visual rows from raw
|
||||
coordinates" approach:**
|
||||
- v1 (group by rounded y only): 1113 "issues", almost all false positives.
|
||||
- v2 (group by (`_column_for_x` tag, rounded y), using the same ±20pt
|
||||
tolerance `extract/spans.py` uses for informational span tagging): dropped
|
||||
to 32, but a real false-positive class remained — kerning jitter (e.g.
|
||||
"mefloquin"'s 'l'/'o' origins differ by only 0.095pt, well inside normal
|
||||
font kerning) was treated as a reversal with no decrease tolerance, and
|
||||
the ±20pt column tolerance creates an *overlapping* accepted x-range for
|
||||
"left" (24-319) and "right" (288-582) — a right-column paragraph
|
||||
starting near x=299 was misclassified "left" and merged with an unrelated
|
||||
left-column line sharing the same y.
|
||||
- v3 (this version — group by (PyMuPDF's own `block` index, rounded y)):
|
||||
the real fix. Two paragraphs from genuinely different columns (e.g. page
|
||||
1104: one block starting at x=299.4, another at x=35.4, both at y=70.4)
|
||||
turned out to sit in **different PyMuPDF blocks**, while page 714's 3
|
||||
genuinely-misordered fragments sit in the **same block** (block 20) split
|
||||
across multiple `line` entries. Block identity — PyMuPDF's own layout
|
||||
analysis, already validated in ADR 0003 to respect this document's
|
||||
two-column structure — is a reliable discriminator that no fixed
|
||||
x-coordinate threshold can be, since real paragraph start positions vary
|
||||
enough to overlap any hand-picked column boundary. A minimum-decrease
|
||||
threshold (`_MIN_DECREASE_PT`, well above observed kerning jitter <0.3pt
|
||||
and well below observed real defects >2pt) still guards against sub-pixel
|
||||
jitter within a block/row. The header band (running page number + drug
|
||||
name, two unrelated boilerplate fields sharing a y-coordinate — stripped
|
||||
before chunking regardless, outlier-catalog item 13) is excluded outright.
|
||||
|
||||
Both checks are cheap (seconds per full-book pass) and must run over 100% of
|
||||
pages, not sampled, per ADR 0003's standing rigor bar.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Sequence, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
_ROW_Y_PRECISION = 1 # decimal places; same-baseline chars share y to <0.01pt in practice
|
||||
_HEADER_BAND_Y = 60.0 # page number + running drug name live here; boilerplate, stripped separately
|
||||
_MIN_DECREASE_PT = 1.0 # observed kerning jitter <0.3pt; observed real defects >2pt — safely between
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GlyphOrderIssue:
|
||||
physical_page: int
|
||||
span_bbox: Tuple[float, float, float, float]
|
||||
original_text: str
|
||||
corrected_text: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReadingOrderIssue:
|
||||
physical_page: int
|
||||
block_index: int
|
||||
row_y: float
|
||||
extracted_text: str
|
||||
corrected_text: str
|
||||
|
||||
|
||||
def is_reversed_order(x_origins: Sequence[float]) -> bool:
|
||||
"""True if every consecutive pair strictly decreases in x — the exact
|
||||
shape of the confirmed within-span defect. A normal LTR span's
|
||||
x-origins strictly increase; requiring *every* pair to decrease (not
|
||||
just "not sorted") avoids false-triggering on ordinary spans.
|
||||
"""
|
||||
if len(x_origins) < 2:
|
||||
return False
|
||||
# strict=False on purpose: this is the adjacent-pair idiom, so the two
|
||||
# sequences differ in length by one by construction.
|
||||
return all(b < a for a, b in zip(x_origins, x_origins[1:], strict=False))
|
||||
|
||||
|
||||
def scan_glyph_order(doc: fitz.Document) -> List[GlyphOrderIssue]:
|
||||
issues: List[GlyphOrderIssue] = []
|
||||
for pno in range(doc.page_count):
|
||||
for block in doc[pno].get_text("rawdict").get("blocks", []):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
chars = span.get("chars", [])
|
||||
if not chars:
|
||||
continue
|
||||
x_origins = [c["origin"][0] for c in chars]
|
||||
if is_reversed_order(x_origins):
|
||||
issues.append(GlyphOrderIssue(
|
||||
physical_page=pno,
|
||||
span_bbox=tuple(span["bbox"]),
|
||||
original_text="".join(c["c"] for c in chars),
|
||||
corrected_text="".join(c["c"] for c in reversed(chars)),
|
||||
))
|
||||
return issues
|
||||
|
||||
|
||||
def _has_significant_backward_jump(xs: Sequence[float], min_decrease: float) -> bool:
|
||||
return any(b < a - min_decrease for a, b in zip(xs, xs[1:], strict=False))
|
||||
|
||||
|
||||
def find_reading_order_issues(
|
||||
chars_by_row: Dict[Tuple[int, float], List[Tuple[float, str]]],
|
||||
min_decrease: float = _MIN_DECREASE_PT,
|
||||
) -> List[ReadingOrderIssue]:
|
||||
"""Pure logic, unit-testable without a real PDF: given characters already
|
||||
grouped by (block_index, row_y) in raw extraction order, flag a row only
|
||||
when it contains a backward x-jump larger than `min_decrease` — ordinary
|
||||
font kerning produces sub-0.3pt jitter (see module docstring's v2
|
||||
entry), so a plain "resorting changes the text" check without this
|
||||
threshold is not reliable; it self-corrupts already-correct text.
|
||||
Grouping by block index (not a hand-picked x-coordinate column
|
||||
boundary) is what the caller must guarantee — see module docstring's
|
||||
v1/v2/v3 history for why a coordinate-based row reconstruction alone is
|
||||
not safe.
|
||||
"""
|
||||
issues = []
|
||||
for (block_index, y), chars in chars_by_row.items():
|
||||
if len(chars) < 2:
|
||||
continue
|
||||
xs = [x for x, _ in chars]
|
||||
if not _has_significant_backward_jump(xs, min_decrease):
|
||||
continue
|
||||
extracted = "".join(c for _, c in chars)
|
||||
corrected = "".join(c for _, c in sorted(chars, key=lambda t: t[0]))
|
||||
if extracted != corrected:
|
||||
issues.append(ReadingOrderIssue(
|
||||
physical_page=-1, block_index=block_index, row_y=y,
|
||||
extracted_text=extracted, corrected_text=corrected,
|
||||
))
|
||||
return issues
|
||||
|
||||
|
||||
def scan_reading_order(doc: fitz.Document) -> List[ReadingOrderIssue]:
|
||||
issues: List[ReadingOrderIssue] = []
|
||||
for pno in range(doc.page_count):
|
||||
rows: Dict[Tuple[int, float], List[Tuple[float, str]]] = defaultdict(list)
|
||||
for block_index, block in enumerate(doc[pno].get_text("rawdict").get("blocks", [])):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
for c in span.get("chars", []):
|
||||
x, y = c["origin"]
|
||||
if y < _HEADER_BAND_Y:
|
||||
continue
|
||||
rows[(block_index, round(y, _ROW_Y_PRECISION))].append((x, c["c"]))
|
||||
for issue in find_reading_order_issues(rows):
|
||||
issues.append(ReadingOrderIssue(
|
||||
physical_page=pno, block_index=issue.block_index, row_y=issue.row_y,
|
||||
extracted_text=issue.extracted_text, corrected_text=issue.corrected_text,
|
||||
))
|
||||
return issues
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Persists the extracted span stream so re-segmentation doesn't require
|
||||
re-running PyMuPDF over the whole PDF every time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from .models import Span
|
||||
|
||||
|
||||
def write_spans_jsonl(spans: Iterable[Span], path: Path) -> int:
|
||||
count = 0
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
for span in spans:
|
||||
f.write(json.dumps(span.__dict__, ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def read_spans_jsonl(path: Path) -> Iterator[Span]:
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
yield Span(**json.loads(line))
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Data model for text spans extracted from the source PDF.
|
||||
|
||||
A Span is one PyMuPDF text span (a run of characters sharing one font/size),
|
||||
tagged with page and column position. This is the sole unit `segment/`
|
||||
consumes — it never touches PyMuPDF or fitz.Document directly (see ADR 0003
|
||||
and docs/pdf-parsing-outlier-catalog.md for why: PyMuPDF is the validated
|
||||
sole general-text extractor for this document).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Span:
|
||||
physical_page: int
|
||||
printed_page: Optional[int]
|
||||
column: str # "left" | "right" | "full_width" | "unknown"
|
||||
block: int
|
||||
line: int
|
||||
span_index: int
|
||||
x0: float
|
||||
y0: float
|
||||
x1: float
|
||||
y1: float
|
||||
text: str
|
||||
font: str
|
||||
size: float
|
||||
|
||||
@property
|
||||
def bold(self) -> bool:
|
||||
return "Bold" in self.font
|
||||
|
||||
@property
|
||||
def span_id(self) -> str:
|
||||
"""Stable identifier for one source span.
|
||||
|
||||
Built from PyMuPDF's own page/block/line/span indices, so the same PDF
|
||||
always yields the same id — a counter would renumber whenever anything
|
||||
upstream changed, which makes downstream provenance unverifiable.
|
||||
"""
|
||||
return f"p{self.physical_page}_b{self.block}_l{self.line}_s{self.span_index}"
|
||||
@@ -0,0 +1,108 @@
|
||||
"""Text that was drawn as vector outlines instead of text operators.
|
||||
|
||||
Physical page 714 prints 17 lines of ordinary Gatifloxacin prose that no text
|
||||
extractor returns: `page.get_text()` omits them, `page.search_for()` finds
|
||||
nothing, `pdfplumber` and `opendataloader-pdf` omit them too. They are not
|
||||
text at all in the file — each line is a filled path of ~1,600-1,800 items,
|
||||
shaped exactly like one line of type and filled with the body-text colour.
|
||||
|
||||
Nothing that asks a text layer can see this, which is why it survived every
|
||||
earlier check in this project. It was found by masking extracted spans over a
|
||||
rendered page and looking at the ink that was left.
|
||||
|
||||
Detection is deliberately shape-based, not content-based: a filled drawing
|
||||
with hundreds of path items whose box is the height of one line and at least
|
||||
30pt wide. Recovery cannot be automatic — the glyphs carry no character
|
||||
codes — so these regions are reported for transcription, never guessed at.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator, List, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
# Measured whole-document. Full outlined lines carry 1,126-1,831 path items;
|
||||
# single outlined glyphs carry 39-45. Ordinary decoration (the running-header
|
||||
# rule, cell borders) carries 1-2, so 30 separates them cleanly. Lowering the
|
||||
# threshold from 200 to 30 was checked before it was applied: it adds 29 runs
|
||||
# and no new page, all on pages 714 and 736, which were already affected.
|
||||
MIN_PATH_ITEMS = 30
|
||||
MIN_RUN_WIDTH_PT = 2.0
|
||||
RUN_HEIGHT_RANGE_PT = (3.0, 20.0)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OutlinedTextRun:
|
||||
"""A run of type that exists only as vector paths — a line, or one glyph.
|
||||
|
||||
`text` stays empty unless a human (or a rendered-page reading) fills it
|
||||
in: the paths carry no character codes, so any text here is a
|
||||
transcription and must be recorded as one.
|
||||
"""
|
||||
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
path_items: int
|
||||
text: str = ""
|
||||
|
||||
@property
|
||||
def is_transcribed(self) -> bool:
|
||||
return bool(self.text)
|
||||
|
||||
|
||||
def _is_outlined_run(drawing: dict, page_width: float) -> bool:
|
||||
rect = drawing["rect"]
|
||||
low, high = RUN_HEIGHT_RANGE_PT
|
||||
return (
|
||||
drawing["type"] == "f"
|
||||
and len(drawing["items"]) >= MIN_PATH_ITEMS
|
||||
and low <= rect.height <= high
|
||||
and rect.width >= MIN_RUN_WIDTH_PT
|
||||
and rect.x0 >= 0
|
||||
and rect.x1 <= page_width + 1
|
||||
)
|
||||
|
||||
|
||||
def detect_outlined_text(
|
||||
doc: "fitz.Document", pages: Iterable[int] | None = None,
|
||||
) -> Iterator[OutlinedTextRun]:
|
||||
"""Yield every run of vector-outlined type in the document."""
|
||||
page_numbers = range(doc.page_count) if pages is None else pages
|
||||
for number in page_numbers:
|
||||
page = doc[number]
|
||||
for drawing in page.get_drawings():
|
||||
if not _is_outlined_run(drawing, page.rect.x1):
|
||||
continue
|
||||
rect = drawing["rect"]
|
||||
yield OutlinedTextRun(
|
||||
physical_page=number,
|
||||
bbox=(round(rect.x0, 2), round(rect.y0, 2),
|
||||
round(rect.x1, 2), round(rect.y1, 2)),
|
||||
path_items=len(drawing["items"]),
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_TRANSCRIPTIONS_PATH = (
|
||||
Path(__file__).resolve().parents[2] / "data" / "verified"
|
||||
/ "outlined_text_transcriptions.json"
|
||||
)
|
||||
|
||||
|
||||
def load_transcribed_runs(path: "Path | None" = None) -> List[OutlinedTextRun]:
|
||||
"""Read the transcribed runs. Every `text` here was read off a rendering."""
|
||||
source = path or DEFAULT_TRANSCRIPTIONS_PATH
|
||||
if not source.exists():
|
||||
return []
|
||||
payload = json.loads(source.read_text(encoding="utf-8"))
|
||||
return [
|
||||
OutlinedTextRun(
|
||||
physical_page=run["physical_page"],
|
||||
bbox=tuple(run["bbox"]),
|
||||
path_items=run["path_items"],
|
||||
text=run["text"],
|
||||
)
|
||||
for run in payload["runs"]
|
||||
]
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Maps physical (0-indexed) page numbers to the book's own printed folio
|
||||
number, by reading the isolated numeric token in each page's header band.
|
||||
|
||||
Required because ADR 0003's page-range rules ("monographs run printed pages
|
||||
99-1496") are meaningless without a real per-page mapping — verified rather
|
||||
than assumed to be a constant offset, since front matter in some books uses
|
||||
roman numerals or restarts numbering. In this book the mapping is empirically
|
||||
a constant (physical + 1) across the entire 1668 pages (verified against the
|
||||
milestone pages: physical 36->printed 37, physical 98->printed 99, physical
|
||||
100->printed 101 "Abacavir", physical 1496->printed 1497), but this module
|
||||
still reads the real folio per page rather than hard-coding that constant, so
|
||||
a future edition/scan with different numbering does not silently mis-map.
|
||||
|
||||
Confirmed real false-conflict case (physical page 1243, "RIBOFLAVIN (Vitamin
|
||||
B2)" monograph, found via a whole-book `cli validate` run and confirmed by
|
||||
rendering the page to an image): the monograph's own title sits high enough
|
||||
on the page that its "2" subscript (size 5.83) falls inside the header band
|
||||
alongside the real folio "1244" (size 10.0), producing two conflicting
|
||||
digit-only candidates and silently dropping the printed page — and with it
|
||||
the entire monograph, since every span on the page then fails the
|
||||
printed-page-range check. A genuine folio is set in the header's own running
|
||||
font size, never a subscript's reduced size, so preferring the
|
||||
largest-font-size candidate(s) resolves this without weakening the
|
||||
"never guess on a real conflict" rule for pages with, e.g., two same-size
|
||||
candidates (still returns None).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
_FOLIO_RE = re.compile(r"^\d{1,4}$")
|
||||
HEADER_BAND_Y = 60.0 # printed folio always appears in the top header band
|
||||
|
||||
|
||||
def build_page_map(doc: fitz.Document) -> Dict[int, Optional[int]]:
|
||||
"""Returns {physical_page: printed_page_or_None}. None means no folio
|
||||
was recoverable (blank/separator pages, title pages) — a valid state,
|
||||
not an error.
|
||||
"""
|
||||
return {pno: _read_folio(doc[pno]) for pno in range(doc.page_count)}
|
||||
|
||||
|
||||
def _read_folio(page: fitz.Page) -> Optional[int]:
|
||||
candidates = [] # (text, size) pairs
|
||||
for block in page.get_text("dict").get("blocks", []):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
text = span["text"].strip()
|
||||
if span["bbox"][1] < HEADER_BAND_Y and _FOLIO_RE.match(text):
|
||||
candidates.append((text, span["size"]))
|
||||
return pick_folio(candidates)
|
||||
|
||||
|
||||
def pick_folio(candidates: List[Tuple[str, float]]) -> Optional[int]:
|
||||
"""Pure decision logic, given the header-band digit-only (text, size)
|
||||
candidates already collected from a page: which one is the real folio.
|
||||
"""
|
||||
if not candidates:
|
||||
return None # blank/separator page: unrecoverable, never guess.
|
||||
|
||||
max_size = max(size for _, size in candidates)
|
||||
largest = {text for text, size in candidates if size == max_size}
|
||||
if len(largest) == 1:
|
||||
return int(largest.pop())
|
||||
# still conflicting even after dropping smaller-font stray digits
|
||||
# (e.g. subscripts): genuinely ambiguous, never guess.
|
||||
return None
|
||||
@@ -0,0 +1,241 @@
|
||||
"""Put transcribed vector-outlined text back into the span stream.
|
||||
|
||||
Outlier-catalog item 24: 51 runs of type on 5 pages exist only as filled
|
||||
vector paths, so no extractor emits a span for them. They were transcribed by
|
||||
reading rendered crops (`data/verified/outlined_text_transcriptions.json`).
|
||||
This module is what makes that transcription part of the corpus rather than a
|
||||
note beside it.
|
||||
|
||||
Placement is geometric, not textual. A run that vertically overlaps an
|
||||
existing visual line is a character (or fragment) dropped out of *that* line
|
||||
and is spliced into it in x order — this is the common case and the damaging
|
||||
one, because a missing diacritic turns "Độ ổn định" into "Độ n định" and
|
||||
still reads as ordinary prose. A run that overlaps no line is a whole missing
|
||||
line and is inserted at a line boundary, ordered by column then y, so the
|
||||
reading order the rest of the pipeline depends on is preserved.
|
||||
|
||||
Synthetic spans are marked by `SYNTHETIC_LINE_BASE` in their line index, so
|
||||
their provenance ids stay distinguishable from real extracted spans forever.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import replace
|
||||
from typing import Iterable, List, Sequence, Tuple
|
||||
|
||||
from .models import Span
|
||||
from .outlined_text import OutlinedTextRun
|
||||
from .page_map import HEADER_BAND_Y
|
||||
from .spans import classify_column
|
||||
|
||||
# Line indices at or above this never come from PyMuPDF — a real page has
|
||||
# nothing close to this many lines in a block.
|
||||
SYNTHETIC_LINE_BASE = 100_000
|
||||
|
||||
SYNTHETIC_FONT = "TimesNewRomanPSMT"
|
||||
SYNTHETIC_SIZE = 9.5
|
||||
|
||||
# A run counts as belonging to an existing line when their vertical extents
|
||||
# overlap by at least this fraction of the run's height.
|
||||
LINE_OVERLAP_RATIO = 0.5
|
||||
|
||||
|
||||
def _vertical_overlap(a: Tuple[float, float], b: Tuple[float, float]) -> float:
|
||||
return max(0.0, min(a[1], b[1]) - max(a[0], b[0]))
|
||||
|
||||
|
||||
def _line_groups(spans: Sequence[Span]) -> List[Tuple[int, int, List[Span]]]:
|
||||
"""Consecutive spans sharing a visual line, with their index range.
|
||||
|
||||
Mirrors `normalize.group_visual_lines`' notion of a line so that a span
|
||||
spliced here lands in the same group there.
|
||||
"""
|
||||
groups: List[Tuple[int, int, List[Span]]] = []
|
||||
for index, span in enumerate(spans):
|
||||
key = (span.physical_page, span.block, span.line)
|
||||
if groups and (groups[-1][2][0].physical_page, groups[-1][2][0].block,
|
||||
groups[-1][2][0].line) == key:
|
||||
start, _, members = groups[-1]
|
||||
members.append(span)
|
||||
groups[-1] = (start, index, members)
|
||||
else:
|
||||
groups.append((index, index, [span]))
|
||||
return groups
|
||||
|
||||
|
||||
def _synthetic_span(run: OutlinedTextRun, template: Span | None,
|
||||
line_index: int) -> Span:
|
||||
x0, y0, x1, y1 = run.bbox
|
||||
column = classify_column(run.bbox)
|
||||
if y0 < HEADER_BAND_Y:
|
||||
# Part of the running header, which is a full-width band. Tagging it
|
||||
# as such lets the one existing boilerplate rule strip it, instead of
|
||||
# this module deciding separately what boilerplate is.
|
||||
column = "full_width"
|
||||
if template is not None:
|
||||
return replace(
|
||||
template,
|
||||
column=template.column if y0 >= HEADER_BAND_Y else column,
|
||||
line=line_index,
|
||||
span_index=0,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
)
|
||||
return Span(
|
||||
physical_page=run.physical_page,
|
||||
printed_page=None,
|
||||
column=column,
|
||||
block=0,
|
||||
line=line_index,
|
||||
span_index=0,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
size=SYNTHETIC_SIZE,
|
||||
)
|
||||
|
||||
|
||||
def char_boxes(doc, page_number: int) -> List[Tuple[str, Tuple[float, ...]]]:
|
||||
"""Per-character boxes for one page, in extraction order.
|
||||
|
||||
Needed because a dropped glyph usually sits *inside* an extracted span,
|
||||
not between two of them: on physical page 714 the span
|
||||
`'Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu'` runs from x=35.5 to
|
||||
x=223.0 and the missing 'ở' belongs at x=167. Splicing at span boundaries
|
||||
put it at the end and produced 'tuở ổi'. Character geometry is the only
|
||||
thing that says where the hole actually is.
|
||||
"""
|
||||
boxes = []
|
||||
for block in doc[page_number].get_text("rawdict")["blocks"]:
|
||||
for line in block.get("lines", []):
|
||||
for span in line["spans"]:
|
||||
for char in span["chars"]:
|
||||
boxes.append((char["c"], char["bbox"]))
|
||||
return boxes
|
||||
|
||||
|
||||
def _split_offset(span: Span, run: OutlinedTextRun,
|
||||
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> int | None:
|
||||
"""Character offset inside `span.text` where the run's glyph belongs."""
|
||||
inside = [
|
||||
box for box in boxes
|
||||
if span.x0 - 0.5 <= box[1][0] and box[1][2] <= span.x1 + 0.5
|
||||
and span.y0 - 1.0 <= box[1][1] and box[1][3] <= span.y1 + 1.0
|
||||
]
|
||||
if len(inside) != len(span.text):
|
||||
return None
|
||||
for offset, (_, bbox) in enumerate(inside):
|
||||
if bbox[0] >= run.bbox[2] - 0.5:
|
||||
return offset
|
||||
return None
|
||||
|
||||
|
||||
def _splice_into_line(spans: List[Span], group, run: OutlinedTextRun,
|
||||
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> None:
|
||||
start, end, members = group
|
||||
anchor = members[0]
|
||||
synthetic = replace(
|
||||
anchor,
|
||||
span_index=SYNTHETIC_LINE_BASE,
|
||||
x0=run.bbox[0], y0=run.bbox[1], x1=run.bbox[2], y1=run.bbox[3],
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
)
|
||||
|
||||
for offset, member in enumerate(members):
|
||||
if not (member.x0 <= run.bbox[0] and run.bbox[2] <= member.x1):
|
||||
continue
|
||||
split_at = _split_offset(member, run, boxes)
|
||||
if split_at is None or split_at == 0:
|
||||
continue
|
||||
index = start + offset
|
||||
left = replace(member, text=member.text[:split_at], x1=run.bbox[0])
|
||||
right = replace(member, text=member.text[split_at:],
|
||||
span_index=member.span_index + SYNTHETIC_LINE_BASE,
|
||||
x0=run.bbox[2])
|
||||
# A split at the very end of a span leaves a fragment holding nothing
|
||||
# but a space. Dropping it costs no text — `join_visual_line` decides
|
||||
# spacing from the horizontal gap, not from a span's own padding.
|
||||
pieces = [p for p in (left, synthetic, right) if p.text.strip()]
|
||||
spans[index:index + 1] = pieces
|
||||
return
|
||||
|
||||
position = end + 1
|
||||
for offset, member in enumerate(members):
|
||||
if run.bbox[0] < member.x0:
|
||||
position = start + offset
|
||||
break
|
||||
spans.insert(position, synthetic)
|
||||
|
||||
|
||||
def _insert_as_new_line(spans: List[Span], run: OutlinedTextRun,
|
||||
line_index: int) -> None:
|
||||
column = classify_column(run.bbox)
|
||||
if run.bbox[1] < HEADER_BAND_Y:
|
||||
column = "full_width"
|
||||
|
||||
position = len(spans)
|
||||
template = None
|
||||
for start, _, members in _line_groups(spans):
|
||||
first = members[0]
|
||||
if first.physical_page < run.physical_page:
|
||||
template = first
|
||||
continue
|
||||
if first.physical_page > run.physical_page:
|
||||
position = start
|
||||
break
|
||||
if first.column == column:
|
||||
template = first
|
||||
if first.y0 > run.bbox[1]:
|
||||
position = start
|
||||
break
|
||||
elif template is not None and first.column != column and position == len(spans):
|
||||
# first line of the next column on this page — the run belongs
|
||||
# before it if we never found a lower line in its own column
|
||||
position = start
|
||||
spans.insert(position, _synthetic_span(run, template, line_index))
|
||||
|
||||
|
||||
def merge_outlined_runs(
|
||||
spans: Iterable[Span], runs: Sequence[OutlinedTextRun], doc=None,
|
||||
) -> List[Span]:
|
||||
"""Return the span stream with every transcribed run put back in place.
|
||||
|
||||
`doc` enables character-accurate splicing of a glyph that fell out of the
|
||||
middle of an extracted span. Without it the run can only be placed at a
|
||||
span boundary, which is wrong for exactly the case that matters most.
|
||||
"""
|
||||
merged = list(spans)
|
||||
ordered = sorted(runs, key=lambda r: (r.physical_page, r.bbox[1], r.bbox[0]))
|
||||
boxes_cache: dict = {}
|
||||
for offset, run in enumerate(ordered):
|
||||
if not run.text:
|
||||
continue
|
||||
run_extent = (run.bbox[1], run.bbox[3])
|
||||
run_column = classify_column(run.bbox)
|
||||
target = None
|
||||
for group in _line_groups(merged):
|
||||
first = group[2][0]
|
||||
if first.physical_page != run.physical_page:
|
||||
continue
|
||||
# Column, not just height: this book sets two columns, so a
|
||||
# right-column run sits at the same y as an unrelated left-column
|
||||
# line. Without this, page 714's "…làm thay đ" was spliced onto
|
||||
# the left column and "…ổi nồng độ glucose máu" stayed broken.
|
||||
if first.column != run_column:
|
||||
continue
|
||||
line_extent = (min(s.y0 for s in group[2]),
|
||||
max(s.y1 for s in group[2]))
|
||||
overlap = _vertical_overlap(run_extent, line_extent)
|
||||
height = max(run.bbox[3] - run.bbox[1], 0.1)
|
||||
if overlap / height >= LINE_OVERLAP_RATIO:
|
||||
target = group
|
||||
break
|
||||
if target is not None:
|
||||
if doc is not None and run.physical_page not in boxes_cache:
|
||||
boxes_cache[run.physical_page] = char_boxes(doc, run.physical_page)
|
||||
_splice_into_line(merged, target, run,
|
||||
boxes_cache.get(run.physical_page, ()))
|
||||
else:
|
||||
_insert_as_new_line(merged, run, SYNTHETIC_LINE_BASE + offset)
|
||||
return merged
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Continuous cross-page span extraction.
|
||||
|
||||
Per ADR 0003: the pipeline must consume text as one continuous cross-page
|
||||
stream, never per-page silos, so that multi-line headings and paragraphs
|
||||
spanning a page/column break can be handled correctly downstream. This
|
||||
module's only job is to yield that stream in reading order; it does not
|
||||
decide what is a heading or a monograph boundary (that's `segment/`'s job).
|
||||
|
||||
Column tagging uses the bounding-box ranges confirmed by inspection in ADR
|
||||
0003 (left column x~44-299, right column x~308-562, page width ~595).
|
||||
|
||||
An earlier version of this module trusted PyMuPDF's own raw block order to
|
||||
already sequence left-then-right correctly, validated only against one
|
||||
example page during ADR 0003. Confirmed wrong via a whole-document
|
||||
character-diff against an independent parser (opendataloader-pdf) plus
|
||||
visual page reads: on 12 of 1398 monograph-range pages (e.g. physical page
|
||||
1100, the OXYBUTYNIN/OXYMETAZOLIN boundary), PyMuPDF's raw block order
|
||||
emits the *right* column before the *left* column. Left uncorrected, this
|
||||
silently corrupts monograph data at a column-reversed page's drug boundary
|
||||
— the wrong column's section content (e.g. "Chống chỉ định") gets appended
|
||||
to whichever monograph is still open when it's encountered, overwriting
|
||||
that monograph's real section and leaving the next monograph missing it.
|
||||
Fixed by explicitly sorting blocks (full_width header band first, then
|
||||
left column, then right column, each by y-position) instead of trusting
|
||||
raw order — full_width blocks are confirmed to be page-header material
|
||||
only in this document (real monograph titles and section headings sit
|
||||
within one column's x-range), so this ordering matches the book's actual
|
||||
two-column-with-running-header layout.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterator
|
||||
|
||||
import fitz
|
||||
|
||||
from .models import Span
|
||||
from .page_map import build_page_map
|
||||
|
||||
_LEFT_COLUMN_X = (44.0, 299.0)
|
||||
_RIGHT_COLUMN_X = (308.0, 562.0)
|
||||
_COLUMN_TOLERANCE = 20.0
|
||||
_FULL_WIDTH_MIN = 400.0
|
||||
_COLUMN_SORT_RANK = {"full_width": 0, "left": 1, "right": 2, "unknown": 3}
|
||||
|
||||
|
||||
def extract_spans(doc: fitz.Document) -> Iterator[Span]:
|
||||
page_map = build_page_map(doc)
|
||||
for pno in range(doc.page_count):
|
||||
printed = page_map[pno]
|
||||
page_dict = doc[pno].get_text("dict")
|
||||
blocks = _sort_blocks_reading_order(page_dict.get("blocks", []))
|
||||
for block_idx, block in enumerate(blocks):
|
||||
column = classify_column(block.get("bbox"))
|
||||
for line_idx, line in enumerate(block.get("lines", [])):
|
||||
for span_idx, span in enumerate(line.get("spans", [])):
|
||||
text = span["text"]
|
||||
if not text.strip():
|
||||
continue
|
||||
x0, y0, x1, y1 = span["bbox"]
|
||||
yield Span(
|
||||
physical_page=pno,
|
||||
printed_page=printed,
|
||||
column=column,
|
||||
block=block_idx,
|
||||
line=line_idx,
|
||||
span_index=span_idx,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=text,
|
||||
font=span["font"],
|
||||
size=span["size"],
|
||||
)
|
||||
|
||||
|
||||
def _sort_blocks_reading_order(blocks: list) -> list:
|
||||
"""Full_width header band first, then left column, then right column,
|
||||
each by y-position — see module docstring for the confirmed real bug
|
||||
this replaces (trusting PyMuPDF's raw block order).
|
||||
"""
|
||||
return sorted(
|
||||
blocks,
|
||||
key=lambda b: (_COLUMN_SORT_RANK[classify_column(b.get("bbox"))], b.get("bbox", (0, 0, 0, 0))[1]),
|
||||
)
|
||||
|
||||
|
||||
def classify_column(bbox) -> str:
|
||||
"""Which of the book's two columns a box sits in (or the header band)."""
|
||||
if bbox is None:
|
||||
return "unknown"
|
||||
x0, _, x1, _ = bbox
|
||||
if (x1 - x0) >= _FULL_WIDTH_MIN:
|
||||
return "full_width"
|
||||
mid = (x0 + x1) / 2
|
||||
# Exact containment before tolerance. The two tolerance bands overlap
|
||||
# between x=288 and x=319, and testing left first put anything in that
|
||||
# strip in the left column — invisible for a full-width block, wrong for a
|
||||
# narrow one. A single 4pt glyph at x=315 on physical page 714 was
|
||||
# classified left, so the 'ổ' missing from "Độ ổn định" could not be
|
||||
# matched to its own line and the corruption survived the repair.
|
||||
if _LEFT_COLUMN_X[0] <= mid <= _LEFT_COLUMN_X[1]:
|
||||
return "left"
|
||||
if _RIGHT_COLUMN_X[0] <= mid <= _RIGHT_COLUMN_X[1]:
|
||||
return "right"
|
||||
if _LEFT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _LEFT_COLUMN_X[1] + _COLUMN_TOLERANCE:
|
||||
return "left"
|
||||
if _RIGHT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _RIGHT_COLUMN_X[1] + _COLUMN_TOLERANCE:
|
||||
return "right"
|
||||
return "unknown"
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Text normalization shared by the pipeline and any validation script.
|
||||
|
||||
Kept as its own stage so the rules live in exactly one place (CLAUDE.md's DRY
|
||||
rule): `segment/` applies them when assembling section text, and audits
|
||||
import the same functions rather than re-implementing them.
|
||||
"""
|
||||
from .glyphs import PUA_SUBSTITUTIONS, find_unmapped_pua, substitute_pua
|
||||
from .text_flow import SPACE_GAP_PT, group_visual_lines, join_spans, join_visual_line
|
||||
|
||||
__all__ = [
|
||||
"PUA_SUBSTITUTIONS",
|
||||
"SPACE_GAP_PT",
|
||||
"find_unmapped_pua",
|
||||
"substitute_pua",
|
||||
"group_visual_lines",
|
||||
"join_spans",
|
||||
"join_visual_line",
|
||||
]
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Private-use-area glyph substitution.
|
||||
|
||||
The source PDF sets several symbols in the `SymbolTiger` / `Symbol` fonts,
|
||||
which PyMuPDF faithfully returns as raw Unicode private-use-area codepoints.
|
||||
Left untranslated they reach embeddings as junk — and 74 of the 86
|
||||
occurrences in this corpus are the comparison operators inside dosing
|
||||
sentences, where losing the operator changes clinical meaning ("liều ≤ 100
|
||||
mg" is not "liều 100 mg").
|
||||
|
||||
Every entry below was located in the source PDF, rendered to an image, and
|
||||
read visually — none inferred from surrounding context. Counts and the page
|
||||
each was confirmed on are recorded in docs/progress-log.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
PUA_LO, PUA_HI = 0xE000, 0xF8FF
|
||||
|
||||
# codepoint -> replacement, with the page the glyph was visually confirmed on
|
||||
PUA_SUBSTITUTIONS: dict[str, str] = {
|
||||
"": "≥", # p.141 "trẻ em ≥ 10 tuổi"
|
||||
"": "≤", # p.169 "liều ≤ 100 mg"
|
||||
"": "α", # p.334 "Streptococcus α tan huyết"
|
||||
"": "→", # p.1027 "HCO₃⁻ + H⁺ → H₂CO₃"
|
||||
"": "®", # p.891 "Plasma Lyte® 56/5%"
|
||||
"": "₁", # p.957 "alpha₁-acid glycoprotein"
|
||||
"": "↓", # p.1033 "rhodanese ↓" (catalysis arrow)
|
||||
"": "γ", # p.1352 "interferon - γ"
|
||||
}
|
||||
|
||||
_TABLE = str.maketrans(PUA_SUBSTITUTIONS)
|
||||
|
||||
|
||||
def substitute_pua(text: str) -> str:
|
||||
return text.translate(_TABLE)
|
||||
|
||||
|
||||
def find_unmapped_pua(text: str) -> list[str]:
|
||||
"""PUA codepoints with no verified replacement.
|
||||
|
||||
Returned rather than silently passed through: an unmapped glyph means the
|
||||
corpus contains a symbol nobody has visually confirmed yet, which must be
|
||||
surfaced instead of embedded as junk.
|
||||
"""
|
||||
return sorted({
|
||||
ch for ch in text
|
||||
if PUA_LO <= ord(ch) <= PUA_HI and ch not in PUA_SUBSTITUTIONS
|
||||
})
|
||||
@@ -0,0 +1,83 @@
|
||||
"""Rejoin PDF spans into flowing text.
|
||||
|
||||
`segment/assembler.py` originally appended one line per *span*, so any visual
|
||||
line that the PDF split into several spans (an italic run, a subscript, a
|
||||
symbol-font glyph) became several "lines". Measured on the whole corpus that
|
||||
produced 99,501 mid-sentence line breaks across 71.8% of sections and 11,612
|
||||
sub-4-character fragment lines — e.g. `"cytochrom P\n450\ngây"`,
|
||||
`"(\nfeline immunodeficiency virus\n)"`, `"Cl\ncr\n< 50 ml/"`.
|
||||
|
||||
Text alone cannot tell a mid-word span split from a genuine line wrap, so the
|
||||
join is driven by geometry instead: PyMuPDF's own `(block, line)` indices say
|
||||
which spans share a visual line, and the horizontal gap says whether a space
|
||||
belongs between them.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterable, List, Sequence, Tuple
|
||||
|
||||
from ..extract.models import Span
|
||||
|
||||
# Horizontal gap (pt) above which two spans on one visual line are separated
|
||||
# by a real space. Kerning noise between adjacent glyph runs sits well under
|
||||
# 1pt; a space at this book's 9.5-10pt body size is ≈2.4pt.
|
||||
SPACE_GAP_PT = 1.0
|
||||
|
||||
_SENTENCE_END = ".;:!?"
|
||||
|
||||
|
||||
def _line_key(span: Span) -> Tuple[int, int, int]:
|
||||
return (span.physical_page, span.block, span.line)
|
||||
|
||||
|
||||
def group_visual_lines(spans: Sequence[Span]) -> List[List[Span]]:
|
||||
"""Group consecutive spans that share a visual line, preserving order."""
|
||||
lines: List[List[Span]] = []
|
||||
for span in spans:
|
||||
if lines and _line_key(lines[-1][0]) == _line_key(span):
|
||||
lines[-1].append(span)
|
||||
else:
|
||||
lines.append([span])
|
||||
return lines
|
||||
|
||||
|
||||
def join_visual_line(spans: Sequence[Span]) -> str:
|
||||
"""Concatenate one visual line, inserting a space only where one exists."""
|
||||
out = ""
|
||||
previous: Span | None = None
|
||||
for span in spans:
|
||||
text = span.text
|
||||
if previous is not None:
|
||||
gap = span.x0 - previous.x1
|
||||
needs_space = (
|
||||
gap >= SPACE_GAP_PT
|
||||
and not out.endswith(" ")
|
||||
and not text.startswith(" ")
|
||||
)
|
||||
if needs_space:
|
||||
out += " "
|
||||
out += text
|
||||
previous = span
|
||||
return out.strip()
|
||||
|
||||
|
||||
def join_spans(spans: Iterable[Span]) -> str:
|
||||
"""Rejoin spans into flowing text.
|
||||
|
||||
A visual line that does not end a sentence is treated as a soft wrap and
|
||||
joined to the next line with a space; a line ending in sentence
|
||||
punctuation keeps its newline, which preserves paragraph and list
|
||||
structure for display and citation.
|
||||
"""
|
||||
lines = [join_visual_line(group) for group in group_visual_lines(list(spans))]
|
||||
lines = [line for line in lines if line]
|
||||
if not lines:
|
||||
return ""
|
||||
|
||||
out = lines[0]
|
||||
for line in lines[1:]:
|
||||
if out.rstrip().endswith(tuple(_SENTENCE_END)):
|
||||
out += "\n" + line
|
||||
else:
|
||||
out += " " + line
|
||||
return out
|
||||
@@ -0,0 +1,32 @@
|
||||
from .assembler import DuplicateDrugIdError, assemble
|
||||
from .atc import ATCResult, extract_atc_codes, is_stated_absent, normalize_atc_candidate
|
||||
from .detector import detect_monograph_titles, detect_section_headings
|
||||
from .io import read_monographs_jsonl, write_monographs_jsonl
|
||||
from .merge import merge_multiline_headings
|
||||
from .models import Heading, Monograph, SectionSpan
|
||||
from .units import normalize_unit_token, validate_unit_tokens
|
||||
from .vocab import SECTION_DEFS, SectionDef, is_part_divider, match_section, normalize_heading_text
|
||||
|
||||
__all__ = [
|
||||
"Heading",
|
||||
"SectionSpan",
|
||||
"Monograph",
|
||||
"assemble",
|
||||
"DuplicateDrugIdError",
|
||||
"detect_monograph_titles",
|
||||
"detect_section_headings",
|
||||
"merge_multiline_headings",
|
||||
"write_monographs_jsonl",
|
||||
"read_monographs_jsonl",
|
||||
"ATCResult",
|
||||
"extract_atc_codes",
|
||||
"is_stated_absent",
|
||||
"normalize_atc_candidate",
|
||||
"normalize_unit_token",
|
||||
"validate_unit_tokens",
|
||||
"SectionDef",
|
||||
"SECTION_DEFS",
|
||||
"match_section",
|
||||
"is_part_divider",
|
||||
"normalize_heading_text",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,485 @@
|
||||
"""Assembles a raw span stream into ordered Monograph records.
|
||||
|
||||
Three simple passes, each independently easy to reason about — avoids a
|
||||
single tangled state machine (SRP: classify, then merge titles, then build):
|
||||
|
||||
1. Classify each span in reading order as a title candidate, a section
|
||||
heading, or body text.
|
||||
2. Coalesce consecutive title-candidate spans into single merged Heading
|
||||
events via `merge.merge_multiline_headings` (handles both the multi-line
|
||||
wrap and same-line font-size-split cases — see merge.py).
|
||||
3. Walk the resulting flat event stream once, building Monograph records.
|
||||
|
||||
Handles the confirmed real "qualifier line" case (outlier-catalog item 18):
|
||||
a monograph title can legitimately repeat (e.g. two distinct "SALBUTAMOL"
|
||||
monographs, "Dùng trong hô hấp" vs "Dùng trong sản khoa") disambiguated by a
|
||||
bold, parenthesized, non-all-caps line directly beneath the title. That
|
||||
qualifier is folded into `drug_id` so two legitimate entries don't collide;
|
||||
a genuine duplicate `drug_id` (no qualifier, same name) raises rather than
|
||||
silently overwriting, since the one apparent duplicate found during ADR
|
||||
0003's investigation (GONADOTROPIN) turned out to be a detector artifact,
|
||||
not real — a real second collision should be surfaced, not hidden.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from typing import Iterator, List, Optional, Union
|
||||
|
||||
from ..extract.models import Span
|
||||
from ..extract.page_map import HEADER_BAND_Y
|
||||
from ..normalize import join_spans, substitute_pua
|
||||
from ..tables.classify import QUARANTINE_SHAPES
|
||||
from .atc import extract_atc_codes
|
||||
from .detector import in_monograph_range, is_monograph_title_candidate
|
||||
from .merge import merge_multiline_headings, merge_same_line_bold_fragments
|
||||
from .models import (
|
||||
PART_PROSE,
|
||||
PART_TABLE,
|
||||
Heading,
|
||||
Monograph,
|
||||
SectionPart,
|
||||
SectionSpan,
|
||||
TableBlock,
|
||||
)
|
||||
from .vocab import (
|
||||
SectionDef,
|
||||
is_part_divider,
|
||||
match_section,
|
||||
match_section_with_inline_value,
|
||||
)
|
||||
|
||||
_QUALIFIER_RE = re.compile(r"^\(.+\)$")
|
||||
|
||||
|
||||
def _is_page_boilerplate(span: Span) -> bool:
|
||||
"""Confirmed real (outlier-catalog item 13, measured via a whole-book
|
||||
`assemble()` run): the running header ("DTQGVN 2" + page number +
|
||||
current monograph name, e.g. physical page 1008's "DTQGVN 2" / "1009" /
|
||||
"Morphin sulfat") was falling through every classification branch below
|
||||
into plain body text, since it matches no section heading and isn't a
|
||||
real all-caps title — silently splicing itself into the *middle* of
|
||||
whatever section happens to be open when a physical page turns (1,374
|
||||
of 11,409 sections / 671 of 682 monographs affected). It's reliably
|
||||
identifiable independent of its (non-vocabulary) text: always the
|
||||
full-page-width block in the header band, same signal `page_map.py`
|
||||
already uses to read the folio.
|
||||
"""
|
||||
return span.column == "full_width" and span.y0 < HEADER_BAND_Y
|
||||
|
||||
|
||||
class DuplicateDrugIdError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _SectionEvent:
|
||||
section_def: SectionDef
|
||||
span: Span
|
||||
inline_value: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _TextEvent:
|
||||
span: Span
|
||||
|
||||
|
||||
_Event = Union[Heading, _SectionEvent, _TextEvent] # Heading == a title event
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
normalized = unicodedata.normalize("NFKD", text)
|
||||
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
|
||||
return re.sub(r"[^a-z0-9]+", "_", ascii_text.lower()).strip("_")
|
||||
|
||||
|
||||
def _is_body_line_that_reads_like_a_label(span: Span, items: List) -> bool:
|
||||
"""A plain line that repeats a section name, sitting under a heading.
|
||||
|
||||
Confirmed real and clinically material: FLUOROURACIL (physical page 681)
|
||||
prints `Thời kỳ mang thai` / `Chống chỉ định.` and `Thời kỳ cho con bú` /
|
||||
`Chống chỉ định.`, verified by rendering the page. The body line matches
|
||||
the section vocabulary, so it was read as a heading — leaving both
|
||||
pregnancy and lactation sections empty and dropping the statement that
|
||||
fluorouracil is contraindicated in both.
|
||||
|
||||
The book never prints an empty section, so a *non-bold* label immediately
|
||||
after a heading is that heading's body. Boldness still cannot be required
|
||||
in general (outlier item 20: `Mã ATC: N06AA09.` is a plain span), which is
|
||||
why this is narrowed to the directly-under-a-heading position.
|
||||
"""
|
||||
if span.bold:
|
||||
return False
|
||||
return bool(items) and isinstance(items[-1], _SectionEvent)
|
||||
|
||||
|
||||
def _classify(spans: List[Span]) -> List[Union[Span, _SectionEvent, _TextEvent]]:
|
||||
"""Pass 1: tag each span. Title candidates are left as raw Span objects
|
||||
(pass 2 groups + merges them); everything else becomes a typed event.
|
||||
|
||||
Section matching does NOT require `span.bold` — confirmed real (outlier
|
||||
item 20): AMITRIPTYLIN's "Mã ATC: N06AA09." is a single **plain, non-bold**
|
||||
span (Abacavir's equivalent is bold "Mã ATC: " + a separate plain value
|
||||
span), inconsistent across the book's ~700 individually-authored
|
||||
monographs (the book's own foreword notes "biên soạn bởi nhiều tác giả").
|
||||
Matching by exact vocabulary text (not styling) is the reliable signal,
|
||||
same lesson as "don't gate on font size" (ADR 0003 item 10) applied to
|
||||
boldness instead.
|
||||
"""
|
||||
items: List[Union[Span, _SectionEvent, _TextEvent]] = []
|
||||
for span in spans:
|
||||
if not span.text.strip():
|
||||
continue
|
||||
if _is_page_boilerplate(span):
|
||||
continue
|
||||
if is_monograph_title_candidate(span):
|
||||
items.append(span)
|
||||
continue
|
||||
section_def = match_section(span.text)
|
||||
if section_def is not None and not _is_body_line_that_reads_like_a_label(
|
||||
span, items
|
||||
):
|
||||
items.append(_SectionEvent(section_def, span))
|
||||
continue
|
||||
if section_def is not None:
|
||||
items.append(_TextEvent(span))
|
||||
continue
|
||||
inline = match_section_with_inline_value(span.text)
|
||||
if inline is not None:
|
||||
items.append(_SectionEvent(inline[0], span, inline_value=inline[1]))
|
||||
else:
|
||||
items.append(_TextEvent(span))
|
||||
return items
|
||||
|
||||
|
||||
def _coalesce_titles(items: List[Union[Span, _SectionEvent, _TextEvent]]) -> List[_Event]:
|
||||
"""Pass 2: merge consecutive raw title-candidate Span runs into single
|
||||
Heading events, preserving the order of everything else.
|
||||
"""
|
||||
events: List[_Event] = []
|
||||
run: List[Span] = []
|
||||
|
||||
def flush_run():
|
||||
if run:
|
||||
events.extend(merge_multiline_headings(list(run)))
|
||||
run.clear()
|
||||
|
||||
for item in items:
|
||||
if isinstance(item, Span):
|
||||
run.append(item)
|
||||
else:
|
||||
flush_run()
|
||||
events.append(item)
|
||||
flush_run()
|
||||
return events
|
||||
|
||||
|
||||
def _is_qualifier_line(span: Span) -> bool:
|
||||
text = span.text.strip()
|
||||
return span.bold and not text.isupper() and bool(_QUALIFIER_RE.match(text))
|
||||
|
||||
|
||||
_ANCHOR_LOOKAHEAD = 6
|
||||
_ANCHOR_SECTION_KEY = "ten_chung_quoc_te"
|
||||
|
||||
|
||||
def _has_anchor_ahead(events: List[_Event], title_index: int) -> bool:
|
||||
"""Every real monograph documents "Tên chung quốc tế" as its very first
|
||||
section (the book's own template, item 2 — see vocab.py docstring).
|
||||
Loosening this to "any known section" was tried and reverted: it let
|
||||
a real, different false positive through (outlier item 21) — individual
|
||||
statin names ("SIMVASTATIN", "LOVASTATIN", ...) are bold+all-caps+short
|
||||
sub-headings *inside* the class-level "CÁC CHẤT ỨC CHẾ HMG-CoA
|
||||
REDUCTASE" monograph, each immediately followed by their own "Liều
|
||||
lượng và cách dùng" sub-section but NOT by "Tên chung quốc tế" (that
|
||||
section belongs only to the parent class monograph) — the loose
|
||||
"any section" check couldn't tell this apart from a real monograph
|
||||
start, but the strict "Tên chung quốc tế specifically" check correctly
|
||||
rejects it, since the specific book-documented template guarantees this
|
||||
exact section is always first for genuine top-level monographs.
|
||||
|
||||
Still correctly rejects the other confirmed false positive (outlier
|
||||
item 19: "HSV"/"CMV" table column headers), which aren't followed by
|
||||
ANY recognized section, let alone this specific one.
|
||||
"""
|
||||
for j in range(title_index + 1, min(title_index + 1 + _ANCHOR_LOOKAHEAD, len(events))):
|
||||
event = events[j]
|
||||
if isinstance(event, Heading) and event.is_monograph_title:
|
||||
return False
|
||||
if isinstance(event, _SectionEvent) and event.section_def.key == _ANCHOR_SECTION_KEY:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _filter_false_positive_titles(events: List[_Event]) -> List[_Event]:
|
||||
"""Pass 2.5: drop title-shaped candidates that aren't followed by any
|
||||
recognized section anchor before the next title candidate.
|
||||
"""
|
||||
return [
|
||||
event for i, event in enumerate(events)
|
||||
if not (isinstance(event, Heading) and event.is_monograph_title)
|
||||
or _has_anchor_ahead(events, i)
|
||||
]
|
||||
|
||||
|
||||
def _region_for(table_index, span: Span):
|
||||
"""The table region a span sits in, if any."""
|
||||
if not table_index:
|
||||
return None
|
||||
for region in table_index.get(span.physical_page, ()):
|
||||
if region.contains(span.x0, span.y0, span.x1, span.y1):
|
||||
return region
|
||||
return None
|
||||
|
||||
|
||||
SPAN_STATE_TEXT = "normalized_text"
|
||||
SPAN_STATE_TABLE = "table"
|
||||
SPAN_STATE_QUARANTINED = "quarantined"
|
||||
SPAN_STATE_BOILERPLATE = "boilerplate_excluded"
|
||||
SPAN_STATE_HEADING = "heading"
|
||||
SPAN_STATE_OUT_OF_SCOPE = "out_of_scope"
|
||||
SPAN_STATE_UNASSIGNED = "unassigned"
|
||||
# Deliberately dropped, not missed: the book's own part-divider titles
|
||||
# ("CÁC CHUYÊN LUẬN THUỐC" etc.) are structure, not content. Reporting them
|
||||
# as `unassigned` would make a clean acceptance target of unassigned == 0
|
||||
# impossible to state honestly.
|
||||
SPAN_STATE_STRUCTURAL = "structural_excluded"
|
||||
|
||||
|
||||
def assemble(spans: List[Span], table_index=None, ledger: Optional[list] = None) -> Iterator[Monograph]:
|
||||
"""Assemble monographs from spans.
|
||||
|
||||
`table_index` maps a physical page to the table regions on it (see
|
||||
`tables.index_by_page`). When supplied, spans falling inside a region are
|
||||
diverted into `Monograph.tables` instead of section prose — measured
|
||||
reason: physical page 109's dosage-form table was otherwise concatenated
|
||||
cell by cell into a section body. Omitting it keeps the previous
|
||||
behaviour, so callers without a region map still work.
|
||||
"""
|
||||
raw_chars = sum(len(s.text) for s in spans)
|
||||
spans = merge_same_line_bold_fragments(spans)
|
||||
events = _filter_false_positive_titles(_coalesce_titles(_classify(spans)))
|
||||
|
||||
# Span-level coverage ledger. Character counts alone cannot balance here
|
||||
# (normalization joins, substitutes and drops characters), so every span
|
||||
# is given a state first and characters are aggregated from that.
|
||||
states: dict = {}
|
||||
if ledger is not None:
|
||||
for s in spans:
|
||||
if not s.text.strip():
|
||||
states[id(s)] = "whitespace_only"
|
||||
elif _is_page_boilerplate(s):
|
||||
states[id(s)] = SPAN_STATE_BOILERPLATE
|
||||
elif is_part_divider(s.text):
|
||||
states[id(s)] = SPAN_STATE_STRUCTURAL
|
||||
elif not in_monograph_range(s):
|
||||
states[id(s)] = SPAN_STATE_OUT_OF_SCOPE
|
||||
elif is_monograph_title_candidate(s):
|
||||
# title spans are merged into a Heading event and lose their
|
||||
# link back to the source span, so they are accounted for here
|
||||
# using the same predicate the classifier uses
|
||||
states[id(s)] = SPAN_STATE_HEADING
|
||||
else:
|
||||
states[id(s)] = SPAN_STATE_UNASSIGNED
|
||||
|
||||
def mark(span: Span, state: str):
|
||||
if ledger is not None:
|
||||
states[id(span)] = state
|
||||
|
||||
seen_ids: set = set()
|
||||
current: Optional[Monograph] = None
|
||||
current_section_key: Optional[str] = None
|
||||
runs: List[tuple] = [] # ordered [(region_or_None, [spans])]
|
||||
inline_prefix: str = ""
|
||||
awaiting_qualifier = False
|
||||
|
||||
def append_span(span: Span, region):
|
||||
"""Keep spans in reading order, starting a new run whenever the
|
||||
prose/table context changes — this is what preserves the real
|
||||
prose -> table -> prose sequence inside one section."""
|
||||
key = region.table_id if region is not None else None
|
||||
if runs and runs[-1][0] == key:
|
||||
runs[-1][1].append(span)
|
||||
else:
|
||||
runs.append((key, [span], region))
|
||||
|
||||
def build_parts() -> List[SectionPart]:
|
||||
parts: List[SectionPart] = []
|
||||
for entry in runs:
|
||||
key, collected = entry[0], entry[1]
|
||||
region = entry[2] if len(entry) > 2 else None
|
||||
if not collected:
|
||||
continue
|
||||
text = substitute_pua(join_spans(collected))
|
||||
if not text.strip():
|
||||
continue
|
||||
pages = [s_.physical_page for s_ in collected]
|
||||
xs0 = min(s_.x0 for s_ in collected); ys0 = min(s_.y0 for s_ in collected)
|
||||
xs1 = max(s_.x1 for s_ in collected); ys1 = max(s_.y1 for s_ in collected)
|
||||
ids = [s_.span_id for s_ in collected]
|
||||
if key is None:
|
||||
parts.append(SectionPart(
|
||||
kind=PART_PROSE, text=text, physical_page=min(pages),
|
||||
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
|
||||
))
|
||||
else:
|
||||
parts.append(SectionPart(
|
||||
kind=PART_TABLE, text=text, physical_page=min(pages),
|
||||
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
|
||||
table_id=key,
|
||||
# deterministic: derived from the first source span, so the
|
||||
# same PDF always produces the same id. A counter suffix
|
||||
# would merely hide a duplicate rather than identify it.
|
||||
table_part_id=f"{key}@{ids[0]}",
|
||||
continuation_group=key,
|
||||
shape=region.shape if region is not None else None,
|
||||
quarantined=(region.shape in QUARANTINE_SHAPES)
|
||||
if region is not None else False,
|
||||
))
|
||||
if inline_prefix:
|
||||
head = SectionPart(
|
||||
kind=PART_PROSE, text=inline_prefix,
|
||||
physical_page=parts[0].physical_page if parts else 0,
|
||||
bbox=parts[0].bbox if parts else [0.0, 0.0, 0.0, 0.0],
|
||||
)
|
||||
parts.insert(0, head)
|
||||
return parts
|
||||
|
||||
def close_current_section():
|
||||
nonlocal runs, inline_prefix
|
||||
if current is not None and current_section_key is None and runs:
|
||||
# spans seen after the title but before any section heading
|
||||
current.preamble.extend(build_parts())
|
||||
if current is not None and current_section_key is not None:
|
||||
existing = current.sections[current_section_key]
|
||||
addition = build_parts()
|
||||
# A section heading can legitimately appear twice inside one
|
||||
# monograph (measured: 33 monographs, 38 occurrences — e.g.
|
||||
# CEFAMANDOL's "Liều lượng và cách dùng" resumes on physical page
|
||||
# 339 after a renal-dosing table). Replacing the SectionSpan here
|
||||
# silently destroyed everything captured before the repeat, so
|
||||
# the parts are concatenated instead. The first heading stays the
|
||||
# section's provenance anchor.
|
||||
combined = list(existing.parts) + addition
|
||||
current.sections[current_section_key] = SectionSpan(
|
||||
key=existing.key, display_name=existing.display_name,
|
||||
heading=existing.heading,
|
||||
# `text` is prose only. Table parts stay in `parts` with their
|
||||
# own provenance and quarantine flag, so anything reading
|
||||
# `.text` (the chunker included) cannot pick up linearised
|
||||
# cells by accident — the ordering is preserved in `parts`.
|
||||
text="\n".join(
|
||||
p_.text for p_ in combined
|
||||
if p_.kind == PART_PROSE and not p_.quarantined and p_.text
|
||||
).strip(),
|
||||
parts=combined,
|
||||
)
|
||||
for part in addition:
|
||||
if part.kind == PART_TABLE:
|
||||
current.tables.append(TableBlock(
|
||||
table_id=part.table_id, shape=part.shape or "",
|
||||
physical_page=part.physical_page, bbox=list(part.bbox),
|
||||
section_key=current_section_key, text=part.text,
|
||||
quarantined=part.quarantined,
|
||||
table_part_id=part.table_part_id,
|
||||
continuation_group=part.continuation_group,
|
||||
source_span_ids=list(part.source_span_ids),
|
||||
))
|
||||
runs = []
|
||||
inline_prefix = ""
|
||||
|
||||
def finalize(monograph: Monograph) -> Monograph:
|
||||
# Duplicate check happens here, not at title-detection time: the
|
||||
# qualifier line (if any) is only known a few events later, so
|
||||
# checking at open-time would false-positive on the legitimate
|
||||
# SALBUTAMOL case (outlier item 18) before the qualifier resolves.
|
||||
if monograph.drug_id in seen_ids:
|
||||
raise DuplicateDrugIdError(
|
||||
f"duplicate drug_id '{monograph.drug_id}' (title '{monograph.drug_name}', "
|
||||
f"physical page {monograph.source_page_range[0]}) — check for a qualifier "
|
||||
f"line (outlier item 18) before assuming this is a real collision"
|
||||
)
|
||||
seen_ids.add(monograph.drug_id)
|
||||
if "ma_atc" in monograph.sections:
|
||||
result = extract_atc_codes(monograph.sections["ma_atc"].text)
|
||||
monograph.atc_codes = result.codes
|
||||
monograph.atc_stated_absent = result.stated_absent
|
||||
return monograph
|
||||
|
||||
for event in events:
|
||||
if isinstance(event, Heading) and event.is_monograph_title:
|
||||
close_current_section()
|
||||
if current is not None:
|
||||
yield finalize(current)
|
||||
current = Monograph(
|
||||
drug_id=_slugify(event.text), drug_name=event.text,
|
||||
source_page_range=[event.physical_page, event.physical_page],
|
||||
)
|
||||
current_section_key = None
|
||||
awaiting_qualifier = True
|
||||
for src in getattr(event, "source_spans", ()) or ():
|
||||
mark(src, SPAN_STATE_HEADING)
|
||||
continue
|
||||
|
||||
if current is None:
|
||||
continue # front matter / general chapters before the first monograph
|
||||
|
||||
if isinstance(event, _SectionEvent):
|
||||
close_current_section()
|
||||
current_section_key = event.section_def.key
|
||||
if event.section_def.key not in current.sections:
|
||||
current.sections[event.section_def.key] = SectionSpan(
|
||||
key=event.section_def.key,
|
||||
display_name=event.section_def.display_name,
|
||||
heading=Heading(
|
||||
text=event.section_def.display_name,
|
||||
physical_page=event.span.physical_page, y0=event.span.y0,
|
||||
is_monograph_title=False, section_key=event.section_def.key,
|
||||
),
|
||||
text="",
|
||||
)
|
||||
if event.inline_value:
|
||||
inline_prefix = event.inline_value
|
||||
awaiting_qualifier = False
|
||||
mark(event.span, SPAN_STATE_HEADING)
|
||||
current.source_page_range[1] = max(current.source_page_range[1], event.span.physical_page)
|
||||
continue
|
||||
|
||||
# _TextEvent
|
||||
span = event.span
|
||||
if awaiting_qualifier and _is_qualifier_line(span):
|
||||
text = span.text.strip()
|
||||
current.drug_id = f"{current.drug_id}_{_slugify(text)}"
|
||||
current.drug_name = f"{current.drug_name} {text}"
|
||||
awaiting_qualifier = False
|
||||
mark(span, SPAN_STATE_HEADING)
|
||||
continue
|
||||
awaiting_qualifier = False
|
||||
|
||||
if not in_monograph_range(span):
|
||||
continue
|
||||
current.source_page_range[1] = max(current.source_page_range[1], span.physical_page)
|
||||
|
||||
region = _region_for(table_index, span)
|
||||
append_span(span, region)
|
||||
if region is not None:
|
||||
mark(span, SPAN_STATE_QUARANTINED
|
||||
if region.shape in QUARANTINE_SHAPES else SPAN_STATE_TABLE)
|
||||
else:
|
||||
mark(span, SPAN_STATE_TEXT)
|
||||
|
||||
if ledger is not None:
|
||||
ledger.append({"raw_chars_before_merge": raw_chars})
|
||||
for s_obj in spans:
|
||||
ledger.append({
|
||||
"state": states[id(s_obj)],
|
||||
"physical_page": s_obj.physical_page,
|
||||
"bbox": [s_obj.x0, s_obj.y0, s_obj.x1, s_obj.y1],
|
||||
"chars": len(s_obj.text),
|
||||
"text": s_obj.text[:60],
|
||||
})
|
||||
|
||||
close_current_section()
|
||||
if current is not None:
|
||||
yield finalize(current)
|
||||
@@ -0,0 +1,135 @@
|
||||
"""ATC-code extraction and normalization.
|
||||
|
||||
Confirmed real text-extraction noise (outlier-catalog item 12c), found while
|
||||
investigating why 22/680 monographs appeared to have zero ATC codes — two
|
||||
distinct causes, both extraction noise rather than missing content:
|
||||
- **Stray internal whitespace** splitting one code into two tokens, e.g.
|
||||
"L01X X02" (should be "L01XX02"), "J04A C01" (should be "J04AC01").
|
||||
- **Digit/letter confusion**: a literal "0" rendered/typeset as "O", e.g.
|
||||
"NO3AX12" (should be "N03AX12").
|
||||
A third, genuinely different outcome: the source text explicitly states
|
||||
"Mã ATC: Chưa có." / "Không có." — a valid "no ATC assigned yet" data state,
|
||||
not an error, and must never be conflated with a parse failure.
|
||||
|
||||
Two more real defects found via a real whole-book `assemble()` run (not
|
||||
assumed, measured against actual monograph text):
|
||||
- **Trailing sentence punctuation**: "Mã ATC: J05AF06." — Abacavir's real
|
||||
field text ends the sentence with a period that isn't part of the code;
|
||||
an earlier version without this fix silently produced zero codes for
|
||||
every single-code monograph ending in ".".
|
||||
- **Per-code parenthetical annotations in multi-ATC monographs**: INSULIN's
|
||||
real field lists all 20 codes each with a species/type note, e.g. "A10AB01
|
||||
(người); A10AB02 (bò); A10AB03 (lợn); ..." — without stripping the
|
||||
trailing "(...)" before length-checking, only 2 of 20 codes survived (the
|
||||
two that happened to have a line-wrap fall between the code and its
|
||||
parenthetical, accidentally isolating the bare code) — a striking example
|
||||
of why this needs whole-corpus validation, not a single clean example.
|
||||
|
||||
A fourth defect, found via the same method (12 vaccine monographs -
|
||||
VẮC XIN SỞI among them - appeared zero-ATC-and-not-stated-absent): the
|
||||
segment split ran *before* parenthetical annotations were stripped, so an
|
||||
annotation containing its own comma broke the split, e.g. "Mã ATC: J07BD01
|
||||
(Measles, live attenuated)." split on "," into "J07BD01 (Measles" and
|
||||
" live attenuated)." — neither a recoverable code shape. INSULIN's
|
||||
Vietnamese annotations ("người", "bò", "lợn") never contain a comma, so
|
||||
this only surfaced with vaccines' English annotations. Fixed by stripping
|
||||
*all* parenthetical groups from the whole field text before splitting,
|
||||
not just a trailing one per already-split segment.
|
||||
|
||||
A fifth defect, same method (15 more monographs, e.g. ALCURONIUM CLORID,
|
||||
AMLODIPIN): some real monographs render the bold section label as "Mã ATC"
|
||||
with no colon, and the colon belongs to the *value* span instead, e.g.
|
||||
bold "Mã ATC" + plain ": M03AA01." (Abacavir's equivalent is bold "Mã ATC:
|
||||
" + plain "J05AF06.", colon on the label side). The heading still matches
|
||||
correctly (`vocab.normalize_heading_text` already strips a trailing
|
||||
colon from either side), but the captured field text keeps the leading
|
||||
": " from the value span, making the stripped candidate 8 characters
|
||||
(":M03AA01") instead of 7 — silently failing the length check. Fixed by
|
||||
taking only the text after the last ":" per segment before normalizing —
|
||||
a strict generalization of the leading-colon strip (see the sixth defect
|
||||
below) that still normalizes a plain "N03AX12" unchanged (no colon to
|
||||
split on).
|
||||
|
||||
A sixth defect, same method (7 monographs with multiple salt/ester forms,
|
||||
e.g. ARGININ, ENALAPRIL, VASOPRESSIN, the INTERFERON and gonadotropin
|
||||
entries): each form is its own "Name: CODE" line rather than a bare code,
|
||||
e.g. "Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01" — the whole
|
||||
segment (including the name) was compared against the 7-character code
|
||||
shape and rejected. Solved by trying the text after the last colon first:
|
||||
"Arginin glutamat: A05BA01" -> "A05BA01".
|
||||
|
||||
A seventh defect, same method (1 monograph, the class-level "CÁC CHẤT ỨC
|
||||
CHẾ HMG-CoA REDUCTASE"): its "Mã ATC" field lists every statin the
|
||||
*opposite* way round, code first — "C10A A01: Simvastatin\nC10A A02:
|
||||
Lovastatin\n..." — so "take the text after the colon" extracts the drug
|
||||
name, not the code. Since a real drug name essentially never happens to
|
||||
match the strict 7-character ATC shape, trying the after-colon part first
|
||||
and falling back to the before-colon part costs nothing for the "Name:
|
||||
CODE" case (defect six) while recovering this reversed "CODE: Name" case
|
||||
too, without needing to special-case either monograph.
|
||||
|
||||
This module returns all three ATC-presence outcomes distinctly (found /
|
||||
recovered-from-noise / stated-absent), never collapsed into one boolean,
|
||||
per the outlier catalog's explicit guidance.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List, Optional
|
||||
|
||||
_ATC_PATTERN = re.compile(r"^[A-Z]\d{2}[A-Z]{2}\d{2}$")
|
||||
_DIGIT_POSITIONS = (1, 2, 5, 6) # 0-indexed positions that must be digits
|
||||
_ABSENT_MARKERS = ("chưa có", "không có")
|
||||
_SEGMENT_SPLIT_RE = re.compile(r"[,;\n]")
|
||||
_PAREN_RE = re.compile(r"\([^()]*\)")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ATCResult:
|
||||
codes: List[str] = field(default_factory=list)
|
||||
stated_absent: bool = False
|
||||
|
||||
|
||||
def is_stated_absent(field_text: str) -> bool:
|
||||
normalized = field_text.strip().lower()
|
||||
return any(marker in normalized for marker in _ABSENT_MARKERS)
|
||||
|
||||
|
||||
def normalize_atc_candidate(raw: str) -> Optional[str]:
|
||||
"""Tries each side of the last ":" (whole string if there is none) as
|
||||
the code, after-colon first since "Name: CODE" is the far more common
|
||||
real shape ("CODE: Name" is confirmed real too, but rare) — returns the
|
||||
first side that normalizes to a valid ATC shape. Normalizing strips
|
||||
trailing sentence punctuation and internal whitespace (fixes the
|
||||
split-token case), then fixes O/0 confusion only at the code's known
|
||||
digit positions (never touches the letter positions, so a genuine "X" in
|
||||
"L01XX02" is left alone). Parenthetical annotations must already be
|
||||
stripped by the caller — see `extract_atc_codes`.
|
||||
"""
|
||||
parts = raw.rsplit(":", 1)
|
||||
candidates = [parts[-1]] if len(parts) == 1 else [parts[1], parts[0]]
|
||||
for part in candidates:
|
||||
stripped = re.sub(r"\s+", "", part.upper()).rstrip(".,;")
|
||||
if len(stripped) != 7:
|
||||
continue
|
||||
chars = list(stripped)
|
||||
for i in _DIGIT_POSITIONS:
|
||||
if chars[i] == "O":
|
||||
chars[i] = "0"
|
||||
candidate = "".join(chars)
|
||||
if _ATC_PATTERN.match(candidate):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def extract_atc_codes(field_text: str) -> ATCResult:
|
||||
if is_stated_absent(field_text):
|
||||
return ATCResult(codes=[], stated_absent=True)
|
||||
without_annotations = _PAREN_RE.sub("", field_text)
|
||||
codes = []
|
||||
for segment in _SEGMENT_SPLIT_RE.split(without_annotations):
|
||||
candidate = normalize_atc_candidate(segment)
|
||||
if candidate:
|
||||
codes.append(candidate)
|
||||
return ATCResult(codes=codes, stated_absent=False)
|
||||
@@ -0,0 +1,89 @@
|
||||
"""Monograph and section boundary detection.
|
||||
|
||||
Validated signal (ADR 0003): monograph titles are bold + all-caps + short
|
||||
line length, scoped to printed pages 99-1496 — font **size** is explicitly
|
||||
NOT part of the rule (a size>=9.8 threshold silently dropped ~15% of real
|
||||
monographs). Section headings are bold spans cross-checked against the
|
||||
known (open/extensible) vocabulary in `vocab.py`, no all-caps requirement
|
||||
(most section headings, e.g. "Chỉ định", are not all-caps).
|
||||
|
||||
Known false positive, explicitly excluded rather than tuned around (outlier
|
||||
item 12d): "CÁC CHUYÊN LUẬN THUỐC" and other part-divider titles sit exactly
|
||||
at the printed-page-99 boundary and are bold + all-caps + short, identical
|
||||
in shape to a real monograph title.
|
||||
|
||||
"All-caps" itself is not 100% reliable either (confirmed real, outlier item
|
||||
21): the class-level monograph "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE" embeds
|
||||
the mixed-case abbreviation "CoA" (Coenzyme A) — a strict `text.isupper()`
|
||||
check silently dropped this entire monograph. `_is_mostly_upper` tolerates
|
||||
a small number of lowercase letters (a strict superset of `isupper()`, so
|
||||
no previously-valid case is excluded) rather than requiring zero.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterator, List
|
||||
|
||||
from ..extract.models import Span
|
||||
from .merge import merge_multiline_headings
|
||||
from .models import Heading
|
||||
from .vocab import is_part_divider, match_section
|
||||
|
||||
MONOGRAPH_PRINTED_PAGE_START = 99
|
||||
MONOGRAPH_PRINTED_PAGE_END = 1496
|
||||
_MIN_TITLE_LEN = 3
|
||||
_MAX_TITLE_LEN = 60
|
||||
_MAX_LOWERCASE_RATIO = 0.10 # HMG-CoA: 1/27 = 3.7% (real title) vs "Mã ATC:": 1/5 = 20% (real
|
||||
# section label, correctly rejected) — a ratio, not an absolute count, is what separates a
|
||||
# long title with one embedded mixed-case abbreviation from a short label with a normal
|
||||
# lowercase diacritic (found via a real regression: an earlier absolute-count version of
|
||||
# this check let "Mã ATC:" through as a false title candidate).
|
||||
|
||||
|
||||
def _is_mostly_upper(text: str) -> bool:
|
||||
letters = [c for c in text if c.isalpha()]
|
||||
if not letters:
|
||||
return False
|
||||
lowercase_ratio = sum(1 for c in letters if c.islower()) / len(letters)
|
||||
return lowercase_ratio <= _MAX_LOWERCASE_RATIO
|
||||
|
||||
|
||||
def in_monograph_range(span: Span) -> bool:
|
||||
return (
|
||||
span.printed_page is not None
|
||||
and MONOGRAPH_PRINTED_PAGE_START <= span.printed_page <= MONOGRAPH_PRINTED_PAGE_END
|
||||
)
|
||||
|
||||
|
||||
def is_monograph_title_candidate(span: Span) -> bool:
|
||||
text = span.text.strip()
|
||||
if not (span.bold and _is_mostly_upper(text)):
|
||||
return False
|
||||
if not (_MIN_TITLE_LEN <= len(text) <= _MAX_TITLE_LEN):
|
||||
return False
|
||||
if not in_monograph_range(span):
|
||||
return False
|
||||
if is_part_divider(text):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def detect_monograph_titles(spans: List[Span]) -> Iterator[Heading]:
|
||||
"""`spans` must be in reading order (as `extract_spans` yields them)."""
|
||||
candidates = [s for s in spans if is_monograph_title_candidate(s)]
|
||||
yield from merge_multiline_headings(candidates)
|
||||
|
||||
|
||||
def detect_section_headings(spans: List[Span]) -> Iterator[Heading]:
|
||||
for span in spans:
|
||||
if not span.bold or not in_monograph_range(span):
|
||||
continue
|
||||
section_def = match_section(span.text)
|
||||
if section_def is None:
|
||||
continue
|
||||
yield Heading(
|
||||
text=section_def.display_name,
|
||||
physical_page=span.physical_page,
|
||||
y0=span.y0,
|
||||
is_monograph_title=False,
|
||||
section_key=section_def.key,
|
||||
)
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Pure I/O boundary for Monograph records — kept separate from detection/
|
||||
assembly logic so those stay testable without disk (Clean Architecture).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from .models import Heading, Monograph, SectionPart, SectionSpan, TableBlock
|
||||
|
||||
|
||||
def _heading_to_dict(h: Heading) -> dict:
|
||||
return {
|
||||
"text": h.text, "physical_page": h.physical_page, "y0": h.y0,
|
||||
"is_monograph_title": h.is_monograph_title, "section_key": h.section_key,
|
||||
}
|
||||
|
||||
|
||||
def _heading_from_dict(d: dict) -> Heading:
|
||||
return Heading(**d)
|
||||
|
||||
|
||||
def _monograph_to_dict(m: Monograph) -> dict:
|
||||
return {
|
||||
"drug_id": m.drug_id,
|
||||
"drug_name": m.drug_name,
|
||||
"source_page_range": m.source_page_range,
|
||||
"atc_codes": m.atc_codes,
|
||||
"atc_stated_absent": m.atc_stated_absent,
|
||||
"sections": {
|
||||
key: {
|
||||
"key": s.key, "display_name": s.display_name,
|
||||
"heading": _heading_to_dict(s.heading), "text": s.text,
|
||||
"parts": [
|
||||
{
|
||||
"kind": p.kind, "text": p.text,
|
||||
"physical_page": p.physical_page, "bbox": p.bbox,
|
||||
"source_span_ids": p.source_span_ids,
|
||||
"table_id": p.table_id, "table_part_id": p.table_part_id,
|
||||
"continuation_group": p.continuation_group,
|
||||
"shape": p.shape, "quarantined": p.quarantined,
|
||||
}
|
||||
for p in s.parts
|
||||
],
|
||||
}
|
||||
for key, s in m.sections.items()
|
||||
},
|
||||
"preamble": [
|
||||
{
|
||||
"kind": p.kind, "text": p.text, "physical_page": p.physical_page,
|
||||
"bbox": p.bbox, "source_span_ids": p.source_span_ids,
|
||||
"quarantined": p.quarantined,
|
||||
}
|
||||
for p in m.preamble
|
||||
],
|
||||
"tables": [
|
||||
{
|
||||
"table_id": t.table_id, "shape": t.shape,
|
||||
"table_part_id": t.table_part_id,
|
||||
"continuation_group": t.continuation_group,
|
||||
"source_span_ids": t.source_span_ids,
|
||||
"physical_page": t.physical_page, "bbox": t.bbox,
|
||||
"section_key": t.section_key, "text": t.text,
|
||||
"quarantined": t.quarantined,
|
||||
}
|
||||
for t in m.tables
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _monograph_from_dict(d: dict) -> Monograph:
|
||||
sections = {
|
||||
key: SectionSpan(
|
||||
key=s["key"], display_name=s["display_name"],
|
||||
heading=_heading_from_dict(s["heading"]), text=s["text"],
|
||||
parts=[
|
||||
SectionPart(
|
||||
kind=p["kind"], text=p["text"],
|
||||
physical_page=p["physical_page"], bbox=p["bbox"],
|
||||
source_span_ids=p.get("source_span_ids", []),
|
||||
table_id=p.get("table_id"), table_part_id=p.get("table_part_id"),
|
||||
continuation_group=p.get("continuation_group"),
|
||||
shape=p.get("shape"), quarantined=p.get("quarantined", False),
|
||||
)
|
||||
for p in s.get("parts", [])
|
||||
],
|
||||
)
|
||||
for key, s in d["sections"].items()
|
||||
}
|
||||
return Monograph(
|
||||
drug_id=d["drug_id"], drug_name=d["drug_name"],
|
||||
source_page_range=d["source_page_range"], sections=sections,
|
||||
atc_codes=d.get("atc_codes", []), atc_stated_absent=d.get("atc_stated_absent", False),
|
||||
preamble=[
|
||||
SectionPart(
|
||||
kind=p["kind"], text=p["text"], physical_page=p["physical_page"],
|
||||
bbox=p["bbox"], source_span_ids=p.get("source_span_ids", []),
|
||||
quarantined=p.get("quarantined", False),
|
||||
)
|
||||
for p in d.get("preamble", [])
|
||||
],
|
||||
tables=[
|
||||
TableBlock(
|
||||
table_id=t["table_id"], shape=t["shape"],
|
||||
physical_page=t["physical_page"], bbox=t["bbox"],
|
||||
section_key=t.get("section_key"), text=t["text"],
|
||||
quarantined=t.get("quarantined", False),
|
||||
table_part_id=t.get("table_part_id"),
|
||||
continuation_group=t.get("continuation_group"),
|
||||
source_span_ids=t.get("source_span_ids", []),
|
||||
)
|
||||
for t in d.get("tables", [])
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def write_monographs_jsonl(monographs: Iterable[Monograph], path: Path) -> int:
|
||||
count = 0
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
for m in monographs:
|
||||
f.write(json.dumps(_monograph_to_dict(m), ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
yield _monograph_from_dict(json.loads(line))
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Multi-line monograph-title merging, and same-line bold-run reassembly.
|
||||
|
||||
Two distinct real fragmentation shapes were confirmed, both requiring merge:
|
||||
|
||||
1. **Multi-line wrap** (ADR 0003's original finding, dominant cause of its
|
||||
recall gap and of the GONADOTROPIN false-collision, outlier item 11):
|
||||
long titles wrap across 2+ physical lines, e.g. physical page 1371 has
|
||||
"THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG" (y0=664.46) immediately followed by
|
||||
"GONADOTROPIN" (y0=676.24) — a ~11.8pt line-height step, same page.
|
||||
2. **Same-line font-size split, found by visually inspecting a real page**
|
||||
(physical page 113, rendered to an image and read directly — not
|
||||
inferred from coordinates alone): "ACICLOVIR" is split into two spans,
|
||||
"ACIC" (size 10.0) and "LOVIR" (size 9.5), touching with a ~0.5pt y0
|
||||
difference and near-zero x-gap. An earlier version of this module
|
||||
required exact font-size equality to merge, which correctly handled
|
||||
case 1 (GONADOTROPIN: both fragments size 9.5) but silently missed case
|
||||
2 — the same "font size is not reliable" lesson from ADR 0003 applies
|
||||
*within* a single title's fragments, not just across different
|
||||
monographs. Fixed by dropping the size-equality requirement; the y-gap
|
||||
+ same-page check alone is sufficient (a real next-monograph title is
|
||||
always much farther down the page/on a different page, given a full
|
||||
monograph's worth of section content in between).
|
||||
|
||||
The join character between merged fragments must differ by case: case 1
|
||||
needs a space (distinct words across a real line wrap); case 2 needs no
|
||||
space (mid-word split, "ACIC" + "LOVIR" = "ACICLOVIR", not "ACIC LOVIR").
|
||||
Distinguished by the y0 gap: small (<= `_SAME_LINE_Y_TOLERANCE`) means same
|
||||
visual line -> concatenate directly; larger means a real new line -> join
|
||||
with a space.
|
||||
|
||||
Candidate spans passed in here are already filtered by the caller (bold +
|
||||
all-caps + short + in the monograph page range) — this module only decides
|
||||
which *consecutive* candidates belong to the same title and how to join them.
|
||||
|
||||
A third, unrelated fragmentation shape was confirmed via a whole-book
|
||||
`cli validate` run against the back-of-book index (5 real monographs -
|
||||
GUAIFENESIN, MEPHENESIN, NATRI THIOSULFAT, RAMIPRIL, TENOXICAM - silently
|
||||
dropped): PyMuPDF splits some bold section-heading runs into several spans
|
||||
around diacritic characters even though the text is a single, visually
|
||||
unbroken line in the rendered page (confirmed by rendering physical page
|
||||
759 to an image and reading it directly — "Tên chung quốc tế" looks
|
||||
completely normal to a human reader; the fragmentation exists only in
|
||||
PyMuPDF's span boundaries, not the document). Confirmed page 759's actual
|
||||
spans: "Tên chung qu" (y0=157.614), "ố" (y0=157.33), "c t" (y0=157.614),
|
||||
"ế" (y0=157.33), ": " (y0=157.614) — all within `_SAME_LINE_Y_TOLERANCE`,
|
||||
so `merge_same_line_bold_fragments` (applied to *all* bold spans, not just
|
||||
title candidates, before section-vocabulary matching) reassembles them the
|
||||
same way case 2 above reassembles "ACIC" + "LOVIR".
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
from typing import Iterator, List
|
||||
|
||||
from ..extract.models import Span
|
||||
from .models import Heading
|
||||
|
||||
_MAX_LINE_GAP_PT = 20.0 # comfortably above the confirmed ~11.8pt wrap case
|
||||
_SAME_LINE_Y_TOLERANCE = 3.0 # comfortably above the confirmed ~0.5pt same-line split
|
||||
|
||||
|
||||
def merge_same_line_bold_fragments(spans: List[Span]) -> List[Span]:
|
||||
"""Reassemble consecutive bold spans that PyMuPDF split mid-line (same
|
||||
page, same visual line) back into one span, so downstream section-vocab
|
||||
matching sees the real text instead of a diacritic-boundary fragment.
|
||||
|
||||
Non-bold spans and spans on different lines pass through unchanged.
|
||||
Provenance (page/column/block/line/span_index/y-position/font/size) is
|
||||
kept from the first fragment; only `text` and `x1` are updated, so the
|
||||
merged span still traces back to its exact source region.
|
||||
"""
|
||||
merged: List[Span] = []
|
||||
buffer: List[Span] = []
|
||||
|
||||
def flush():
|
||||
if not buffer:
|
||||
return
|
||||
if len(buffer) == 1:
|
||||
merged.append(buffer[0])
|
||||
else:
|
||||
merged.append(dataclasses.replace(
|
||||
buffer[0], text="".join(s.text for s in buffer), x1=buffer[-1].x1,
|
||||
))
|
||||
|
||||
for span in spans:
|
||||
same_line_bold_run = (
|
||||
buffer and span.bold and buffer[-1].bold
|
||||
and span.physical_page == buffer[-1].physical_page
|
||||
and abs(span.y0 - buffer[-1].y0) <= _SAME_LINE_Y_TOLERANCE
|
||||
)
|
||||
if same_line_bold_run:
|
||||
buffer.append(span)
|
||||
else:
|
||||
flush()
|
||||
buffer = [span]
|
||||
flush()
|
||||
return merged
|
||||
|
||||
|
||||
def _same_title_run(prev: Span, curr: Span) -> bool:
|
||||
gap = curr.y0 - prev.y0
|
||||
return curr.physical_page == prev.physical_page and 0 <= gap <= _MAX_LINE_GAP_PT
|
||||
|
||||
|
||||
def merge_multiline_headings(candidates: List[Span]) -> Iterator[Heading]:
|
||||
"""`candidates` must already be in reading order (as extract_spans
|
||||
yields them) and pre-filtered to heading candidates only.
|
||||
"""
|
||||
buffer: List[Span] = []
|
||||
for span in candidates:
|
||||
if buffer and _same_title_run(buffer[-1], span):
|
||||
buffer.append(span)
|
||||
else:
|
||||
if buffer:
|
||||
yield _flush(buffer)
|
||||
buffer = [span]
|
||||
if buffer:
|
||||
yield _flush(buffer)
|
||||
|
||||
|
||||
def _flush(buffer: List[Span]) -> Heading:
|
||||
parts = [buffer[0].text.strip()]
|
||||
for prev, curr in zip(buffer, buffer[1:], strict=False):
|
||||
same_line = abs(curr.y0 - prev.y0) <= _SAME_LINE_Y_TOLERANCE
|
||||
parts.append("" if same_line else " ")
|
||||
parts.append(curr.text.strip())
|
||||
first = buffer[0]
|
||||
return Heading(
|
||||
text="".join(parts),
|
||||
physical_page=first.physical_page,
|
||||
y0=first.y0,
|
||||
is_monograph_title=True,
|
||||
)
|
||||
@@ -0,0 +1,95 @@
|
||||
"""Data model for segmented output, matching docs/architecture.md's contract:
|
||||
{drug_id, drug_name, source_page_range, sections: {...}}
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Heading:
|
||||
text: str
|
||||
physical_page: int
|
||||
y0: float
|
||||
is_monograph_title: bool
|
||||
section_key: Optional[str] = None
|
||||
|
||||
|
||||
PART_PROSE = "prose"
|
||||
PART_TABLE = "table"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionPart:
|
||||
"""One contiguous run of a section, in reading order.
|
||||
|
||||
A section is not uniformly prose: a dosing section routinely reads
|
||||
prose -> table -> prose. Flattening that to a single string loses both the
|
||||
ordering and the ability to say which part a sentence came from, so the
|
||||
parts are kept in sequence with their own provenance.
|
||||
"""
|
||||
kind: str
|
||||
text: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
source_span_ids: List[str] = field(default_factory=list)
|
||||
table_id: Optional[str] = None
|
||||
table_part_id: Optional[str] = None
|
||||
continuation_group: Optional[str] = None
|
||||
shape: Optional[str] = None
|
||||
quarantined: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionSpan:
|
||||
key: str
|
||||
display_name: str
|
||||
heading: Heading
|
||||
text: str
|
||||
parts: List[SectionPart] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def prose_text(self) -> str:
|
||||
"""Only the parts safe to read as prose — excludes quarantined ones."""
|
||||
return "\n".join(
|
||||
p.text for p in self.parts
|
||||
if p.kind == PART_PROSE and not p.quarantined and p.text
|
||||
).strip()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TableBlock:
|
||||
"""Text lifted out of a table region, kept beside the prose instead of
|
||||
inside it.
|
||||
|
||||
`quarantined` marks content whose flattened text is actively misleading
|
||||
(a 2D lookup grid means nothing without its row and column headers) —
|
||||
such a block must not be embedded or cited as if it were prose.
|
||||
"""
|
||||
table_id: str
|
||||
shape: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
section_key: Optional[str]
|
||||
text: str
|
||||
quarantined: bool = False
|
||||
table_part_id: Optional[str] = None
|
||||
continuation_group: Optional[str] = None
|
||||
source_span_ids: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Monograph:
|
||||
drug_id: str
|
||||
drug_name: str
|
||||
source_page_range: List[int]
|
||||
sections: Dict[str, SectionSpan] = field(default_factory=dict)
|
||||
atc_codes: List[str] = field(default_factory=list)
|
||||
atc_stated_absent: bool = False
|
||||
tables: List[TableBlock] = field(default_factory=list)
|
||||
# Text between the monograph title and its first section heading. Real and
|
||||
# clinically important — e.g. ARTEMETHER (physical page 210) opens with the
|
||||
# regulatory notice that single-agent artemisinin products were withdrawn
|
||||
# to limit resistance. It belongs to no section, so it was being dropped.
|
||||
preamble: List[SectionPart] = field(default_factory=list)
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Dosing-unit token validation (mg/mcg/mmol/g/ml).
|
||||
|
||||
Unlike `atc.py`'s whitespace-split defect (confirmed with real examples,
|
||||
outlier-catalog item 12c), a targeted regex scan of the full monograph page
|
||||
range (99-1496 printed) for the analogous unit-token pattern (a unit like
|
||||
"mg" split into "m g" by a stray internal space) found **zero occurrences**
|
||||
— this is NOT a confirmed defect in this corpus. This module exists as a
|
||||
defensive check by analogy, per the project's explicit dosing-safety
|
||||
requirement: a silent mg/mcg confusion is a 1000x dosing error, and the
|
||||
book's own "Người lớn"/"Trẻ em" dosing-population split appears on the
|
||||
majority of monograph pages (outlier-catalog item 17), so the cost of an
|
||||
undetected unit-token corruption is high enough to check for even without a
|
||||
confirmed prior occurrence — but callers must not describe what this module
|
||||
guards against as "a confirmed real defect," only as a validated absence
|
||||
plus a standing defensive gate.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
_KNOWN_UNITS = ("mg", "mcg", "mmol", "microgam", "g", "ml", "iu", "đvqt")
|
||||
_UNIT_PATTERN = re.compile(
|
||||
"^(" + "|".join(re.escape(u) for u in _KNOWN_UNITS) + ")$", re.IGNORECASE
|
||||
)
|
||||
|
||||
|
||||
def normalize_unit_token(raw: str) -> Optional[str]:
|
||||
"""Strips internal whitespace (defends against a stray-space split, the
|
||||
same failure class as the confirmed ATC whitespace-split defect) and
|
||||
validates against the known dosing-unit vocabulary. Returns the
|
||||
lowercase canonical unit string, or None if unrecognized.
|
||||
"""
|
||||
stripped = re.sub(r"\s+", "", raw).lower()
|
||||
return stripped if _UNIT_PATTERN.match(stripped) else None
|
||||
|
||||
|
||||
def validate_unit_tokens(tokens: list) -> "list[str]":
|
||||
"""Returns the subset of `tokens` that fail normalization — callers use
|
||||
this to flag a dosing section for manual review, not to silently drop
|
||||
or auto-correct (unlike ATC codes, there is no confirmed-safe recovery
|
||||
rule here since no real corruption pattern has been observed yet).
|
||||
"""
|
||||
return [t for t in tokens if normalize_unit_token(t) is None]
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Section-name taxonomy for drug monographs.
|
||||
|
||||
Canonical list transcribed directly from the book's own documented template
|
||||
(physical page 38, printed page 39, "HƯỚNG DẪN SỬ DỤNG DƯỢC THƯ QUỐC GIA
|
||||
VIỆT NAM") and cross-checked against real bold headings in the Abacavir/
|
||||
Acarbose monographs (physical pages 100-102). The book documents 19 fields
|
||||
per monograph, of which #1 ("Tên chuyên luận thuốc") is the monograph title
|
||||
itself (handled by `detector.detect_monograph_titles`, not a section) —
|
||||
leaving 18 documented sections. `ten_thuong_mai` ("Tên thương mại") is a
|
||||
19th, real, but *undocumented* field confirmed present in real monographs
|
||||
(outlier-catalog item 12) — open/closed taxonomy: add new entries here as
|
||||
they're found, never change the matching logic in detector.py.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionDef:
|
||||
key: str
|
||||
display_name: str
|
||||
# Real spelling variants observed in the book itself. The source is not
|
||||
# typographically consistent: it prints "qui chế" 469 times against the
|
||||
# documented "quy chế", and carries assorted typos ("Mã ACT", "sử trí",
|
||||
# "Chống chỉ đinh"). Whitespace and look-alike-character differences are
|
||||
# NOT listed here — `_lookup_key` folds those away for every entry at
|
||||
# once, so this stays a list of genuinely different wordings.
|
||||
aliases: Tuple[str, ...] = ()
|
||||
|
||||
@property
|
||||
def labels(self) -> Tuple[str, ...]:
|
||||
return (self.display_name,) + self.aliases
|
||||
|
||||
|
||||
SECTION_DEFS = [
|
||||
SectionDef("ten_chung_quoc_te", "Tên chung quốc tế", ("Ten chung quốc tế",)),
|
||||
SectionDef("ma_atc", "Mã ATC", ("Mã ACT",)),
|
||||
SectionDef("loai_thuoc", "Loại thuốc", ("Loại thuôc", "Lọai thuốc", "Phân loại thuốc")),
|
||||
SectionDef("dang_thuoc_va_ham_luong", "Dạng thuốc và hàm lượng",
|
||||
("Dạng dùng và hàm lượng",)),
|
||||
SectionDef("duoc_ly_va_co_che_tac_dung", "Dược lý và cơ chế tác dụng",
|
||||
("Dược lí và cơ chế tác dụng", "Dược lý học và cơ chế tác dụng")),
|
||||
SectionDef("chi_dinh", "Chỉ định"),
|
||||
SectionDef("chong_chi_dinh", "Chống chỉ định", ("Chống chỉ đinh",)),
|
||||
SectionDef("than_trong", "Thận trọng"),
|
||||
SectionDef("thoi_ky_mang_thai", "Thời kỳ mang thai", ("Thời kì mang thai",)),
|
||||
SectionDef("thoi_ky_cho_con_bu", "Thời kỳ cho con bú", ("Thời kì cho con bú",)),
|
||||
SectionDef("tac_dung_khong_mong_muon", "Tác dụng không mong muốn (ADR)",
|
||||
("Tác dụng không mong muốn", "Tác dụng không mong muốn ADR")),
|
||||
SectionDef("huong_dan_xu_tri_adr", "Hướng dẫn cách xử trí ADR",
|
||||
("Hướng dẫn xử trí ADR", "Hướng dẫn cách sử trí ADR",
|
||||
"Hướng dẫn cách xử trí các ADR")),
|
||||
SectionDef("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
("Liều lượng cách dùng", "Liều lượng, cách dùng",
|
||||
"Liều dùng và cách dùng", "Liều lượng và cách sử dụng")),
|
||||
SectionDef("tuong_tac_thuoc", "Tương tác thuốc"),
|
||||
SectionDef("do_on_dinh_va_bao_quan", "Độ ổn định và bảo quản"),
|
||||
SectionDef("tuong_ky", "Tương kỵ"),
|
||||
SectionDef("qua_lieu_va_xu_tri", "Quá liều và xử trí",
|
||||
("Quá liều và cách xử trí", "Quá liều và xử lý",
|
||||
"Quá liều cấp tính và xử trí")),
|
||||
SectionDef("thong_tin_quy_che", "Thông tin quy chế",
|
||||
("Thông tin qui chế", "Thông tin về qui chế", "Thông tin và quy chế")),
|
||||
SectionDef("ten_thuong_mai", "Tên thương mại"),
|
||||
]
|
||||
|
||||
# Near-miss strings deliberately NOT treated as section headings, recorded so
|
||||
# a later reader does not "helpfully" add them: "Thể trọng" is body weight,
|
||||
# not "Thận trọng" (caution); "Tác dụng không mong muốn của opioid" is a
|
||||
# drug-specific sub-heading inside a section, not the section itself.
|
||||
REJECTED_NEAR_MISSES = frozenset({"Thể trọng", "Tác dụng không mong muốn của opioid"})
|
||||
|
||||
# Part/section-divider titles (from the book's own table of contents) that
|
||||
# are bold + all-caps + short, exactly like a monograph title, but are NOT
|
||||
# drug monographs — confirmed false positive, outlier-catalog item 12d.
|
||||
PART_DIVIDER_TITLES = {
|
||||
"CÁC CHUYÊN LUẬN CHUNG",
|
||||
"CÁC CHUYÊN LUẬN THUỐC",
|
||||
"CÁC PHỤ LỤC",
|
||||
}
|
||||
|
||||
_TRAILING_PUNCT_RE = re.compile(r"[:.\s]+$")
|
||||
_WHITESPACE_RE = re.compile(r"\s+")
|
||||
_ALL_WHITESPACE_RE = re.compile(r"\s")
|
||||
|
||||
# Look-alike characters the typesetting mixes with their correct forms:
|
||||
# U+00D0 LATIN CAPITAL LETTER ETH is used where U+0110 LATIN CAPITAL LETTER D
|
||||
# WITH STROKE belongs ("Ðộ ổn định" vs "Độ ổn định"), and NFC does not unify
|
||||
# them because they are genuinely distinct codepoints that merely look alike.
|
||||
_CONFUSABLES = str.maketrans({"Ð": "Đ", "ð": "đ"})
|
||||
|
||||
|
||||
def normalize_heading_text(text: str) -> str:
|
||||
"""Strip trailing colon/period/whitespace and collapse internal runs so
|
||||
"Tên chung quốc tế:" and "Tên chung quốc tế" (both observed verbatim in
|
||||
real monographs) render the same. Preserves single spaces — this is the
|
||||
display form, not the lookup form.
|
||||
"""
|
||||
normalized = _TRAILING_PUNCT_RE.sub("", text.strip())
|
||||
return _WHITESPACE_RE.sub(" ", normalized)
|
||||
|
||||
|
||||
def _lookup_key(text: str) -> str:
|
||||
"""Fold away the differences that are typesetting noise, not wording.
|
||||
|
||||
The source splits and joins headings inconsistently — "Chỉđịnh",
|
||||
"H ướng dẫn cách xử trí ADR", "Tác dụng khôngmong muốn (ADR)" and
|
||||
"Độổn định và bảo quản" all appear — so whitespace is removed entirely
|
||||
rather than enumerated as aliases. Case and look-alike characters are
|
||||
folded for the same reason.
|
||||
"""
|
||||
folded = unicodedata.normalize("NFC", normalize_heading_text(text))
|
||||
folded = folded.translate(_CONFUSABLES)
|
||||
return _ALL_WHITESPACE_RE.sub("", folded).lower()
|
||||
|
||||
|
||||
_LOOKUP: Dict[str, SectionDef] = {
|
||||
_lookup_key(label): d for d in SECTION_DEFS for label in d.labels
|
||||
}
|
||||
|
||||
|
||||
def match_section(text: str) -> Optional[SectionDef]:
|
||||
return _LOOKUP.get(_lookup_key(text))
|
||||
|
||||
|
||||
# Sorted longest-label-first so a prefix check never matches a shorter
|
||||
# label that happens to also be a prefix of a longer one (none currently
|
||||
# collide, but this is a cheap, permanent safety property to keep).
|
||||
_PREFIX_CANDIDATES: list = sorted(
|
||||
((label, d) for d in SECTION_DEFS for label in d.labels),
|
||||
key=lambda pair: -len(pair[0]),
|
||||
)
|
||||
|
||||
|
||||
def match_section_with_inline_value(text: str) -> Optional[Tuple[SectionDef, str]]:
|
||||
"""Handles a real, confirmed structural variant (outlier item 20):
|
||||
some monographs render a section heading and its value as ONE
|
||||
non-bold, non-separated span, e.g. AMITRIPTYLIN's "Mã ATC: N06AA09."
|
||||
(Abacavir's equivalent is bold "Mã ATC: " + separate plain "J05AF06.").
|
||||
Returns (matched section, remaining value text) or None.
|
||||
"""
|
||||
stripped = text.strip()
|
||||
for label, section_def in _PREFIX_CANDIDATES:
|
||||
if stripped[: len(label)].lower() != label.lower():
|
||||
continue
|
||||
remainder = stripped[len(label):].lstrip()
|
||||
if remainder.startswith(":"):
|
||||
return section_def, remainder[1:].strip()
|
||||
return None
|
||||
|
||||
|
||||
def is_part_divider(text: str) -> bool:
|
||||
return normalize_heading_text(text).upper() in PART_DIVIDER_TITLES
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Table stage: detect tabular regions so their text stops leaking into prose.
|
||||
|
||||
Scope note: this stage locates and classifies table *regions*. Reconstructing
|
||||
correct rows and columns is deliberately not attempted here — see ADR 0003
|
||||
and outlier-catalog items 5-7 for why that is a separate, harder problem.
|
||||
"""
|
||||
from .classify import (
|
||||
QUARANTINE_SHAPES,
|
||||
SHAPE_CROSS_PAGE,
|
||||
SHAPE_FORMULA_2D,
|
||||
SHAPE_GRID_2D,
|
||||
SHAPE_MULTI_HEADER,
|
||||
SHAPE_SINGLE_COLUMN_BOXED,
|
||||
SHAPE_NOT_TABLE_FULL_PAGE,
|
||||
SHAPE_SIMPLE,
|
||||
classify_shape,
|
||||
)
|
||||
from .detect import detect_table_regions
|
||||
from .io import index_by_page, read_regions_json, write_regions_json
|
||||
from .models import TableRegion
|
||||
|
||||
__all__ = [
|
||||
"QUARANTINE_SHAPES",
|
||||
"SHAPE_CROSS_PAGE",
|
||||
"SHAPE_FORMULA_2D",
|
||||
"SHAPE_GRID_2D",
|
||||
"SHAPE_MULTI_HEADER",
|
||||
"SHAPE_SINGLE_COLUMN_BOXED",
|
||||
"SHAPE_NOT_TABLE_FULL_PAGE",
|
||||
"SHAPE_SIMPLE",
|
||||
"TableRegion",
|
||||
"classify_shape",
|
||||
"detect_table_regions",
|
||||
"index_by_page",
|
||||
"read_regions_json",
|
||||
"write_regions_json",
|
||||
]
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Shape classification for detected table regions.
|
||||
|
||||
Whole-corpus measurement: of 200 regions `pdfplumber.find_tables()` reports,
|
||||
22 are not tables at all (17 cover a whole page — e.g. the copyright page —
|
||||
and 5 are single-column text blocks such as the epilepsy classification
|
||||
list). Routing every region through one generic reconstructor would treat
|
||||
those 22 as tables, so shape is decided first and handling follows from it.
|
||||
|
||||
Open/closed: adding a shape means adding a rule here, not editing callers.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
PAGE_WIDTH, PAGE_HEIGHT = 595.3, 841.9
|
||||
FULL_PAGE_AREA_RATIO = 0.75
|
||||
|
||||
SHAPE_SIMPLE = "simple_table"
|
||||
SHAPE_MULTI_HEADER = "multi_level_or_merged_header"
|
||||
SHAPE_CROSS_PAGE = "cross_page_continuation"
|
||||
SHAPE_GRID_2D = "grid_2d_numeric"
|
||||
SHAPE_NOT_TABLE_FULL_PAGE = "not_a_table_full_page"
|
||||
SHAPE_SINGLE_COLUMN_BOXED = "single_column_boxed_list"
|
||||
|
||||
# Not produced by table detection — a stacked fraction is not a table — but it
|
||||
# is the same kind of object as far as assembly is concerned: a rectangle whose
|
||||
# spans must be lifted out of prose rather than run together. Measured on
|
||||
# NETILMICIN (physical page 1042) and AMPICILIN VÀ SULBACTAM (202): linearised,
|
||||
# the numerator lands before the '=' and the division reads as multiplication.
|
||||
SHAPE_FORMULA_2D = "formula_2d"
|
||||
|
||||
# Shapes whose flattened text must not be embedded or cited as prose until a
|
||||
# real row/column reconstruction exists. Any multi-column table loses its
|
||||
# cell semantics when linearised — a 2D lookup grid most severely (its values
|
||||
# are meaningless without both headers, outlier-catalog item 7), but a plain
|
||||
# dosing table is no safer to quote once its columns are run together.
|
||||
# `single_column_boxed_list` is excluded deliberately: one column linearises
|
||||
# correctly, so it reads as ordinary text (physical page 55's "Bảng 2").
|
||||
QUARANTINE_SHAPES = frozenset({
|
||||
SHAPE_GRID_2D,
|
||||
SHAPE_SIMPLE,
|
||||
SHAPE_MULTI_HEADER,
|
||||
SHAPE_CROSS_PAGE,
|
||||
SHAPE_FORMULA_2D,
|
||||
})
|
||||
|
||||
|
||||
def _area_ratio(bbox) -> float:
|
||||
x0, y0, x1, y1 = bbox
|
||||
return abs((x1 - x0) * (y1 - y0)) / (PAGE_WIDTH * PAGE_HEIGHT)
|
||||
|
||||
|
||||
def classify_shape(
|
||||
bbox,
|
||||
n_rows: int,
|
||||
n_cols: int,
|
||||
first_row: List[str],
|
||||
starts_near_top: bool,
|
||||
all_cells_numeric: bool,
|
||||
) -> str:
|
||||
if _area_ratio(bbox) >= FULL_PAGE_AREA_RATIO:
|
||||
return SHAPE_NOT_TABLE_FULL_PAGE
|
||||
# Only a single *column* is degenerate. A single ROW with several columns
|
||||
# is the opposite of degenerate — it is the orphaned continuation row of
|
||||
# a table broken across a page (outlier-catalog item 5), the case where
|
||||
# losing the content is most damaging because a row without its header
|
||||
# cannot be interpreted. Verified visually: physical pages 62 and 72 are
|
||||
# exactly this (1x3, with cell rules visible), and an earlier version of
|
||||
# this rule discarded both as "not a table".
|
||||
# One column inside a ruled box. Structurally not a row/column table, but
|
||||
# the book may still number it as one — physical page 55 is captioned
|
||||
# "Bảng 2: Phân loại quốc tế các cơn động kinh (1989)" and is a nested
|
||||
# numbered list drawn inside a frame. Named for what it is rather than
|
||||
# "not a table": single-column content linearises correctly and must stay
|
||||
# in the text, unlike a real 2D table.
|
||||
if n_cols <= 1:
|
||||
return SHAPE_SINGLE_COLUMN_BOXED
|
||||
if n_rows <= 1:
|
||||
return SHAPE_CROSS_PAGE
|
||||
if all_cells_numeric and n_cols >= 4:
|
||||
return SHAPE_GRID_2D
|
||||
|
||||
cells = [(c or "").strip() for c in first_row]
|
||||
textual = sum(
|
||||
1 for c in cells
|
||||
if c and not c.replace(",", "").replace(".", "").replace("-", "").isdigit()
|
||||
)
|
||||
if starts_near_top and textual <= 1:
|
||||
return SHAPE_CROSS_PAGE
|
||||
if cells and any(not c for c in cells) and textual >= 1:
|
||||
return SHAPE_MULTI_HEADER
|
||||
return SHAPE_SIMPLE
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Table region detection.
|
||||
|
||||
`pdfplumber` is used here and nowhere else in the pipeline: ADR 0003 records
|
||||
that its general text extraction scrambles reading order on this document,
|
||||
so it is kept strictly to table geometry, where it is the only tool that
|
||||
works. PyMuPDF remains the sole text extractor.
|
||||
|
||||
Detection is slow (≈17 minutes over the 1668-page book), so the result is
|
||||
written once to a region map and reused — see `io.py`. The detection itself
|
||||
lives here, in the pipeline, rather than in a throwaway script.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Iterator, List
|
||||
|
||||
import pdfplumber
|
||||
|
||||
from .classify import classify_shape
|
||||
from .models import TableRegion
|
||||
|
||||
TOP_BAND_Y = 120.0
|
||||
|
||||
|
||||
def _all_numeric(data) -> bool:
|
||||
values = [(c or "").strip() for row in data for c in row]
|
||||
values = [v for v in values if v]
|
||||
if not values:
|
||||
return False
|
||||
return all(
|
||||
v.replace(",", "").replace(".", "").replace("-", "").isdigit()
|
||||
for v in values
|
||||
)
|
||||
|
||||
|
||||
def detect_table_regions(pdf_path: Path) -> Iterator[TableRegion]:
|
||||
with pdfplumber.open(pdf_path) as pdf:
|
||||
for page_number, page in enumerate(pdf.pages):
|
||||
try:
|
||||
found = page.find_tables()
|
||||
except Exception:
|
||||
continue
|
||||
for index, table in enumerate(found):
|
||||
data = table.extract() or []
|
||||
first_row: List[str] = [
|
||||
(c or "").strip() for c in (data[0] if data else [])
|
||||
]
|
||||
n_rows = len(data)
|
||||
n_cols = max((len(r) for r in data), default=0)
|
||||
bbox = tuple(round(v, 1) for v in table.bbox)
|
||||
yield TableRegion(
|
||||
table_id=f"p{page_number}_t{index}",
|
||||
physical_page=page_number,
|
||||
bbox=bbox,
|
||||
n_rows=n_rows,
|
||||
n_cols=n_cols,
|
||||
shape=classify_shape(
|
||||
bbox=bbox,
|
||||
n_rows=n_rows,
|
||||
n_cols=n_cols,
|
||||
first_row=first_row,
|
||||
starts_near_top=bbox[1] < TOP_BAND_Y,
|
||||
all_cells_numeric=_all_numeric(data),
|
||||
),
|
||||
first_row=first_row[:8],
|
||||
)
|
||||
# pdfplumber caches every parsed object per page; without this the
|
||||
# 1668-page book grows the process past 6 GB and the run dies on
|
||||
# a paging-file error rather than finishing.
|
||||
page.flush_cache()
|
||||
page.get_textmap.cache_clear()
|
||||
@@ -0,0 +1,42 @@
|
||||
"""Filesystem boundary for the table stage."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List
|
||||
|
||||
from .models import TableRegion
|
||||
|
||||
|
||||
def write_regions_json(regions: Iterable[TableRegion], path: Path) -> int:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
rows = [asdict(r) for r in regions]
|
||||
path.write_text(json.dumps(rows, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
return len(rows)
|
||||
|
||||
|
||||
def read_regions_json(path: Path) -> List[TableRegion]:
|
||||
rows = json.loads(path.read_text(encoding="utf-8"))
|
||||
return [
|
||||
TableRegion(
|
||||
table_id=r["table_id"],
|
||||
physical_page=r["physical_page"],
|
||||
bbox=tuple(r["bbox"]),
|
||||
n_rows=r["n_rows"],
|
||||
n_cols=r["n_cols"],
|
||||
shape=r["shape"],
|
||||
first_row=r.get("first_row", []),
|
||||
)
|
||||
for r in rows
|
||||
]
|
||||
|
||||
|
||||
def index_by_page(regions: Iterable[TableRegion]) -> Dict[int, List[TableRegion]]:
|
||||
"""Group real table regions by page for O(1) lookup during assembly."""
|
||||
index: Dict[int, List[TableRegion]] = {}
|
||||
for region in regions:
|
||||
if not region.is_real_table:
|
||||
continue
|
||||
index.setdefault(region.physical_page, []).append(region)
|
||||
return index
|
||||
@@ -0,0 +1,39 @@
|
||||
"""Table region model.
|
||||
|
||||
A region is a rectangle on one page that holds tabular content. It is
|
||||
deliberately separate from the table's *contents*: the pipeline's first
|
||||
obligation is to stop tabular text leaking into prose (measured: page 109's
|
||||
dosage-form table was being concatenated cell-by-cell into a section body),
|
||||
which needs only the geometry. Reconstructing rows and columns correctly is
|
||||
a later, harder step.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List, Tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TableRegion:
|
||||
table_id: str
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
n_rows: int
|
||||
n_cols: int
|
||||
shape: str
|
||||
first_row: List[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def is_real_table(self) -> bool:
|
||||
return not self.shape.startswith("not_a_table")
|
||||
|
||||
def contains(self, x0: float, y0: float, x1: float, y1: float) -> bool:
|
||||
"""True when a span's box lies (mostly) inside this region.
|
||||
|
||||
Uses the span's centre rather than full containment: PyMuPDF span
|
||||
boxes and pdfplumber table boxes come from different engines and
|
||||
disagree by a point or two at the edges.
|
||||
"""
|
||||
cx, cy = (x0 + x1) / 2, (y0 + y1) / 2
|
||||
left, top, right, bottom = self.bbox
|
||||
return left <= cx <= right and top <= cy <= bottom
|
||||
@@ -0,0 +1,51 @@
|
||||
from .back_index import GroundTruthEntry, parse_back_index
|
||||
from .metrics import RecallPrecisionResult, compute_recall_precision
|
||||
from .readiness import (
|
||||
Gate,
|
||||
corpus_size,
|
||||
evaluate,
|
||||
evaluate_chunks,
|
||||
read_chunks,
|
||||
read_monographs,
|
||||
)
|
||||
from .residual_ink import (
|
||||
ANTIALIAS_SPECK,
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
HEADER_BAND_FRAGMENT,
|
||||
HEADER_RULE,
|
||||
RULE_FRAGMENT,
|
||||
TABLE_FRAME,
|
||||
TEXT_AS_VECTOR_OUTLINE,
|
||||
UNCLASSIFIED,
|
||||
PageContext,
|
||||
ResidualRegion,
|
||||
classify,
|
||||
scan_document,
|
||||
scan_page,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"GroundTruthEntry",
|
||||
"parse_back_index",
|
||||
"RecallPrecisionResult",
|
||||
"compute_recall_precision",
|
||||
"Gate",
|
||||
"evaluate",
|
||||
"evaluate_chunks",
|
||||
"read_chunks",
|
||||
"corpus_size",
|
||||
"read_monographs",
|
||||
"PageContext",
|
||||
"ResidualRegion",
|
||||
"classify",
|
||||
"scan_page",
|
||||
"scan_document",
|
||||
"HEADER_RULE",
|
||||
"TABLE_FRAME",
|
||||
"TEXT_AS_VECTOR_OUTLINE",
|
||||
"FRACTION_BAR_CANDIDATE",
|
||||
"HEADER_BAND_FRAGMENT",
|
||||
"RULE_FRAGMENT",
|
||||
"ANTIALIAS_SPECK",
|
||||
"UNCLASSIFIED",
|
||||
]
|
||||
@@ -0,0 +1,52 @@
|
||||
"""Parses the book's own "Mục lục tra cứu" (back-of-book index) into
|
||||
page-verified ground truth — per ADR 0003, this is the correct validation
|
||||
source (exact page numbers per generic name), not the front-matter drug list
|
||||
(no page numbers).
|
||||
|
||||
Real format confirmed by reading physical pages 1530+ directly:
|
||||
- Genuine generic-name entries: "Abacavir, 101" (name, comma, printed page).
|
||||
- Brand-name cross-references: "Ziagen - Abacavir, 101" / "ABAB -
|
||||
Paracetamol, 1118" (brand " - " generic, page) — skipped for ground
|
||||
truth, per ADR 0003.
|
||||
- Section-letter headers ("A", "B", ...) and running header/footer
|
||||
boilerplate lines don't match the entry pattern and are naturally
|
||||
ignored, not specially cased.
|
||||
|
||||
Known limitation, inherited from the already-validated ADR 0003 approach
|
||||
(not newly introduced here): a handful of genuine compound-name entries in
|
||||
the book use " - " *within* the generic name itself (e.g. "Carbidopa -
|
||||
levodopa"), which this parser's cross-reference exclusion will also skip —
|
||||
the same trade-off the original 91.7%-recall validation already made
|
||||
successfully, not re-litigated here.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import List
|
||||
|
||||
import fitz
|
||||
|
||||
BACK_INDEX_START_PHYSICAL = 1530 # printed 1531 — first page of real entries ("A" section)
|
||||
|
||||
_ENTRY_RE = re.compile(r"^(.+?),\s*(\d+)\s*$")
|
||||
_CROSS_REF_MARKER = " - "
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GroundTruthEntry:
|
||||
name: str
|
||||
printed_page: int
|
||||
|
||||
|
||||
def parse_back_index(doc: fitz.Document, start_physical_page: int = BACK_INDEX_START_PHYSICAL) -> List[GroundTruthEntry]:
|
||||
entries: List[GroundTruthEntry] = []
|
||||
for pno in range(start_physical_page, doc.page_count):
|
||||
for line in doc[pno].get_text().split("\n"):
|
||||
line = line.strip()
|
||||
if not line or _CROSS_REF_MARKER in line:
|
||||
continue
|
||||
match = _ENTRY_RE.match(line)
|
||||
if match:
|
||||
entries.append(GroundTruthEntry(name=match.group(1).strip(), printed_page=int(match.group(2))))
|
||||
return entries
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Monograph-boundary recall/precision against the back-of-book index.
|
||||
|
||||
Per ADR 0003, only recall was ever measured before (91.7%, 665/725) — this
|
||||
module adds precision (never measured previously) alongside recall, per the
|
||||
approved eval-framework plan.
|
||||
|
||||
Page comparison: `GroundTruthEntry.printed_page` is a *printed* page number;
|
||||
`Monograph.source_page_range` is *physical*. The physical->printed offset
|
||||
was empirically confirmed constant (+1) across every tested milestone page
|
||||
in Phase 1.1 (`extract/page_map.py`) — reused here rather than re-derived.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import List
|
||||
|
||||
from ..segment.models import Monograph
|
||||
from .back_index import GroundTruthEntry
|
||||
|
||||
PRINTED_PAGE_OFFSET = 1
|
||||
PAGE_TOLERANCE = 2
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecallPrecisionResult:
|
||||
recall: float
|
||||
precision: float
|
||||
matched_count: int
|
||||
total_ground_truth: int
|
||||
total_detected: int
|
||||
unmatched_ground_truth: List[GroundTruthEntry]
|
||||
unmatched_detected: List[Monograph]
|
||||
|
||||
|
||||
_WHITESPACE_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _normalize_name(name: str) -> str:
|
||||
# collapse-whitespace: confirmed real case — "ALVERIN CITRAT" (double
|
||||
# space, likely a genuine PDF-rendering artifact) failed to match
|
||||
# ground truth's "Alverin citrat" under plain strip+upper, found via a
|
||||
# real `cli validate` run (4 of 12 unmatched-detected monographs had
|
||||
# this exact shape: ALVERIN CITRAT, OXYMETAZOLIN HYDROCLORID,
|
||||
# TERBUTALIN SULFAT, TIOTROPIUM BROMID).
|
||||
return _WHITESPACE_RE.sub(" ", name.strip()).upper()
|
||||
|
||||
|
||||
def _monograph_start_printed_page(monograph: Monograph) -> int:
|
||||
return monograph.source_page_range[0] + PRINTED_PAGE_OFFSET
|
||||
|
||||
|
||||
def _names_match(entry_name: str, drug_name: str) -> bool:
|
||||
a, b = _normalize_name(entry_name), _normalize_name(drug_name)
|
||||
return a in b or b in a
|
||||
|
||||
|
||||
def _names_match_exactly(entry_name: str, drug_name: str) -> bool:
|
||||
return _normalize_name(entry_name) == _normalize_name(drug_name)
|
||||
|
||||
|
||||
def _find_match(entry: GroundTruthEntry, monographs: List[Monograph]):
|
||||
# Exact match first, substring fallback only if no exact match exists:
|
||||
# confirmed real case, "Isosorbid" and "Isosorbid dinitrat" are two
|
||||
# distinct real monographs a page apart. A substring-only search finds
|
||||
# "Isosorbid" for BOTH ground-truth entries (it's a substring of
|
||||
# "Isosorbid dinitrat" too) and, being first in page order, wins via
|
||||
# `next()` for both — leaving the real "Isosorbid dinitrat" monograph
|
||||
# spuriously unmatched. Same shape confirmed for "Ampicilin" /
|
||||
# "Ampicilin và sulbactam". Trying each entry's exact match across all
|
||||
# monographs before falling back to substring resolves both without
|
||||
# needing order-dependent tie-breaking.
|
||||
in_tolerance = [
|
||||
m for m in monographs
|
||||
if abs(_monograph_start_printed_page(m) - entry.printed_page) <= PAGE_TOLERANCE
|
||||
]
|
||||
return next(
|
||||
(m for m in in_tolerance if _names_match_exactly(entry.name, m.drug_name)),
|
||||
next((m for m in in_tolerance if _names_match(entry.name, m.drug_name)), None),
|
||||
)
|
||||
|
||||
|
||||
def compute_recall_precision(
|
||||
monographs: List[Monograph], ground_truth: List[GroundTruthEntry],
|
||||
) -> RecallPrecisionResult:
|
||||
matched_gt = []
|
||||
unmatched_gt = []
|
||||
matched_detected_ids: set = set()
|
||||
|
||||
for entry in ground_truth:
|
||||
match = _find_match(entry, monographs)
|
||||
if match is not None:
|
||||
matched_gt.append(entry)
|
||||
matched_detected_ids.add(match.drug_id)
|
||||
else:
|
||||
unmatched_gt.append(entry)
|
||||
|
||||
unmatched_detected = [m for m in monographs if m.drug_id not in matched_detected_ids]
|
||||
return RecallPrecisionResult(
|
||||
recall=len(matched_gt) / len(ground_truth) if ground_truth else 0.0,
|
||||
precision=len(matched_detected_ids) / len(monographs) if monographs else 0.0,
|
||||
matched_count=len(matched_gt),
|
||||
total_ground_truth=len(ground_truth),
|
||||
total_detected=len(monographs),
|
||||
unmatched_ground_truth=unmatched_gt,
|
||||
unmatched_detected=unmatched_detected,
|
||||
)
|
||||
@@ -0,0 +1,202 @@
|
||||
"""Named gates that must hold before the corpus is chunked.
|
||||
|
||||
Chunking bakes whatever it is given into embeddings, where defects stop being
|
||||
inspectable. So the question this module answers is not "did the pipeline
|
||||
run" but "is the text going in actually the text on the page". Each gate is
|
||||
reported on its own line with its own number and its own target — a single
|
||||
pass/fail would hide exactly the problems that took a whole session to find.
|
||||
|
||||
Every gate here is computed from the artefacts, never remembered from an
|
||||
earlier run: quoting a number from before a code change is the specific
|
||||
mistake this project keeps catching.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List, Sequence
|
||||
|
||||
PUA_RANGE = (0xE000, 0xF8FF)
|
||||
REPLACEMENT_CHAR = "�"
|
||||
|
||||
# Strings that were confirmed by eye to be corruption, each traced to a
|
||||
# dropped vector-outlined glyph (outlier-catalog item 24). They are checked
|
||||
# literally: if one reappears, the repair regressed.
|
||||
KNOWN_CORRUPTIONS = (
|
||||
"Độ n định",
|
||||
"≥ 1 tu i",
|
||||
"tại ch :",
|
||||
)
|
||||
|
||||
# Fragments of 2D formulas that must never sit in prose, where the missing
|
||||
# fraction bar turns a division into a multiplication.
|
||||
FORMULA_FRAGMENTS = (
|
||||
"Thể trọng (kg)",
|
||||
"(140 - tuổi) x cân nặng",
|
||||
"x (140 - số tuổi)",
|
||||
"Giá trị Clcr của bệnh nhân",
|
||||
"218 x P x",
|
||||
"× trọng lượng cơ thể (kg)",
|
||||
"Cân nặng (kg) x liều",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Gate:
|
||||
name: str
|
||||
count: int
|
||||
target: int = 0
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def passed(self) -> bool:
|
||||
return self.count == self.target
|
||||
|
||||
|
||||
def _section_texts(monograph: dict) -> Iterable[str]:
|
||||
for section in (monograph.get("sections") or {}).values():
|
||||
yield section.get("text") or ""
|
||||
|
||||
|
||||
def _count_pua(text: str) -> int:
|
||||
return sum(1 for ch in text if PUA_RANGE[0] <= ord(ch) <= PUA_RANGE[1])
|
||||
|
||||
|
||||
def evaluate(monographs: Sequence[dict],
|
||||
transcribed_runs: Sequence[dict] = ()) -> List[Gate]:
|
||||
"""Compute every readiness gate over the whole corpus."""
|
||||
pua = replacement = empty = no_provenance = 0
|
||||
corruptions: Dict[str, int] = {c: 0 for c in KNOWN_CORRUPTIONS}
|
||||
formula_leaks: Dict[str, int] = {f: 0 for f in FORMULA_FRAGMENTS}
|
||||
unflagged_blocks = 0
|
||||
ids: Dict[str, int] = {}
|
||||
no_page_range = 0
|
||||
corpus = []
|
||||
|
||||
for monograph in monographs:
|
||||
ids[monograph["drug_id"]] = ids.get(monograph["drug_id"], 0) + 1
|
||||
if not monograph.get("source_page_range"):
|
||||
no_page_range += 1
|
||||
for section in (monograph.get("sections") or {}).values():
|
||||
text = section.get("text") or ""
|
||||
corpus.append(text)
|
||||
if not text.strip():
|
||||
empty += 1
|
||||
if not section.get("parts"):
|
||||
no_provenance += 1
|
||||
pua += _count_pua(text)
|
||||
replacement += text.count(REPLACEMENT_CHAR)
|
||||
for phrase in KNOWN_CORRUPTIONS:
|
||||
corruptions[phrase] += text.count(phrase)
|
||||
for phrase in FORMULA_FRAGMENTS:
|
||||
formula_leaks[phrase] += text.count(phrase)
|
||||
for block in monograph.get("tables") or []:
|
||||
if not block.get("quarantined"):
|
||||
unflagged_blocks += 1
|
||||
|
||||
joined = "\n".join(corpus)
|
||||
unmerged = [
|
||||
run for run in transcribed_runs
|
||||
if len(run["text"].strip()) > 2 and run["text"].strip() not in joined
|
||||
]
|
||||
|
||||
return [
|
||||
Gate("outlined_run_not_merged", len(unmerged),
|
||||
detail="; ".join(f"p{r['physical_page']} {r['text'][:40]!r}"
|
||||
for r in unmerged[:5])),
|
||||
Gate("known_corruption_string", sum(corruptions.values()),
|
||||
detail=", ".join(f"{k!r}={v}" for k, v in corruptions.items() if v)),
|
||||
Gate("formula_fragment_in_prose", sum(formula_leaks.values()),
|
||||
detail=", ".join(f"{k!r}={v}" for k, v in formula_leaks.items() if v)),
|
||||
Gate("pua_char", pua),
|
||||
Gate("replacement_char_ufffd", replacement),
|
||||
Gate("empty_section", empty),
|
||||
Gate("section_without_provenance", no_provenance),
|
||||
Gate("unflagged_quarantine_block", unflagged_blocks),
|
||||
Gate("duplicate_drug_id", sum(1 for n in ids.values() if n > 1)),
|
||||
Gate("monograph_without_page_range", no_page_range),
|
||||
]
|
||||
|
||||
|
||||
def corpus_size(monographs: Sequence[dict]) -> Dict[str, int]:
|
||||
"""Informational, not a gate: how much text chunking would consume."""
|
||||
sections = [t for m in monographs for t in _section_texts(m)]
|
||||
return {
|
||||
"monographs": len(monographs),
|
||||
"sections": len(sections),
|
||||
"section_chars": sum(len(t) for t in sections),
|
||||
"quarantined_blocks": sum(len(m.get("tables") or []) for m in monographs),
|
||||
}
|
||||
|
||||
|
||||
def read_monographs(path: Path) -> List[dict]:
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
return [json.loads(line) for line in handle if line.strip()]
|
||||
|
||||
|
||||
def evaluate_chunks(monographs: Sequence[dict],
|
||||
chunks: Sequence[dict]) -> List[Gate]:
|
||||
"""ADR 0006 gates: a chunk must never hide that a block was lifted.
|
||||
|
||||
The failure being guarded against is silent, not visible: a chunk of
|
||||
AMPICILIN VÀ SULBACTAM's dosing section is grammatical, complete-looking
|
||||
prose with the renal-dosing table absent and nothing marking the absence.
|
||||
Measured: 127 of 167 lifted blocks came out of `liều lượng và cách dùng`.
|
||||
"""
|
||||
blocks_by_section: Dict[tuple, list] = {}
|
||||
block_ids: Dict[str, str] = {}
|
||||
block_texts: Dict[str, str] = {}
|
||||
for monograph in monographs:
|
||||
for block in monograph.get("tables") or []:
|
||||
key = (monograph["drug_id"], block.get("section_key"))
|
||||
blocks_by_section.setdefault(key, []).append(block)
|
||||
block_ids[block["table_id"]] = monograph["drug_id"]
|
||||
if block.get("text"):
|
||||
block_texts[block["table_id"]] = block["text"]
|
||||
|
||||
referenced: Dict[tuple, set] = {}
|
||||
unknown_id = missing_provenance = leaked = 0
|
||||
descriptors = 0
|
||||
descriptor_without_attachment = 0
|
||||
|
||||
for chunk in chunks:
|
||||
attachments = chunk.get("attachments") or []
|
||||
if chunk.get("chunk_kind") == "block_descriptor":
|
||||
descriptors += 1
|
||||
if not attachments:
|
||||
descriptor_without_attachment += 1
|
||||
key = (chunk["drug_id"], chunk["section_key"])
|
||||
for attachment in attachments:
|
||||
referenced.setdefault(key, set()).add(attachment["block_id"])
|
||||
if block_ids.get(attachment["block_id"]) != chunk["drug_id"]:
|
||||
unknown_id += 1
|
||||
if attachment.get("physical_page") is None or not attachment.get("bbox"):
|
||||
missing_provenance += 1
|
||||
body = chunk.get("text") or ""
|
||||
for attachment in attachments:
|
||||
source = block_texts.get(attachment["block_id"], "")
|
||||
probe = source.strip()[:60]
|
||||
if len(probe) > 20 and probe in body:
|
||||
leaked += 1
|
||||
|
||||
unreferenced = 0
|
||||
for key, blocks in blocks_by_section.items():
|
||||
seen = referenced.get(key, set())
|
||||
unreferenced += sum(1 for b in blocks if b["table_id"] not in seen)
|
||||
|
||||
total_blocks = sum(len(v) for v in blocks_by_section.values())
|
||||
return [
|
||||
Gate("section_block_without_chunk_reference", unreferenced),
|
||||
Gate("attachment_block_id_unknown", unknown_id),
|
||||
Gate("attachment_without_page_or_bbox", missing_provenance),
|
||||
Gate("block_text_leaked_into_chunk_text", leaked),
|
||||
Gate("descriptor_chunk_without_attachment", descriptor_without_attachment),
|
||||
Gate("descriptor_count_vs_block_count", descriptors, target=total_blocks,
|
||||
detail=f"{descriptors} descriptors for {total_blocks} blocks"),
|
||||
]
|
||||
|
||||
|
||||
def read_chunks(path: Path) -> List[dict]:
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
return [json.loads(line) for line in handle if line.strip()]
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Residual-ink coverage check: what is on the page that the text layer never emitted.
|
||||
|
||||
Every other check in this project asks a detector whether it found something.
|
||||
This one asks the page. It renders each page, whites out every pixel covered
|
||||
by a span the extractor actually produced, and reports the ink that survives.
|
||||
Whatever survives is content the text layer cannot account for — vector
|
||||
rules, fraction bars, figures.
|
||||
|
||||
Why it earns its place: the two confirmed 2D-formula corruptions
|
||||
(NETILMICIN physical page 1042, AMPICILIN VÀ SULBACTAM physical page 202)
|
||||
are invisible to both table detectors in this repo — `pdfplumber` reports 0
|
||||
regions on those pages and so does `opendataloader-pdf`. The two tools share
|
||||
a blind spot because both need ruling lines. Pixels do not share it: the
|
||||
fraction bar is ink, so it survives the mask and gets reported.
|
||||
|
||||
The check needs no ground truth and no sampling — measured at 0.06 s/page,
|
||||
so all 1668 pages run in under two minutes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, Iterable, Iterator, List, Sequence, Tuple
|
||||
|
||||
import fitz
|
||||
import numpy as np
|
||||
from scipy import ndimage
|
||||
|
||||
from ..extract.outlined_text import OutlinedTextRun, detect_outlined_text
|
||||
from ..tables.models import TableRegion
|
||||
|
||||
RENDER_DPI = 150
|
||||
INK_THRESHOLD = 200
|
||||
|
||||
# Measured, not guessed: at 1.0pt the mask eats the fraction bar itself —
|
||||
# page 1042's bar survives as 9.1pt of its true 188.6pt. At 0.5pt the full
|
||||
# bar survives, and a 10-page prose sample produced the same region count as
|
||||
# 1.0pt (11 regions), i.e. the looser padding adds no noise.
|
||||
MASK_PAD_PT = 0.5
|
||||
|
||||
THIN_HEIGHT_PT = 3.0
|
||||
HEADER_BAND_PT = 60.0
|
||||
RULE_MIN_WIDTH_PT = 400.0
|
||||
BAR_MIN_WIDTH_PT = 10.0
|
||||
|
||||
# A glyph outline can extend a fraction past the filled path's own box, so the
|
||||
# overlap test is given room: without it, three ink fragments on page 714 sit
|
||||
# just outside their line's box and read as unexplained text.
|
||||
OUTLINE_TOLERANCE_PT = 2.0
|
||||
|
||||
# Smaller than any mark a real glyph leaves. Measured against the 1,054
|
||||
# components of confirmed outlined text on the five affected pages: the
|
||||
# smallest is well above this, so the rule cannot swallow real text.
|
||||
SPECK_EXTENT_PT = 2.0
|
||||
|
||||
HEADER_RULE = "header_rule"
|
||||
TABLE_FRAME = "table_frame"
|
||||
TEXT_AS_VECTOR_OUTLINE = "text_as_vector_outline"
|
||||
FRACTION_BAR_CANDIDATE = "fraction_bar_candidate"
|
||||
HEADER_BAND_FRAGMENT = "header_band_fragment"
|
||||
RULE_FRAGMENT = "rule_fragment"
|
||||
ANTIALIAS_SPECK = "antialias_speck"
|
||||
UNCLASSIFIED = "unclassified"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResidualRegion:
|
||||
"""Ink left on a page after masking every extracted span.
|
||||
|
||||
Provenance is the point: `physical_page` + `bbox` locate the region in the
|
||||
source PDF exactly, so any verdict about it can be re-checked by eye.
|
||||
"""
|
||||
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
ink_px: int
|
||||
|
||||
@property
|
||||
def width_pt(self) -> float:
|
||||
return self.bbox[2] - self.bbox[0]
|
||||
|
||||
@property
|
||||
def height_pt(self) -> float:
|
||||
return self.bbox[3] - self.bbox[1]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PageContext:
|
||||
"""What else is known to be on the page, for naming residual ink.
|
||||
|
||||
Carried as one object so a new kind of context is a new field here rather
|
||||
than a new positional argument on every predicate.
|
||||
"""
|
||||
|
||||
tables: Sequence[TableRegion] = ()
|
||||
outlined_runs: Sequence[OutlinedTextRun] = ()
|
||||
|
||||
|
||||
def _overlaps(bbox, other) -> bool:
|
||||
return not (bbox[2] < other[0] or bbox[0] > other[2]
|
||||
or bbox[3] < other[1] or bbox[1] > other[3])
|
||||
|
||||
|
||||
def _is_header_rule(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return (
|
||||
region.height_pt <= THIN_HEIGHT_PT
|
||||
and region.bbox[1] < HEADER_BAND_PT
|
||||
and region.width_pt >= RULE_MIN_WIDTH_PT
|
||||
)
|
||||
|
||||
|
||||
def _is_table_frame(region: ResidualRegion, context: PageContext) -> bool:
|
||||
return any(table.contains(*region.bbox) for table in context.tables)
|
||||
|
||||
|
||||
def _is_outlined_text(region: ResidualRegion, context: PageContext) -> bool:
|
||||
pad = OUTLINE_TOLERANCE_PT
|
||||
return any(
|
||||
_overlaps(region.bbox,
|
||||
(line.bbox[0] - pad, line.bbox[1] - pad,
|
||||
line.bbox[2] + pad, line.bbox[3] + pad))
|
||||
for line in context.outlined_runs
|
||||
)
|
||||
|
||||
|
||||
def _is_fraction_bar(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return region.height_pt <= THIN_HEIGHT_PT and region.width_pt >= BAR_MIN_WIDTH_PT
|
||||
|
||||
|
||||
def _is_header_band_fragment(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
"""Leftovers of the running-header rule, chopped up by the text over it.
|
||||
|
||||
Confirmed by eye on physical page 382: a 31.7 x 9.6pt L-shape that is the
|
||||
header rule meeting a vertical tick, split into its own component because
|
||||
the header text's mask cut the rule either side of it.
|
||||
"""
|
||||
return region.bbox[3] <= HEADER_BAND_PT
|
||||
|
||||
|
||||
def _is_rule_fragment(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return min(region.width_pt, region.height_pt) <= THIN_HEIGHT_PT
|
||||
|
||||
|
||||
def _is_speck(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return (region.width_pt < SPECK_EXTENT_PT
|
||||
and region.height_pt < SPECK_EXTENT_PT)
|
||||
|
||||
|
||||
# Open/closed: a new residual kind is a new entry here, not an edit to the
|
||||
# existing predicates. Order matters — first match wins. Outlined text is
|
||||
# tested before the fraction-bar shape rule, which its underline-like
|
||||
# fragments would otherwise satisfy; the header band is tested before it too,
|
||||
# because the header text's mask cuts the running rule into short pieces that
|
||||
# are bar-shaped (8 of them on physical page 382 alone).
|
||||
_RULES: List[Tuple[str, Callable[[ResidualRegion, PageContext], bool]]] = [
|
||||
(HEADER_RULE, _is_header_rule),
|
||||
(TABLE_FRAME, _is_table_frame),
|
||||
(TEXT_AS_VECTOR_OUTLINE, _is_outlined_text),
|
||||
(HEADER_BAND_FRAGMENT, _is_header_band_fragment),
|
||||
(FRACTION_BAR_CANDIDATE, _is_fraction_bar),
|
||||
(ANTIALIAS_SPECK, _is_speck),
|
||||
(RULE_FRAGMENT, _is_rule_fragment),
|
||||
]
|
||||
|
||||
|
||||
def classify(region: ResidualRegion, context: PageContext | None = None) -> str:
|
||||
"""Name what a residual region is. Pure — no PDF, no rendering."""
|
||||
context = context or PageContext()
|
||||
for kind, predicate in _RULES:
|
||||
if predicate(region, context):
|
||||
return kind
|
||||
return UNCLASSIFIED
|
||||
|
||||
|
||||
def _ink_boxes(mask: "np.ndarray") -> Iterator[Tuple[int, int, int, int]]:
|
||||
"""One box per connected blob of surviving ink.
|
||||
|
||||
Two cheaper splits were tried first and both misreport real pages. Cutting
|
||||
into horizontal bands only merges a table in the left column with one in
|
||||
the right column, so the merged box's centre lands in the gutter, matches
|
||||
no table region, and physical page 209's ADR table is reported as
|
||||
unaccounted-for ink. Adding a column-run split then cuts a single table
|
||||
grid into its individual rules, because masking the text leaves the rules
|
||||
standing with empty gaps between them. A table grid is one connected
|
||||
object and a fraction bar is another, so connectivity is the property that
|
||||
actually separates them.
|
||||
"""
|
||||
labelled, _ = ndimage.label(mask, structure=np.ones((3, 3), dtype=bool))
|
||||
for top_bottom, left_right in ndimage.find_objects(labelled) or []:
|
||||
yield left_right.start, top_bottom.start, left_right.stop - 1, top_bottom.stop - 1
|
||||
|
||||
|
||||
def scan_page(page: "fitz.Page", dpi: int = RENDER_DPI) -> List[ResidualRegion]:
|
||||
"""Render one page, mask its extracted spans, return the surviving ink."""
|
||||
scale = dpi / 72.0
|
||||
pixmap = page.get_pixmap(dpi=dpi, colorspace=fitz.csGRAY)
|
||||
image = np.frombuffer(pixmap.samples, dtype=np.uint8).reshape(
|
||||
pixmap.height, pixmap.width
|
||||
).copy()
|
||||
|
||||
for block in page.get_text("dict")["blocks"]:
|
||||
for line in block.get("lines", []):
|
||||
for span in line["spans"]:
|
||||
x0, y0, x1, y1 = span["bbox"]
|
||||
top = max(0, int((y0 - MASK_PAD_PT) * scale))
|
||||
bottom = min(pixmap.height, int((y1 + MASK_PAD_PT) * scale) + 1)
|
||||
left = max(0, int((x0 - MASK_PAD_PT) * scale))
|
||||
right = min(pixmap.width, int((x1 + MASK_PAD_PT) * scale) + 1)
|
||||
image[top:bottom, left:right] = 255
|
||||
|
||||
mask = image < INK_THRESHOLD
|
||||
return [
|
||||
ResidualRegion(
|
||||
physical_page=page.number,
|
||||
bbox=(
|
||||
round(left / scale, 2),
|
||||
round(top / scale, 2),
|
||||
round(right / scale, 2),
|
||||
round(bottom / scale, 2),
|
||||
),
|
||||
ink_px=int(mask[top:bottom + 1, left:right + 1].sum()),
|
||||
)
|
||||
for left, top, right, bottom in _ink_boxes(mask)
|
||||
]
|
||||
|
||||
|
||||
def scan_document(
|
||||
doc: "fitz.Document",
|
||||
tables_by_page: Dict[int, List[TableRegion]] | None = None,
|
||||
pages: Iterable[int] | None = None,
|
||||
) -> Iterator[Tuple[ResidualRegion, str]]:
|
||||
"""Yield every residual region in the document with its classification."""
|
||||
tables_by_page = tables_by_page or {}
|
||||
page_numbers = list(range(doc.page_count) if pages is None else pages)
|
||||
|
||||
outlines: Dict[int, List[OutlinedTextRun]] = {}
|
||||
for line in detect_outlined_text(doc, page_numbers):
|
||||
outlines.setdefault(line.physical_page, []).append(line)
|
||||
|
||||
for number in page_numbers:
|
||||
context = PageContext(
|
||||
tables=tables_by_page.get(number, ()),
|
||||
outlined_runs=outlines.get(number, ()),
|
||||
)
|
||||
for region in scan_page(doc[number]):
|
||||
yield region, classify(region, context)
|
||||
@@ -3,8 +3,14 @@ name = "ingestion"
|
||||
version = "0.0.0"
|
||||
description = "Offline batch pipeline: PDF -> monographs -> chunks -> embeddings -> Qdrant"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = []
|
||||
dependencies = ["pymupdf>=1.24", "numpy>=1.26", "scipy>=1.11"]
|
||||
|
||||
[project.optional-dependencies]
|
||||
dev = ["pytest>=7.4"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["ingestion*"]
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ingestion.chunk import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
SCHEMA_VERSION,
|
||||
chunk_monograph,
|
||||
chunk_section,
|
||||
write_chunks_jsonl,
|
||||
)
|
||||
from ingestion.chunk.chunker import _is_label_row, describe_block
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from ingestion.tables import SHAPE_FORMULA_2D, SHAPE_MULTI_HEADER, SHAPE_SIMPLE
|
||||
|
||||
|
||||
def _section(key, display, text, page=202):
|
||||
return SectionSpan(
|
||||
key=key, display_name=display,
|
||||
heading=Heading(text=display, physical_page=page, y0=100.0,
|
||||
is_monograph_title=False, section_key=key),
|
||||
text=text,
|
||||
)
|
||||
|
||||
|
||||
def _monograph(sections, tables=()):
|
||||
return Monograph(
|
||||
drug_id="ampicilin_va_sulbactam",
|
||||
drug_name="AMPICILIN VÀ SULBACTAM",
|
||||
source_page_range=[200, 203],
|
||||
sections={s.key: s for s in sections},
|
||||
atc_codes=["J01CR01"],
|
||||
tables=list(tables),
|
||||
)
|
||||
|
||||
|
||||
def _block(block_id="p202_t0", shape=SHAPE_SIMPLE, section_key="lieu_luong_va_cach_dung"):
|
||||
return TableBlock(
|
||||
table_id=block_id, shape=shape, physical_page=202,
|
||||
bbox=[299.0, 189.6, 552.4, 300.5], section_key=section_key,
|
||||
text="Độ thanh thải creatinin Nửa đời Liều 1,5 - 3,0 g",
|
||||
quarantined=True,
|
||||
)
|
||||
|
||||
|
||||
def test_a_section_whose_table_was_lifted_says_so():
|
||||
"""The defect this exists to prevent is silent, not visible.
|
||||
|
||||
Without the reference, this chunk is grammatical, complete-looking prose
|
||||
with the renal-dosing table absent and nothing marking the absence.
|
||||
"""
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(monograph)
|
||||
|
||||
prose = [c for c in chunks if c.chunk_kind == CHUNK_KIND_PROSE]
|
||||
assert len(prose) == 1
|
||||
assert prose[0].has_quarantined_content is True
|
||||
assert [a.block_id for a in prose[0].attachments] == ["p202_t0"]
|
||||
assert prose[0].attachments[0].physical_page == 202
|
||||
assert prose[0].attachments[0].bbox
|
||||
|
||||
|
||||
def test_a_lifted_block_gets_its_own_retrievable_descriptor():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
descriptors = [c for c in chunk_monograph(monograph)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
assert len(descriptors) == 1
|
||||
assert "AMPICILIN VÀ SULBACTAM" in descriptors[0].text
|
||||
assert "Liều lượng và cách dùng" in descriptors[0].text
|
||||
# printed page, which is what a reader holding the book looks for
|
||||
assert "trang 203" in descriptors[0].text
|
||||
|
||||
|
||||
def test_no_cell_value_ever_reaches_the_descriptor_text():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
block = _block()
|
||||
monograph = _monograph([section], [block])
|
||||
descriptors = [c for c in chunk_monograph(monograph, {"p202_t0": []})
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
assert "1,5 - 3,0 g" not in descriptors[0].text
|
||||
|
||||
|
||||
def test_a_header_row_carrying_a_number_is_refused():
|
||||
"""AMIODARON, physical page 183 — a real case, caught by a gate.
|
||||
|
||||
pdfplumber reported the first row as
|
||||
"Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)", i.e. a
|
||||
dose inside what it called a header, from an extraction never verified by
|
||||
eye. Measured: 42 of 124 simple-table headers (34%) contain a digit.
|
||||
"""
|
||||
assert _is_label_row(["Các Statin", "Khởi đầu", "Liều duy trì"]) is True
|
||||
assert _is_label_row(["Liều 720 mg/ngày (0,5 mg/phút)"]) is False
|
||||
assert _is_label_row(["x" * 45]) is False
|
||||
assert _is_label_row([]) is False
|
||||
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(
|
||||
monograph, {"p202_t0": ["Liều 720 mg/ngày (0,5 mg/phút)"]})
|
||||
descriptor = next(c for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert "720" not in descriptor.text
|
||||
assert descriptor.attachments[0].header_row == []
|
||||
|
||||
|
||||
def test_only_a_simple_table_contributes_a_header():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
header = {"p202_t0": ["Nhóm", "Liều"]}
|
||||
for shape, expected in ((SHAPE_SIMPLE, ["Nhóm", "Liều"]),
|
||||
(SHAPE_MULTI_HEADER, [])):
|
||||
monograph = _monograph([section], [_block(shape=shape)])
|
||||
descriptor = next(c for c in chunk_monograph(monograph, header)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert descriptor.attachments[0].header_row == expected
|
||||
|
||||
|
||||
def test_a_formula_block_is_described_as_a_formula():
|
||||
section = _section("than_trong", "Thận trọng", "Prose.")
|
||||
block = _block(block_id="p1042_f0", shape=SHAPE_FORMULA_2D,
|
||||
section_key="than_trong")
|
||||
monograph = _monograph([section], [block])
|
||||
descriptor = next(c for c in chunk_monograph(monograph)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert "công thức" in descriptor.text
|
||||
assert "bảng" not in descriptor.text
|
||||
|
||||
|
||||
def test_attachments_do_not_change_the_prose_text():
|
||||
"""The condition under which this feature was accepted at all."""
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
|
||||
with_block = chunk_section(_monograph([section], [_block()]), section,
|
||||
[_block()])
|
||||
without = chunk_section(_monograph([section]), section)
|
||||
prose_with = [c for c in with_block if c.chunk_kind == CHUNK_KIND_PROSE]
|
||||
assert [c.text for c in prose_with] == [c.text for c in without]
|
||||
assert [c.chunk_id for c in prose_with] == [c.chunk_id for c in without]
|
||||
|
||||
|
||||
def test_a_section_with_no_text_but_a_block_still_yields_the_descriptor():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "")
|
||||
chunks = chunk_monograph(_monograph([section], [_block()]))
|
||||
assert [c.chunk_kind for c in chunks] == [CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
|
||||
|
||||
def test_written_chunks_declare_their_schema_version(tmp_path: Path):
|
||||
section = _section("chi_dinh", "Chỉ định", "Nhiễm khuẩn.")
|
||||
chunks = chunk_monograph(_monograph([section]))
|
||||
out = tmp_path / "chunks.jsonl"
|
||||
assert write_chunks_jsonl(chunks, out) == 1
|
||||
record = json.loads(out.read_text(encoding="utf-8").splitlines()[0])
|
||||
assert record["schema_version"] == SCHEMA_VERSION
|
||||
assert record["chunk_kind"] == CHUNK_KIND_PROSE
|
||||
assert record["has_quarantined_content"] is False
|
||||
|
||||
|
||||
def test_describe_block_names_the_page_even_with_no_header():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(monograph)
|
||||
attachment = next(c for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR).attachments[0]
|
||||
text = describe_block(monograph, section, attachment)
|
||||
assert "trang 203" in text
|
||||
assert "không trích dẫn được dưới dạng văn bản" in text
|
||||
@@ -0,0 +1,49 @@
|
||||
import pytest
|
||||
|
||||
from ingestion.cli import build_parser
|
||||
|
||||
|
||||
def test_run_subcommand_parses_required_pdf_arg():
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(["run", "--pdf", "some.pdf"])
|
||||
assert args.command == "run"
|
||||
assert args.pdf == "some.pdf"
|
||||
assert args.out == "data/processed/monographs.jsonl"
|
||||
|
||||
|
||||
def test_run_subcommand_accepts_custom_out():
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(["run", "--pdf", "a.pdf", "--out", "b.jsonl"])
|
||||
assert args.out == "b.jsonl"
|
||||
|
||||
|
||||
def test_run_requires_pdf_arg():
|
||||
parser = build_parser()
|
||||
with pytest.raises(SystemExit):
|
||||
parser.parse_args(["run"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("command", ["visual-diff", "scaffold-golden"])
|
||||
def test_not_yet_implemented_commands_raise_explicitly(command):
|
||||
parser = build_parser()
|
||||
args = parser.parse_args([command])
|
||||
with pytest.raises(NotImplementedError):
|
||||
args.func(args)
|
||||
|
||||
|
||||
def test_run_reports_missing_pdf_file(tmp_path, capsys):
|
||||
parser = build_parser()
|
||||
missing = tmp_path / "does_not_exist.pdf"
|
||||
args = parser.parse_args(["run", "--pdf", str(missing)])
|
||||
exit_code = args.func(args)
|
||||
assert exit_code == 1
|
||||
assert "not found" in capsys.readouterr().err
|
||||
|
||||
|
||||
def test_validate_reports_missing_pdf_file(tmp_path, capsys):
|
||||
parser = build_parser()
|
||||
missing = tmp_path / "does_not_exist.pdf"
|
||||
args = parser.parse_args(["validate", "--pdf", str(missing)])
|
||||
exit_code = args.func(args)
|
||||
assert exit_code == 1
|
||||
assert "not found" in capsys.readouterr().err
|
||||
@@ -0,0 +1,75 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ingestion.extract.formulas import (
|
||||
FORMULA_BAND_HEIGHT_PT,
|
||||
FORMULA_SIDE_MARGIN_PT,
|
||||
load_formula_regions,
|
||||
)
|
||||
from ingestion.tables import QUARANTINE_SHAPES, SHAPE_FORMULA_2D
|
||||
|
||||
VERIFIED = (Path(__file__).resolve().parents[1] / "data" / "verified"
|
||||
/ "formula_regions_2d.json")
|
||||
TRANSCRIPTIONS = (Path(__file__).resolve().parents[1] / "data" / "verified"
|
||||
/ "outlined_text_transcriptions.json")
|
||||
|
||||
|
||||
def test_a_2d_formula_is_always_quarantined():
|
||||
# linearised, "a / b" reads as "a x b" — a dosing error, not a cosmetic one
|
||||
assert SHAPE_FORMULA_2D in QUARANTINE_SHAPES
|
||||
|
||||
|
||||
def test_verified_formula_regions_load_with_the_confirmed_pages():
|
||||
regions = load_formula_regions()
|
||||
assert {r.physical_page for r in regions} == {
|
||||
43, 92, 147, 202, 325, 349, 1042, 1043, 1132, 1402,
|
||||
}
|
||||
assert all(r.shape == SHAPE_FORMULA_2D for r in regions)
|
||||
|
||||
|
||||
def test_the_region_covers_numerator_and_denominator_not_just_the_bar():
|
||||
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
|
||||
bar = next(r for r in payload["regions"] if r["physical_page"] == 1042)
|
||||
region = next(r for r in load_formula_regions() if r.physical_page == 1042)
|
||||
x0, y0, x1, y1 = bar["bar_bbox"]
|
||||
assert region.bbox[1] == y0 - FORMULA_BAND_HEIGHT_PT
|
||||
assert region.bbox[3] == y1 + FORMULA_BAND_HEIGHT_PT
|
||||
assert region.bbox[0] == x0 - FORMULA_SIDE_MARGIN_PT
|
||||
|
||||
|
||||
def test_the_barless_adenosin_formula_is_recorded_as_a_recall_limit():
|
||||
"""The source prints no bar, so no geometric detector can find it.
|
||||
|
||||
Recorded so a later reader does not mistake the fraction-bar scan for
|
||||
complete formula coverage — how many bar-less formulas the book contains
|
||||
has never been measured.
|
||||
"""
|
||||
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
|
||||
barless = [r for r in payload["regions"] if r.get("source_prints_no_bar")]
|
||||
assert [r["physical_page"] for r in barless] == [147]
|
||||
assert "UNMEASURED" in payload["recall_limit"]
|
||||
|
||||
|
||||
def test_outlined_text_transcriptions_cover_every_detected_run():
|
||||
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
|
||||
runs = payload["runs"]
|
||||
assert len(runs) == 51
|
||||
assert all(r["text"] for r in runs), "a run with no transcription is data loss"
|
||||
pages = {}
|
||||
for run in runs:
|
||||
pages[run["physical_page"]] = pages.get(run["physical_page"], 0) + 1
|
||||
assert pages == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}
|
||||
|
||||
|
||||
def test_single_glyph_transcriptions_name_the_line_they_were_dropped_from():
|
||||
"""The subtlest form of the defect: one character missing mid-sentence.
|
||||
|
||||
"Độ ổn định" extracts as "Độ n định" and reads as ordinary text, so
|
||||
nothing downstream can notice. Keeping the owning line in the record is
|
||||
what makes the repair checkable.
|
||||
"""
|
||||
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
|
||||
singles = [r for r in payload["runs"] if r["single_glyph"]]
|
||||
assert len(singles) == 29
|
||||
with_context = [r for r in singles if r["extracted_line_it_belongs_to"]]
|
||||
assert with_context, "no dropped glyph could be tied back to its line"
|
||||
@@ -0,0 +1,79 @@
|
||||
from ingestion.extract.glyph_order import find_reading_order_issues, is_reversed_order
|
||||
|
||||
|
||||
def test_normal_ltr_span_not_flagged():
|
||||
# ordinary increasing x-origins, as any normal left-to-right span has
|
||||
assert not is_reversed_order([264.7, 269.4, 271.6, 276.3, 278.5])
|
||||
|
||||
|
||||
def test_confirmed_page_1373_defect_shape_is_flagged():
|
||||
# exact x-origins read via get_text("rawdict") from physical page 1373's
|
||||
# affected span (" tịx 4 =" reversed) — see docs/pdf-parsing-outlier-catalog.md item 9
|
||||
x_origins = [66.32, 64.17, 61.53, 58.89, 54.14, 51.98, 47.23]
|
||||
assert is_reversed_order(x_origins)
|
||||
|
||||
|
||||
def test_single_char_span_not_flagged():
|
||||
assert not is_reversed_order([100.0])
|
||||
|
||||
|
||||
def test_empty_span_not_flagged():
|
||||
assert not is_reversed_order([])
|
||||
|
||||
|
||||
def test_tied_x_origins_not_flagged_as_reversed():
|
||||
# equal x-origins (e.g. stacked/overlapping glyphs) are not "decreasing"
|
||||
assert not is_reversed_order([100.0, 100.0, 100.0])
|
||||
|
||||
|
||||
def test_correctly_ordered_row_not_flagged():
|
||||
row = {(20, 550.9): [(518.0, "n"), (525.2, "h"), (532.6, "i"), (536.8, "e")]}
|
||||
assert find_reading_order_issues(row) == []
|
||||
|
||||
|
||||
def test_confirmed_page_714_row_misorder_is_flagged():
|
||||
# reproduces the real page-714 finding: within one PyMuPDF block (20),
|
||||
# 4 line fragments are emitted out of x-order ("quản ", " ộ", "đ tệih",
|
||||
# "n " concatenated) that reconstruct correctly ("...nhiệt độ") when
|
||||
# re-sorted by x-origin — see outlier catalog item 9.
|
||||
row = {
|
||||
(20, 550.9): [
|
||||
(518.06, "n"), (525.20, " "),
|
||||
(546.59, " "), (553.71, "ộ"),
|
||||
(541.84, "đ"), (539.45, " "), (536.81, "t"), (532.59, "ệ"), (529.95, "i"), (525.20, "h"),
|
||||
]
|
||||
}
|
||||
issues = find_reading_order_issues(row)
|
||||
assert len(issues) == 1
|
||||
assert issues[0].extracted_text != issues[0].corrected_text
|
||||
|
||||
|
||||
def test_different_blocks_at_same_y_not_merged():
|
||||
# regression test for a real false positive: two DIFFERENT paragraphs in
|
||||
# different PyMuPDF blocks (a right-column paragraph starting at x=299.4
|
||||
# and a left-column paragraph starting at x=35.4, page 1104) coincide at
|
||||
# the same y — grouping by block index (not a hand-picked x-coordinate
|
||||
# column boundary) is what keeps them from being merged into one "row".
|
||||
# This is the caller's responsibility (scan_reading_order groups by real
|
||||
# PyMuPDF block index); find_reading_order_issues just trusts its input
|
||||
# is already correctly grouped, which these two dict entries demonstrate.
|
||||
row_block_1 = {(1, 70.4): [(299.39, "m"), (306.78, "ô")]}
|
||||
row_block_4 = {(4, 70.4): [(35.43, "d"), (40.18, "e")]}
|
||||
assert find_reading_order_issues(row_block_1) == []
|
||||
assert find_reading_order_issues(row_block_4) == []
|
||||
|
||||
|
||||
def test_kerning_jitter_not_flagged_as_reading_order_defect():
|
||||
# regression test for a real false positive found by running against the
|
||||
# actual PDF: "mefloquin" ('l' at x=491.566, 'o' at x=491.471 — a
|
||||
# 0.095pt kerning-driven dip) was previously "corrected" into the wrong
|
||||
# word "mefolquin". A row-level check with no decrease tolerance treats
|
||||
# ordinary kerning as a defect and corrupts already-correct text.
|
||||
row = {
|
||||
(5, 449.7): [
|
||||
(474.865, "m"), (482.161, "e"), (486.284, "f"),
|
||||
(491.566, "l"), (491.471, "o"), (496.126, "q"),
|
||||
(500.781, "u"), (505.436, "i"), (507.982, "n"),
|
||||
]
|
||||
}
|
||||
assert find_reading_order_issues(row) == []
|
||||
@@ -0,0 +1,27 @@
|
||||
from ingestion.extract.page_map import pick_folio
|
||||
|
||||
|
||||
def test_single_candidate_is_the_folio():
|
||||
assert pick_folio([("101", 10.0)]) == 101
|
||||
|
||||
|
||||
def test_no_candidates_is_unrecoverable():
|
||||
assert pick_folio([]) is None
|
||||
|
||||
|
||||
def test_confirmed_riboflavin_subscript_conflict_resolved_by_size():
|
||||
# exact (text, size) pairs read from physical page 1243's header band:
|
||||
# the real folio "1244" (size 10.0, matching the rest of the running
|
||||
# header) and the "2" subscript from "Vitamin B2" (size 5.83), which
|
||||
# happens to fall in the same y<60 header band because the RIBOFLAVIN
|
||||
# title sits high on the page — see module docstring. Silently dropped
|
||||
# the whole monograph before this fix, confirmed via a whole-book
|
||||
# `cli validate` run and by rendering the page to an image.
|
||||
candidates = [("1244", 10.0), ("2", 5.83)]
|
||||
assert pick_folio(candidates) == 1244
|
||||
|
||||
|
||||
def test_genuine_same_size_conflict_still_returns_none():
|
||||
# two same-size digit-only candidates: real ambiguity, must not guess
|
||||
candidates = [("101", 10.0), ("205", 10.0)]
|
||||
assert pick_folio(candidates) is None
|
||||
@@ -0,0 +1,67 @@
|
||||
from ingestion.extract.spans import classify_column, _sort_blocks_reading_order
|
||||
|
||||
|
||||
def _block(x0, y0, x1, y1):
|
||||
return {"bbox": (x0, y0, x1, y1)}
|
||||
|
||||
|
||||
def testclassify_column_left():
|
||||
assert classify_column((35.0, 100.0, 280.0, 120.0)) == "left"
|
||||
|
||||
|
||||
def testclassify_column_right():
|
||||
assert classify_column((299.0, 100.0, 553.0, 120.0)) == "right"
|
||||
|
||||
|
||||
def testclassify_column_full_width_header():
|
||||
assert classify_column((35.0, 34.0, 552.0, 48.0)) == "full_width"
|
||||
|
||||
|
||||
def testclassify_column_none_bbox_is_unknown():
|
||||
assert classify_column(None) == "unknown"
|
||||
|
||||
|
||||
def test_confirmed_real_oxymetazolin_page_reversed_order_is_corrected():
|
||||
# exact bboxes from physical page 1100 (the OXYBUTYNIN/OXYMETAZOLIN
|
||||
# boundary — see spans.py module docstring): PyMuPDF's raw block order
|
||||
# is [header, right x7, left x8], right column before left. An earlier
|
||||
# version of this module trusted that raw order, silently attributing
|
||||
# OXYMETAZOLIN's "Chống chỉ định" (right column) to the still-open
|
||||
# OXYBUTYNIN monograph. Confirmed via a whole-book cli validate run,
|
||||
# a whole-document cross-tool character-diff, and rendering the page.
|
||||
raw_order = [
|
||||
_block(34.96, 34.39, 552.10, 47.72), # 0: full_width header
|
||||
_block(299.39, 60.46, 553.72, 121.96), # 1: right
|
||||
_block(299.39, 124.33, 553.72, 368.94), # 2: right
|
||||
_block(299.39, 371.31, 553.72, 408.39), # 3: right
|
||||
_block(35.43, 60.77, 289.77, 330.56), # 4: left (Xử trí: ...)
|
||||
_block(35.43, 379.21, 231.23, 391.87), # 5: left (Tên chung quốc tế)
|
||||
]
|
||||
sorted_blocks = _sort_blocks_reading_order(raw_order)
|
||||
columns_in_order = [classify_column(b["bbox"]) for b in sorted_blocks]
|
||||
assert columns_in_order == ["full_width", "left", "left", "right", "right", "right"]
|
||||
|
||||
|
||||
def test_already_correct_order_is_left_unchanged_in_content():
|
||||
blocks = [
|
||||
_block(35.0, 60.0, 280.0, 100.0), # left
|
||||
_block(35.0, 110.0, 280.0, 150.0), # left, further down
|
||||
_block(299.0, 60.0, 553.0, 100.0), # right
|
||||
]
|
||||
sorted_blocks = _sort_blocks_reading_order(blocks)
|
||||
assert sorted_blocks == blocks
|
||||
|
||||
|
||||
def test_a_narrow_box_between_the_columns_belongs_to_the_right_column():
|
||||
"""The two tolerance bands overlap between x=288 and x=319.
|
||||
|
||||
Testing left first put everything in that strip in the left column. It is
|
||||
invisible for a full-width block and wrong for a narrow one: a single 4pt
|
||||
glyph at x=315 on physical page 714 was classified left, so the 'ổ'
|
||||
missing from "Độ ổn định" could not be matched to its own line and the
|
||||
corruption survived the repair.
|
||||
"""
|
||||
assert classify_column((313.7, 506.2, 317.8, 514.8)) == "right"
|
||||
assert classify_column((35.4, 500.0, 289.7, 510.0)) == "left"
|
||||
# a box that lands in neither range still resolves by tolerance
|
||||
assert classify_column((300.0, 500.0, 305.0, 510.0)) == "left"
|
||||
@@ -0,0 +1,95 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.normalize import (
|
||||
PUA_SUBSTITUTIONS,
|
||||
find_unmapped_pua,
|
||||
group_visual_lines,
|
||||
join_spans,
|
||||
substitute_pua,
|
||||
)
|
||||
|
||||
|
||||
def _span(text, *, page=100, block=0, line=0, index=0, x0=50.0, x1=None, y0=100.0):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column="left",
|
||||
block=block, line=line, span_index=index,
|
||||
x0=x0, y0=y0, x1=(x0 + len(text) * 4.5) if x1 is None else x1, y1=y0 + 10,
|
||||
text=text, font="Tiger", size=9.5,
|
||||
)
|
||||
|
||||
|
||||
def test_pua_map_covers_every_codepoint_confirmed_in_the_corpus():
|
||||
# all 8 were located in the source PDF, rendered, and read visually —
|
||||
# see docs/progress-log.md for the page each was confirmed on
|
||||
assert PUA_SUBSTITUTIONS[""] == "≥"
|
||||
assert PUA_SUBSTITUTIONS[""] == "≤"
|
||||
assert PUA_SUBSTITUTIONS[""] == "α"
|
||||
assert PUA_SUBSTITUTIONS[""] == "→"
|
||||
assert PUA_SUBSTITUTIONS[""] == "®"
|
||||
assert PUA_SUBSTITUTIONS[""] == "₁"
|
||||
assert PUA_SUBSTITUTIONS[""] == "↓"
|
||||
assert PUA_SUBSTITUTIONS[""] == "γ"
|
||||
|
||||
|
||||
def test_comparison_operators_in_real_dosing_sentences_are_restored():
|
||||
# the clinically dangerous case: without this, "liều ≤ 100 mg" reaches
|
||||
# embeddings as "liều 100 mg" and the operator is lost
|
||||
assert substitute_pua("trẻ em 10 tuổi") == "trẻ em ≥ 10 tuổi"
|
||||
assert substitute_pua("liều 100 mg") == "liều ≤ 100 mg"
|
||||
|
||||
|
||||
def test_unmapped_pua_is_reported_not_silently_passed_through():
|
||||
assert find_unmapped_pua("liều 100 mg") == []
|
||||
assert find_unmapped_pua("bất ngờ đây") == [""]
|
||||
|
||||
|
||||
def test_subscript_span_rejoins_without_a_spurious_space():
|
||||
# real corpus case: "cytochrom P450" arrived as "cytochrom P\n450\ngây"
|
||||
spans = [
|
||||
_span("cytochrom P", x0=50.0, x1=100.0),
|
||||
_span("450", x0=100.2, x1=110.0),
|
||||
_span(" gây chuyển hóa.", x0=110.1, x1=180.0),
|
||||
]
|
||||
assert join_spans(spans) == "cytochrom P450 gây chuyển hóa."
|
||||
|
||||
|
||||
def test_italic_run_inside_parentheses_rejoins_on_one_line():
|
||||
# real corpus case: "(\nfeline immunodeficiency virus\n)"
|
||||
spans = [
|
||||
_span("(", x0=50.0, x1=53.0),
|
||||
_span("feline immunodeficiency virus", x0=53.1, x1=180.0),
|
||||
_span(")", x0=180.1, x1=183.0),
|
||||
]
|
||||
assert join_spans(spans) == "(feline immunodeficiency virus)"
|
||||
|
||||
|
||||
def test_wrap_without_sentence_end_is_joined_with_a_space():
|
||||
spans = [
|
||||
_span("không nhai. Nếu", line=0, y0=100.0),
|
||||
_span("uống viên thuốc", line=1, y0=112.0),
|
||||
]
|
||||
assert join_spans(spans) == "không nhai. Nếu uống viên thuốc"
|
||||
|
||||
|
||||
def test_sentence_end_keeps_the_line_break():
|
||||
spans = [
|
||||
_span("Liều người lớn: 10 mg.", line=0, y0=100.0),
|
||||
_span("Trẻ em: 5 mg.", line=1, y0=112.0),
|
||||
]
|
||||
assert join_spans(spans) == "Liều người lớn: 10 mg.\nTrẻ em: 5 mg."
|
||||
|
||||
|
||||
def test_wide_gap_on_one_line_still_yields_a_space():
|
||||
spans = [
|
||||
_span("Người bệnh", x0=50.0, x1=100.0),
|
||||
_span("100 kg", x0=104.0, x1=130.0),
|
||||
]
|
||||
assert join_spans(spans) == "Người bệnh 100 kg"
|
||||
|
||||
|
||||
def test_visual_lines_group_by_pymupdf_block_and_line_indices():
|
||||
spans = [
|
||||
_span("a", block=0, line=0), _span("b", block=0, line=0),
|
||||
_span("c", block=0, line=1),
|
||||
_span("d", block=1, line=0),
|
||||
]
|
||||
assert [len(g) for g in group_visual_lines(spans)] == [2, 1, 1]
|
||||
@@ -0,0 +1,332 @@
|
||||
import pytest
|
||||
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.assembler import DuplicateDrugIdError, assemble
|
||||
|
||||
|
||||
def _span(text, page, y0, bold=True, size=9.5, printed=None, column="left"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=printed if printed is not None else page + 1,
|
||||
column=column, block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=("TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"), size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_basic_single_monograph_with_sections_and_body():
|
||||
spans = [
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir (Acyclovir-like).", 100, 92.0, bold=False),
|
||||
_span("Mã ATC:", 100, 104.0),
|
||||
_span("J05AF06", 100, 116.0, bold=False),
|
||||
_span("Chỉ định", 101, 60.0),
|
||||
_span("Điều trị nhiễm HIV.", 101, 72.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
m = monographs[0]
|
||||
assert m.drug_id == "abacavir"
|
||||
assert m.drug_name == "ABACAVIR"
|
||||
assert m.source_page_range == [100, 101]
|
||||
assert m.sections["ten_chung_quoc_te"].text == "Abacavir (Acyclovir-like)."
|
||||
assert m.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
|
||||
assert m.atc_codes == ["J05AF06"]
|
||||
assert m.atc_stated_absent is False
|
||||
|
||||
|
||||
def test_non_bold_combined_heading_value_span_confirmed_real_amitriptylin_case():
|
||||
# AMITRIPTYLIN's real "Mã ATC:" heading is a single non-bold span
|
||||
# combining label and value ("Mã ATC: N06AA09."), unlike Abacavir's
|
||||
# bold-label + separate-value spans — see outlier item 20.
|
||||
spans = [
|
||||
_span("AMITRIPTYLIN", 184, 60.0),
|
||||
_span("Tên chung quốc tế: ", 184, 85.0),
|
||||
_span("Amitriptyline.", 184, 85.2, bold=False),
|
||||
_span("Mã ATC: N06AA09.", 184, 100.0, bold=False),
|
||||
_span("Loại thuốc:", 184, 115.0),
|
||||
_span("Thuốc chống trầm cảm.", 184, 115.2, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.sections["ma_atc"].text == "N06AA09."
|
||||
assert m.atc_codes == ["N06AA09"]
|
||||
|
||||
|
||||
def test_atc_stated_absent_propagates():
|
||||
spans = [
|
||||
_span("ADIPIODON", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 72.0),
|
||||
_span("Adipiodon.", 100, 84.0, bold=False),
|
||||
_span("Mã ATC:", 100, 96.0),
|
||||
_span("Chưa có.", 100, 108.0, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.atc_codes == []
|
||||
assert m.atc_stated_absent is True
|
||||
|
||||
|
||||
def test_qualifier_line_disambiguates_same_name_monographs():
|
||||
# reproduces the confirmed real SALBUTAMOL case (outlier item 18):
|
||||
# same base title, disambiguated by a bold non-caps parenthesized line.
|
||||
spans = [
|
||||
_span("SALBUTAMOL", 1261, 60.0),
|
||||
_span("(Dùng trong hô hấp)", 1261, 72.0),
|
||||
_span("Tên chung quốc tế:", 1261, 84.0),
|
||||
_span("Salbutamol.", 1261, 96.0, bold=False),
|
||||
_span("Chỉ định", 1261, 108.0),
|
||||
_span("Điều trị hen.", 1261, 120.0, bold=False),
|
||||
_span("SALBUTAMOL", 1263, 60.0),
|
||||
_span("(Dùng trong sản khoa)", 1263, 72.0),
|
||||
_span("Tên chung quốc tế:", 1263, 84.0),
|
||||
_span("Salbutamol.", 1263, 96.0, bold=False),
|
||||
_span("Chỉ định", 1263, 108.0),
|
||||
_span("Điều trị dọa sinh non.", 1263, 120.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 2
|
||||
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
|
||||
assert monographs[0].drug_name == "SALBUTAMOL (Dùng trong hô hấp)"
|
||||
assert monographs[1].drug_id == "salbutamol_dung_trong_san_khoa"
|
||||
assert monographs[0].sections["chi_dinh"].text == "Điều trị hen."
|
||||
assert monographs[1].sections["chi_dinh"].text == "Điều trị dọa sinh non."
|
||||
|
||||
|
||||
def test_genuine_duplicate_drug_id_raises():
|
||||
spans = [
|
||||
_span("FOOBARDRUG", 200, 60.0),
|
||||
_span("Tên chung quốc tế:", 200, 72.0),
|
||||
_span("Foobardrug.", 200, 84.0, bold=False),
|
||||
_span("Chỉ định", 200, 96.0),
|
||||
_span("A.", 200, 108.0, bold=False),
|
||||
_span("FOOBARDRUG", 300, 60.0),
|
||||
_span("Tên chung quốc tế:", 300, 72.0),
|
||||
_span("Foobardrug.", 300, 84.0, bold=False),
|
||||
_span("Chỉ định", 300, 96.0),
|
||||
_span("B.", 300, 108.0, bold=False),
|
||||
]
|
||||
with pytest.raises(DuplicateDrugIdError):
|
||||
list(assemble(spans))
|
||||
|
||||
|
||||
def test_gonadotropin_wrap_does_not_falsely_trigger_duplicate_check():
|
||||
# regression: the multi-line wrap must merge BEFORE the duplicate check
|
||||
# runs, so this is never treated as two separate "GONADOTROPIN" titles
|
||||
spans = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 676.2354736328125),
|
||||
_span("Tên chung quốc tế:", 1371, 690.0),
|
||||
_span("Gonadorelin.", 1371, 700.0, bold=False),
|
||||
_span("Chỉ định", 1371, 712.0),
|
||||
_span("X.", 1371, 724.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_name == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
|
||||
|
||||
|
||||
def test_front_matter_before_first_monograph_is_ignored():
|
||||
spans = [
|
||||
_span("Some front matter heading", 5, 60.0, bold=False, printed=6),
|
||||
_span("random body text", 5, 72.0, bold=False, printed=6),
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir.", 100, 92.0, bold=False),
|
||||
_span("Chỉ định", 100, 104.0),
|
||||
_span("X.", 100, 116.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "abacavir"
|
||||
|
||||
|
||||
def test_empty_spans_yields_nothing():
|
||||
assert list(assemble([])) == []
|
||||
|
||||
|
||||
def test_table_header_false_positive_not_treated_as_monograph():
|
||||
# reproduces the confirmed real "HSV"/"CMV" table-column-header case
|
||||
# (outlier item 19, physical page 698, inside the Foscarnet natri
|
||||
# monograph's dosing table) — bold+all-caps+short, identical shape to a
|
||||
# real title, but never followed by "Tên chung quốc tế" before the next
|
||||
# real title. Must not be treated as a monograph boundary.
|
||||
spans = [
|
||||
_span("FOSCARNET NATRI", 690, 60.0),
|
||||
_span("Tên chung quốc tế:", 690, 80.0),
|
||||
_span("Foscarnet.", 690, 92.0, bold=False),
|
||||
_span("Chỉ định", 690, 104.0),
|
||||
_span("Điều trị CMV.", 690, 116.0, bold=False),
|
||||
_span("HSV", 698, 523.0),
|
||||
_span("HSV", 698, 523.0),
|
||||
_span("CMV", 698, 523.0),
|
||||
_span("CMV", 698, 523.0),
|
||||
_span("40 mg/kg cách nhau 12 giờ", 698, 540.0, bold=False),
|
||||
_span("ARTEMETHER", 700, 60.0),
|
||||
_span("Tên chung quốc tế:", 700, 80.0),
|
||||
_span("Artemether.", 700, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert [m.drug_id for m in monographs] == ["foscarnet_natri", "artemether"]
|
||||
# the table row's numbers/labels stay attached to Foscarnet's Chỉ định
|
||||
# section body (dropped from a dedicated section, which is fine — no
|
||||
# false monograph boundary is what matters here)
|
||||
assert "hsv" not in monographs[0].drug_id
|
||||
assert "cmv" not in monographs[0].drug_id
|
||||
|
||||
|
||||
def test_real_title_immediately_followed_by_anchor_is_kept():
|
||||
spans = [
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir.", 100, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "abacavir"
|
||||
|
||||
|
||||
def test_real_title_with_qualifier_before_anchor_is_still_kept():
|
||||
# the anchor lookahead must tolerate one intervening qualifier-line
|
||||
# event (the SALBUTAMOL case), not just immediate adjacency
|
||||
spans = [
|
||||
_span("SALBUTAMOL", 1261, 60.0),
|
||||
_span("(Dùng trong hô hấp)", 1261, 72.0),
|
||||
_span("Tên chung quốc tế:", 1261, 84.0),
|
||||
_span("Salbutamol.", 1261, 96.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
|
||||
|
||||
|
||||
def test_class_level_monograph_sub_heading_not_treated_as_own_monograph():
|
||||
# reproduces the confirmed real case (outlier item 21): "SIMVASTATIN" is
|
||||
# a bold+all-caps+short sub-heading *inside* the class-level "CÁC CHẤT
|
||||
# ỨC CHẾ HMG-CoA REDUCTASE" monograph, immediately followed by its own
|
||||
# "Liều lượng và cách dùng" but NOT by "Tên chung quốc tế" (that section
|
||||
# belongs only to the parent). Must stay folded into the parent, not
|
||||
# become its own monograph.
|
||||
spans = [
|
||||
_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 60.0, printed=285),
|
||||
_span("Tên chung quốc tế:", 284, 72.0, printed=285),
|
||||
_span("Simvastatin, Lovastatin.", 284, 84.0, bold=False, printed=285),
|
||||
_span("Chỉ định", 284, 96.0, printed=285),
|
||||
_span("Tăng lipid huyết.", 284, 108.0, bold=False, printed=285),
|
||||
_span("SIMVASTATIN", 285, 60.0, printed=286),
|
||||
_span("Liều lượng và cách dùng", 285, 72.0, printed=286),
|
||||
_span("Uống 10 - 20 mg mỗi tối.", 285, 84.0, bold=False, printed=286),
|
||||
_span("LOVASTATIN", 285, 96.0, printed=286),
|
||||
_span("Liều lượng và cách dùng", 285, 108.0, printed=286),
|
||||
_span("Uống 20 mg mỗi ngày.", 285, 120.0, bold=False, printed=286),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_name == "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"
|
||||
# the sub-headings' own dosing text stays attached to the parent
|
||||
# monograph's content rather than vanishing or becoming new monographs
|
||||
assert "Uống 20 mg mỗi ngày." in monographs[0].sections["lieu_luong_va_cach_dung"].text
|
||||
|
||||
|
||||
def test_running_header_boilerplate_stripped_from_mid_section_body_confirmed_real_morphin_case():
|
||||
# exact confirmed real case: physical page 1008's running header
|
||||
# ("DTQGVN 2" / "1009" / "Morphin sulfat", all column="full_width",
|
||||
# y0~34, well inside the header band) falls squarely in the middle of
|
||||
# MORPHIN SULFAT's "Liều lượng và cách dùng" section, which spans the
|
||||
# page 1007->1008 boundary — see outlier-catalog item 13 / assembler.py
|
||||
# module docstring. Whole-corpus measured: 1,374/11,409 sections (12.0%)
|
||||
# affected before this fix, 671/682 monographs (98.4%) had at least one.
|
||||
spans = [
|
||||
_span("MORPHIN SULFAT", 1007, 60.0),
|
||||
_span("Tên chung quốc tế:", 1007, 80.0),
|
||||
_span("Morphini sulfas.", 1007, 92.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 1007, 700.0),
|
||||
_span("Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu", 1007, 785.4, bold=False),
|
||||
_span("DTQGVN 2", 1008, 34.6, bold=False, column="full_width"),
|
||||
_span("1009", 1008, 34.6, bold=False, column="full_width"),
|
||||
_span("Morphin sulfat", 1008, 34.4, bold=False, column="full_width"),
|
||||
_span("uống viên thuốc giải phóng chậm thì không được nghiền.", 1008, 60.8, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
section_text = m.sections["lieu_luong_va_cach_dung"].text
|
||||
assert "DTQGVN" not in section_text
|
||||
assert "1009" not in section_text
|
||||
# the two body spans are one sentence broken by a page boundary: "Nếu"
|
||||
# does not end a sentence, so normalize/text_flow rejoins them with a
|
||||
# space rather than preserving the PDF's visual wrap as a hard newline
|
||||
assert section_text == (
|
||||
"Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu "
|
||||
"uống viên thuốc giải phóng chậm thì không được nghiền."
|
||||
)
|
||||
|
||||
|
||||
def test_last_real_monograph_in_book_still_kept_near_end_of_input():
|
||||
# anchor lookahead must not require a "next title" to exist — the very
|
||||
# last monograph in the book has no following title at all
|
||||
spans = [
|
||||
_span("ZOLPIDEM", 1494, 60.0),
|
||||
_span("Tên chung quốc tế:", 1494, 80.0),
|
||||
_span("Zolpidem.", 1494, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "zolpidem"
|
||||
|
||||
|
||||
def test_repeated_section_heading_appends_instead_of_overwriting():
|
||||
# measured real case: 33 monographs repeat a section heading (38
|
||||
# occurrences). CEFAMANDOL's "Liều lượng và cách dùng" resumes on
|
||||
# physical page 339 after a renal-dosing table; the old code replaced the
|
||||
# SectionSpan, destroying everything captured before the repeat — for
|
||||
# CEFAMANDOL that left the dosing section holding only the table.
|
||||
spans = [
|
||||
_span("CEFAMANDOL", 338, 60.0),
|
||||
_span("Tên chung quốc tế", 338, 80.0),
|
||||
_span("Cefamandolum.", 338, 92.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 338, 400.0),
|
||||
_span("Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần.", 338, 412.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 339, 200.0),
|
||||
_span("Suy thận: giảm liều theo độ thanh thải creatinin.", 339, 212.0, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
text = m.sections["lieu_luong_va_cach_dung"].text
|
||||
assert "Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần." in text
|
||||
assert "Suy thận: giảm liều theo độ thanh thải creatinin." in text
|
||||
# the first heading stays the provenance anchor
|
||||
assert m.sections["lieu_luong_va_cach_dung"].heading.physical_page == 338
|
||||
|
||||
|
||||
def test_a_plain_label_line_under_a_heading_is_body_not_a_new_section():
|
||||
"""FLUOROURACIL, physical page 681 — verified by rendering the page.
|
||||
|
||||
The book prints "Thời kỳ mang thai" / "Chống chỉ định." and "Thời kỳ cho
|
||||
con bú" / "Chống chỉ định.". The body line matches the section vocabulary,
|
||||
so it was read as a heading and both sections came out empty — dropping
|
||||
the statement that fluorouracil is contraindicated in pregnancy and while
|
||||
breastfeeding.
|
||||
"""
|
||||
spans = [
|
||||
_span("FLUOROURACIL", 681, 60.0),
|
||||
_span("Tên chung quốc tế", 681, 80.0),
|
||||
_span("Fluorouracilum.", 681, 92.0, bold=False),
|
||||
_span("Chống chỉ định", 681, 110.0),
|
||||
_span("Suy tủy nặng.", 681, 122.0, bold=False),
|
||||
_span("Thời kỳ mang thai", 681, 140.0),
|
||||
_span("Chống chỉ định.", 681, 152.0, bold=False),
|
||||
_span("Thời kỳ cho con bú", 681, 170.0),
|
||||
_span("Chống chỉ định.", 681, 182.0, bold=False),
|
||||
]
|
||||
monograph = list(assemble(spans))[0]
|
||||
assert monograph.sections["thoi_ky_mang_thai"].text == "Chống chỉ định."
|
||||
assert monograph.sections["thoi_ky_cho_con_bu"].text == "Chống chỉ định."
|
||||
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
|
||||
|
||||
|
||||
def test_a_bold_label_line_still_opens_its_section():
|
||||
spans = [
|
||||
_span("FLUOROURACIL", 681, 60.0),
|
||||
_span("Tên chung quốc tế", 681, 80.0),
|
||||
_span("Fluorouracilum.", 681, 92.0, bold=False),
|
||||
_span("Chống chỉ định", 681, 110.0),
|
||||
_span("Suy tủy nặng.", 681, 122.0, bold=False),
|
||||
]
|
||||
monograph = list(assemble(spans))[0]
|
||||
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
|
||||
@@ -0,0 +1,144 @@
|
||||
from ingestion.segment.atc import extract_atc_codes, is_stated_absent, normalize_atc_candidate
|
||||
|
||||
|
||||
def test_stray_whitespace_split_j04a_c01_recovered():
|
||||
assert normalize_atc_candidate("J04A C01") == "J04AC01"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_n05b_a06_recovered():
|
||||
assert normalize_atc_candidate("N05B A06") == "N05BA06"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_l01x_x02_recovered():
|
||||
assert normalize_atc_candidate("L01X X02") == "L01XX02"
|
||||
|
||||
|
||||
def test_digit_letter_confusion_no3ax12_recovered():
|
||||
assert normalize_atc_candidate("NO3AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_digit_letter_confusion_jo1dc07_recovered():
|
||||
assert normalize_atc_candidate("JO1DC07") == "J01DC07"
|
||||
|
||||
|
||||
def test_clean_code_passes_through():
|
||||
assert normalize_atc_candidate("N03AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_garbage_not_recovered():
|
||||
assert normalize_atc_candidate("NOT AN ATC CODE") is None
|
||||
assert normalize_atc_candidate("") is None
|
||||
|
||||
|
||||
def test_stated_absent_chua_co():
|
||||
assert is_stated_absent("Mã ATC: Chưa có.") is True
|
||||
|
||||
|
||||
def test_stated_absent_khong_co():
|
||||
assert is_stated_absent("Không có.") is True
|
||||
|
||||
|
||||
def test_stated_present_not_flagged_absent():
|
||||
assert is_stated_absent("N03AX12") is False
|
||||
|
||||
|
||||
def test_extract_single_code():
|
||||
result = extract_atc_codes("N03AX12")
|
||||
assert result.codes == ["N03AX12"]
|
||||
assert result.stated_absent is False
|
||||
|
||||
|
||||
def test_extract_multi_code_insulin_style():
|
||||
result = extract_atc_codes("A10AB01, A10AC01, A10AD01")
|
||||
assert result.codes == ["A10AB01", "A10AC01", "A10AD01"]
|
||||
|
||||
|
||||
def test_extract_multi_code_with_noise_mixed_in():
|
||||
# one clean code, one noisy code recovered, matching the real corpus
|
||||
# pattern where a monograph has some clean and some noisy ATC entries
|
||||
result = extract_atc_codes("N03AX12, J04A C01")
|
||||
assert result.codes == ["N03AX12", "J04AC01"]
|
||||
|
||||
|
||||
def test_extract_stated_absent_returns_no_codes():
|
||||
result = extract_atc_codes("Mã ATC: Chưa có.")
|
||||
assert result.codes == []
|
||||
assert result.stated_absent is True
|
||||
|
||||
|
||||
def test_trailing_period_recovered_confirmed_real_abacavir_case():
|
||||
# real field text is "J05AF06." — a sentence-ending period, not part of
|
||||
# the code; an earlier version silently produced zero codes here.
|
||||
assert normalize_atc_candidate("J05AF06.") == "J05AF06"
|
||||
result = extract_atc_codes("J05AF06.")
|
||||
assert result.codes == ["J05AF06"]
|
||||
|
||||
|
||||
def test_species_annotation_stripped_confirmed_real_insulin_case():
|
||||
# annotation-stripping is extract_atc_codes's job (must run before the
|
||||
# comma/semicolon split, see below) — normalize_atc_candidate itself
|
||||
# only normalizes an already-isolated code token.
|
||||
result = extract_atc_codes("A10AB01 (người); A10AB02 (bò)")
|
||||
assert result.codes == ["A10AB01", "A10AB02"]
|
||||
|
||||
|
||||
def test_leading_colon_from_value_span_stripped_confirmed_real_alcuronium_case():
|
||||
# real field text for ALCURONIUM CLORID (physical page 152): the bold
|
||||
# label span is "Mã ATC" with no colon, and the plain value span is
|
||||
# ": M03AA01." — the colon belongs to the value side here, not the
|
||||
# label side (Abacavir's equivalent has it on the label side instead:
|
||||
# "Mã ATC: " + "J05AF06."). See atc.py module docstring, defect 5.
|
||||
assert normalize_atc_candidate(": M03AA01.") == "M03AA01"
|
||||
result = extract_atc_codes(": M03AA01.")
|
||||
assert result.codes == ["M03AA01"]
|
||||
|
||||
|
||||
def test_name_prefixed_code_stripped_confirmed_real_arginin_case():
|
||||
# real field text for ARGININ (physical page 204): two salt forms, each
|
||||
# its own "Name: CODE" line, not a bare code — see atc.py module
|
||||
# docstring, defect 6.
|
||||
assert normalize_atc_candidate("Arginin glutamat: A05BA01") == "A05BA01"
|
||||
result = extract_atc_codes("Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01")
|
||||
assert result.codes == ["A05BA01", "B05XB01"]
|
||||
|
||||
|
||||
def test_plain_code_with_no_colon_still_normalizes():
|
||||
assert normalize_atc_candidate("N03AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_reversed_code_first_shape_confirmed_real_hmg_coa_case():
|
||||
# real field text for CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE (physical page
|
||||
# 284): each statin is "CODE: Name", the opposite order from the
|
||||
# "Name: CODE" shape above — see atc.py module docstring, defect 7.
|
||||
# "C10A A01" also has the already-fixed stray-whitespace split.
|
||||
assert normalize_atc_candidate("C10A A01: Simvastatin") == "C10AA01"
|
||||
result = extract_atc_codes("C10A A01: Simvastatin\nC10A A02: Lovastatin")
|
||||
assert result.codes == ["C10AA01", "C10AA02"]
|
||||
|
||||
|
||||
def test_annotation_containing_a_comma_does_not_break_the_split_confirmed_vaccine_case():
|
||||
# real field text for VẮC XIN SỞI (physical page 1437): the English
|
||||
# annotation "(Measles, live attenuated)" contains its own comma. An
|
||||
# earlier version split on "," *before* stripping the annotation,
|
||||
# breaking "J07BD01 (Measles, live attenuated)." into two unrecoverable
|
||||
# fragments and silently returning zero codes — see atc.py module
|
||||
# docstring, defect 4.
|
||||
result = extract_atc_codes("J07BD01 (Measles, live attenuated).")
|
||||
assert result.codes == ["J07BD01"]
|
||||
|
||||
|
||||
def test_extract_all_20_insulin_codes_from_real_field_text():
|
||||
# exact real field text for INSULIN (physical page 809) — see atc.py
|
||||
# module docstring; confirms the fix recovers all 20, not just 2.
|
||||
field_text = (
|
||||
"A10AB01 (người); A10AB02 (bò); A10AB03 (lợn);\n"
|
||||
"A10AB04 (lispro); A10AB05 (aspart); A10AB06 (glulisin);\n"
|
||||
"A10AC01 (người); A10AC02 (bò); A10AC03 (lợn); A10AC04\n"
|
||||
"(lispro); A10AD01 (người), A10AD02 (bò), A10AD03 (lợn),\n"
|
||||
"A10AD04 (lispro), A10AE01 (người); A10AE02 (bò); A10AE03\n"
|
||||
"(lợn); A10AE04 (glargin); A10AE05 (detemir), A10AF01 (người)."
|
||||
)
|
||||
result = extract_atc_codes(field_text)
|
||||
assert len(result.codes) == 20
|
||||
assert "A10AB01" in result.codes
|
||||
assert "A10AF01" in result.codes
|
||||
@@ -0,0 +1,88 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.detector import detect_monograph_titles, detect_section_headings
|
||||
|
||||
|
||||
def _span(text, physical_page, printed_page, y0=100.0, bold=True, size=10.0):
|
||||
font = "TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"
|
||||
return Span(
|
||||
physical_page=physical_page, printed_page=printed_page, column="left",
|
||||
block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=font, size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_confirmed_part_divider_excluded_at_page_99_boundary():
|
||||
# "CÁC CHUYÊN LUẬN THUỐC" at physical page 98 / printed 99 — bold,
|
||||
# all-caps, short: identical shape to a real title, must be excluded.
|
||||
spans = [_span("CÁC CHUYÊN LUẬN THUỐC", 98, 99), _span("ABACAVIR", 100, 101)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["ABACAVIR"]
|
||||
|
||||
|
||||
def test_monograph_title_outside_page_range_excluded():
|
||||
# bold all-caps short text in front matter (e.g. an org name) must not
|
||||
# be picked up — scoping to printed 99-1496 is required, not optional.
|
||||
spans = [_span("BỘ Y TẾ", 2, 3), _span("ABACAVIR", 100, 101)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["ABACAVIR"]
|
||||
|
||||
|
||||
def test_gonadotropin_wrap_detected_as_one_title():
|
||||
spans = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 1372, y0=664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 1372, y0=676.2354736328125),
|
||||
]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"]
|
||||
|
||||
|
||||
def test_non_bold_all_caps_text_not_a_title_candidate():
|
||||
spans = [_span("NOT BOLD BUT CAPS", 100, 101, bold=False)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_lowercase_bold_text_not_a_title_candidate():
|
||||
spans = [_span("Abacavir", 100, 101)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_short_section_label_with_normal_diacritic_not_a_title_candidate():
|
||||
# regression: an earlier absolute-count (not ratio) version of the
|
||||
# mixed-case tolerance let "Mã ATC:" through as a false title candidate
|
||||
# — its single lowercase diacritic ('ã') is normal Vietnamese
|
||||
# orthography, not a HMG-CoA-style embedded abbreviation. A ratio
|
||||
# threshold correctly rejects this short label (1/5 = 20% lowercase)
|
||||
# while still accepting the long HMG-CoA title (1/27 = 3.7%).
|
||||
spans = [_span("Mã ATC:", 100, 101)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_confirmed_hmg_coa_mixed_case_title_still_detected():
|
||||
# "CoA" (Coenzyme A) is a real mixed-case abbreviation embedded in an
|
||||
# otherwise all-caps title — outlier item 21. A strict isupper() check
|
||||
# silently dropped this entire class-level monograph from the corpus.
|
||||
spans = [_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 285)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"]
|
||||
|
||||
|
||||
def test_section_heading_matched_with_and_without_trailing_colon():
|
||||
spans = [
|
||||
_span("Tên chung quốc tế:", 100, 101, bold=True, size=9.5),
|
||||
_span("Chỉ định", 100, 101, bold=True, size=9.5),
|
||||
]
|
||||
headings = list(detect_section_headings(spans))
|
||||
assert [h.section_key for h in headings] == ["ten_chung_quoc_te", "chi_dinh"]
|
||||
|
||||
|
||||
def test_unknown_bold_text_not_matched_as_section():
|
||||
# e.g. "Cách dùng:" — a real sub-heading within "Liều lượng và cách
|
||||
# dùng" that is NOT one of the known top-level section names.
|
||||
spans = [_span("Cách dùng:", 100, 101, bold=True, size=9.5)]
|
||||
assert list(detect_section_headings(spans)) == []
|
||||
|
||||
|
||||
def test_section_heading_outside_monograph_range_excluded():
|
||||
spans = [_span("Chỉ định", 5, 6, bold=True, size=9.5)]
|
||||
assert list(detect_section_headings(spans)) == []
|
||||
@@ -0,0 +1,72 @@
|
||||
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan
|
||||
|
||||
|
||||
def test_round_trip_preserves_all_fields(tmp_path):
|
||||
heading = Heading(text="Chỉ định", physical_page=100, y0=80.0, is_monograph_title=False, section_key="chi_dinh")
|
||||
section = SectionSpan(key="chi_dinh", display_name="Chỉ định", heading=heading, text="Điều trị nhiễm HIV.")
|
||||
monograph = Monograph(
|
||||
drug_id="abacavir", drug_name="ABACAVIR", source_page_range=[100, 101],
|
||||
sections={"chi_dinh": section}, atc_codes=["J05AF06"], atc_stated_absent=False,
|
||||
)
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
count = write_monographs_jsonl([monograph], path)
|
||||
assert count == 1
|
||||
|
||||
result = list(read_monographs_jsonl(path))
|
||||
assert len(result) == 1
|
||||
r = result[0]
|
||||
assert r.drug_id == "abacavir"
|
||||
assert r.drug_name == "ABACAVIR"
|
||||
assert r.source_page_range == [100, 101]
|
||||
assert r.atc_codes == ["J05AF06"]
|
||||
assert r.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
|
||||
assert r.sections["chi_dinh"].heading.section_key == "chi_dinh"
|
||||
|
||||
|
||||
def test_multiple_monographs_round_trip(tmp_path):
|
||||
m1 = Monograph(drug_id="a", drug_name="A", source_page_range=[1, 2])
|
||||
m2 = Monograph(drug_id="b", drug_name="B", source_page_range=[3, 4])
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
write_monographs_jsonl([m1, m2], path)
|
||||
result = list(read_monographs_jsonl(path))
|
||||
assert [r.drug_id for r in result] == ["a", "b"]
|
||||
|
||||
|
||||
def test_empty_write_produces_empty_file(tmp_path):
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
count = write_monographs_jsonl([], path)
|
||||
assert count == 0
|
||||
assert list(read_monographs_jsonl(path)) == []
|
||||
|
||||
|
||||
def test_table_blocks_survive_a_write_read_round_trip(tmp_path):
|
||||
# the lifted table blocks were being computed in memory and then dropped
|
||||
# at the file boundary — 148 blocks existed in the run summary but the
|
||||
# JSONL had no "tables" key at all
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
|
||||
|
||||
heading = Heading(text="Liều lượng và cách dùng", physical_page=339, y0=200.0,
|
||||
is_monograph_title=False, section_key="lieu_luong_va_cach_dung")
|
||||
m = Monograph(
|
||||
drug_id="cefamandol", drug_name="CEFAMANDOL", source_page_range=[338, 340],
|
||||
sections={"lieu_luong_va_cach_dung": SectionSpan(
|
||||
key="lieu_luong_va_cach_dung", display_name="Liều lượng và cách dùng",
|
||||
heading=heading, text="Cách dùng ...")},
|
||||
tables=[TableBlock(
|
||||
table_id="p339_t0", shape="simple_table", physical_page=339,
|
||||
bbox=[40.0, 380.0, 400.0, 620.0],
|
||||
section_key="lieu_luong_va_cach_dung",
|
||||
text="80 - 50 750 mg - 2 g, 6 giờ/lần.", quarantined=True)],
|
||||
)
|
||||
path = tmp_path / "m.jsonl"
|
||||
write_monographs_jsonl([m], path)
|
||||
back = list(read_monographs_jsonl(path))[0]
|
||||
assert len(back.tables) == 1
|
||||
t = back.tables[0]
|
||||
assert t.table_id == "p339_t0"
|
||||
assert t.physical_page == 339
|
||||
assert t.bbox == [40.0, 380.0, 400.0, 620.0]
|
||||
assert t.quarantined is True
|
||||
assert "750 mg - 2 g" in t.text
|
||||
@@ -0,0 +1,134 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.merge import merge_multiline_headings, merge_same_line_bold_fragments
|
||||
|
||||
|
||||
def _span(text, page, y0, size=9.5, font="TimesNewRomanPS-BoldMT"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column="right",
|
||||
block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=font, size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_confirmed_gonadotropin_wrap_merges_into_one_heading():
|
||||
# exact bboxes from physical page 1371 (0-indexed) — see module docstring
|
||||
candidates = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 676.2354736328125),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
|
||||
|
||||
|
||||
def test_unrelated_single_line_titles_on_different_pages_not_merged():
|
||||
candidates = [
|
||||
_span("GONADOTROPIN", 755, 200.0),
|
||||
_span("HYDROCORTISON", 900, 300.0),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 2
|
||||
assert [h.text for h in headings] == ["GONADOTROPIN", "HYDROCORTISON"]
|
||||
|
||||
|
||||
def test_large_y_gap_on_same_page_not_merged():
|
||||
# two genuinely separate single-line titles far apart on the same page
|
||||
# (e.g. two short monographs stacked in one column) must not merge
|
||||
candidates = [
|
||||
_span("ATENOLOL", 219, 100.0),
|
||||
_span("ATRACURIUM BESYLAT", 219, 500.0),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 2
|
||||
|
||||
|
||||
def test_confirmed_aciclovir_same_line_split_merges_without_space():
|
||||
# exact bboxes from physical page 113 (0-indexed), found by rendering the
|
||||
# page to an image and reading it directly: "ACIC" (size 10.0) and
|
||||
# "LOVIR" (size 9.5) are one word split into two spans on the same
|
||||
# visual line — different font size, ~0.5pt y0 gap, near-zero x-gap.
|
||||
# Must merge WITHOUT a space ("ACICLOVIR", not "ACIC LOVIR") — see
|
||||
# module docstring.
|
||||
candidates = [
|
||||
_span("ACIC", 113, 515.1914672851562, size=10.0),
|
||||
_span("LOVIR", 113, 515.7044677734375, size=9.5),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "ACICLOVIR"
|
||||
|
||||
|
||||
def test_wrap_and_same_line_split_use_different_join_characters():
|
||||
# a genuine line-wrap (large y-gap) still joins with a space even when
|
||||
# font size differs, since size is no longer part of the merge decision
|
||||
candidates = [
|
||||
_span("FIRST LINE", 100, 200.0, size=10.0),
|
||||
_span("SECOND LINE", 100, 212.0, size=9.5),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "FIRST LINE SECOND LINE"
|
||||
|
||||
|
||||
def test_single_candidate_yields_one_heading():
|
||||
headings = list(merge_multiline_headings([_span("ABACAVIR", 100, 60.29)]))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "ABACAVIR"
|
||||
|
||||
|
||||
def test_empty_input_yields_nothing():
|
||||
assert list(merge_multiline_headings([])) == []
|
||||
|
||||
|
||||
def test_confirmed_ten_chung_quoc_te_diacritic_split_reassembles():
|
||||
# exact fragments + y0 from physical page 759's "GUAIFENESIN" monograph,
|
||||
# found via a whole-book `cli validate` run (the monograph was silently
|
||||
# dropped because "Tên chung quốc tế" never matched the section
|
||||
# vocabulary) and confirmed by rendering the page to an image: to a
|
||||
# human reader the line looks completely normal, but PyMuPDF splits it
|
||||
# into 5 spans around the diacritic characters — see module docstring.
|
||||
fragments = [
|
||||
_span("Tên chung qu", 759, 157.614),
|
||||
_span("ố", 759, 157.33),
|
||||
_span("c t", 759, 157.614),
|
||||
_span("ế", 759, 157.33),
|
||||
_span(": ", 759, 157.614),
|
||||
]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 1
|
||||
assert merged[0].text == "Tên chung quốc tế: "
|
||||
|
||||
|
||||
def test_non_bold_spans_pass_through_unmerged():
|
||||
fragments = [
|
||||
_span("Guaifenesin", 759, 157.24, font="TimesNewRomanPSMT"),
|
||||
_span(".", 759, 157.24, font="TimesNewRomanPSMT"),
|
||||
]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 2
|
||||
|
||||
|
||||
def test_bold_spans_on_different_lines_not_merged():
|
||||
fragments = [_span("Chỉ định", 100, 200.0), _span("Chống chỉ định", 100, 220.0)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 2
|
||||
|
||||
|
||||
def test_merged_span_keeps_provenance_of_first_fragment():
|
||||
fragments = [_span("Tên chung qu", 759, 157.614), _span("ố", 759, 157.33)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert merged[0].physical_page == 759
|
||||
assert merged[0].printed_page == 760
|
||||
assert merged[0].x0 == fragments[0].x0
|
||||
assert merged[0].x1 == fragments[-1].x1
|
||||
|
||||
|
||||
def test_single_bold_span_passes_through_unchanged():
|
||||
fragments = [_span("ABACAVIR", 100, 60.29)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert merged == fragments
|
||||
|
||||
|
||||
def test_empty_input_to_same_line_merge_yields_nothing():
|
||||
assert merge_same_line_bold_fragments([]) == []
|
||||
@@ -0,0 +1,108 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment import assemble
|
||||
from ingestion.tables import SHAPE_GRID_2D, SHAPE_SIMPLE, TableRegion, index_by_page
|
||||
|
||||
|
||||
def _span(text, page, y0, *, bold=False, x0=50.0, block=0, line=0, column="left"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column=column,
|
||||
block=block, line=line, span_index=0,
|
||||
x0=x0, y0=y0, x1=x0 + len(text) * 4.5, y1=y0 + 10,
|
||||
text=text, font="Tiger-Bold" if bold else "Tiger", size=9.5,
|
||||
)
|
||||
|
||||
|
||||
def _monograph_spans(extra):
|
||||
return [
|
||||
_span("PARACETAMOL", 109, 60.0, bold=True),
|
||||
_span("Tên chung quốc tế", 109, 80.0, bold=True),
|
||||
_span("Paracetamolum.", 109, 92.0),
|
||||
_span("Dạng thuốc và hàm lượng", 109, 200.0, bold=True),
|
||||
] + extra
|
||||
|
||||
|
||||
def test_table_spans_are_lifted_out_of_section_prose():
|
||||
# real measured case: physical page 109's dosage-form table was being
|
||||
# concatenated cell by cell into the section body
|
||||
# ('Viên nén' + '1' + '1 - 4' + '8 - 12' + 'Viên nang tác' ...)
|
||||
spans = _monograph_spans([
|
||||
_span("Thuốc dùng đường uống.", 109, 220.0),
|
||||
_span("Viên nén", 109, 400.0, block=5),
|
||||
_span("1", 109, 400.0, block=5, x0=200.0),
|
||||
_span("1 - 4", 109, 400.0, block=5, x0=260.0),
|
||||
_span("Sau khi uống hấp thu nhanh.", 109, 600.0, block=9),
|
||||
])
|
||||
# the region must cover the table's first column too — it starts at the
|
||||
# left margin, same x as body prose
|
||||
region = TableRegion("p109_t0", 109, (40.0, 380.0, 400.0, 460.0), 3, 3, SHAPE_SIMPLE)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
|
||||
body = m.sections["dang_thuoc_va_ham_luong"].text
|
||||
assert "Viên nén" not in body
|
||||
assert "1 - 4" not in body
|
||||
assert "Thuốc dùng đường uống." in body
|
||||
assert "Sau khi uống hấp thu nhanh." in body
|
||||
|
||||
assert len(m.tables) == 1
|
||||
block = m.tables[0]
|
||||
assert block.table_id == "p109_t0"
|
||||
assert "Viên nén" in block.text and "1 - 4" in block.text
|
||||
assert block.section_key == "dang_thuoc_va_ham_luong"
|
||||
assert block.physical_page == 109
|
||||
# every multi-column table is quarantined until a real row/column
|
||||
# reconstruction exists — its linearised text is not safe to cite as prose
|
||||
assert block.quarantined is True
|
||||
|
||||
|
||||
def test_without_a_region_map_behaviour_is_unchanged():
|
||||
spans = _monograph_spans([
|
||||
_span("Thuốc dùng đường uống.", 109, 220.0),
|
||||
_span("Viên nén", 109, 400.0, block=5),
|
||||
])
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.tables == []
|
||||
assert "Viên nén" in m.sections["dang_thuoc_va_ham_luong"].text
|
||||
|
||||
|
||||
def test_2d_grid_block_is_quarantined():
|
||||
# a 2D lookup grid's flattened text is meaningless without row/column
|
||||
# headers (outlier item 7) — it must be marked, not silently embedded
|
||||
spans = _monograph_spans([_span("0,52", 109, 400.0, block=5, x0=200.0)])
|
||||
region = TableRegion("p109_t1", 109, (150.0, 380.0, 400.0, 460.0), 6, 5, SHAPE_GRID_2D)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
assert len(m.tables) == 1
|
||||
assert m.tables[0].quarantined is True
|
||||
|
||||
|
||||
def test_non_table_regions_are_never_lifted():
|
||||
# the 17 full-page false positives must not swallow a whole page of prose
|
||||
spans = _monograph_spans([_span("Thuốc dùng đường uống.", 109, 220.0)])
|
||||
region = TableRegion("p109_t0", 109, (0.0, 0.0, 595.3, 836.2), 1, 2,
|
||||
"not_a_table_full_page")
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
assert m.tables == []
|
||||
assert "Thuốc dùng đường uống." in m.sections["dang_thuoc_va_ham_luong"].text
|
||||
|
||||
|
||||
def test_table_block_ids_stay_unique_when_a_section_resumes():
|
||||
# a region flushed twice (section closes, then resumes) must not emit two
|
||||
# blocks with the same table_id — provenance ids have to be unique
|
||||
spans = [
|
||||
_span("CEFAMANDOL", 339, 60.0, bold=True),
|
||||
_span("Tên chung quốc tế", 339, 80.0, bold=True),
|
||||
_span("Cefamandolum.", 339, 92.0),
|
||||
_span("Liều lượng và cách dùng", 339, 200.0, bold=True),
|
||||
_span("80 - 50", 339, 400.0, block=5),
|
||||
_span("Liều lượng và cách dùng", 339, 500.0, bold=True),
|
||||
_span("< 25 - 10", 339, 600.0, block=9),
|
||||
]
|
||||
region = TableRegion("p339_t0", 339, (40.0, 380.0, 400.0, 620.0), 5, 2, SHAPE_SIMPLE)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
# table_id is deterministic per REGION, so two parts of one table share
|
||||
# it on purpose; table_part_id is the unique key, derived from the first
|
||||
# source span rather than a counter (a counter would renumber whenever
|
||||
# anything upstream shifted, hiding rather than identifying a duplicate)
|
||||
assert len({t.table_part_id for t in m.tables}) == len(m.tables)
|
||||
assert {t.continuation_group for t in m.tables} == {"p339_t0"}
|
||||
assert all(t.table_part_id.startswith("p339_t0@") for t in m.tables)
|
||||
assert all(t.quarantined for t in m.tables)
|
||||
@@ -0,0 +1,30 @@
|
||||
from ingestion.segment.units import normalize_unit_token, validate_unit_tokens
|
||||
|
||||
|
||||
def test_clean_unit_passes_through():
|
||||
assert normalize_unit_token("mg") == "mg"
|
||||
assert normalize_unit_token("mcg") == "mcg"
|
||||
assert normalize_unit_token("mmol") == "mmol"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_recovered_by_analogy_to_atc():
|
||||
assert normalize_unit_token("m g") == "mg"
|
||||
assert normalize_unit_token("m cg") == "mcg"
|
||||
|
||||
|
||||
def test_case_insensitive():
|
||||
assert normalize_unit_token("MG") == "mg"
|
||||
|
||||
|
||||
def test_unknown_token_not_recovered():
|
||||
assert normalize_unit_token("xyz") is None
|
||||
assert normalize_unit_token("") is None
|
||||
|
||||
|
||||
def test_validate_unit_tokens_flags_only_bad_ones():
|
||||
bad = validate_unit_tokens(["mg", "mcg", "xyz", "ml"])
|
||||
assert bad == ["xyz"]
|
||||
|
||||
|
||||
def test_validate_unit_tokens_empty_when_all_valid():
|
||||
assert validate_unit_tokens(["mg", "mcg", "mmol"]) == []
|
||||
@@ -0,0 +1,83 @@
|
||||
from ingestion.segment.vocab import match_section, match_section_with_inline_value
|
||||
|
||||
|
||||
def test_exact_label_match_with_trailing_colon():
|
||||
d = match_section("Tên chung quốc tế:")
|
||||
assert d is not None and d.key == "ten_chung_quoc_te"
|
||||
|
||||
|
||||
def test_exact_label_match_without_trailing_colon():
|
||||
d = match_section("Chỉ định")
|
||||
assert d is not None and d.key == "chi_dinh"
|
||||
|
||||
|
||||
def test_inline_value_combined_span_confirmed_real_amitriptylin_case():
|
||||
# AMITRIPTYLIN's real "Mã ATC:" field is one non-bold span combining
|
||||
# label and value: "Mã ATC: N06AA09." — see outlier item 20.
|
||||
result = match_section_with_inline_value("Mã ATC: N06AA09.")
|
||||
assert result is not None
|
||||
section_def, value = result
|
||||
assert section_def.key == "ma_atc"
|
||||
assert value == "N06AA09."
|
||||
|
||||
|
||||
def test_inline_value_not_matched_when_no_colon_follows():
|
||||
assert match_section_with_inline_value("Mã ATC something else entirely") is None
|
||||
|
||||
|
||||
def test_inline_value_does_not_confuse_plain_body_text():
|
||||
assert match_section_with_inline_value("Bệnh nhân cần theo dõi chặt chẽ.") is None
|
||||
|
||||
|
||||
def test_exact_match_takes_priority_over_prefix_for_label_only_span():
|
||||
d = match_section("Mã ATC:")
|
||||
assert d is not None and d.key == "ma_atc"
|
||||
|
||||
|
||||
def test_real_spelling_variants_found_in_the_book_all_match():
|
||||
# measured whole-corpus: 42 distinct near-miss heading strings, 542
|
||||
# occurrences, none of which matched before aliases were added. The
|
||||
# heaviest is "Thông tin qui chế" (469x) — the book prints "qui" where
|
||||
# its own documented template says "quy", which cost 586 of 682
|
||||
# monographs their thong_tin_quy_che section entirely.
|
||||
from ingestion.segment.vocab import match_section
|
||||
cases = {
|
||||
"Thông tin qui chế": "thong_tin_quy_che",
|
||||
"Thông tin về qui chế": "thong_tin_quy_che",
|
||||
"Thông tin và quy chế": "thong_tin_quy_che",
|
||||
"Mã ACT": "ma_atc",
|
||||
"Chống chỉ đinh": "chong_chi_dinh",
|
||||
"Thời kì mang thai": "thoi_ky_mang_thai",
|
||||
"Thời kì cho con bú": "thoi_ky_cho_con_bu",
|
||||
"Dược lí và cơ chế tác dụng": "duoc_ly_va_co_che_tac_dung",
|
||||
"Hướng dẫn cách sử trí ADR": "huong_dan_xu_tri_adr",
|
||||
"Quá liều và xử lý": "qua_lieu_va_xu_tri",
|
||||
"Lọai thuốc": "loai_thuoc",
|
||||
}
|
||||
for text, expected_key in cases.items():
|
||||
matched = match_section(text)
|
||||
assert matched is not None, f"{text!r} should match a section"
|
||||
assert matched.key == expected_key
|
||||
|
||||
|
||||
def test_typesetting_noise_is_folded_without_needing_an_alias_each():
|
||||
# missing/extra spaces and the Ð/Đ look-alike are handled by the lookup
|
||||
# key, not enumerated per-variant
|
||||
from ingestion.segment.vocab import match_section
|
||||
assert match_section("Chỉđịnh").key == "chi_dinh"
|
||||
assert match_section("Chống chỉđịnh").key == "chong_chi_dinh"
|
||||
assert match_section("Độổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
||||
assert match_section("Ðộ ổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
||||
assert match_section("H ướng dẫn cách xử trí ADR").key == "huong_dan_xu_tri_adr"
|
||||
assert match_section("Tư ơng kỵ").key == "tuong_ky"
|
||||
assert match_section("Tác dụng khôngmong muốn (ADR)").key == "tac_dung_khong_mong_muon"
|
||||
assert match_section("Thận trọng.").key == "than_trong"
|
||||
|
||||
|
||||
def test_near_misses_that_are_not_sections_stay_unmatched():
|
||||
# "Thể trọng" is body weight, not "Thận trọng" (caution) — a 0.84
|
||||
# similarity that must NOT become an alias; the opioid string is a
|
||||
# drug-specific sub-heading inside a section, not the section itself
|
||||
from ingestion.segment.vocab import match_section
|
||||
assert match_section("Thể trọng") is None
|
||||
assert match_section("Tác dụng không mong muốn của opioid") is None
|
||||
@@ -0,0 +1,107 @@
|
||||
from ingestion.segment.models import Monograph
|
||||
from ingestion.validation.back_index import GroundTruthEntry
|
||||
from ingestion.validation.metrics import compute_recall_precision
|
||||
|
||||
|
||||
def _mono(drug_id, drug_name, start_physical):
|
||||
return Monograph(drug_id=drug_id, drug_name=drug_name, source_page_range=[start_physical, start_physical + 1])
|
||||
|
||||
|
||||
def test_perfect_match_recall_and_precision_are_one():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 1.0
|
||||
assert result.matched_count == 1
|
||||
|
||||
|
||||
def test_missed_ground_truth_entry_lowers_recall_not_precision():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [
|
||||
GroundTruthEntry(name="Abacavir", printed_page=101),
|
||||
GroundTruthEntry(name="Acarbose", printed_page=103),
|
||||
]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 0.5
|
||||
assert result.precision == 1.0
|
||||
assert len(result.unmatched_ground_truth) == 1
|
||||
assert result.unmatched_ground_truth[0].name == "Acarbose"
|
||||
|
||||
|
||||
def test_spurious_detected_monograph_lowers_precision_not_recall():
|
||||
monographs = [
|
||||
_mono("abacavir", "ABACAVIR", 100),
|
||||
_mono("cac_chuyen_luan_thuoc", "CÁC CHUYÊN LUẬN THUỐC", 98),
|
||||
]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 0.5
|
||||
assert len(result.unmatched_detected) == 1
|
||||
|
||||
|
||||
def test_page_tolerance_allows_small_offset():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=103)] # +2 tolerance
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
|
||||
|
||||
def test_page_beyond_tolerance_does_not_match():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=110)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_qualifier_suffixed_name_still_matches_base_ground_truth_name():
|
||||
# SALBUTAMOL (Dùng trong hô hấp) should still match a ground-truth
|
||||
# entry that just says "Salbutamol"
|
||||
monographs = [_mono("salbutamol_dung_trong_ho_hap", "SALBUTAMOL (Dùng trong hô hấp)", 1261)]
|
||||
ground_truth = [GroundTruthEntry(name="Salbutamol", printed_page=1262)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
|
||||
|
||||
def test_empty_ground_truth_gives_zero_recall_not_error():
|
||||
result = compute_recall_precision([_mono("a", "A", 1)], [])
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_empty_monographs_gives_zero_precision_not_error():
|
||||
result = compute_recall_precision([], [GroundTruthEntry(name="A", printed_page=1)])
|
||||
assert result.precision == 0.0
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_exact_match_preferred_over_substring_steal_confirmed_real_case():
|
||||
# Confirmed real case from a whole-book `cli validate` run: "ISOSORBID"
|
||||
# and "ISOSORBID DINITRAT" are two distinct, correctly-segmented
|
||||
# monographs a page apart. A pure substring match lets the shorter name
|
||||
# "steal" both ground-truth entries (it's a substring of the longer one
|
||||
# too) via `next()`'s order-dependent first match, leaving the real
|
||||
# "ISOSORBID DINITRAT" monograph spuriously unmatched even though an
|
||||
# exact match for it exists.
|
||||
monographs = [
|
||||
_mono("isosorbid", "ISOSORBID", 844),
|
||||
_mono("isosorbid_dinitrat", "ISOSORBID DINITRAT", 845),
|
||||
]
|
||||
ground_truth = [
|
||||
GroundTruthEntry(name="Isosorbid", printed_page=845),
|
||||
GroundTruthEntry(name="Isosorbid dinitrat", printed_page=846),
|
||||
]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 1.0
|
||||
assert len(result.unmatched_detected) == 0
|
||||
|
||||
|
||||
def test_double_space_in_detected_name_still_matches_confirmed_real_case():
|
||||
# confirmed real case from a whole-book `cli validate` run: "ALVERIN
|
||||
# CITRAT" (double space) failed to match ground truth's single-spaced
|
||||
# "Alverin citrat" under plain strip+upper comparison.
|
||||
monographs = [_mono("alverin_citrat", "ALVERIN CITRAT", 171)]
|
||||
ground_truth = [GroundTruthEntry(name="Alverin citrat", printed_page=172)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
@@ -0,0 +1,124 @@
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from ingestion.extract import OutlinedTextRun
|
||||
from ingestion.tables import TableRegion
|
||||
from ingestion.validation import (
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
HEADER_RULE,
|
||||
RULE_FRAGMENT,
|
||||
TABLE_FRAME,
|
||||
TEXT_AS_VECTOR_OUTLINE,
|
||||
UNCLASSIFIED,
|
||||
PageContext,
|
||||
ResidualRegion,
|
||||
classify,
|
||||
scan_page,
|
||||
)
|
||||
from ingestion.validation.residual_ink import FRACTION_BAR_CANDIDATE as BAR
|
||||
|
||||
PDF_PATH = Path(__file__).resolve().parents[1] / "data" / "raw" / (
|
||||
"duoc-thu-quoc-gia-viet-nam-2018.pdf"
|
||||
)
|
||||
needs_pdf = pytest.mark.skipif(not PDF_PATH.exists(), reason="source PDF not present")
|
||||
|
||||
|
||||
def _region(x0, y0, x1, y1, page=100, ink=500):
|
||||
return ResidualRegion(physical_page=page, bbox=(x0, y0, x1, y1), ink_px=ink)
|
||||
|
||||
|
||||
def test_running_header_rule_is_named_not_left_unclassified():
|
||||
# measured on real pages: a ~516pt wide, 0pt tall rule at y≈48-52 appears
|
||||
# on essentially every page of the book
|
||||
assert classify(_region(36.0, 48.5, 552.0, 48.5)) == HEADER_RULE
|
||||
|
||||
|
||||
def test_a_thin_bar_below_the_header_band_is_a_fraction_bar_candidate():
|
||||
# NETILMICIN, physical page 1042: the Cockcroft-Gault fraction bar
|
||||
assert classify(_region(97.9, 492.0, 286.5, 492.0)) == FRACTION_BAR_CANDIDATE
|
||||
|
||||
|
||||
def test_ink_inside_a_known_table_region_is_a_table_frame_not_a_formula():
|
||||
table = TableRegion(
|
||||
table_id="p202_t0", physical_page=202, bbox=(299.0, 189.6, 552.4, 300.5),
|
||||
n_rows=4, n_cols=3, shape="simple_table",
|
||||
)
|
||||
region = _region(299.0, 189.6, 552.4, 300.5, page=202)
|
||||
assert classify(region, PageContext(tables=[table])) == TABLE_FRAME
|
||||
# ...and the same geometry with no table map degrades to "look at it",
|
||||
# never to a silent pass
|
||||
assert classify(region) == UNCLASSIFIED
|
||||
|
||||
|
||||
def test_a_wide_rule_outside_the_header_band_is_not_treated_as_a_header_rule():
|
||||
assert classify(_region(36.0, 700.0, 552.0, 700.0)) == FRACTION_BAR_CANDIDATE
|
||||
|
||||
|
||||
def test_a_tall_block_of_unaccounted_ink_stays_unclassified():
|
||||
# a figure or an image of text must never be silently absorbed by a rule
|
||||
assert classify(_region(100.0, 300.0, 400.0, 500.0)) == UNCLASSIFIED
|
||||
|
||||
|
||||
def test_hairline_shorter_than_the_minimum_bar_width_is_a_rule_fragment():
|
||||
# too short to be a fraction bar, too thin to be anything but a rule
|
||||
assert classify(_region(100.0, 300.0, 105.0, 300.0)) == RULE_FRAGMENT
|
||||
|
||||
|
||||
@needs_pdf
|
||||
@pytest.mark.parametrize(
|
||||
"page,expected_bar_width_pt",
|
||||
[
|
||||
(1042, 188.6), # NETILMICIN — Cockcroft-Gault
|
||||
(202, 118.1), # AMPICILIN VÀ SULBACTAM — Cockcroft-Gault
|
||||
],
|
||||
)
|
||||
def test_confirmed_2d_formula_bars_survive_the_span_mask(page, expected_bar_width_pt):
|
||||
"""Regression fixture for the two visually confirmed corrupted formulas.
|
||||
|
||||
Both pages are reported as having zero tables by `pdfplumber` and zero by
|
||||
`opendataloader-pdf`; the bar is only findable as ink. If the mask padding
|
||||
is ever loosened again the bar disappears (at 1.0pt page 1042's bar
|
||||
shrinks from 188.6pt to 9.1pt) — this test is what catches that.
|
||||
"""
|
||||
import fitz
|
||||
|
||||
doc = fitz.open(PDF_PATH)
|
||||
bars = [
|
||||
r for r in scan_page(doc[page])
|
||||
if classify(r) == BAR and r.bbox[1] > 60.0
|
||||
]
|
||||
assert bars, f"no fraction-bar candidate found on physical page {page}"
|
||||
assert max(b.width_pt for b in bars) == pytest.approx(expected_bar_width_pt, abs=1.0)
|
||||
|
||||
|
||||
def test_vector_outlined_text_is_named_rather_than_left_unclassified():
|
||||
# physical page 714 prints 17 lines of Gatifloxacin prose as filled paths;
|
||||
# no text extractor returns them, so the gate must name the defect
|
||||
line = OutlinedTextRun(
|
||||
physical_page=714, bbox=(35.3, 75.8, 286.7, 84.4), path_items=1638,
|
||||
)
|
||||
region = _region(35.5, 76.0, 120.0, 84.0, page=714)
|
||||
context = PageContext(outlined_runs=[line])
|
||||
assert classify(region, context) == TEXT_AS_VECTOR_OUTLINE
|
||||
# an untranscribed line must never be mistaken for recovered content
|
||||
assert not line.is_transcribed
|
||||
|
||||
|
||||
@needs_pdf
|
||||
def test_outlined_text_lines_are_found_on_exactly_the_five_known_pages():
|
||||
"""Whole-document regression: 51 outlined runs on 5 pages.
|
||||
|
||||
Cross-checked two ways at the time of writing — the drawing-shape scan
|
||||
below, and independently by counting glyph-shaped leftovers in the
|
||||
residual-ink mask, which found the same five pages.
|
||||
"""
|
||||
import fitz
|
||||
|
||||
from ingestion.extract import detect_outlined_text
|
||||
|
||||
lines = list(detect_outlined_text(fitz.open(PDF_PATH)))
|
||||
by_page = {}
|
||||
for line in lines:
|
||||
by_page[line.physical_page] = by_page.get(line.physical_page, 0) + 1
|
||||
assert by_page == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}
|
||||
Reference in New Issue
Block a user