Files
duocthu/ingestion/tests/test_segment_detector.py
T

89 lines
3.9 KiB
Python

from ingestion.extract.models import Span
from ingestion.segment.detector import detect_monograph_titles, detect_section_headings
def _span(text, physical_page, printed_page, y0=100.0, bold=True, size=10.0):
font = "TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"
return Span(
physical_page=physical_page, printed_page=printed_page, column="left",
block=0, line=0, span_index=0,
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
text=text, font=font, size=size,
)
def test_confirmed_part_divider_excluded_at_page_99_boundary():
# "CÁC CHUYÊN LUẬN THUỐC" at physical page 98 / printed 99 — bold,
# all-caps, short: identical shape to a real title, must be excluded.
spans = [_span("CÁC CHUYÊN LUẬN THUỐC", 98, 99), _span("ABACAVIR", 100, 101)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["ABACAVIR"]
def test_monograph_title_outside_page_range_excluded():
# bold all-caps short text in front matter (e.g. an org name) must not
# be picked up — scoping to printed 99-1496 is required, not optional.
spans = [_span("BỘ Y TẾ", 2, 3), _span("ABACAVIR", 100, 101)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["ABACAVIR"]
def test_gonadotropin_wrap_detected_as_one_title():
spans = [
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 1372, y0=664.4554443359375),
_span("GONADOTROPIN", 1371, 1372, y0=676.2354736328125),
]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"]
def test_non_bold_all_caps_text_not_a_title_candidate():
spans = [_span("NOT BOLD BUT CAPS", 100, 101, bold=False)]
assert list(detect_monograph_titles(spans)) == []
def test_lowercase_bold_text_not_a_title_candidate():
spans = [_span("Abacavir", 100, 101)]
assert list(detect_monograph_titles(spans)) == []
def test_short_section_label_with_normal_diacritic_not_a_title_candidate():
# regression: an earlier absolute-count (not ratio) version of the
# mixed-case tolerance let "Mã ATC:" through as a false title candidate
# — its single lowercase diacritic ('ã') is normal Vietnamese
# orthography, not a HMG-CoA-style embedded abbreviation. A ratio
# threshold correctly rejects this short label (1/5 = 20% lowercase)
# while still accepting the long HMG-CoA title (1/27 = 3.7%).
spans = [_span("Mã ATC:", 100, 101)]
assert list(detect_monograph_titles(spans)) == []
def test_confirmed_hmg_coa_mixed_case_title_still_detected():
# "CoA" (Coenzyme A) is a real mixed-case abbreviation embedded in an
# otherwise all-caps title — outlier item 21. A strict isupper() check
# silently dropped this entire class-level monograph from the corpus.
spans = [_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 285)]
titles = [h.text for h in detect_monograph_titles(spans)]
assert titles == ["CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"]
def test_section_heading_matched_with_and_without_trailing_colon():
spans = [
_span("Tên chung quốc tế:", 100, 101, bold=True, size=9.5),
_span("Chỉ định", 100, 101, bold=True, size=9.5),
]
headings = list(detect_section_headings(spans))
assert [h.section_key for h in headings] == ["ten_chung_quoc_te", "chi_dinh"]
def test_unknown_bold_text_not_matched_as_section():
# e.g. "Cách dùng:" — a real sub-heading within "Liều lượng và cách
# dùng" that is NOT one of the known top-level section names.
spans = [_span("Cách dùng:", 100, 101, bold=True, size=9.5)]
assert list(detect_section_headings(spans)) == []
def test_section_heading_outside_monograph_range_excluded():
spans = [_span("Chỉ định", 5, 6, bold=True, size=9.5)]
assert list(detect_section_headings(spans)) == []