88 lines
4.1 KiB
Python
88 lines
4.1 KiB
Python
from ingestion.segment.vocab import match_section, match_section_with_inline_value
|
|
|
|
|
|
def test_exact_label_match_with_trailing_colon():
|
|
d = match_section("Tên chung quốc tế:")
|
|
assert d is not None and d.key == "ten_chung_quoc_te"
|
|
|
|
|
|
def test_exact_label_match_without_trailing_colon():
|
|
d = match_section("Chỉ định")
|
|
assert d is not None and d.key == "chi_dinh"
|
|
|
|
|
|
def test_inline_value_combined_span_confirmed_real_amitriptylin_case():
|
|
# AMITRIPTYLIN's real "Mã ATC:" field is one non-bold span combining
|
|
# label and value: "Mã ATC: N06AA09." — see outlier item 20.
|
|
result = match_section_with_inline_value("Mã ATC: N06AA09.")
|
|
assert result is not None
|
|
section_def, value = result
|
|
assert section_def.key == "ma_atc"
|
|
assert value == "N06AA09."
|
|
|
|
|
|
def test_inline_value_not_matched_when_no_colon_follows():
|
|
assert match_section_with_inline_value("Mã ATC something else entirely") is None
|
|
|
|
|
|
def test_inline_value_does_not_confuse_plain_body_text():
|
|
assert match_section_with_inline_value("Bệnh nhân cần theo dõi chặt chẽ.") is None
|
|
|
|
|
|
def test_exact_match_takes_priority_over_prefix_for_label_only_span():
|
|
d = match_section("Mã ATC:")
|
|
assert d is not None and d.key == "ma_atc"
|
|
|
|
|
|
def test_real_spelling_variants_found_in_the_book_all_match():
|
|
# measured whole-corpus: 42 distinct near-miss heading strings, 542
|
|
# occurrences, none of which matched before aliases were added. The
|
|
# heaviest is "Thông tin qui chế" (469x) — the book prints "qui" where
|
|
# its own documented template says "quy", which cost 586 of 682
|
|
# monographs their thong_tin_quy_che section entirely.
|
|
from ingestion.segment.vocab import match_section
|
|
cases = {
|
|
"Thông tin qui chế": "thong_tin_quy_che",
|
|
"Thông tin về qui chế": "thong_tin_quy_che",
|
|
"Thông tin và quy chế": "thong_tin_quy_che",
|
|
"Mã ACT": "ma_atc",
|
|
"Chống chỉ đinh": "chong_chi_dinh",
|
|
"Thời kì mang thai": "thoi_ky_mang_thai",
|
|
"Thời kì cho con bú": "thoi_ky_cho_con_bu",
|
|
"Dược lí và cơ chế tác dụng": "duoc_ly_va_co_che_tac_dung",
|
|
"Hướng dẫn cách sử trí ADR": "huong_dan_xu_tri_adr",
|
|
"Quá liều và xử lý": "qua_lieu_va_xu_tri",
|
|
"Lọai thuốc": "loai_thuoc",
|
|
"Tên chung quốc tế và mã ATC": "ten_chung_quoc_te",
|
|
"Dạng bào chế và hàm lượng": "dang_thuoc_va_ham_luong",
|
|
"Liều lượng và cách dùng giải độc tố uốn ván hấp phụ đơn giá":
|
|
"lieu_luong_va_cach_dung",
|
|
}
|
|
for text, expected_key in cases.items():
|
|
matched = match_section(text)
|
|
assert matched is not None, f"{text!r} should match a section"
|
|
assert matched.key == expected_key
|
|
|
|
|
|
def test_typesetting_noise_is_folded_without_needing_an_alias_each():
|
|
# missing/extra spaces and the Ð/Đ look-alike are handled by the lookup
|
|
# key, not enumerated per-variant
|
|
from ingestion.segment.vocab import match_section
|
|
assert match_section("Chỉđịnh").key == "chi_dinh"
|
|
assert match_section("Chống chỉđịnh").key == "chong_chi_dinh"
|
|
assert match_section("Độổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
|
assert match_section("Ðộ ổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
|
assert match_section("H ướng dẫn cách xử trí ADR").key == "huong_dan_xu_tri_adr"
|
|
assert match_section("Tư ơng kỵ").key == "tuong_ky"
|
|
assert match_section("Tác dụng khôngmong muốn (ADR)").key == "tac_dung_khong_mong_muon"
|
|
assert match_section("Thận trọng.").key == "than_trong"
|
|
|
|
|
|
def test_near_misses_that_are_not_sections_stay_unmatched():
|
|
# "Thể trọng" is body weight, not "Thận trọng" (caution) — a 0.84
|
|
# similarity that must NOT become an alias; the opioid string is a
|
|
# drug-specific sub-heading inside a section, not the section itself
|
|
from ingestion.segment.vocab import match_section
|
|
assert match_section("Thể trọng") is None
|
|
assert match_section("Tác dụng không mong muốn của opioid") is None
|