Fix every real lint finding and drop degenerate splice fragments

This commit is contained in:
2026-08-01 13:51:38 +07:00
parent 967b917001
commit 834d9e51b0
69 changed files with 10119 additions and 39 deletions
+95
View File
@@ -0,0 +1,95 @@
from ingestion.extract.models import Span
from ingestion.normalize import (
PUA_SUBSTITUTIONS,
find_unmapped_pua,
group_visual_lines,
join_spans,
substitute_pua,
)
def _span(text, *, page=100, block=0, line=0, index=0, x0=50.0, x1=None, y0=100.0):
return Span(
physical_page=page, printed_page=page + 1, column="left",
block=block, line=line, span_index=index,
x0=x0, y0=y0, x1=(x0 + len(text) * 4.5) if x1 is None else x1, y1=y0 + 10,
text=text, font="Tiger", size=9.5,
)
def test_pua_map_covers_every_codepoint_confirmed_in_the_corpus():
# all 8 were located in the source PDF, rendered, and read visually —
# see docs/progress-log.md for the page each was confirmed on
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "α"
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "®"
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == ""
assert PUA_SUBSTITUTIONS[""] == "γ"
def test_comparison_operators_in_real_dosing_sentences_are_restored():
# the clinically dangerous case: without this, "liều ≤ 100 mg" reaches
# embeddings as "liều  100 mg" and the operator is lost
assert substitute_pua("trẻ em  10 tuổi") == "trẻ em ≥ 10 tuổi"
assert substitute_pua("liều  100 mg") == "liều ≤ 100 mg"
def test_unmapped_pua_is_reported_not_silently_passed_through():
assert find_unmapped_pua("liều  100 mg") == []
assert find_unmapped_pua("bất ngờ  đây") == [""]
def test_subscript_span_rejoins_without_a_spurious_space():
# real corpus case: "cytochrom P450" arrived as "cytochrom P\n450\ngây"
spans = [
_span("cytochrom P", x0=50.0, x1=100.0),
_span("450", x0=100.2, x1=110.0),
_span(" gây chuyển hóa.", x0=110.1, x1=180.0),
]
assert join_spans(spans) == "cytochrom P450 gây chuyển hóa."
def test_italic_run_inside_parentheses_rejoins_on_one_line():
# real corpus case: "(\nfeline immunodeficiency virus\n)"
spans = [
_span("(", x0=50.0, x1=53.0),
_span("feline immunodeficiency virus", x0=53.1, x1=180.0),
_span(")", x0=180.1, x1=183.0),
]
assert join_spans(spans) == "(feline immunodeficiency virus)"
def test_wrap_without_sentence_end_is_joined_with_a_space():
spans = [
_span("không nhai. Nếu", line=0, y0=100.0),
_span("uống viên thuốc", line=1, y0=112.0),
]
assert join_spans(spans) == "không nhai. Nếu uống viên thuốc"
def test_sentence_end_keeps_the_line_break():
spans = [
_span("Liều người lớn: 10 mg.", line=0, y0=100.0),
_span("Trẻ em: 5 mg.", line=1, y0=112.0),
]
assert join_spans(spans) == "Liều người lớn: 10 mg.\nTrẻ em: 5 mg."
def test_wide_gap_on_one_line_still_yields_a_space():
spans = [
_span("Người bệnh", x0=50.0, x1=100.0),
_span("100 kg", x0=104.0, x1=130.0),
]
assert join_spans(spans) == "Người bệnh 100 kg"
def test_visual_lines_group_by_pymupdf_block_and_line_indices():
spans = [
_span("a", block=0, line=0), _span("b", block=0, line=0),
_span("c", block=0, line=1),
_span("d", block=1, line=0),
]
assert [len(g) for g in group_visual_lines(spans)] == [2, 1, 1]