Fix every real lint finding and drop degenerate splice fragments
This commit is contained in:
@@ -1,11 +1,15 @@
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
LOG_PATH = os.path.join(ROOT, "docs", "progress-log.md")
|
||||
README_PATH = os.path.join(ROOT, "README.md")
|
||||
ADR_DIR = os.path.join(ROOT, "docs", "adr")
|
||||
|
||||
def latest_entry():
|
||||
|
||||
def latest_progress_entry():
|
||||
try:
|
||||
with open(LOG_PATH, encoding="utf-8") as f:
|
||||
text = f.read()
|
||||
@@ -16,16 +20,56 @@ def latest_entry():
|
||||
return part.strip()
|
||||
return ""
|
||||
|
||||
entry = latest_entry()
|
||||
if entry:
|
||||
context = (
|
||||
"Project: Duoc Thu RAG medical chatbot (D:\\VSF-DUOCTHU). "
|
||||
"Latest entry from docs/progress-log.md (read that file and "
|
||||
"CLAUDE.md for full status before assuming anything):\n\n" + entry
|
||||
|
||||
def readme_text():
|
||||
try:
|
||||
with open(README_PATH, encoding="utf-8") as f:
|
||||
return f.read().strip()
|
||||
except FileNotFoundError:
|
||||
return ""
|
||||
|
||||
|
||||
def adr_index():
|
||||
lines = []
|
||||
for path in sorted(glob.glob(os.path.join(ADR_DIR, "*.md"))):
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
first_line = f.readline().strip()
|
||||
except OSError:
|
||||
continue
|
||||
title = re.sub(r"^#\s*", "", first_line)
|
||||
lines.append(f"- `docs/adr/{os.path.basename(path)}`: {title}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
sections = [
|
||||
"Project: Duoc Thu RAG medical chatbot (D:\\VSF-DUOCTHU). "
|
||||
"This is an automated orientation summary, not the full picture — read "
|
||||
"the referenced files (README.md, docs/architecture.md, the specific "
|
||||
"ADR, CLAUDE.md) before making claims about scope, architecture, or "
|
||||
"what already exists."
|
||||
]
|
||||
|
||||
readme = readme_text()
|
||||
if readme:
|
||||
sections.append("## README.md\n\n" + readme)
|
||||
|
||||
adrs = adr_index()
|
||||
if adrs:
|
||||
sections.append(
|
||||
"## Architecture decision records (docs/adr/) — titles only, "
|
||||
"read the full ADR before relying on its rationale/consequences:\n\n"
|
||||
+ adrs
|
||||
)
|
||||
|
||||
progress = latest_progress_entry()
|
||||
if progress:
|
||||
sections.append("## Latest entry from docs/progress-log.md\n\n" + progress)
|
||||
|
||||
if len(sections) > 1:
|
||||
print(json.dumps({
|
||||
"hookSpecificOutput": {
|
||||
"hookEventName": "SessionStart",
|
||||
"additionalContext": context,
|
||||
"additionalContext": "\n\n".join(sections),
|
||||
}
|
||||
}))
|
||||
|
||||
@@ -23,6 +23,11 @@ venv/
|
||||
.env.*
|
||||
!.env.example
|
||||
|
||||
# Investigation scratch — temporary evidence tools and their rendered output.
|
||||
# Per CLAUDE.md these are deleted once their finding lands in a test, fixture,
|
||||
# ADR or the outlier catalog; they are never imported by production code.
|
||||
ingestion/scratch/
|
||||
|
||||
# Ingestion large/derived artifacts (regeneratable — never commit)
|
||||
ingestion/data/interim/*
|
||||
!ingestion/data/interim/.gitkeep
|
||||
|
||||
@@ -31,6 +31,16 @@ See `docs/pdf-parsing-outlier-catalog.md` and
|
||||
`docs/adr/0003-pdf-parsing-strategy.md` for the concrete track record this
|
||||
rule comes from.
|
||||
|
||||
**What "verified" means:** when reporting something as verified, state (1)
|
||||
the command/test/script/manual check that was run, (2) the exact input
|
||||
scope, (3) the expected invariant or acceptance condition, (4) the observed
|
||||
result, and (5) any part of the requested scope that was *not* covered.
|
||||
Distinct scopes (unit test, regression fixture, selected-page sample,
|
||||
selected monographs, all detected monographs, full 1668-page document) are
|
||||
not interchangeable — don't describe one as another. Avoid words like
|
||||
"fully verified", "complete", "all", "no data lost", or "production-ready"
|
||||
unless the checks actually performed support that literal claim.
|
||||
|
||||
## Real code follows Clean Code / Clean Architecture / SoC / DRY / SOLID
|
||||
|
||||
Applies to anything meant to be committed as part of the actual system
|
||||
@@ -70,3 +80,62 @@ checklist is what actually catches that, the same reasoning behind the
|
||||
where the *why* isn't obvious from the code itself) — matches the
|
||||
no-comments-unless-non-obvious style already used throughout this
|
||||
project's docs and ADRs.
|
||||
|
||||
## Preserve provenance
|
||||
|
||||
Every extracted or transformed unit must retain enough provenance to trace
|
||||
it back to the source document — depending on the data type, this may
|
||||
include document id, page number, source block/span id, bounding box,
|
||||
reading-order position, table id and row/column coordinates, formula
|
||||
source span, monograph id, section path, and extraction method/parser
|
||||
version.
|
||||
|
||||
**Why:** this is a medical reference book being turned into a chatbot's
|
||||
knowledge base — if an answer is wrong, being able to trace a chunk back to
|
||||
the exact page/span it came from is how it gets debugged and corrected.
|
||||
Normalized text that "looks right" is not the same guarantee as text that
|
||||
is traceable.
|
||||
|
||||
**How to apply:**
|
||||
- Don't discard provenance fields just because the normalized text appears
|
||||
correct — a text value that can't be traced back to its source is not a
|
||||
fully validated extraction result.
|
||||
- When adding a new pipeline stage or record type, carry existing
|
||||
provenance fields through rather than dropping them at the boundary.
|
||||
|
||||
## Investigation scripts are evidence tools, not production code
|
||||
|
||||
One-off investigation scripts (e.g. scanning the corpus to check a
|
||||
hypothesis) may optimize for speed, but they must:
|
||||
- be clearly named or located as temporary investigation code;
|
||||
- state or record the scope they scanned;
|
||||
- output enough information to reproduce or inspect the finding;
|
||||
- not be imported by production code, and not become the only
|
||||
implementation of a parsing rule;
|
||||
- not be cited as whole-document evidence unless they actually covered the
|
||||
whole document;
|
||||
- be deleted after their finding is captured in a regression test, fixture,
|
||||
ADR, or the outlier catalog.
|
||||
|
||||
When an investigation uncovers a real parsing rule, move that rule into the
|
||||
production implementation and validate both the production code and the
|
||||
regression fixture — per [[DRY]] above, the rule should end up living in
|
||||
exactly one place.
|
||||
|
||||
## Definition of done
|
||||
|
||||
A task is not complete merely because code was written. Before reporting
|
||||
completion:
|
||||
- run the most relevant available tests and validation commands, and
|
||||
report the exact commands/checks run and whether each passed or failed;
|
||||
- state the validation scope (see "What 'verified' means" above);
|
||||
- add or update a regression fixture for each parser bug fixed;
|
||||
- confirm intended provenance fields remain present;
|
||||
- check for silent loss of expected monographs, sections, tables, formulas,
|
||||
or source references when the task could affect them;
|
||||
- avoid whole-document claims when only sample validation was performed;
|
||||
- list anything not tested, not measured, blocked, or still uncertain.
|
||||
|
||||
If only part of the task is complete, report the completed and incomplete
|
||||
parts separately — don't hide failing tests, unexpected counts, incomplete
|
||||
coverage, or contradictory evidence to present a cleaner status.
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
# ADR 0004: Chunking strategy for drug monographs — validated against real per-section measurements
|
||||
|
||||
## Status
|
||||
|
||||
Accepted for the monograph range (printed pp. 99-1496) only. General
|
||||
chapters (pp. 37-98) and appendices (pp. 1497-1528) are explicitly out of
|
||||
scope — see Consequences.
|
||||
|
||||
## Context
|
||||
|
||||
`docs/architecture.md`'s original "Chunking" paragraph specified `(drug,
|
||||
section)` as the chunk unit, a ~500-800 token budget, and a 400-token/
|
||||
50-overlap sliding window for oversized sections. Those numbers were written
|
||||
before segmentation existed — a plausible guess, never checked against real
|
||||
per-section text length.
|
||||
|
||||
Phase 1 (extract → segment → validate) is now real, tested code producing
|
||||
682 real monographs from the full 1668-page source PDF. This session ran
|
||||
`python -m ingestion.cli run` for real and measured actual per-section
|
||||
length across the whole corpus with a temporary investigation script
|
||||
(`ingestion/scratch/chunking_stats_survey.py`, deleted after this ADR
|
||||
captured its findings, per this project's investigation-script rule) —
|
||||
something that had never been measured before this ADR.
|
||||
|
||||
## What was actually measured (whole corpus, 682 monographs)
|
||||
|
||||
- Sections per monograph: min 11, median 17, max 19 (of ~18-19 known
|
||||
section keys in `segment/vocab.py`'s open taxonomy).
|
||||
- Whole-monograph length: median 11,480 chars, p90 19,068 chars, max 38,786
|
||||
chars.
|
||||
- Per-section length, converted to a **chars/4 token estimate — an
|
||||
estimate, not a real tokenizer count**:
|
||||
- Most of the ~18 section types sit comfortably under 800 estimated
|
||||
tokens even at their p90 (e.g. `chi_dinh` p90≈268 tok, `dang_thuoc_va_
|
||||
ham_luong` p90≈115 tok, `tac_dung_khong_mong_muon` p90≈481 tok).
|
||||
- **Two sections routinely exceed 800 tokens**:
|
||||
`duoc_ly_va_co_che_tac_dung` (242 of 678 monographs that have this
|
||||
section, 35.7%, max ≈3542 tok) and `lieu_luong_va_cach_dung` (200 of
|
||||
675, 29.6%, max ≈3631 tok).
|
||||
- A smaller tail also exceeds it: `than_trong` (25/680, 3.7%),
|
||||
`tuong_tac_thuoc` (22/642, 3.4%).
|
||||
- This means: the original 800-token ceiling is directionally correct
|
||||
(it clears ~16 of 18 section types at their p90 with room to spare),
|
||||
but "sub-chunk in that case" is not a rare hedge as originally implied
|
||||
— it is the routine path for roughly a third of all monographs, on two
|
||||
specific, named, high-clinical-importance sections (mechanism of
|
||||
action and dosing).
|
||||
|
||||
**A separate, blocking bug was found while gathering this data, not fixed
|
||||
by this ADR** (out of scope — belongs to `extract`/`segment`, owned by a
|
||||
parallel session at the time of writing): running header/footer
|
||||
boilerplate ("DTQGVN 2" + page number + repeated drug name, tagged
|
||||
`column="full_width"` in `extract/spans.py`) is never filtered out of
|
||||
section body text before it reaches `SectionSpan.text`. Measured:
|
||||
1,374 of 11,409 sections (12.0%) contain a literal "DTQGVN" string
|
||||
mid-text; 671 of 682 monographs (98.4%) have at least one affected section
|
||||
(e.g. MORPHIN SULFAT's `lieu_luong_va_cach_dung`: `"...Nếu\nDTQGVN 2\n1009\n
|
||||
Morphin sulfat\nuống viên thuốc..."`). This is `docs/pdf-parsing-outlier-
|
||||
catalog.md` item 13's known risk, measured whole-corpus for the first time
|
||||
here. **Chunking must not run against real data until this is fixed** —
|
||||
otherwise boilerplate is baked into embeddings and can surface mid-sentence
|
||||
in a chunk shown to a doctor or pharmacist.
|
||||
|
||||
## Decision
|
||||
|
||||
1. **Chunk unit stays `(drug_id, section_key)`** — matches
|
||||
`segment/models.py`'s existing `Monograph.sections: Dict[str,
|
||||
SectionSpan]`, matches how a doctor/pharmacist would query ("what does
|
||||
it say about liều dùng"), and lets a citation point at one clinical
|
||||
section rather than a whole 2,000-19,000-char monograph.
|
||||
2. **Token budget: keep the 800-token ceiling** (chars/4 estimate) as the
|
||||
split trigger. Below it, a section is one chunk, verbatim. This is now a
|
||||
validated choice, not a guess.
|
||||
3. **Sub-chunking only applies to the long-tail sections above** (~30-36%
|
||||
of monographs for the two named sections, a few percent for the rest).
|
||||
Method: **sentence-boundary-aware sliding window**, replacing the
|
||||
originally-guessed fixed-character window. Target ~600-700 tokens per
|
||||
sub-chunk (headroom under the 800 ceiling), ~1 sentence / 50-80 token
|
||||
overlap between adjacent sub-chunks. Split only at a sentence boundary
|
||||
(`.`/`;`/`:` followed by whitespace + capital letter), explicitly not
|
||||
treating a Vietnamese decimal comma (e.g. "0,425") as a boundary.
|
||||
4. **Why sentence-aware, not line- or character-based**: `assembler.py`
|
||||
joins `body_lines` one line per PyMuPDF *span*, i.e. one PDF visual
|
||||
line-wrap point — not a semantic paragraph or sentence boundary. A blind
|
||||
character/line window can split a sentence mid-way. This is a real,
|
||||
measured risk here, not theoretical: outlier-catalog item 17 found
|
||||
adult/child dosing splits ("Người lớn"/"Trẻ em") appear on 1,121 of
|
||||
~1,400 monograph-range pages — a chunk boundary landing inside one of
|
||||
those sentences would be a patient-safety-relevant defect, not a
|
||||
cosmetic one.
|
||||
5. **Chunk metadata / provenance** (extends the existing `drug_name,
|
||||
section_type, source_page_range, chunk_id` list in `docs/architecture.md`
|
||||
— per CLAUDE.md's provenance rule): `chunk_id`
|
||||
(`{drug_id}__{section_key}__{part_index}`), `drug_id`, `drug_name`,
|
||||
`section_key`, `section_display_name`, `atc_codes` (inherited from the
|
||||
monograph — enables ATC-class-filtered retrieval), `source_page_range`
|
||||
(monograph-level, see Consequences), `part_index`/`part_count` (`0`/`1`
|
||||
for un-split sections, keeps the schema uniform across all chunks).
|
||||
|
||||
## Consequences
|
||||
|
||||
- **Scope**: this decision covers the monograph range only. General
|
||||
chapters and appendices contain real tables and 2D stacked-fraction
|
||||
formulas (`docs/document-profile.md`, investigation in progress as of
|
||||
this ADR) that need their own structural survey before any chunking rule
|
||||
can be designed for them — do not extend this ADR's rules to those ranges
|
||||
without a fresh investigation.
|
||||
- **Hard prerequisite**: the boilerplate-leakage bug described above must
|
||||
be fixed in `extract`/`segment` before this chunking design is run
|
||||
against real data for ingestion. This ADR does not fix it.
|
||||
- **Known gap — sub-compound tagging inside class-level monographs**: 25.5%
|
||||
of the corpus has more than one ATC code per monograph (outlier item
|
||||
12a), e.g. "VITAMIN D VÀ CÁC THUỐC TƯƠNG TỰ" documents dosing for 7
|
||||
different analogues inside one `lieu_luong_va_cach_dung` section. No
|
||||
reliable structural signal was found in sampled text to split a section
|
||||
by sub-compound — a chunk from this section is tagged with the class
|
||||
name only, not the specific analogue a query might target. Deferred to
|
||||
golden-dataset-driven eval rather than guessed at now.
|
||||
- **Known gap — sub-chunk page precision**: `source_page_range` is
|
||||
monograph-level, not sub-chunk-exact. A sub-chunk from late in a
|
||||
multi-page section inherits the whole monograph's page range rather than
|
||||
its own precise page, because per-line page tracking doesn't currently
|
||||
exist in `SectionSpan`/`Heading`. The monograph + section-heading page is
|
||||
still real, checkable provenance, but this is a known precision gap, not
|
||||
full sub-chunk traceability. Flagged as a future improvement.
|
||||
- **Not yet built**: the Vietnamese sentence-boundary splitter itself
|
||||
(abbreviation handling, decimal-comma handling, ATC-code-period handling)
|
||||
is specified here as a rule, not implemented or unit-tested. Building and
|
||||
testing it is a separate, later task (`ingestion/ingestion/chunk/`, which
|
||||
does not exist yet).
|
||||
@@ -0,0 +1,216 @@
|
||||
# ADR 0005: `segment/` output contract needed by `chunk/` — structure-preserving, not flattened
|
||||
|
||||
## Status
|
||||
|
||||
Proposed. **Contract/schema only — no implementation.** `segment/models.py`,
|
||||
`segment/assembler.py`, and `segment/io.py` are actively owned by a parallel
|
||||
session on the same checkout at the time of writing; this ADR specifies what
|
||||
`chunk/` needs from `segment/`'s output precisely enough to implement and
|
||||
test, but does not touch those files itself. Supersedes part of ADR 0004
|
||||
(see "Relationship to ADR 0004" below) — ADR 0004's `(drug_id, section_key)`
|
||||
chunk-unit-as-leaf assumption is corrected here to `(drug_id, section_key)`
|
||||
as a **parent**, with sentence-window splitting demoted from primary
|
||||
strategy to fallback.
|
||||
|
||||
## Context
|
||||
|
||||
ADR 0004 designed chunking against `segment/models.py`'s current output:
|
||||
`SectionSpan.text` is a single flattened string per section (`"\n".join(
|
||||
body_line.strip() for ...)`), with all per-line style (`Span.bold`) and
|
||||
per-line page position discarded once the string is built (confirmed by
|
||||
reading `assembler.py`: `body_lines.append(span.text.strip())` keeps only
|
||||
`span.text`, nothing else). Review of ADR 0004 surfaced four real problems
|
||||
that trace back to this flattening, not to the chunking algorithm itself:
|
||||
|
||||
1. **A section is not a single semantic unit.** `liều lượng và cách dùng`
|
||||
and `tương tác thuốc` routinely contain multiple distinct facts (dosing
|
||||
per patient population, dosing per organ-function impairment, multiple
|
||||
separate drug interactions) that a doctor may want to retrieve
|
||||
independently. Measured: an explicit population marker ("Người lớn"/
|
||||
"Trẻ em"/"Trẻ sơ sinh"/"Suy thận"/"Suy gan" immediately followed by `:`
|
||||
or `.`) appears in **303 of 675 monographs (44.9%)** that have a `liều
|
||||
lượng và cách dùng` section — this is common, not an edge case.
|
||||
2. **A blind sentence-boundary sliding window (ADR 0004's original
|
||||
sub-chunking method) can still split two different facts into the same
|
||||
chunk, or split one fact across two chunks**, because it has no way to
|
||||
know a population/interaction boundary exists — that information exists
|
||||
in the source (as a bold or otherwise visually distinct sub-heading, per
|
||||
direct reading of MORPHIN SULFAT/VITAMIN D section text: lines like
|
||||
"Thuốc uống", "Cách dùng:" render as isolated bold short lines in the
|
||||
PDF) but is discarded before `chunk/` ever sees it.
|
||||
3. **Tables inside the monograph range are not addressed at all.** ADR 0004
|
||||
implicitly assumed monograph-range sections are prose. `docs/pdf-parsing-
|
||||
outlier-catalog.md` item 19 already documents a real table (dosing by
|
||||
renal function, HSV/CMV columns) inside a monograph body (Foscarnet
|
||||
natri, physical page 698) — flattening a table's rows into
|
||||
newline-joined body text destroys its row/column structure exactly the
|
||||
way outlier item 7 already describes for the appendix's 2D nomogram
|
||||
table. A whole-range survey to size this properly is in progress
|
||||
alongside this ADR (see "Not yet resolved" below).
|
||||
4. **Provenance is section-level, not chunk-level**, because per-line
|
||||
`physical_page`/`y0` (which `Span` already carries — see
|
||||
`extract/models.py`) is discarded at the same flattening point. For a
|
||||
section spanning several physical pages, a sub-chunk built from its
|
||||
final third currently has no way to know its own real page — it can
|
||||
only inherit the whole monograph's `source_page_range`. For medical
|
||||
citations this is not precise enough.
|
||||
|
||||
**A fifth, independently-found data-quality bug makes precise provenance
|
||||
even more necessary, not less**: the corpus's last-processed monograph
|
||||
(ZOLPIDEM) is never closed until true end-of-stream, and `assembler._classify`
|
||||
calls `match_section()`/`match_section_with_inline_value()` on every span
|
||||
with **no `in_monograph_range` gate** (unlike `_TextEvent` handling, which
|
||||
does check it). A spurious bold-text match on physical page 1655 — deep in
|
||||
the back-of-book "Mục lục tra cứu" brand-name index, confirmed by reading
|
||||
that page directly — overwrote ZOLPIDEM's real `tương tác thuốc`
|
||||
`SectionSpan` with an empty one and corrupted its `source_page_range` to
|
||||
`[1492, 1655]`. This is real content loss (measured: exactly 1 monograph
|
||||
affected, the last one processed — every other monograph is closed on
|
||||
schedule by the next monograph title, which *is* range-gated). Flagged for
|
||||
the session that owns `extract`/`segment`, not fixed here.
|
||||
|
||||
## Decision
|
||||
|
||||
Extend `segment/models.py`'s `SectionSpan` with a structured, line-level
|
||||
representation, additive to (not replacing) the existing flat `text` field
|
||||
— `chunk/` becomes a real, structure-aware consumer instead of re-deriving
|
||||
structure from a flattened string via ad hoc regex.
|
||||
|
||||
### New/changed types (`segment/models.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class BodyLine:
|
||||
text: str
|
||||
physical_page: int
|
||||
y0: float
|
||||
bold: bool # Span.bold, preserved instead of discarded
|
||||
|
||||
@dataclass
|
||||
class SectionSpan:
|
||||
key: str
|
||||
display_name: str
|
||||
heading: Heading
|
||||
text: str # UNCHANGED meaning, kept for
|
||||
# backward compat (see invariant below)
|
||||
lines: List[BodyLine] = field(default_factory=list) # NEW
|
||||
```
|
||||
|
||||
`lines` carries exactly the per-line signal `chunk/` needs to do its own
|
||||
job (population/subheading detection, precise page provenance) without
|
||||
`segment/` having to know anything about chunking — `segment/`'s
|
||||
responsibility stays "detect boundaries and preserve source structure," not
|
||||
"decide what a retrieval unit is" (Clean Architecture / SoC, per
|
||||
CLAUDE.md). Specifically, this is deliberately **not** a `is_subheading:
|
||||
bool` field computed by `segment/` — classifying "is this line a
|
||||
subheading a chunker should split on" is a chunking-time decision (what
|
||||
counts as a good split point can vary by strategy/eval results), not a
|
||||
segmentation-time one. `segment/` should stop discarding the raw signal
|
||||
(`bold`, `y0`, `physical_page`) it already has per span; it should not also
|
||||
start doing chunk-shaping judgment calls.
|
||||
|
||||
### Invariants
|
||||
|
||||
1. `text == "\n".join(l.text for l in lines).strip()` for every
|
||||
`SectionSpan`, for the lifetime of this contract — `lines` is a strictly
|
||||
additive refinement, never a divergent second source of truth. Any
|
||||
change to how body text is assembled (e.g. the boilerplate-stripping fix
|
||||
already applied by the other session) must update both fields from the
|
||||
same filtered span list, not `text` alone.
|
||||
2. `lines` is in reading order, matching the order `text`'s lines already
|
||||
implicitly have.
|
||||
3. Every `BodyLine.physical_page` satisfies `detector.in_monograph_range`
|
||||
for a `Span` on that page — i.e., **no line in any `SectionSpan.lines`
|
||||
may come from outside the monograph's real printed-page range**. This is
|
||||
the ZOLPIDEM bug's exact failure mode stated as an invariant: it was
|
||||
violated (a spurious section event was accepted from a fully
|
||||
out-of-range page precisely because no such check existed for section
|
||||
*events*, only for body *text* events). Enforcing this invariant closes
|
||||
that bug as a side effect, but the invariant is stated here as a
|
||||
contract requirement independent of any specific fix implementation.
|
||||
4. Every currently-open monograph must be finalized exactly once, at either
|
||||
(a) the next monograph title, or (b) true end-of-stream — with no third
|
||||
path (e.g., a stray out-of-range section match) able to silently mutate
|
||||
an already-"complete" monograph's sections after point (a) would
|
||||
otherwise have applied. (This is a restatement of invariant 3 from the
|
||||
monograph-lifecycle side, not a new requirement.)
|
||||
|
||||
### Migration impact
|
||||
|
||||
- **`segment/io.py`** (`_monograph_to_dict`/`_monograph_from_dict`,
|
||||
`write_monographs_jsonl`/`read_monographs_jsonl`): additive — serialize
|
||||
`lines` alongside the existing `text`/`heading` fields per section.
|
||||
Existing consumers reading only `text` (e.g. `segment/atc.py`'s
|
||||
`extract_atc_codes`, which regexes over `SectionSpan.text`) need no
|
||||
change, per invariant 1.
|
||||
- **`ingestion/data/processed/monographs.jsonl`**: schema grows a new
|
||||
optional-shaped field (`sections[key].lines`). No `schema_version` field
|
||||
currently exists in the serialized dict (checked `io.py` directly) —
|
||||
worth adding as part of this change, both for this migration and because
|
||||
`docs/architecture.md` already assumes "collection aliasing allows
|
||||
re-ingesting with a changed chunking strategy," which implies the
|
||||
ingestion output itself should be able to declare which schema shape it
|
||||
is.
|
||||
- **Existing 110 tests**: unaffected if invariant 1 holds — no assertion in
|
||||
the current suite inspects `lines` (it doesn't exist yet), and `text`'s
|
||||
value/semantics are unchanged.
|
||||
- **New tests required** (this ADR specifies them; implementation and the
|
||||
actual test code are not part of this ADR):
|
||||
1. Regression test reproducing the ZOLPIDEM failure shape: a synthetic
|
||||
span stream — last monograph's title and real sections, followed by
|
||||
spans whose `printed_page` is out of `in_monograph_range` but whose
|
||||
text matches a `vocab.py` section label — asserting the monograph
|
||||
closes with its real sections intact and the out-of-range spurious
|
||||
match is ignored, not accepted.
|
||||
2. `SectionSpan.lines` fixture test: using the real MORPHIN SULFAT
|
||||
boilerplate-fix fixture already in `tests/test_segment_assembler.py`,
|
||||
assert `lines` preserves the correct `bold`/`physical_page`/`y0` per
|
||||
retained line (and that stripped boilerplate lines are absent from
|
||||
`lines` too, not just from `text`).
|
||||
3. Round-trip test: `write_monographs_jsonl` → `read_monographs_jsonl`
|
||||
preserves `lines` exactly (dataclass equality per line).
|
||||
4. Whole-corpus invariant-1 check: for a real `cli run` output, assert
|
||||
`text == "\n".join(l.text for l in lines).strip()` holds for every
|
||||
section of every monograph, not a sample.
|
||||
|
||||
## Relationship to ADR 0004
|
||||
|
||||
ADR 0004's chunk-unit decision (`(drug_id, section_key)`) is **not**
|
||||
discarded — a section is still the natural *parent* grouping (matches how a
|
||||
clinician thinks, matches `Monograph.sections`). What changes: ADR 0004
|
||||
described a section as directly *the* chunk when under the 800-token
|
||||
ceiling, with sentence-window splitting as the fallback for oversized
|
||||
sections. Per the review above, splitting must instead **first** attempt to
|
||||
break at real structural boundaries available in `SectionSpan.lines` (a
|
||||
bold, short, isolated line — the same "subheading" shape already visually
|
||||
confirmed for route-of-administration/population sub-headers — or an
|
||||
explicit population/organ-function marker), with the sentence-window method
|
||||
demoted to a fallback for the remaining prose that has no such marker. The
|
||||
exact splitting algorithm (how a "subheading-shaped line" is defined
|
||||
precisely, in code) is a `chunk/`-side implementation detail *enabled* by
|
||||
this contract, not decided by it.
|
||||
|
||||
## Not yet resolved (explicitly out of scope for this ADR)
|
||||
|
||||
- **Table/formula content blocks.** A separate whole-monograph-range survey
|
||||
(pdfplumber `find_tables()` + PyMuPDF math-symbol scan, physical pages
|
||||
98-1494 excluding blank page 99 — the exact set `detector.
|
||||
in_monograph_range` accepts, not an assumed offset) is in progress at the
|
||||
time of writing, per explicit user instruction to measure before deciding
|
||||
a table/formula chunk-unit strategy. This ADR's `BodyLine`
|
||||
contract covers **text content only**; a table/formula region should
|
||||
*not* currently be flattened into `BodyLine`s (doing so would repeat
|
||||
exactly the "destroys row/column meaning" mistake outlier item 7 already
|
||||
documents) — but the precise `ContentBlock`/table-row/formula-unit shape
|
||||
is deferred to a follow-up revision of this ADR once the survey reports
|
||||
real numbers (how many monographs/sections affected, page-break
|
||||
continuation frequency, multi-tier headers, merged cells, footnotes).
|
||||
- **Paragraph-boundary detection** (grouping consecutive `BodyLine`s into a
|
||||
flowing paragraph vs. a new one) is left to `chunk/`, using the same
|
||||
kind of y-gap heuristic `segment/merge.py` already validates for
|
||||
multi-line title wraps (`_MAX_LINE_GAP_PT`) — `BodyLine.y0` is sufficient
|
||||
raw signal for `chunk/` to compute this itself; `segment/` does not need
|
||||
to pre-compute paragraph grouping.
|
||||
- **The actual `chunk/` splitting implementation** (subheading detector,
|
||||
population-marker regex, sentence-window fallback) is not part of this
|
||||
ADR — this ADR defines the data contract that implementation will consume.
|
||||
@@ -0,0 +1,171 @@
|
||||
# ADR 0006: chunks must carry references to lifted table/formula blocks
|
||||
|
||||
## Status
|
||||
|
||||
Proposed, with implementation to follow immediately. Resolves the item ADR
|
||||
0005 explicitly deferred ("Table/formula content blocks … the precise
|
||||
`ContentBlock`/table-row/formula-unit shape is deferred to a follow-up
|
||||
revision of this ADR once the survey reports real numbers"). The survey has
|
||||
reported.
|
||||
|
||||
## Context
|
||||
|
||||
`segment/` now lifts table and formula regions out of section prose and
|
||||
quarantines them (ADR 0003 lineage, outlier-catalog items 7, 8, 24, 25).
|
||||
That was the right move — linearised, AMPICILIN VÀ SULBACTAM's
|
||||
Cockcroft-Gault fraction read as `Clcr (ml/phút) = 72 x creatinin huyết
|
||||
thanh`, i.e. a division presented as a multiplication, in a renal-dosing
|
||||
section.
|
||||
|
||||
But `chunk/models.py` has no field that refers to a lifted block. Measured on
|
||||
the current whole-corpus output:
|
||||
|
||||
| quantity | value |
|
||||
|---|---|
|
||||
| lifted blocks | 167, all quarantined |
|
||||
| monographs affected | 96 of 683 (**14.1%**) |
|
||||
| sections affected | 108 |
|
||||
| **blocks in `lieu_luong_va_cach_dung`** | **127 (76%)** |
|
||||
| next largest section | `duoc_ly_va_co_che_tac_dung`, 16 |
|
||||
| shapes | simple_table 136, multi_level_or_merged_header 16, formula_2d 14, cross_page_continuation 1 |
|
||||
|
||||
So three quarters of everything removed from prose was removed from the
|
||||
dosing section, in a drug formulary, for an audience of doctors and
|
||||
pharmacists.
|
||||
|
||||
**The failure this creates is silent, not visible.** A chunk of AMPICILIN VÀ
|
||||
SULBACTAM's `lieu_luong_va_cach_dung` is grammatical, complete-looking prose
|
||||
with the renal-dosing table absent and nothing marking the absence. Retrieval
|
||||
ranks it, the model answers from it, and neither has any way to know a table
|
||||
was taken out. A visible error would be safer than this.
|
||||
|
||||
A second, quieter failure: a table is currently **unreachable**. Nothing in
|
||||
the index represents it, so "bảng liều theo chức năng thận của ampicilin"
|
||||
cannot retrieve it even in principle.
|
||||
|
||||
## Decision
|
||||
|
||||
Chunks reference blocks; blocks' content never becomes embedded text.
|
||||
|
||||
### 1. `Chunk` gains typed attachments
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class ChunkAttachment:
|
||||
block_id: str
|
||||
kind: str # "table" | "formula"
|
||||
shape: str # simple_table | multi_level_or_merged_header |
|
||||
# cross_page_continuation | formula_2d
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
quarantined: bool
|
||||
header_row: List[str] = () # simple_table only; see caveat below
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Chunk:
|
||||
...
|
||||
chunk_kind: str = "prose" # "prose" | "block_descriptor"
|
||||
attachments: List[ChunkAttachment] = ()
|
||||
has_quarantined_content: bool = False
|
||||
```
|
||||
|
||||
`has_quarantined_content` is derivable from `attachments`, and is serialized
|
||||
anyway. A consumer that never looks at `attachments` must still be unable to
|
||||
miss the fact — the whole defect being fixed here is a consumer not knowing
|
||||
what it was not told.
|
||||
|
||||
### 2. One descriptor chunk per block, built from metadata only
|
||||
|
||||
A block also gets its own chunk so it is retrievable at all:
|
||||
|
||||
```
|
||||
chunk_id = "{drug_id}:{section_key}:block:{block_id}"
|
||||
chunk_kind = "block_descriptor"
|
||||
text = "AMPICILIN VÀ SULBACTAM — Liều lượng và cách dùng — bảng,
|
||||
trang in 204. Cột: Độ thanh thải creatinin | Nửa đời |
|
||||
Liều ampicilin/sulbactam."
|
||||
```
|
||||
|
||||
The text is assembled from the drug name, the section display name, the kind,
|
||||
the printed page and — for `simple_table` only — the header row. **No cell
|
||||
value ever appears.** A header row is a row of labels; linearising it cannot
|
||||
invent a numeric relationship, which is precisely what linearising a body row
|
||||
does. For every other shape the header is omitted, because
|
||||
`multi_level_or_merged_header` is the shape whose header extraction is least
|
||||
trustworthy.
|
||||
|
||||
Caveat recorded in the schema itself: `header_row` comes from
|
||||
`pdfplumber.find_tables()`'s first row and has **not** been verified by eye
|
||||
(the 180 real tables' individual shapes are rule-derived; only the 20
|
||||
"not a table" verdicts were visually confirmed). It is retrieval bait, never
|
||||
an answer.
|
||||
|
||||
### 3. The answer layer's obligations (binding on `ai-service`)
|
||||
|
||||
These are stated here because they are the reason the schema exists; they are
|
||||
not implemented by `ingestion/`.
|
||||
|
||||
1. A retrieved chunk with `has_quarantined_content: true` **must** cause the
|
||||
answer to state that a table or formula exists at the cited page, and to
|
||||
surface its rendered crop. The answer may not present itself as complete.
|
||||
2. A `block_descriptor` chunk may be answered **only** with the crop. It must
|
||||
never be paraphrased, and its `header_row` must never be presented as the
|
||||
table's content.
|
||||
3. No chunk carrying a quarantined attachment may be used to state a numeric
|
||||
dose. If the dose is in the table, the answer is the crop plus the page.
|
||||
|
||||
### 4. `schema_version`
|
||||
|
||||
`monographs.jsonl` and the chunk output both gain `schema_version`. ADR 0005
|
||||
flagged its absence; a schema that now has two chunk kinds and an attachment
|
||||
list cannot be safely consumed without one.
|
||||
|
||||
## Alternatives rejected
|
||||
|
||||
- **Flatten the block into the chunk text.** This is the defect, not the fix
|
||||
— it reproduces `Clcr = 72 x creatinin` exactly.
|
||||
- **Chunk the block's linearised text as an ordinary chunk.** Worse than
|
||||
flattening: it makes unsafe text independently retrievable *as prose*, with
|
||||
its quarantine flag one dereference away from being ignored.
|
||||
- **Drop the blocks.** Silent loss, and contrary to the standing rule that
|
||||
unreconstructable content is quarantined with full provenance, never
|
||||
deleted.
|
||||
- **Rely on the prose saying "xem bảng".** The prose often does not, and a
|
||||
retrieval layer cannot act on an unstructured hint.
|
||||
- **Wait for row/column reconstruction and do this once.** Reconstruction is
|
||||
days of work and would leave the corpus unchunkable meanwhile; worse, it
|
||||
would make the schema question look answered when the *silent-incompleteness*
|
||||
problem is independent of whether the rows are recovered. Reconstruction
|
||||
later populates `rows` on the same attachment without touching consumers.
|
||||
|
||||
## Why a crop is a legitimate answer, not a placeholder
|
||||
|
||||
For doctors and pharmacists a rendered crop of the source page is the
|
||||
highest-fidelity response available: it *is* the book, and it is verifiable at
|
||||
a glance. Reconstruction earns its keep for a different job — comparing or
|
||||
combining values across drugs, which is the synthesis use case this product
|
||||
exists for — not for single-table lookup.
|
||||
|
||||
## Invariants and gates
|
||||
|
||||
Added to `cli chunk-ready` and to the chunk stage's own tests:
|
||||
|
||||
1. `section_with_lifted_block_but_no_chunk_reference = 0`
|
||||
2. `attachment_block_id_unknown = 0` — every referenced id exists on the
|
||||
monograph
|
||||
3. `attachment_without_page_or_bbox = 0`
|
||||
4. `block_text_leaked_into_chunk_text = 0` — no chunk's embedded text
|
||||
contains a quarantined block's text
|
||||
5. `descriptor_chunk_count == block_count`
|
||||
6. `descriptor_chunk_without_attachment = 0`
|
||||
|
||||
## Consequences
|
||||
|
||||
- Prose chunks shrink slightly in trustworthiness terms but grow in honesty:
|
||||
the ones missing a table now say so.
|
||||
- The index gains 167 descriptor chunks (≈1.4% of the expected chunk count),
|
||||
each cheap and none carrying unsafe text.
|
||||
- `ai-service` cannot be built to answer a dosing question from prose alone
|
||||
for the 108 affected sections without violating a stated contract.
|
||||
- The 14 `formula_2d` attachments make the two Cockcroft-Gault formulas
|
||||
answerable as crops today, which they are not now.
|
||||
+25
-6
@@ -80,12 +80,31 @@ methodology, cross-tool comparison, and validation numbers.
|
||||
`ingestion/data/processed/monographs.jsonl` and validated both
|
||||
automatically (see ADR 0003) and via manual spot-check in
|
||||
`ingestion/notebooks/`.
|
||||
3. **Chunking**: each `(drug, section)` pair is the natural chunk unit;
|
||||
never split a section unless it exceeds a token budget (~500-800 tokens),
|
||||
in which case sub-chunk with a sliding window (400 tokens, 50 overlap),
|
||||
tagging the same drug+section metadata plus `part_index`. Every chunk
|
||||
carries `drug_name`, `section_type`, `source_page_range`, `chunk_id` as
|
||||
Qdrant payload — this is what makes citations possible.
|
||||
3. **Chunking** (monograph range only, pp. 99-1496 — see
|
||||
`docs/adr/0004-chunking-strategy.md` for the full measured rationale):
|
||||
each `(drug_id, section_key)` pair is the chunk unit; a section stays one
|
||||
chunk if it's under an **800-token ceiling** (chars/4 estimate — a
|
||||
validated line, not a guess: whole-corpus measurement across 682
|
||||
monographs shows ~16 of 18 section types clear it comfortably at their
|
||||
p90). Two sections routinely exceed it — `dược lý và cơ chế tác dụng`
|
||||
(35.7% of monographs that have it) and `liều lượng và cách dùng`
|
||||
(29.6%) — sub-chunking is the **routine** path for those two, not a rare
|
||||
edge case. Oversized sections are split with a **sentence-boundary-aware
|
||||
sliding window** (~600-700 tokens/sub-chunk, ~1 sentence/50-80 token
|
||||
overlap), never a blind character/line window — PDF line-wrap points
|
||||
are not safe cut points, and a mid-sentence split risks separating an
|
||||
adult/child dosing instruction (a measured, common pattern — outlier
|
||||
catalog item 17) into two chunks. Every chunk carries `chunk_id`,
|
||||
`drug_id`, `drug_name`, `section_key`, `section_display_name`,
|
||||
`atc_codes`, `source_page_range`, `part_index`/`part_count` as Qdrant
|
||||
payload — this is what makes citations possible. **Known open gaps**
|
||||
(see ADR 0004): sub-compound tagging inside class-level/multi-ATC
|
||||
monographs (25.5% of the corpus) is not yet solved; `source_page_range`
|
||||
is monograph-level, not sub-chunk-exact; chunking for general chapters/
|
||||
appendices is a separate, not-yet-designed task; a confirmed
|
||||
header/footer-boilerplate leak into section text (98.4% of monographs
|
||||
affected) must be fixed upstream before this design runs against real
|
||||
data.
|
||||
4. **Embedding + load**: OpenAI `text-embedding-3-small` in batches, upserted
|
||||
into a versioned Qdrant collection (`drug_monographs_v1`) keyed by
|
||||
`chunk_id` for idempotent re-runs; collection aliasing allows re-ingesting
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
# Document Profile — Dược thư quốc gia Việt Nam 2018
|
||||
|
||||
Reverse-engineering survey of the source PDF (`duoc-thu-quoc-gia-viet-nam-2018.pdf`,
|
||||
1668 pages) to catalog every distinct page/content type BEFORE deciding what
|
||||
parser modules to build. **Classification only — nothing here changes the
|
||||
parsing pipeline.** Purpose: give real numbers to decide which content types
|
||||
are common enough to deserve a dedicated pipeline stage, per the "leverage
|
||||
the existing pipeline + add supplementary handling" direction agreed with
|
||||
the user (not a full architecture rewrite).
|
||||
|
||||
Method, per this project's standing rules ([[feedback-rigorous-validation]],
|
||||
ADR 0003): every count below is a **whole-document** scan (all 1668 pages,
|
||||
not a sample), classification rules are stated explicitly so any number can
|
||||
be independently re-checked, and every non-trivial claim is cross-checked
|
||||
with a second tool (`opendataloader-pdf`, the tool ADR 0003 validated for
|
||||
this purpose — **not** `pdfplumber`, which ADR 0003 already found scrambles
|
||||
reading order on this document) and/or a rendered-page-image visual read.
|
||||
|
||||
Reproducible script: `ingestion/scratch/document_profile_group1.py`
|
||||
(investigation code per CLAUDE.md's rules — temporary, not imported by
|
||||
production code; delete once this doc + any resulting regression fixtures
|
||||
fully capture its findings).
|
||||
|
||||
**Note on page numbering**: all page numbers below are physical/0-indexed
|
||||
(PyMuPDF convention). A PDF viewer's page counter is 1-indexed:
|
||||
`viewer page N == physical page N-1`.
|
||||
|
||||
## Group 1 — objectively measurable (done, verified)
|
||||
|
||||
| Category | Rule | Count | Verification |
|
||||
|---|---|---|---|
|
||||
| 2-column | page has both `column="left"` and `column="right"` spans (ADR 0003 bbox ranges) | 1628 | rule-based, matches known monograph-body layout |
|
||||
| Mixed/other layout | page has a set of column tags not matching the other 3 buckets | 32 | **100% manually viewed** (rendered every page) — see breakdown below, zero anomalies |
|
||||
| Full-width only | only `column="full_width"` spans | 5 | pages 3, 5, 37, 97, 1497 — all print-layout blank/divider-adjacent pages |
|
||||
| No text extracted | zero spans on the page | 2 | pages 99, 1666 |
|
||||
| Single-column-side | only `left` or only `right`, no `full_width` | 1 | page 1495 — near-empty (1 span), boundary page right at the monograph range end (1496) |
|
||||
| Near-empty (<20 chars) | `doc[p].get_text().strip()` length | 7 | pages 3, 5, 37, 99, 1495, 1497, 1666 — all print-layout blank/separator pages, consistent with ADR 0003's earlier finding of 6 (this scan found 1 more, page 5, confirmed same nature by direct read) |
|
||||
| Embedded images | `doc[p].get_images(full=True)` non-empty | 0 | 2 independent scans, 2 sessions, same result — **zero scanned pages in this document, no OCR needed** |
|
||||
| Chemical reaction equations (confirmed) | manual read of every regex candidate's context | **2** | see "Formula/notation" below — corrected from an initial loose-regex count of 25 |
|
||||
| Ion/electrolyte notation (Na+, Ca2+, Cl-, etc.) | same regex, reclassified after context read | ~23 pages (of the 25 original candidates) | common prose notation, not a "formula" needing special parsing — but subscript/superscript preservation matters, see below |
|
||||
| Comparison-operator notation (ADR frequency thresholds, "ADR > 1/100") | regex: digit adjacent to `<`/`>` | **933** | this is a **standard template pattern**, not an outlier — appears in the "Tác dụng không mong muốn (ADR)" section of most monographs, flagged by the user directly from a real page (Zolpidem, physical page 1494) |
|
||||
|
||||
### Mixed/other layout — full breakdown (32/32 pages viewed)
|
||||
|
||||
None are parsing anomalies. All are legitimate non-monograph content:
|
||||
|
||||
- **Front-matter title/cover/copyright pages**: 0, 1, 2
|
||||
- **Foreword**: 6
|
||||
- **Committee/personnel roster** (name lists, 2-column but different geometry than monograph body): 7, 9, 10, 11
|
||||
- **Table of contents**: 8
|
||||
- **"Danh mục các chuyên luận thuốc"** — Vietnamese\|English drug-name reference table, 2-column but different bbox geometry than the monograph body column rule (hence not tagged `two_column`): 12, 13, 14, 15, 16, 17, 18, 19, 21, 22, 24, 25, 26, 27, 28, 29, 30, 31 (18 pages; pages 20 and 23 of this same table happened to match the monograph-body bbox rule and are already counted under `two_column`)
|
||||
- **"Ký hiệu chữ viết tắt"** — abbreviation table, 3 columns (abbreviation \| English \| Vietnamese): 33
|
||||
- **Part-divider title pages**: 36 ("CÁC CHUYÊN LUẬN CHUNG"), 98 ("CÁC CHUYÊN LUẬN THUỐC"), 1496 ("CÁC PHỤ LỤC"), 1528 ("MỤC LỤC TRA CỨU")
|
||||
- **Blank separator**: 1529
|
||||
- **Colophon (print/publisher info)**: 1667
|
||||
|
||||
Potentially useful finding for future scope: the Vietnamese\|English name table
|
||||
(18-20 pages) could seed a synonym/alias table for search, if that's ever
|
||||
wanted — currently out of scope, noted only.
|
||||
|
||||
### Formula/notation — corrected finding
|
||||
|
||||
An initial loose regex found 25 candidate pages. **Reading the actual context
|
||||
of every match (cross-checked with `opendataloader-pdf`, not just PyMuPDF)
|
||||
showed this was the wrong classification** — most matches are ion/electrolyte
|
||||
charge notation (Na⁺, K⁺, Ca²⁺, Cl⁻, Mg²⁺, Fe²⁺/Fe³⁺, HCO₃⁻, PO₄³⁻, NH₄⁺),
|
||||
which is common, ordinary prose notation throughout the pharmacology text,
|
||||
not a distinct "formula" content type. Two unrelated `+`-adjacent patterns
|
||||
were also caught by the same regex and are semantically different again:
|
||||
"CD4+" (immunology cell-marker notation, not a chemical charge) and
|
||||
"O2 + N2O" (anesthetic gas mixture percentages).
|
||||
|
||||
**Only 2 pages have a genuine chemical reaction equation:**
|
||||
1. Physical page 1033 (already known, outlier-catalog item 16): cyanide
|
||||
antidote mechanism, `Na2S2O3 + CN⁻ → SCN⁻ + Na2SO3` — the reaction arrow
|
||||
extracts as a Private-Use-Area glyph (U+F0AF), not standard Unicode.
|
||||
2. Physical page 1027 (**new finding this session**, printed page 1028,
|
||||
"Natri bicarbonat"): buffer equation `HCO₃⁻ + H⁺ → H₂CO₃ → CO₂ + H₂O`,
|
||||
confirmed by rendering the page to an image — the source PDF renders
|
||||
this with real visual subscript/superscript.
|
||||
|
||||
**Real cross-cutting issue found, not yet sized or fixed**: both PyMuPDF's
|
||||
and `opendataloader-pdf`'s plain-text extraction **flatten subscript/
|
||||
superscript formatting** — the bicarbonate equation extracts as flat text
|
||||
("HCO-3+ H+ ... H2CO3 ... CO2 + H2O", digits inline, no vertical
|
||||
positioning info kept in the text string alone, though bbox/font-size data
|
||||
for the small subscript run is still recoverable from raw spans if a future
|
||||
stage needs to reconstruct it). This affects ion notation too, and likely
|
||||
also formula-adjacent abbreviations like "CD4", "Ca²⁺", "vitamin B₂/B₆/B₁₂"
|
||||
site-wide, not just these 2 pages — **the true scope of subscript/superscript
|
||||
loss has not been measured yet**, only observed on this one confirmed page.
|
||||
|
||||
### Mathematical formulas — separate from chemistry, found after the user
|
||||
asked "what about math" (this profile initially only scanned for chemistry-
|
||||
shaped tokens and missed this category entirely — a real gap, not a
|
||||
deliberate scope decision)
|
||||
|
||||
Whole-document regex scan for math symbols (full 1668 pages), initially run
|
||||
with PyMuPDF only — **caught by the user re-checking my methodology**
|
||||
("đừng dùng 1 con pymu" — don't rely on just one tool) — then re-verified
|
||||
against `opendataloader-pdf`'s independent whole-document text extraction
|
||||
(125s for all 1668 pages):
|
||||
|
||||
| Symbol | Meaning | Pages found (PyMuPDF) | Total occurrences: PyMuPDF | Total occurrences: opendataloader-pdf |
|
||||
|---|---|---|---|---|
|
||||
| `±` | mean ± SD | 44 | 95 | 95 ✅ |
|
||||
| `≤` | less-than-or-equal (dosing/lab thresholds) | 91 | 178 | 178 ✅ |
|
||||
| `≥` | greater-than-or-equal (dosing/lab thresholds) | 144 | 244 | 245 (off by 1, unexplained, not chased further — negligible vs. the total) |
|
||||
| `×` | multiplication | 19 | 50 | 50 ✅ |
|
||||
| `√`, `÷` | square root, division | 0 | 0 | 0 ✅ |
|
||||
|
||||
Two independent tools agree almost exactly (only the `≥` total differs, by
|
||||
1 out of 245) — real cross-tool evidence the symbol counts aren't a
|
||||
single-tool artifact, not just an assertion.
|
||||
|
||||
`≤`/`≥` join the already-found `<`/`>` (933 pages) as further evidence that
|
||||
**threshold/comparison notation is a pervasive, standard part of this book's
|
||||
dosing and lab-value template**, not a rare outlier — same conclusion as
|
||||
before, now with more symbols confirmed.
|
||||
|
||||
**`×` (19 pages) was individually context-checked (not just counted)** —
|
||||
splits into two real, different things:
|
||||
- **9 pages** use `×` only as dosing-frequency shorthand ("200 mg × 1
|
||||
lần/ngày" = "200mg, once a day") or scientific notation ("18 × 10⁶")
|
||||
— not a standalone formula: pages 61, 91, 153, 155, 516, 716, 794, 974, 1412.
|
||||
- **10 pages have genuine standalone calculation formulas** (variable =
|
||||
expression), found in the general-chapters section (printed 37-98,
|
||||
physical ~36-97) and one appendix: pages 43, 92, 94, 147, 206, 699, 853,
|
||||
1274, 1359, 1498. Examples: Cockcroft-Gault creatinine clearance
|
||||
(`Clcr(nam) = (140-tuổi)×thể trọng / (Ccr×72)`), MDRD GFR (`GFR(nam) =
|
||||
186 × (Ccr)^-1,154 × (tuổi)^-0,203`), the DuBois body-surface-area formula
|
||||
(`S = W^0.425 × H^0.725 × 71.84`, physical page 1498, Appendix 1),
|
||||
elimination half-life (`t½ = 0,693×Vd/Cl`), clearance (`Cl = Q×E`).
|
||||
|
||||
**Severe finding, confirmed visually, worse than the subscript-flattening
|
||||
issue above**: physical pages 43 and 94 (printed 44, 95 — "Sử dụng thuốc ở
|
||||
người suy giảm chức năng gan, thận" and the pharmacokinetics general
|
||||
chapter) were rendered to images and read directly. The PDF itself shows
|
||||
clean, properly typeset **stacked fractions** (numerator over denominator,
|
||||
e.g. `Cl_TP = D/AUC`, `t½ = 0,693×Vd/Cl`). But the plain-text extraction of
|
||||
these same formulas comes out **scrambled, not just subscript-flattened** —
|
||||
e.g. page 94's `Cl = Q × E = (Ca-Cv)/Ca` extracts as the fragment sequence
|
||||
`"Cl = Q × E = | a | v | a | C | C | C | Q | − | × |"`, unreadable and not
|
||||
recoverable by a simple flatten-subscript fix. This is a genuine reading-
|
||||
order defect specific to stacked-fraction layout, distinct from (and more
|
||||
severe than) the subscript-loss issue, confirmed on 2 pages so far — **not
|
||||
yet measured across all 10 real-formula pages**, only these 2 were rendered
|
||||
and read.
|
||||
|
||||
**Scope honesty**: the `×`/`±`/`≤`/`≥` regex families are still just
|
||||
*candidate* signals for "this page has notable math content" — a formula
|
||||
using only `/` for a fraction, or only superscript exponents with no `×` at
|
||||
all, would not be caught by this scan. The 10-page "genuine formula" count
|
||||
should be read as a lower bound, not a confirmed total.
|
||||
|
||||
**This also confirms a bigger open gap**: both real formulas and real data
|
||||
tables (Bảng 3, Bảng 4 — bordered tables with rows/columns, seen on page 43
|
||||
during the visual check) live in the **general chapters section (printed
|
||||
37-98)**, which per [[project-medical-chatbot-status]] memory has "never
|
||||
been structurally investigated." Group 2 below must cover this range, not
|
||||
just the monograph body.
|
||||
|
||||
## Group 2 — heading / table / list types
|
||||
|
||||
### Tables — in progress, NOT yet a trustworthy number
|
||||
|
||||
`opendataloader-pdf`'s JSON output (whole-document, converted in 99s) has
|
||||
built-in structural typing (`heading`/`table`/`list`/`paragraph`/`caption`),
|
||||
so this was tried first instead of hand-writing a table detector.
|
||||
|
||||
**Indexing pitfall caught before it became a wrong report**: opendataloader's
|
||||
`page number` field is **1-indexed** (confirmed via the RIBOFLAVIN reference
|
||||
point — its title lands at `page number: 1244`, and this document's
|
||||
physical(0-indexed)+1 == printed page always coincide, per ADR 0003's
|
||||
confirmed constant +1 offset — so `page number - 1 == PyMuPDF physical
|
||||
page`). An initial table-count query used the raw `page number` value
|
||||
unconverted and produced a count that only *coincidentally* matched a
|
||||
"2 tables" ground-truth check by luck — re-verified correctly afterward:
|
||||
physical page 43 (`page number 44`) shows 2 tables with captions "Bảng 3.
|
||||
Phân loại mức độ suy thận theo creatinin..." and "Bảng 4: ...tốc độ lọc cầu
|
||||
thận (GFR)" — an exact match to the page rendered and read directly
|
||||
earlier in this investigation.
|
||||
|
||||
**Current whole-document numbers from opendataloader-pdf alone (converted
|
||||
to physical 0-indexed pages)**:
|
||||
- 170 table elements, on 129 distinct pages.
|
||||
- 107 of those pages are inside the monograph range (98-1495 physical); 22
|
||||
are in the general-chapters range (physical 42-92, i.e. printed 43-93);
|
||||
none found yet in the appendices range beyond page 1498 and 1509.
|
||||
|
||||
**This count is NOT yet trustworthy as a final number** — it comes from a
|
||||
single tool, spot-checked correct on only 1 of 129 pages so far. Per
|
||||
ADR 0003, opendataloader's higher-level structural classifier (confirmed
|
||||
inconsistent for headings specifically) has an unknown reliability for
|
||||
tables specifically. Cross-checking now with `pdfplumber`'s
|
||||
`find_tables()`/`extract_tables()` — the tool ADR 0003 explicitly kept
|
||||
around *only* for table extraction (unlike its general text extraction,
|
||||
which is confirmed broken on this document) — whole-document run in
|
||||
progress, slower than opendataloader's, not complete as of this entry.
|
||||
**Do not cite the 170/129 numbers above as confirmed until this second
|
||||
tool's results are compared.**
|
||||
|
||||
### Headings, lists — not started
|
||||
|
||||
Requires proposing a taxonomy from real samples (per the "propose first,
|
||||
user reviews" approach agreed for this doc), since unlike Group 1's layout
|
||||
checks there's no purely objective rule to classify these — pending. The
|
||||
opendataloader JSON also has `heading` (3165) and `list` (1624) element
|
||||
counts whole-document, but per the table-count lesson above these should
|
||||
not be quoted as real numbers until cross-checked the same way.
|
||||
|
||||
## Known gaps in this profile itself
|
||||
|
||||
- Comparison-operator (933 pages) and ion-notation (~23 pages) candidates
|
||||
were pattern-matched but not each individually opened — the sample checks
|
||||
done (Zolpidem page for comparison-operators, all formula-regex contexts
|
||||
for ion notation) are consistent enough to trust the *category*, but a
|
||||
page-by-page audit of all 933/23 was not performed.
|
||||
- No table detection exists yet in this profile (Group 2 will need to define
|
||||
a table-detection rule before it can be counted). Confirmed real bordered
|
||||
tables exist at least on physical page 43 ("Bảng 3", "Bảng 4" — suy thận
|
||||
classification), found incidentally while visually checking a math
|
||||
formula, not from a deliberate table search.
|
||||
- General chapters (37-98 printed) and appendices (1497-1528 printed) have
|
||||
only been surveyed for Group 1's layout/blank/image/formula/math
|
||||
dimensions here — their own internal structure (headings, lists, full
|
||||
table inventory within those sections) is still unsurveyed. This range
|
||||
is now confirmed to contain real formulas and real tables (see Math
|
||||
section above), so it must be explicitly in scope for Group 2, not
|
||||
treated as monograph-adjacent filler.
|
||||
@@ -0,0 +1,116 @@
|
||||
# Kế hoạch phủ toàn bộ nội dung PDF (text + bảng + công thức + outlier)
|
||||
|
||||
**Trạng thái**: kế hoạch đang thực thi, lập 2026-07-31. Các ô ghi `[chờ đo]`
|
||||
là số liệu chưa có tại thời điểm viết — không được trích dẫn cho đến khi
|
||||
điền bằng kết quả chạy thật.
|
||||
|
||||
## Mục tiêu, phát biểu chính xác
|
||||
|
||||
Có hai mục tiêu thường bị gộp làm một. Kế hoạch này chỉ nhận mục tiêu A cho
|
||||
cuối ngày, và phát biểu rõ B là việc dài hơn.
|
||||
|
||||
| | Mục tiêu | Nhận cho cuối ngày? |
|
||||
|---|---|---|
|
||||
| **A** | **Phủ toàn bộ, không mất âm thầm**: mọi ký tự trong 1668 trang đều rơi vào đúng một rổ đầu ra hoặc vào rổ `unassigned` đếm được; mọi đối tượng không đáng tin đều bị gắn cờ tường minh; provenance giữ nguyên | **Có** |
|
||||
| **B** | **Đúng 100% đã chứng minh**: mọi bảng và công thức đã đối chiếu ground truth | **Không** — cần đối chiếu thủ công toàn bộ, là công người, không phải công máy |
|
||||
|
||||
Tuyên bố "parse được toàn bộ" chỉ hợp lệ theo nghĩa A. Bất kỳ báo cáo nào
|
||||
cũng phải nói rõ đang nói về A hay B.
|
||||
|
||||
## Vì sao không xây một bộ reconstruct tổng quát
|
||||
|
||||
Chưa biết trong sách có bao nhiêu bảng, bao nhiêu dạng cấu trúc, bao nhiêu
|
||||
trang continuation. Xây một bộ tổng quát trước khi biết phân bố dạng là đầu
|
||||
tư mù. Thứ tự bắt buộc: **kiểm kê → phân loại dạng → chọn đường xử lý theo
|
||||
từng dạng → mới code**.
|
||||
|
||||
## Giai đoạn
|
||||
|
||||
### A. Kiểm kê toàn corpus (đang chạy)
|
||||
|
||||
Script tạm `ingestion/scratch/inventory_tables_formulas.py`, scope toàn bộ
|
||||
1668 trang, xuất provenance từng đối tượng để soi lại được.
|
||||
|
||||
| Đại lượng | Kết quả |
|
||||
|---|---|
|
||||
| Số bảng pdfplumber tìm được / số trang có bảng | `[chờ đo]` |
|
||||
| Phân bố số cột | `[chờ đo]` |
|
||||
| Ứng viên continuation (bảng ở đầu trang/cột, không header) | `[chờ đo]` |
|
||||
| Lưới toàn số ≥4 cột (ứng viên 2D lookup, catalog item 7) | `[chờ đo]` |
|
||||
| Ứng viên công thức: fraction_bar / PUA / small_font_numeric | `[chờ đo]` |
|
||||
|
||||
Kiểm kê này **cố tình thiên về recall**: bắt thừa còn hơn bỏ sót; độ chính
|
||||
xác đo sau bằng kiểm tra trực quan.
|
||||
|
||||
### B. Sổ cái phủ ký tự — đây là eval chứng minh "trích xuất được"
|
||||
|
||||
Với mỗi trang trong 1668 trang, đối chiếu:
|
||||
|
||||
```
|
||||
chars_trên_trang_gốc == chars_vào_section_text
|
||||
+ chars_vào_ô_bảng
|
||||
+ chars_vào_vùng_công_thức
|
||||
+ chars_vào_front_matter / phụ lục
|
||||
+ chars_unassigned
|
||||
```
|
||||
|
||||
`unassigned` phải ra **một con số cụ thể kèm danh sách trang/bbox**, không
|
||||
phải một lời khẳng định. Đây là điểm khác biệt so với mọi eval trước đó
|
||||
trong dự án: recall/precision hiện tại chỉ đo **phát hiện ranh giới chuyên
|
||||
luận**, không đo nội dung; sổ cái này đo nội dung ở mức ký tự, whole-document,
|
||||
không phải mẫu.
|
||||
|
||||
Giới hạn phải nói rõ: sổ cái chứng minh **không mất**, không chứng minh
|
||||
**đúng thứ tự** hay **đúng ngữ nghĩa**. Thứ tự đã có kiểm tra riêng
|
||||
(`scan_reading_order`, `scan_glyph_order`); ngữ nghĩa thuộc mục tiêu B.
|
||||
|
||||
### C. Định tuyến theo dạng, mỗi dạng một đường
|
||||
|
||||
| Dạng | Xử lý | Metadata bắt buộc |
|
||||
|---|---|---|
|
||||
| Bảng có kẻ khung, header dạng chữ | Trích ô thật | `table_id`, `row`, `col`, `page`, `bbox` |
|
||||
| Bảng ngắt trang/cột (catalog item 5-6) | Gắn lại header gốc vào phần tiếp | thêm `continues_from` |
|
||||
| Lưới toàn số 2D (item 7) | **Không** chunk thành text | `do_not_cite: true` + giữ công thức đi kèm |
|
||||
| Công thức 1D (mũ inline) | Giữ nguyên text | `formula_kind: "1d"` |
|
||||
| Công thức 2D (có fraction bar) | Gắn cờ, giữ bbox + ảnh crop | `needs_review: true` |
|
||||
| Ký tự PUA (item: mũi tên lỗi) | Bảng thay thế tường minh | `pua_substituted` |
|
||||
|
||||
Mở/đóng theo SOLID: thêm một dạng mới = thêm một entry định tuyến, không
|
||||
sửa code đang chạy.
|
||||
|
||||
### D. Vùng ngoài chuyên luận
|
||||
|
||||
General chapters (tr. 37-98) và phụ lục (tr. 1497-1528) hiện **nằm ngoài
|
||||
phạm vi hoàn toàn** — pipeline chỉ sinh 682 chuyên luận. Hai vùng này phải
|
||||
hoặc vào sổ cái phủ, hoặc bị loại trừ tường minh kèm con số ký tự bị loại.
|
||||
Không được im lặng bỏ qua.
|
||||
|
||||
### E. Artifact bằng chứng
|
||||
|
||||
Mỗi đối tượng bị gắn cờ sinh một ảnh crop theo bbox đặt cạnh text trích ra,
|
||||
để mọi tuyên bố eval soi tận mắt được. Tự đọc ảnh để kiểm chứng, không đẩy
|
||||
việc kiểm tra sang người dùng.
|
||||
|
||||
## Số đo cần báo riêng, không gộp
|
||||
|
||||
Theo yêu cầu tránh gộp chỉ số che lấp điểm yếu:
|
||||
|
||||
- **detection recall** của detector trên golden set — bắt được bao nhiêu %
|
||||
đối tượng thật
|
||||
- **false positive** — bắt nhầm bao nhiêu
|
||||
- **số đối tượng chưa phân loại** — bao nhiêu cái detector không biết xếp vào
|
||||
đâu
|
||||
- **structural accuracy** — bảng tái tạo đúng hàng/cột bao nhiêu %
|
||||
- **semantic fidelity** — nội dung ô đúng bao nhiêu %
|
||||
|
||||
Detector dựa trên bbox là **heuristic**: nó tìm ứng viên, không chứng minh
|
||||
đã bắt hết mọi phân số, chỉ số, căn, ma trận hay lưới 2D. Mọi báo cáo phải
|
||||
đi kèm ba số đầu, không được nói suông "detector hoạt động tốt".
|
||||
|
||||
## Nợ kỹ thuật đã biết, chưa xử lý
|
||||
|
||||
- Ground truth từ Mục lục tra cứu **chưa được làm sạch**: chứa entry tham
|
||||
chiếu chéo lặp (ví dụ `"- CoA reductase, 285"` xuất hiện hơn 10 lần trong
|
||||
danh sách unmatched). Mẫu số 1064 hiện tại vì thế không đáng tin để chốt;
|
||||
ADR 0003 dùng mẫu số 725 nên hai lần đo **không so sánh trực tiếp được**.
|
||||
- Nội dung text chuyên luận chưa từng được đo độ chính xác so với nguồn.
|
||||
@@ -166,11 +166,14 @@ unreliable, always flag them") or its opposite ("formulas extract fine, no
|
||||
special handling needed") — neither is true here. The determining factor is
|
||||
whether the formula's visual layout is fundamentally 1D (left-to-right,
|
||||
like an inline exponent) or 2D (a fraction, a matrix, stacked terms).
|
||||
**Check:** no cheap automatic detector was built for this distinction yet —
|
||||
treat any equation/formula-like content as a manual-review candidate,
|
||||
especially anything with a fraction bar, until a real detector exists
|
||||
(e.g. checking for large vertical bbox gaps between adjacent glyphs that
|
||||
should be visually stacked).
|
||||
**Check:** a detector now exists — `residual_ink.py`'s
|
||||
`fraction_bar_candidate`, which finds the bar as ink no extracted span
|
||||
accounts for. Measured on this book: **precision 16/23 = 69.6%** (the misses
|
||||
are decorative underlines and table borders), recall unknown, and it is blind
|
||||
by construction to a fraction printed without a bar (item 25). Its output is
|
||||
therefore a review queue, not a verdict: all 23 candidates were rendered and
|
||||
read one at a time before any was acted on, and only the confirmed ones went
|
||||
into `ingestion/data/verified/formula_regions_2d.json`.
|
||||
**Generalizes:** yes — any technical/medical/scientific PDF with inline
|
||||
math will have this exact split; don't assume all formulas behave the same
|
||||
way in extraction.
|
||||
@@ -179,25 +182,65 @@ way in extraction.
|
||||
|
||||
## Character/glyph-level risks
|
||||
|
||||
### 9. Rare reversed (right-to-left) glyph-order defect
|
||||
**What it looks like:** confirmed exactly once across the entire
|
||||
1668-page book (physical page 1373): one short text run's glyphs are
|
||||
positioned in **descending x-order** rather than ascending, producing
|
||||
scrambled output (e.g. `" = tịx 8 yàgn gnàh uềil gnổt(..."`) that reverses
|
||||
character-by-character back to the correct Vietnamese sentence
|
||||
(`"(4 xịt = 800 microgam) vào buổi chiều..."`).
|
||||
**Why it matters:** this is a genuine, confirmed data-corruption risk, not
|
||||
theoretical — but it's also extremely rare (1 occurrence in 1668 pages), so
|
||||
it must be *detected*, not assumed to be either absent or common.
|
||||
**Check:** group text fragments into visual rows by rounded y-coordinate,
|
||||
then check whether x-coordinates are non-decreasing across the row; flag
|
||||
(and optionally auto-correct by re-sorting on x) any row that isn't. This
|
||||
full-book check runs in about 20 seconds.
|
||||
### 9. Rare reversed/misordered glyph defects — corrected count: 2, not 1
|
||||
**What it looks like:** re-implemented as real, tested production code
|
||||
(`ingestion/ingestion/extract/glyph_order.py`) rather than trusted from the
|
||||
earlier exploratory script's claim. Found **two distinct shapes**, not the
|
||||
one originally reported:
|
||||
1. **Within-span character reversal** (physical page 1373, the originally
|
||||
reported case): one span's glyphs are positioned in descending x-order,
|
||||
producing `" = tịx 8 yàgn gnàh uềil gnổt(..."`, which reverses
|
||||
character-by-character back to `"(4 xịt = 800 microgam) vào buổi
|
||||
chiều..."`.
|
||||
2. **Cross-fragment row misordering, newly found** (physical page 714): a
|
||||
single visual row is split by PyMuPDF into multiple `line` objects
|
||||
*within one block* that are then emitted out of left-to-right order —
|
||||
each fragment's own characters are fine, but concatenating fragments in
|
||||
extraction order produces `"...bảo quản ộđ tệihn "` instead of the
|
||||
correct `"...bảo quản nhiệt độ "`. This is a different underlying shape
|
||||
from item 1 (multiple mis-ordered fragments, not one reversed span) and
|
||||
was missed by the original narrower (within-span-only) check — the
|
||||
ADR 0003 claim of "exactly 1 occurrence in the whole book" undercounted
|
||||
the real defect population; corrected here.
|
||||
**Getting a trustworthy count took three detector iterations** (documented
|
||||
in the module's own docstring) — the first naive whole-book implementation
|
||||
of the row-level check reported **1113** "issues," almost all false
|
||||
positives from two mechanisms: (a) ordinary font-kerning jitter (e.g. in
|
||||
"mefloquin," two adjacent glyphs differ by 0.095pt — normal kerning, not a
|
||||
defect) treated as a reversal with no decrease-tolerance, which then
|
||||
actively *corrupted* correct text into "mefolquin"; and (b) reconstructing
|
||||
"visual rows" from raw x/y coordinates using a hand-picked column-boundary
|
||||
threshold, which misclassifies a paragraph that happens to start near the
|
||||
natural column gap (confirmed real case: a right-column paragraph starting
|
||||
at x=299.4 got merged with an unrelated left-column paragraph at the same
|
||||
y). The fix that survived whole-book testing: group by PyMuPDF's own
|
||||
`block` index (already validated in ADR 0003 to respect this document's
|
||||
column structure) instead of re-deriving columns from coordinates, plus a
|
||||
minimum-decrease threshold (1.0pt — safely between the ~0.3pt kerning noise
|
||||
floor and the >2pt real-defect magnitude). Final whole-book result: **11
|
||||
row-level issues on 5 pages** — 3 of those pages (92, 94, 805) are formula
|
||||
regions already flagged as unreliable in item 8 below (2D-layout formulas
|
||||
scramble on extraction; this check's "corrected" text for those rows should
|
||||
**not** be trusted or auto-applied, same as item 8's existing guidance),
|
||||
leaving exactly the 2 genuine prose defects above (pages 714, 1373).
|
||||
**Why it matters:** both genuine defects are confirmed real data-corruption
|
||||
risks, not theoretical — but both are also extremely rare (2 occurrences in
|
||||
1668 pages of prose), so they must be *detected*, not assumed either absent
|
||||
or common. Equally important: a naive implementation of "the obvious check"
|
||||
can itself introduce false positives and even actively corrupt correct
|
||||
text — this detector's own false-positive history is as important a lesson
|
||||
as the defects it catches.
|
||||
**Check:** `ingestion.extract.scan_glyph_order` (within-span) and
|
||||
`ingestion.extract.scan_reading_order` (cross-fragment, grouped by real
|
||||
PyMuPDF block index + row y, with a 1.0pt minimum-decrease threshold and
|
||||
header-band exclusion). Both run in seconds over the full book.
|
||||
**Generalizes:** yes, directly — this is a cheap, universal sanity check
|
||||
worth running on any PDF text-extraction pipeline as a standing QA gate,
|
||||
regardless of source document, since it catches a class of PDF-authoring
|
||||
defects (RTL/BiDi overrides, corrupted content streams) that have nothing
|
||||
to do with this book specifically.
|
||||
regardless of source document. The false-positive history also generalizes:
|
||||
any "reconstruct visual rows from raw coordinates" approach needs a
|
||||
decrease-tolerance (font kerning is universal) and should prefer the
|
||||
source tool's own layout-analysis groupings (blocks/lines) over hand-picked
|
||||
coordinate thresholds wherever available.
|
||||
|
||||
---
|
||||
|
||||
@@ -424,12 +467,349 @@ rather than treating them as failures.
|
||||
|
||||
---
|
||||
|
||||
### 15. No embedded images anywhere in the book — measured, not assumed
|
||||
**What it looks like:** a whole-book scan of `page.get_images(full=True)` across
|
||||
all 1668 pages returns **zero** embedded raster/vector images, confirmed via
|
||||
PyMuPDF's own image extraction API (not just "the text doesn't mention an
|
||||
image").
|
||||
**Why it matters:** avoids over-investing in image/caption validation tooling
|
||||
for a corpus that has no images to validate — but this must be a measured
|
||||
fact, not an assumption from the book's general description as "text-heavy."
|
||||
**Generalizes:** the check (`get_images(full=True)` summed over every page)
|
||||
is a cheap one-line whole-document verification worth running on any PDF
|
||||
before deciding whether image-handling code is needed at all.
|
||||
|
||||
### 16. Chemical reaction arrows render as Private-Use-Area glyphs, not Unicode arrows
|
||||
**What it looks like:** confirmed real example — physical page 1033 contains a
|
||||
genuine chemical reaction equation (`Na2S2O3 + CN⁻ → SCN⁻ + Na2SO3`, part of
|
||||
the cyanide-antidote/rhodanese mechanism description). The reaction arrow
|
||||
extracts as a Private-Use-Area codepoint (``), not a standard Unicode
|
||||
arrow (`→`) — the source PDF's font maps a custom symbol glyph (likely from a
|
||||
symbol/wingdings-style embedded font) into a PUA slot, and raw text extraction
|
||||
faithfully returns that codepoint rather than a human-readable arrow.
|
||||
**Why it matters:** any pipeline that treats extracted text as directly
|
||||
human-readable/citable will surface a mangled or invisible character where a
|
||||
reaction arrow should be; a naive keyword/embedding step over raw text would
|
||||
either silently drop it (if PUA codepoints get filtered as junk) or leave a
|
||||
confusing tofu/box character in a chunk shown to a doctor or pharmacist.
|
||||
**Check:** scan extracted text for codepoints in the Unicode Private Use Area
|
||||
ranges (`U+E000–U+F8FF`) — cheap and generalizes to any custom-glyph symbol
|
||||
substitution, not just arrows.
|
||||
**Handling:** for now, flag any monograph/section containing a PUA codepoint
|
||||
for manual review or map known PUA codepoints (e.g. this book's `` →
|
||||
`→`) via an explicit substitution table; do not pass raw PUA codepoints
|
||||
through to chunking/embedding untranslated.
|
||||
**Generalizes:** yes — any PDF built from print-authoring software that uses a
|
||||
symbol font for arrows/special glyphs (common in scientific/medical/chemistry
|
||||
documents) can exhibit this; always check for PUA codepoints in extracted
|
||||
text as a standing sanity check, not just assume standard Unicode symbols.
|
||||
**Confirmed real chemical formula in the corpus, but rare:** a regex scan for
|
||||
molecular-formula-shaped tokens (`[A-Z][a-z]?\d{1,3}` repeated) across the
|
||||
monograph page range found 9 raw hits; manual inspection found most are
|
||||
**false positives** (`H5N1` = flu strain name, `P2Y12` = a receptor name, not
|
||||
molecular formulas) and only one confirmed genuine chemical formula/equation
|
||||
(the Na2S2O3 case above) — real chemical notation exists in this corpus but
|
||||
is genuinely rare, not a systemic pattern requiring a general chemistry
|
||||
parser.
|
||||
|
||||
### 17. Adult/child dosing-population splits are the norm, not an edge case
|
||||
**What it looks like:** measured via a whole-monograph-range text scan for
|
||||
"Người lớn"/"Trẻ em"/"Trẻ sơ sinh" (adult/child/newborn) — these terms appear
|
||||
on **1121 of ~1400** monograph-range pages, i.e. the large majority of drug
|
||||
monographs split dosing by patient population.
|
||||
**Why it matters:** this is exactly the kind of structural content where a
|
||||
segmentation/chunking bug that interleaves or merges adjacent subsections
|
||||
(e.g. a table/list continuation bug, see items 5-6) would be a genuine
|
||||
patient-safety risk, not just a data-quality nicety — mixing an adult dose
|
||||
into a child-dose chunk (or vice versa) is a plausible, concrete failure
|
||||
mode given how common this structure is.
|
||||
**Handling:** treat "does this monograph's dosing section correctly keep
|
||||
adult/child/newborn subsections un-interleaved" as a standing validation
|
||||
check (not a rare-case afterthought), given the measured prevalence.
|
||||
**Generalizes:** yes — any clinical/pharmacological reference document
|
||||
organized with population-specific subsections has this same risk profile;
|
||||
measure real prevalence before deciding how much validation effort a
|
||||
structural risk deserves (same methodology lesson as item 12a).
|
||||
|
||||
### 18. A monograph title can legitimately repeat — disambiguated by a bold, non-caps qualifier line
|
||||
**What it looks like:** confirmed real example, found while smoke-testing
|
||||
the real `segment/detector.py` against the full book: "SALBUTAMOL" is
|
||||
detected as a monograph title **twice** (physical pages 1261 and 1263).
|
||||
Rendering both pages to images and reading them directly (not inferred from
|
||||
coordinates) confirmed these are two genuinely different, complete
|
||||
monographs — "SALBUTAMOL (Dùng trong hô hấp)" (respiratory use) and
|
||||
"SALBUTAMOL (Dùng trong sản khoa)" (obstetric/tocolytic use) — each with
|
||||
its own full 18-section template. The qualifier ("(Dùng trong hô hấp)" /
|
||||
"(Dùng trong sản khoa)") is a bold line immediately below the all-caps
|
||||
title, but is **not itself all-caps** (mixed case inside the parens), so it
|
||||
is correctly excluded from `detect_monograph_titles`'s all-caps candidate
|
||||
filter — it must instead be captured as a *separate* signal and folded into
|
||||
the monograph's disambiguating identity downstream.
|
||||
**Why it matters:** an assembler that derives `drug_id` from the title text
|
||||
alone (e.g. a simple slug of "SALBUTAMOL") will produce a real collision
|
||||
between two legitimately different monographs — this is **not** the same
|
||||
failure mode as the already-fixed GONADOTROPIN false-collision (that one
|
||||
was a detector artifact from unmerged multi-line wrapping; this one is a
|
||||
genuine same-name-different-monograph case that must be preserved, not
|
||||
merged away).
|
||||
**Handling (for Phase 1.3's assembler):** after detecting a monograph title,
|
||||
check for an immediately-following bold, parenthesized, non-all-caps line
|
||||
directly below it (same page, small y-gap) and include it in `drug_id`
|
||||
generation when present, so "salbutamol_ho_hap" and "salbutamol_san_khoa"
|
||||
remain distinct rather than colliding as "salbutamol" twice. The
|
||||
`assembler.py` duplicate-drug_id check (outlier-catalog reasoning already
|
||||
established: raise on a genuine duplicate rather than silently overwriting)
|
||||
must be designed with this real case in mind, or it will incorrectly reject
|
||||
a legitimate second "SALBUTAMOL" entry.
|
||||
**Generalizes:** yes — any drug/entity reference work that documents the
|
||||
same base substance under multiple distinct use-contexts (formulation,
|
||||
indication, route) can have this exact pattern; never assume a title string
|
||||
alone is a unique key without checking for a disambiguating qualifier line.
|
||||
|
||||
### 19. Table column headers can be bold + all-caps + short — identical shape to a real title
|
||||
**What it looks like:** confirmed real example, found via a whole-book
|
||||
`assemble()` run raising a duplicate-drug_id error: "HSV" and "CMV" each
|
||||
appear twice as bold, all-caps, short (3-char) spans on physical page 698 —
|
||||
not drug names at all, but **column headers in a dosing-by-renal-function
|
||||
table** inside the "Foscarnet natri" monograph ("Liều đối với HSV / HSV /
|
||||
CMV / CMV"). Bold+all-caps+short is exactly the monograph-title signal
|
||||
(item 10/12d), so this is a genuine detector ambiguity, not a coding bug.
|
||||
**Why it matters:** unlike item 12d's part-divider titles (a small,
|
||||
enumerable, fixed set of known strings), a table's column headers are
|
||||
unbounded and content-dependent (any future table could use "HSV", "CMV",
|
||||
or something else entirely as a header) — an exclusion list approach
|
||||
doesn't generalize here the way it did for part-dividers.
|
||||
**Handling:** require a **structural anchor** rather than a text exclusion
|
||||
list: a real monograph title is always followed shortly by at least one
|
||||
recognized section heading from the vocabulary (in practice, always "Tên
|
||||
chung quốc tế" first) before the next title-shaped candidate. A
|
||||
table-header false positive is not — the table's own cells are numbers/
|
||||
plain text, matching no vocabulary entry. Implemented as
|
||||
`assembler._filter_false_positive_titles` (lookahead of 6 events, checked
|
||||
against the same coalesced event stream already built for assembly — no
|
||||
separate detection pass, no duplicated logic).
|
||||
**Generalizes:** yes, more broadly than item 12d — any document where
|
||||
section/entity boundaries are marked by a *shape* (bold+caps+short) that a
|
||||
table, list, or figure caption could coincidentally also match should
|
||||
verify a **structural follow-on anchor**, not just a shape match or a
|
||||
denylist of known bad strings, since the space of possible false-shaped
|
||||
content (table headers, figure labels, pull-quotes) is unbounded while the
|
||||
space of "what a real boundary is followed by" is small and known.
|
||||
|
||||
### 20. Section headings are not consistently bold across monographs — some combine label+value in one plain span
|
||||
**What it looks like:** confirmed real example, found by investigating why
|
||||
a whole-book `assemble()` run showed 48 monographs with zero ATC codes and
|
||||
not stated-absent (far more than the ~13-14 the original spot-check
|
||||
extrapolated). AMITRIPTYLIN's real "Mã ATC:" field is a **single, plain
|
||||
(non-bold)** span containing the label AND value together: `"Mã ATC:
|
||||
N06AA09."` — unlike Abacavir's equivalent, which is a bold `"Mã ATC: "`
|
||||
label span followed by a separate plain `"J05AF06."` value span. Both
|
||||
render visually similar but have completely different span/style
|
||||
structure. Given the book's own foreword states it was "biên soạn bởi
|
||||
nhiều tác giả" (written by many authors), this kind of per-author styling
|
||||
inconsistency across ~700 individually-authored monographs is plausible
|
||||
and, once checked, confirmed real — not a one-off.
|
||||
**Why it matters:** a detector that requires `span.bold` to recognize a
|
||||
section heading (reasonable-looking given every *title* is confirmed bold)
|
||||
silently drops entire sections for a meaningful fraction of the corpus —
|
||||
this directly caused undercounted ATC codes (and, structurally, would
|
||||
equally affect any other section) for monographs using this looser style.
|
||||
**Handling:** match section headings by **vocabulary text**, not by
|
||||
boldness — the same "don't gate on a styling attribute, only content is
|
||||
reliable" lesson as item 10 (font size), now applied to boldness. Also
|
||||
handle the "label + value combined in one span" shape explicitly (a prefix
|
||||
match: does the span start with a known label followed by ":", with the
|
||||
remainder treated as the section's inline value) rather than assuming
|
||||
label and value are always separate spans.
|
||||
**Generalizes:** yes — any print-authored reference work assembled from
|
||||
many individual authors/editors over a long production process should
|
||||
expect inconsistent low-level styling of nominally-identical structural
|
||||
elements; verify a structural signal (styling) against the *content* it's
|
||||
supposed to correlate with, across a large real sample, before trusting it
|
||||
as a universal discriminator — the same methodology lesson as item 10,
|
||||
found again independently here.
|
||||
|
||||
### 21. "All-caps" is not 100% reliable either — and a class-level monograph's own internal sub-headings can masquerade as new monographs
|
||||
**What it looks like:** two distinct confirmed real findings from the same
|
||||
investigation:
|
||||
1. The class-level monograph "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE" embeds the
|
||||
mixed-case abbreviation "CoA" (Coenzyme A) inside an otherwise all-caps
|
||||
title. A strict `text.isupper()` check requires *zero* lowercase
|
||||
letters, so this single embedded abbreviation caused the entire
|
||||
monograph to be silently dropped from the corpus — found only by
|
||||
directly checking whether this specific, previously-known (outlier item
|
||||
12a) class-level monograph was present in a real whole-book `assemble()`
|
||||
run, and discovering it was not.
|
||||
2. Within that same class-level monograph, individual statin names
|
||||
("SIMVASTATIN", "LOVASTATIN", "PRAVASTATIN", "FLUVASTATIN") appear as
|
||||
their own bold+all-caps+short sub-headings, each introducing its own
|
||||
"Liều lượng và cách dùng" sub-section — shape-identical to a real
|
||||
monograph title, and (after fix 1 above made the loosened "any known
|
||||
section" anchor check pass) briefly became a second false-positive
|
||||
category alongside item 19's table headers, since these sub-headings
|
||||
*are* followed by a recognized section, just never by "Tên chung quốc
|
||||
tế" specifically (that section belongs only to the parent).
|
||||
**Why it matters:** together these show that neither "all-caps" nor "loosen
|
||||
the anchor to any section" is safe in isolation — the fix for one false
|
||||
positive (item 19, HSV/CMV) reopened a different one (SIMVASTATIN) until
|
||||
the anchor check was tightened back to the *specific* section the book's
|
||||
own template guarantees is always first for a genuine top-level monograph.
|
||||
**Handling:** `detector._is_mostly_upper` uses a **lowercase-letter ratio**
|
||||
(≤10%), not an absolute count — an earlier absolute-count version (≤2
|
||||
lowercase letters) let a real regression through: "Mã ATC:" has only 1
|
||||
lowercase letter (a normal Vietnamese diacritic, 'ã') but that's 20% of its
|
||||
5 letters, correctly rejected by the ratio while HMG-CoA's 1/27 ≈ 3.7%
|
||||
correctly passes. `assembler._has_anchor_ahead`
|
||||
requires specifically the "ten_chung_quoc_te" section key, not just any
|
||||
recognized section, since that is the one invariant the book's documented
|
||||
template actually guarantees is unique to real top-level monographs.
|
||||
**Generalizes:** yes — (1) don't assume a styling/casing convention holds
|
||||
with zero exceptions across an entire corpus, even one confirmed exception
|
||||
matters at whole-corpus scale; (2) when a document has nested substructure
|
||||
that mimics top-level structure (a class monograph containing per-item
|
||||
sub-entries), the anchor used to confirm a real boundary must be the most
|
||||
*specific* invariant available, not just "some known follow-on content" —
|
||||
a looser check that fixes one false positive can silently reopen another.
|
||||
|
||||
### 22. Running-header boilerplate was never actually stripped, despite item 13's warning — measured whole-corpus at 98.4% of monographs affected
|
||||
**What it looks like:** the running header at the top of every physical page
|
||||
("DTQGVN 2" + printed page number + the current monograph's name, e.g.
|
||||
physical page 1008's "DTQGVN 2" / "1009" / "Morphin sulfat", tagged
|
||||
`column="full_width"` by `extract/spans.py`) matches no section heading and
|
||||
isn't a real all-caps title, so it fell through every classification branch
|
||||
in `assembler._classify` into plain body text — splicing itself into the
|
||||
*middle* of whatever section is open when a physical page turns. Real
|
||||
example, MORPHIN SULFAT's `liều lượng và cách dùng`: `"...Nếu\nDTQGVN 2\n
|
||||
1009\nMorphin sulfat\nuống viên thuốc..."` — the header text lands inside a
|
||||
real dosing sentence.
|
||||
**Why it matters:** item 13 (above) already *warned* "strip the fixed
|
||||
boilerplate before parsing content" back when the extraction layer was
|
||||
first built, but that step was never actually implemented in `assembler.py`
|
||||
— the warning existed in the catalog without a corresponding code path or
|
||||
test enforcing it, and nothing caught the gap until a whole-corpus
|
||||
measurement was actually run. Measured: **1,374 of 11,409 sections (12.0%)
|
||||
contained a literal "DTQGVN" string mid-text; 671 of 682 monographs (98.4%)
|
||||
had at least one affected section** — this is not a rare edge case, it's
|
||||
the default outcome for any section whose text happens to cross a physical
|
||||
page boundary (i.e. most sections longer than about half a page). Left
|
||||
unfixed, boilerplate gets baked into chunks and embeddings and can surface
|
||||
mid-sentence in a citation shown to a doctor/pharmacist.
|
||||
**Handling:** `assembler._is_page_boilerplate` drops any span with
|
||||
`column == "full_width"` and `y0 < HEADER_BAND_Y` (the same header-band
|
||||
threshold `page_map.py` already uses to read the folio) before it reaches
|
||||
any other classification branch. Whole-corpus re-measurement after the fix:
|
||||
0 of 11,409 sections contain "DTQGVN". Regression test uses the exact real
|
||||
MORPHIN SULFAT span shape.
|
||||
**Generalizes:** a documented risk in this catalog is not the same as a
|
||||
verified-fixed risk — "we know this could happen" needs a whole-corpus
|
||||
measurement (not just a warning paragraph) before it can be crossed off,
|
||||
and ideally a regression test that would fail if the fix were ever reverted.
|
||||
|
||||
### 23. PyMuPDF's raw block order doesn't reliably sequence left-column-before-right-column — confirmed wrong on 12 of 1398 pages
|
||||
**What it looks like:** `extract/spans.py` originally trusted PyMuPDF's own
|
||||
block iteration order to already emit left-column content before
|
||||
right-column content, validated only against one example page during ADR
|
||||
0003. On physical page 1100 (the OXYBUTYNIN/OXYMETAZOLIN monograph
|
||||
boundary) and 11 other pages, PyMuPDF's raw block order emits the *right*
|
||||
column first. Since `assembler.assemble` appends section content to
|
||||
whichever monograph is currently open, this silently attributed
|
||||
OXYMETAZOLIN's right-column sections (Chống chỉ định, Thận trọng, Thời kỳ
|
||||
mang thai, Thời kỳ cho con bú, ADR, Hướng dẫn xử trí ADR, Liều lượng và
|
||||
cách dùng) to the still-open OXYBUTYNIN monograph — overwriting
|
||||
OXYBUTYNIN's real sections and leaving OXYMETAZOLIN missing all 7.
|
||||
**Why it matters:** medically relevant — wrong contraindication/ADR content
|
||||
silently attached to the wrong drug. Found via a whole-document
|
||||
(1668-page) character-similarity diff against an independent parser
|
||||
(`opendataloader-pdf`), not from a sample; confirmed by rendering the page
|
||||
to an image and reading it directly, then confirmed again in the actual
|
||||
`assemble()` output.
|
||||
**Handling:** `extract.spans._sort_blocks_reading_order` explicitly sorts
|
||||
each page's blocks by (full_width header band first, then left column,
|
||||
then right column) and then by y-position, instead of trusting raw PyMuPDF
|
||||
order. Whole-range (99-1496) re-scan after the fix: 0 pages with the
|
||||
reversed-order signature (was 12). Directly verified OXYBUTYNIN's and
|
||||
OXYMETAZOLIN's `assemble()`-produced sections are now distinct and
|
||||
drug-appropriate.
|
||||
**Generalizes:** don't trust an upstream library's element ordering just
|
||||
because it happened to be correct on the one page checked during initial
|
||||
validation — for a whole-corpus pipeline, explicitly sort by the actual
|
||||
signal you care about (here: visual column position) rather than an
|
||||
implicit "the library probably does this right" assumption.
|
||||
|
||||
### 24. Some text exists only as vector outlines — no text extractor can read it, and single dropped glyphs corrupt otherwise-clean sentences
|
||||
**What it looks like:** physical page 714 prints 17 full lines of ordinary
|
||||
GATIFLOXACIN prose that `page.get_text()` does not return, `page.search_for()`
|
||||
cannot find, and neither `pdfplumber` nor `opendataloader-pdf` returns either.
|
||||
`page.get_drawings()` shows why: each line is a filled path of 1,126-1,831
|
||||
items, shaped exactly like one line of type and filled with the body-text
|
||||
colour. The same defect occurs at glyph granularity (39-45 path items), and
|
||||
that form is far more dangerous — a single Vietnamese diacritic character
|
||||
drops out of a line that otherwise extracts perfectly: `Độ ổn định` extracts
|
||||
as `Độ n định`, `≥ 1 tuổi` as `≥ 1 tu i`, `tại chỗ` as `tại ch `. The result
|
||||
reads as ordinary text, so no structural check, no count and no cross-tool
|
||||
comparison notices it.
|
||||
**Why it matters:** this is silent loss of clinical prose in a drug
|
||||
formulary, and it is invisible to every check that asks a text layer a
|
||||
question. It survived a whole-document span-coverage ledger reporting
|
||||
`unassigned = 0`, because the spans that existed were all routed correctly —
|
||||
the missing content was never a span at all.
|
||||
**Check:** render the page, white out every extracted span's bbox, and look
|
||||
at the ink that survives (`ingestion/validation/residual_ink.py`, ~0.06
|
||||
s/page). Confirm with `page.get_drawings()`: a filled path with ≥30 items
|
||||
whose box is 3-20pt tall is type, not decoration (real decoration on this
|
||||
book carries 1-2 items).
|
||||
**Handling:** `ingestion/extract/outlined_text.py` detects the runs;
|
||||
recovery cannot be automatic because the paths carry no character codes, so
|
||||
each run was rendered and transcribed by reading it, into
|
||||
`ingestion/data/verified/outlined_text_transcriptions.json` with page, bbox,
|
||||
and the extracted line it belongs to. Whole-document scope: **51 runs on 5
|
||||
pages** (714 ×31, 736 ×16, 1373, 1444, 1445 ×2), 1,116 characters.
|
||||
**Generalizes:** yes — any PDF produced by a layout tool that converts
|
||||
selected text to outlines (common when a font cannot be embedded) has this.
|
||||
Never treat "the text layer returned something for this page" as evidence
|
||||
the page was fully extracted; compare against the rendered pixels.
|
||||
|
||||
### 25. A fraction can be printed with no fraction bar at all, so no geometric detector can find it
|
||||
**What it looks like:** ADENOSIN (physical page 147) prints its infusion-rate
|
||||
formula as three plain lines — `Tốc độ truyền dịch (ml/phút) = 0,140
|
||||
(mg/kg/phút) × trọng lượng cơ thể (kg)` / `Nồng độ adenosin (3 mg/ml).` —
|
||||
with **no rule drawn between numerator and denominator**, confirmed by
|
||||
rendering the region and reading it. Extracted linearly it reads as a
|
||||
multiplication chain, i.e. the division silently disappears.
|
||||
**Why it matters:** it defeats the detector that catches every other 2D
|
||||
formula in this book. The fraction-bar signal (item 8, and
|
||||
`residual_ink.py`'s `fraction_bar_candidate`) finds ink; there is no ink to
|
||||
find here. It was caught only because a prose-leak gate matched its text.
|
||||
**Check:** there is no cheap automatic check. Treat any line ending in a
|
||||
unit-bearing quantity immediately followed by a line that is itself a
|
||||
unit-bearing quantity as a division candidate for human review.
|
||||
**Handling:** quarantined via the verified region list with
|
||||
`source_prints_no_bar: true`. The count of bar-less formulas in this book is
|
||||
**unmeasured** — recorded as `recall_limit` in
|
||||
`ingestion/data/verified/formula_regions_2d.json` so the bar scan is never
|
||||
mistaken for complete formula coverage.
|
||||
**Generalizes:** yes — measured precision of the fraction-bar rule on this
|
||||
book is **16/23 = 69.6%**, and its recall is unknown. A geometric heuristic
|
||||
finds candidates; it never proves absence.
|
||||
|
||||
## Not yet investigated (flagged for future work, not silently ignored)
|
||||
|
||||
- **Footnote-style superscript reference markers** (seen as `a, b, c, d` in
|
||||
one table) — not yet checked for whether the footnote text stays
|
||||
correctly associated with its marker/row during extraction.
|
||||
- **Formula detection heuristic** (item 8) — no automatic detector exists
|
||||
yet to flag 2D-formula regions before they're trusted as chunk content.
|
||||
- **How many bar-less formulas exist** (item 25) — one confirmed, total
|
||||
unmeasured; no geometric signal can bound it.
|
||||
- **Merging the 51 transcribed outlined runs back into monograph text**
|
||||
(item 24) — transcribed and stored, but the corpus still contains
|
||||
`Độ n định`.
|
||||
- **2D grid table reconstruction** (item 7) — no implementation yet for
|
||||
recovering row/column-correct values from a nomogram-style table.
|
||||
- **Exact shortest monograph name+page** — a quick unmerged crude scan (no
|
||||
multi-line title merge) gave a different longest-monograph ranking than
|
||||
the already-documented authoritative one (item 12e: "AMOXICILIN VÀ KALI
|
||||
CLAVULANAT", 45,623 chars), meaning the crude scan's numbers are not
|
||||
reliable enough to name an exact shortest monograph — deferred to the real
|
||||
Phase 1.2 detector (with proper multi-line merge and back-index-validated
|
||||
boundaries), which will produce a trustworthy number as a side effect of
|
||||
its own validation run, rather than trusting today's quick, differently-
|
||||
scoped script.
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,208 @@
|
||||
# Verification strategy — how extraction is actually measured
|
||||
|
||||
**Short answer to "do you compare characters?": no.** Character comparison
|
||||
was tried and rejected twice, for reasons recorded below. What is used
|
||||
instead is a ladder of instruments, each answering a *different* question,
|
||||
each with a stated blind spot. No single number means "the parse is correct",
|
||||
and this document exists so nobody later mistakes one rung for another.
|
||||
|
||||
Status: written 2026-08-01, after the residual-ink work. Every figure quoted
|
||||
here was measured on the whole 1668-page document unless said otherwise.
|
||||
|
||||
---
|
||||
|
||||
## The rule that governs everything below
|
||||
|
||||
**An instrument must be checked before its output is believed.** In this
|
||||
project the measuring device has been wrong before the data was, repeatedly.
|
||||
Only after an instrument survives its own check does its number get quoted.
|
||||
|
||||
Three confirmed cases, all from 2026-08-01:
|
||||
|
||||
| what was nearly reported | why it was wrong |
|
||||
|---|---|
|
||||
| "extraction ratio 0.6656, 835 pages below 98%" | `get_texttrace()` counts glyphs painted *outside* the page rectangle — 4,717,407 of them, on pages that are visually blank |
|
||||
| "ratio 0.8023, 1642 of 1668 pages below 95%" (after clipping to the page) | Vietnamese diacritics are painted as two glyphs and extracted as one character, so the deficit is systematic and meaningless |
|
||||
| "page 209's ADR table is unaccounted-for ink" | the residual scan's horizontal banding merged the left and right columns, so the box's centre landed in the gutter and matched no table |
|
||||
|
||||
Earlier sessions add three more: a gate comparing post-merge spans against
|
||||
raw spans, one ordering parts by page-y in a two-column book, and one
|
||||
treating a legitimately resuming section as an ordering violation.
|
||||
|
||||
Corollary: **a non-zero gate is not automatically a data bug.** Check the
|
||||
gate, then the data.
|
||||
|
||||
---
|
||||
|
||||
## Why not character comparison
|
||||
|
||||
1. **Characters cannot be balanced across normalization.** The pipeline joins
|
||||
spans that share a visual line, substitutes PUA codepoints for real
|
||||
glyphs, and strips separators. A character in, character out ledger cannot
|
||||
close, so a mismatch tells you nothing.
|
||||
2. **Glyph counts cannot stand in for characters.** See the table above —
|
||||
both attempts produced confident, wrong numbers.
|
||||
3. **Comparing extracted text against another extractor's text measures
|
||||
agreement, not truth**, and on this document the tools share a blind spot
|
||||
(§3).
|
||||
|
||||
What replaced it: balance at the **span** level (a unit that survives the
|
||||
pipeline), and verify at the **pixel** level (a unit that owes nothing to any
|
||||
extractor).
|
||||
|
||||
---
|
||||
|
||||
## Layer 1 — Span routing ledger: did every span land somewhere?
|
||||
|
||||
`cli coverage`. Each of the 252,733 merged spans is assigned exactly one
|
||||
state and characters are aggregated from the states.
|
||||
|
||||
| state | spans | chars |
|
||||
|---|---|---|
|
||||
| normalized_text | 177,679 | 8,182,049 |
|
||||
| out_of_scope | 53,374 | 897,692 |
|
||||
| heading | 12,764 | 221,266 |
|
||||
| boilerplate_excluded | 4,976 | 47,609 |
|
||||
| quarantined | 3,937 | 48,989 |
|
||||
| structural_excluded | 3 | 53 |
|
||||
| **unassigned** | **0** | **0** |
|
||||
|
||||
**Proves:** nothing the extractor produced was dropped without a name.
|
||||
**Does not prove:** that routed content survived downstream. A section-
|
||||
overwrite bug was invisible to this ledger — spans were correctly marked
|
||||
`normalized_text`, then their section was overwritten later.
|
||||
**Does not prove:** that the extractor produced everything on the page. That
|
||||
is Layer 2's job, and it is the gap that mattered most.
|
||||
|
||||
---
|
||||
|
||||
## Layer 2 — Residual ink: what is on the page that no span accounts for?
|
||||
|
||||
`cli residual-ink`. Render the page, white out every pixel covered by an
|
||||
extracted span's bbox, measure the ink that survives, and give every
|
||||
surviving region a name. Needs no ground truth, no sampling, and no second
|
||||
tool. Measured cost: **0.06 s/page, all 1668 pages in under two minutes.**
|
||||
|
||||
| kind | regions |
|
||||
|---|---|
|
||||
| header_rule | 1,649 |
|
||||
| text_as_vector_outline | 1,061 |
|
||||
| table_frame | 959 |
|
||||
| antialias_speck | 220 |
|
||||
| fraction_bar_candidate | 23 |
|
||||
| rule_fragment | 10 |
|
||||
| header_band_fragment | 9 |
|
||||
| **unclassified** | **0** |
|
||||
|
||||
This is the only instrument here that does not ask a text layer a question,
|
||||
which is why it found what everything else missed: **51 runs of type that
|
||||
exist only as vector paths** (outlier-catalog item 24), invisible to
|
||||
PyMuPDF, pdfplumber and opendataloader-pdf alike.
|
||||
|
||||
**Proves:** every mark on all 1668 pages is accounted for by name.
|
||||
**Does not prove:** that the names are right. `unclassified = 0` means every
|
||||
region was *named*, not that every verdict was checked by eye. Of the seven
|
||||
kinds, only `text_as_vector_outline` and `fraction_bar_candidate` were
|
||||
confirmed exhaustively; the rest were confirmed on sampled examples.
|
||||
**Calibration matters:** at 1.0pt of mask padding the check ate the very
|
||||
fraction bars it exists to find (page 1042's bar shrank from 188.6pt to
|
||||
9.1pt). 0.5pt was chosen by measurement, and a regression test pins it.
|
||||
|
||||
---
|
||||
|
||||
## Layer 3 — Cross-tool agreement: useful, and routinely over-claimed
|
||||
|
||||
Inside the monograph range, `pdfplumber` and `opendataloader-pdf` agree
|
||||
*exactly* on where tables are: same 112 pages, same per-page count, zero
|
||||
pages found by only one. That looks like strong evidence and is not.
|
||||
|
||||
**On physical page 1042, both report zero tables.** There is a
|
||||
Cockcroft-Gault fraction on that page. Both tools need ruling lines; the bar
|
||||
is a drawn line but not a table, so neither sees it. The same holds on 202.
|
||||
|
||||
**Rule adopted:** agreement between two tools that share a failure mode
|
||||
measures *consistency*, never *recall*. Cross-tool agreement may be reported
|
||||
as a reproducibility check and never as coverage evidence.
|
||||
|
||||
Where it is genuinely useful: opendataloader's whole-book JSON carries 141
|
||||
tables / 826 rows / 2,468 cells with per-cell page, bbox, row, column and
|
||||
span — a second independent source of table structure, already on disk.
|
||||
|
||||
---
|
||||
|
||||
## Layer 4 — Visual census: the only instrument that yields content verdicts
|
||||
|
||||
Render the region, read the image, record the verdict. This is what turns a
|
||||
candidate into a fact, and it is the only layer that can say what the text
|
||||
*says*.
|
||||
|
||||
**Census when the population is small enough to enumerate.** This is stronger
|
||||
than any confidence interval, so prefer it whenever possible:
|
||||
|
||||
| population | size | status |
|
||||
|---|---|---|
|
||||
| fraction-bar candidates | 23 | **all 23 read.** 16 real, 7 not → precision **69.6%** |
|
||||
| vector-outlined runs | 51 | **all 51 read and transcribed** (1,116 characters) |
|
||||
| "not a table" verdicts | 20 | all 20 read (found 2 wrong) |
|
||||
| table blocks | 155 | not started |
|
||||
|
||||
**Sampling only when a census is impossible**, and then with the arithmetic
|
||||
stated. Rule of three: inspect *n* items, find **0** defects, and the 95%
|
||||
upper bound on the defect rate is ≈ 3/n. So "≤ 1% error" costs **n ≥ 300 with
|
||||
zero defects**; "≤ 5%" costs n ≥ 60. Any "99%" claim that cannot name its *n*
|
||||
is not a measurement.
|
||||
|
||||
**Risk-based, not random**, when sampling: 100% of table pages, formula
|
||||
pages, monograph boundaries, parser-warning pages and unusual-layout pages,
|
||||
plus a sample of normal pages.
|
||||
|
||||
---
|
||||
|
||||
## Layer 5 — Invariants the book itself supplies
|
||||
|
||||
The source is redundant, and each redundancy is a free check that needs no
|
||||
human ground truth. A violation is a proof of a defect.
|
||||
|
||||
- back-of-book index → monograph boundaries (in use: 92.9% recall / 99.1%
|
||||
precision, on a denominator that is **not yet cleaned**)
|
||||
- `"Bảng N"` captions → every caption must have a detected table (in use:
|
||||
32/33)
|
||||
- cross-references (`"xem Liều lượng và cách dùng"`) → must resolve to a
|
||||
section that exists in the same monograph (**not built**)
|
||||
- ATC codes → must match the WHO shape `[A-Z]\d\d[A-Z][A-Z]\d\d` (**not
|
||||
built**)
|
||||
- dose ranges (`"4 - 7,5 mg/kg"`) → must parse as two ordered numbers
|
||||
(**not built**)
|
||||
|
||||
---
|
||||
|
||||
## Layer 6 — Fail safe at the point of use
|
||||
|
||||
Detection is never complete, so the system must stay safe when it misses.
|
||||
|
||||
- every chunk carries `page` + `bbox`; every answer carries a citation
|
||||
- the UI shows the **rendered source crop** beside the answer, so a
|
||||
pharmacist verifies against the book in seconds
|
||||
- `quarantined` content and `formula_kind: 2d` never enter the model's
|
||||
context as prose — crop or refuse, never linearised text
|
||||
|
||||
This is what makes the two Cockcroft-Gault formulas safe *today*, before any
|
||||
reconstruction exists: left in prose they read as multiplication, which is a
|
||||
dosing error.
|
||||
|
||||
---
|
||||
|
||||
## What may and may not be said in a report
|
||||
|
||||
- Name the **denominator** every time. "99%" of characters, pages, tables,
|
||||
formulas and monographs are five different claims.
|
||||
- Distinguish **detected / named / verified**. `unclassified = 0` is "named".
|
||||
- A heuristic finds **candidates**; it never proves absence. The fraction-bar
|
||||
rule is 69.6% precise and its recall is unknown — and known to be below
|
||||
100%, because ADENOSIN (page 147) prints a fraction with no bar at all.
|
||||
- Never write "100%", "complete", "all", "no data lost" or "production-ready"
|
||||
unless the checks performed support the literal claim.
|
||||
- The honest current shape: *"the parser processed 1668/1668 pages;
|
||||
structural checks and 145 tests pass; nothing is lost without being
|
||||
counted. Content accuracy is NOT confirmed at 100% because there is no
|
||||
human-reviewed ground truth for the whole document to diff against."*
|
||||
@@ -0,0 +1,185 @@
|
||||
{
|
||||
"note": "2D (stacked-fraction) formula regions, every one confirmed by rendering the page and reading it. bbox is the fraction bar itself; numerator and denominator sit above and below it.",
|
||||
"verified_on": "2026-08-01",
|
||||
"method": "residual-ink fraction_bar_candidate, then visual inspection of all 23 candidates",
|
||||
"regions": [
|
||||
{
|
||||
"physical_page": 43,
|
||||
"bar_bbox": [
|
||||
133.44,
|
||||
308.16,
|
||||
222.72,
|
||||
308.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
430.08,
|
||||
603.36,
|
||||
465.6,
|
||||
603.36
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
362.88,
|
||||
704.64,
|
||||
386.4,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
400.32,
|
||||
704.64,
|
||||
422.88,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
456.96,
|
||||
704.64,
|
||||
480.0,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 92,
|
||||
"bar_bbox": [
|
||||
505.44,
|
||||
704.64,
|
||||
522.72,
|
||||
704.64
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 202,
|
||||
"bar_bbox": [
|
||||
371.52,
|
||||
151.68,
|
||||
489.6,
|
||||
151.68
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 325,
|
||||
"bar_bbox": [
|
||||
445.92,
|
||||
562.08,
|
||||
544.32,
|
||||
562.56
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 325,
|
||||
"bar_bbox": [
|
||||
439.68,
|
||||
623.52,
|
||||
560.16,
|
||||
623.52
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 349,
|
||||
"bar_bbox": [
|
||||
384.48,
|
||||
336.0,
|
||||
491.52,
|
||||
336.48
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1042,
|
||||
"bar_bbox": [
|
||||
97.92,
|
||||
492.0,
|
||||
286.56,
|
||||
492.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1043,
|
||||
"bar_bbox": [
|
||||
94.56,
|
||||
536.16,
|
||||
140.16,
|
||||
536.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1043,
|
||||
"bar_bbox": [
|
||||
145.92,
|
||||
536.16,
|
||||
244.8,
|
||||
536.16
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1132,
|
||||
"bar_bbox": [
|
||||
63.36,
|
||||
498.72,
|
||||
239.52,
|
||||
498.72
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1402,
|
||||
"bar_bbox": [
|
||||
120.96,
|
||||
711.84,
|
||||
201.6,
|
||||
711.84
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 1402,
|
||||
"bar_bbox": [
|
||||
180.48,
|
||||
770.4,
|
||||
205.92,
|
||||
770.4
|
||||
]
|
||||
},
|
||||
{
|
||||
"physical_page": 147,
|
||||
"bar_bbox": [
|
||||
307.9,
|
||||
672.0,
|
||||
428.5,
|
||||
672.0
|
||||
],
|
||||
"source_prints_no_bar": true,
|
||||
"note": "ADENOSIN infusion-rate formula. The source page prints three plain lines with no fraction bar at all, so no geometric detector can find it — confirmed by rendering the region and reading it. Left in prose it reads as a multiplication chain. Quarantined on the strength of the reading, and flagged for human confirmation of the intended division."
|
||||
}
|
||||
],
|
||||
"rejected": [
|
||||
{
|
||||
"physical_page": 4,
|
||||
"reason": "decorative underlines on the Ministry decision page"
|
||||
},
|
||||
{
|
||||
"physical_page": 63,
|
||||
"reason": "ruled box around a treatment-protocol paragraph"
|
||||
},
|
||||
{
|
||||
"physical_page": 845,
|
||||
"reason": "table header cell border"
|
||||
},
|
||||
{
|
||||
"physical_page": 878,
|
||||
"reason": "table header cell border"
|
||||
},
|
||||
{
|
||||
"physical_page": 1667,
|
||||
"reason": "rule above the colophon on the last page"
|
||||
}
|
||||
],
|
||||
"recall_limit": "The fraction-bar signal cannot find a fraction the source never typeset. ADENOSIN (physical page 147) is one confirmed case, found only because a prose-leak gate matched its text. The true number of bar-less formulas in the book is UNMEASURED."
|
||||
}
|
||||
@@ -0,0 +1,671 @@
|
||||
{
|
||||
"note": "Text that exists in the PDF only as vector outlines. No extractor returns it (PyMuPDF, pdfplumber and opendataloader-pdf all omit it). Every 'text' value below is a transcription read off the rendered page, not extracted data.",
|
||||
"transcribed_on": "2026-08-01",
|
||||
"method": "ingestion.extract.detect_outlined_text located the runs; each run was rendered at 210-300 dpi and read directly",
|
||||
"confidence": "Full-line runs are read with high confidence. Single-glyph runs are Vietnamese diacritic characters dropped out of an otherwise-extracted line; the glyph identity is legible but these should still be spot-checked by a human before the corpus is treated as complete.",
|
||||
"runs": [
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
470.32,
|
||||
37.64,
|
||||
521.85,
|
||||
44.57
|
||||
],
|
||||
"path_items": 337,
|
||||
"text": "Gatifloxacin",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.31,
|
||||
75.76,
|
||||
286.72,
|
||||
84.41
|
||||
],
|
||||
"path_items": 1638,
|
||||
"text": "Nghiên cứu trên động vật, gatifloxacin gây ngộ độc cho thai.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.77,
|
||||
87.91,
|
||||
287.05,
|
||||
96.56
|
||||
],
|
||||
"path_items": 1714,
|
||||
"text": "Gatifloxacin chỉ sử dụng cho phụ nữ có thai khi lợi ích vượt trội so",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.67,
|
||||
127.07,
|
||||
551.33,
|
||||
137.58
|
||||
],
|
||||
"path_items": 1765,
|
||||
"text": "Thuốc kháng acid (antacid): Gatifloxacin bị giảm hấp thu khi sử",
|
||||
"extracted_line_it_belongs_to": "Do chưa biết thuốc có phân bố vào sữa mẹ khi dùng trên người hay ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.51,
|
||||
139.65,
|
||||
287.22,
|
||||
150.17
|
||||
],
|
||||
"path_items": 1831,
|
||||
"text": "không, cần thận trọng khi sử dụng gatifloxacin cho phụ nữ đang",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
151.2,
|
||||
550.66,
|
||||
161.71
|
||||
],
|
||||
"path_items": 1787,
|
||||
"text": "cần dùng gatifloxacin ít nhất 4 giờ trước khi dùng các antacid này.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.44,
|
||||
177.2,
|
||||
455.19,
|
||||
185.84
|
||||
],
|
||||
"path_items": 1126,
|
||||
"text": "học có ý nghĩa lâm sàng với gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.66,
|
||||
223.59,
|
||||
397.93,
|
||||
234.1
|
||||
],
|
||||
"path_items": 792,
|
||||
"text": "giảm hấp thu gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": "Mắt: Chứng sưng viêm mi mắt, xuất huyết kết mạc, rát kết mạc, ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.66,
|
||||
247.72,
|
||||
551.32,
|
||||
258.23
|
||||
],
|
||||
"path_items": 1731,
|
||||
"text": "giữa warfarin và gatifloxacin, nhưng do một số quinolon có khả",
|
||||
"extracted_line_it_belongs_to": "khô mắt, phù, rát, viêm giác mạc, giảm thị lực, kích ứng kết mạc.",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
321.98,
|
||||
363.88,
|
||||
330.62
|
||||
],
|
||||
"path_items": 480,
|
||||
"text": "của gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": "Thần kinh: Căng thẳng, kích động, lo lắng, mất ngủ, hoa mắt, giấc ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.78,
|
||||
441.28,
|
||||
287.38,
|
||||
451.79
|
||||
],
|
||||
"path_items": 1778,
|
||||
"text": "Cần ngừng gatifloxacin trong các trường hợp: Bắt đầu có các biểu",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
452.82,
|
||||
461.52,
|
||||
463.34
|
||||
],
|
||||
"path_items": 1115,
|
||||
"text": "Gatifloxacin dùng với các thuốc làm thay đ",
|
||||
"extracted_line_it_belongs_to": "hiện ban da hoặc bất kỳ dấu hiệu nào của phản ứng quá mẫn, có ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
461.95,
|
||||
452.82,
|
||||
466.05,
|
||||
461.42
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "i nồng độ glucose máu ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
35.72,
|
||||
503.9,
|
||||
81.99,
|
||||
512.55
|
||||
],
|
||||
"path_items": 373,
|
||||
"text": "gatifloxacin.",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
313.73,
|
||||
506.23,
|
||||
317.83,
|
||||
514.82
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
542.15,
|
||||
506.23,
|
||||
546.25,
|
||||
514.82
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Độ n định: Dung dịch sau khi pha loãng trong dịch tương hợp n ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
450.53,
|
||||
520.35,
|
||||
455.28,
|
||||
526.89
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "định trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.71,
|
||||
532.42,
|
||||
304.45,
|
||||
538.95
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": " nhiệt độ 2 - 8 oC. Dung dịch pha loãng này (trừ pha trong natri ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
385.77,
|
||||
542.42,
|
||||
389.87,
|
||||
551.01
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
36.09,
|
||||
543.49,
|
||||
287.58,
|
||||
554.0
|
||||
],
|
||||
"path_items": 1809,
|
||||
"text": "Ghi chú: Đối với gatifloxacin dạng viên và dạng tiêm, nhà sản xuất",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
513.35,
|
||||
544.48,
|
||||
518.1,
|
||||
551.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "bicarbonat 5%) có thể n định tới 6 tháng nếu bảo quản ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
522.12,
|
||||
554.48,
|
||||
526.22,
|
||||
563.08
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "-25 đến -10 oC, sau khi đưa ra khỏi tủ lạnh sâu, tiếp tục n định ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
426.75,
|
||||
568.61,
|
||||
431.5,
|
||||
575.14
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
525.62,
|
||||
568.61,
|
||||
530.37,
|
||||
575.14
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "trong vòng 14 ngày nếu bảo quản nhiệt độ 20 - 26 oC hoặc nhiệt ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.47,
|
||||
619.81,
|
||||
551.17,
|
||||
630.32
|
||||
],
|
||||
"path_items": 1692,
|
||||
"text": "Vì có rất ít các thông tin về tương ky của gatifloxacin, nên không",
|
||||
"extracted_line_it_belongs_to": "Tiêm truyền tĩnh mạch dưới dạng dung dịch 2 mg/ml trong 60 phút.",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
105.83,
|
||||
628.89,
|
||||
109.93,
|
||||
637.14
|
||||
],
|
||||
"path_items": 39,
|
||||
"text": "ỗ",
|
||||
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.48,
|
||||
631.87,
|
||||
551.03,
|
||||
642.39
|
||||
],
|
||||
"path_items": 1739,
|
||||
"text": "thêm bất kỳ một thuốc nào khác vào dịch truyền gatifloxacin hoặc",
|
||||
"extracted_line_it_belongs_to": "Thuốc dùng tại ch : Chỉ dùng nhỏ vào mắt bị viêm; tránh để tiếp ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
391.91,
|
||||
685.48,
|
||||
396.01,
|
||||
693.72
|
||||
],
|
||||
"path_items": 39,
|
||||
"text": "ỗ",
|
||||
"extracted_line_it_belongs_to": "triệu chứng và điều trị h trợ, bao gồm: Gây nôn và rửa dạ dày để ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
223.28,
|
||||
713.6,
|
||||
227.39,
|
||||
722.19
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
167.4,
|
||||
715.66,
|
||||
172.14,
|
||||
722.19
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu i và người lớn:",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 714,
|
||||
"bbox": [
|
||||
299.72,
|
||||
786.64,
|
||||
551.3,
|
||||
797.16
|
||||
],
|
||||
"path_items": 1687,
|
||||
"text": "Gatifloxacin thuộc Danh mục nguyên liệu và thuốc thành phẩm",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
448.56,
|
||||
88.17,
|
||||
453.3,
|
||||
94.7
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "không màu, đóng kín tránh ánh sáng điều kiện lạnh 2 - 8 oC; ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
266.53,
|
||||
98.19,
|
||||
270.63,
|
||||
106.79
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
282.68,
|
||||
100.26,
|
||||
287.43,
|
||||
106.79
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "dưới da hoặc tiêm bắp. Đối với người lớn và trẻ em từ 3 tu i tr ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
129.96,
|
||||
122.5,
|
||||
209.16,
|
||||
133.01
|
||||
],
|
||||
"path_items": 580,
|
||||
"text": "nh tổn thương dây th",
|
||||
"extracted_line_it_belongs_to": "vào vùng cơ mông để trá",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
172.81,
|
||||
221.76,
|
||||
177.56,
|
||||
228.3
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "Liều thường dùng của GMDCUV người lớn và trẻ em để dự ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
91.42,
|
||||
280.45,
|
||||
95.52,
|
||||
289.05
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "tiêm các liều b sung với các khoảng cách là 4 tuần.",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
157.23,
|
||||
332.32,
|
||||
161.98,
|
||||
338.85
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "lại. Liều thông thường HTCUV người lớn và trẻ em để dự phòng ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
192.28,
|
||||
369.96,
|
||||
197.03,
|
||||
376.5
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "chậm trễ trong bắt đầu tiêm phòng hoặc người có thể trọng quá ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
399.06,
|
||||
424.38,
|
||||
403.8,
|
||||
430.91
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "huyết thanh của người trư ng thành khỏe mạnh đã được tạo miễn ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
187.41,
|
||||
513.01,
|
||||
192.16,
|
||||
519.54
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "GMDCUV hoặc HTCUV không ảnh hư ng tới đáp ứng miễn dịch ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
280.17,
|
||||
625.95,
|
||||
284.91,
|
||||
632.48
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "miễn dịch đối với một vài loại vắc xin virus sống (vắc xin virus s i ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
282.99,
|
||||
699.18,
|
||||
287.09,
|
||||
707.78
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "dịch hoặc huyết thanh ngựa thì nên dùng thêm một liều vắc xin b ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
419.76,
|
||||
726.48,
|
||||
424.51,
|
||||
733.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
442.01,
|
||||
726.48,
|
||||
446.75,
|
||||
733.01
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "phòng thí nghiệm và bị ảnh hư ng b i phương pháp xét nghiệm. ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
305.51,
|
||||
737.0,
|
||||
309.61,
|
||||
745.6
|
||||
],
|
||||
"path_items": 43,
|
||||
"text": "ổ",
|
||||
"extracted_line_it_belongs_to": "Do các chế phẩm có chứa globulin miễn dịch không có biểu hiện ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 736,
|
||||
"bbox": [
|
||||
62.37,
|
||||
751.45,
|
||||
67.12,
|
||||
757.98
|
||||
],
|
||||
"path_items": 45,
|
||||
"text": "ở",
|
||||
"extracted_line_it_belongs_to": "ảnh hư ng tới các đáp ứng miễn dịch của vắc xin uống virus bại ",
|
||||
"single_glyph": true
|
||||
},
|
||||
{
|
||||
"physical_page": 1373,
|
||||
"bbox": [
|
||||
43.81,
|
||||
98.58,
|
||||
295.66,
|
||||
109.1
|
||||
],
|
||||
"path_items": 1458,
|
||||
"text": "Nếu phối hợp với flutamid ở giai đoạn T2b - T4 (B2 - C), điều trị",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1444,
|
||||
"bbox": [
|
||||
35.72,
|
||||
136.22,
|
||||
287.35,
|
||||
146.74
|
||||
],
|
||||
"path_items": 1731,
|
||||
"text": "Trimovax (Sanofi Pasteur): Một liều vắc xin chứa virus sống giảm",
|
||||
"extracted_line_it_belongs_to": null,
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1445,
|
||||
"bbox": [
|
||||
308.28,
|
||||
114.39,
|
||||
559.87,
|
||||
123.03
|
||||
],
|
||||
"path_items": 1440,
|
||||
"text": "(Typhoid, inactivated, whole cell), J07AP03 (Typhoid, purified",
|
||||
"extracted_line_it_belongs_to": "thể xảy ra 5 ngày sau khi tiêm: Sốt (có thể dự phòng bằng các loại ",
|
||||
"single_glyph": false
|
||||
},
|
||||
{
|
||||
"physical_page": 1445,
|
||||
"bbox": [
|
||||
115.46,
|
||||
646.46,
|
||||
208.24,
|
||||
655.11
|
||||
],
|
||||
"path_items": 660,
|
||||
"text": "Haemophilus influenzae",
|
||||
"extracted_line_it_belongs_to": "khác như vắc xin ",
|
||||
"single_glyph": false
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
from .chunker import chunk_all, chunk_monograph, chunk_section, estimate_tokens
|
||||
from .io import read_monographs_jsonl, write_chunks_jsonl
|
||||
from .models import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
SCHEMA_VERSION,
|
||||
Chunk,
|
||||
ChunkAttachment,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"Chunk",
|
||||
"ChunkAttachment",
|
||||
"SCHEMA_VERSION",
|
||||
"CHUNK_KIND_PROSE",
|
||||
"CHUNK_KIND_BLOCK_DESCRIPTOR",
|
||||
"chunk_all",
|
||||
"chunk_monograph",
|
||||
"chunk_section",
|
||||
"estimate_tokens",
|
||||
"read_monographs_jsonl",
|
||||
"write_chunks_jsonl",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Section -> chunk logic (pure; no filesystem, no embedding client).
|
||||
|
||||
ADR 0004: chunk unit is `(drug_id, section_key)`. A section under the token
|
||||
ceiling becomes one chunk verbatim. Only the long-tail sections above it are
|
||||
sub-chunked, with a sentence-boundary-aware sliding window.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, Iterable, Iterator, List, Sequence
|
||||
|
||||
from ..segment.models import Monograph, SectionSpan, TableBlock
|
||||
from ..tables.classify import SHAPE_FORMULA_2D, SHAPE_SIMPLE
|
||||
from .models import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
Chunk,
|
||||
ChunkAttachment,
|
||||
)
|
||||
from .sentences import split_sentences
|
||||
|
||||
CEILING_TOKENS = 800
|
||||
TARGET_TOKENS = 650
|
||||
OVERLAP_TOKENS = 65
|
||||
|
||||
# Physical -> printed page. Empirically constant across every tested
|
||||
# milestone page (extract/page_map.py, ADR 0003); the descriptor quotes the
|
||||
# printed number because that is what a reader holding the book looks for.
|
||||
PRINTED_PAGE_OFFSET = 1
|
||||
|
||||
KIND_TABLE = "table"
|
||||
KIND_FORMULA = "formula"
|
||||
|
||||
# A header row is only safe to embed when it is genuinely a row of labels.
|
||||
# Measured on the corpus: 42 of 124 simple-table headers (34%) contain a
|
||||
# digit, and AMIODARON's (physical page 183) is
|
||||
# "Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)" — a dose,
|
||||
# inside what pdfplumber called a header, from an extraction never verified by
|
||||
# eye. A label carrying no digit cannot be mistaken for a dose; a long cell is
|
||||
# content rather than a label.
|
||||
_DIGIT = re.compile(r"\d")
|
||||
HEADER_CELL_MAX_CHARS = 40
|
||||
|
||||
|
||||
def _is_label_row(cells: Sequence[str]) -> bool:
|
||||
kept = [c for c in cells if c and c.strip()]
|
||||
if not kept:
|
||||
return False
|
||||
return all(
|
||||
not _DIGIT.search(cell) and len(cell.strip()) <= HEADER_CELL_MAX_CHARS
|
||||
for cell in kept
|
||||
)
|
||||
|
||||
|
||||
def estimate_tokens(text: str) -> int:
|
||||
"""ADR 0004's chars/4 estimate — an estimate, not a tokenizer count."""
|
||||
return len(text) // 4
|
||||
|
||||
|
||||
def _pack(sentences: List[str]) -> List[List[str]]:
|
||||
"""Greedily pack sentences up to TARGET_TOKENS, overlapping by OVERLAP_TOKENS.
|
||||
|
||||
A single sentence longer than the target becomes its own part rather than
|
||||
being cut mid-sentence — the caller flags it instead of splitting it.
|
||||
"""
|
||||
parts: List[List[str]] = []
|
||||
current: List[str] = []
|
||||
current_tokens = 0
|
||||
|
||||
for sentence in sentences:
|
||||
tokens = estimate_tokens(sentence)
|
||||
if current and current_tokens + tokens > TARGET_TOKENS:
|
||||
parts.append(current)
|
||||
overlap: List[str] = []
|
||||
acc = 0
|
||||
for prev in reversed(current):
|
||||
overlap.insert(0, prev)
|
||||
acc += estimate_tokens(prev)
|
||||
if acc >= OVERLAP_TOKENS:
|
||||
break
|
||||
current = list(overlap)
|
||||
current_tokens = sum(estimate_tokens(s) for s in current)
|
||||
current.append(sentence)
|
||||
current_tokens += tokens
|
||||
|
||||
if current:
|
||||
parts.append(current)
|
||||
return parts
|
||||
|
||||
|
||||
def _block_kind(block: TableBlock) -> str:
|
||||
return KIND_FORMULA if block.shape == SHAPE_FORMULA_2D else KIND_TABLE
|
||||
|
||||
|
||||
def _attachment(block: TableBlock, header_row: List[str]) -> ChunkAttachment:
|
||||
return ChunkAttachment(
|
||||
block_id=block.table_id,
|
||||
kind=_block_kind(block),
|
||||
shape=block.shape,
|
||||
physical_page=block.physical_page,
|
||||
bbox=list(block.bbox),
|
||||
quarantined=block.quarantined,
|
||||
# Only a simple table's first row can be a row of plain labels, and
|
||||
# only when it actually reads like one. A multi-level or merged header
|
||||
# is the shape whose extraction is least trustworthy, so it
|
||||
# contributes nothing rather than something wrong.
|
||||
header_row=(list(header_row)
|
||||
if block.shape == SHAPE_SIMPLE and _is_label_row(header_row)
|
||||
else []),
|
||||
)
|
||||
|
||||
|
||||
def _blocks_by_section(monograph: Monograph) -> Dict[str, List[TableBlock]]:
|
||||
grouped: Dict[str, List[TableBlock]] = {}
|
||||
for block in monograph.tables:
|
||||
if block.section_key:
|
||||
grouped.setdefault(block.section_key, []).append(block)
|
||||
return grouped
|
||||
|
||||
|
||||
def describe_block(monograph: Monograph, section: SectionSpan,
|
||||
attachment: ChunkAttachment) -> str:
|
||||
"""Retrieval text for a block, built only from metadata.
|
||||
|
||||
No cell value ever appears here. A header row is a row of labels;
|
||||
linearising it cannot invent a numeric relationship, which is exactly what
|
||||
linearising a body row does.
|
||||
"""
|
||||
noun = "công thức" if attachment.kind == KIND_FORMULA else "bảng"
|
||||
printed = attachment.physical_page + PRINTED_PAGE_OFFSET
|
||||
text = (f"{monograph.drug_name} — {section.display_name} — {noun}, "
|
||||
f"trang {printed}.")
|
||||
if attachment.header_row:
|
||||
columns = " | ".join(c.replace("\n", " ").strip()
|
||||
for c in attachment.header_row if c and c.strip())
|
||||
if columns:
|
||||
text += f" Cột: {columns}."
|
||||
text += (" Nội dung chỉ tra cứu được trên ảnh trang gốc, "
|
||||
"không trích dẫn được dưới dạng văn bản.")
|
||||
return text
|
||||
|
||||
|
||||
def chunk_section(monograph: Monograph, section: SectionSpan,
|
||||
blocks: Sequence[TableBlock] = (),
|
||||
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
|
||||
header_rows = header_rows or {}
|
||||
attachments = [_attachment(b, header_rows.get(b.table_id, [])) for b in blocks]
|
||||
quarantined = any(a.quarantined for a in attachments)
|
||||
|
||||
def build(body: str, part_index: int, part_count: int) -> Chunk:
|
||||
tokens = estimate_tokens(body)
|
||||
return Chunk(
|
||||
chunk_id=f"{monograph.drug_id}__{section.key}__{part_index}",
|
||||
drug_id=monograph.drug_id,
|
||||
drug_name=monograph.drug_name,
|
||||
section_key=section.key,
|
||||
section_display_name=section.display_name,
|
||||
text=body,
|
||||
heading_physical_page=section.heading.physical_page,
|
||||
source_page_range=list(monograph.source_page_range),
|
||||
atc_codes=list(monograph.atc_codes),
|
||||
part_index=part_index,
|
||||
part_count=part_count,
|
||||
est_tokens=tokens,
|
||||
oversized=tokens > CEILING_TOKENS,
|
||||
chunk_kind=CHUNK_KIND_PROSE,
|
||||
attachments=list(attachments),
|
||||
has_quarantined_content=quarantined,
|
||||
)
|
||||
|
||||
text = section.text.strip()
|
||||
prose: List[Chunk] = []
|
||||
if text:
|
||||
if estimate_tokens(text) <= CEILING_TOKENS:
|
||||
prose = [build(text, 0, 1)]
|
||||
else:
|
||||
parts = _pack(split_sentences(text))
|
||||
bodies = [b for b in ("".join(p).strip() for p in parts) if b]
|
||||
prose = [build(b, i, len(bodies)) for i, b in enumerate(bodies)]
|
||||
|
||||
descriptors = []
|
||||
for attachment in attachments:
|
||||
body = describe_block(monograph, section, attachment)
|
||||
descriptors.append(Chunk(
|
||||
chunk_id=f"{monograph.drug_id}__{section.key}__block__{attachment.block_id}",
|
||||
drug_id=monograph.drug_id,
|
||||
drug_name=monograph.drug_name,
|
||||
section_key=section.key,
|
||||
section_display_name=section.display_name,
|
||||
text=body,
|
||||
heading_physical_page=section.heading.physical_page,
|
||||
source_page_range=list(monograph.source_page_range),
|
||||
atc_codes=list(monograph.atc_codes),
|
||||
est_tokens=estimate_tokens(body),
|
||||
chunk_kind=CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
attachments=[attachment],
|
||||
has_quarantined_content=attachment.quarantined,
|
||||
))
|
||||
return prose + descriptors
|
||||
|
||||
|
||||
def chunk_monograph(monograph: Monograph,
|
||||
header_rows: Dict[str, List[str]] | None = None) -> List[Chunk]:
|
||||
grouped = _blocks_by_section(monograph)
|
||||
chunks: List[Chunk] = []
|
||||
for section in monograph.sections.values():
|
||||
chunks.extend(chunk_section(monograph, section,
|
||||
grouped.get(section.key, ()), header_rows))
|
||||
return chunks
|
||||
|
||||
|
||||
def chunk_all(monographs: Iterable[Monograph],
|
||||
header_rows: Dict[str, List[str]] | None = None) -> Iterator[Chunk]:
|
||||
for monograph in monographs:
|
||||
yield from chunk_monograph(monograph, header_rows)
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Filesystem boundary for the chunk stage — kept out of the pure logic."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from ..segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from .models import SCHEMA_VERSION, Chunk
|
||||
|
||||
|
||||
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
|
||||
with path.open(encoding="utf-8") as fh:
|
||||
for line in fh:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
raw = json.loads(line)
|
||||
sections = {}
|
||||
for key, s in raw.get("sections", {}).items():
|
||||
h = s["heading"]
|
||||
sections[key] = SectionSpan(
|
||||
key=s["key"],
|
||||
display_name=s["display_name"],
|
||||
heading=Heading(
|
||||
text=h["text"],
|
||||
physical_page=h["physical_page"],
|
||||
y0=h["y0"],
|
||||
is_monograph_title=h["is_monograph_title"],
|
||||
section_key=h.get("section_key"),
|
||||
),
|
||||
text=s["text"],
|
||||
)
|
||||
yield Monograph(
|
||||
drug_id=raw["drug_id"],
|
||||
drug_name=raw["drug_name"],
|
||||
source_page_range=raw["source_page_range"],
|
||||
sections=sections,
|
||||
atc_codes=raw.get("atc_codes", []),
|
||||
atc_stated_absent=raw.get("atc_stated_absent", False),
|
||||
tables=[
|
||||
TableBlock(
|
||||
table_id=t["table_id"],
|
||||
shape=t["shape"],
|
||||
physical_page=t["physical_page"],
|
||||
bbox=t["bbox"],
|
||||
section_key=t.get("section_key"),
|
||||
text=t.get("text", ""),
|
||||
quarantined=t.get("quarantined", False),
|
||||
table_part_id=t.get("table_part_id"),
|
||||
continuation_group=t.get("continuation_group"),
|
||||
source_span_ids=t.get("source_span_ids", []),
|
||||
)
|
||||
for t in raw.get("tables", [])
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def write_chunks_jsonl(chunks: Iterable[Chunk], path: Path) -> int:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
count = 0
|
||||
with path.open("w", encoding="utf-8") as fh:
|
||||
for chunk in chunks:
|
||||
# ADR 0005 flagged the absence of a version and ADR 0006 made it
|
||||
# necessary: the record now has two chunk kinds and an attachment
|
||||
# list, so a consumer must be able to tell which shape it has.
|
||||
record = {"schema_version": SCHEMA_VERSION, **asdict(chunk)}
|
||||
fh.write(json.dumps(record, ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Chunk record — the unit handed to embedding/indexing.
|
||||
|
||||
Provenance fields follow ADR 0004 and CLAUDE.md's provenance rule: a chunk
|
||||
must carry enough to trace it back to a monograph, a section, and the page
|
||||
its section heading was found on.
|
||||
|
||||
ADR 0006 adds attachments. `segment/` lifts tables and 2D formulas out of
|
||||
section prose because linearising them is actively wrong — AMPICILIN VÀ
|
||||
SULBACTAM's Cockcroft-Gault fraction read as `Clcr (ml/phút) = 72 x
|
||||
creatinin huyết thanh`, a division presented as a multiplication in a
|
||||
renal-dosing section. Without a reference back, a chunk of that section is
|
||||
grammatical, complete-looking prose with the dosing table silently absent.
|
||||
Measured: 127 of 167 lifted blocks (76%) came out of `liều lượng và cách
|
||||
dùng`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List
|
||||
|
||||
SCHEMA_VERSION = 2
|
||||
|
||||
CHUNK_KIND_PROSE = "prose"
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR = "block_descriptor"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ChunkAttachment:
|
||||
"""A table or formula that was lifted out of this chunk's section.
|
||||
|
||||
`bbox` + `physical_page` are what let the answer layer render the source
|
||||
crop, which for a quarantined block is the only faithful answer available.
|
||||
"""
|
||||
|
||||
block_id: str
|
||||
kind: str # "table" | "formula"
|
||||
shape: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
quarantined: bool = True
|
||||
# First row of a `simple_table`, used to make the block findable. Comes
|
||||
# from pdfplumber and has NOT been verified by eye — the 180 real tables'
|
||||
# shapes are rule-derived. Retrieval bait, never an answer.
|
||||
header_row: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Chunk:
|
||||
chunk_id: str
|
||||
drug_id: str
|
||||
drug_name: str
|
||||
section_key: str
|
||||
section_display_name: str
|
||||
text: str
|
||||
heading_physical_page: int
|
||||
source_page_range: List[int]
|
||||
atc_codes: List[str] = field(default_factory=list)
|
||||
part_index: int = 0
|
||||
part_count: int = 1
|
||||
est_tokens: int = 0
|
||||
oversized: bool = False
|
||||
chunk_kind: str = CHUNK_KIND_PROSE
|
||||
attachments: List[ChunkAttachment] = field(default_factory=list)
|
||||
# Derivable from `attachments`, stored anyway: the defect this schema
|
||||
# exists to prevent is a consumer not knowing what it was not told.
|
||||
has_quarantined_content: bool = False
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Vietnamese sentence-boundary splitting for medical formulary text.
|
||||
|
||||
ADR 0004 requires splitting at sentence boundaries rather than a blind
|
||||
character window: `segment/assembler.py` joins body lines at PDF visual
|
||||
line-wrap points, so a character window can land mid-sentence — and outlier
|
||||
item 17 measured adult/child dosing sentences ("Người lớn"/"Trẻ em") on
|
||||
1,121 of ~1,400 monograph pages, where a mid-sentence cut is a
|
||||
patient-safety defect rather than a cosmetic one.
|
||||
|
||||
Boundary rule: `.`, `;`, `:`, `?` or `!` followed by whitespace and an
|
||||
opening character (uppercase letter or digit), minus the exclusions below.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List
|
||||
|
||||
_TERMINATORS = ".;:?!"
|
||||
|
||||
# Tokens that end in '.' but do not end a sentence.
|
||||
_ABBREVIATIONS = frozenset({
|
||||
"v.v", "vv", "tr", "tp", "ts", "bs", "gs", "pgs", "ths", "dr", "st",
|
||||
"no", "nxb", "cs", "kg", "mg", "ml", "mcg", "gr", "hb", "tm", "tb",
|
||||
})
|
||||
|
||||
_OPENS_SENTENCE = re.compile(r"[A-ZÀ-Ỹ0-9(\-–]")
|
||||
_TRAILING_TOKEN = re.compile(r"([\wÀ-ỹ.]+)\.$")
|
||||
|
||||
|
||||
def _is_abbreviation(left: str) -> bool:
|
||||
m = _TRAILING_TOKEN.search(left.rstrip())
|
||||
if not m:
|
||||
return False
|
||||
token = m.group(1).rstrip(".").lower()
|
||||
if token in _ABBREVIATIONS:
|
||||
return True
|
||||
# single letter -> an initial ("P." in a name), not a sentence end
|
||||
return len(token) == 1 and token.isalpha()
|
||||
|
||||
|
||||
def _is_decimal_or_numbering(text: str, i: int) -> bool:
|
||||
"""A period/comma sitting between digits, or a list numbering like '1. '."""
|
||||
if text[i] != ".":
|
||||
return False
|
||||
prev_ch = text[i - 1] if i > 0 else ""
|
||||
next_ch = text[i + 1] if i + 1 < len(text) else ""
|
||||
if prev_ch.isdigit() and next_ch.isdigit():
|
||||
return True
|
||||
# "1." / "12." starting a numbered list item: digits preceded by start/newline
|
||||
j = i - 1
|
||||
while j >= 0 and text[j].isdigit():
|
||||
j -= 1
|
||||
if j < i - 1 and (j < 0 or text[j] in "\n \t("):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def split_sentences(text: str) -> List[str]:
|
||||
"""Split into sentence-ish units, preserving all characters.
|
||||
|
||||
Concatenating the result (without added separators) reproduces the input
|
||||
exactly — no character is dropped, which the coverage ledger depends on.
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
|
||||
out: List[str] = []
|
||||
start = 0
|
||||
i = 0
|
||||
n = len(text)
|
||||
while i < n:
|
||||
ch = text[i]
|
||||
if ch not in _TERMINATORS:
|
||||
i += 1
|
||||
continue
|
||||
if _is_decimal_or_numbering(text, i):
|
||||
i += 1
|
||||
continue
|
||||
|
||||
j = i + 1
|
||||
if j < n and text[j] in "\")]”’":
|
||||
j += 1
|
||||
ws_start = j
|
||||
while j < n and text[j].isspace():
|
||||
j += 1
|
||||
if j == ws_start or j >= n:
|
||||
i += 1
|
||||
continue
|
||||
if not _OPENS_SENTENCE.match(text[j]):
|
||||
i += 1
|
||||
continue
|
||||
if ch == "." and _is_abbreviation(text[start:i + 1]):
|
||||
i += 1
|
||||
continue
|
||||
|
||||
out.append(text[start:j])
|
||||
start = j
|
||||
i = j
|
||||
|
||||
if start < n:
|
||||
out.append(text[start:])
|
||||
return out
|
||||
@@ -0,0 +1,516 @@
|
||||
"""CLI entry point: `python -m ingestion.cli <subcommand>`.
|
||||
|
||||
`run` is the real Phase 1 pipeline (extract -> segment -> write). `validate`,
|
||||
`visual-diff`, and `scaffold-golden` are Phase 1.4-1.7 work — declared here
|
||||
now (per the approved plan's CLI contract) but not yet implemented; they
|
||||
raise `NotImplementedError` explicitly rather than silently no-op-ing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
from .extract import (
|
||||
extract_spans,
|
||||
load_transcribed_runs,
|
||||
merge_outlined_runs,
|
||||
index_formula_regions_by_page,
|
||||
load_formula_regions,
|
||||
scan_glyph_order,
|
||||
scan_reading_order,
|
||||
)
|
||||
from .chunk import (
|
||||
CHUNK_KIND_PROSE,
|
||||
chunk_all,
|
||||
read_monographs_jsonl,
|
||||
write_chunks_jsonl,
|
||||
)
|
||||
from .segment import DuplicateDrugIdError, assemble, write_monographs_jsonl
|
||||
from .tables import (
|
||||
detect_table_regions,
|
||||
index_by_page,
|
||||
read_regions_json,
|
||||
write_regions_json,
|
||||
)
|
||||
from .validation import (
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
corpus_size,
|
||||
evaluate,
|
||||
evaluate_chunks,
|
||||
read_chunks,
|
||||
read_monographs,
|
||||
UNCLASSIFIED,
|
||||
compute_recall_precision,
|
||||
parse_back_index,
|
||||
)
|
||||
from .validation import scan_document as scan_residual_ink
|
||||
|
||||
|
||||
def _extracted_and_repaired_spans(doc, verbose: bool = False):
|
||||
"""The span stream, with vector-outlined text put back into it.
|
||||
|
||||
Outlier-catalog item 24: 51 runs of type exist only as vector paths, so
|
||||
extraction alone leaves holes mid-sentence ("Độ ổn định" -> "Độ n định").
|
||||
Every command that builds monographs must repair the stream the same way,
|
||||
or the ledger and the output describe different pipelines.
|
||||
"""
|
||||
spans = list(extract_spans(doc))
|
||||
if verbose:
|
||||
print(f"extracted {len(spans)} spans")
|
||||
runs = load_transcribed_runs()
|
||||
if not runs:
|
||||
return spans
|
||||
spans = merge_outlined_runs(spans, runs, doc=doc)
|
||||
if verbose:
|
||||
chars = sum(len(r.text) for r in runs)
|
||||
print(f"merged {len(runs)} transcribed vector-outlined runs "
|
||||
f"({chars} characters) back into the stream")
|
||||
return spans
|
||||
|
||||
|
||||
def _region_index(tables_arg, verbose: bool = False):
|
||||
"""Every region whose spans must be lifted out of prose, keyed by page.
|
||||
|
||||
`run` and `coverage` must build this the same way — when `coverage` loaded
|
||||
only tables while `run` also loaded formulas, the ledger described a
|
||||
pipeline that was not the one producing the output.
|
||||
"""
|
||||
index = {}
|
||||
regions_path = Path(tables_arg) if tables_arg else None
|
||||
if regions_path and regions_path.exists():
|
||||
index = index_by_page(read_regions_json(regions_path))
|
||||
if verbose:
|
||||
real = sum(len(v) for v in index.values())
|
||||
print(f"loaded {real} table regions on {len(index)} pages "
|
||||
f"from {regions_path}")
|
||||
elif regions_path and verbose:
|
||||
print(f"note: no table region map at {regions_path} — table text will "
|
||||
f"stay in section prose (run 'detect-tables' to produce one)")
|
||||
|
||||
formulas = load_formula_regions()
|
||||
for page, page_formulas in index_formula_regions_by_page(formulas).items():
|
||||
index.setdefault(page, []).extend(page_formulas)
|
||||
if formulas and verbose:
|
||||
print(f"loaded {len(formulas)} verified 2D formula regions on "
|
||||
f"{len({f.physical_page for f in formulas})} pages")
|
||||
return index or None
|
||||
|
||||
|
||||
def _cmd_run(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
print(f"opened {pdf_path} ({doc.page_count} pages)")
|
||||
|
||||
glyph_issues = scan_glyph_order(doc)
|
||||
reading_issues = scan_reading_order(doc)
|
||||
total_defects = len(glyph_issues) + len(reading_issues)
|
||||
if total_defects:
|
||||
print(
|
||||
f"glyph/reading-order sanity gate: {len(glyph_issues)} within-span + "
|
||||
f"{len(reading_issues)} cross-fragment issue(s) found "
|
||||
f"(see docs/pdf-parsing-outlier-catalog.md item 9 for known cases; "
|
||||
f"formula-region issues are expected there, not auto-corrected)."
|
||||
)
|
||||
|
||||
spans = _extracted_and_repaired_spans(doc, verbose=True)
|
||||
|
||||
table_index = _region_index(args.tables, verbose=True)
|
||||
|
||||
try:
|
||||
monographs = list(assemble(spans, table_index=table_index))
|
||||
except DuplicateDrugIdError as e:
|
||||
print(f"error: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
table_blocks = sum(len(m.tables) for m in monographs)
|
||||
quarantined = sum(1 for m in monographs for t in m.tables if t.quarantined)
|
||||
if table_index:
|
||||
print(f"lifted {table_blocks} table blocks out of section prose "
|
||||
f"({quarantined} quarantined)")
|
||||
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
count = write_monographs_jsonl(monographs, out_path)
|
||||
print(f"wrote {count} monographs to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_validate(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
spans = list(extract_spans(doc))
|
||||
try:
|
||||
monographs = list(assemble(spans))
|
||||
except DuplicateDrugIdError as e:
|
||||
print(f"error: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
ground_truth = parse_back_index(doc)
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
|
||||
print(f"detected monographs: {result.total_detected}")
|
||||
print(f"ground-truth entries: {result.total_ground_truth}")
|
||||
print(f"recall: {result.recall:.1%} ({result.matched_count}/{result.total_ground_truth})")
|
||||
print(f"precision: {result.precision:.1%}")
|
||||
if result.unmatched_ground_truth:
|
||||
print(f"\nunmatched ground-truth entries (first 20 of {len(result.unmatched_ground_truth)}):")
|
||||
for entry in result.unmatched_ground_truth[:20]:
|
||||
print(f" {entry.name}, {entry.printed_page}")
|
||||
if result.unmatched_detected:
|
||||
print(f"\nunmatched detected monographs (first 20 of {len(result.unmatched_detected)}):")
|
||||
for m in result.unmatched_detected[:20]:
|
||||
print(f" {m.drug_name} (physical page {m.source_page_range[0]})")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_detect_tables(args: argparse.Namespace) -> int:
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
regions = list(detect_table_regions(pdf_path))
|
||||
out_path = Path(args.out)
|
||||
count = write_regions_json(regions, out_path)
|
||||
|
||||
shapes = Counter(r.shape for r in regions)
|
||||
real = sum(1 for r in regions if r.is_real_table)
|
||||
print(f"detected {count} candidate regions on "
|
||||
f"{len({r.physical_page for r in regions})} pages")
|
||||
for shape, n in shapes.most_common():
|
||||
print(f" {shape:32} {n:5}")
|
||||
print(f"real tables: {real} not tables: {count - real}")
|
||||
print(f"wrote {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_coverage(args: argparse.Namespace) -> int:
|
||||
"""Span-level coverage ledger: where did every span end up?
|
||||
|
||||
Characters cannot be balanced directly — normalization joins, substitutes
|
||||
and drops them — so each span is assigned a state and characters are
|
||||
aggregated from those states.
|
||||
"""
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
spans = _extracted_and_repaired_spans(doc)
|
||||
|
||||
table_index = _region_index(args.tables)
|
||||
|
||||
ledger: list = []
|
||||
list(assemble(spans, table_index=table_index, ledger=ledger))
|
||||
|
||||
header, rows = ledger[0], ledger[1:]
|
||||
by_state = Counter(r["state"] for r in rows)
|
||||
chars = Counter()
|
||||
for r in rows:
|
||||
chars[r["state"]] += r["chars"]
|
||||
total_spans = len(rows)
|
||||
total_chars = sum(chars.values())
|
||||
|
||||
print(f"SCOPE: {pdf_path} — all {doc.page_count} pages")
|
||||
print(f"raw chars before span merge: {header['raw_chars_before_merge']:,}")
|
||||
print(f"spans after merge: {total_spans:,} chars: {total_chars:,}")
|
||||
print()
|
||||
print(f"{'state':<24}{'spans':>10}{'% spans':>10}{'chars':>14}{'% chars':>10}")
|
||||
print("-" * 68)
|
||||
for state, n in by_state.most_common():
|
||||
print(f"{state:<24}{n:>10,}{n/total_spans*100:>9.1f}%"
|
||||
f"{chars[state]:>14,}{chars[state]/total_chars*100:>9.1f}%")
|
||||
|
||||
unassigned = [r for r in rows if r["state"] == "unassigned"]
|
||||
print()
|
||||
print(f"UNASSIGNED: {len(unassigned):,} spans, "
|
||||
f"{sum(r['chars'] for r in unassigned):,} chars")
|
||||
if unassigned:
|
||||
pages = Counter(r["physical_page"] for r in unassigned)
|
||||
print(f" on {len(pages)} pages; worst: {pages.most_common(10)}")
|
||||
print(" first 10 examples:")
|
||||
for r in unassigned[:10]:
|
||||
print(f" p{r['physical_page']} {r['bbox']} {r['text']!r}")
|
||||
|
||||
if args.out:
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(json.dumps(ledger, ensure_ascii=False), encoding="utf-8")
|
||||
print(f"wrote full ledger to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_residual_ink(args: argparse.Namespace) -> int:
|
||||
"""Ask the page, not a detector: what ink did the text layer never emit?
|
||||
|
||||
Gate: `unclassified` must reach 0 — every surviving region has to be
|
||||
named, not silently tolerated.
|
||||
"""
|
||||
pdf_path = Path(args.pdf)
|
||||
if not pdf_path.exists():
|
||||
print(f"error: PDF not found: {pdf_path}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
tables_by_page = None
|
||||
regions_path = Path(args.tables) if args.tables else None
|
||||
if regions_path and regions_path.exists():
|
||||
tables_by_page = index_by_page(read_regions_json(regions_path))
|
||||
print(f"loaded table regions on {len(tables_by_page)} pages from {regions_path}")
|
||||
|
||||
pages = range(doc.page_count) if args.pages is None else _parse_pages(args.pages)
|
||||
pages = list(pages)
|
||||
findings = list(scan_residual_ink(doc, tables_by_page, pages))
|
||||
|
||||
kinds = Counter(kind for _, kind in findings)
|
||||
print(f"SCOPE: {pdf_path} — {len(pages)} of {doc.page_count} pages")
|
||||
print(f"residual regions: {len(findings)}")
|
||||
for kind, n in kinds.most_common():
|
||||
print(f" {kind:<26}{n:>7}")
|
||||
print(f"\nGATE unclassified = {kinds[UNCLASSIFIED]} (target 0)")
|
||||
|
||||
flagged = [(r, k) for r, k in findings
|
||||
if k in (FRACTION_BAR_CANDIDATE, UNCLASSIFIED)]
|
||||
print(f"needs eyes on it: {len(flagged)} region(s) on "
|
||||
f"{len({r.physical_page for r, _ in flagged})} pages")
|
||||
for region, kind in flagged[:20]:
|
||||
print(f" p{region.physical_page:<5} {kind:<24} "
|
||||
f"w={region.width_pt:6.1f} h={region.height_pt:5.1f} "
|
||||
f"bbox={region.bbox}")
|
||||
|
||||
if args.out:
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
payload = [
|
||||
{"physical_page": r.physical_page, "bbox": list(r.bbox),
|
||||
"ink_px": r.ink_px, "width_pt": round(r.width_pt, 2),
|
||||
"height_pt": round(r.height_pt, 2), "kind": k}
|
||||
for r, k in findings
|
||||
]
|
||||
out_path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
||||
print(f"wrote {len(payload)} regions to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _parse_pages(spec: str):
|
||||
"""Parse '202', '200-210' or '202,1042' into physical page numbers."""
|
||||
pages = []
|
||||
for part in spec.split(","):
|
||||
if "-" in part:
|
||||
start, end = part.split("-", 1)
|
||||
pages.extend(range(int(start), int(end) + 1))
|
||||
else:
|
||||
pages.append(int(part))
|
||||
return pages
|
||||
|
||||
|
||||
def _cmd_chunk(args: argparse.Namespace) -> int:
|
||||
"""Build retrieval chunks from the segmented monographs."""
|
||||
monographs_path = Path(args.monographs)
|
||||
if not monographs_path.exists():
|
||||
print(f"error: no monographs at {monographs_path} — run 'run' first",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
|
||||
header_rows = {}
|
||||
regions_path = Path(args.tables) if args.tables else None
|
||||
if regions_path and regions_path.exists():
|
||||
header_rows = {r.table_id: r.first_row
|
||||
for r in read_regions_json(regions_path)}
|
||||
|
||||
monographs = list(read_monographs_jsonl(monographs_path))
|
||||
chunks = list(chunk_all(monographs, header_rows))
|
||||
|
||||
kinds = Counter(c.chunk_kind for c in chunks)
|
||||
with_attachments = sum(1 for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_PROSE and c.attachments)
|
||||
oversized = sum(1 for c in chunks if c.oversized)
|
||||
tokens = sum(c.est_tokens for c in chunks)
|
||||
|
||||
print(f"SCOPE: {monographs_path} — {len(monographs)} monographs")
|
||||
print(f"chunks: {len(chunks)}")
|
||||
for kind, n in kinds.most_common():
|
||||
print(f" {kind:<22}{n:>7}")
|
||||
print(f"prose chunks carrying a lifted block: {with_attachments}")
|
||||
print(f"oversized (over the {800}-token ceiling): {oversized}")
|
||||
print(f"estimated tokens (chars/4, an estimate): {tokens:,}")
|
||||
|
||||
out_path = Path(args.out)
|
||||
written = write_chunks_jsonl(chunks, out_path)
|
||||
print(f"wrote {written} chunks to {out_path}")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_chunk_ready(args: argparse.Namespace) -> int:
|
||||
"""Every gate that must hold before the corpus may be chunked.
|
||||
|
||||
Chunking turns text into embeddings, where a defect stops being
|
||||
inspectable — so each invariant is printed with its own count and its own
|
||||
target rather than folded into one verdict.
|
||||
"""
|
||||
monographs_path = Path(args.monographs)
|
||||
if not monographs_path.exists():
|
||||
print(f"error: no monographs at {monographs_path} — run 'run' first",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
|
||||
monographs = read_monographs(monographs_path)
|
||||
runs = load_transcribed_runs()
|
||||
gates = evaluate(monographs, [
|
||||
{"physical_page": r.physical_page, "text": r.text} for r in runs
|
||||
])
|
||||
|
||||
size = corpus_size(monographs)
|
||||
print(f"SCOPE: {monographs_path} — {size['monographs']} monographs, "
|
||||
f"{size['sections']} sections, {size['section_chars']:,} characters")
|
||||
print(f" {size['quarantined_blocks']} quarantined table/formula blocks "
|
||||
f"(excluded from prose, citable only with their source crop)")
|
||||
print()
|
||||
print(f"{'gate':<34}{'count':>8}{'target':>8} result")
|
||||
print("-" * 62)
|
||||
for gate in gates:
|
||||
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
|
||||
f"{'PASS' if gate.passed else 'FAIL'}"
|
||||
+ (f" {gate.detail}" if gate.detail else ""))
|
||||
|
||||
chunks_path = Path(args.chunks)
|
||||
if chunks_path.exists():
|
||||
chunk_gates = evaluate_chunks(monographs, read_chunks(chunks_path))
|
||||
print()
|
||||
print(f"ADR 0006 — chunk references ({chunks_path}):")
|
||||
for gate in chunk_gates:
|
||||
print(f"{gate.name:<34}{gate.count:>8}{gate.target:>8} "
|
||||
f"{'PASS' if gate.passed else 'FAIL'}"
|
||||
+ (f" {gate.detail}" if gate.detail else ""))
|
||||
gates = gates + chunk_gates
|
||||
else:
|
||||
print()
|
||||
print(f"note: no chunks at {chunks_path} — ADR 0006 gates not run "
|
||||
f"(run 'chunk' to produce them)")
|
||||
|
||||
failed = [g for g in gates if not g.passed]
|
||||
print()
|
||||
if failed:
|
||||
print(f"NOT READY TO CHUNK — {len(failed)} gate(s) failing: "
|
||||
+ ", ".join(g.name for g in failed))
|
||||
return 1
|
||||
print("READY TO CHUNK — every gate above met its target.")
|
||||
print("Not proven by these gates: content accuracy against the source "
|
||||
"(no whole-document human-reviewed ground truth exists), table "
|
||||
"row/column reconstruction, and recall for borderless tables and "
|
||||
"bar-less formulas.")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_not_implemented(name: str):
|
||||
def _cmd(_args: argparse.Namespace) -> int:
|
||||
raise NotImplementedError(
|
||||
f"'{name}' is planned (see the approved segmentation/eval plan) but not yet built."
|
||||
)
|
||||
return _cmd
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(prog="python -m ingestion.cli")
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
|
||||
p_run = sub.add_parser("run", help="Extract + segment the PDF into monographs.jsonl")
|
||||
p_run.add_argument("--pdf", required=True, help="Path to the source PDF")
|
||||
p_run.add_argument(
|
||||
"--out", default="data/processed/monographs.jsonl",
|
||||
help="Output JSONL path (default: data/processed/monographs.jsonl)",
|
||||
)
|
||||
p_run.add_argument(
|
||||
"--tables", default="data/processed/table_regions.json",
|
||||
help="Table region map from 'detect-tables'. When present, table text "
|
||||
"is lifted out of section prose (default: "
|
||||
"data/processed/table_regions.json)",
|
||||
)
|
||||
p_run.set_defaults(func=_cmd_run)
|
||||
|
||||
p_validate = sub.add_parser("validate", help="Whole-book recall/precision vs. back-of-book index")
|
||||
p_validate.add_argument("--pdf", required=True)
|
||||
p_validate.set_defaults(func=_cmd_validate)
|
||||
|
||||
p_tables = sub.add_parser(
|
||||
"detect-tables",
|
||||
help="Locate + classify table regions (slow; result is cached and reused)",
|
||||
)
|
||||
p_tables.add_argument("--pdf", required=True)
|
||||
p_tables.add_argument(
|
||||
"--out", default="data/processed/table_regions.json",
|
||||
help="Output region-map path (default: data/processed/table_regions.json)",
|
||||
)
|
||||
p_tables.set_defaults(func=_cmd_detect_tables)
|
||||
|
||||
p_cov = sub.add_parser(
|
||||
"coverage", help="Span-level coverage ledger — where every span ended up")
|
||||
p_cov.add_argument("--pdf", required=True)
|
||||
p_cov.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_cov.add_argument("--out", default="data/processed/coverage_ledger.json")
|
||||
p_cov.set_defaults(func=_cmd_coverage)
|
||||
|
||||
p_residual = sub.add_parser(
|
||||
"residual-ink",
|
||||
help="Ink on the page that no extracted span accounts for "
|
||||
"(no ground truth needed; gate: unclassified = 0)",
|
||||
)
|
||||
p_residual.add_argument("--pdf", required=True)
|
||||
p_residual.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_residual.add_argument(
|
||||
"--pages", default=None,
|
||||
help="Limit to pages, e.g. '202' or '200-210' or '202,1042' "
|
||||
"(default: every page)",
|
||||
)
|
||||
p_residual.add_argument("--out", default="data/processed/residual_ink.json")
|
||||
p_residual.set_defaults(func=_cmd_residual_ink)
|
||||
|
||||
p_ready = sub.add_parser(
|
||||
"chunk-ready",
|
||||
help="Named gates that must all hold before chunking (garbage-in guard)")
|
||||
p_ready.add_argument(
|
||||
"--monographs", default="data/processed/monographs.jsonl")
|
||||
p_ready.add_argument("--chunks", default="data/processed/chunks.jsonl")
|
||||
p_ready.set_defaults(func=_cmd_chunk_ready)
|
||||
|
||||
p_chunk = sub.add_parser(
|
||||
"chunk", help="Build retrieval chunks (ADR 0004/0005/0006)")
|
||||
p_chunk.add_argument("--monographs", default="data/processed/monographs.jsonl")
|
||||
p_chunk.add_argument("--tables", default="data/processed/table_regions.json")
|
||||
p_chunk.add_argument("--out", default="data/processed/chunks.jsonl")
|
||||
p_chunk.set_defaults(func=_cmd_chunk)
|
||||
|
||||
p_visual = sub.add_parser("visual-diff", help="Render a page with detected boundaries overlaid")
|
||||
p_visual.set_defaults(func=_cmd_not_implemented("visual-diff"))
|
||||
|
||||
p_scaffold = sub.add_parser("scaffold-golden", help="Draft golden-set entries for human review")
|
||||
p_scaffold.set_defaults(func=_cmd_not_implemented("scaffold-golden"))
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv=None) -> int:
|
||||
if hasattr(sys.stdout, "reconfigure"):
|
||||
sys.stdout.reconfigure(encoding="utf-8")
|
||||
sys.stderr.reconfigure(encoding="utf-8")
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(argv)
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,36 @@
|
||||
from .glyph_order import (
|
||||
GlyphOrderIssue,
|
||||
ReadingOrderIssue,
|
||||
find_reading_order_issues,
|
||||
is_reversed_order,
|
||||
scan_glyph_order,
|
||||
scan_reading_order,
|
||||
)
|
||||
from .formulas import index_formula_regions_by_page, load_formula_regions
|
||||
from .models import Span
|
||||
from .outlined_text import (
|
||||
OutlinedTextRun,
|
||||
detect_outlined_text,
|
||||
load_transcribed_runs,
|
||||
)
|
||||
from .repair import merge_outlined_runs
|
||||
from .page_map import build_page_map
|
||||
from .spans import extract_spans
|
||||
|
||||
__all__ = [
|
||||
"Span",
|
||||
"OutlinedTextRun",
|
||||
"load_formula_regions",
|
||||
"index_formula_regions_by_page",
|
||||
"detect_outlined_text",
|
||||
"load_transcribed_runs",
|
||||
"merge_outlined_runs",
|
||||
"build_page_map",
|
||||
"extract_spans",
|
||||
"GlyphOrderIssue",
|
||||
"is_reversed_order",
|
||||
"scan_glyph_order",
|
||||
"ReadingOrderIssue",
|
||||
"find_reading_order_issues",
|
||||
"scan_reading_order",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
"""2D (stacked-fraction) formula regions, loaded from a verified list.
|
||||
|
||||
Why a curated file and not a detector: the residual-ink check produces
|
||||
*candidates* — thin ink bars that no extracted span accounts for — and its
|
||||
measured precision on this book is 16 of 23, **69.6%**. The seven misses are
|
||||
decorative underlines on the Ministry decision page, ruled boxes and table
|
||||
borders. A 70%-precise rule must not be allowed to quarantine content on its
|
||||
own, so every candidate was rendered and read, and only the confirmed ones
|
||||
are listed in `data/verified/formula_regions_2d.json`.
|
||||
|
||||
The stored bbox is the fraction bar itself. The numerator sits above it and
|
||||
the denominator below, so the bar is grown vertically here to cover the whole
|
||||
formula. The growth factor is deliberately generous: over-capturing a line of
|
||||
neighbouring prose into a quarantined block is recoverable, leaving half a
|
||||
formula in the prose is not.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict, List
|
||||
|
||||
from ..tables.classify import SHAPE_FORMULA_2D
|
||||
from ..tables.models import TableRegion
|
||||
|
||||
# One line of body type on this book measures ~10.5pt; a fraction spans the
|
||||
# numerator line, the bar and the denominator line.
|
||||
FORMULA_BAND_HEIGHT_PT = 13.0
|
||||
|
||||
# Wide on purpose. The bar is often narrower than the numerator above it, and
|
||||
# a numerator span can carry leading spaces that push its box's centre well to
|
||||
# the left of the bar: at 4pt of margin, AMPICILIN VÀ SULBACTAM's numerator
|
||||
# 'Thể trọng (kg)' stayed behind in the prose while the rest of the fraction
|
||||
# was lifted. A fraction sits alone on its lines, so taking most of the column
|
||||
# width costs at worst a neighbouring line inside a quarantined block.
|
||||
FORMULA_SIDE_MARGIN_PT = 95.0
|
||||
|
||||
DEFAULT_VERIFIED_PATH = (
|
||||
Path(__file__).resolve().parents[2] / "data" / "verified"
|
||||
/ "formula_regions_2d.json"
|
||||
)
|
||||
|
||||
|
||||
def load_formula_regions(path: Path | None = None) -> List[TableRegion]:
|
||||
"""Read the verified 2D-formula regions as page regions to divert."""
|
||||
source = path or DEFAULT_VERIFIED_PATH
|
||||
if not source.exists():
|
||||
return []
|
||||
|
||||
payload = json.loads(source.read_text(encoding="utf-8"))
|
||||
regions = []
|
||||
for index, entry in enumerate(payload["regions"]):
|
||||
x0, y0, x1, y1 = entry["bar_bbox"]
|
||||
regions.append(
|
||||
TableRegion(
|
||||
table_id=f"p{entry['physical_page']}_f{index}",
|
||||
physical_page=entry["physical_page"],
|
||||
bbox=(x0 - FORMULA_SIDE_MARGIN_PT, y0 - FORMULA_BAND_HEIGHT_PT,
|
||||
x1 + FORMULA_SIDE_MARGIN_PT, y1 + FORMULA_BAND_HEIGHT_PT),
|
||||
n_rows=2,
|
||||
n_cols=1,
|
||||
shape=SHAPE_FORMULA_2D,
|
||||
)
|
||||
)
|
||||
return regions
|
||||
|
||||
|
||||
def index_formula_regions_by_page(
|
||||
regions: List[TableRegion],
|
||||
) -> Dict[int, List[TableRegion]]:
|
||||
index: Dict[int, List[TableRegion]] = {}
|
||||
for region in regions:
|
||||
index.setdefault(region.physical_page, []).append(region)
|
||||
return index
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Mandatory glyph/reading-order sanity check.
|
||||
|
||||
Two distinct defect shapes were confirmed by testing this module against the
|
||||
real PDF (not assumed from the ADR description alone):
|
||||
|
||||
1. **Within-span glyph reversal** (`scan_glyph_order` / `GlyphOrderIssue`):
|
||||
physical page 1373 (0-indexed) contains a span whose characters are
|
||||
positioned in strictly decreasing x-origin order, producing scrambled
|
||||
text (e.g. "= tịx 8 y..." instead of "y 8 xịt ="). Matches ADR 0003's
|
||||
original description.
|
||||
|
||||
2. **Cross-span row misordering within one PyMuPDF block**
|
||||
(`scan_reading_order` / `ReadingOrderIssue`) — a genuinely different,
|
||||
previously undocumented shape found while testing this module end-to-end:
|
||||
physical page 714 has a visual text row split into multiple PyMuPDF line
|
||||
objects, within a single `block`, that are emitted out of left-to-right
|
||||
order relative to each other (each individual span's own characters are
|
||||
fine, but the fragments interleave incorrectly), e.g. the row "...bảo
|
||||
quản nhiệt độ..." is emitted as fragments "quản ", " ộ", "đ tệih", "n " in
|
||||
that (wrong) order. Concatenating characters in raw extraction order
|
||||
produces garbled text; re-sorting the *same* characters within one visual
|
||||
row by x-origin recovers the correct reading order exactly. This means
|
||||
ADR 0003's "exactly 1 occurrence in the whole book" claim was based on a
|
||||
narrower (within-span-only) check and undercounted the real defect
|
||||
population — corrected here, see docs/pdf-parsing-outlier-catalog.md
|
||||
item 9 update.
|
||||
|
||||
**Getting the row-grouping key right took three iterations, each caught by
|
||||
running against the real book rather than trusting the first result (per
|
||||
CLAUDE.md's no-fabrication rule) — recorded here since the failure modes
|
||||
generalize to any from-scratch "reconstruct visual rows from raw
|
||||
coordinates" approach:**
|
||||
- v1 (group by rounded y only): 1113 "issues", almost all false positives.
|
||||
- v2 (group by (`_column_for_x` tag, rounded y), using the same ±20pt
|
||||
tolerance `extract/spans.py` uses for informational span tagging): dropped
|
||||
to 32, but a real false-positive class remained — kerning jitter (e.g.
|
||||
"mefloquin"'s 'l'/'o' origins differ by only 0.095pt, well inside normal
|
||||
font kerning) was treated as a reversal with no decrease tolerance, and
|
||||
the ±20pt column tolerance creates an *overlapping* accepted x-range for
|
||||
"left" (24-319) and "right" (288-582) — a right-column paragraph
|
||||
starting near x=299 was misclassified "left" and merged with an unrelated
|
||||
left-column line sharing the same y.
|
||||
- v3 (this version — group by (PyMuPDF's own `block` index, rounded y)):
|
||||
the real fix. Two paragraphs from genuinely different columns (e.g. page
|
||||
1104: one block starting at x=299.4, another at x=35.4, both at y=70.4)
|
||||
turned out to sit in **different PyMuPDF blocks**, while page 714's 3
|
||||
genuinely-misordered fragments sit in the **same block** (block 20) split
|
||||
across multiple `line` entries. Block identity — PyMuPDF's own layout
|
||||
analysis, already validated in ADR 0003 to respect this document's
|
||||
two-column structure — is a reliable discriminator that no fixed
|
||||
x-coordinate threshold can be, since real paragraph start positions vary
|
||||
enough to overlap any hand-picked column boundary. A minimum-decrease
|
||||
threshold (`_MIN_DECREASE_PT`, well above observed kerning jitter <0.3pt
|
||||
and well below observed real defects >2pt) still guards against sub-pixel
|
||||
jitter within a block/row. The header band (running page number + drug
|
||||
name, two unrelated boilerplate fields sharing a y-coordinate — stripped
|
||||
before chunking regardless, outlier-catalog item 13) is excluded outright.
|
||||
|
||||
Both checks are cheap (seconds per full-book pass) and must run over 100% of
|
||||
pages, not sampled, per ADR 0003's standing rigor bar.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Sequence, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
_ROW_Y_PRECISION = 1 # decimal places; same-baseline chars share y to <0.01pt in practice
|
||||
_HEADER_BAND_Y = 60.0 # page number + running drug name live here; boilerplate, stripped separately
|
||||
_MIN_DECREASE_PT = 1.0 # observed kerning jitter <0.3pt; observed real defects >2pt — safely between
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GlyphOrderIssue:
|
||||
physical_page: int
|
||||
span_bbox: Tuple[float, float, float, float]
|
||||
original_text: str
|
||||
corrected_text: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReadingOrderIssue:
|
||||
physical_page: int
|
||||
block_index: int
|
||||
row_y: float
|
||||
extracted_text: str
|
||||
corrected_text: str
|
||||
|
||||
|
||||
def is_reversed_order(x_origins: Sequence[float]) -> bool:
|
||||
"""True if every consecutive pair strictly decreases in x — the exact
|
||||
shape of the confirmed within-span defect. A normal LTR span's
|
||||
x-origins strictly increase; requiring *every* pair to decrease (not
|
||||
just "not sorted") avoids false-triggering on ordinary spans.
|
||||
"""
|
||||
if len(x_origins) < 2:
|
||||
return False
|
||||
# strict=False on purpose: this is the adjacent-pair idiom, so the two
|
||||
# sequences differ in length by one by construction.
|
||||
return all(b < a for a, b in zip(x_origins, x_origins[1:], strict=False))
|
||||
|
||||
|
||||
def scan_glyph_order(doc: fitz.Document) -> List[GlyphOrderIssue]:
|
||||
issues: List[GlyphOrderIssue] = []
|
||||
for pno in range(doc.page_count):
|
||||
for block in doc[pno].get_text("rawdict").get("blocks", []):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
chars = span.get("chars", [])
|
||||
if not chars:
|
||||
continue
|
||||
x_origins = [c["origin"][0] for c in chars]
|
||||
if is_reversed_order(x_origins):
|
||||
issues.append(GlyphOrderIssue(
|
||||
physical_page=pno,
|
||||
span_bbox=tuple(span["bbox"]),
|
||||
original_text="".join(c["c"] for c in chars),
|
||||
corrected_text="".join(c["c"] for c in reversed(chars)),
|
||||
))
|
||||
return issues
|
||||
|
||||
|
||||
def _has_significant_backward_jump(xs: Sequence[float], min_decrease: float) -> bool:
|
||||
return any(b < a - min_decrease for a, b in zip(xs, xs[1:], strict=False))
|
||||
|
||||
|
||||
def find_reading_order_issues(
|
||||
chars_by_row: Dict[Tuple[int, float], List[Tuple[float, str]]],
|
||||
min_decrease: float = _MIN_DECREASE_PT,
|
||||
) -> List[ReadingOrderIssue]:
|
||||
"""Pure logic, unit-testable without a real PDF: given characters already
|
||||
grouped by (block_index, row_y) in raw extraction order, flag a row only
|
||||
when it contains a backward x-jump larger than `min_decrease` — ordinary
|
||||
font kerning produces sub-0.3pt jitter (see module docstring's v2
|
||||
entry), so a plain "resorting changes the text" check without this
|
||||
threshold is not reliable; it self-corrupts already-correct text.
|
||||
Grouping by block index (not a hand-picked x-coordinate column
|
||||
boundary) is what the caller must guarantee — see module docstring's
|
||||
v1/v2/v3 history for why a coordinate-based row reconstruction alone is
|
||||
not safe.
|
||||
"""
|
||||
issues = []
|
||||
for (block_index, y), chars in chars_by_row.items():
|
||||
if len(chars) < 2:
|
||||
continue
|
||||
xs = [x for x, _ in chars]
|
||||
if not _has_significant_backward_jump(xs, min_decrease):
|
||||
continue
|
||||
extracted = "".join(c for _, c in chars)
|
||||
corrected = "".join(c for _, c in sorted(chars, key=lambda t: t[0]))
|
||||
if extracted != corrected:
|
||||
issues.append(ReadingOrderIssue(
|
||||
physical_page=-1, block_index=block_index, row_y=y,
|
||||
extracted_text=extracted, corrected_text=corrected,
|
||||
))
|
||||
return issues
|
||||
|
||||
|
||||
def scan_reading_order(doc: fitz.Document) -> List[ReadingOrderIssue]:
|
||||
issues: List[ReadingOrderIssue] = []
|
||||
for pno in range(doc.page_count):
|
||||
rows: Dict[Tuple[int, float], List[Tuple[float, str]]] = defaultdict(list)
|
||||
for block_index, block in enumerate(doc[pno].get_text("rawdict").get("blocks", [])):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
for c in span.get("chars", []):
|
||||
x, y = c["origin"]
|
||||
if y < _HEADER_BAND_Y:
|
||||
continue
|
||||
rows[(block_index, round(y, _ROW_Y_PRECISION))].append((x, c["c"]))
|
||||
for issue in find_reading_order_issues(rows):
|
||||
issues.append(ReadingOrderIssue(
|
||||
physical_page=pno, block_index=issue.block_index, row_y=issue.row_y,
|
||||
extracted_text=issue.extracted_text, corrected_text=issue.corrected_text,
|
||||
))
|
||||
return issues
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Persists the extracted span stream so re-segmentation doesn't require
|
||||
re-running PyMuPDF over the whole PDF every time.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from .models import Span
|
||||
|
||||
|
||||
def write_spans_jsonl(spans: Iterable[Span], path: Path) -> int:
|
||||
count = 0
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
for span in spans:
|
||||
f.write(json.dumps(span.__dict__, ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def read_spans_jsonl(path: Path) -> Iterator[Span]:
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
yield Span(**json.loads(line))
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Data model for text spans extracted from the source PDF.
|
||||
|
||||
A Span is one PyMuPDF text span (a run of characters sharing one font/size),
|
||||
tagged with page and column position. This is the sole unit `segment/`
|
||||
consumes — it never touches PyMuPDF or fitz.Document directly (see ADR 0003
|
||||
and docs/pdf-parsing-outlier-catalog.md for why: PyMuPDF is the validated
|
||||
sole general-text extractor for this document).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Span:
|
||||
physical_page: int
|
||||
printed_page: Optional[int]
|
||||
column: str # "left" | "right" | "full_width" | "unknown"
|
||||
block: int
|
||||
line: int
|
||||
span_index: int
|
||||
x0: float
|
||||
y0: float
|
||||
x1: float
|
||||
y1: float
|
||||
text: str
|
||||
font: str
|
||||
size: float
|
||||
|
||||
@property
|
||||
def bold(self) -> bool:
|
||||
return "Bold" in self.font
|
||||
|
||||
@property
|
||||
def span_id(self) -> str:
|
||||
"""Stable identifier for one source span.
|
||||
|
||||
Built from PyMuPDF's own page/block/line/span indices, so the same PDF
|
||||
always yields the same id — a counter would renumber whenever anything
|
||||
upstream changed, which makes downstream provenance unverifiable.
|
||||
"""
|
||||
return f"p{self.physical_page}_b{self.block}_l{self.line}_s{self.span_index}"
|
||||
@@ -0,0 +1,108 @@
|
||||
"""Text that was drawn as vector outlines instead of text operators.
|
||||
|
||||
Physical page 714 prints 17 lines of ordinary Gatifloxacin prose that no text
|
||||
extractor returns: `page.get_text()` omits them, `page.search_for()` finds
|
||||
nothing, `pdfplumber` and `opendataloader-pdf` omit them too. They are not
|
||||
text at all in the file — each line is a filled path of ~1,600-1,800 items,
|
||||
shaped exactly like one line of type and filled with the body-text colour.
|
||||
|
||||
Nothing that asks a text layer can see this, which is why it survived every
|
||||
earlier check in this project. It was found by masking extracted spans over a
|
||||
rendered page and looking at the ink that was left.
|
||||
|
||||
Detection is deliberately shape-based, not content-based: a filled drawing
|
||||
with hundreds of path items whose box is the height of one line and at least
|
||||
30pt wide. Recovery cannot be automatic — the glyphs carry no character
|
||||
codes — so these regions are reported for transcription, never guessed at.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator, List, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
# Measured whole-document. Full outlined lines carry 1,126-1,831 path items;
|
||||
# single outlined glyphs carry 39-45. Ordinary decoration (the running-header
|
||||
# rule, cell borders) carries 1-2, so 30 separates them cleanly. Lowering the
|
||||
# threshold from 200 to 30 was checked before it was applied: it adds 29 runs
|
||||
# and no new page, all on pages 714 and 736, which were already affected.
|
||||
MIN_PATH_ITEMS = 30
|
||||
MIN_RUN_WIDTH_PT = 2.0
|
||||
RUN_HEIGHT_RANGE_PT = (3.0, 20.0)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OutlinedTextRun:
|
||||
"""A run of type that exists only as vector paths — a line, or one glyph.
|
||||
|
||||
`text` stays empty unless a human (or a rendered-page reading) fills it
|
||||
in: the paths carry no character codes, so any text here is a
|
||||
transcription and must be recorded as one.
|
||||
"""
|
||||
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
path_items: int
|
||||
text: str = ""
|
||||
|
||||
@property
|
||||
def is_transcribed(self) -> bool:
|
||||
return bool(self.text)
|
||||
|
||||
|
||||
def _is_outlined_run(drawing: dict, page_width: float) -> bool:
|
||||
rect = drawing["rect"]
|
||||
low, high = RUN_HEIGHT_RANGE_PT
|
||||
return (
|
||||
drawing["type"] == "f"
|
||||
and len(drawing["items"]) >= MIN_PATH_ITEMS
|
||||
and low <= rect.height <= high
|
||||
and rect.width >= MIN_RUN_WIDTH_PT
|
||||
and rect.x0 >= 0
|
||||
and rect.x1 <= page_width + 1
|
||||
)
|
||||
|
||||
|
||||
def detect_outlined_text(
|
||||
doc: "fitz.Document", pages: Iterable[int] | None = None,
|
||||
) -> Iterator[OutlinedTextRun]:
|
||||
"""Yield every run of vector-outlined type in the document."""
|
||||
page_numbers = range(doc.page_count) if pages is None else pages
|
||||
for number in page_numbers:
|
||||
page = doc[number]
|
||||
for drawing in page.get_drawings():
|
||||
if not _is_outlined_run(drawing, page.rect.x1):
|
||||
continue
|
||||
rect = drawing["rect"]
|
||||
yield OutlinedTextRun(
|
||||
physical_page=number,
|
||||
bbox=(round(rect.x0, 2), round(rect.y0, 2),
|
||||
round(rect.x1, 2), round(rect.y1, 2)),
|
||||
path_items=len(drawing["items"]),
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_TRANSCRIPTIONS_PATH = (
|
||||
Path(__file__).resolve().parents[2] / "data" / "verified"
|
||||
/ "outlined_text_transcriptions.json"
|
||||
)
|
||||
|
||||
|
||||
def load_transcribed_runs(path: "Path | None" = None) -> List[OutlinedTextRun]:
|
||||
"""Read the transcribed runs. Every `text` here was read off a rendering."""
|
||||
source = path or DEFAULT_TRANSCRIPTIONS_PATH
|
||||
if not source.exists():
|
||||
return []
|
||||
payload = json.loads(source.read_text(encoding="utf-8"))
|
||||
return [
|
||||
OutlinedTextRun(
|
||||
physical_page=run["physical_page"],
|
||||
bbox=tuple(run["bbox"]),
|
||||
path_items=run["path_items"],
|
||||
text=run["text"],
|
||||
)
|
||||
for run in payload["runs"]
|
||||
]
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Maps physical (0-indexed) page numbers to the book's own printed folio
|
||||
number, by reading the isolated numeric token in each page's header band.
|
||||
|
||||
Required because ADR 0003's page-range rules ("monographs run printed pages
|
||||
99-1496") are meaningless without a real per-page mapping — verified rather
|
||||
than assumed to be a constant offset, since front matter in some books uses
|
||||
roman numerals or restarts numbering. In this book the mapping is empirically
|
||||
a constant (physical + 1) across the entire 1668 pages (verified against the
|
||||
milestone pages: physical 36->printed 37, physical 98->printed 99, physical
|
||||
100->printed 101 "Abacavir", physical 1496->printed 1497), but this module
|
||||
still reads the real folio per page rather than hard-coding that constant, so
|
||||
a future edition/scan with different numbering does not silently mis-map.
|
||||
|
||||
Confirmed real false-conflict case (physical page 1243, "RIBOFLAVIN (Vitamin
|
||||
B2)" monograph, found via a whole-book `cli validate` run and confirmed by
|
||||
rendering the page to an image): the monograph's own title sits high enough
|
||||
on the page that its "2" subscript (size 5.83) falls inside the header band
|
||||
alongside the real folio "1244" (size 10.0), producing two conflicting
|
||||
digit-only candidates and silently dropping the printed page — and with it
|
||||
the entire monograph, since every span on the page then fails the
|
||||
printed-page-range check. A genuine folio is set in the header's own running
|
||||
font size, never a subscript's reduced size, so preferring the
|
||||
largest-font-size candidate(s) resolves this without weakening the
|
||||
"never guess on a real conflict" rule for pages with, e.g., two same-size
|
||||
candidates (still returns None).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import fitz
|
||||
|
||||
_FOLIO_RE = re.compile(r"^\d{1,4}$")
|
||||
HEADER_BAND_Y = 60.0 # printed folio always appears in the top header band
|
||||
|
||||
|
||||
def build_page_map(doc: fitz.Document) -> Dict[int, Optional[int]]:
|
||||
"""Returns {physical_page: printed_page_or_None}. None means no folio
|
||||
was recoverable (blank/separator pages, title pages) — a valid state,
|
||||
not an error.
|
||||
"""
|
||||
return {pno: _read_folio(doc[pno]) for pno in range(doc.page_count)}
|
||||
|
||||
|
||||
def _read_folio(page: fitz.Page) -> Optional[int]:
|
||||
candidates = [] # (text, size) pairs
|
||||
for block in page.get_text("dict").get("blocks", []):
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
text = span["text"].strip()
|
||||
if span["bbox"][1] < HEADER_BAND_Y and _FOLIO_RE.match(text):
|
||||
candidates.append((text, span["size"]))
|
||||
return pick_folio(candidates)
|
||||
|
||||
|
||||
def pick_folio(candidates: List[Tuple[str, float]]) -> Optional[int]:
|
||||
"""Pure decision logic, given the header-band digit-only (text, size)
|
||||
candidates already collected from a page: which one is the real folio.
|
||||
"""
|
||||
if not candidates:
|
||||
return None # blank/separator page: unrecoverable, never guess.
|
||||
|
||||
max_size = max(size for _, size in candidates)
|
||||
largest = {text for text, size in candidates if size == max_size}
|
||||
if len(largest) == 1:
|
||||
return int(largest.pop())
|
||||
# still conflicting even after dropping smaller-font stray digits
|
||||
# (e.g. subscripts): genuinely ambiguous, never guess.
|
||||
return None
|
||||
@@ -0,0 +1,241 @@
|
||||
"""Put transcribed vector-outlined text back into the span stream.
|
||||
|
||||
Outlier-catalog item 24: 51 runs of type on 5 pages exist only as filled
|
||||
vector paths, so no extractor emits a span for them. They were transcribed by
|
||||
reading rendered crops (`data/verified/outlined_text_transcriptions.json`).
|
||||
This module is what makes that transcription part of the corpus rather than a
|
||||
note beside it.
|
||||
|
||||
Placement is geometric, not textual. A run that vertically overlaps an
|
||||
existing visual line is a character (or fragment) dropped out of *that* line
|
||||
and is spliced into it in x order — this is the common case and the damaging
|
||||
one, because a missing diacritic turns "Độ ổn định" into "Độ n định" and
|
||||
still reads as ordinary prose. A run that overlaps no line is a whole missing
|
||||
line and is inserted at a line boundary, ordered by column then y, so the
|
||||
reading order the rest of the pipeline depends on is preserved.
|
||||
|
||||
Synthetic spans are marked by `SYNTHETIC_LINE_BASE` in their line index, so
|
||||
their provenance ids stay distinguishable from real extracted spans forever.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import replace
|
||||
from typing import Iterable, List, Sequence, Tuple
|
||||
|
||||
from .models import Span
|
||||
from .outlined_text import OutlinedTextRun
|
||||
from .page_map import HEADER_BAND_Y
|
||||
from .spans import classify_column
|
||||
|
||||
# Line indices at or above this never come from PyMuPDF — a real page has
|
||||
# nothing close to this many lines in a block.
|
||||
SYNTHETIC_LINE_BASE = 100_000
|
||||
|
||||
SYNTHETIC_FONT = "TimesNewRomanPSMT"
|
||||
SYNTHETIC_SIZE = 9.5
|
||||
|
||||
# A run counts as belonging to an existing line when their vertical extents
|
||||
# overlap by at least this fraction of the run's height.
|
||||
LINE_OVERLAP_RATIO = 0.5
|
||||
|
||||
|
||||
def _vertical_overlap(a: Tuple[float, float], b: Tuple[float, float]) -> float:
|
||||
return max(0.0, min(a[1], b[1]) - max(a[0], b[0]))
|
||||
|
||||
|
||||
def _line_groups(spans: Sequence[Span]) -> List[Tuple[int, int, List[Span]]]:
|
||||
"""Consecutive spans sharing a visual line, with their index range.
|
||||
|
||||
Mirrors `normalize.group_visual_lines`' notion of a line so that a span
|
||||
spliced here lands in the same group there.
|
||||
"""
|
||||
groups: List[Tuple[int, int, List[Span]]] = []
|
||||
for index, span in enumerate(spans):
|
||||
key = (span.physical_page, span.block, span.line)
|
||||
if groups and (groups[-1][2][0].physical_page, groups[-1][2][0].block,
|
||||
groups[-1][2][0].line) == key:
|
||||
start, _, members = groups[-1]
|
||||
members.append(span)
|
||||
groups[-1] = (start, index, members)
|
||||
else:
|
||||
groups.append((index, index, [span]))
|
||||
return groups
|
||||
|
||||
|
||||
def _synthetic_span(run: OutlinedTextRun, template: Span | None,
|
||||
line_index: int) -> Span:
|
||||
x0, y0, x1, y1 = run.bbox
|
||||
column = classify_column(run.bbox)
|
||||
if y0 < HEADER_BAND_Y:
|
||||
# Part of the running header, which is a full-width band. Tagging it
|
||||
# as such lets the one existing boilerplate rule strip it, instead of
|
||||
# this module deciding separately what boilerplate is.
|
||||
column = "full_width"
|
||||
if template is not None:
|
||||
return replace(
|
||||
template,
|
||||
column=template.column if y0 >= HEADER_BAND_Y else column,
|
||||
line=line_index,
|
||||
span_index=0,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
)
|
||||
return Span(
|
||||
physical_page=run.physical_page,
|
||||
printed_page=None,
|
||||
column=column,
|
||||
block=0,
|
||||
line=line_index,
|
||||
span_index=0,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
size=SYNTHETIC_SIZE,
|
||||
)
|
||||
|
||||
|
||||
def char_boxes(doc, page_number: int) -> List[Tuple[str, Tuple[float, ...]]]:
|
||||
"""Per-character boxes for one page, in extraction order.
|
||||
|
||||
Needed because a dropped glyph usually sits *inside* an extracted span,
|
||||
not between two of them: on physical page 714 the span
|
||||
`'Viêm màng tiếp hợp nhiễm khuẩn trẻ em ≥ 1 tu'` runs from x=35.5 to
|
||||
x=223.0 and the missing 'ở' belongs at x=167. Splicing at span boundaries
|
||||
put it at the end and produced 'tuở ổi'. Character geometry is the only
|
||||
thing that says where the hole actually is.
|
||||
"""
|
||||
boxes = []
|
||||
for block in doc[page_number].get_text("rawdict")["blocks"]:
|
||||
for line in block.get("lines", []):
|
||||
for span in line["spans"]:
|
||||
for char in span["chars"]:
|
||||
boxes.append((char["c"], char["bbox"]))
|
||||
return boxes
|
||||
|
||||
|
||||
def _split_offset(span: Span, run: OutlinedTextRun,
|
||||
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> int | None:
|
||||
"""Character offset inside `span.text` where the run's glyph belongs."""
|
||||
inside = [
|
||||
box for box in boxes
|
||||
if span.x0 - 0.5 <= box[1][0] and box[1][2] <= span.x1 + 0.5
|
||||
and span.y0 - 1.0 <= box[1][1] and box[1][3] <= span.y1 + 1.0
|
||||
]
|
||||
if len(inside) != len(span.text):
|
||||
return None
|
||||
for offset, (_, bbox) in enumerate(inside):
|
||||
if bbox[0] >= run.bbox[2] - 0.5:
|
||||
return offset
|
||||
return None
|
||||
|
||||
|
||||
def _splice_into_line(spans: List[Span], group, run: OutlinedTextRun,
|
||||
boxes: Sequence[Tuple[str, Tuple[float, ...]]]) -> None:
|
||||
start, end, members = group
|
||||
anchor = members[0]
|
||||
synthetic = replace(
|
||||
anchor,
|
||||
span_index=SYNTHETIC_LINE_BASE,
|
||||
x0=run.bbox[0], y0=run.bbox[1], x1=run.bbox[2], y1=run.bbox[3],
|
||||
text=run.text,
|
||||
font=SYNTHETIC_FONT,
|
||||
)
|
||||
|
||||
for offset, member in enumerate(members):
|
||||
if not (member.x0 <= run.bbox[0] and run.bbox[2] <= member.x1):
|
||||
continue
|
||||
split_at = _split_offset(member, run, boxes)
|
||||
if split_at is None or split_at == 0:
|
||||
continue
|
||||
index = start + offset
|
||||
left = replace(member, text=member.text[:split_at], x1=run.bbox[0])
|
||||
right = replace(member, text=member.text[split_at:],
|
||||
span_index=member.span_index + SYNTHETIC_LINE_BASE,
|
||||
x0=run.bbox[2])
|
||||
# A split at the very end of a span leaves a fragment holding nothing
|
||||
# but a space. Dropping it costs no text — `join_visual_line` decides
|
||||
# spacing from the horizontal gap, not from a span's own padding.
|
||||
pieces = [p for p in (left, synthetic, right) if p.text.strip()]
|
||||
spans[index:index + 1] = pieces
|
||||
return
|
||||
|
||||
position = end + 1
|
||||
for offset, member in enumerate(members):
|
||||
if run.bbox[0] < member.x0:
|
||||
position = start + offset
|
||||
break
|
||||
spans.insert(position, synthetic)
|
||||
|
||||
|
||||
def _insert_as_new_line(spans: List[Span], run: OutlinedTextRun,
|
||||
line_index: int) -> None:
|
||||
column = classify_column(run.bbox)
|
||||
if run.bbox[1] < HEADER_BAND_Y:
|
||||
column = "full_width"
|
||||
|
||||
position = len(spans)
|
||||
template = None
|
||||
for start, _, members in _line_groups(spans):
|
||||
first = members[0]
|
||||
if first.physical_page < run.physical_page:
|
||||
template = first
|
||||
continue
|
||||
if first.physical_page > run.physical_page:
|
||||
position = start
|
||||
break
|
||||
if first.column == column:
|
||||
template = first
|
||||
if first.y0 > run.bbox[1]:
|
||||
position = start
|
||||
break
|
||||
elif template is not None and first.column != column and position == len(spans):
|
||||
# first line of the next column on this page — the run belongs
|
||||
# before it if we never found a lower line in its own column
|
||||
position = start
|
||||
spans.insert(position, _synthetic_span(run, template, line_index))
|
||||
|
||||
|
||||
def merge_outlined_runs(
|
||||
spans: Iterable[Span], runs: Sequence[OutlinedTextRun], doc=None,
|
||||
) -> List[Span]:
|
||||
"""Return the span stream with every transcribed run put back in place.
|
||||
|
||||
`doc` enables character-accurate splicing of a glyph that fell out of the
|
||||
middle of an extracted span. Without it the run can only be placed at a
|
||||
span boundary, which is wrong for exactly the case that matters most.
|
||||
"""
|
||||
merged = list(spans)
|
||||
ordered = sorted(runs, key=lambda r: (r.physical_page, r.bbox[1], r.bbox[0]))
|
||||
boxes_cache: dict = {}
|
||||
for offset, run in enumerate(ordered):
|
||||
if not run.text:
|
||||
continue
|
||||
run_extent = (run.bbox[1], run.bbox[3])
|
||||
run_column = classify_column(run.bbox)
|
||||
target = None
|
||||
for group in _line_groups(merged):
|
||||
first = group[2][0]
|
||||
if first.physical_page != run.physical_page:
|
||||
continue
|
||||
# Column, not just height: this book sets two columns, so a
|
||||
# right-column run sits at the same y as an unrelated left-column
|
||||
# line. Without this, page 714's "…làm thay đ" was spliced onto
|
||||
# the left column and "…ổi nồng độ glucose máu" stayed broken.
|
||||
if first.column != run_column:
|
||||
continue
|
||||
line_extent = (min(s.y0 for s in group[2]),
|
||||
max(s.y1 for s in group[2]))
|
||||
overlap = _vertical_overlap(run_extent, line_extent)
|
||||
height = max(run.bbox[3] - run.bbox[1], 0.1)
|
||||
if overlap / height >= LINE_OVERLAP_RATIO:
|
||||
target = group
|
||||
break
|
||||
if target is not None:
|
||||
if doc is not None and run.physical_page not in boxes_cache:
|
||||
boxes_cache[run.physical_page] = char_boxes(doc, run.physical_page)
|
||||
_splice_into_line(merged, target, run,
|
||||
boxes_cache.get(run.physical_page, ()))
|
||||
else:
|
||||
_insert_as_new_line(merged, run, SYNTHETIC_LINE_BASE + offset)
|
||||
return merged
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Continuous cross-page span extraction.
|
||||
|
||||
Per ADR 0003: the pipeline must consume text as one continuous cross-page
|
||||
stream, never per-page silos, so that multi-line headings and paragraphs
|
||||
spanning a page/column break can be handled correctly downstream. This
|
||||
module's only job is to yield that stream in reading order; it does not
|
||||
decide what is a heading or a monograph boundary (that's `segment/`'s job).
|
||||
|
||||
Column tagging uses the bounding-box ranges confirmed by inspection in ADR
|
||||
0003 (left column x~44-299, right column x~308-562, page width ~595).
|
||||
|
||||
An earlier version of this module trusted PyMuPDF's own raw block order to
|
||||
already sequence left-then-right correctly, validated only against one
|
||||
example page during ADR 0003. Confirmed wrong via a whole-document
|
||||
character-diff against an independent parser (opendataloader-pdf) plus
|
||||
visual page reads: on 12 of 1398 monograph-range pages (e.g. physical page
|
||||
1100, the OXYBUTYNIN/OXYMETAZOLIN boundary), PyMuPDF's raw block order
|
||||
emits the *right* column before the *left* column. Left uncorrected, this
|
||||
silently corrupts monograph data at a column-reversed page's drug boundary
|
||||
— the wrong column's section content (e.g. "Chống chỉ định") gets appended
|
||||
to whichever monograph is still open when it's encountered, overwriting
|
||||
that monograph's real section and leaving the next monograph missing it.
|
||||
Fixed by explicitly sorting blocks (full_width header band first, then
|
||||
left column, then right column, each by y-position) instead of trusting
|
||||
raw order — full_width blocks are confirmed to be page-header material
|
||||
only in this document (real monograph titles and section headings sit
|
||||
within one column's x-range), so this ordering matches the book's actual
|
||||
two-column-with-running-header layout.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterator
|
||||
|
||||
import fitz
|
||||
|
||||
from .models import Span
|
||||
from .page_map import build_page_map
|
||||
|
||||
_LEFT_COLUMN_X = (44.0, 299.0)
|
||||
_RIGHT_COLUMN_X = (308.0, 562.0)
|
||||
_COLUMN_TOLERANCE = 20.0
|
||||
_FULL_WIDTH_MIN = 400.0
|
||||
_COLUMN_SORT_RANK = {"full_width": 0, "left": 1, "right": 2, "unknown": 3}
|
||||
|
||||
|
||||
def extract_spans(doc: fitz.Document) -> Iterator[Span]:
|
||||
page_map = build_page_map(doc)
|
||||
for pno in range(doc.page_count):
|
||||
printed = page_map[pno]
|
||||
page_dict = doc[pno].get_text("dict")
|
||||
blocks = _sort_blocks_reading_order(page_dict.get("blocks", []))
|
||||
for block_idx, block in enumerate(blocks):
|
||||
column = classify_column(block.get("bbox"))
|
||||
for line_idx, line in enumerate(block.get("lines", [])):
|
||||
for span_idx, span in enumerate(line.get("spans", [])):
|
||||
text = span["text"]
|
||||
if not text.strip():
|
||||
continue
|
||||
x0, y0, x1, y1 = span["bbox"]
|
||||
yield Span(
|
||||
physical_page=pno,
|
||||
printed_page=printed,
|
||||
column=column,
|
||||
block=block_idx,
|
||||
line=line_idx,
|
||||
span_index=span_idx,
|
||||
x0=x0, y0=y0, x1=x1, y1=y1,
|
||||
text=text,
|
||||
font=span["font"],
|
||||
size=span["size"],
|
||||
)
|
||||
|
||||
|
||||
def _sort_blocks_reading_order(blocks: list) -> list:
|
||||
"""Full_width header band first, then left column, then right column,
|
||||
each by y-position — see module docstring for the confirmed real bug
|
||||
this replaces (trusting PyMuPDF's raw block order).
|
||||
"""
|
||||
return sorted(
|
||||
blocks,
|
||||
key=lambda b: (_COLUMN_SORT_RANK[classify_column(b.get("bbox"))], b.get("bbox", (0, 0, 0, 0))[1]),
|
||||
)
|
||||
|
||||
|
||||
def classify_column(bbox) -> str:
|
||||
"""Which of the book's two columns a box sits in (or the header band)."""
|
||||
if bbox is None:
|
||||
return "unknown"
|
||||
x0, _, x1, _ = bbox
|
||||
if (x1 - x0) >= _FULL_WIDTH_MIN:
|
||||
return "full_width"
|
||||
mid = (x0 + x1) / 2
|
||||
# Exact containment before tolerance. The two tolerance bands overlap
|
||||
# between x=288 and x=319, and testing left first put anything in that
|
||||
# strip in the left column — invisible for a full-width block, wrong for a
|
||||
# narrow one. A single 4pt glyph at x=315 on physical page 714 was
|
||||
# classified left, so the 'ổ' missing from "Độ ổn định" could not be
|
||||
# matched to its own line and the corruption survived the repair.
|
||||
if _LEFT_COLUMN_X[0] <= mid <= _LEFT_COLUMN_X[1]:
|
||||
return "left"
|
||||
if _RIGHT_COLUMN_X[0] <= mid <= _RIGHT_COLUMN_X[1]:
|
||||
return "right"
|
||||
if _LEFT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _LEFT_COLUMN_X[1] + _COLUMN_TOLERANCE:
|
||||
return "left"
|
||||
if _RIGHT_COLUMN_X[0] - _COLUMN_TOLERANCE <= mid <= _RIGHT_COLUMN_X[1] + _COLUMN_TOLERANCE:
|
||||
return "right"
|
||||
return "unknown"
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Text normalization shared by the pipeline and any validation script.
|
||||
|
||||
Kept as its own stage so the rules live in exactly one place (CLAUDE.md's DRY
|
||||
rule): `segment/` applies them when assembling section text, and audits
|
||||
import the same functions rather than re-implementing them.
|
||||
"""
|
||||
from .glyphs import PUA_SUBSTITUTIONS, find_unmapped_pua, substitute_pua
|
||||
from .text_flow import SPACE_GAP_PT, group_visual_lines, join_spans, join_visual_line
|
||||
|
||||
__all__ = [
|
||||
"PUA_SUBSTITUTIONS",
|
||||
"SPACE_GAP_PT",
|
||||
"find_unmapped_pua",
|
||||
"substitute_pua",
|
||||
"group_visual_lines",
|
||||
"join_spans",
|
||||
"join_visual_line",
|
||||
]
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Private-use-area glyph substitution.
|
||||
|
||||
The source PDF sets several symbols in the `SymbolTiger` / `Symbol` fonts,
|
||||
which PyMuPDF faithfully returns as raw Unicode private-use-area codepoints.
|
||||
Left untranslated they reach embeddings as junk — and 74 of the 86
|
||||
occurrences in this corpus are the comparison operators inside dosing
|
||||
sentences, where losing the operator changes clinical meaning ("liều ≤ 100
|
||||
mg" is not "liều 100 mg").
|
||||
|
||||
Every entry below was located in the source PDF, rendered to an image, and
|
||||
read visually — none inferred from surrounding context. Counts and the page
|
||||
each was confirmed on are recorded in docs/progress-log.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
PUA_LO, PUA_HI = 0xE000, 0xF8FF
|
||||
|
||||
# codepoint -> replacement, with the page the glyph was visually confirmed on
|
||||
PUA_SUBSTITUTIONS: dict[str, str] = {
|
||||
"": "≥", # p.141 "trẻ em ≥ 10 tuổi"
|
||||
"": "≤", # p.169 "liều ≤ 100 mg"
|
||||
"": "α", # p.334 "Streptococcus α tan huyết"
|
||||
"": "→", # p.1027 "HCO₃⁻ + H⁺ → H₂CO₃"
|
||||
"": "®", # p.891 "Plasma Lyte® 56/5%"
|
||||
"": "₁", # p.957 "alpha₁-acid glycoprotein"
|
||||
"": "↓", # p.1033 "rhodanese ↓" (catalysis arrow)
|
||||
"": "γ", # p.1352 "interferon - γ"
|
||||
}
|
||||
|
||||
_TABLE = str.maketrans(PUA_SUBSTITUTIONS)
|
||||
|
||||
|
||||
def substitute_pua(text: str) -> str:
|
||||
return text.translate(_TABLE)
|
||||
|
||||
|
||||
def find_unmapped_pua(text: str) -> list[str]:
|
||||
"""PUA codepoints with no verified replacement.
|
||||
|
||||
Returned rather than silently passed through: an unmapped glyph means the
|
||||
corpus contains a symbol nobody has visually confirmed yet, which must be
|
||||
surfaced instead of embedded as junk.
|
||||
"""
|
||||
return sorted({
|
||||
ch for ch in text
|
||||
if PUA_LO <= ord(ch) <= PUA_HI and ch not in PUA_SUBSTITUTIONS
|
||||
})
|
||||
@@ -0,0 +1,83 @@
|
||||
"""Rejoin PDF spans into flowing text.
|
||||
|
||||
`segment/assembler.py` originally appended one line per *span*, so any visual
|
||||
line that the PDF split into several spans (an italic run, a subscript, a
|
||||
symbol-font glyph) became several "lines". Measured on the whole corpus that
|
||||
produced 99,501 mid-sentence line breaks across 71.8% of sections and 11,612
|
||||
sub-4-character fragment lines — e.g. `"cytochrom P\n450\ngây"`,
|
||||
`"(\nfeline immunodeficiency virus\n)"`, `"Cl\ncr\n< 50 ml/"`.
|
||||
|
||||
Text alone cannot tell a mid-word span split from a genuine line wrap, so the
|
||||
join is driven by geometry instead: PyMuPDF's own `(block, line)` indices say
|
||||
which spans share a visual line, and the horizontal gap says whether a space
|
||||
belongs between them.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterable, List, Sequence, Tuple
|
||||
|
||||
from ..extract.models import Span
|
||||
|
||||
# Horizontal gap (pt) above which two spans on one visual line are separated
|
||||
# by a real space. Kerning noise between adjacent glyph runs sits well under
|
||||
# 1pt; a space at this book's 9.5-10pt body size is ≈2.4pt.
|
||||
SPACE_GAP_PT = 1.0
|
||||
|
||||
_SENTENCE_END = ".;:!?"
|
||||
|
||||
|
||||
def _line_key(span: Span) -> Tuple[int, int, int]:
|
||||
return (span.physical_page, span.block, span.line)
|
||||
|
||||
|
||||
def group_visual_lines(spans: Sequence[Span]) -> List[List[Span]]:
|
||||
"""Group consecutive spans that share a visual line, preserving order."""
|
||||
lines: List[List[Span]] = []
|
||||
for span in spans:
|
||||
if lines and _line_key(lines[-1][0]) == _line_key(span):
|
||||
lines[-1].append(span)
|
||||
else:
|
||||
lines.append([span])
|
||||
return lines
|
||||
|
||||
|
||||
def join_visual_line(spans: Sequence[Span]) -> str:
|
||||
"""Concatenate one visual line, inserting a space only where one exists."""
|
||||
out = ""
|
||||
previous: Span | None = None
|
||||
for span in spans:
|
||||
text = span.text
|
||||
if previous is not None:
|
||||
gap = span.x0 - previous.x1
|
||||
needs_space = (
|
||||
gap >= SPACE_GAP_PT
|
||||
and not out.endswith(" ")
|
||||
and not text.startswith(" ")
|
||||
)
|
||||
if needs_space:
|
||||
out += " "
|
||||
out += text
|
||||
previous = span
|
||||
return out.strip()
|
||||
|
||||
|
||||
def join_spans(spans: Iterable[Span]) -> str:
|
||||
"""Rejoin spans into flowing text.
|
||||
|
||||
A visual line that does not end a sentence is treated as a soft wrap and
|
||||
joined to the next line with a space; a line ending in sentence
|
||||
punctuation keeps its newline, which preserves paragraph and list
|
||||
structure for display and citation.
|
||||
"""
|
||||
lines = [join_visual_line(group) for group in group_visual_lines(list(spans))]
|
||||
lines = [line for line in lines if line]
|
||||
if not lines:
|
||||
return ""
|
||||
|
||||
out = lines[0]
|
||||
for line in lines[1:]:
|
||||
if out.rstrip().endswith(tuple(_SENTENCE_END)):
|
||||
out += "\n" + line
|
||||
else:
|
||||
out += " " + line
|
||||
return out
|
||||
@@ -0,0 +1,32 @@
|
||||
from .assembler import DuplicateDrugIdError, assemble
|
||||
from .atc import ATCResult, extract_atc_codes, is_stated_absent, normalize_atc_candidate
|
||||
from .detector import detect_monograph_titles, detect_section_headings
|
||||
from .io import read_monographs_jsonl, write_monographs_jsonl
|
||||
from .merge import merge_multiline_headings
|
||||
from .models import Heading, Monograph, SectionSpan
|
||||
from .units import normalize_unit_token, validate_unit_tokens
|
||||
from .vocab import SECTION_DEFS, SectionDef, is_part_divider, match_section, normalize_heading_text
|
||||
|
||||
__all__ = [
|
||||
"Heading",
|
||||
"SectionSpan",
|
||||
"Monograph",
|
||||
"assemble",
|
||||
"DuplicateDrugIdError",
|
||||
"detect_monograph_titles",
|
||||
"detect_section_headings",
|
||||
"merge_multiline_headings",
|
||||
"write_monographs_jsonl",
|
||||
"read_monographs_jsonl",
|
||||
"ATCResult",
|
||||
"extract_atc_codes",
|
||||
"is_stated_absent",
|
||||
"normalize_atc_candidate",
|
||||
"normalize_unit_token",
|
||||
"validate_unit_tokens",
|
||||
"SectionDef",
|
||||
"SECTION_DEFS",
|
||||
"match_section",
|
||||
"is_part_divider",
|
||||
"normalize_heading_text",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,485 @@
|
||||
"""Assembles a raw span stream into ordered Monograph records.
|
||||
|
||||
Three simple passes, each independently easy to reason about — avoids a
|
||||
single tangled state machine (SRP: classify, then merge titles, then build):
|
||||
|
||||
1. Classify each span in reading order as a title candidate, a section
|
||||
heading, or body text.
|
||||
2. Coalesce consecutive title-candidate spans into single merged Heading
|
||||
events via `merge.merge_multiline_headings` (handles both the multi-line
|
||||
wrap and same-line font-size-split cases — see merge.py).
|
||||
3. Walk the resulting flat event stream once, building Monograph records.
|
||||
|
||||
Handles the confirmed real "qualifier line" case (outlier-catalog item 18):
|
||||
a monograph title can legitimately repeat (e.g. two distinct "SALBUTAMOL"
|
||||
monographs, "Dùng trong hô hấp" vs "Dùng trong sản khoa") disambiguated by a
|
||||
bold, parenthesized, non-all-caps line directly beneath the title. That
|
||||
qualifier is folded into `drug_id` so two legitimate entries don't collide;
|
||||
a genuine duplicate `drug_id` (no qualifier, same name) raises rather than
|
||||
silently overwriting, since the one apparent duplicate found during ADR
|
||||
0003's investigation (GONADOTROPIN) turned out to be a detector artifact,
|
||||
not real — a real second collision should be surfaced, not hidden.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from typing import Iterator, List, Optional, Union
|
||||
|
||||
from ..extract.models import Span
|
||||
from ..extract.page_map import HEADER_BAND_Y
|
||||
from ..normalize import join_spans, substitute_pua
|
||||
from ..tables.classify import QUARANTINE_SHAPES
|
||||
from .atc import extract_atc_codes
|
||||
from .detector import in_monograph_range, is_monograph_title_candidate
|
||||
from .merge import merge_multiline_headings, merge_same_line_bold_fragments
|
||||
from .models import (
|
||||
PART_PROSE,
|
||||
PART_TABLE,
|
||||
Heading,
|
||||
Monograph,
|
||||
SectionPart,
|
||||
SectionSpan,
|
||||
TableBlock,
|
||||
)
|
||||
from .vocab import (
|
||||
SectionDef,
|
||||
is_part_divider,
|
||||
match_section,
|
||||
match_section_with_inline_value,
|
||||
)
|
||||
|
||||
_QUALIFIER_RE = re.compile(r"^\(.+\)$")
|
||||
|
||||
|
||||
def _is_page_boilerplate(span: Span) -> bool:
|
||||
"""Confirmed real (outlier-catalog item 13, measured via a whole-book
|
||||
`assemble()` run): the running header ("DTQGVN 2" + page number +
|
||||
current monograph name, e.g. physical page 1008's "DTQGVN 2" / "1009" /
|
||||
"Morphin sulfat") was falling through every classification branch below
|
||||
into plain body text, since it matches no section heading and isn't a
|
||||
real all-caps title — silently splicing itself into the *middle* of
|
||||
whatever section happens to be open when a physical page turns (1,374
|
||||
of 11,409 sections / 671 of 682 monographs affected). It's reliably
|
||||
identifiable independent of its (non-vocabulary) text: always the
|
||||
full-page-width block in the header band, same signal `page_map.py`
|
||||
already uses to read the folio.
|
||||
"""
|
||||
return span.column == "full_width" and span.y0 < HEADER_BAND_Y
|
||||
|
||||
|
||||
class DuplicateDrugIdError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _SectionEvent:
|
||||
section_def: SectionDef
|
||||
span: Span
|
||||
inline_value: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _TextEvent:
|
||||
span: Span
|
||||
|
||||
|
||||
_Event = Union[Heading, _SectionEvent, _TextEvent] # Heading == a title event
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
normalized = unicodedata.normalize("NFKD", text)
|
||||
ascii_text = normalized.encode("ascii", "ignore").decode("ascii")
|
||||
return re.sub(r"[^a-z0-9]+", "_", ascii_text.lower()).strip("_")
|
||||
|
||||
|
||||
def _is_body_line_that_reads_like_a_label(span: Span, items: List) -> bool:
|
||||
"""A plain line that repeats a section name, sitting under a heading.
|
||||
|
||||
Confirmed real and clinically material: FLUOROURACIL (physical page 681)
|
||||
prints `Thời kỳ mang thai` / `Chống chỉ định.` and `Thời kỳ cho con bú` /
|
||||
`Chống chỉ định.`, verified by rendering the page. The body line matches
|
||||
the section vocabulary, so it was read as a heading — leaving both
|
||||
pregnancy and lactation sections empty and dropping the statement that
|
||||
fluorouracil is contraindicated in both.
|
||||
|
||||
The book never prints an empty section, so a *non-bold* label immediately
|
||||
after a heading is that heading's body. Boldness still cannot be required
|
||||
in general (outlier item 20: `Mã ATC: N06AA09.` is a plain span), which is
|
||||
why this is narrowed to the directly-under-a-heading position.
|
||||
"""
|
||||
if span.bold:
|
||||
return False
|
||||
return bool(items) and isinstance(items[-1], _SectionEvent)
|
||||
|
||||
|
||||
def _classify(spans: List[Span]) -> List[Union[Span, _SectionEvent, _TextEvent]]:
|
||||
"""Pass 1: tag each span. Title candidates are left as raw Span objects
|
||||
(pass 2 groups + merges them); everything else becomes a typed event.
|
||||
|
||||
Section matching does NOT require `span.bold` — confirmed real (outlier
|
||||
item 20): AMITRIPTYLIN's "Mã ATC: N06AA09." is a single **plain, non-bold**
|
||||
span (Abacavir's equivalent is bold "Mã ATC: " + a separate plain value
|
||||
span), inconsistent across the book's ~700 individually-authored
|
||||
monographs (the book's own foreword notes "biên soạn bởi nhiều tác giả").
|
||||
Matching by exact vocabulary text (not styling) is the reliable signal,
|
||||
same lesson as "don't gate on font size" (ADR 0003 item 10) applied to
|
||||
boldness instead.
|
||||
"""
|
||||
items: List[Union[Span, _SectionEvent, _TextEvent]] = []
|
||||
for span in spans:
|
||||
if not span.text.strip():
|
||||
continue
|
||||
if _is_page_boilerplate(span):
|
||||
continue
|
||||
if is_monograph_title_candidate(span):
|
||||
items.append(span)
|
||||
continue
|
||||
section_def = match_section(span.text)
|
||||
if section_def is not None and not _is_body_line_that_reads_like_a_label(
|
||||
span, items
|
||||
):
|
||||
items.append(_SectionEvent(section_def, span))
|
||||
continue
|
||||
if section_def is not None:
|
||||
items.append(_TextEvent(span))
|
||||
continue
|
||||
inline = match_section_with_inline_value(span.text)
|
||||
if inline is not None:
|
||||
items.append(_SectionEvent(inline[0], span, inline_value=inline[1]))
|
||||
else:
|
||||
items.append(_TextEvent(span))
|
||||
return items
|
||||
|
||||
|
||||
def _coalesce_titles(items: List[Union[Span, _SectionEvent, _TextEvent]]) -> List[_Event]:
|
||||
"""Pass 2: merge consecutive raw title-candidate Span runs into single
|
||||
Heading events, preserving the order of everything else.
|
||||
"""
|
||||
events: List[_Event] = []
|
||||
run: List[Span] = []
|
||||
|
||||
def flush_run():
|
||||
if run:
|
||||
events.extend(merge_multiline_headings(list(run)))
|
||||
run.clear()
|
||||
|
||||
for item in items:
|
||||
if isinstance(item, Span):
|
||||
run.append(item)
|
||||
else:
|
||||
flush_run()
|
||||
events.append(item)
|
||||
flush_run()
|
||||
return events
|
||||
|
||||
|
||||
def _is_qualifier_line(span: Span) -> bool:
|
||||
text = span.text.strip()
|
||||
return span.bold and not text.isupper() and bool(_QUALIFIER_RE.match(text))
|
||||
|
||||
|
||||
_ANCHOR_LOOKAHEAD = 6
|
||||
_ANCHOR_SECTION_KEY = "ten_chung_quoc_te"
|
||||
|
||||
|
||||
def _has_anchor_ahead(events: List[_Event], title_index: int) -> bool:
|
||||
"""Every real monograph documents "Tên chung quốc tế" as its very first
|
||||
section (the book's own template, item 2 — see vocab.py docstring).
|
||||
Loosening this to "any known section" was tried and reverted: it let
|
||||
a real, different false positive through (outlier item 21) — individual
|
||||
statin names ("SIMVASTATIN", "LOVASTATIN", ...) are bold+all-caps+short
|
||||
sub-headings *inside* the class-level "CÁC CHẤT ỨC CHẾ HMG-CoA
|
||||
REDUCTASE" monograph, each immediately followed by their own "Liều
|
||||
lượng và cách dùng" sub-section but NOT by "Tên chung quốc tế" (that
|
||||
section belongs only to the parent class monograph) — the loose
|
||||
"any section" check couldn't tell this apart from a real monograph
|
||||
start, but the strict "Tên chung quốc tế specifically" check correctly
|
||||
rejects it, since the specific book-documented template guarantees this
|
||||
exact section is always first for genuine top-level monographs.
|
||||
|
||||
Still correctly rejects the other confirmed false positive (outlier
|
||||
item 19: "HSV"/"CMV" table column headers), which aren't followed by
|
||||
ANY recognized section, let alone this specific one.
|
||||
"""
|
||||
for j in range(title_index + 1, min(title_index + 1 + _ANCHOR_LOOKAHEAD, len(events))):
|
||||
event = events[j]
|
||||
if isinstance(event, Heading) and event.is_monograph_title:
|
||||
return False
|
||||
if isinstance(event, _SectionEvent) and event.section_def.key == _ANCHOR_SECTION_KEY:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _filter_false_positive_titles(events: List[_Event]) -> List[_Event]:
|
||||
"""Pass 2.5: drop title-shaped candidates that aren't followed by any
|
||||
recognized section anchor before the next title candidate.
|
||||
"""
|
||||
return [
|
||||
event for i, event in enumerate(events)
|
||||
if not (isinstance(event, Heading) and event.is_monograph_title)
|
||||
or _has_anchor_ahead(events, i)
|
||||
]
|
||||
|
||||
|
||||
def _region_for(table_index, span: Span):
|
||||
"""The table region a span sits in, if any."""
|
||||
if not table_index:
|
||||
return None
|
||||
for region in table_index.get(span.physical_page, ()):
|
||||
if region.contains(span.x0, span.y0, span.x1, span.y1):
|
||||
return region
|
||||
return None
|
||||
|
||||
|
||||
SPAN_STATE_TEXT = "normalized_text"
|
||||
SPAN_STATE_TABLE = "table"
|
||||
SPAN_STATE_QUARANTINED = "quarantined"
|
||||
SPAN_STATE_BOILERPLATE = "boilerplate_excluded"
|
||||
SPAN_STATE_HEADING = "heading"
|
||||
SPAN_STATE_OUT_OF_SCOPE = "out_of_scope"
|
||||
SPAN_STATE_UNASSIGNED = "unassigned"
|
||||
# Deliberately dropped, not missed: the book's own part-divider titles
|
||||
# ("CÁC CHUYÊN LUẬN THUỐC" etc.) are structure, not content. Reporting them
|
||||
# as `unassigned` would make a clean acceptance target of unassigned == 0
|
||||
# impossible to state honestly.
|
||||
SPAN_STATE_STRUCTURAL = "structural_excluded"
|
||||
|
||||
|
||||
def assemble(spans: List[Span], table_index=None, ledger: Optional[list] = None) -> Iterator[Monograph]:
|
||||
"""Assemble monographs from spans.
|
||||
|
||||
`table_index` maps a physical page to the table regions on it (see
|
||||
`tables.index_by_page`). When supplied, spans falling inside a region are
|
||||
diverted into `Monograph.tables` instead of section prose — measured
|
||||
reason: physical page 109's dosage-form table was otherwise concatenated
|
||||
cell by cell into a section body. Omitting it keeps the previous
|
||||
behaviour, so callers without a region map still work.
|
||||
"""
|
||||
raw_chars = sum(len(s.text) for s in spans)
|
||||
spans = merge_same_line_bold_fragments(spans)
|
||||
events = _filter_false_positive_titles(_coalesce_titles(_classify(spans)))
|
||||
|
||||
# Span-level coverage ledger. Character counts alone cannot balance here
|
||||
# (normalization joins, substitutes and drops characters), so every span
|
||||
# is given a state first and characters are aggregated from that.
|
||||
states: dict = {}
|
||||
if ledger is not None:
|
||||
for s in spans:
|
||||
if not s.text.strip():
|
||||
states[id(s)] = "whitespace_only"
|
||||
elif _is_page_boilerplate(s):
|
||||
states[id(s)] = SPAN_STATE_BOILERPLATE
|
||||
elif is_part_divider(s.text):
|
||||
states[id(s)] = SPAN_STATE_STRUCTURAL
|
||||
elif not in_monograph_range(s):
|
||||
states[id(s)] = SPAN_STATE_OUT_OF_SCOPE
|
||||
elif is_monograph_title_candidate(s):
|
||||
# title spans are merged into a Heading event and lose their
|
||||
# link back to the source span, so they are accounted for here
|
||||
# using the same predicate the classifier uses
|
||||
states[id(s)] = SPAN_STATE_HEADING
|
||||
else:
|
||||
states[id(s)] = SPAN_STATE_UNASSIGNED
|
||||
|
||||
def mark(span: Span, state: str):
|
||||
if ledger is not None:
|
||||
states[id(span)] = state
|
||||
|
||||
seen_ids: set = set()
|
||||
current: Optional[Monograph] = None
|
||||
current_section_key: Optional[str] = None
|
||||
runs: List[tuple] = [] # ordered [(region_or_None, [spans])]
|
||||
inline_prefix: str = ""
|
||||
awaiting_qualifier = False
|
||||
|
||||
def append_span(span: Span, region):
|
||||
"""Keep spans in reading order, starting a new run whenever the
|
||||
prose/table context changes — this is what preserves the real
|
||||
prose -> table -> prose sequence inside one section."""
|
||||
key = region.table_id if region is not None else None
|
||||
if runs and runs[-1][0] == key:
|
||||
runs[-1][1].append(span)
|
||||
else:
|
||||
runs.append((key, [span], region))
|
||||
|
||||
def build_parts() -> List[SectionPart]:
|
||||
parts: List[SectionPart] = []
|
||||
for entry in runs:
|
||||
key, collected = entry[0], entry[1]
|
||||
region = entry[2] if len(entry) > 2 else None
|
||||
if not collected:
|
||||
continue
|
||||
text = substitute_pua(join_spans(collected))
|
||||
if not text.strip():
|
||||
continue
|
||||
pages = [s_.physical_page for s_ in collected]
|
||||
xs0 = min(s_.x0 for s_ in collected); ys0 = min(s_.y0 for s_ in collected)
|
||||
xs1 = max(s_.x1 for s_ in collected); ys1 = max(s_.y1 for s_ in collected)
|
||||
ids = [s_.span_id for s_ in collected]
|
||||
if key is None:
|
||||
parts.append(SectionPart(
|
||||
kind=PART_PROSE, text=text, physical_page=min(pages),
|
||||
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
|
||||
))
|
||||
else:
|
||||
parts.append(SectionPart(
|
||||
kind=PART_TABLE, text=text, physical_page=min(pages),
|
||||
bbox=[xs0, ys0, xs1, ys1], source_span_ids=ids,
|
||||
table_id=key,
|
||||
# deterministic: derived from the first source span, so the
|
||||
# same PDF always produces the same id. A counter suffix
|
||||
# would merely hide a duplicate rather than identify it.
|
||||
table_part_id=f"{key}@{ids[0]}",
|
||||
continuation_group=key,
|
||||
shape=region.shape if region is not None else None,
|
||||
quarantined=(region.shape in QUARANTINE_SHAPES)
|
||||
if region is not None else False,
|
||||
))
|
||||
if inline_prefix:
|
||||
head = SectionPart(
|
||||
kind=PART_PROSE, text=inline_prefix,
|
||||
physical_page=parts[0].physical_page if parts else 0,
|
||||
bbox=parts[0].bbox if parts else [0.0, 0.0, 0.0, 0.0],
|
||||
)
|
||||
parts.insert(0, head)
|
||||
return parts
|
||||
|
||||
def close_current_section():
|
||||
nonlocal runs, inline_prefix
|
||||
if current is not None and current_section_key is None and runs:
|
||||
# spans seen after the title but before any section heading
|
||||
current.preamble.extend(build_parts())
|
||||
if current is not None and current_section_key is not None:
|
||||
existing = current.sections[current_section_key]
|
||||
addition = build_parts()
|
||||
# A section heading can legitimately appear twice inside one
|
||||
# monograph (measured: 33 monographs, 38 occurrences — e.g.
|
||||
# CEFAMANDOL's "Liều lượng và cách dùng" resumes on physical page
|
||||
# 339 after a renal-dosing table). Replacing the SectionSpan here
|
||||
# silently destroyed everything captured before the repeat, so
|
||||
# the parts are concatenated instead. The first heading stays the
|
||||
# section's provenance anchor.
|
||||
combined = list(existing.parts) + addition
|
||||
current.sections[current_section_key] = SectionSpan(
|
||||
key=existing.key, display_name=existing.display_name,
|
||||
heading=existing.heading,
|
||||
# `text` is prose only. Table parts stay in `parts` with their
|
||||
# own provenance and quarantine flag, so anything reading
|
||||
# `.text` (the chunker included) cannot pick up linearised
|
||||
# cells by accident — the ordering is preserved in `parts`.
|
||||
text="\n".join(
|
||||
p_.text for p_ in combined
|
||||
if p_.kind == PART_PROSE and not p_.quarantined and p_.text
|
||||
).strip(),
|
||||
parts=combined,
|
||||
)
|
||||
for part in addition:
|
||||
if part.kind == PART_TABLE:
|
||||
current.tables.append(TableBlock(
|
||||
table_id=part.table_id, shape=part.shape or "",
|
||||
physical_page=part.physical_page, bbox=list(part.bbox),
|
||||
section_key=current_section_key, text=part.text,
|
||||
quarantined=part.quarantined,
|
||||
table_part_id=part.table_part_id,
|
||||
continuation_group=part.continuation_group,
|
||||
source_span_ids=list(part.source_span_ids),
|
||||
))
|
||||
runs = []
|
||||
inline_prefix = ""
|
||||
|
||||
def finalize(monograph: Monograph) -> Monograph:
|
||||
# Duplicate check happens here, not at title-detection time: the
|
||||
# qualifier line (if any) is only known a few events later, so
|
||||
# checking at open-time would false-positive on the legitimate
|
||||
# SALBUTAMOL case (outlier item 18) before the qualifier resolves.
|
||||
if monograph.drug_id in seen_ids:
|
||||
raise DuplicateDrugIdError(
|
||||
f"duplicate drug_id '{monograph.drug_id}' (title '{monograph.drug_name}', "
|
||||
f"physical page {monograph.source_page_range[0]}) — check for a qualifier "
|
||||
f"line (outlier item 18) before assuming this is a real collision"
|
||||
)
|
||||
seen_ids.add(monograph.drug_id)
|
||||
if "ma_atc" in monograph.sections:
|
||||
result = extract_atc_codes(monograph.sections["ma_atc"].text)
|
||||
monograph.atc_codes = result.codes
|
||||
monograph.atc_stated_absent = result.stated_absent
|
||||
return monograph
|
||||
|
||||
for event in events:
|
||||
if isinstance(event, Heading) and event.is_monograph_title:
|
||||
close_current_section()
|
||||
if current is not None:
|
||||
yield finalize(current)
|
||||
current = Monograph(
|
||||
drug_id=_slugify(event.text), drug_name=event.text,
|
||||
source_page_range=[event.physical_page, event.physical_page],
|
||||
)
|
||||
current_section_key = None
|
||||
awaiting_qualifier = True
|
||||
for src in getattr(event, "source_spans", ()) or ():
|
||||
mark(src, SPAN_STATE_HEADING)
|
||||
continue
|
||||
|
||||
if current is None:
|
||||
continue # front matter / general chapters before the first monograph
|
||||
|
||||
if isinstance(event, _SectionEvent):
|
||||
close_current_section()
|
||||
current_section_key = event.section_def.key
|
||||
if event.section_def.key not in current.sections:
|
||||
current.sections[event.section_def.key] = SectionSpan(
|
||||
key=event.section_def.key,
|
||||
display_name=event.section_def.display_name,
|
||||
heading=Heading(
|
||||
text=event.section_def.display_name,
|
||||
physical_page=event.span.physical_page, y0=event.span.y0,
|
||||
is_monograph_title=False, section_key=event.section_def.key,
|
||||
),
|
||||
text="",
|
||||
)
|
||||
if event.inline_value:
|
||||
inline_prefix = event.inline_value
|
||||
awaiting_qualifier = False
|
||||
mark(event.span, SPAN_STATE_HEADING)
|
||||
current.source_page_range[1] = max(current.source_page_range[1], event.span.physical_page)
|
||||
continue
|
||||
|
||||
# _TextEvent
|
||||
span = event.span
|
||||
if awaiting_qualifier and _is_qualifier_line(span):
|
||||
text = span.text.strip()
|
||||
current.drug_id = f"{current.drug_id}_{_slugify(text)}"
|
||||
current.drug_name = f"{current.drug_name} {text}"
|
||||
awaiting_qualifier = False
|
||||
mark(span, SPAN_STATE_HEADING)
|
||||
continue
|
||||
awaiting_qualifier = False
|
||||
|
||||
if not in_monograph_range(span):
|
||||
continue
|
||||
current.source_page_range[1] = max(current.source_page_range[1], span.physical_page)
|
||||
|
||||
region = _region_for(table_index, span)
|
||||
append_span(span, region)
|
||||
if region is not None:
|
||||
mark(span, SPAN_STATE_QUARANTINED
|
||||
if region.shape in QUARANTINE_SHAPES else SPAN_STATE_TABLE)
|
||||
else:
|
||||
mark(span, SPAN_STATE_TEXT)
|
||||
|
||||
if ledger is not None:
|
||||
ledger.append({"raw_chars_before_merge": raw_chars})
|
||||
for s_obj in spans:
|
||||
ledger.append({
|
||||
"state": states[id(s_obj)],
|
||||
"physical_page": s_obj.physical_page,
|
||||
"bbox": [s_obj.x0, s_obj.y0, s_obj.x1, s_obj.y1],
|
||||
"chars": len(s_obj.text),
|
||||
"text": s_obj.text[:60],
|
||||
})
|
||||
|
||||
close_current_section()
|
||||
if current is not None:
|
||||
yield finalize(current)
|
||||
@@ -0,0 +1,135 @@
|
||||
"""ATC-code extraction and normalization.
|
||||
|
||||
Confirmed real text-extraction noise (outlier-catalog item 12c), found while
|
||||
investigating why 22/680 monographs appeared to have zero ATC codes — two
|
||||
distinct causes, both extraction noise rather than missing content:
|
||||
- **Stray internal whitespace** splitting one code into two tokens, e.g.
|
||||
"L01X X02" (should be "L01XX02"), "J04A C01" (should be "J04AC01").
|
||||
- **Digit/letter confusion**: a literal "0" rendered/typeset as "O", e.g.
|
||||
"NO3AX12" (should be "N03AX12").
|
||||
A third, genuinely different outcome: the source text explicitly states
|
||||
"Mã ATC: Chưa có." / "Không có." — a valid "no ATC assigned yet" data state,
|
||||
not an error, and must never be conflated with a parse failure.
|
||||
|
||||
Two more real defects found via a real whole-book `assemble()` run (not
|
||||
assumed, measured against actual monograph text):
|
||||
- **Trailing sentence punctuation**: "Mã ATC: J05AF06." — Abacavir's real
|
||||
field text ends the sentence with a period that isn't part of the code;
|
||||
an earlier version without this fix silently produced zero codes for
|
||||
every single-code monograph ending in ".".
|
||||
- **Per-code parenthetical annotations in multi-ATC monographs**: INSULIN's
|
||||
real field lists all 20 codes each with a species/type note, e.g. "A10AB01
|
||||
(người); A10AB02 (bò); A10AB03 (lợn); ..." — without stripping the
|
||||
trailing "(...)" before length-checking, only 2 of 20 codes survived (the
|
||||
two that happened to have a line-wrap fall between the code and its
|
||||
parenthetical, accidentally isolating the bare code) — a striking example
|
||||
of why this needs whole-corpus validation, not a single clean example.
|
||||
|
||||
A fourth defect, found via the same method (12 vaccine monographs -
|
||||
VẮC XIN SỞI among them - appeared zero-ATC-and-not-stated-absent): the
|
||||
segment split ran *before* parenthetical annotations were stripped, so an
|
||||
annotation containing its own comma broke the split, e.g. "Mã ATC: J07BD01
|
||||
(Measles, live attenuated)." split on "," into "J07BD01 (Measles" and
|
||||
" live attenuated)." — neither a recoverable code shape. INSULIN's
|
||||
Vietnamese annotations ("người", "bò", "lợn") never contain a comma, so
|
||||
this only surfaced with vaccines' English annotations. Fixed by stripping
|
||||
*all* parenthetical groups from the whole field text before splitting,
|
||||
not just a trailing one per already-split segment.
|
||||
|
||||
A fifth defect, same method (15 more monographs, e.g. ALCURONIUM CLORID,
|
||||
AMLODIPIN): some real monographs render the bold section label as "Mã ATC"
|
||||
with no colon, and the colon belongs to the *value* span instead, e.g.
|
||||
bold "Mã ATC" + plain ": M03AA01." (Abacavir's equivalent is bold "Mã ATC:
|
||||
" + plain "J05AF06.", colon on the label side). The heading still matches
|
||||
correctly (`vocab.normalize_heading_text` already strips a trailing
|
||||
colon from either side), but the captured field text keeps the leading
|
||||
": " from the value span, making the stripped candidate 8 characters
|
||||
(":M03AA01") instead of 7 — silently failing the length check. Fixed by
|
||||
taking only the text after the last ":" per segment before normalizing —
|
||||
a strict generalization of the leading-colon strip (see the sixth defect
|
||||
below) that still normalizes a plain "N03AX12" unchanged (no colon to
|
||||
split on).
|
||||
|
||||
A sixth defect, same method (7 monographs with multiple salt/ester forms,
|
||||
e.g. ARGININ, ENALAPRIL, VASOPRESSIN, the INTERFERON and gonadotropin
|
||||
entries): each form is its own "Name: CODE" line rather than a bare code,
|
||||
e.g. "Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01" — the whole
|
||||
segment (including the name) was compared against the 7-character code
|
||||
shape and rejected. Solved by trying the text after the last colon first:
|
||||
"Arginin glutamat: A05BA01" -> "A05BA01".
|
||||
|
||||
A seventh defect, same method (1 monograph, the class-level "CÁC CHẤT ỨC
|
||||
CHẾ HMG-CoA REDUCTASE"): its "Mã ATC" field lists every statin the
|
||||
*opposite* way round, code first — "C10A A01: Simvastatin\nC10A A02:
|
||||
Lovastatin\n..." — so "take the text after the colon" extracts the drug
|
||||
name, not the code. Since a real drug name essentially never happens to
|
||||
match the strict 7-character ATC shape, trying the after-colon part first
|
||||
and falling back to the before-colon part costs nothing for the "Name:
|
||||
CODE" case (defect six) while recovering this reversed "CODE: Name" case
|
||||
too, without needing to special-case either monograph.
|
||||
|
||||
This module returns all three ATC-presence outcomes distinctly (found /
|
||||
recovered-from-noise / stated-absent), never collapsed into one boolean,
|
||||
per the outlier catalog's explicit guidance.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List, Optional
|
||||
|
||||
_ATC_PATTERN = re.compile(r"^[A-Z]\d{2}[A-Z]{2}\d{2}$")
|
||||
_DIGIT_POSITIONS = (1, 2, 5, 6) # 0-indexed positions that must be digits
|
||||
_ABSENT_MARKERS = ("chưa có", "không có")
|
||||
_SEGMENT_SPLIT_RE = re.compile(r"[,;\n]")
|
||||
_PAREN_RE = re.compile(r"\([^()]*\)")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ATCResult:
|
||||
codes: List[str] = field(default_factory=list)
|
||||
stated_absent: bool = False
|
||||
|
||||
|
||||
def is_stated_absent(field_text: str) -> bool:
|
||||
normalized = field_text.strip().lower()
|
||||
return any(marker in normalized for marker in _ABSENT_MARKERS)
|
||||
|
||||
|
||||
def normalize_atc_candidate(raw: str) -> Optional[str]:
|
||||
"""Tries each side of the last ":" (whole string if there is none) as
|
||||
the code, after-colon first since "Name: CODE" is the far more common
|
||||
real shape ("CODE: Name" is confirmed real too, but rare) — returns the
|
||||
first side that normalizes to a valid ATC shape. Normalizing strips
|
||||
trailing sentence punctuation and internal whitespace (fixes the
|
||||
split-token case), then fixes O/0 confusion only at the code's known
|
||||
digit positions (never touches the letter positions, so a genuine "X" in
|
||||
"L01XX02" is left alone). Parenthetical annotations must already be
|
||||
stripped by the caller — see `extract_atc_codes`.
|
||||
"""
|
||||
parts = raw.rsplit(":", 1)
|
||||
candidates = [parts[-1]] if len(parts) == 1 else [parts[1], parts[0]]
|
||||
for part in candidates:
|
||||
stripped = re.sub(r"\s+", "", part.upper()).rstrip(".,;")
|
||||
if len(stripped) != 7:
|
||||
continue
|
||||
chars = list(stripped)
|
||||
for i in _DIGIT_POSITIONS:
|
||||
if chars[i] == "O":
|
||||
chars[i] = "0"
|
||||
candidate = "".join(chars)
|
||||
if _ATC_PATTERN.match(candidate):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def extract_atc_codes(field_text: str) -> ATCResult:
|
||||
if is_stated_absent(field_text):
|
||||
return ATCResult(codes=[], stated_absent=True)
|
||||
without_annotations = _PAREN_RE.sub("", field_text)
|
||||
codes = []
|
||||
for segment in _SEGMENT_SPLIT_RE.split(without_annotations):
|
||||
candidate = normalize_atc_candidate(segment)
|
||||
if candidate:
|
||||
codes.append(candidate)
|
||||
return ATCResult(codes=codes, stated_absent=False)
|
||||
@@ -0,0 +1,89 @@
|
||||
"""Monograph and section boundary detection.
|
||||
|
||||
Validated signal (ADR 0003): monograph titles are bold + all-caps + short
|
||||
line length, scoped to printed pages 99-1496 — font **size** is explicitly
|
||||
NOT part of the rule (a size>=9.8 threshold silently dropped ~15% of real
|
||||
monographs). Section headings are bold spans cross-checked against the
|
||||
known (open/extensible) vocabulary in `vocab.py`, no all-caps requirement
|
||||
(most section headings, e.g. "Chỉ định", are not all-caps).
|
||||
|
||||
Known false positive, explicitly excluded rather than tuned around (outlier
|
||||
item 12d): "CÁC CHUYÊN LUẬN THUỐC" and other part-divider titles sit exactly
|
||||
at the printed-page-99 boundary and are bold + all-caps + short, identical
|
||||
in shape to a real monograph title.
|
||||
|
||||
"All-caps" itself is not 100% reliable either (confirmed real, outlier item
|
||||
21): the class-level monograph "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE" embeds
|
||||
the mixed-case abbreviation "CoA" (Coenzyme A) — a strict `text.isupper()`
|
||||
check silently dropped this entire monograph. `_is_mostly_upper` tolerates
|
||||
a small number of lowercase letters (a strict superset of `isupper()`, so
|
||||
no previously-valid case is excluded) rather than requiring zero.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterator, List
|
||||
|
||||
from ..extract.models import Span
|
||||
from .merge import merge_multiline_headings
|
||||
from .models import Heading
|
||||
from .vocab import is_part_divider, match_section
|
||||
|
||||
MONOGRAPH_PRINTED_PAGE_START = 99
|
||||
MONOGRAPH_PRINTED_PAGE_END = 1496
|
||||
_MIN_TITLE_LEN = 3
|
||||
_MAX_TITLE_LEN = 60
|
||||
_MAX_LOWERCASE_RATIO = 0.10 # HMG-CoA: 1/27 = 3.7% (real title) vs "Mã ATC:": 1/5 = 20% (real
|
||||
# section label, correctly rejected) — a ratio, not an absolute count, is what separates a
|
||||
# long title with one embedded mixed-case abbreviation from a short label with a normal
|
||||
# lowercase diacritic (found via a real regression: an earlier absolute-count version of
|
||||
# this check let "Mã ATC:" through as a false title candidate).
|
||||
|
||||
|
||||
def _is_mostly_upper(text: str) -> bool:
|
||||
letters = [c for c in text if c.isalpha()]
|
||||
if not letters:
|
||||
return False
|
||||
lowercase_ratio = sum(1 for c in letters if c.islower()) / len(letters)
|
||||
return lowercase_ratio <= _MAX_LOWERCASE_RATIO
|
||||
|
||||
|
||||
def in_monograph_range(span: Span) -> bool:
|
||||
return (
|
||||
span.printed_page is not None
|
||||
and MONOGRAPH_PRINTED_PAGE_START <= span.printed_page <= MONOGRAPH_PRINTED_PAGE_END
|
||||
)
|
||||
|
||||
|
||||
def is_monograph_title_candidate(span: Span) -> bool:
|
||||
text = span.text.strip()
|
||||
if not (span.bold and _is_mostly_upper(text)):
|
||||
return False
|
||||
if not (_MIN_TITLE_LEN <= len(text) <= _MAX_TITLE_LEN):
|
||||
return False
|
||||
if not in_monograph_range(span):
|
||||
return False
|
||||
if is_part_divider(text):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def detect_monograph_titles(spans: List[Span]) -> Iterator[Heading]:
|
||||
"""`spans` must be in reading order (as `extract_spans` yields them)."""
|
||||
candidates = [s for s in spans if is_monograph_title_candidate(s)]
|
||||
yield from merge_multiline_headings(candidates)
|
||||
|
||||
|
||||
def detect_section_headings(spans: List[Span]) -> Iterator[Heading]:
|
||||
for span in spans:
|
||||
if not span.bold or not in_monograph_range(span):
|
||||
continue
|
||||
section_def = match_section(span.text)
|
||||
if section_def is None:
|
||||
continue
|
||||
yield Heading(
|
||||
text=section_def.display_name,
|
||||
physical_page=span.physical_page,
|
||||
y0=span.y0,
|
||||
is_monograph_title=False,
|
||||
section_key=section_def.key,
|
||||
)
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Pure I/O boundary for Monograph records — kept separate from detection/
|
||||
assembly logic so those stay testable without disk (Clean Architecture).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
from .models import Heading, Monograph, SectionPart, SectionSpan, TableBlock
|
||||
|
||||
|
||||
def _heading_to_dict(h: Heading) -> dict:
|
||||
return {
|
||||
"text": h.text, "physical_page": h.physical_page, "y0": h.y0,
|
||||
"is_monograph_title": h.is_monograph_title, "section_key": h.section_key,
|
||||
}
|
||||
|
||||
|
||||
def _heading_from_dict(d: dict) -> Heading:
|
||||
return Heading(**d)
|
||||
|
||||
|
||||
def _monograph_to_dict(m: Monograph) -> dict:
|
||||
return {
|
||||
"drug_id": m.drug_id,
|
||||
"drug_name": m.drug_name,
|
||||
"source_page_range": m.source_page_range,
|
||||
"atc_codes": m.atc_codes,
|
||||
"atc_stated_absent": m.atc_stated_absent,
|
||||
"sections": {
|
||||
key: {
|
||||
"key": s.key, "display_name": s.display_name,
|
||||
"heading": _heading_to_dict(s.heading), "text": s.text,
|
||||
"parts": [
|
||||
{
|
||||
"kind": p.kind, "text": p.text,
|
||||
"physical_page": p.physical_page, "bbox": p.bbox,
|
||||
"source_span_ids": p.source_span_ids,
|
||||
"table_id": p.table_id, "table_part_id": p.table_part_id,
|
||||
"continuation_group": p.continuation_group,
|
||||
"shape": p.shape, "quarantined": p.quarantined,
|
||||
}
|
||||
for p in s.parts
|
||||
],
|
||||
}
|
||||
for key, s in m.sections.items()
|
||||
},
|
||||
"preamble": [
|
||||
{
|
||||
"kind": p.kind, "text": p.text, "physical_page": p.physical_page,
|
||||
"bbox": p.bbox, "source_span_ids": p.source_span_ids,
|
||||
"quarantined": p.quarantined,
|
||||
}
|
||||
for p in m.preamble
|
||||
],
|
||||
"tables": [
|
||||
{
|
||||
"table_id": t.table_id, "shape": t.shape,
|
||||
"table_part_id": t.table_part_id,
|
||||
"continuation_group": t.continuation_group,
|
||||
"source_span_ids": t.source_span_ids,
|
||||
"physical_page": t.physical_page, "bbox": t.bbox,
|
||||
"section_key": t.section_key, "text": t.text,
|
||||
"quarantined": t.quarantined,
|
||||
}
|
||||
for t in m.tables
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _monograph_from_dict(d: dict) -> Monograph:
|
||||
sections = {
|
||||
key: SectionSpan(
|
||||
key=s["key"], display_name=s["display_name"],
|
||||
heading=_heading_from_dict(s["heading"]), text=s["text"],
|
||||
parts=[
|
||||
SectionPart(
|
||||
kind=p["kind"], text=p["text"],
|
||||
physical_page=p["physical_page"], bbox=p["bbox"],
|
||||
source_span_ids=p.get("source_span_ids", []),
|
||||
table_id=p.get("table_id"), table_part_id=p.get("table_part_id"),
|
||||
continuation_group=p.get("continuation_group"),
|
||||
shape=p.get("shape"), quarantined=p.get("quarantined", False),
|
||||
)
|
||||
for p in s.get("parts", [])
|
||||
],
|
||||
)
|
||||
for key, s in d["sections"].items()
|
||||
}
|
||||
return Monograph(
|
||||
drug_id=d["drug_id"], drug_name=d["drug_name"],
|
||||
source_page_range=d["source_page_range"], sections=sections,
|
||||
atc_codes=d.get("atc_codes", []), atc_stated_absent=d.get("atc_stated_absent", False),
|
||||
preamble=[
|
||||
SectionPart(
|
||||
kind=p["kind"], text=p["text"], physical_page=p["physical_page"],
|
||||
bbox=p["bbox"], source_span_ids=p.get("source_span_ids", []),
|
||||
quarantined=p.get("quarantined", False),
|
||||
)
|
||||
for p in d.get("preamble", [])
|
||||
],
|
||||
tables=[
|
||||
TableBlock(
|
||||
table_id=t["table_id"], shape=t["shape"],
|
||||
physical_page=t["physical_page"], bbox=t["bbox"],
|
||||
section_key=t.get("section_key"), text=t["text"],
|
||||
quarantined=t.get("quarantined", False),
|
||||
table_part_id=t.get("table_part_id"),
|
||||
continuation_group=t.get("continuation_group"),
|
||||
source_span_ids=t.get("source_span_ids", []),
|
||||
)
|
||||
for t in d.get("tables", [])
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def write_monographs_jsonl(monographs: Iterable[Monograph], path: Path) -> int:
|
||||
count = 0
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
for m in monographs:
|
||||
f.write(json.dumps(_monograph_to_dict(m), ensure_ascii=False) + "\n")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def read_monographs_jsonl(path: Path) -> Iterator[Monograph]:
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
yield _monograph_from_dict(json.loads(line))
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Multi-line monograph-title merging, and same-line bold-run reassembly.
|
||||
|
||||
Two distinct real fragmentation shapes were confirmed, both requiring merge:
|
||||
|
||||
1. **Multi-line wrap** (ADR 0003's original finding, dominant cause of its
|
||||
recall gap and of the GONADOTROPIN false-collision, outlier item 11):
|
||||
long titles wrap across 2+ physical lines, e.g. physical page 1371 has
|
||||
"THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG" (y0=664.46) immediately followed by
|
||||
"GONADOTROPIN" (y0=676.24) — a ~11.8pt line-height step, same page.
|
||||
2. **Same-line font-size split, found by visually inspecting a real page**
|
||||
(physical page 113, rendered to an image and read directly — not
|
||||
inferred from coordinates alone): "ACICLOVIR" is split into two spans,
|
||||
"ACIC" (size 10.0) and "LOVIR" (size 9.5), touching with a ~0.5pt y0
|
||||
difference and near-zero x-gap. An earlier version of this module
|
||||
required exact font-size equality to merge, which correctly handled
|
||||
case 1 (GONADOTROPIN: both fragments size 9.5) but silently missed case
|
||||
2 — the same "font size is not reliable" lesson from ADR 0003 applies
|
||||
*within* a single title's fragments, not just across different
|
||||
monographs. Fixed by dropping the size-equality requirement; the y-gap
|
||||
+ same-page check alone is sufficient (a real next-monograph title is
|
||||
always much farther down the page/on a different page, given a full
|
||||
monograph's worth of section content in between).
|
||||
|
||||
The join character between merged fragments must differ by case: case 1
|
||||
needs a space (distinct words across a real line wrap); case 2 needs no
|
||||
space (mid-word split, "ACIC" + "LOVIR" = "ACICLOVIR", not "ACIC LOVIR").
|
||||
Distinguished by the y0 gap: small (<= `_SAME_LINE_Y_TOLERANCE`) means same
|
||||
visual line -> concatenate directly; larger means a real new line -> join
|
||||
with a space.
|
||||
|
||||
Candidate spans passed in here are already filtered by the caller (bold +
|
||||
all-caps + short + in the monograph page range) — this module only decides
|
||||
which *consecutive* candidates belong to the same title and how to join them.
|
||||
|
||||
A third, unrelated fragmentation shape was confirmed via a whole-book
|
||||
`cli validate` run against the back-of-book index (5 real monographs -
|
||||
GUAIFENESIN, MEPHENESIN, NATRI THIOSULFAT, RAMIPRIL, TENOXICAM - silently
|
||||
dropped): PyMuPDF splits some bold section-heading runs into several spans
|
||||
around diacritic characters even though the text is a single, visually
|
||||
unbroken line in the rendered page (confirmed by rendering physical page
|
||||
759 to an image and reading it directly — "Tên chung quốc tế" looks
|
||||
completely normal to a human reader; the fragmentation exists only in
|
||||
PyMuPDF's span boundaries, not the document). Confirmed page 759's actual
|
||||
spans: "Tên chung qu" (y0=157.614), "ố" (y0=157.33), "c t" (y0=157.614),
|
||||
"ế" (y0=157.33), ": " (y0=157.614) — all within `_SAME_LINE_Y_TOLERANCE`,
|
||||
so `merge_same_line_bold_fragments` (applied to *all* bold spans, not just
|
||||
title candidates, before section-vocabulary matching) reassembles them the
|
||||
same way case 2 above reassembles "ACIC" + "LOVIR".
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
from typing import Iterator, List
|
||||
|
||||
from ..extract.models import Span
|
||||
from .models import Heading
|
||||
|
||||
_MAX_LINE_GAP_PT = 20.0 # comfortably above the confirmed ~11.8pt wrap case
|
||||
_SAME_LINE_Y_TOLERANCE = 3.0 # comfortably above the confirmed ~0.5pt same-line split
|
||||
|
||||
|
||||
def merge_same_line_bold_fragments(spans: List[Span]) -> List[Span]:
|
||||
"""Reassemble consecutive bold spans that PyMuPDF split mid-line (same
|
||||
page, same visual line) back into one span, so downstream section-vocab
|
||||
matching sees the real text instead of a diacritic-boundary fragment.
|
||||
|
||||
Non-bold spans and spans on different lines pass through unchanged.
|
||||
Provenance (page/column/block/line/span_index/y-position/font/size) is
|
||||
kept from the first fragment; only `text` and `x1` are updated, so the
|
||||
merged span still traces back to its exact source region.
|
||||
"""
|
||||
merged: List[Span] = []
|
||||
buffer: List[Span] = []
|
||||
|
||||
def flush():
|
||||
if not buffer:
|
||||
return
|
||||
if len(buffer) == 1:
|
||||
merged.append(buffer[0])
|
||||
else:
|
||||
merged.append(dataclasses.replace(
|
||||
buffer[0], text="".join(s.text for s in buffer), x1=buffer[-1].x1,
|
||||
))
|
||||
|
||||
for span in spans:
|
||||
same_line_bold_run = (
|
||||
buffer and span.bold and buffer[-1].bold
|
||||
and span.physical_page == buffer[-1].physical_page
|
||||
and abs(span.y0 - buffer[-1].y0) <= _SAME_LINE_Y_TOLERANCE
|
||||
)
|
||||
if same_line_bold_run:
|
||||
buffer.append(span)
|
||||
else:
|
||||
flush()
|
||||
buffer = [span]
|
||||
flush()
|
||||
return merged
|
||||
|
||||
|
||||
def _same_title_run(prev: Span, curr: Span) -> bool:
|
||||
gap = curr.y0 - prev.y0
|
||||
return curr.physical_page == prev.physical_page and 0 <= gap <= _MAX_LINE_GAP_PT
|
||||
|
||||
|
||||
def merge_multiline_headings(candidates: List[Span]) -> Iterator[Heading]:
|
||||
"""`candidates` must already be in reading order (as extract_spans
|
||||
yields them) and pre-filtered to heading candidates only.
|
||||
"""
|
||||
buffer: List[Span] = []
|
||||
for span in candidates:
|
||||
if buffer and _same_title_run(buffer[-1], span):
|
||||
buffer.append(span)
|
||||
else:
|
||||
if buffer:
|
||||
yield _flush(buffer)
|
||||
buffer = [span]
|
||||
if buffer:
|
||||
yield _flush(buffer)
|
||||
|
||||
|
||||
def _flush(buffer: List[Span]) -> Heading:
|
||||
parts = [buffer[0].text.strip()]
|
||||
for prev, curr in zip(buffer, buffer[1:], strict=False):
|
||||
same_line = abs(curr.y0 - prev.y0) <= _SAME_LINE_Y_TOLERANCE
|
||||
parts.append("" if same_line else " ")
|
||||
parts.append(curr.text.strip())
|
||||
first = buffer[0]
|
||||
return Heading(
|
||||
text="".join(parts),
|
||||
physical_page=first.physical_page,
|
||||
y0=first.y0,
|
||||
is_monograph_title=True,
|
||||
)
|
||||
@@ -0,0 +1,95 @@
|
||||
"""Data model for segmented output, matching docs/architecture.md's contract:
|
||||
{drug_id, drug_name, source_page_range, sections: {...}}
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Heading:
|
||||
text: str
|
||||
physical_page: int
|
||||
y0: float
|
||||
is_monograph_title: bool
|
||||
section_key: Optional[str] = None
|
||||
|
||||
|
||||
PART_PROSE = "prose"
|
||||
PART_TABLE = "table"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionPart:
|
||||
"""One contiguous run of a section, in reading order.
|
||||
|
||||
A section is not uniformly prose: a dosing section routinely reads
|
||||
prose -> table -> prose. Flattening that to a single string loses both the
|
||||
ordering and the ability to say which part a sentence came from, so the
|
||||
parts are kept in sequence with their own provenance.
|
||||
"""
|
||||
kind: str
|
||||
text: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
source_span_ids: List[str] = field(default_factory=list)
|
||||
table_id: Optional[str] = None
|
||||
table_part_id: Optional[str] = None
|
||||
continuation_group: Optional[str] = None
|
||||
shape: Optional[str] = None
|
||||
quarantined: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionSpan:
|
||||
key: str
|
||||
display_name: str
|
||||
heading: Heading
|
||||
text: str
|
||||
parts: List[SectionPart] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def prose_text(self) -> str:
|
||||
"""Only the parts safe to read as prose — excludes quarantined ones."""
|
||||
return "\n".join(
|
||||
p.text for p in self.parts
|
||||
if p.kind == PART_PROSE and not p.quarantined and p.text
|
||||
).strip()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TableBlock:
|
||||
"""Text lifted out of a table region, kept beside the prose instead of
|
||||
inside it.
|
||||
|
||||
`quarantined` marks content whose flattened text is actively misleading
|
||||
(a 2D lookup grid means nothing without its row and column headers) —
|
||||
such a block must not be embedded or cited as if it were prose.
|
||||
"""
|
||||
table_id: str
|
||||
shape: str
|
||||
physical_page: int
|
||||
bbox: List[float]
|
||||
section_key: Optional[str]
|
||||
text: str
|
||||
quarantined: bool = False
|
||||
table_part_id: Optional[str] = None
|
||||
continuation_group: Optional[str] = None
|
||||
source_span_ids: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Monograph:
|
||||
drug_id: str
|
||||
drug_name: str
|
||||
source_page_range: List[int]
|
||||
sections: Dict[str, SectionSpan] = field(default_factory=dict)
|
||||
atc_codes: List[str] = field(default_factory=list)
|
||||
atc_stated_absent: bool = False
|
||||
tables: List[TableBlock] = field(default_factory=list)
|
||||
# Text between the monograph title and its first section heading. Real and
|
||||
# clinically important — e.g. ARTEMETHER (physical page 210) opens with the
|
||||
# regulatory notice that single-agent artemisinin products were withdrawn
|
||||
# to limit resistance. It belongs to no section, so it was being dropped.
|
||||
preamble: List[SectionPart] = field(default_factory=list)
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Dosing-unit token validation (mg/mcg/mmol/g/ml).
|
||||
|
||||
Unlike `atc.py`'s whitespace-split defect (confirmed with real examples,
|
||||
outlier-catalog item 12c), a targeted regex scan of the full monograph page
|
||||
range (99-1496 printed) for the analogous unit-token pattern (a unit like
|
||||
"mg" split into "m g" by a stray internal space) found **zero occurrences**
|
||||
— this is NOT a confirmed defect in this corpus. This module exists as a
|
||||
defensive check by analogy, per the project's explicit dosing-safety
|
||||
requirement: a silent mg/mcg confusion is a 1000x dosing error, and the
|
||||
book's own "Người lớn"/"Trẻ em" dosing-population split appears on the
|
||||
majority of monograph pages (outlier-catalog item 17), so the cost of an
|
||||
undetected unit-token corruption is high enough to check for even without a
|
||||
confirmed prior occurrence — but callers must not describe what this module
|
||||
guards against as "a confirmed real defect," only as a validated absence
|
||||
plus a standing defensive gate.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
_KNOWN_UNITS = ("mg", "mcg", "mmol", "microgam", "g", "ml", "iu", "đvqt")
|
||||
_UNIT_PATTERN = re.compile(
|
||||
"^(" + "|".join(re.escape(u) for u in _KNOWN_UNITS) + ")$", re.IGNORECASE
|
||||
)
|
||||
|
||||
|
||||
def normalize_unit_token(raw: str) -> Optional[str]:
|
||||
"""Strips internal whitespace (defends against a stray-space split, the
|
||||
same failure class as the confirmed ATC whitespace-split defect) and
|
||||
validates against the known dosing-unit vocabulary. Returns the
|
||||
lowercase canonical unit string, or None if unrecognized.
|
||||
"""
|
||||
stripped = re.sub(r"\s+", "", raw).lower()
|
||||
return stripped if _UNIT_PATTERN.match(stripped) else None
|
||||
|
||||
|
||||
def validate_unit_tokens(tokens: list) -> "list[str]":
|
||||
"""Returns the subset of `tokens` that fail normalization — callers use
|
||||
this to flag a dosing section for manual review, not to silently drop
|
||||
or auto-correct (unlike ATC codes, there is no confirmed-safe recovery
|
||||
rule here since no real corruption pattern has been observed yet).
|
||||
"""
|
||||
return [t for t in tokens if normalize_unit_token(t) is None]
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Section-name taxonomy for drug monographs.
|
||||
|
||||
Canonical list transcribed directly from the book's own documented template
|
||||
(physical page 38, printed page 39, "HƯỚNG DẪN SỬ DỤNG DƯỢC THƯ QUỐC GIA
|
||||
VIỆT NAM") and cross-checked against real bold headings in the Abacavir/
|
||||
Acarbose monographs (physical pages 100-102). The book documents 19 fields
|
||||
per monograph, of which #1 ("Tên chuyên luận thuốc") is the monograph title
|
||||
itself (handled by `detector.detect_monograph_titles`, not a section) —
|
||||
leaving 18 documented sections. `ten_thuong_mai` ("Tên thương mại") is a
|
||||
19th, real, but *undocumented* field confirmed present in real monographs
|
||||
(outlier-catalog item 12) — open/closed taxonomy: add new entries here as
|
||||
they're found, never change the matching logic in detector.py.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SectionDef:
|
||||
key: str
|
||||
display_name: str
|
||||
# Real spelling variants observed in the book itself. The source is not
|
||||
# typographically consistent: it prints "qui chế" 469 times against the
|
||||
# documented "quy chế", and carries assorted typos ("Mã ACT", "sử trí",
|
||||
# "Chống chỉ đinh"). Whitespace and look-alike-character differences are
|
||||
# NOT listed here — `_lookup_key` folds those away for every entry at
|
||||
# once, so this stays a list of genuinely different wordings.
|
||||
aliases: Tuple[str, ...] = ()
|
||||
|
||||
@property
|
||||
def labels(self) -> Tuple[str, ...]:
|
||||
return (self.display_name,) + self.aliases
|
||||
|
||||
|
||||
SECTION_DEFS = [
|
||||
SectionDef("ten_chung_quoc_te", "Tên chung quốc tế", ("Ten chung quốc tế",)),
|
||||
SectionDef("ma_atc", "Mã ATC", ("Mã ACT",)),
|
||||
SectionDef("loai_thuoc", "Loại thuốc", ("Loại thuôc", "Lọai thuốc", "Phân loại thuốc")),
|
||||
SectionDef("dang_thuoc_va_ham_luong", "Dạng thuốc và hàm lượng",
|
||||
("Dạng dùng và hàm lượng",)),
|
||||
SectionDef("duoc_ly_va_co_che_tac_dung", "Dược lý và cơ chế tác dụng",
|
||||
("Dược lí và cơ chế tác dụng", "Dược lý học và cơ chế tác dụng")),
|
||||
SectionDef("chi_dinh", "Chỉ định"),
|
||||
SectionDef("chong_chi_dinh", "Chống chỉ định", ("Chống chỉ đinh",)),
|
||||
SectionDef("than_trong", "Thận trọng"),
|
||||
SectionDef("thoi_ky_mang_thai", "Thời kỳ mang thai", ("Thời kì mang thai",)),
|
||||
SectionDef("thoi_ky_cho_con_bu", "Thời kỳ cho con bú", ("Thời kì cho con bú",)),
|
||||
SectionDef("tac_dung_khong_mong_muon", "Tác dụng không mong muốn (ADR)",
|
||||
("Tác dụng không mong muốn", "Tác dụng không mong muốn ADR")),
|
||||
SectionDef("huong_dan_xu_tri_adr", "Hướng dẫn cách xử trí ADR",
|
||||
("Hướng dẫn xử trí ADR", "Hướng dẫn cách sử trí ADR",
|
||||
"Hướng dẫn cách xử trí các ADR")),
|
||||
SectionDef("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
("Liều lượng cách dùng", "Liều lượng, cách dùng",
|
||||
"Liều dùng và cách dùng", "Liều lượng và cách sử dụng")),
|
||||
SectionDef("tuong_tac_thuoc", "Tương tác thuốc"),
|
||||
SectionDef("do_on_dinh_va_bao_quan", "Độ ổn định và bảo quản"),
|
||||
SectionDef("tuong_ky", "Tương kỵ"),
|
||||
SectionDef("qua_lieu_va_xu_tri", "Quá liều và xử trí",
|
||||
("Quá liều và cách xử trí", "Quá liều và xử lý",
|
||||
"Quá liều cấp tính và xử trí")),
|
||||
SectionDef("thong_tin_quy_che", "Thông tin quy chế",
|
||||
("Thông tin qui chế", "Thông tin về qui chế", "Thông tin và quy chế")),
|
||||
SectionDef("ten_thuong_mai", "Tên thương mại"),
|
||||
]
|
||||
|
||||
# Near-miss strings deliberately NOT treated as section headings, recorded so
|
||||
# a later reader does not "helpfully" add them: "Thể trọng" is body weight,
|
||||
# not "Thận trọng" (caution); "Tác dụng không mong muốn của opioid" is a
|
||||
# drug-specific sub-heading inside a section, not the section itself.
|
||||
REJECTED_NEAR_MISSES = frozenset({"Thể trọng", "Tác dụng không mong muốn của opioid"})
|
||||
|
||||
# Part/section-divider titles (from the book's own table of contents) that
|
||||
# are bold + all-caps + short, exactly like a monograph title, but are NOT
|
||||
# drug monographs — confirmed false positive, outlier-catalog item 12d.
|
||||
PART_DIVIDER_TITLES = {
|
||||
"CÁC CHUYÊN LUẬN CHUNG",
|
||||
"CÁC CHUYÊN LUẬN THUỐC",
|
||||
"CÁC PHỤ LỤC",
|
||||
}
|
||||
|
||||
_TRAILING_PUNCT_RE = re.compile(r"[:.\s]+$")
|
||||
_WHITESPACE_RE = re.compile(r"\s+")
|
||||
_ALL_WHITESPACE_RE = re.compile(r"\s")
|
||||
|
||||
# Look-alike characters the typesetting mixes with their correct forms:
|
||||
# U+00D0 LATIN CAPITAL LETTER ETH is used where U+0110 LATIN CAPITAL LETTER D
|
||||
# WITH STROKE belongs ("Ðộ ổn định" vs "Độ ổn định"), and NFC does not unify
|
||||
# them because they are genuinely distinct codepoints that merely look alike.
|
||||
_CONFUSABLES = str.maketrans({"Ð": "Đ", "ð": "đ"})
|
||||
|
||||
|
||||
def normalize_heading_text(text: str) -> str:
|
||||
"""Strip trailing colon/period/whitespace and collapse internal runs so
|
||||
"Tên chung quốc tế:" and "Tên chung quốc tế" (both observed verbatim in
|
||||
real monographs) render the same. Preserves single spaces — this is the
|
||||
display form, not the lookup form.
|
||||
"""
|
||||
normalized = _TRAILING_PUNCT_RE.sub("", text.strip())
|
||||
return _WHITESPACE_RE.sub(" ", normalized)
|
||||
|
||||
|
||||
def _lookup_key(text: str) -> str:
|
||||
"""Fold away the differences that are typesetting noise, not wording.
|
||||
|
||||
The source splits and joins headings inconsistently — "Chỉđịnh",
|
||||
"H ướng dẫn cách xử trí ADR", "Tác dụng khôngmong muốn (ADR)" and
|
||||
"Độổn định và bảo quản" all appear — so whitespace is removed entirely
|
||||
rather than enumerated as aliases. Case and look-alike characters are
|
||||
folded for the same reason.
|
||||
"""
|
||||
folded = unicodedata.normalize("NFC", normalize_heading_text(text))
|
||||
folded = folded.translate(_CONFUSABLES)
|
||||
return _ALL_WHITESPACE_RE.sub("", folded).lower()
|
||||
|
||||
|
||||
_LOOKUP: Dict[str, SectionDef] = {
|
||||
_lookup_key(label): d for d in SECTION_DEFS for label in d.labels
|
||||
}
|
||||
|
||||
|
||||
def match_section(text: str) -> Optional[SectionDef]:
|
||||
return _LOOKUP.get(_lookup_key(text))
|
||||
|
||||
|
||||
# Sorted longest-label-first so a prefix check never matches a shorter
|
||||
# label that happens to also be a prefix of a longer one (none currently
|
||||
# collide, but this is a cheap, permanent safety property to keep).
|
||||
_PREFIX_CANDIDATES: list = sorted(
|
||||
((label, d) for d in SECTION_DEFS for label in d.labels),
|
||||
key=lambda pair: -len(pair[0]),
|
||||
)
|
||||
|
||||
|
||||
def match_section_with_inline_value(text: str) -> Optional[Tuple[SectionDef, str]]:
|
||||
"""Handles a real, confirmed structural variant (outlier item 20):
|
||||
some monographs render a section heading and its value as ONE
|
||||
non-bold, non-separated span, e.g. AMITRIPTYLIN's "Mã ATC: N06AA09."
|
||||
(Abacavir's equivalent is bold "Mã ATC: " + separate plain "J05AF06.").
|
||||
Returns (matched section, remaining value text) or None.
|
||||
"""
|
||||
stripped = text.strip()
|
||||
for label, section_def in _PREFIX_CANDIDATES:
|
||||
if stripped[: len(label)].lower() != label.lower():
|
||||
continue
|
||||
remainder = stripped[len(label):].lstrip()
|
||||
if remainder.startswith(":"):
|
||||
return section_def, remainder[1:].strip()
|
||||
return None
|
||||
|
||||
|
||||
def is_part_divider(text: str) -> bool:
|
||||
return normalize_heading_text(text).upper() in PART_DIVIDER_TITLES
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Table stage: detect tabular regions so their text stops leaking into prose.
|
||||
|
||||
Scope note: this stage locates and classifies table *regions*. Reconstructing
|
||||
correct rows and columns is deliberately not attempted here — see ADR 0003
|
||||
and outlier-catalog items 5-7 for why that is a separate, harder problem.
|
||||
"""
|
||||
from .classify import (
|
||||
QUARANTINE_SHAPES,
|
||||
SHAPE_CROSS_PAGE,
|
||||
SHAPE_FORMULA_2D,
|
||||
SHAPE_GRID_2D,
|
||||
SHAPE_MULTI_HEADER,
|
||||
SHAPE_SINGLE_COLUMN_BOXED,
|
||||
SHAPE_NOT_TABLE_FULL_PAGE,
|
||||
SHAPE_SIMPLE,
|
||||
classify_shape,
|
||||
)
|
||||
from .detect import detect_table_regions
|
||||
from .io import index_by_page, read_regions_json, write_regions_json
|
||||
from .models import TableRegion
|
||||
|
||||
__all__ = [
|
||||
"QUARANTINE_SHAPES",
|
||||
"SHAPE_CROSS_PAGE",
|
||||
"SHAPE_FORMULA_2D",
|
||||
"SHAPE_GRID_2D",
|
||||
"SHAPE_MULTI_HEADER",
|
||||
"SHAPE_SINGLE_COLUMN_BOXED",
|
||||
"SHAPE_NOT_TABLE_FULL_PAGE",
|
||||
"SHAPE_SIMPLE",
|
||||
"TableRegion",
|
||||
"classify_shape",
|
||||
"detect_table_regions",
|
||||
"index_by_page",
|
||||
"read_regions_json",
|
||||
"write_regions_json",
|
||||
]
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Shape classification for detected table regions.
|
||||
|
||||
Whole-corpus measurement: of 200 regions `pdfplumber.find_tables()` reports,
|
||||
22 are not tables at all (17 cover a whole page — e.g. the copyright page —
|
||||
and 5 are single-column text blocks such as the epilepsy classification
|
||||
list). Routing every region through one generic reconstructor would treat
|
||||
those 22 as tables, so shape is decided first and handling follows from it.
|
||||
|
||||
Open/closed: adding a shape means adding a rule here, not editing callers.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
PAGE_WIDTH, PAGE_HEIGHT = 595.3, 841.9
|
||||
FULL_PAGE_AREA_RATIO = 0.75
|
||||
|
||||
SHAPE_SIMPLE = "simple_table"
|
||||
SHAPE_MULTI_HEADER = "multi_level_or_merged_header"
|
||||
SHAPE_CROSS_PAGE = "cross_page_continuation"
|
||||
SHAPE_GRID_2D = "grid_2d_numeric"
|
||||
SHAPE_NOT_TABLE_FULL_PAGE = "not_a_table_full_page"
|
||||
SHAPE_SINGLE_COLUMN_BOXED = "single_column_boxed_list"
|
||||
|
||||
# Not produced by table detection — a stacked fraction is not a table — but it
|
||||
# is the same kind of object as far as assembly is concerned: a rectangle whose
|
||||
# spans must be lifted out of prose rather than run together. Measured on
|
||||
# NETILMICIN (physical page 1042) and AMPICILIN VÀ SULBACTAM (202): linearised,
|
||||
# the numerator lands before the '=' and the division reads as multiplication.
|
||||
SHAPE_FORMULA_2D = "formula_2d"
|
||||
|
||||
# Shapes whose flattened text must not be embedded or cited as prose until a
|
||||
# real row/column reconstruction exists. Any multi-column table loses its
|
||||
# cell semantics when linearised — a 2D lookup grid most severely (its values
|
||||
# are meaningless without both headers, outlier-catalog item 7), but a plain
|
||||
# dosing table is no safer to quote once its columns are run together.
|
||||
# `single_column_boxed_list` is excluded deliberately: one column linearises
|
||||
# correctly, so it reads as ordinary text (physical page 55's "Bảng 2").
|
||||
QUARANTINE_SHAPES = frozenset({
|
||||
SHAPE_GRID_2D,
|
||||
SHAPE_SIMPLE,
|
||||
SHAPE_MULTI_HEADER,
|
||||
SHAPE_CROSS_PAGE,
|
||||
SHAPE_FORMULA_2D,
|
||||
})
|
||||
|
||||
|
||||
def _area_ratio(bbox) -> float:
|
||||
x0, y0, x1, y1 = bbox
|
||||
return abs((x1 - x0) * (y1 - y0)) / (PAGE_WIDTH * PAGE_HEIGHT)
|
||||
|
||||
|
||||
def classify_shape(
|
||||
bbox,
|
||||
n_rows: int,
|
||||
n_cols: int,
|
||||
first_row: List[str],
|
||||
starts_near_top: bool,
|
||||
all_cells_numeric: bool,
|
||||
) -> str:
|
||||
if _area_ratio(bbox) >= FULL_PAGE_AREA_RATIO:
|
||||
return SHAPE_NOT_TABLE_FULL_PAGE
|
||||
# Only a single *column* is degenerate. A single ROW with several columns
|
||||
# is the opposite of degenerate — it is the orphaned continuation row of
|
||||
# a table broken across a page (outlier-catalog item 5), the case where
|
||||
# losing the content is most damaging because a row without its header
|
||||
# cannot be interpreted. Verified visually: physical pages 62 and 72 are
|
||||
# exactly this (1x3, with cell rules visible), and an earlier version of
|
||||
# this rule discarded both as "not a table".
|
||||
# One column inside a ruled box. Structurally not a row/column table, but
|
||||
# the book may still number it as one — physical page 55 is captioned
|
||||
# "Bảng 2: Phân loại quốc tế các cơn động kinh (1989)" and is a nested
|
||||
# numbered list drawn inside a frame. Named for what it is rather than
|
||||
# "not a table": single-column content linearises correctly and must stay
|
||||
# in the text, unlike a real 2D table.
|
||||
if n_cols <= 1:
|
||||
return SHAPE_SINGLE_COLUMN_BOXED
|
||||
if n_rows <= 1:
|
||||
return SHAPE_CROSS_PAGE
|
||||
if all_cells_numeric and n_cols >= 4:
|
||||
return SHAPE_GRID_2D
|
||||
|
||||
cells = [(c or "").strip() for c in first_row]
|
||||
textual = sum(
|
||||
1 for c in cells
|
||||
if c and not c.replace(",", "").replace(".", "").replace("-", "").isdigit()
|
||||
)
|
||||
if starts_near_top and textual <= 1:
|
||||
return SHAPE_CROSS_PAGE
|
||||
if cells and any(not c for c in cells) and textual >= 1:
|
||||
return SHAPE_MULTI_HEADER
|
||||
return SHAPE_SIMPLE
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Table region detection.
|
||||
|
||||
`pdfplumber` is used here and nowhere else in the pipeline: ADR 0003 records
|
||||
that its general text extraction scrambles reading order on this document,
|
||||
so it is kept strictly to table geometry, where it is the only tool that
|
||||
works. PyMuPDF remains the sole text extractor.
|
||||
|
||||
Detection is slow (≈17 minutes over the 1668-page book), so the result is
|
||||
written once to a region map and reused — see `io.py`. The detection itself
|
||||
lives here, in the pipeline, rather than in a throwaway script.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Iterator, List
|
||||
|
||||
import pdfplumber
|
||||
|
||||
from .classify import classify_shape
|
||||
from .models import TableRegion
|
||||
|
||||
TOP_BAND_Y = 120.0
|
||||
|
||||
|
||||
def _all_numeric(data) -> bool:
|
||||
values = [(c or "").strip() for row in data for c in row]
|
||||
values = [v for v in values if v]
|
||||
if not values:
|
||||
return False
|
||||
return all(
|
||||
v.replace(",", "").replace(".", "").replace("-", "").isdigit()
|
||||
for v in values
|
||||
)
|
||||
|
||||
|
||||
def detect_table_regions(pdf_path: Path) -> Iterator[TableRegion]:
|
||||
with pdfplumber.open(pdf_path) as pdf:
|
||||
for page_number, page in enumerate(pdf.pages):
|
||||
try:
|
||||
found = page.find_tables()
|
||||
except Exception:
|
||||
continue
|
||||
for index, table in enumerate(found):
|
||||
data = table.extract() or []
|
||||
first_row: List[str] = [
|
||||
(c or "").strip() for c in (data[0] if data else [])
|
||||
]
|
||||
n_rows = len(data)
|
||||
n_cols = max((len(r) for r in data), default=0)
|
||||
bbox = tuple(round(v, 1) for v in table.bbox)
|
||||
yield TableRegion(
|
||||
table_id=f"p{page_number}_t{index}",
|
||||
physical_page=page_number,
|
||||
bbox=bbox,
|
||||
n_rows=n_rows,
|
||||
n_cols=n_cols,
|
||||
shape=classify_shape(
|
||||
bbox=bbox,
|
||||
n_rows=n_rows,
|
||||
n_cols=n_cols,
|
||||
first_row=first_row,
|
||||
starts_near_top=bbox[1] < TOP_BAND_Y,
|
||||
all_cells_numeric=_all_numeric(data),
|
||||
),
|
||||
first_row=first_row[:8],
|
||||
)
|
||||
# pdfplumber caches every parsed object per page; without this the
|
||||
# 1668-page book grows the process past 6 GB and the run dies on
|
||||
# a paging-file error rather than finishing.
|
||||
page.flush_cache()
|
||||
page.get_textmap.cache_clear()
|
||||
@@ -0,0 +1,42 @@
|
||||
"""Filesystem boundary for the table stage."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List
|
||||
|
||||
from .models import TableRegion
|
||||
|
||||
|
||||
def write_regions_json(regions: Iterable[TableRegion], path: Path) -> int:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
rows = [asdict(r) for r in regions]
|
||||
path.write_text(json.dumps(rows, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
return len(rows)
|
||||
|
||||
|
||||
def read_regions_json(path: Path) -> List[TableRegion]:
|
||||
rows = json.loads(path.read_text(encoding="utf-8"))
|
||||
return [
|
||||
TableRegion(
|
||||
table_id=r["table_id"],
|
||||
physical_page=r["physical_page"],
|
||||
bbox=tuple(r["bbox"]),
|
||||
n_rows=r["n_rows"],
|
||||
n_cols=r["n_cols"],
|
||||
shape=r["shape"],
|
||||
first_row=r.get("first_row", []),
|
||||
)
|
||||
for r in rows
|
||||
]
|
||||
|
||||
|
||||
def index_by_page(regions: Iterable[TableRegion]) -> Dict[int, List[TableRegion]]:
|
||||
"""Group real table regions by page for O(1) lookup during assembly."""
|
||||
index: Dict[int, List[TableRegion]] = {}
|
||||
for region in regions:
|
||||
if not region.is_real_table:
|
||||
continue
|
||||
index.setdefault(region.physical_page, []).append(region)
|
||||
return index
|
||||
@@ -0,0 +1,39 @@
|
||||
"""Table region model.
|
||||
|
||||
A region is a rectangle on one page that holds tabular content. It is
|
||||
deliberately separate from the table's *contents*: the pipeline's first
|
||||
obligation is to stop tabular text leaking into prose (measured: page 109's
|
||||
dosage-form table was being concatenated cell-by-cell into a section body),
|
||||
which needs only the geometry. Reconstructing rows and columns correctly is
|
||||
a later, harder step.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import List, Tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TableRegion:
|
||||
table_id: str
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
n_rows: int
|
||||
n_cols: int
|
||||
shape: str
|
||||
first_row: List[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def is_real_table(self) -> bool:
|
||||
return not self.shape.startswith("not_a_table")
|
||||
|
||||
def contains(self, x0: float, y0: float, x1: float, y1: float) -> bool:
|
||||
"""True when a span's box lies (mostly) inside this region.
|
||||
|
||||
Uses the span's centre rather than full containment: PyMuPDF span
|
||||
boxes and pdfplumber table boxes come from different engines and
|
||||
disagree by a point or two at the edges.
|
||||
"""
|
||||
cx, cy = (x0 + x1) / 2, (y0 + y1) / 2
|
||||
left, top, right, bottom = self.bbox
|
||||
return left <= cx <= right and top <= cy <= bottom
|
||||
@@ -0,0 +1,51 @@
|
||||
from .back_index import GroundTruthEntry, parse_back_index
|
||||
from .metrics import RecallPrecisionResult, compute_recall_precision
|
||||
from .readiness import (
|
||||
Gate,
|
||||
corpus_size,
|
||||
evaluate,
|
||||
evaluate_chunks,
|
||||
read_chunks,
|
||||
read_monographs,
|
||||
)
|
||||
from .residual_ink import (
|
||||
ANTIALIAS_SPECK,
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
HEADER_BAND_FRAGMENT,
|
||||
HEADER_RULE,
|
||||
RULE_FRAGMENT,
|
||||
TABLE_FRAME,
|
||||
TEXT_AS_VECTOR_OUTLINE,
|
||||
UNCLASSIFIED,
|
||||
PageContext,
|
||||
ResidualRegion,
|
||||
classify,
|
||||
scan_document,
|
||||
scan_page,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"GroundTruthEntry",
|
||||
"parse_back_index",
|
||||
"RecallPrecisionResult",
|
||||
"compute_recall_precision",
|
||||
"Gate",
|
||||
"evaluate",
|
||||
"evaluate_chunks",
|
||||
"read_chunks",
|
||||
"corpus_size",
|
||||
"read_monographs",
|
||||
"PageContext",
|
||||
"ResidualRegion",
|
||||
"classify",
|
||||
"scan_page",
|
||||
"scan_document",
|
||||
"HEADER_RULE",
|
||||
"TABLE_FRAME",
|
||||
"TEXT_AS_VECTOR_OUTLINE",
|
||||
"FRACTION_BAR_CANDIDATE",
|
||||
"HEADER_BAND_FRAGMENT",
|
||||
"RULE_FRAGMENT",
|
||||
"ANTIALIAS_SPECK",
|
||||
"UNCLASSIFIED",
|
||||
]
|
||||
@@ -0,0 +1,52 @@
|
||||
"""Parses the book's own "Mục lục tra cứu" (back-of-book index) into
|
||||
page-verified ground truth — per ADR 0003, this is the correct validation
|
||||
source (exact page numbers per generic name), not the front-matter drug list
|
||||
(no page numbers).
|
||||
|
||||
Real format confirmed by reading physical pages 1530+ directly:
|
||||
- Genuine generic-name entries: "Abacavir, 101" (name, comma, printed page).
|
||||
- Brand-name cross-references: "Ziagen - Abacavir, 101" / "ABAB -
|
||||
Paracetamol, 1118" (brand " - " generic, page) — skipped for ground
|
||||
truth, per ADR 0003.
|
||||
- Section-letter headers ("A", "B", ...) and running header/footer
|
||||
boilerplate lines don't match the entry pattern and are naturally
|
||||
ignored, not specially cased.
|
||||
|
||||
Known limitation, inherited from the already-validated ADR 0003 approach
|
||||
(not newly introduced here): a handful of genuine compound-name entries in
|
||||
the book use " - " *within* the generic name itself (e.g. "Carbidopa -
|
||||
levodopa"), which this parser's cross-reference exclusion will also skip —
|
||||
the same trade-off the original 91.7%-recall validation already made
|
||||
successfully, not re-litigated here.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import List
|
||||
|
||||
import fitz
|
||||
|
||||
BACK_INDEX_START_PHYSICAL = 1530 # printed 1531 — first page of real entries ("A" section)
|
||||
|
||||
_ENTRY_RE = re.compile(r"^(.+?),\s*(\d+)\s*$")
|
||||
_CROSS_REF_MARKER = " - "
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GroundTruthEntry:
|
||||
name: str
|
||||
printed_page: int
|
||||
|
||||
|
||||
def parse_back_index(doc: fitz.Document, start_physical_page: int = BACK_INDEX_START_PHYSICAL) -> List[GroundTruthEntry]:
|
||||
entries: List[GroundTruthEntry] = []
|
||||
for pno in range(start_physical_page, doc.page_count):
|
||||
for line in doc[pno].get_text().split("\n"):
|
||||
line = line.strip()
|
||||
if not line or _CROSS_REF_MARKER in line:
|
||||
continue
|
||||
match = _ENTRY_RE.match(line)
|
||||
if match:
|
||||
entries.append(GroundTruthEntry(name=match.group(1).strip(), printed_page=int(match.group(2))))
|
||||
return entries
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Monograph-boundary recall/precision against the back-of-book index.
|
||||
|
||||
Per ADR 0003, only recall was ever measured before (91.7%, 665/725) — this
|
||||
module adds precision (never measured previously) alongside recall, per the
|
||||
approved eval-framework plan.
|
||||
|
||||
Page comparison: `GroundTruthEntry.printed_page` is a *printed* page number;
|
||||
`Monograph.source_page_range` is *physical*. The physical->printed offset
|
||||
was empirically confirmed constant (+1) across every tested milestone page
|
||||
in Phase 1.1 (`extract/page_map.py`) — reused here rather than re-derived.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import List
|
||||
|
||||
from ..segment.models import Monograph
|
||||
from .back_index import GroundTruthEntry
|
||||
|
||||
PRINTED_PAGE_OFFSET = 1
|
||||
PAGE_TOLERANCE = 2
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecallPrecisionResult:
|
||||
recall: float
|
||||
precision: float
|
||||
matched_count: int
|
||||
total_ground_truth: int
|
||||
total_detected: int
|
||||
unmatched_ground_truth: List[GroundTruthEntry]
|
||||
unmatched_detected: List[Monograph]
|
||||
|
||||
|
||||
_WHITESPACE_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _normalize_name(name: str) -> str:
|
||||
# collapse-whitespace: confirmed real case — "ALVERIN CITRAT" (double
|
||||
# space, likely a genuine PDF-rendering artifact) failed to match
|
||||
# ground truth's "Alverin citrat" under plain strip+upper, found via a
|
||||
# real `cli validate` run (4 of 12 unmatched-detected monographs had
|
||||
# this exact shape: ALVERIN CITRAT, OXYMETAZOLIN HYDROCLORID,
|
||||
# TERBUTALIN SULFAT, TIOTROPIUM BROMID).
|
||||
return _WHITESPACE_RE.sub(" ", name.strip()).upper()
|
||||
|
||||
|
||||
def _monograph_start_printed_page(monograph: Monograph) -> int:
|
||||
return monograph.source_page_range[0] + PRINTED_PAGE_OFFSET
|
||||
|
||||
|
||||
def _names_match(entry_name: str, drug_name: str) -> bool:
|
||||
a, b = _normalize_name(entry_name), _normalize_name(drug_name)
|
||||
return a in b or b in a
|
||||
|
||||
|
||||
def _names_match_exactly(entry_name: str, drug_name: str) -> bool:
|
||||
return _normalize_name(entry_name) == _normalize_name(drug_name)
|
||||
|
||||
|
||||
def _find_match(entry: GroundTruthEntry, monographs: List[Monograph]):
|
||||
# Exact match first, substring fallback only if no exact match exists:
|
||||
# confirmed real case, "Isosorbid" and "Isosorbid dinitrat" are two
|
||||
# distinct real monographs a page apart. A substring-only search finds
|
||||
# "Isosorbid" for BOTH ground-truth entries (it's a substring of
|
||||
# "Isosorbid dinitrat" too) and, being first in page order, wins via
|
||||
# `next()` for both — leaving the real "Isosorbid dinitrat" monograph
|
||||
# spuriously unmatched. Same shape confirmed for "Ampicilin" /
|
||||
# "Ampicilin và sulbactam". Trying each entry's exact match across all
|
||||
# monographs before falling back to substring resolves both without
|
||||
# needing order-dependent tie-breaking.
|
||||
in_tolerance = [
|
||||
m for m in monographs
|
||||
if abs(_monograph_start_printed_page(m) - entry.printed_page) <= PAGE_TOLERANCE
|
||||
]
|
||||
return next(
|
||||
(m for m in in_tolerance if _names_match_exactly(entry.name, m.drug_name)),
|
||||
next((m for m in in_tolerance if _names_match(entry.name, m.drug_name)), None),
|
||||
)
|
||||
|
||||
|
||||
def compute_recall_precision(
|
||||
monographs: List[Monograph], ground_truth: List[GroundTruthEntry],
|
||||
) -> RecallPrecisionResult:
|
||||
matched_gt = []
|
||||
unmatched_gt = []
|
||||
matched_detected_ids: set = set()
|
||||
|
||||
for entry in ground_truth:
|
||||
match = _find_match(entry, monographs)
|
||||
if match is not None:
|
||||
matched_gt.append(entry)
|
||||
matched_detected_ids.add(match.drug_id)
|
||||
else:
|
||||
unmatched_gt.append(entry)
|
||||
|
||||
unmatched_detected = [m for m in monographs if m.drug_id not in matched_detected_ids]
|
||||
return RecallPrecisionResult(
|
||||
recall=len(matched_gt) / len(ground_truth) if ground_truth else 0.0,
|
||||
precision=len(matched_detected_ids) / len(monographs) if monographs else 0.0,
|
||||
matched_count=len(matched_gt),
|
||||
total_ground_truth=len(ground_truth),
|
||||
total_detected=len(monographs),
|
||||
unmatched_ground_truth=unmatched_gt,
|
||||
unmatched_detected=unmatched_detected,
|
||||
)
|
||||
@@ -0,0 +1,202 @@
|
||||
"""Named gates that must hold before the corpus is chunked.
|
||||
|
||||
Chunking bakes whatever it is given into embeddings, where defects stop being
|
||||
inspectable. So the question this module answers is not "did the pipeline
|
||||
run" but "is the text going in actually the text on the page". Each gate is
|
||||
reported on its own line with its own number and its own target — a single
|
||||
pass/fail would hide exactly the problems that took a whole session to find.
|
||||
|
||||
Every gate here is computed from the artefacts, never remembered from an
|
||||
earlier run: quoting a number from before a code change is the specific
|
||||
mistake this project keeps catching.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List, Sequence
|
||||
|
||||
PUA_RANGE = (0xE000, 0xF8FF)
|
||||
REPLACEMENT_CHAR = "�"
|
||||
|
||||
# Strings that were confirmed by eye to be corruption, each traced to a
|
||||
# dropped vector-outlined glyph (outlier-catalog item 24). They are checked
|
||||
# literally: if one reappears, the repair regressed.
|
||||
KNOWN_CORRUPTIONS = (
|
||||
"Độ n định",
|
||||
"≥ 1 tu i",
|
||||
"tại ch :",
|
||||
)
|
||||
|
||||
# Fragments of 2D formulas that must never sit in prose, where the missing
|
||||
# fraction bar turns a division into a multiplication.
|
||||
FORMULA_FRAGMENTS = (
|
||||
"Thể trọng (kg)",
|
||||
"(140 - tuổi) x cân nặng",
|
||||
"x (140 - số tuổi)",
|
||||
"Giá trị Clcr của bệnh nhân",
|
||||
"218 x P x",
|
||||
"× trọng lượng cơ thể (kg)",
|
||||
"Cân nặng (kg) x liều",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Gate:
|
||||
name: str
|
||||
count: int
|
||||
target: int = 0
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def passed(self) -> bool:
|
||||
return self.count == self.target
|
||||
|
||||
|
||||
def _section_texts(monograph: dict) -> Iterable[str]:
|
||||
for section in (monograph.get("sections") or {}).values():
|
||||
yield section.get("text") or ""
|
||||
|
||||
|
||||
def _count_pua(text: str) -> int:
|
||||
return sum(1 for ch in text if PUA_RANGE[0] <= ord(ch) <= PUA_RANGE[1])
|
||||
|
||||
|
||||
def evaluate(monographs: Sequence[dict],
|
||||
transcribed_runs: Sequence[dict] = ()) -> List[Gate]:
|
||||
"""Compute every readiness gate over the whole corpus."""
|
||||
pua = replacement = empty = no_provenance = 0
|
||||
corruptions: Dict[str, int] = {c: 0 for c in KNOWN_CORRUPTIONS}
|
||||
formula_leaks: Dict[str, int] = {f: 0 for f in FORMULA_FRAGMENTS}
|
||||
unflagged_blocks = 0
|
||||
ids: Dict[str, int] = {}
|
||||
no_page_range = 0
|
||||
corpus = []
|
||||
|
||||
for monograph in monographs:
|
||||
ids[monograph["drug_id"]] = ids.get(monograph["drug_id"], 0) + 1
|
||||
if not monograph.get("source_page_range"):
|
||||
no_page_range += 1
|
||||
for section in (monograph.get("sections") or {}).values():
|
||||
text = section.get("text") or ""
|
||||
corpus.append(text)
|
||||
if not text.strip():
|
||||
empty += 1
|
||||
if not section.get("parts"):
|
||||
no_provenance += 1
|
||||
pua += _count_pua(text)
|
||||
replacement += text.count(REPLACEMENT_CHAR)
|
||||
for phrase in KNOWN_CORRUPTIONS:
|
||||
corruptions[phrase] += text.count(phrase)
|
||||
for phrase in FORMULA_FRAGMENTS:
|
||||
formula_leaks[phrase] += text.count(phrase)
|
||||
for block in monograph.get("tables") or []:
|
||||
if not block.get("quarantined"):
|
||||
unflagged_blocks += 1
|
||||
|
||||
joined = "\n".join(corpus)
|
||||
unmerged = [
|
||||
run for run in transcribed_runs
|
||||
if len(run["text"].strip()) > 2 and run["text"].strip() not in joined
|
||||
]
|
||||
|
||||
return [
|
||||
Gate("outlined_run_not_merged", len(unmerged),
|
||||
detail="; ".join(f"p{r['physical_page']} {r['text'][:40]!r}"
|
||||
for r in unmerged[:5])),
|
||||
Gate("known_corruption_string", sum(corruptions.values()),
|
||||
detail=", ".join(f"{k!r}={v}" for k, v in corruptions.items() if v)),
|
||||
Gate("formula_fragment_in_prose", sum(formula_leaks.values()),
|
||||
detail=", ".join(f"{k!r}={v}" for k, v in formula_leaks.items() if v)),
|
||||
Gate("pua_char", pua),
|
||||
Gate("replacement_char_ufffd", replacement),
|
||||
Gate("empty_section", empty),
|
||||
Gate("section_without_provenance", no_provenance),
|
||||
Gate("unflagged_quarantine_block", unflagged_blocks),
|
||||
Gate("duplicate_drug_id", sum(1 for n in ids.values() if n > 1)),
|
||||
Gate("monograph_without_page_range", no_page_range),
|
||||
]
|
||||
|
||||
|
||||
def corpus_size(monographs: Sequence[dict]) -> Dict[str, int]:
|
||||
"""Informational, not a gate: how much text chunking would consume."""
|
||||
sections = [t for m in monographs for t in _section_texts(m)]
|
||||
return {
|
||||
"monographs": len(monographs),
|
||||
"sections": len(sections),
|
||||
"section_chars": sum(len(t) for t in sections),
|
||||
"quarantined_blocks": sum(len(m.get("tables") or []) for m in monographs),
|
||||
}
|
||||
|
||||
|
||||
def read_monographs(path: Path) -> List[dict]:
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
return [json.loads(line) for line in handle if line.strip()]
|
||||
|
||||
|
||||
def evaluate_chunks(monographs: Sequence[dict],
|
||||
chunks: Sequence[dict]) -> List[Gate]:
|
||||
"""ADR 0006 gates: a chunk must never hide that a block was lifted.
|
||||
|
||||
The failure being guarded against is silent, not visible: a chunk of
|
||||
AMPICILIN VÀ SULBACTAM's dosing section is grammatical, complete-looking
|
||||
prose with the renal-dosing table absent and nothing marking the absence.
|
||||
Measured: 127 of 167 lifted blocks came out of `liều lượng và cách dùng`.
|
||||
"""
|
||||
blocks_by_section: Dict[tuple, list] = {}
|
||||
block_ids: Dict[str, str] = {}
|
||||
block_texts: Dict[str, str] = {}
|
||||
for monograph in monographs:
|
||||
for block in monograph.get("tables") or []:
|
||||
key = (monograph["drug_id"], block.get("section_key"))
|
||||
blocks_by_section.setdefault(key, []).append(block)
|
||||
block_ids[block["table_id"]] = monograph["drug_id"]
|
||||
if block.get("text"):
|
||||
block_texts[block["table_id"]] = block["text"]
|
||||
|
||||
referenced: Dict[tuple, set] = {}
|
||||
unknown_id = missing_provenance = leaked = 0
|
||||
descriptors = 0
|
||||
descriptor_without_attachment = 0
|
||||
|
||||
for chunk in chunks:
|
||||
attachments = chunk.get("attachments") or []
|
||||
if chunk.get("chunk_kind") == "block_descriptor":
|
||||
descriptors += 1
|
||||
if not attachments:
|
||||
descriptor_without_attachment += 1
|
||||
key = (chunk["drug_id"], chunk["section_key"])
|
||||
for attachment in attachments:
|
||||
referenced.setdefault(key, set()).add(attachment["block_id"])
|
||||
if block_ids.get(attachment["block_id"]) != chunk["drug_id"]:
|
||||
unknown_id += 1
|
||||
if attachment.get("physical_page") is None or not attachment.get("bbox"):
|
||||
missing_provenance += 1
|
||||
body = chunk.get("text") or ""
|
||||
for attachment in attachments:
|
||||
source = block_texts.get(attachment["block_id"], "")
|
||||
probe = source.strip()[:60]
|
||||
if len(probe) > 20 and probe in body:
|
||||
leaked += 1
|
||||
|
||||
unreferenced = 0
|
||||
for key, blocks in blocks_by_section.items():
|
||||
seen = referenced.get(key, set())
|
||||
unreferenced += sum(1 for b in blocks if b["table_id"] not in seen)
|
||||
|
||||
total_blocks = sum(len(v) for v in blocks_by_section.values())
|
||||
return [
|
||||
Gate("section_block_without_chunk_reference", unreferenced),
|
||||
Gate("attachment_block_id_unknown", unknown_id),
|
||||
Gate("attachment_without_page_or_bbox", missing_provenance),
|
||||
Gate("block_text_leaked_into_chunk_text", leaked),
|
||||
Gate("descriptor_chunk_without_attachment", descriptor_without_attachment),
|
||||
Gate("descriptor_count_vs_block_count", descriptors, target=total_blocks,
|
||||
detail=f"{descriptors} descriptors for {total_blocks} blocks"),
|
||||
]
|
||||
|
||||
|
||||
def read_chunks(path: Path) -> List[dict]:
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
return [json.loads(line) for line in handle if line.strip()]
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Residual-ink coverage check: what is on the page that the text layer never emitted.
|
||||
|
||||
Every other check in this project asks a detector whether it found something.
|
||||
This one asks the page. It renders each page, whites out every pixel covered
|
||||
by a span the extractor actually produced, and reports the ink that survives.
|
||||
Whatever survives is content the text layer cannot account for — vector
|
||||
rules, fraction bars, figures.
|
||||
|
||||
Why it earns its place: the two confirmed 2D-formula corruptions
|
||||
(NETILMICIN physical page 1042, AMPICILIN VÀ SULBACTAM physical page 202)
|
||||
are invisible to both table detectors in this repo — `pdfplumber` reports 0
|
||||
regions on those pages and so does `opendataloader-pdf`. The two tools share
|
||||
a blind spot because both need ruling lines. Pixels do not share it: the
|
||||
fraction bar is ink, so it survives the mask and gets reported.
|
||||
|
||||
The check needs no ground truth and no sampling — measured at 0.06 s/page,
|
||||
so all 1668 pages run in under two minutes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, Iterable, Iterator, List, Sequence, Tuple
|
||||
|
||||
import fitz
|
||||
import numpy as np
|
||||
from scipy import ndimage
|
||||
|
||||
from ..extract.outlined_text import OutlinedTextRun, detect_outlined_text
|
||||
from ..tables.models import TableRegion
|
||||
|
||||
RENDER_DPI = 150
|
||||
INK_THRESHOLD = 200
|
||||
|
||||
# Measured, not guessed: at 1.0pt the mask eats the fraction bar itself —
|
||||
# page 1042's bar survives as 9.1pt of its true 188.6pt. At 0.5pt the full
|
||||
# bar survives, and a 10-page prose sample produced the same region count as
|
||||
# 1.0pt (11 regions), i.e. the looser padding adds no noise.
|
||||
MASK_PAD_PT = 0.5
|
||||
|
||||
THIN_HEIGHT_PT = 3.0
|
||||
HEADER_BAND_PT = 60.0
|
||||
RULE_MIN_WIDTH_PT = 400.0
|
||||
BAR_MIN_WIDTH_PT = 10.0
|
||||
|
||||
# A glyph outline can extend a fraction past the filled path's own box, so the
|
||||
# overlap test is given room: without it, three ink fragments on page 714 sit
|
||||
# just outside their line's box and read as unexplained text.
|
||||
OUTLINE_TOLERANCE_PT = 2.0
|
||||
|
||||
# Smaller than any mark a real glyph leaves. Measured against the 1,054
|
||||
# components of confirmed outlined text on the five affected pages: the
|
||||
# smallest is well above this, so the rule cannot swallow real text.
|
||||
SPECK_EXTENT_PT = 2.0
|
||||
|
||||
HEADER_RULE = "header_rule"
|
||||
TABLE_FRAME = "table_frame"
|
||||
TEXT_AS_VECTOR_OUTLINE = "text_as_vector_outline"
|
||||
FRACTION_BAR_CANDIDATE = "fraction_bar_candidate"
|
||||
HEADER_BAND_FRAGMENT = "header_band_fragment"
|
||||
RULE_FRAGMENT = "rule_fragment"
|
||||
ANTIALIAS_SPECK = "antialias_speck"
|
||||
UNCLASSIFIED = "unclassified"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResidualRegion:
|
||||
"""Ink left on a page after masking every extracted span.
|
||||
|
||||
Provenance is the point: `physical_page` + `bbox` locate the region in the
|
||||
source PDF exactly, so any verdict about it can be re-checked by eye.
|
||||
"""
|
||||
|
||||
physical_page: int
|
||||
bbox: Tuple[float, float, float, float]
|
||||
ink_px: int
|
||||
|
||||
@property
|
||||
def width_pt(self) -> float:
|
||||
return self.bbox[2] - self.bbox[0]
|
||||
|
||||
@property
|
||||
def height_pt(self) -> float:
|
||||
return self.bbox[3] - self.bbox[1]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PageContext:
|
||||
"""What else is known to be on the page, for naming residual ink.
|
||||
|
||||
Carried as one object so a new kind of context is a new field here rather
|
||||
than a new positional argument on every predicate.
|
||||
"""
|
||||
|
||||
tables: Sequence[TableRegion] = ()
|
||||
outlined_runs: Sequence[OutlinedTextRun] = ()
|
||||
|
||||
|
||||
def _overlaps(bbox, other) -> bool:
|
||||
return not (bbox[2] < other[0] or bbox[0] > other[2]
|
||||
or bbox[3] < other[1] or bbox[1] > other[3])
|
||||
|
||||
|
||||
def _is_header_rule(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return (
|
||||
region.height_pt <= THIN_HEIGHT_PT
|
||||
and region.bbox[1] < HEADER_BAND_PT
|
||||
and region.width_pt >= RULE_MIN_WIDTH_PT
|
||||
)
|
||||
|
||||
|
||||
def _is_table_frame(region: ResidualRegion, context: PageContext) -> bool:
|
||||
return any(table.contains(*region.bbox) for table in context.tables)
|
||||
|
||||
|
||||
def _is_outlined_text(region: ResidualRegion, context: PageContext) -> bool:
|
||||
pad = OUTLINE_TOLERANCE_PT
|
||||
return any(
|
||||
_overlaps(region.bbox,
|
||||
(line.bbox[0] - pad, line.bbox[1] - pad,
|
||||
line.bbox[2] + pad, line.bbox[3] + pad))
|
||||
for line in context.outlined_runs
|
||||
)
|
||||
|
||||
|
||||
def _is_fraction_bar(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return region.height_pt <= THIN_HEIGHT_PT and region.width_pt >= BAR_MIN_WIDTH_PT
|
||||
|
||||
|
||||
def _is_header_band_fragment(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
"""Leftovers of the running-header rule, chopped up by the text over it.
|
||||
|
||||
Confirmed by eye on physical page 382: a 31.7 x 9.6pt L-shape that is the
|
||||
header rule meeting a vertical tick, split into its own component because
|
||||
the header text's mask cut the rule either side of it.
|
||||
"""
|
||||
return region.bbox[3] <= HEADER_BAND_PT
|
||||
|
||||
|
||||
def _is_rule_fragment(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return min(region.width_pt, region.height_pt) <= THIN_HEIGHT_PT
|
||||
|
||||
|
||||
def _is_speck(region: ResidualRegion, _context: PageContext) -> bool:
|
||||
return (region.width_pt < SPECK_EXTENT_PT
|
||||
and region.height_pt < SPECK_EXTENT_PT)
|
||||
|
||||
|
||||
# Open/closed: a new residual kind is a new entry here, not an edit to the
|
||||
# existing predicates. Order matters — first match wins. Outlined text is
|
||||
# tested before the fraction-bar shape rule, which its underline-like
|
||||
# fragments would otherwise satisfy; the header band is tested before it too,
|
||||
# because the header text's mask cuts the running rule into short pieces that
|
||||
# are bar-shaped (8 of them on physical page 382 alone).
|
||||
_RULES: List[Tuple[str, Callable[[ResidualRegion, PageContext], bool]]] = [
|
||||
(HEADER_RULE, _is_header_rule),
|
||||
(TABLE_FRAME, _is_table_frame),
|
||||
(TEXT_AS_VECTOR_OUTLINE, _is_outlined_text),
|
||||
(HEADER_BAND_FRAGMENT, _is_header_band_fragment),
|
||||
(FRACTION_BAR_CANDIDATE, _is_fraction_bar),
|
||||
(ANTIALIAS_SPECK, _is_speck),
|
||||
(RULE_FRAGMENT, _is_rule_fragment),
|
||||
]
|
||||
|
||||
|
||||
def classify(region: ResidualRegion, context: PageContext | None = None) -> str:
|
||||
"""Name what a residual region is. Pure — no PDF, no rendering."""
|
||||
context = context or PageContext()
|
||||
for kind, predicate in _RULES:
|
||||
if predicate(region, context):
|
||||
return kind
|
||||
return UNCLASSIFIED
|
||||
|
||||
|
||||
def _ink_boxes(mask: "np.ndarray") -> Iterator[Tuple[int, int, int, int]]:
|
||||
"""One box per connected blob of surviving ink.
|
||||
|
||||
Two cheaper splits were tried first and both misreport real pages. Cutting
|
||||
into horizontal bands only merges a table in the left column with one in
|
||||
the right column, so the merged box's centre lands in the gutter, matches
|
||||
no table region, and physical page 209's ADR table is reported as
|
||||
unaccounted-for ink. Adding a column-run split then cuts a single table
|
||||
grid into its individual rules, because masking the text leaves the rules
|
||||
standing with empty gaps between them. A table grid is one connected
|
||||
object and a fraction bar is another, so connectivity is the property that
|
||||
actually separates them.
|
||||
"""
|
||||
labelled, _ = ndimage.label(mask, structure=np.ones((3, 3), dtype=bool))
|
||||
for top_bottom, left_right in ndimage.find_objects(labelled) or []:
|
||||
yield left_right.start, top_bottom.start, left_right.stop - 1, top_bottom.stop - 1
|
||||
|
||||
|
||||
def scan_page(page: "fitz.Page", dpi: int = RENDER_DPI) -> List[ResidualRegion]:
|
||||
"""Render one page, mask its extracted spans, return the surviving ink."""
|
||||
scale = dpi / 72.0
|
||||
pixmap = page.get_pixmap(dpi=dpi, colorspace=fitz.csGRAY)
|
||||
image = np.frombuffer(pixmap.samples, dtype=np.uint8).reshape(
|
||||
pixmap.height, pixmap.width
|
||||
).copy()
|
||||
|
||||
for block in page.get_text("dict")["blocks"]:
|
||||
for line in block.get("lines", []):
|
||||
for span in line["spans"]:
|
||||
x0, y0, x1, y1 = span["bbox"]
|
||||
top = max(0, int((y0 - MASK_PAD_PT) * scale))
|
||||
bottom = min(pixmap.height, int((y1 + MASK_PAD_PT) * scale) + 1)
|
||||
left = max(0, int((x0 - MASK_PAD_PT) * scale))
|
||||
right = min(pixmap.width, int((x1 + MASK_PAD_PT) * scale) + 1)
|
||||
image[top:bottom, left:right] = 255
|
||||
|
||||
mask = image < INK_THRESHOLD
|
||||
return [
|
||||
ResidualRegion(
|
||||
physical_page=page.number,
|
||||
bbox=(
|
||||
round(left / scale, 2),
|
||||
round(top / scale, 2),
|
||||
round(right / scale, 2),
|
||||
round(bottom / scale, 2),
|
||||
),
|
||||
ink_px=int(mask[top:bottom + 1, left:right + 1].sum()),
|
||||
)
|
||||
for left, top, right, bottom in _ink_boxes(mask)
|
||||
]
|
||||
|
||||
|
||||
def scan_document(
|
||||
doc: "fitz.Document",
|
||||
tables_by_page: Dict[int, List[TableRegion]] | None = None,
|
||||
pages: Iterable[int] | None = None,
|
||||
) -> Iterator[Tuple[ResidualRegion, str]]:
|
||||
"""Yield every residual region in the document with its classification."""
|
||||
tables_by_page = tables_by_page or {}
|
||||
page_numbers = list(range(doc.page_count) if pages is None else pages)
|
||||
|
||||
outlines: Dict[int, List[OutlinedTextRun]] = {}
|
||||
for line in detect_outlined_text(doc, page_numbers):
|
||||
outlines.setdefault(line.physical_page, []).append(line)
|
||||
|
||||
for number in page_numbers:
|
||||
context = PageContext(
|
||||
tables=tables_by_page.get(number, ()),
|
||||
outlined_runs=outlines.get(number, ()),
|
||||
)
|
||||
for region in scan_page(doc[number]):
|
||||
yield region, classify(region, context)
|
||||
@@ -3,8 +3,14 @@ name = "ingestion"
|
||||
version = "0.0.0"
|
||||
description = "Offline batch pipeline: PDF -> monographs -> chunks -> embeddings -> Qdrant"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = []
|
||||
dependencies = ["pymupdf>=1.24", "numpy>=1.26", "scipy>=1.11"]
|
||||
|
||||
[project.optional-dependencies]
|
||||
dev = ["pytest>=7.4"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["ingestion*"]
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ingestion.chunk import (
|
||||
CHUNK_KIND_BLOCK_DESCRIPTOR,
|
||||
CHUNK_KIND_PROSE,
|
||||
SCHEMA_VERSION,
|
||||
chunk_monograph,
|
||||
chunk_section,
|
||||
write_chunks_jsonl,
|
||||
)
|
||||
from ingestion.chunk.chunker import _is_label_row, describe_block
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from ingestion.tables import SHAPE_FORMULA_2D, SHAPE_MULTI_HEADER, SHAPE_SIMPLE
|
||||
|
||||
|
||||
def _section(key, display, text, page=202):
|
||||
return SectionSpan(
|
||||
key=key, display_name=display,
|
||||
heading=Heading(text=display, physical_page=page, y0=100.0,
|
||||
is_monograph_title=False, section_key=key),
|
||||
text=text,
|
||||
)
|
||||
|
||||
|
||||
def _monograph(sections, tables=()):
|
||||
return Monograph(
|
||||
drug_id="ampicilin_va_sulbactam",
|
||||
drug_name="AMPICILIN VÀ SULBACTAM",
|
||||
source_page_range=[200, 203],
|
||||
sections={s.key: s for s in sections},
|
||||
atc_codes=["J01CR01"],
|
||||
tables=list(tables),
|
||||
)
|
||||
|
||||
|
||||
def _block(block_id="p202_t0", shape=SHAPE_SIMPLE, section_key="lieu_luong_va_cach_dung"):
|
||||
return TableBlock(
|
||||
table_id=block_id, shape=shape, physical_page=202,
|
||||
bbox=[299.0, 189.6, 552.4, 300.5], section_key=section_key,
|
||||
text="Độ thanh thải creatinin Nửa đời Liều 1,5 - 3,0 g",
|
||||
quarantined=True,
|
||||
)
|
||||
|
||||
|
||||
def test_a_section_whose_table_was_lifted_says_so():
|
||||
"""The defect this exists to prevent is silent, not visible.
|
||||
|
||||
Without the reference, this chunk is grammatical, complete-looking prose
|
||||
with the renal-dosing table absent and nothing marking the absence.
|
||||
"""
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(monograph)
|
||||
|
||||
prose = [c for c in chunks if c.chunk_kind == CHUNK_KIND_PROSE]
|
||||
assert len(prose) == 1
|
||||
assert prose[0].has_quarantined_content is True
|
||||
assert [a.block_id for a in prose[0].attachments] == ["p202_t0"]
|
||||
assert prose[0].attachments[0].physical_page == 202
|
||||
assert prose[0].attachments[0].bbox
|
||||
|
||||
|
||||
def test_a_lifted_block_gets_its_own_retrievable_descriptor():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
descriptors = [c for c in chunk_monograph(monograph)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
assert len(descriptors) == 1
|
||||
assert "AMPICILIN VÀ SULBACTAM" in descriptors[0].text
|
||||
assert "Liều lượng và cách dùng" in descriptors[0].text
|
||||
# printed page, which is what a reader holding the book looks for
|
||||
assert "trang 203" in descriptors[0].text
|
||||
|
||||
|
||||
def test_no_cell_value_ever_reaches_the_descriptor_text():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
block = _block()
|
||||
monograph = _monograph([section], [block])
|
||||
descriptors = [c for c in chunk_monograph(monograph, {"p202_t0": []})
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
assert "1,5 - 3,0 g" not in descriptors[0].text
|
||||
|
||||
|
||||
def test_a_header_row_carrying_a_number_is_refused():
|
||||
"""AMIODARON, physical page 183 — a real case, caught by a gate.
|
||||
|
||||
pdfplumber reported the first row as
|
||||
"Thời gian liệu pháp tĩnh mạch Liều 720 mg/ngày (0,5 mg/phút)", i.e. a
|
||||
dose inside what it called a header, from an extraction never verified by
|
||||
eye. Measured: 42 of 124 simple-table headers (34%) contain a digit.
|
||||
"""
|
||||
assert _is_label_row(["Các Statin", "Khởi đầu", "Liều duy trì"]) is True
|
||||
assert _is_label_row(["Liều 720 mg/ngày (0,5 mg/phút)"]) is False
|
||||
assert _is_label_row(["x" * 45]) is False
|
||||
assert _is_label_row([]) is False
|
||||
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(
|
||||
monograph, {"p202_t0": ["Liều 720 mg/ngày (0,5 mg/phút)"]})
|
||||
descriptor = next(c for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert "720" not in descriptor.text
|
||||
assert descriptor.attachments[0].header_row == []
|
||||
|
||||
|
||||
def test_only_a_simple_table_contributes_a_header():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
header = {"p202_t0": ["Nhóm", "Liều"]}
|
||||
for shape, expected in ((SHAPE_SIMPLE, ["Nhóm", "Liều"]),
|
||||
(SHAPE_MULTI_HEADER, [])):
|
||||
monograph = _monograph([section], [_block(shape=shape)])
|
||||
descriptor = next(c for c in chunk_monograph(monograph, header)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert descriptor.attachments[0].header_row == expected
|
||||
|
||||
|
||||
def test_a_formula_block_is_described_as_a_formula():
|
||||
section = _section("than_trong", "Thận trọng", "Prose.")
|
||||
block = _block(block_id="p1042_f0", shape=SHAPE_FORMULA_2D,
|
||||
section_key="than_trong")
|
||||
monograph = _monograph([section], [block])
|
||||
descriptor = next(c for c in chunk_monograph(monograph)
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR)
|
||||
assert "công thức" in descriptor.text
|
||||
assert "bảng" not in descriptor.text
|
||||
|
||||
|
||||
def test_attachments_do_not_change_the_prose_text():
|
||||
"""The condition under which this feature was accepted at all."""
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng",
|
||||
"Liều thường dùng cho người lớn là 1,5 - 3 g mỗi 6 giờ.")
|
||||
with_block = chunk_section(_monograph([section], [_block()]), section,
|
||||
[_block()])
|
||||
without = chunk_section(_monograph([section]), section)
|
||||
prose_with = [c for c in with_block if c.chunk_kind == CHUNK_KIND_PROSE]
|
||||
assert [c.text for c in prose_with] == [c.text for c in without]
|
||||
assert [c.chunk_id for c in prose_with] == [c.chunk_id for c in without]
|
||||
|
||||
|
||||
def test_a_section_with_no_text_but_a_block_still_yields_the_descriptor():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "")
|
||||
chunks = chunk_monograph(_monograph([section], [_block()]))
|
||||
assert [c.chunk_kind for c in chunks] == [CHUNK_KIND_BLOCK_DESCRIPTOR]
|
||||
|
||||
|
||||
def test_written_chunks_declare_their_schema_version(tmp_path: Path):
|
||||
section = _section("chi_dinh", "Chỉ định", "Nhiễm khuẩn.")
|
||||
chunks = chunk_monograph(_monograph([section]))
|
||||
out = tmp_path / "chunks.jsonl"
|
||||
assert write_chunks_jsonl(chunks, out) == 1
|
||||
record = json.loads(out.read_text(encoding="utf-8").splitlines()[0])
|
||||
assert record["schema_version"] == SCHEMA_VERSION
|
||||
assert record["chunk_kind"] == CHUNK_KIND_PROSE
|
||||
assert record["has_quarantined_content"] is False
|
||||
|
||||
|
||||
def test_describe_block_names_the_page_even_with_no_header():
|
||||
section = _section("lieu_luong_va_cach_dung", "Liều lượng và cách dùng", "Prose.")
|
||||
monograph = _monograph([section], [_block()])
|
||||
chunks = chunk_monograph(monograph)
|
||||
attachment = next(c for c in chunks
|
||||
if c.chunk_kind == CHUNK_KIND_BLOCK_DESCRIPTOR).attachments[0]
|
||||
text = describe_block(monograph, section, attachment)
|
||||
assert "trang 203" in text
|
||||
assert "không trích dẫn được dưới dạng văn bản" in text
|
||||
@@ -0,0 +1,49 @@
|
||||
import pytest
|
||||
|
||||
from ingestion.cli import build_parser
|
||||
|
||||
|
||||
def test_run_subcommand_parses_required_pdf_arg():
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(["run", "--pdf", "some.pdf"])
|
||||
assert args.command == "run"
|
||||
assert args.pdf == "some.pdf"
|
||||
assert args.out == "data/processed/monographs.jsonl"
|
||||
|
||||
|
||||
def test_run_subcommand_accepts_custom_out():
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(["run", "--pdf", "a.pdf", "--out", "b.jsonl"])
|
||||
assert args.out == "b.jsonl"
|
||||
|
||||
|
||||
def test_run_requires_pdf_arg():
|
||||
parser = build_parser()
|
||||
with pytest.raises(SystemExit):
|
||||
parser.parse_args(["run"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("command", ["visual-diff", "scaffold-golden"])
|
||||
def test_not_yet_implemented_commands_raise_explicitly(command):
|
||||
parser = build_parser()
|
||||
args = parser.parse_args([command])
|
||||
with pytest.raises(NotImplementedError):
|
||||
args.func(args)
|
||||
|
||||
|
||||
def test_run_reports_missing_pdf_file(tmp_path, capsys):
|
||||
parser = build_parser()
|
||||
missing = tmp_path / "does_not_exist.pdf"
|
||||
args = parser.parse_args(["run", "--pdf", str(missing)])
|
||||
exit_code = args.func(args)
|
||||
assert exit_code == 1
|
||||
assert "not found" in capsys.readouterr().err
|
||||
|
||||
|
||||
def test_validate_reports_missing_pdf_file(tmp_path, capsys):
|
||||
parser = build_parser()
|
||||
missing = tmp_path / "does_not_exist.pdf"
|
||||
args = parser.parse_args(["validate", "--pdf", str(missing)])
|
||||
exit_code = args.func(args)
|
||||
assert exit_code == 1
|
||||
assert "not found" in capsys.readouterr().err
|
||||
@@ -0,0 +1,75 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ingestion.extract.formulas import (
|
||||
FORMULA_BAND_HEIGHT_PT,
|
||||
FORMULA_SIDE_MARGIN_PT,
|
||||
load_formula_regions,
|
||||
)
|
||||
from ingestion.tables import QUARANTINE_SHAPES, SHAPE_FORMULA_2D
|
||||
|
||||
VERIFIED = (Path(__file__).resolve().parents[1] / "data" / "verified"
|
||||
/ "formula_regions_2d.json")
|
||||
TRANSCRIPTIONS = (Path(__file__).resolve().parents[1] / "data" / "verified"
|
||||
/ "outlined_text_transcriptions.json")
|
||||
|
||||
|
||||
def test_a_2d_formula_is_always_quarantined():
|
||||
# linearised, "a / b" reads as "a x b" — a dosing error, not a cosmetic one
|
||||
assert SHAPE_FORMULA_2D in QUARANTINE_SHAPES
|
||||
|
||||
|
||||
def test_verified_formula_regions_load_with_the_confirmed_pages():
|
||||
regions = load_formula_regions()
|
||||
assert {r.physical_page for r in regions} == {
|
||||
43, 92, 147, 202, 325, 349, 1042, 1043, 1132, 1402,
|
||||
}
|
||||
assert all(r.shape == SHAPE_FORMULA_2D for r in regions)
|
||||
|
||||
|
||||
def test_the_region_covers_numerator_and_denominator_not_just_the_bar():
|
||||
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
|
||||
bar = next(r for r in payload["regions"] if r["physical_page"] == 1042)
|
||||
region = next(r for r in load_formula_regions() if r.physical_page == 1042)
|
||||
x0, y0, x1, y1 = bar["bar_bbox"]
|
||||
assert region.bbox[1] == y0 - FORMULA_BAND_HEIGHT_PT
|
||||
assert region.bbox[3] == y1 + FORMULA_BAND_HEIGHT_PT
|
||||
assert region.bbox[0] == x0 - FORMULA_SIDE_MARGIN_PT
|
||||
|
||||
|
||||
def test_the_barless_adenosin_formula_is_recorded_as_a_recall_limit():
|
||||
"""The source prints no bar, so no geometric detector can find it.
|
||||
|
||||
Recorded so a later reader does not mistake the fraction-bar scan for
|
||||
complete formula coverage — how many bar-less formulas the book contains
|
||||
has never been measured.
|
||||
"""
|
||||
payload = json.loads(VERIFIED.read_text(encoding="utf-8"))
|
||||
barless = [r for r in payload["regions"] if r.get("source_prints_no_bar")]
|
||||
assert [r["physical_page"] for r in barless] == [147]
|
||||
assert "UNMEASURED" in payload["recall_limit"]
|
||||
|
||||
|
||||
def test_outlined_text_transcriptions_cover_every_detected_run():
|
||||
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
|
||||
runs = payload["runs"]
|
||||
assert len(runs) == 51
|
||||
assert all(r["text"] for r in runs), "a run with no transcription is data loss"
|
||||
pages = {}
|
||||
for run in runs:
|
||||
pages[run["physical_page"]] = pages.get(run["physical_page"], 0) + 1
|
||||
assert pages == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}
|
||||
|
||||
|
||||
def test_single_glyph_transcriptions_name_the_line_they_were_dropped_from():
|
||||
"""The subtlest form of the defect: one character missing mid-sentence.
|
||||
|
||||
"Độ ổn định" extracts as "Độ n định" and reads as ordinary text, so
|
||||
nothing downstream can notice. Keeping the owning line in the record is
|
||||
what makes the repair checkable.
|
||||
"""
|
||||
payload = json.loads(TRANSCRIPTIONS.read_text(encoding="utf-8"))
|
||||
singles = [r for r in payload["runs"] if r["single_glyph"]]
|
||||
assert len(singles) == 29
|
||||
with_context = [r for r in singles if r["extracted_line_it_belongs_to"]]
|
||||
assert with_context, "no dropped glyph could be tied back to its line"
|
||||
@@ -0,0 +1,79 @@
|
||||
from ingestion.extract.glyph_order import find_reading_order_issues, is_reversed_order
|
||||
|
||||
|
||||
def test_normal_ltr_span_not_flagged():
|
||||
# ordinary increasing x-origins, as any normal left-to-right span has
|
||||
assert not is_reversed_order([264.7, 269.4, 271.6, 276.3, 278.5])
|
||||
|
||||
|
||||
def test_confirmed_page_1373_defect_shape_is_flagged():
|
||||
# exact x-origins read via get_text("rawdict") from physical page 1373's
|
||||
# affected span (" tịx 4 =" reversed) — see docs/pdf-parsing-outlier-catalog.md item 9
|
||||
x_origins = [66.32, 64.17, 61.53, 58.89, 54.14, 51.98, 47.23]
|
||||
assert is_reversed_order(x_origins)
|
||||
|
||||
|
||||
def test_single_char_span_not_flagged():
|
||||
assert not is_reversed_order([100.0])
|
||||
|
||||
|
||||
def test_empty_span_not_flagged():
|
||||
assert not is_reversed_order([])
|
||||
|
||||
|
||||
def test_tied_x_origins_not_flagged_as_reversed():
|
||||
# equal x-origins (e.g. stacked/overlapping glyphs) are not "decreasing"
|
||||
assert not is_reversed_order([100.0, 100.0, 100.0])
|
||||
|
||||
|
||||
def test_correctly_ordered_row_not_flagged():
|
||||
row = {(20, 550.9): [(518.0, "n"), (525.2, "h"), (532.6, "i"), (536.8, "e")]}
|
||||
assert find_reading_order_issues(row) == []
|
||||
|
||||
|
||||
def test_confirmed_page_714_row_misorder_is_flagged():
|
||||
# reproduces the real page-714 finding: within one PyMuPDF block (20),
|
||||
# 4 line fragments are emitted out of x-order ("quản ", " ộ", "đ tệih",
|
||||
# "n " concatenated) that reconstruct correctly ("...nhiệt độ") when
|
||||
# re-sorted by x-origin — see outlier catalog item 9.
|
||||
row = {
|
||||
(20, 550.9): [
|
||||
(518.06, "n"), (525.20, " "),
|
||||
(546.59, " "), (553.71, "ộ"),
|
||||
(541.84, "đ"), (539.45, " "), (536.81, "t"), (532.59, "ệ"), (529.95, "i"), (525.20, "h"),
|
||||
]
|
||||
}
|
||||
issues = find_reading_order_issues(row)
|
||||
assert len(issues) == 1
|
||||
assert issues[0].extracted_text != issues[0].corrected_text
|
||||
|
||||
|
||||
def test_different_blocks_at_same_y_not_merged():
|
||||
# regression test for a real false positive: two DIFFERENT paragraphs in
|
||||
# different PyMuPDF blocks (a right-column paragraph starting at x=299.4
|
||||
# and a left-column paragraph starting at x=35.4, page 1104) coincide at
|
||||
# the same y — grouping by block index (not a hand-picked x-coordinate
|
||||
# column boundary) is what keeps them from being merged into one "row".
|
||||
# This is the caller's responsibility (scan_reading_order groups by real
|
||||
# PyMuPDF block index); find_reading_order_issues just trusts its input
|
||||
# is already correctly grouped, which these two dict entries demonstrate.
|
||||
row_block_1 = {(1, 70.4): [(299.39, "m"), (306.78, "ô")]}
|
||||
row_block_4 = {(4, 70.4): [(35.43, "d"), (40.18, "e")]}
|
||||
assert find_reading_order_issues(row_block_1) == []
|
||||
assert find_reading_order_issues(row_block_4) == []
|
||||
|
||||
|
||||
def test_kerning_jitter_not_flagged_as_reading_order_defect():
|
||||
# regression test for a real false positive found by running against the
|
||||
# actual PDF: "mefloquin" ('l' at x=491.566, 'o' at x=491.471 — a
|
||||
# 0.095pt kerning-driven dip) was previously "corrected" into the wrong
|
||||
# word "mefolquin". A row-level check with no decrease tolerance treats
|
||||
# ordinary kerning as a defect and corrupts already-correct text.
|
||||
row = {
|
||||
(5, 449.7): [
|
||||
(474.865, "m"), (482.161, "e"), (486.284, "f"),
|
||||
(491.566, "l"), (491.471, "o"), (496.126, "q"),
|
||||
(500.781, "u"), (505.436, "i"), (507.982, "n"),
|
||||
]
|
||||
}
|
||||
assert find_reading_order_issues(row) == []
|
||||
@@ -0,0 +1,27 @@
|
||||
from ingestion.extract.page_map import pick_folio
|
||||
|
||||
|
||||
def test_single_candidate_is_the_folio():
|
||||
assert pick_folio([("101", 10.0)]) == 101
|
||||
|
||||
|
||||
def test_no_candidates_is_unrecoverable():
|
||||
assert pick_folio([]) is None
|
||||
|
||||
|
||||
def test_confirmed_riboflavin_subscript_conflict_resolved_by_size():
|
||||
# exact (text, size) pairs read from physical page 1243's header band:
|
||||
# the real folio "1244" (size 10.0, matching the rest of the running
|
||||
# header) and the "2" subscript from "Vitamin B2" (size 5.83), which
|
||||
# happens to fall in the same y<60 header band because the RIBOFLAVIN
|
||||
# title sits high on the page — see module docstring. Silently dropped
|
||||
# the whole monograph before this fix, confirmed via a whole-book
|
||||
# `cli validate` run and by rendering the page to an image.
|
||||
candidates = [("1244", 10.0), ("2", 5.83)]
|
||||
assert pick_folio(candidates) == 1244
|
||||
|
||||
|
||||
def test_genuine_same_size_conflict_still_returns_none():
|
||||
# two same-size digit-only candidates: real ambiguity, must not guess
|
||||
candidates = [("101", 10.0), ("205", 10.0)]
|
||||
assert pick_folio(candidates) is None
|
||||
@@ -0,0 +1,67 @@
|
||||
from ingestion.extract.spans import classify_column, _sort_blocks_reading_order
|
||||
|
||||
|
||||
def _block(x0, y0, x1, y1):
|
||||
return {"bbox": (x0, y0, x1, y1)}
|
||||
|
||||
|
||||
def testclassify_column_left():
|
||||
assert classify_column((35.0, 100.0, 280.0, 120.0)) == "left"
|
||||
|
||||
|
||||
def testclassify_column_right():
|
||||
assert classify_column((299.0, 100.0, 553.0, 120.0)) == "right"
|
||||
|
||||
|
||||
def testclassify_column_full_width_header():
|
||||
assert classify_column((35.0, 34.0, 552.0, 48.0)) == "full_width"
|
||||
|
||||
|
||||
def testclassify_column_none_bbox_is_unknown():
|
||||
assert classify_column(None) == "unknown"
|
||||
|
||||
|
||||
def test_confirmed_real_oxymetazolin_page_reversed_order_is_corrected():
|
||||
# exact bboxes from physical page 1100 (the OXYBUTYNIN/OXYMETAZOLIN
|
||||
# boundary — see spans.py module docstring): PyMuPDF's raw block order
|
||||
# is [header, right x7, left x8], right column before left. An earlier
|
||||
# version of this module trusted that raw order, silently attributing
|
||||
# OXYMETAZOLIN's "Chống chỉ định" (right column) to the still-open
|
||||
# OXYBUTYNIN monograph. Confirmed via a whole-book cli validate run,
|
||||
# a whole-document cross-tool character-diff, and rendering the page.
|
||||
raw_order = [
|
||||
_block(34.96, 34.39, 552.10, 47.72), # 0: full_width header
|
||||
_block(299.39, 60.46, 553.72, 121.96), # 1: right
|
||||
_block(299.39, 124.33, 553.72, 368.94), # 2: right
|
||||
_block(299.39, 371.31, 553.72, 408.39), # 3: right
|
||||
_block(35.43, 60.77, 289.77, 330.56), # 4: left (Xử trí: ...)
|
||||
_block(35.43, 379.21, 231.23, 391.87), # 5: left (Tên chung quốc tế)
|
||||
]
|
||||
sorted_blocks = _sort_blocks_reading_order(raw_order)
|
||||
columns_in_order = [classify_column(b["bbox"]) for b in sorted_blocks]
|
||||
assert columns_in_order == ["full_width", "left", "left", "right", "right", "right"]
|
||||
|
||||
|
||||
def test_already_correct_order_is_left_unchanged_in_content():
|
||||
blocks = [
|
||||
_block(35.0, 60.0, 280.0, 100.0), # left
|
||||
_block(35.0, 110.0, 280.0, 150.0), # left, further down
|
||||
_block(299.0, 60.0, 553.0, 100.0), # right
|
||||
]
|
||||
sorted_blocks = _sort_blocks_reading_order(blocks)
|
||||
assert sorted_blocks == blocks
|
||||
|
||||
|
||||
def test_a_narrow_box_between_the_columns_belongs_to_the_right_column():
|
||||
"""The two tolerance bands overlap between x=288 and x=319.
|
||||
|
||||
Testing left first put everything in that strip in the left column. It is
|
||||
invisible for a full-width block and wrong for a narrow one: a single 4pt
|
||||
glyph at x=315 on physical page 714 was classified left, so the 'ổ'
|
||||
missing from "Độ ổn định" could not be matched to its own line and the
|
||||
corruption survived the repair.
|
||||
"""
|
||||
assert classify_column((313.7, 506.2, 317.8, 514.8)) == "right"
|
||||
assert classify_column((35.4, 500.0, 289.7, 510.0)) == "left"
|
||||
# a box that lands in neither range still resolves by tolerance
|
||||
assert classify_column((300.0, 500.0, 305.0, 510.0)) == "left"
|
||||
@@ -0,0 +1,95 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.normalize import (
|
||||
PUA_SUBSTITUTIONS,
|
||||
find_unmapped_pua,
|
||||
group_visual_lines,
|
||||
join_spans,
|
||||
substitute_pua,
|
||||
)
|
||||
|
||||
|
||||
def _span(text, *, page=100, block=0, line=0, index=0, x0=50.0, x1=None, y0=100.0):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column="left",
|
||||
block=block, line=line, span_index=index,
|
||||
x0=x0, y0=y0, x1=(x0 + len(text) * 4.5) if x1 is None else x1, y1=y0 + 10,
|
||||
text=text, font="Tiger", size=9.5,
|
||||
)
|
||||
|
||||
|
||||
def test_pua_map_covers_every_codepoint_confirmed_in_the_corpus():
|
||||
# all 8 were located in the source PDF, rendered, and read visually —
|
||||
# see docs/progress-log.md for the page each was confirmed on
|
||||
assert PUA_SUBSTITUTIONS[""] == "≥"
|
||||
assert PUA_SUBSTITUTIONS[""] == "≤"
|
||||
assert PUA_SUBSTITUTIONS[""] == "α"
|
||||
assert PUA_SUBSTITUTIONS[""] == "→"
|
||||
assert PUA_SUBSTITUTIONS[""] == "®"
|
||||
assert PUA_SUBSTITUTIONS[""] == "₁"
|
||||
assert PUA_SUBSTITUTIONS[""] == "↓"
|
||||
assert PUA_SUBSTITUTIONS[""] == "γ"
|
||||
|
||||
|
||||
def test_comparison_operators_in_real_dosing_sentences_are_restored():
|
||||
# the clinically dangerous case: without this, "liều ≤ 100 mg" reaches
|
||||
# embeddings as "liều 100 mg" and the operator is lost
|
||||
assert substitute_pua("trẻ em 10 tuổi") == "trẻ em ≥ 10 tuổi"
|
||||
assert substitute_pua("liều 100 mg") == "liều ≤ 100 mg"
|
||||
|
||||
|
||||
def test_unmapped_pua_is_reported_not_silently_passed_through():
|
||||
assert find_unmapped_pua("liều 100 mg") == []
|
||||
assert find_unmapped_pua("bất ngờ đây") == [""]
|
||||
|
||||
|
||||
def test_subscript_span_rejoins_without_a_spurious_space():
|
||||
# real corpus case: "cytochrom P450" arrived as "cytochrom P\n450\ngây"
|
||||
spans = [
|
||||
_span("cytochrom P", x0=50.0, x1=100.0),
|
||||
_span("450", x0=100.2, x1=110.0),
|
||||
_span(" gây chuyển hóa.", x0=110.1, x1=180.0),
|
||||
]
|
||||
assert join_spans(spans) == "cytochrom P450 gây chuyển hóa."
|
||||
|
||||
|
||||
def test_italic_run_inside_parentheses_rejoins_on_one_line():
|
||||
# real corpus case: "(\nfeline immunodeficiency virus\n)"
|
||||
spans = [
|
||||
_span("(", x0=50.0, x1=53.0),
|
||||
_span("feline immunodeficiency virus", x0=53.1, x1=180.0),
|
||||
_span(")", x0=180.1, x1=183.0),
|
||||
]
|
||||
assert join_spans(spans) == "(feline immunodeficiency virus)"
|
||||
|
||||
|
||||
def test_wrap_without_sentence_end_is_joined_with_a_space():
|
||||
spans = [
|
||||
_span("không nhai. Nếu", line=0, y0=100.0),
|
||||
_span("uống viên thuốc", line=1, y0=112.0),
|
||||
]
|
||||
assert join_spans(spans) == "không nhai. Nếu uống viên thuốc"
|
||||
|
||||
|
||||
def test_sentence_end_keeps_the_line_break():
|
||||
spans = [
|
||||
_span("Liều người lớn: 10 mg.", line=0, y0=100.0),
|
||||
_span("Trẻ em: 5 mg.", line=1, y0=112.0),
|
||||
]
|
||||
assert join_spans(spans) == "Liều người lớn: 10 mg.\nTrẻ em: 5 mg."
|
||||
|
||||
|
||||
def test_wide_gap_on_one_line_still_yields_a_space():
|
||||
spans = [
|
||||
_span("Người bệnh", x0=50.0, x1=100.0),
|
||||
_span("100 kg", x0=104.0, x1=130.0),
|
||||
]
|
||||
assert join_spans(spans) == "Người bệnh 100 kg"
|
||||
|
||||
|
||||
def test_visual_lines_group_by_pymupdf_block_and_line_indices():
|
||||
spans = [
|
||||
_span("a", block=0, line=0), _span("b", block=0, line=0),
|
||||
_span("c", block=0, line=1),
|
||||
_span("d", block=1, line=0),
|
||||
]
|
||||
assert [len(g) for g in group_visual_lines(spans)] == [2, 1, 1]
|
||||
@@ -0,0 +1,332 @@
|
||||
import pytest
|
||||
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.assembler import DuplicateDrugIdError, assemble
|
||||
|
||||
|
||||
def _span(text, page, y0, bold=True, size=9.5, printed=None, column="left"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=printed if printed is not None else page + 1,
|
||||
column=column, block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=("TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"), size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_basic_single_monograph_with_sections_and_body():
|
||||
spans = [
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir (Acyclovir-like).", 100, 92.0, bold=False),
|
||||
_span("Mã ATC:", 100, 104.0),
|
||||
_span("J05AF06", 100, 116.0, bold=False),
|
||||
_span("Chỉ định", 101, 60.0),
|
||||
_span("Điều trị nhiễm HIV.", 101, 72.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
m = monographs[0]
|
||||
assert m.drug_id == "abacavir"
|
||||
assert m.drug_name == "ABACAVIR"
|
||||
assert m.source_page_range == [100, 101]
|
||||
assert m.sections["ten_chung_quoc_te"].text == "Abacavir (Acyclovir-like)."
|
||||
assert m.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
|
||||
assert m.atc_codes == ["J05AF06"]
|
||||
assert m.atc_stated_absent is False
|
||||
|
||||
|
||||
def test_non_bold_combined_heading_value_span_confirmed_real_amitriptylin_case():
|
||||
# AMITRIPTYLIN's real "Mã ATC:" heading is a single non-bold span
|
||||
# combining label and value ("Mã ATC: N06AA09."), unlike Abacavir's
|
||||
# bold-label + separate-value spans — see outlier item 20.
|
||||
spans = [
|
||||
_span("AMITRIPTYLIN", 184, 60.0),
|
||||
_span("Tên chung quốc tế: ", 184, 85.0),
|
||||
_span("Amitriptyline.", 184, 85.2, bold=False),
|
||||
_span("Mã ATC: N06AA09.", 184, 100.0, bold=False),
|
||||
_span("Loại thuốc:", 184, 115.0),
|
||||
_span("Thuốc chống trầm cảm.", 184, 115.2, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.sections["ma_atc"].text == "N06AA09."
|
||||
assert m.atc_codes == ["N06AA09"]
|
||||
|
||||
|
||||
def test_atc_stated_absent_propagates():
|
||||
spans = [
|
||||
_span("ADIPIODON", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 72.0),
|
||||
_span("Adipiodon.", 100, 84.0, bold=False),
|
||||
_span("Mã ATC:", 100, 96.0),
|
||||
_span("Chưa có.", 100, 108.0, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.atc_codes == []
|
||||
assert m.atc_stated_absent is True
|
||||
|
||||
|
||||
def test_qualifier_line_disambiguates_same_name_monographs():
|
||||
# reproduces the confirmed real SALBUTAMOL case (outlier item 18):
|
||||
# same base title, disambiguated by a bold non-caps parenthesized line.
|
||||
spans = [
|
||||
_span("SALBUTAMOL", 1261, 60.0),
|
||||
_span("(Dùng trong hô hấp)", 1261, 72.0),
|
||||
_span("Tên chung quốc tế:", 1261, 84.0),
|
||||
_span("Salbutamol.", 1261, 96.0, bold=False),
|
||||
_span("Chỉ định", 1261, 108.0),
|
||||
_span("Điều trị hen.", 1261, 120.0, bold=False),
|
||||
_span("SALBUTAMOL", 1263, 60.0),
|
||||
_span("(Dùng trong sản khoa)", 1263, 72.0),
|
||||
_span("Tên chung quốc tế:", 1263, 84.0),
|
||||
_span("Salbutamol.", 1263, 96.0, bold=False),
|
||||
_span("Chỉ định", 1263, 108.0),
|
||||
_span("Điều trị dọa sinh non.", 1263, 120.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 2
|
||||
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
|
||||
assert monographs[0].drug_name == "SALBUTAMOL (Dùng trong hô hấp)"
|
||||
assert monographs[1].drug_id == "salbutamol_dung_trong_san_khoa"
|
||||
assert monographs[0].sections["chi_dinh"].text == "Điều trị hen."
|
||||
assert monographs[1].sections["chi_dinh"].text == "Điều trị dọa sinh non."
|
||||
|
||||
|
||||
def test_genuine_duplicate_drug_id_raises():
|
||||
spans = [
|
||||
_span("FOOBARDRUG", 200, 60.0),
|
||||
_span("Tên chung quốc tế:", 200, 72.0),
|
||||
_span("Foobardrug.", 200, 84.0, bold=False),
|
||||
_span("Chỉ định", 200, 96.0),
|
||||
_span("A.", 200, 108.0, bold=False),
|
||||
_span("FOOBARDRUG", 300, 60.0),
|
||||
_span("Tên chung quốc tế:", 300, 72.0),
|
||||
_span("Foobardrug.", 300, 84.0, bold=False),
|
||||
_span("Chỉ định", 300, 96.0),
|
||||
_span("B.", 300, 108.0, bold=False),
|
||||
]
|
||||
with pytest.raises(DuplicateDrugIdError):
|
||||
list(assemble(spans))
|
||||
|
||||
|
||||
def test_gonadotropin_wrap_does_not_falsely_trigger_duplicate_check():
|
||||
# regression: the multi-line wrap must merge BEFORE the duplicate check
|
||||
# runs, so this is never treated as two separate "GONADOTROPIN" titles
|
||||
spans = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 676.2354736328125),
|
||||
_span("Tên chung quốc tế:", 1371, 690.0),
|
||||
_span("Gonadorelin.", 1371, 700.0, bold=False),
|
||||
_span("Chỉ định", 1371, 712.0),
|
||||
_span("X.", 1371, 724.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_name == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
|
||||
|
||||
|
||||
def test_front_matter_before_first_monograph_is_ignored():
|
||||
spans = [
|
||||
_span("Some front matter heading", 5, 60.0, bold=False, printed=6),
|
||||
_span("random body text", 5, 72.0, bold=False, printed=6),
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir.", 100, 92.0, bold=False),
|
||||
_span("Chỉ định", 100, 104.0),
|
||||
_span("X.", 100, 116.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "abacavir"
|
||||
|
||||
|
||||
def test_empty_spans_yields_nothing():
|
||||
assert list(assemble([])) == []
|
||||
|
||||
|
||||
def test_table_header_false_positive_not_treated_as_monograph():
|
||||
# reproduces the confirmed real "HSV"/"CMV" table-column-header case
|
||||
# (outlier item 19, physical page 698, inside the Foscarnet natri
|
||||
# monograph's dosing table) — bold+all-caps+short, identical shape to a
|
||||
# real title, but never followed by "Tên chung quốc tế" before the next
|
||||
# real title. Must not be treated as a monograph boundary.
|
||||
spans = [
|
||||
_span("FOSCARNET NATRI", 690, 60.0),
|
||||
_span("Tên chung quốc tế:", 690, 80.0),
|
||||
_span("Foscarnet.", 690, 92.0, bold=False),
|
||||
_span("Chỉ định", 690, 104.0),
|
||||
_span("Điều trị CMV.", 690, 116.0, bold=False),
|
||||
_span("HSV", 698, 523.0),
|
||||
_span("HSV", 698, 523.0),
|
||||
_span("CMV", 698, 523.0),
|
||||
_span("CMV", 698, 523.0),
|
||||
_span("40 mg/kg cách nhau 12 giờ", 698, 540.0, bold=False),
|
||||
_span("ARTEMETHER", 700, 60.0),
|
||||
_span("Tên chung quốc tế:", 700, 80.0),
|
||||
_span("Artemether.", 700, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert [m.drug_id for m in monographs] == ["foscarnet_natri", "artemether"]
|
||||
# the table row's numbers/labels stay attached to Foscarnet's Chỉ định
|
||||
# section body (dropped from a dedicated section, which is fine — no
|
||||
# false monograph boundary is what matters here)
|
||||
assert "hsv" not in monographs[0].drug_id
|
||||
assert "cmv" not in monographs[0].drug_id
|
||||
|
||||
|
||||
def test_real_title_immediately_followed_by_anchor_is_kept():
|
||||
spans = [
|
||||
_span("ABACAVIR", 100, 60.0),
|
||||
_span("Tên chung quốc tế:", 100, 80.0),
|
||||
_span("Abacavir.", 100, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "abacavir"
|
||||
|
||||
|
||||
def test_real_title_with_qualifier_before_anchor_is_still_kept():
|
||||
# the anchor lookahead must tolerate one intervening qualifier-line
|
||||
# event (the SALBUTAMOL case), not just immediate adjacency
|
||||
spans = [
|
||||
_span("SALBUTAMOL", 1261, 60.0),
|
||||
_span("(Dùng trong hô hấp)", 1261, 72.0),
|
||||
_span("Tên chung quốc tế:", 1261, 84.0),
|
||||
_span("Salbutamol.", 1261, 96.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "salbutamol_dung_trong_ho_hap"
|
||||
|
||||
|
||||
def test_class_level_monograph_sub_heading_not_treated_as_own_monograph():
|
||||
# reproduces the confirmed real case (outlier item 21): "SIMVASTATIN" is
|
||||
# a bold+all-caps+short sub-heading *inside* the class-level "CÁC CHẤT
|
||||
# ỨC CHẾ HMG-CoA REDUCTASE" monograph, immediately followed by its own
|
||||
# "Liều lượng và cách dùng" but NOT by "Tên chung quốc tế" (that section
|
||||
# belongs only to the parent). Must stay folded into the parent, not
|
||||
# become its own monograph.
|
||||
spans = [
|
||||
_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 60.0, printed=285),
|
||||
_span("Tên chung quốc tế:", 284, 72.0, printed=285),
|
||||
_span("Simvastatin, Lovastatin.", 284, 84.0, bold=False, printed=285),
|
||||
_span("Chỉ định", 284, 96.0, printed=285),
|
||||
_span("Tăng lipid huyết.", 284, 108.0, bold=False, printed=285),
|
||||
_span("SIMVASTATIN", 285, 60.0, printed=286),
|
||||
_span("Liều lượng và cách dùng", 285, 72.0, printed=286),
|
||||
_span("Uống 10 - 20 mg mỗi tối.", 285, 84.0, bold=False, printed=286),
|
||||
_span("LOVASTATIN", 285, 96.0, printed=286),
|
||||
_span("Liều lượng và cách dùng", 285, 108.0, printed=286),
|
||||
_span("Uống 20 mg mỗi ngày.", 285, 120.0, bold=False, printed=286),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_name == "CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"
|
||||
# the sub-headings' own dosing text stays attached to the parent
|
||||
# monograph's content rather than vanishing or becoming new monographs
|
||||
assert "Uống 20 mg mỗi ngày." in monographs[0].sections["lieu_luong_va_cach_dung"].text
|
||||
|
||||
|
||||
def test_running_header_boilerplate_stripped_from_mid_section_body_confirmed_real_morphin_case():
|
||||
# exact confirmed real case: physical page 1008's running header
|
||||
# ("DTQGVN 2" / "1009" / "Morphin sulfat", all column="full_width",
|
||||
# y0~34, well inside the header band) falls squarely in the middle of
|
||||
# MORPHIN SULFAT's "Liều lượng và cách dùng" section, which spans the
|
||||
# page 1007->1008 boundary — see outlier-catalog item 13 / assembler.py
|
||||
# module docstring. Whole-corpus measured: 1,374/11,409 sections (12.0%)
|
||||
# affected before this fix, 671/682 monographs (98.4%) had at least one.
|
||||
spans = [
|
||||
_span("MORPHIN SULFAT", 1007, 60.0),
|
||||
_span("Tên chung quốc tế:", 1007, 80.0),
|
||||
_span("Morphini sulfas.", 1007, 92.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 1007, 700.0),
|
||||
_span("Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu", 1007, 785.4, bold=False),
|
||||
_span("DTQGVN 2", 1008, 34.6, bold=False, column="full_width"),
|
||||
_span("1009", 1008, 34.6, bold=False, column="full_width"),
|
||||
_span("Morphin sulfat", 1008, 34.4, bold=False, column="full_width"),
|
||||
_span("uống viên thuốc giải phóng chậm thì không được nghiền.", 1008, 60.8, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
section_text = m.sections["lieu_luong_va_cach_dung"].text
|
||||
assert "DTQGVN" not in section_text
|
||||
assert "1009" not in section_text
|
||||
# the two body spans are one sentence broken by a page boundary: "Nếu"
|
||||
# does not end a sentence, so normalize/text_flow rejoins them with a
|
||||
# space rather than preserving the PDF's visual wrap as a hard newline
|
||||
assert section_text == (
|
||||
"Với thuốc viên (viên nang hoặc viên nén) không nhai. Nếu "
|
||||
"uống viên thuốc giải phóng chậm thì không được nghiền."
|
||||
)
|
||||
|
||||
|
||||
def test_last_real_monograph_in_book_still_kept_near_end_of_input():
|
||||
# anchor lookahead must not require a "next title" to exist — the very
|
||||
# last monograph in the book has no following title at all
|
||||
spans = [
|
||||
_span("ZOLPIDEM", 1494, 60.0),
|
||||
_span("Tên chung quốc tế:", 1494, 80.0),
|
||||
_span("Zolpidem.", 1494, 92.0, bold=False),
|
||||
]
|
||||
monographs = list(assemble(spans))
|
||||
assert len(monographs) == 1
|
||||
assert monographs[0].drug_id == "zolpidem"
|
||||
|
||||
|
||||
def test_repeated_section_heading_appends_instead_of_overwriting():
|
||||
# measured real case: 33 monographs repeat a section heading (38
|
||||
# occurrences). CEFAMANDOL's "Liều lượng và cách dùng" resumes on
|
||||
# physical page 339 after a renal-dosing table; the old code replaced the
|
||||
# SectionSpan, destroying everything captured before the repeat — for
|
||||
# CEFAMANDOL that left the dosing section holding only the table.
|
||||
spans = [
|
||||
_span("CEFAMANDOL", 338, 60.0),
|
||||
_span("Tên chung quốc tế", 338, 80.0),
|
||||
_span("Cefamandolum.", 338, 92.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 338, 400.0),
|
||||
_span("Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần.", 338, 412.0, bold=False),
|
||||
_span("Liều lượng và cách dùng", 339, 200.0),
|
||||
_span("Suy thận: giảm liều theo độ thanh thải creatinin.", 339, 212.0, bold=False),
|
||||
]
|
||||
m = list(assemble(spans))[0]
|
||||
text = m.sections["lieu_luong_va_cach_dung"].text
|
||||
assert "Người lớn: 500 mg - 1 g, 4 - 8 giờ/lần." in text
|
||||
assert "Suy thận: giảm liều theo độ thanh thải creatinin." in text
|
||||
# the first heading stays the provenance anchor
|
||||
assert m.sections["lieu_luong_va_cach_dung"].heading.physical_page == 338
|
||||
|
||||
|
||||
def test_a_plain_label_line_under_a_heading_is_body_not_a_new_section():
|
||||
"""FLUOROURACIL, physical page 681 — verified by rendering the page.
|
||||
|
||||
The book prints "Thời kỳ mang thai" / "Chống chỉ định." and "Thời kỳ cho
|
||||
con bú" / "Chống chỉ định.". The body line matches the section vocabulary,
|
||||
so it was read as a heading and both sections came out empty — dropping
|
||||
the statement that fluorouracil is contraindicated in pregnancy and while
|
||||
breastfeeding.
|
||||
"""
|
||||
spans = [
|
||||
_span("FLUOROURACIL", 681, 60.0),
|
||||
_span("Tên chung quốc tế", 681, 80.0),
|
||||
_span("Fluorouracilum.", 681, 92.0, bold=False),
|
||||
_span("Chống chỉ định", 681, 110.0),
|
||||
_span("Suy tủy nặng.", 681, 122.0, bold=False),
|
||||
_span("Thời kỳ mang thai", 681, 140.0),
|
||||
_span("Chống chỉ định.", 681, 152.0, bold=False),
|
||||
_span("Thời kỳ cho con bú", 681, 170.0),
|
||||
_span("Chống chỉ định.", 681, 182.0, bold=False),
|
||||
]
|
||||
monograph = list(assemble(spans))[0]
|
||||
assert monograph.sections["thoi_ky_mang_thai"].text == "Chống chỉ định."
|
||||
assert monograph.sections["thoi_ky_cho_con_bu"].text == "Chống chỉ định."
|
||||
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
|
||||
|
||||
|
||||
def test_a_bold_label_line_still_opens_its_section():
|
||||
spans = [
|
||||
_span("FLUOROURACIL", 681, 60.0),
|
||||
_span("Tên chung quốc tế", 681, 80.0),
|
||||
_span("Fluorouracilum.", 681, 92.0, bold=False),
|
||||
_span("Chống chỉ định", 681, 110.0),
|
||||
_span("Suy tủy nặng.", 681, 122.0, bold=False),
|
||||
]
|
||||
monograph = list(assemble(spans))[0]
|
||||
assert monograph.sections["chong_chi_dinh"].text == "Suy tủy nặng."
|
||||
@@ -0,0 +1,144 @@
|
||||
from ingestion.segment.atc import extract_atc_codes, is_stated_absent, normalize_atc_candidate
|
||||
|
||||
|
||||
def test_stray_whitespace_split_j04a_c01_recovered():
|
||||
assert normalize_atc_candidate("J04A C01") == "J04AC01"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_n05b_a06_recovered():
|
||||
assert normalize_atc_candidate("N05B A06") == "N05BA06"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_l01x_x02_recovered():
|
||||
assert normalize_atc_candidate("L01X X02") == "L01XX02"
|
||||
|
||||
|
||||
def test_digit_letter_confusion_no3ax12_recovered():
|
||||
assert normalize_atc_candidate("NO3AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_digit_letter_confusion_jo1dc07_recovered():
|
||||
assert normalize_atc_candidate("JO1DC07") == "J01DC07"
|
||||
|
||||
|
||||
def test_clean_code_passes_through():
|
||||
assert normalize_atc_candidate("N03AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_garbage_not_recovered():
|
||||
assert normalize_atc_candidate("NOT AN ATC CODE") is None
|
||||
assert normalize_atc_candidate("") is None
|
||||
|
||||
|
||||
def test_stated_absent_chua_co():
|
||||
assert is_stated_absent("Mã ATC: Chưa có.") is True
|
||||
|
||||
|
||||
def test_stated_absent_khong_co():
|
||||
assert is_stated_absent("Không có.") is True
|
||||
|
||||
|
||||
def test_stated_present_not_flagged_absent():
|
||||
assert is_stated_absent("N03AX12") is False
|
||||
|
||||
|
||||
def test_extract_single_code():
|
||||
result = extract_atc_codes("N03AX12")
|
||||
assert result.codes == ["N03AX12"]
|
||||
assert result.stated_absent is False
|
||||
|
||||
|
||||
def test_extract_multi_code_insulin_style():
|
||||
result = extract_atc_codes("A10AB01, A10AC01, A10AD01")
|
||||
assert result.codes == ["A10AB01", "A10AC01", "A10AD01"]
|
||||
|
||||
|
||||
def test_extract_multi_code_with_noise_mixed_in():
|
||||
# one clean code, one noisy code recovered, matching the real corpus
|
||||
# pattern where a monograph has some clean and some noisy ATC entries
|
||||
result = extract_atc_codes("N03AX12, J04A C01")
|
||||
assert result.codes == ["N03AX12", "J04AC01"]
|
||||
|
||||
|
||||
def test_extract_stated_absent_returns_no_codes():
|
||||
result = extract_atc_codes("Mã ATC: Chưa có.")
|
||||
assert result.codes == []
|
||||
assert result.stated_absent is True
|
||||
|
||||
|
||||
def test_trailing_period_recovered_confirmed_real_abacavir_case():
|
||||
# real field text is "J05AF06." — a sentence-ending period, not part of
|
||||
# the code; an earlier version silently produced zero codes here.
|
||||
assert normalize_atc_candidate("J05AF06.") == "J05AF06"
|
||||
result = extract_atc_codes("J05AF06.")
|
||||
assert result.codes == ["J05AF06"]
|
||||
|
||||
|
||||
def test_species_annotation_stripped_confirmed_real_insulin_case():
|
||||
# annotation-stripping is extract_atc_codes's job (must run before the
|
||||
# comma/semicolon split, see below) — normalize_atc_candidate itself
|
||||
# only normalizes an already-isolated code token.
|
||||
result = extract_atc_codes("A10AB01 (người); A10AB02 (bò)")
|
||||
assert result.codes == ["A10AB01", "A10AB02"]
|
||||
|
||||
|
||||
def test_leading_colon_from_value_span_stripped_confirmed_real_alcuronium_case():
|
||||
# real field text for ALCURONIUM CLORID (physical page 152): the bold
|
||||
# label span is "Mã ATC" with no colon, and the plain value span is
|
||||
# ": M03AA01." — the colon belongs to the value side here, not the
|
||||
# label side (Abacavir's equivalent has it on the label side instead:
|
||||
# "Mã ATC: " + "J05AF06."). See atc.py module docstring, defect 5.
|
||||
assert normalize_atc_candidate(": M03AA01.") == "M03AA01"
|
||||
result = extract_atc_codes(": M03AA01.")
|
||||
assert result.codes == ["M03AA01"]
|
||||
|
||||
|
||||
def test_name_prefixed_code_stripped_confirmed_real_arginin_case():
|
||||
# real field text for ARGININ (physical page 204): two salt forms, each
|
||||
# its own "Name: CODE" line, not a bare code — see atc.py module
|
||||
# docstring, defect 6.
|
||||
assert normalize_atc_candidate("Arginin glutamat: A05BA01") == "A05BA01"
|
||||
result = extract_atc_codes("Arginin glutamat: A05BA01\nArginin hydroclorid: B05XB01")
|
||||
assert result.codes == ["A05BA01", "B05XB01"]
|
||||
|
||||
|
||||
def test_plain_code_with_no_colon_still_normalizes():
|
||||
assert normalize_atc_candidate("N03AX12") == "N03AX12"
|
||||
|
||||
|
||||
def test_reversed_code_first_shape_confirmed_real_hmg_coa_case():
|
||||
# real field text for CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE (physical page
|
||||
# 284): each statin is "CODE: Name", the opposite order from the
|
||||
# "Name: CODE" shape above — see atc.py module docstring, defect 7.
|
||||
# "C10A A01" also has the already-fixed stray-whitespace split.
|
||||
assert normalize_atc_candidate("C10A A01: Simvastatin") == "C10AA01"
|
||||
result = extract_atc_codes("C10A A01: Simvastatin\nC10A A02: Lovastatin")
|
||||
assert result.codes == ["C10AA01", "C10AA02"]
|
||||
|
||||
|
||||
def test_annotation_containing_a_comma_does_not_break_the_split_confirmed_vaccine_case():
|
||||
# real field text for VẮC XIN SỞI (physical page 1437): the English
|
||||
# annotation "(Measles, live attenuated)" contains its own comma. An
|
||||
# earlier version split on "," *before* stripping the annotation,
|
||||
# breaking "J07BD01 (Measles, live attenuated)." into two unrecoverable
|
||||
# fragments and silently returning zero codes — see atc.py module
|
||||
# docstring, defect 4.
|
||||
result = extract_atc_codes("J07BD01 (Measles, live attenuated).")
|
||||
assert result.codes == ["J07BD01"]
|
||||
|
||||
|
||||
def test_extract_all_20_insulin_codes_from_real_field_text():
|
||||
# exact real field text for INSULIN (physical page 809) — see atc.py
|
||||
# module docstring; confirms the fix recovers all 20, not just 2.
|
||||
field_text = (
|
||||
"A10AB01 (người); A10AB02 (bò); A10AB03 (lợn);\n"
|
||||
"A10AB04 (lispro); A10AB05 (aspart); A10AB06 (glulisin);\n"
|
||||
"A10AC01 (người); A10AC02 (bò); A10AC03 (lợn); A10AC04\n"
|
||||
"(lispro); A10AD01 (người), A10AD02 (bò), A10AD03 (lợn),\n"
|
||||
"A10AD04 (lispro), A10AE01 (người); A10AE02 (bò); A10AE03\n"
|
||||
"(lợn); A10AE04 (glargin); A10AE05 (detemir), A10AF01 (người)."
|
||||
)
|
||||
result = extract_atc_codes(field_text)
|
||||
assert len(result.codes) == 20
|
||||
assert "A10AB01" in result.codes
|
||||
assert "A10AF01" in result.codes
|
||||
@@ -0,0 +1,88 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.detector import detect_monograph_titles, detect_section_headings
|
||||
|
||||
|
||||
def _span(text, physical_page, printed_page, y0=100.0, bold=True, size=10.0):
|
||||
font = "TimesNewRomanPS-BoldMT" if bold else "TimesNewRomanPSMT"
|
||||
return Span(
|
||||
physical_page=physical_page, printed_page=printed_page, column="left",
|
||||
block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=font, size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_confirmed_part_divider_excluded_at_page_99_boundary():
|
||||
# "CÁC CHUYÊN LUẬN THUỐC" at physical page 98 / printed 99 — bold,
|
||||
# all-caps, short: identical shape to a real title, must be excluded.
|
||||
spans = [_span("CÁC CHUYÊN LUẬN THUỐC", 98, 99), _span("ABACAVIR", 100, 101)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["ABACAVIR"]
|
||||
|
||||
|
||||
def test_monograph_title_outside_page_range_excluded():
|
||||
# bold all-caps short text in front matter (e.g. an org name) must not
|
||||
# be picked up — scoping to printed 99-1496 is required, not optional.
|
||||
spans = [_span("BỘ Y TẾ", 2, 3), _span("ABACAVIR", 100, 101)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["ABACAVIR"]
|
||||
|
||||
|
||||
def test_gonadotropin_wrap_detected_as_one_title():
|
||||
spans = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 1372, y0=664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 1372, y0=676.2354736328125),
|
||||
]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"]
|
||||
|
||||
|
||||
def test_non_bold_all_caps_text_not_a_title_candidate():
|
||||
spans = [_span("NOT BOLD BUT CAPS", 100, 101, bold=False)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_lowercase_bold_text_not_a_title_candidate():
|
||||
spans = [_span("Abacavir", 100, 101)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_short_section_label_with_normal_diacritic_not_a_title_candidate():
|
||||
# regression: an earlier absolute-count (not ratio) version of the
|
||||
# mixed-case tolerance let "Mã ATC:" through as a false title candidate
|
||||
# — its single lowercase diacritic ('ã') is normal Vietnamese
|
||||
# orthography, not a HMG-CoA-style embedded abbreviation. A ratio
|
||||
# threshold correctly rejects this short label (1/5 = 20% lowercase)
|
||||
# while still accepting the long HMG-CoA title (1/27 = 3.7%).
|
||||
spans = [_span("Mã ATC:", 100, 101)]
|
||||
assert list(detect_monograph_titles(spans)) == []
|
||||
|
||||
|
||||
def test_confirmed_hmg_coa_mixed_case_title_still_detected():
|
||||
# "CoA" (Coenzyme A) is a real mixed-case abbreviation embedded in an
|
||||
# otherwise all-caps title — outlier item 21. A strict isupper() check
|
||||
# silently dropped this entire class-level monograph from the corpus.
|
||||
spans = [_span("CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE", 284, 285)]
|
||||
titles = [h.text for h in detect_monograph_titles(spans)]
|
||||
assert titles == ["CÁC CHẤT ỨC CHẾ HMG-CoA REDUCTASE"]
|
||||
|
||||
|
||||
def test_section_heading_matched_with_and_without_trailing_colon():
|
||||
spans = [
|
||||
_span("Tên chung quốc tế:", 100, 101, bold=True, size=9.5),
|
||||
_span("Chỉ định", 100, 101, bold=True, size=9.5),
|
||||
]
|
||||
headings = list(detect_section_headings(spans))
|
||||
assert [h.section_key for h in headings] == ["ten_chung_quoc_te", "chi_dinh"]
|
||||
|
||||
|
||||
def test_unknown_bold_text_not_matched_as_section():
|
||||
# e.g. "Cách dùng:" — a real sub-heading within "Liều lượng và cách
|
||||
# dùng" that is NOT one of the known top-level section names.
|
||||
spans = [_span("Cách dùng:", 100, 101, bold=True, size=9.5)]
|
||||
assert list(detect_section_headings(spans)) == []
|
||||
|
||||
|
||||
def test_section_heading_outside_monograph_range_excluded():
|
||||
spans = [_span("Chỉ định", 5, 6, bold=True, size=9.5)]
|
||||
assert list(detect_section_headings(spans)) == []
|
||||
@@ -0,0 +1,72 @@
|
||||
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan
|
||||
|
||||
|
||||
def test_round_trip_preserves_all_fields(tmp_path):
|
||||
heading = Heading(text="Chỉ định", physical_page=100, y0=80.0, is_monograph_title=False, section_key="chi_dinh")
|
||||
section = SectionSpan(key="chi_dinh", display_name="Chỉ định", heading=heading, text="Điều trị nhiễm HIV.")
|
||||
monograph = Monograph(
|
||||
drug_id="abacavir", drug_name="ABACAVIR", source_page_range=[100, 101],
|
||||
sections={"chi_dinh": section}, atc_codes=["J05AF06"], atc_stated_absent=False,
|
||||
)
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
count = write_monographs_jsonl([monograph], path)
|
||||
assert count == 1
|
||||
|
||||
result = list(read_monographs_jsonl(path))
|
||||
assert len(result) == 1
|
||||
r = result[0]
|
||||
assert r.drug_id == "abacavir"
|
||||
assert r.drug_name == "ABACAVIR"
|
||||
assert r.source_page_range == [100, 101]
|
||||
assert r.atc_codes == ["J05AF06"]
|
||||
assert r.sections["chi_dinh"].text == "Điều trị nhiễm HIV."
|
||||
assert r.sections["chi_dinh"].heading.section_key == "chi_dinh"
|
||||
|
||||
|
||||
def test_multiple_monographs_round_trip(tmp_path):
|
||||
m1 = Monograph(drug_id="a", drug_name="A", source_page_range=[1, 2])
|
||||
m2 = Monograph(drug_id="b", drug_name="B", source_page_range=[3, 4])
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
write_monographs_jsonl([m1, m2], path)
|
||||
result = list(read_monographs_jsonl(path))
|
||||
assert [r.drug_id for r in result] == ["a", "b"]
|
||||
|
||||
|
||||
def test_empty_write_produces_empty_file(tmp_path):
|
||||
path = tmp_path / "monographs.jsonl"
|
||||
count = write_monographs_jsonl([], path)
|
||||
assert count == 0
|
||||
assert list(read_monographs_jsonl(path)) == []
|
||||
|
||||
|
||||
def test_table_blocks_survive_a_write_read_round_trip(tmp_path):
|
||||
# the lifted table blocks were being computed in memory and then dropped
|
||||
# at the file boundary — 148 blocks existed in the run summary but the
|
||||
# JSONL had no "tables" key at all
|
||||
from ingestion.segment.models import Heading, Monograph, SectionSpan, TableBlock
|
||||
from ingestion.segment.io import read_monographs_jsonl, write_monographs_jsonl
|
||||
|
||||
heading = Heading(text="Liều lượng và cách dùng", physical_page=339, y0=200.0,
|
||||
is_monograph_title=False, section_key="lieu_luong_va_cach_dung")
|
||||
m = Monograph(
|
||||
drug_id="cefamandol", drug_name="CEFAMANDOL", source_page_range=[338, 340],
|
||||
sections={"lieu_luong_va_cach_dung": SectionSpan(
|
||||
key="lieu_luong_va_cach_dung", display_name="Liều lượng và cách dùng",
|
||||
heading=heading, text="Cách dùng ...")},
|
||||
tables=[TableBlock(
|
||||
table_id="p339_t0", shape="simple_table", physical_page=339,
|
||||
bbox=[40.0, 380.0, 400.0, 620.0],
|
||||
section_key="lieu_luong_va_cach_dung",
|
||||
text="80 - 50 750 mg - 2 g, 6 giờ/lần.", quarantined=True)],
|
||||
)
|
||||
path = tmp_path / "m.jsonl"
|
||||
write_monographs_jsonl([m], path)
|
||||
back = list(read_monographs_jsonl(path))[0]
|
||||
assert len(back.tables) == 1
|
||||
t = back.tables[0]
|
||||
assert t.table_id == "p339_t0"
|
||||
assert t.physical_page == 339
|
||||
assert t.bbox == [40.0, 380.0, 400.0, 620.0]
|
||||
assert t.quarantined is True
|
||||
assert "750 mg - 2 g" in t.text
|
||||
@@ -0,0 +1,134 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment.merge import merge_multiline_headings, merge_same_line_bold_fragments
|
||||
|
||||
|
||||
def _span(text, page, y0, size=9.5, font="TimesNewRomanPS-BoldMT"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column="right",
|
||||
block=0, line=0, span_index=0,
|
||||
x0=100.0, y0=y0, x1=200.0, y1=y0 + 12.0,
|
||||
text=text, font=font, size=size,
|
||||
)
|
||||
|
||||
|
||||
def test_confirmed_gonadotropin_wrap_merges_into_one_heading():
|
||||
# exact bboxes from physical page 1371 (0-indexed) — see module docstring
|
||||
candidates = [
|
||||
_span("THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG", 1371, 664.4554443359375),
|
||||
_span("GONADOTROPIN", 1371, 676.2354736328125),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "THUỐC TƯƠNG TỰ HORMON GIẢI PHÓNG GONADOTROPIN"
|
||||
|
||||
|
||||
def test_unrelated_single_line_titles_on_different_pages_not_merged():
|
||||
candidates = [
|
||||
_span("GONADOTROPIN", 755, 200.0),
|
||||
_span("HYDROCORTISON", 900, 300.0),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 2
|
||||
assert [h.text for h in headings] == ["GONADOTROPIN", "HYDROCORTISON"]
|
||||
|
||||
|
||||
def test_large_y_gap_on_same_page_not_merged():
|
||||
# two genuinely separate single-line titles far apart on the same page
|
||||
# (e.g. two short monographs stacked in one column) must not merge
|
||||
candidates = [
|
||||
_span("ATENOLOL", 219, 100.0),
|
||||
_span("ATRACURIUM BESYLAT", 219, 500.0),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 2
|
||||
|
||||
|
||||
def test_confirmed_aciclovir_same_line_split_merges_without_space():
|
||||
# exact bboxes from physical page 113 (0-indexed), found by rendering the
|
||||
# page to an image and reading it directly: "ACIC" (size 10.0) and
|
||||
# "LOVIR" (size 9.5) are one word split into two spans on the same
|
||||
# visual line — different font size, ~0.5pt y0 gap, near-zero x-gap.
|
||||
# Must merge WITHOUT a space ("ACICLOVIR", not "ACIC LOVIR") — see
|
||||
# module docstring.
|
||||
candidates = [
|
||||
_span("ACIC", 113, 515.1914672851562, size=10.0),
|
||||
_span("LOVIR", 113, 515.7044677734375, size=9.5),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "ACICLOVIR"
|
||||
|
||||
|
||||
def test_wrap_and_same_line_split_use_different_join_characters():
|
||||
# a genuine line-wrap (large y-gap) still joins with a space even when
|
||||
# font size differs, since size is no longer part of the merge decision
|
||||
candidates = [
|
||||
_span("FIRST LINE", 100, 200.0, size=10.0),
|
||||
_span("SECOND LINE", 100, 212.0, size=9.5),
|
||||
]
|
||||
headings = list(merge_multiline_headings(candidates))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "FIRST LINE SECOND LINE"
|
||||
|
||||
|
||||
def test_single_candidate_yields_one_heading():
|
||||
headings = list(merge_multiline_headings([_span("ABACAVIR", 100, 60.29)]))
|
||||
assert len(headings) == 1
|
||||
assert headings[0].text == "ABACAVIR"
|
||||
|
||||
|
||||
def test_empty_input_yields_nothing():
|
||||
assert list(merge_multiline_headings([])) == []
|
||||
|
||||
|
||||
def test_confirmed_ten_chung_quoc_te_diacritic_split_reassembles():
|
||||
# exact fragments + y0 from physical page 759's "GUAIFENESIN" monograph,
|
||||
# found via a whole-book `cli validate` run (the monograph was silently
|
||||
# dropped because "Tên chung quốc tế" never matched the section
|
||||
# vocabulary) and confirmed by rendering the page to an image: to a
|
||||
# human reader the line looks completely normal, but PyMuPDF splits it
|
||||
# into 5 spans around the diacritic characters — see module docstring.
|
||||
fragments = [
|
||||
_span("Tên chung qu", 759, 157.614),
|
||||
_span("ố", 759, 157.33),
|
||||
_span("c t", 759, 157.614),
|
||||
_span("ế", 759, 157.33),
|
||||
_span(": ", 759, 157.614),
|
||||
]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 1
|
||||
assert merged[0].text == "Tên chung quốc tế: "
|
||||
|
||||
|
||||
def test_non_bold_spans_pass_through_unmerged():
|
||||
fragments = [
|
||||
_span("Guaifenesin", 759, 157.24, font="TimesNewRomanPSMT"),
|
||||
_span(".", 759, 157.24, font="TimesNewRomanPSMT"),
|
||||
]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 2
|
||||
|
||||
|
||||
def test_bold_spans_on_different_lines_not_merged():
|
||||
fragments = [_span("Chỉ định", 100, 200.0), _span("Chống chỉ định", 100, 220.0)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert len(merged) == 2
|
||||
|
||||
|
||||
def test_merged_span_keeps_provenance_of_first_fragment():
|
||||
fragments = [_span("Tên chung qu", 759, 157.614), _span("ố", 759, 157.33)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert merged[0].physical_page == 759
|
||||
assert merged[0].printed_page == 760
|
||||
assert merged[0].x0 == fragments[0].x0
|
||||
assert merged[0].x1 == fragments[-1].x1
|
||||
|
||||
|
||||
def test_single_bold_span_passes_through_unchanged():
|
||||
fragments = [_span("ABACAVIR", 100, 60.29)]
|
||||
merged = merge_same_line_bold_fragments(fragments)
|
||||
assert merged == fragments
|
||||
|
||||
|
||||
def test_empty_input_to_same_line_merge_yields_nothing():
|
||||
assert merge_same_line_bold_fragments([]) == []
|
||||
@@ -0,0 +1,108 @@
|
||||
from ingestion.extract.models import Span
|
||||
from ingestion.segment import assemble
|
||||
from ingestion.tables import SHAPE_GRID_2D, SHAPE_SIMPLE, TableRegion, index_by_page
|
||||
|
||||
|
||||
def _span(text, page, y0, *, bold=False, x0=50.0, block=0, line=0, column="left"):
|
||||
return Span(
|
||||
physical_page=page, printed_page=page + 1, column=column,
|
||||
block=block, line=line, span_index=0,
|
||||
x0=x0, y0=y0, x1=x0 + len(text) * 4.5, y1=y0 + 10,
|
||||
text=text, font="Tiger-Bold" if bold else "Tiger", size=9.5,
|
||||
)
|
||||
|
||||
|
||||
def _monograph_spans(extra):
|
||||
return [
|
||||
_span("PARACETAMOL", 109, 60.0, bold=True),
|
||||
_span("Tên chung quốc tế", 109, 80.0, bold=True),
|
||||
_span("Paracetamolum.", 109, 92.0),
|
||||
_span("Dạng thuốc và hàm lượng", 109, 200.0, bold=True),
|
||||
] + extra
|
||||
|
||||
|
||||
def test_table_spans_are_lifted_out_of_section_prose():
|
||||
# real measured case: physical page 109's dosage-form table was being
|
||||
# concatenated cell by cell into the section body
|
||||
# ('Viên nén' + '1' + '1 - 4' + '8 - 12' + 'Viên nang tác' ...)
|
||||
spans = _monograph_spans([
|
||||
_span("Thuốc dùng đường uống.", 109, 220.0),
|
||||
_span("Viên nén", 109, 400.0, block=5),
|
||||
_span("1", 109, 400.0, block=5, x0=200.0),
|
||||
_span("1 - 4", 109, 400.0, block=5, x0=260.0),
|
||||
_span("Sau khi uống hấp thu nhanh.", 109, 600.0, block=9),
|
||||
])
|
||||
# the region must cover the table's first column too — it starts at the
|
||||
# left margin, same x as body prose
|
||||
region = TableRegion("p109_t0", 109, (40.0, 380.0, 400.0, 460.0), 3, 3, SHAPE_SIMPLE)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
|
||||
body = m.sections["dang_thuoc_va_ham_luong"].text
|
||||
assert "Viên nén" not in body
|
||||
assert "1 - 4" not in body
|
||||
assert "Thuốc dùng đường uống." in body
|
||||
assert "Sau khi uống hấp thu nhanh." in body
|
||||
|
||||
assert len(m.tables) == 1
|
||||
block = m.tables[0]
|
||||
assert block.table_id == "p109_t0"
|
||||
assert "Viên nén" in block.text and "1 - 4" in block.text
|
||||
assert block.section_key == "dang_thuoc_va_ham_luong"
|
||||
assert block.physical_page == 109
|
||||
# every multi-column table is quarantined until a real row/column
|
||||
# reconstruction exists — its linearised text is not safe to cite as prose
|
||||
assert block.quarantined is True
|
||||
|
||||
|
||||
def test_without_a_region_map_behaviour_is_unchanged():
|
||||
spans = _monograph_spans([
|
||||
_span("Thuốc dùng đường uống.", 109, 220.0),
|
||||
_span("Viên nén", 109, 400.0, block=5),
|
||||
])
|
||||
m = list(assemble(spans))[0]
|
||||
assert m.tables == []
|
||||
assert "Viên nén" in m.sections["dang_thuoc_va_ham_luong"].text
|
||||
|
||||
|
||||
def test_2d_grid_block_is_quarantined():
|
||||
# a 2D lookup grid's flattened text is meaningless without row/column
|
||||
# headers (outlier item 7) — it must be marked, not silently embedded
|
||||
spans = _monograph_spans([_span("0,52", 109, 400.0, block=5, x0=200.0)])
|
||||
region = TableRegion("p109_t1", 109, (150.0, 380.0, 400.0, 460.0), 6, 5, SHAPE_GRID_2D)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
assert len(m.tables) == 1
|
||||
assert m.tables[0].quarantined is True
|
||||
|
||||
|
||||
def test_non_table_regions_are_never_lifted():
|
||||
# the 17 full-page false positives must not swallow a whole page of prose
|
||||
spans = _monograph_spans([_span("Thuốc dùng đường uống.", 109, 220.0)])
|
||||
region = TableRegion("p109_t0", 109, (0.0, 0.0, 595.3, 836.2), 1, 2,
|
||||
"not_a_table_full_page")
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
assert m.tables == []
|
||||
assert "Thuốc dùng đường uống." in m.sections["dang_thuoc_va_ham_luong"].text
|
||||
|
||||
|
||||
def test_table_block_ids_stay_unique_when_a_section_resumes():
|
||||
# a region flushed twice (section closes, then resumes) must not emit two
|
||||
# blocks with the same table_id — provenance ids have to be unique
|
||||
spans = [
|
||||
_span("CEFAMANDOL", 339, 60.0, bold=True),
|
||||
_span("Tên chung quốc tế", 339, 80.0, bold=True),
|
||||
_span("Cefamandolum.", 339, 92.0),
|
||||
_span("Liều lượng và cách dùng", 339, 200.0, bold=True),
|
||||
_span("80 - 50", 339, 400.0, block=5),
|
||||
_span("Liều lượng và cách dùng", 339, 500.0, bold=True),
|
||||
_span("< 25 - 10", 339, 600.0, block=9),
|
||||
]
|
||||
region = TableRegion("p339_t0", 339, (40.0, 380.0, 400.0, 620.0), 5, 2, SHAPE_SIMPLE)
|
||||
m = list(assemble(spans, table_index=index_by_page([region])))[0]
|
||||
# table_id is deterministic per REGION, so two parts of one table share
|
||||
# it on purpose; table_part_id is the unique key, derived from the first
|
||||
# source span rather than a counter (a counter would renumber whenever
|
||||
# anything upstream shifted, hiding rather than identifying a duplicate)
|
||||
assert len({t.table_part_id for t in m.tables}) == len(m.tables)
|
||||
assert {t.continuation_group for t in m.tables} == {"p339_t0"}
|
||||
assert all(t.table_part_id.startswith("p339_t0@") for t in m.tables)
|
||||
assert all(t.quarantined for t in m.tables)
|
||||
@@ -0,0 +1,30 @@
|
||||
from ingestion.segment.units import normalize_unit_token, validate_unit_tokens
|
||||
|
||||
|
||||
def test_clean_unit_passes_through():
|
||||
assert normalize_unit_token("mg") == "mg"
|
||||
assert normalize_unit_token("mcg") == "mcg"
|
||||
assert normalize_unit_token("mmol") == "mmol"
|
||||
|
||||
|
||||
def test_stray_whitespace_split_recovered_by_analogy_to_atc():
|
||||
assert normalize_unit_token("m g") == "mg"
|
||||
assert normalize_unit_token("m cg") == "mcg"
|
||||
|
||||
|
||||
def test_case_insensitive():
|
||||
assert normalize_unit_token("MG") == "mg"
|
||||
|
||||
|
||||
def test_unknown_token_not_recovered():
|
||||
assert normalize_unit_token("xyz") is None
|
||||
assert normalize_unit_token("") is None
|
||||
|
||||
|
||||
def test_validate_unit_tokens_flags_only_bad_ones():
|
||||
bad = validate_unit_tokens(["mg", "mcg", "xyz", "ml"])
|
||||
assert bad == ["xyz"]
|
||||
|
||||
|
||||
def test_validate_unit_tokens_empty_when_all_valid():
|
||||
assert validate_unit_tokens(["mg", "mcg", "mmol"]) == []
|
||||
@@ -0,0 +1,83 @@
|
||||
from ingestion.segment.vocab import match_section, match_section_with_inline_value
|
||||
|
||||
|
||||
def test_exact_label_match_with_trailing_colon():
|
||||
d = match_section("Tên chung quốc tế:")
|
||||
assert d is not None and d.key == "ten_chung_quoc_te"
|
||||
|
||||
|
||||
def test_exact_label_match_without_trailing_colon():
|
||||
d = match_section("Chỉ định")
|
||||
assert d is not None and d.key == "chi_dinh"
|
||||
|
||||
|
||||
def test_inline_value_combined_span_confirmed_real_amitriptylin_case():
|
||||
# AMITRIPTYLIN's real "Mã ATC:" field is one non-bold span combining
|
||||
# label and value: "Mã ATC: N06AA09." — see outlier item 20.
|
||||
result = match_section_with_inline_value("Mã ATC: N06AA09.")
|
||||
assert result is not None
|
||||
section_def, value = result
|
||||
assert section_def.key == "ma_atc"
|
||||
assert value == "N06AA09."
|
||||
|
||||
|
||||
def test_inline_value_not_matched_when_no_colon_follows():
|
||||
assert match_section_with_inline_value("Mã ATC something else entirely") is None
|
||||
|
||||
|
||||
def test_inline_value_does_not_confuse_plain_body_text():
|
||||
assert match_section_with_inline_value("Bệnh nhân cần theo dõi chặt chẽ.") is None
|
||||
|
||||
|
||||
def test_exact_match_takes_priority_over_prefix_for_label_only_span():
|
||||
d = match_section("Mã ATC:")
|
||||
assert d is not None and d.key == "ma_atc"
|
||||
|
||||
|
||||
def test_real_spelling_variants_found_in_the_book_all_match():
|
||||
# measured whole-corpus: 42 distinct near-miss heading strings, 542
|
||||
# occurrences, none of which matched before aliases were added. The
|
||||
# heaviest is "Thông tin qui chế" (469x) — the book prints "qui" where
|
||||
# its own documented template says "quy", which cost 586 of 682
|
||||
# monographs their thong_tin_quy_che section entirely.
|
||||
from ingestion.segment.vocab import match_section
|
||||
cases = {
|
||||
"Thông tin qui chế": "thong_tin_quy_che",
|
||||
"Thông tin về qui chế": "thong_tin_quy_che",
|
||||
"Thông tin và quy chế": "thong_tin_quy_che",
|
||||
"Mã ACT": "ma_atc",
|
||||
"Chống chỉ đinh": "chong_chi_dinh",
|
||||
"Thời kì mang thai": "thoi_ky_mang_thai",
|
||||
"Thời kì cho con bú": "thoi_ky_cho_con_bu",
|
||||
"Dược lí và cơ chế tác dụng": "duoc_ly_va_co_che_tac_dung",
|
||||
"Hướng dẫn cách sử trí ADR": "huong_dan_xu_tri_adr",
|
||||
"Quá liều và xử lý": "qua_lieu_va_xu_tri",
|
||||
"Lọai thuốc": "loai_thuoc",
|
||||
}
|
||||
for text, expected_key in cases.items():
|
||||
matched = match_section(text)
|
||||
assert matched is not None, f"{text!r} should match a section"
|
||||
assert matched.key == expected_key
|
||||
|
||||
|
||||
def test_typesetting_noise_is_folded_without_needing_an_alias_each():
|
||||
# missing/extra spaces and the Ð/Đ look-alike are handled by the lookup
|
||||
# key, not enumerated per-variant
|
||||
from ingestion.segment.vocab import match_section
|
||||
assert match_section("Chỉđịnh").key == "chi_dinh"
|
||||
assert match_section("Chống chỉđịnh").key == "chong_chi_dinh"
|
||||
assert match_section("Độổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
||||
assert match_section("Ðộ ổn định và bảo quản").key == "do_on_dinh_va_bao_quan"
|
||||
assert match_section("H ướng dẫn cách xử trí ADR").key == "huong_dan_xu_tri_adr"
|
||||
assert match_section("Tư ơng kỵ").key == "tuong_ky"
|
||||
assert match_section("Tác dụng khôngmong muốn (ADR)").key == "tac_dung_khong_mong_muon"
|
||||
assert match_section("Thận trọng.").key == "than_trong"
|
||||
|
||||
|
||||
def test_near_misses_that_are_not_sections_stay_unmatched():
|
||||
# "Thể trọng" is body weight, not "Thận trọng" (caution) — a 0.84
|
||||
# similarity that must NOT become an alias; the opioid string is a
|
||||
# drug-specific sub-heading inside a section, not the section itself
|
||||
from ingestion.segment.vocab import match_section
|
||||
assert match_section("Thể trọng") is None
|
||||
assert match_section("Tác dụng không mong muốn của opioid") is None
|
||||
@@ -0,0 +1,107 @@
|
||||
from ingestion.segment.models import Monograph
|
||||
from ingestion.validation.back_index import GroundTruthEntry
|
||||
from ingestion.validation.metrics import compute_recall_precision
|
||||
|
||||
|
||||
def _mono(drug_id, drug_name, start_physical):
|
||||
return Monograph(drug_id=drug_id, drug_name=drug_name, source_page_range=[start_physical, start_physical + 1])
|
||||
|
||||
|
||||
def test_perfect_match_recall_and_precision_are_one():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 1.0
|
||||
assert result.matched_count == 1
|
||||
|
||||
|
||||
def test_missed_ground_truth_entry_lowers_recall_not_precision():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [
|
||||
GroundTruthEntry(name="Abacavir", printed_page=101),
|
||||
GroundTruthEntry(name="Acarbose", printed_page=103),
|
||||
]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 0.5
|
||||
assert result.precision == 1.0
|
||||
assert len(result.unmatched_ground_truth) == 1
|
||||
assert result.unmatched_ground_truth[0].name == "Acarbose"
|
||||
|
||||
|
||||
def test_spurious_detected_monograph_lowers_precision_not_recall():
|
||||
monographs = [
|
||||
_mono("abacavir", "ABACAVIR", 100),
|
||||
_mono("cac_chuyen_luan_thuoc", "CÁC CHUYÊN LUẬN THUỐC", 98),
|
||||
]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=101)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 0.5
|
||||
assert len(result.unmatched_detected) == 1
|
||||
|
||||
|
||||
def test_page_tolerance_allows_small_offset():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=103)] # +2 tolerance
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
|
||||
|
||||
def test_page_beyond_tolerance_does_not_match():
|
||||
monographs = [_mono("abacavir", "ABACAVIR", 100)]
|
||||
ground_truth = [GroundTruthEntry(name="Abacavir", printed_page=110)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_qualifier_suffixed_name_still_matches_base_ground_truth_name():
|
||||
# SALBUTAMOL (Dùng trong hô hấp) should still match a ground-truth
|
||||
# entry that just says "Salbutamol"
|
||||
monographs = [_mono("salbutamol_dung_trong_ho_hap", "SALBUTAMOL (Dùng trong hô hấp)", 1261)]
|
||||
ground_truth = [GroundTruthEntry(name="Salbutamol", printed_page=1262)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
|
||||
|
||||
def test_empty_ground_truth_gives_zero_recall_not_error():
|
||||
result = compute_recall_precision([_mono("a", "A", 1)], [])
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_empty_monographs_gives_zero_precision_not_error():
|
||||
result = compute_recall_precision([], [GroundTruthEntry(name="A", printed_page=1)])
|
||||
assert result.precision == 0.0
|
||||
assert result.recall == 0.0
|
||||
|
||||
|
||||
def test_exact_match_preferred_over_substring_steal_confirmed_real_case():
|
||||
# Confirmed real case from a whole-book `cli validate` run: "ISOSORBID"
|
||||
# and "ISOSORBID DINITRAT" are two distinct, correctly-segmented
|
||||
# monographs a page apart. A pure substring match lets the shorter name
|
||||
# "steal" both ground-truth entries (it's a substring of the longer one
|
||||
# too) via `next()`'s order-dependent first match, leaving the real
|
||||
# "ISOSORBID DINITRAT" monograph spuriously unmatched even though an
|
||||
# exact match for it exists.
|
||||
monographs = [
|
||||
_mono("isosorbid", "ISOSORBID", 844),
|
||||
_mono("isosorbid_dinitrat", "ISOSORBID DINITRAT", 845),
|
||||
]
|
||||
ground_truth = [
|
||||
GroundTruthEntry(name="Isosorbid", printed_page=845),
|
||||
GroundTruthEntry(name="Isosorbid dinitrat", printed_page=846),
|
||||
]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
assert result.precision == 1.0
|
||||
assert len(result.unmatched_detected) == 0
|
||||
|
||||
|
||||
def test_double_space_in_detected_name_still_matches_confirmed_real_case():
|
||||
# confirmed real case from a whole-book `cli validate` run: "ALVERIN
|
||||
# CITRAT" (double space) failed to match ground truth's single-spaced
|
||||
# "Alverin citrat" under plain strip+upper comparison.
|
||||
monographs = [_mono("alverin_citrat", "ALVERIN CITRAT", 171)]
|
||||
ground_truth = [GroundTruthEntry(name="Alverin citrat", printed_page=172)]
|
||||
result = compute_recall_precision(monographs, ground_truth)
|
||||
assert result.recall == 1.0
|
||||
@@ -0,0 +1,124 @@
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from ingestion.extract import OutlinedTextRun
|
||||
from ingestion.tables import TableRegion
|
||||
from ingestion.validation import (
|
||||
FRACTION_BAR_CANDIDATE,
|
||||
HEADER_RULE,
|
||||
RULE_FRAGMENT,
|
||||
TABLE_FRAME,
|
||||
TEXT_AS_VECTOR_OUTLINE,
|
||||
UNCLASSIFIED,
|
||||
PageContext,
|
||||
ResidualRegion,
|
||||
classify,
|
||||
scan_page,
|
||||
)
|
||||
from ingestion.validation.residual_ink import FRACTION_BAR_CANDIDATE as BAR
|
||||
|
||||
PDF_PATH = Path(__file__).resolve().parents[1] / "data" / "raw" / (
|
||||
"duoc-thu-quoc-gia-viet-nam-2018.pdf"
|
||||
)
|
||||
needs_pdf = pytest.mark.skipif(not PDF_PATH.exists(), reason="source PDF not present")
|
||||
|
||||
|
||||
def _region(x0, y0, x1, y1, page=100, ink=500):
|
||||
return ResidualRegion(physical_page=page, bbox=(x0, y0, x1, y1), ink_px=ink)
|
||||
|
||||
|
||||
def test_running_header_rule_is_named_not_left_unclassified():
|
||||
# measured on real pages: a ~516pt wide, 0pt tall rule at y≈48-52 appears
|
||||
# on essentially every page of the book
|
||||
assert classify(_region(36.0, 48.5, 552.0, 48.5)) == HEADER_RULE
|
||||
|
||||
|
||||
def test_a_thin_bar_below_the_header_band_is_a_fraction_bar_candidate():
|
||||
# NETILMICIN, physical page 1042: the Cockcroft-Gault fraction bar
|
||||
assert classify(_region(97.9, 492.0, 286.5, 492.0)) == FRACTION_BAR_CANDIDATE
|
||||
|
||||
|
||||
def test_ink_inside_a_known_table_region_is_a_table_frame_not_a_formula():
|
||||
table = TableRegion(
|
||||
table_id="p202_t0", physical_page=202, bbox=(299.0, 189.6, 552.4, 300.5),
|
||||
n_rows=4, n_cols=3, shape="simple_table",
|
||||
)
|
||||
region = _region(299.0, 189.6, 552.4, 300.5, page=202)
|
||||
assert classify(region, PageContext(tables=[table])) == TABLE_FRAME
|
||||
# ...and the same geometry with no table map degrades to "look at it",
|
||||
# never to a silent pass
|
||||
assert classify(region) == UNCLASSIFIED
|
||||
|
||||
|
||||
def test_a_wide_rule_outside_the_header_band_is_not_treated_as_a_header_rule():
|
||||
assert classify(_region(36.0, 700.0, 552.0, 700.0)) == FRACTION_BAR_CANDIDATE
|
||||
|
||||
|
||||
def test_a_tall_block_of_unaccounted_ink_stays_unclassified():
|
||||
# a figure or an image of text must never be silently absorbed by a rule
|
||||
assert classify(_region(100.0, 300.0, 400.0, 500.0)) == UNCLASSIFIED
|
||||
|
||||
|
||||
def test_hairline_shorter_than_the_minimum_bar_width_is_a_rule_fragment():
|
||||
# too short to be a fraction bar, too thin to be anything but a rule
|
||||
assert classify(_region(100.0, 300.0, 105.0, 300.0)) == RULE_FRAGMENT
|
||||
|
||||
|
||||
@needs_pdf
|
||||
@pytest.mark.parametrize(
|
||||
"page,expected_bar_width_pt",
|
||||
[
|
||||
(1042, 188.6), # NETILMICIN — Cockcroft-Gault
|
||||
(202, 118.1), # AMPICILIN VÀ SULBACTAM — Cockcroft-Gault
|
||||
],
|
||||
)
|
||||
def test_confirmed_2d_formula_bars_survive_the_span_mask(page, expected_bar_width_pt):
|
||||
"""Regression fixture for the two visually confirmed corrupted formulas.
|
||||
|
||||
Both pages are reported as having zero tables by `pdfplumber` and zero by
|
||||
`opendataloader-pdf`; the bar is only findable as ink. If the mask padding
|
||||
is ever loosened again the bar disappears (at 1.0pt page 1042's bar
|
||||
shrinks from 188.6pt to 9.1pt) — this test is what catches that.
|
||||
"""
|
||||
import fitz
|
||||
|
||||
doc = fitz.open(PDF_PATH)
|
||||
bars = [
|
||||
r for r in scan_page(doc[page])
|
||||
if classify(r) == BAR and r.bbox[1] > 60.0
|
||||
]
|
||||
assert bars, f"no fraction-bar candidate found on physical page {page}"
|
||||
assert max(b.width_pt for b in bars) == pytest.approx(expected_bar_width_pt, abs=1.0)
|
||||
|
||||
|
||||
def test_vector_outlined_text_is_named_rather_than_left_unclassified():
|
||||
# physical page 714 prints 17 lines of Gatifloxacin prose as filled paths;
|
||||
# no text extractor returns them, so the gate must name the defect
|
||||
line = OutlinedTextRun(
|
||||
physical_page=714, bbox=(35.3, 75.8, 286.7, 84.4), path_items=1638,
|
||||
)
|
||||
region = _region(35.5, 76.0, 120.0, 84.0, page=714)
|
||||
context = PageContext(outlined_runs=[line])
|
||||
assert classify(region, context) == TEXT_AS_VECTOR_OUTLINE
|
||||
# an untranscribed line must never be mistaken for recovered content
|
||||
assert not line.is_transcribed
|
||||
|
||||
|
||||
@needs_pdf
|
||||
def test_outlined_text_lines_are_found_on_exactly_the_five_known_pages():
|
||||
"""Whole-document regression: 51 outlined runs on 5 pages.
|
||||
|
||||
Cross-checked two ways at the time of writing — the drawing-shape scan
|
||||
below, and independently by counting glyph-shaped leftovers in the
|
||||
residual-ink mask, which found the same five pages.
|
||||
"""
|
||||
import fitz
|
||||
|
||||
from ingestion.extract import detect_outlined_text
|
||||
|
||||
lines = list(detect_outlined_text(fitz.open(PDF_PATH)))
|
||||
by_page = {}
|
||||
for line in lines:
|
||||
by_page[line.physical_page] = by_page.get(line.physical_page, 0) + 1
|
||||
assert by_page == {714: 31, 736: 16, 1373: 1, 1444: 1, 1445: 2}
|
||||
Reference in New Issue
Block a user