feat(rex,pfs): EPUB attachments readable by the corpus indexer (F2)
extract_attachment returned "unsupported" for .epub, so CPT 2021/2022 and CPT Changes 2023 had no text at all, and 2018/2019/2024 indexed only from their PDF siblings. Added rex.comments.epub_text — stdlib only, shared with pfs.cpt_epub — that resolves an EPUB's own reading order via META-INF/container.xml -> the OPF's manifest + spine (falling back to sorted .xhtml/.html names when container.xml is missing), and strips tags/entities into plain text. extract_attachment's new .epub branch returns the same ExtractResult shape the PDF branch does; no change needed in llm.source._attachment_sections, which already tries every bib attachment regardless of extension. pfs.cpt_epub used to hard-code "OPS/" as the content-file prefix in three places; it now resolves the same way via epub_text.content_root, so a differently-templated EPUB would still locate its content instead of silently parsing to nothing.
This commit is contained in:
@@ -2,8 +2,11 @@
|
|||||||
EPUB markup — chapter headings, code-entry tables, parenthetical
|
EPUB markup — chapter headings, code-entry tables, parenthetical
|
||||||
instructions, CPT Changes/Assistant citations, and the authoritative
|
instructions, CPT Changes/Assistant citations, and the authoritative
|
||||||
appendix lists — into the dataclasses in ``pfs.cpt_model``. Stdlib only
|
appendix lists — into the dataclasses in ``pfs.cpt_model``. Stdlib only
|
||||||
(``zipfile``, ``html.parser``, ``html``, ``re``); no DuckDB, no bib, no
|
(``zipfile``, ``html.parser``, ``html``, ``re``) plus
|
||||||
network. See docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
|
``rex.comments.epub_text`` (itself stdlib-only — container/spine
|
||||||
|
resolution shared with the corpus indexer's own EPUB text extraction,
|
||||||
|
F2); no DuckDB, no bib, no network. See
|
||||||
|
docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
|
||||||
§1/§3 for what each structure means.
|
§1/§3 for what each structure means.
|
||||||
|
|
||||||
The 2021/2022/2024 Professional editions share one publisher template,
|
The 2021/2022/2024 Professional editions share one publisher template,
|
||||||
@@ -44,6 +47,7 @@ from pfs.cpt_model import (
|
|||||||
CptReference,
|
CptReference,
|
||||||
CptSection,
|
CptSection,
|
||||||
)
|
)
|
||||||
|
from rex.comments.epub_text import content_root
|
||||||
|
|
||||||
# One CPT code: 4 digits + a digit/letter (99490, 0202U, 99213F, 0042T, 0001A).
|
# One CPT code: 4 digits + a digit/letter (99490, 0202U, 99213F, 0042T, 0001A).
|
||||||
_CODE = re.compile(r"\b\d{4}[0-9A-Z]\b")
|
_CODE = re.compile(r"\b\d{4}[0-9A-Z]\b")
|
||||||
@@ -681,7 +685,7 @@ def _body_inner(text: str) -> str:
|
|||||||
return m.group(1) if m else text
|
return m.group(1) if m else text
|
||||||
|
|
||||||
|
|
||||||
def _chapter_groups(names: list[str]) -> list[list[str]]:
|
def _chapter_groups(names: list[str], root: str = "OPS/") -> list[list[str]]:
|
||||||
"""Group a book's chapter body files by chapter number, in book
|
"""Group a book's chapter body files by chapter number, in book
|
||||||
order. A chapter that's been split across several xhtml files
|
order. A chapter that's been split across several xhtml files
|
||||||
(``Chapter10.xhtml`` + ``Chapter10a.xhtml`` in 2019,
|
(``Chapter10.xhtml`` + ``Chapter10a.xhtml`` in 2019,
|
||||||
@@ -706,7 +710,7 @@ def _chapter_groups(names: list[str]) -> list[list[str]]:
|
|||||||
grouped: dict[str, list[str]] = {}
|
grouped: dict[str, list[str]] = {}
|
||||||
for name in names:
|
for name in names:
|
||||||
base = name.rsplit("/", 1)[-1]
|
base = name.rsplit("/", 1)[-1]
|
||||||
if not (name.startswith("OPS/") and base.endswith(".xhtml")):
|
if not (name.startswith(root) and base.endswith(".xhtml")):
|
||||||
continue
|
continue
|
||||||
if "Chapter" not in base or "_Toc" in base:
|
if "Chapter" not in base or "_Toc" in base:
|
||||||
continue
|
continue
|
||||||
@@ -899,11 +903,15 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
|||||||
|
|
||||||
with zipfile.ZipFile(path) as zf:
|
with zipfile.ZipFile(path) as zf:
|
||||||
names = zf.namelist()
|
names = zf.namelist()
|
||||||
|
# Resolved once via container.xml -> the OPF's own path, not
|
||||||
|
# hard-coded — falls back to "OPS/" (every edition checked so
|
||||||
|
# far) when container.xml is missing/malformed (F2).
|
||||||
|
root = content_root(zf)
|
||||||
|
|
||||||
toc_ranges: dict[str, tuple[str, str]] = {}
|
toc_ranges: dict[str, tuple[str, str]] = {}
|
||||||
for name in names:
|
for name in names:
|
||||||
base = name.rsplit("/", 1)[-1]
|
base = name.rsplit("/", 1)[-1]
|
||||||
if name.startswith("OPS/") and base.endswith(".xhtml") and "_Toc" in base:
|
if name.startswith(root) and base.endswith(".xhtml") and "_Toc" in base:
|
||||||
text = zf.read(name).decode("utf-8", errors="replace")
|
text = zf.read(name).decode("utf-8", errors="replace")
|
||||||
tok = _Tokenizer()
|
tok = _Tokenizer()
|
||||||
tok.feed(text)
|
tok.feed(text)
|
||||||
@@ -914,6 +922,14 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
|||||||
codes: list[CptCode] = []
|
codes: list[CptCode] = []
|
||||||
instructions: list[CptInstruction] = []
|
instructions: list[CptInstruction] = []
|
||||||
references: list[CptReference] = []
|
references: list[CptReference] = []
|
||||||
|
chapter_groups = _chapter_groups(names, root)
|
||||||
|
if not chapter_groups:
|
||||||
|
# F4: an empty parse must never reach write_cpt_edition's
|
||||||
|
# delete-then-insert — that would wipe an edition already on
|
||||||
|
# file for nothing. A book with genuinely no chapter body
|
||||||
|
# files (wrong root resolved, or a malformed/non-CPT EPUB)
|
||||||
|
# is a real failure, not a silent zero-row edition.
|
||||||
|
raise ValueError(f"no chapter content found in {path}")
|
||||||
# C10: most chapter-groups carry no div.ch-title of their own (a
|
# C10: most chapter-groups carry no div.ch-title of their own (a
|
||||||
# chapter split across many numbered files for size) — thread
|
# chapter split across many numbered files for size) — thread
|
||||||
# the active chapter title from one group's call into the
|
# the active chapter title from one group's call into the
|
||||||
@@ -921,7 +937,7 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
|||||||
# replaces it. Calls _group directly (not parse_xhtml, whose
|
# replaces it. Calls _group directly (not parse_xhtml, whose
|
||||||
# public 4-tuple return drops this) to carry it.
|
# public 4-tuple return drops this) to carry it.
|
||||||
chapter_title = ""
|
chapter_title = ""
|
||||||
for group in _chapter_groups(names):
|
for group in chapter_groups:
|
||||||
bodies = [
|
bodies = [
|
||||||
_body_inner(zf.read(name).decode("utf-8", errors="replace"))
|
_body_inner(zf.read(name).decode("utf-8", errors="replace"))
|
||||||
for name in group
|
for name in group
|
||||||
@@ -951,9 +967,7 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
|||||||
for name in names:
|
for name in names:
|
||||||
base = name.rsplit("/", 1)[-1]
|
base = name.rsplit("/", 1)[-1]
|
||||||
if not (
|
if not (
|
||||||
name.startswith("OPS/")
|
name.startswith(root) and base.endswith(".xhtml") and "ppendix" in base
|
||||||
and base.endswith(".xhtml")
|
|
||||||
and "ppendix" in base
|
|
||||||
):
|
):
|
||||||
continue
|
continue
|
||||||
letter = _appendix_letter(base)
|
letter = _appendix_letter(base)
|
||||||
|
|||||||
130
src/rex/comments/epub_text.py
Normal file
130
src/rex/comments/epub_text.py
Normal file
@@ -0,0 +1,130 @@
|
|||||||
|
"""Shared EPUB container/spine resolution: reading order and plain text.
|
||||||
|
|
||||||
|
Every EPUB carries ``META-INF/container.xml``, which points at the
|
||||||
|
package document (the "OPF") — the document that lists every content
|
||||||
|
file (the manifest) and the order a reader walks them in (the spine).
|
||||||
|
That's the only reliable way to know a book's own content-file prefix
|
||||||
|
(``OPS/`` for the AMA's CPT books, something else for another
|
||||||
|
publisher) and reading order — the zip's own member order is not
|
||||||
|
necessarily book order (``pfs.cpt_epub`` hit this: 2021/2022/2024's
|
||||||
|
namelist() interleaves chapters arbitrarily).
|
||||||
|
|
||||||
|
Two consumers share this: ``rex.comments.extract`` (F2 — the corpus
|
||||||
|
indexer needs whole-book plain text out of a CPT EPUB attachment) and
|
||||||
|
``pfs.cpt_epub`` (which used to hard-code the ``OPS/`` prefix three
|
||||||
|
times; it now resolves the same way, so a differently-templated EPUB
|
||||||
|
would still locate its content instead of silently finding nothing).
|
||||||
|
|
||||||
|
Stdlib only (``zipfile``, ``xml.etree``, ``re``, ``html``) — no bib, no
|
||||||
|
DuckDB, no network.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
import re
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
from xml.etree import ElementTree as ET
|
||||||
|
|
||||||
|
_CONTAINER_PATH = "META-INF/container.xml"
|
||||||
|
_CONTAINER_NS = {"c": "urn:oasis:names:tc:opendocument:xmlns:container"}
|
||||||
|
_OPF_NS = {"o": "http://www.idpf.org/2007/opf"}
|
||||||
|
|
||||||
|
_TAG_RE = re.compile(r"<[^>]+>")
|
||||||
|
_BLOCK_RE = re.compile(r"<(?:p|div|br|li|h[1-6]|tr)[^>]*>", re.IGNORECASE)
|
||||||
|
_SCRIPT_STYLE_RE = re.compile(
|
||||||
|
r"<(script|style)[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL
|
||||||
|
)
|
||||||
|
_BODY_RE = re.compile(r"<body[^>]*>(.*)</body>", re.IGNORECASE | re.DOTALL)
|
||||||
|
|
||||||
|
|
||||||
|
def _opf_info(zf: zipfile.ZipFile) -> tuple[str, str] | None:
|
||||||
|
"""(content-root dir, OPF member path), resolved via
|
||||||
|
``META-INF/container.xml`` -> the first ``<rootfile>``'s
|
||||||
|
``full-path``. ``None`` when container.xml is absent or malformed."""
|
||||||
|
try:
|
||||||
|
data = zf.read(_CONTAINER_PATH)
|
||||||
|
except KeyError:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
root = ET.fromstring(data)
|
||||||
|
except ET.ParseError:
|
||||||
|
return None
|
||||||
|
rootfile = root.find(".//c:rootfile", _CONTAINER_NS)
|
||||||
|
opf_path = rootfile.get("full-path") if rootfile is not None else None
|
||||||
|
if not opf_path:
|
||||||
|
return None
|
||||||
|
opf_dir = opf_path.rsplit("/", 1)[0] + "/" if "/" in opf_path else ""
|
||||||
|
return opf_dir, opf_path
|
||||||
|
|
||||||
|
|
||||||
|
def content_root(zf: zipfile.ZipFile) -> str:
|
||||||
|
"""The directory prefix (``"OPS/"``, ``""``, ...) holding this
|
||||||
|
book's content files. Falls back to ``"OPS/"`` — the AMA books' own
|
||||||
|
layout, and what every caller hard-coded before this existed — when
|
||||||
|
container.xml can't be resolved."""
|
||||||
|
info = _opf_info(zf)
|
||||||
|
return info[0] if info is not None else "OPS/"
|
||||||
|
|
||||||
|
|
||||||
|
def spine_paths(zf: zipfile.ZipFile) -> list[str] | None:
|
||||||
|
"""Full zip-member paths in the book's spine reading order,
|
||||||
|
resolved via container.xml -> the OPF's manifest + spine. ``None``
|
||||||
|
when the OPF can't be located or parsed — callers fall back to
|
||||||
|
sorted content-file names."""
|
||||||
|
info = _opf_info(zf)
|
||||||
|
if info is None:
|
||||||
|
return None
|
||||||
|
opf_dir, opf_path = info
|
||||||
|
try:
|
||||||
|
opf = ET.fromstring(zf.read(opf_path))
|
||||||
|
except (KeyError, ET.ParseError):
|
||||||
|
return None
|
||||||
|
manifest = {
|
||||||
|
item.get("id"): item.get("href")
|
||||||
|
for item in opf.findall(".//o:manifest/o:item", _OPF_NS)
|
||||||
|
}
|
||||||
|
paths = []
|
||||||
|
for itemref in opf.findall(".//o:spine/o:itemref", _OPF_NS):
|
||||||
|
href = manifest.get(itemref.get("idref"))
|
||||||
|
if href:
|
||||||
|
paths.append(opf_dir + href)
|
||||||
|
return paths or None
|
||||||
|
|
||||||
|
|
||||||
|
def _to_text(raw: str) -> str:
|
||||||
|
"""One content file's markup -> plain text: body only, block tags
|
||||||
|
become line breaks, entities unescaped, whitespace collapsed."""
|
||||||
|
m = _BODY_RE.search(raw)
|
||||||
|
body = m.group(1) if m else raw
|
||||||
|
body = _SCRIPT_STYLE_RE.sub("", body)
|
||||||
|
body = _BLOCK_RE.sub("\n", body)
|
||||||
|
text = html.unescape(_TAG_RE.sub("", body))
|
||||||
|
text = re.sub(r"[ \t]+", " ", text)
|
||||||
|
text = re.sub(r"\n\s*\n+", "\n\n", text)
|
||||||
|
return text.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def extract_text(path: str | Path) -> str:
|
||||||
|
"""Whole-book plain text, in reading order: spine order when the
|
||||||
|
OPF resolves, else every ``.xhtml``/``.html`` member sorted by
|
||||||
|
name. Sections are joined with a blank line; empty sections
|
||||||
|
(nav/cover files with no text) are dropped."""
|
||||||
|
with zipfile.ZipFile(path) as zf:
|
||||||
|
names = spine_paths(zf)
|
||||||
|
if not names:
|
||||||
|
names = sorted(
|
||||||
|
n
|
||||||
|
for n in zf.namelist()
|
||||||
|
if n.lower().endswith((".xhtml", ".html", ".htm"))
|
||||||
|
and not n.startswith("META-INF/")
|
||||||
|
)
|
||||||
|
parts = []
|
||||||
|
for name in names:
|
||||||
|
try:
|
||||||
|
raw = zf.read(name).decode("utf-8", errors="replace")
|
||||||
|
except KeyError:
|
||||||
|
continue
|
||||||
|
parts.append(_to_text(raw))
|
||||||
|
return "\n\n".join(p for p in parts if p.strip())
|
||||||
@@ -45,6 +45,8 @@ def extract_attachment(path: Path, *, ocr_engine=None) -> ExtractResult:
|
|||||||
return _extract_docx(path)
|
return _extract_docx(path)
|
||||||
if suffix in (".txt", ".html", ".htm"):
|
if suffix in (".txt", ".html", ".htm"):
|
||||||
return _extract_text_like(path, suffix)
|
return _extract_text_like(path, suffix)
|
||||||
|
if suffix == ".epub":
|
||||||
|
return _extract_epub(path)
|
||||||
if suffix in _IMAGE_SUFFIXES:
|
if suffix in _IMAGE_SUFFIXES:
|
||||||
# Real CMS comments include scanned-letter images; route to
|
# Real CMS comments include scanned-letter images; route to
|
||||||
# phase-2 OCR queue, don't drop them as "unsupported".
|
# phase-2 OCR queue, don't drop them as "unsupported".
|
||||||
@@ -110,6 +112,24 @@ def _extract_text_like(path: Path, suffix: str) -> ExtractResult:
|
|||||||
return ExtractResult(text=text, status=status, chars=chars)
|
return ExtractResult(text=text, status=status, chars=chars)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_epub(path: Path) -> ExtractResult:
|
||||||
|
"""F2: an EPUB attachment (the AMA's CPT codebooks — 2021/2022 and
|
||||||
|
the CPT Changes annuals ship EPUB-only) is real book text, not an
|
||||||
|
OCR candidate; ``rex.comments.epub_text`` resolves reading order via
|
||||||
|
the container/spine, the same mechanism ``pfs.cpt_epub`` uses for
|
||||||
|
its own content root."""
|
||||||
|
from rex.comments.epub_text import extract_text
|
||||||
|
|
||||||
|
try:
|
||||||
|
text = extract_text(path)
|
||||||
|
except Exception as e: # noqa: BLE001 — a malformed zip/xml is not fatal
|
||||||
|
log.warning("epub extract failed for %s: %s", path, e)
|
||||||
|
return ExtractResult(text="", status="failed", chars=0)
|
||||||
|
chars = len(text)
|
||||||
|
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
|
||||||
|
return ExtractResult(text=text, status=status, chars=chars)
|
||||||
|
|
||||||
|
|
||||||
def _strip_html(s: str) -> str:
|
def _strip_html(s: str) -> str:
|
||||||
"""Cheap HTML→text — break on block tags, drop the rest."""
|
"""Cheap HTML→text — break on block tags, drop the rest."""
|
||||||
s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I)
|
s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I)
|
||||||
|
|||||||
@@ -150,6 +150,27 @@ class TestCorpusDocs:
|
|||||||
assert docs[key].text.startswith("## letter.txt\n\nAttachment body text")
|
assert docs[key].text.startswith("## letter.txt\n\nAttachment body text")
|
||||||
assert [name for name, _ in docs[key].files] == ["letter.txt"]
|
assert [name for name, _ in docs[key].files] == ["letter.txt"]
|
||||||
|
|
||||||
|
def test_epub_attachment_text_included(self, store, tmp_path):
|
||||||
|
"""F2: an EPUB attachment (the AMA's CPT codebooks) is real book
|
||||||
|
text for the corpus indexer, same as a PDF/DOCX/txt attachment."""
|
||||||
|
import zipfile
|
||||||
|
|
||||||
|
key = store.create(Item(item_type="source", title="CPT Professional 2024"))
|
||||||
|
store.add_tag(key, "year:2024")
|
||||||
|
epub = tmp_path / "CPT Professional 2024.epub"
|
||||||
|
with zipfile.ZipFile(epub, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr(
|
||||||
|
"OPS/Chapter01.xhtml",
|
||||||
|
"<html><body><p>"
|
||||||
|
+ "Chronic care management services, first 20 minutes. " * 4
|
||||||
|
+ "</p></body></html>",
|
||||||
|
)
|
||||||
|
store.attach_file(key, epub)
|
||||||
|
docs = {d.key: d for d in iter_corpus_docs(store)}
|
||||||
|
assert "Chronic care management services" in docs[key].text
|
||||||
|
assert [name for name, _ in docs[key].files] == [epub.name]
|
||||||
|
|
||||||
def test_zotero_pdf_fallback_used_when_bib_has_no_attachments(
|
def test_zotero_pdf_fallback_used_when_bib_has_no_attachments(
|
||||||
self, store, tmp_path
|
self, store, tmp_path
|
||||||
):
|
):
|
||||||
|
|||||||
@@ -454,6 +454,65 @@ class TestChapterTitleParenting:
|
|||||||
assert sec.path == ("Surgery", "Other Procedures")
|
assert sec.path == ("Surgery", "Other Procedures")
|
||||||
|
|
||||||
|
|
||||||
|
class TestEmptyParseRaisesF4:
|
||||||
|
"""F4: an empty parse must never reach write_cpt_edition's
|
||||||
|
delete-then-insert silently — parse_epub raises instead."""
|
||||||
|
|
||||||
|
def test_no_chapter_files_at_all_raises(self, tmp_path):
|
||||||
|
path = tmp_path / "empty.epub"
|
||||||
|
_build_epub(path, {}) # mimetype only, no OPS/ChapterNN.xhtml
|
||||||
|
with pytest.raises(ValueError, match="no chapter content found"):
|
||||||
|
parse_epub(path, year=2024)
|
||||||
|
|
||||||
|
def test_files_present_but_none_named_chapter_raises(self, tmp_path):
|
||||||
|
path = tmp_path / "no_chapters.epub"
|
||||||
|
_build_epub(path, {"Frontmatter.xhtml": "<p>Not a chapter file.</p>"})
|
||||||
|
with pytest.raises(ValueError, match="no chapter content found"):
|
||||||
|
parse_epub(path, year=2024)
|
||||||
|
|
||||||
|
|
||||||
|
class TestContentRootResolution:
|
||||||
|
"""F2/F4: cpt_epub resolves the OPS/ content-file prefix via
|
||||||
|
container.xml -> the OPF's own path, not a hard-coded string —
|
||||||
|
falling back to "OPS/" (every edition checked so far, and what
|
||||||
|
``_build_epub``'s fixtures use with no container.xml at all)."""
|
||||||
|
|
||||||
|
def test_falls_back_to_ops_without_a_container_xml(self, tmp_path):
|
||||||
|
path = tmp_path / "sample.epub"
|
||||||
|
_build_epub(path, {"Chapter01.xhtml": _C9_CH2})
|
||||||
|
edition = parse_epub(path, year=2024)
|
||||||
|
assert any(c.code == "54321" for c in edition.codes)
|
||||||
|
|
||||||
|
def test_resolves_a_non_ops_content_root_via_container_xml(self, tmp_path):
|
||||||
|
# A differently-templated EPUB whose content root isn't "OPS/" —
|
||||||
|
# before F2, this silently parsed to nothing.
|
||||||
|
path = tmp_path / "other_root.epub"
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr(
|
||||||
|
"META-INF/container.xml",
|
||||||
|
'<?xml version="1.0"?>\n'
|
||||||
|
'<container version="1.0" '
|
||||||
|
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
|
||||||
|
"<rootfiles>\n"
|
||||||
|
'<rootfile full-path="EPUB/content.opf" '
|
||||||
|
'media-type="application/oebps-package+xml"/>\n'
|
||||||
|
"</rootfiles>\n</container>",
|
||||||
|
)
|
||||||
|
zf.writestr(
|
||||||
|
"EPUB/content.opf",
|
||||||
|
'<?xml version="1.0"?>\n'
|
||||||
|
'<package xmlns="http://www.idpf.org/2007/opf" version="3.0">\n'
|
||||||
|
"<manifest/><spine/>\n</package>",
|
||||||
|
)
|
||||||
|
zf.writestr(
|
||||||
|
"EPUB/Chapter01.xhtml",
|
||||||
|
f"<html><body>{_C9_CH2}</body></html>",
|
||||||
|
)
|
||||||
|
edition = parse_epub(path, year=2024)
|
||||||
|
assert any(c.code == "54321" for c in edition.codes)
|
||||||
|
|
||||||
|
|
||||||
class TestAlternates:
|
class TestAlternates:
|
||||||
def test_ranged_entry_wins_as_canonical(self, c9_edition):
|
def test_ranged_entry_wins_as_canonical(self, c9_edition):
|
||||||
matches = [c for c in c9_edition.codes if c.code == "54321"]
|
matches = [c for c in c9_edition.codes if c.code == "54321"]
|
||||||
|
|||||||
106
tests/rex/comments/test_epub_text.py
Normal file
106
tests/rex/comments/test_epub_text.py
Normal file
@@ -0,0 +1,106 @@
|
|||||||
|
"""rex.comments.epub_text — shared EPUB container/spine resolution."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from rex.comments.epub_text import content_root, extract_text, spine_paths
|
||||||
|
|
||||||
|
_CONTAINER_XML = (
|
||||||
|
'<?xml version="1.0"?>\n'
|
||||||
|
'<container version="1.0" '
|
||||||
|
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
|
||||||
|
"<rootfiles>\n"
|
||||||
|
'<rootfile full-path="OEBPS/content.opf" '
|
||||||
|
'media-type="application/oebps-package+xml"/>\n'
|
||||||
|
"</rootfiles>\n</container>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _opf(spine_idrefs: list[str]) -> str:
|
||||||
|
spine = "\n".join(f'<itemref idref="{i}"/>' for i in spine_idrefs)
|
||||||
|
return (
|
||||||
|
'<?xml version="1.0"?>\n'
|
||||||
|
'<package xmlns="http://www.idpf.org/2007/opf" version="3.0">\n'
|
||||||
|
"<manifest>\n"
|
||||||
|
'<item id="a" href="a.xhtml" media-type="application/xhtml+xml"/>\n'
|
||||||
|
'<item id="b" href="b.xhtml" media-type="application/xhtml+xml"/>\n'
|
||||||
|
"</manifest>\n"
|
||||||
|
f"<spine>\n{spine}\n</spine>\n"
|
||||||
|
"</package>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _build_epub_with_container(path: Path, *, spine_order: list[str]) -> None:
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr("META-INF/container.xml", _CONTAINER_XML)
|
||||||
|
zf.writestr("OEBPS/content.opf", _opf(spine_order))
|
||||||
|
zf.writestr("OEBPS/a.xhtml", "<html><body><p>Section A text.</p></body></html>")
|
||||||
|
zf.writestr("OEBPS/b.xhtml", "<html><body><p>Section B text.</p></body></html>")
|
||||||
|
|
||||||
|
|
||||||
|
def _build_epub_without_container(path: Path) -> None:
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
# Deliberately named so name-sort order differs from any
|
||||||
|
# plausible spine order — proves the fallback really is name-sort.
|
||||||
|
zf.writestr("OPS/z_first.xhtml", "<html><body><p>Z content.</p></body></html>")
|
||||||
|
zf.writestr("OPS/a_second.xhtml", "<html><body><p>A content.</p></body></html>")
|
||||||
|
|
||||||
|
|
||||||
|
class TestContentRoot:
|
||||||
|
def test_resolves_via_container_xml(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
_build_epub_with_container(path, spine_order=["a", "b"])
|
||||||
|
with zipfile.ZipFile(path) as zf:
|
||||||
|
assert content_root(zf) == "OEBPS/"
|
||||||
|
|
||||||
|
def test_falls_back_to_ops_without_container_xml(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
_build_epub_without_container(path)
|
||||||
|
with zipfile.ZipFile(path) as zf:
|
||||||
|
assert content_root(zf) == "OPS/"
|
||||||
|
|
||||||
|
|
||||||
|
class TestSpinePaths:
|
||||||
|
def test_resolves_spine_order_from_the_opf(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
_build_epub_with_container(path, spine_order=["b", "a"])
|
||||||
|
with zipfile.ZipFile(path) as zf:
|
||||||
|
assert spine_paths(zf) == ["OEBPS/b.xhtml", "OEBPS/a.xhtml"]
|
||||||
|
|
||||||
|
def test_none_without_container_xml(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
_build_epub_without_container(path)
|
||||||
|
with zipfile.ZipFile(path) as zf:
|
||||||
|
assert spine_paths(zf) is None
|
||||||
|
|
||||||
|
|
||||||
|
class TestExtractText:
|
||||||
|
def test_extracts_in_spine_order_not_alphabetical(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
# Spine deliberately reversed from alphabetical (a, b) name order.
|
||||||
|
_build_epub_with_container(path, spine_order=["b", "a"])
|
||||||
|
text = extract_text(path)
|
||||||
|
assert text.index("Section B text") < text.index("Section A text")
|
||||||
|
|
||||||
|
def test_falls_back_to_sorted_names_without_container_xml(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
_build_epub_without_container(path)
|
||||||
|
text = extract_text(path)
|
||||||
|
# a_second.xhtml sorts before z_first.xhtml by name.
|
||||||
|
assert text.index("A content") < text.index("Z content")
|
||||||
|
|
||||||
|
def test_strips_tags_and_unescapes_entities(self, tmp_path):
|
||||||
|
path = tmp_path / "book.epub"
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr(
|
||||||
|
"OPS/a.xhtml",
|
||||||
|
"<html><body><p>CPT® codes & values</p></body></html>",
|
||||||
|
)
|
||||||
|
text = extract_text(path)
|
||||||
|
assert "<p>" not in text
|
||||||
|
assert "CPT® codes & values" in text
|
||||||
55
tests/rex/comments/test_extract_epub.py
Normal file
55
tests/rex/comments/test_extract_epub.py
Normal file
@@ -0,0 +1,55 @@
|
|||||||
|
"""F2: extract_attachment's .epub branch."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from rex.comments import extract_attachment
|
||||||
|
|
||||||
|
|
||||||
|
def _build_epub(path: Path, body: str) -> None:
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr("OPS/Chapter01.xhtml", f"<html><body>{body}</body></html>")
|
||||||
|
|
||||||
|
|
||||||
|
def test_epub_extracted_ok(tmp_path: Path):
|
||||||
|
p = tmp_path / "book.epub"
|
||||||
|
_build_epub(p, "<p>" + "A CPT codebook chapter with real text. " * 5 + "</p>")
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ok"
|
||||||
|
assert "CPT codebook chapter" in result.text
|
||||||
|
assert result.chars == len(result.text)
|
||||||
|
|
||||||
|
|
||||||
|
def test_epub_below_threshold_marked_ocr_needed(tmp_path: Path):
|
||||||
|
p = tmp_path / "book.epub"
|
||||||
|
_build_epub(p, "<p>short</p>")
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ocr_needed"
|
||||||
|
|
||||||
|
|
||||||
|
def test_epub_malformed_zip_marked_failed(tmp_path: Path):
|
||||||
|
p = tmp_path / "book.epub"
|
||||||
|
p.write_bytes(b"not a real zip file")
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "failed"
|
||||||
|
assert result.text == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_epub_extracted_in_reading_order(tmp_path: Path):
|
||||||
|
p = tmp_path / "book.epub"
|
||||||
|
with zipfile.ZipFile(p, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr(
|
||||||
|
"OPS/Chapter01.xhtml",
|
||||||
|
"<html><body><p>First chapter content here, plenty of text.</p></body></html>",
|
||||||
|
)
|
||||||
|
zf.writestr(
|
||||||
|
"OPS/Chapter02.xhtml",
|
||||||
|
"<html><body><p>Second chapter content here, plenty of text.</p></body></html>",
|
||||||
|
)
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ok"
|
||||||
|
assert result.text.index("First chapter") < result.text.index("Second chapter")
|
||||||
Reference in New Issue
Block a user