feat(rex,pfs): EPUB attachments readable by the corpus indexer (F2)

extract_attachment returned "unsupported" for .epub, so CPT 2021/2022
and CPT Changes 2023 had no text at all, and 2018/2019/2024 indexed
only from their PDF siblings. Added rex.comments.epub_text — stdlib
only, shared with pfs.cpt_epub — that resolves an EPUB's own reading
order via META-INF/container.xml -> the OPF's manifest + spine (falling
back to sorted .xhtml/.html names when container.xml is missing), and
strips tags/entities into plain text. extract_attachment's new .epub
branch returns the same ExtractResult shape the PDF branch does; no
change needed in llm.source._attachment_sections, which already tries
every bib attachment regardless of extension.

pfs.cpt_epub used to hard-code "OPS/" as the content-file prefix in
three places; it now resolves the same way via epub_text.content_root,
so a differently-templated EPUB would still locate its content instead
of silently parsing to nothing.
This commit is contained in:
kert
2026-09-09 21:11:04 -04:00
parent 00dd3fb9b0
commit 77fd48eb31
7 changed files with 414 additions and 9 deletions

View File

@@ -2,8 +2,11 @@
EPUB markup — chapter headings, code-entry tables, parenthetical EPUB markup — chapter headings, code-entry tables, parenthetical
instructions, CPT Changes/Assistant citations, and the authoritative instructions, CPT Changes/Assistant citations, and the authoritative
appendix lists — into the dataclasses in ``pfs.cpt_model``. Stdlib only appendix lists — into the dataclasses in ``pfs.cpt_model``. Stdlib only
(``zipfile``, ``html.parser``, ``html``, ``re``); no DuckDB, no bib, no (``zipfile``, ``html.parser``, ``html``, ``re``) plus
network. See docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md ``rex.comments.epub_text`` (itself stdlib-only — container/spine
resolution shared with the corpus indexer's own EPUB text extraction,
F2); no DuckDB, no bib, no network. See
docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
§1/§3 for what each structure means. §1/§3 for what each structure means.
The 2021/2022/2024 Professional editions share one publisher template, The 2021/2022/2024 Professional editions share one publisher template,
@@ -44,6 +47,7 @@ from pfs.cpt_model import (
CptReference, CptReference,
CptSection, CptSection,
) )
from rex.comments.epub_text import content_root
# One CPT code: 4 digits + a digit/letter (99490, 0202U, 99213F, 0042T, 0001A). # One CPT code: 4 digits + a digit/letter (99490, 0202U, 99213F, 0042T, 0001A).
_CODE = re.compile(r"\b\d{4}[0-9A-Z]\b") _CODE = re.compile(r"\b\d{4}[0-9A-Z]\b")
@@ -681,7 +685,7 @@ def _body_inner(text: str) -> str:
return m.group(1) if m else text return m.group(1) if m else text
def _chapter_groups(names: list[str]) -> list[list[str]]: def _chapter_groups(names: list[str], root: str = "OPS/") -> list[list[str]]:
"""Group a book's chapter body files by chapter number, in book """Group a book's chapter body files by chapter number, in book
order. A chapter that's been split across several xhtml files order. A chapter that's been split across several xhtml files
(``Chapter10.xhtml`` + ``Chapter10a.xhtml`` in 2019, (``Chapter10.xhtml`` + ``Chapter10a.xhtml`` in 2019,
@@ -706,7 +710,7 @@ def _chapter_groups(names: list[str]) -> list[list[str]]:
grouped: dict[str, list[str]] = {} grouped: dict[str, list[str]] = {}
for name in names: for name in names:
base = name.rsplit("/", 1)[-1] base = name.rsplit("/", 1)[-1]
if not (name.startswith("OPS/") and base.endswith(".xhtml")): if not (name.startswith(root) and base.endswith(".xhtml")):
continue continue
if "Chapter" not in base or "_Toc" in base: if "Chapter" not in base or "_Toc" in base:
continue continue
@@ -899,11 +903,15 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
with zipfile.ZipFile(path) as zf: with zipfile.ZipFile(path) as zf:
names = zf.namelist() names = zf.namelist()
# Resolved once via container.xml -> the OPF's own path, not
# hard-coded — falls back to "OPS/" (every edition checked so
# far) when container.xml is missing/malformed (F2).
root = content_root(zf)
toc_ranges: dict[str, tuple[str, str]] = {} toc_ranges: dict[str, tuple[str, str]] = {}
for name in names: for name in names:
base = name.rsplit("/", 1)[-1] base = name.rsplit("/", 1)[-1]
if name.startswith("OPS/") and base.endswith(".xhtml") and "_Toc" in base: if name.startswith(root) and base.endswith(".xhtml") and "_Toc" in base:
text = zf.read(name).decode("utf-8", errors="replace") text = zf.read(name).decode("utf-8", errors="replace")
tok = _Tokenizer() tok = _Tokenizer()
tok.feed(text) tok.feed(text)
@@ -914,6 +922,14 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
codes: list[CptCode] = [] codes: list[CptCode] = []
instructions: list[CptInstruction] = [] instructions: list[CptInstruction] = []
references: list[CptReference] = [] references: list[CptReference] = []
chapter_groups = _chapter_groups(names, root)
if not chapter_groups:
# F4: an empty parse must never reach write_cpt_edition's
# delete-then-insert — that would wipe an edition already on
# file for nothing. A book with genuinely no chapter body
# files (wrong root resolved, or a malformed/non-CPT EPUB)
# is a real failure, not a silent zero-row edition.
raise ValueError(f"no chapter content found in {path}")
# C10: most chapter-groups carry no div.ch-title of their own (a # C10: most chapter-groups carry no div.ch-title of their own (a
# chapter split across many numbered files for size) — thread # chapter split across many numbered files for size) — thread
# the active chapter title from one group's call into the # the active chapter title from one group's call into the
@@ -921,7 +937,7 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
# replaces it. Calls _group directly (not parse_xhtml, whose # replaces it. Calls _group directly (not parse_xhtml, whose
# public 4-tuple return drops this) to carry it. # public 4-tuple return drops this) to carry it.
chapter_title = "" chapter_title = ""
for group in _chapter_groups(names): for group in chapter_groups:
bodies = [ bodies = [
_body_inner(zf.read(name).decode("utf-8", errors="replace")) _body_inner(zf.read(name).decode("utf-8", errors="replace"))
for name in group for name in group
@@ -951,9 +967,7 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
for name in names: for name in names:
base = name.rsplit("/", 1)[-1] base = name.rsplit("/", 1)[-1]
if not ( if not (
name.startswith("OPS/") name.startswith(root) and base.endswith(".xhtml") and "ppendix" in base
and base.endswith(".xhtml")
and "ppendix" in base
): ):
continue continue
letter = _appendix_letter(base) letter = _appendix_letter(base)

View File

@@ -0,0 +1,130 @@
"""Shared EPUB container/spine resolution: reading order and plain text.
Every EPUB carries ``META-INF/container.xml``, which points at the
package document (the "OPF") — the document that lists every content
file (the manifest) and the order a reader walks them in (the spine).
That's the only reliable way to know a book's own content-file prefix
(``OPS/`` for the AMA's CPT books, something else for another
publisher) and reading order — the zip's own member order is not
necessarily book order (``pfs.cpt_epub`` hit this: 2021/2022/2024's
namelist() interleaves chapters arbitrarily).
Two consumers share this: ``rex.comments.extract`` (F2 — the corpus
indexer needs whole-book plain text out of a CPT EPUB attachment) and
``pfs.cpt_epub`` (which used to hard-code the ``OPS/`` prefix three
times; it now resolves the same way, so a differently-templated EPUB
would still locate its content instead of silently finding nothing).
Stdlib only (``zipfile``, ``xml.etree``, ``re``, ``html``) — no bib, no
DuckDB, no network.
"""
from __future__ import annotations
import html
import re
import zipfile
from pathlib import Path
from xml.etree import ElementTree as ET
_CONTAINER_PATH = "META-INF/container.xml"
_CONTAINER_NS = {"c": "urn:oasis:names:tc:opendocument:xmlns:container"}
_OPF_NS = {"o": "http://www.idpf.org/2007/opf"}
_TAG_RE = re.compile(r"<[^>]+>")
_BLOCK_RE = re.compile(r"<(?:p|div|br|li|h[1-6]|tr)[^>]*>", re.IGNORECASE)
_SCRIPT_STYLE_RE = re.compile(
r"<(script|style)[^>]*>.*?</\1>", re.IGNORECASE | re.DOTALL
)
_BODY_RE = re.compile(r"<body[^>]*>(.*)</body>", re.IGNORECASE | re.DOTALL)
def _opf_info(zf: zipfile.ZipFile) -> tuple[str, str] | None:
"""(content-root dir, OPF member path), resolved via
``META-INF/container.xml`` -> the first ``<rootfile>``'s
``full-path``. ``None`` when container.xml is absent or malformed."""
try:
data = zf.read(_CONTAINER_PATH)
except KeyError:
return None
try:
root = ET.fromstring(data)
except ET.ParseError:
return None
rootfile = root.find(".//c:rootfile", _CONTAINER_NS)
opf_path = rootfile.get("full-path") if rootfile is not None else None
if not opf_path:
return None
opf_dir = opf_path.rsplit("/", 1)[0] + "/" if "/" in opf_path else ""
return opf_dir, opf_path
def content_root(zf: zipfile.ZipFile) -> str:
"""The directory prefix (``"OPS/"``, ``""``, ...) holding this
book's content files. Falls back to ``"OPS/"`` — the AMA books' own
layout, and what every caller hard-coded before this existed — when
container.xml can't be resolved."""
info = _opf_info(zf)
return info[0] if info is not None else "OPS/"
def spine_paths(zf: zipfile.ZipFile) -> list[str] | None:
"""Full zip-member paths in the book's spine reading order,
resolved via container.xml -> the OPF's manifest + spine. ``None``
when the OPF can't be located or parsed — callers fall back to
sorted content-file names."""
info = _opf_info(zf)
if info is None:
return None
opf_dir, opf_path = info
try:
opf = ET.fromstring(zf.read(opf_path))
except (KeyError, ET.ParseError):
return None
manifest = {
item.get("id"): item.get("href")
for item in opf.findall(".//o:manifest/o:item", _OPF_NS)
}
paths = []
for itemref in opf.findall(".//o:spine/o:itemref", _OPF_NS):
href = manifest.get(itemref.get("idref"))
if href:
paths.append(opf_dir + href)
return paths or None
def _to_text(raw: str) -> str:
"""One content file's markup -> plain text: body only, block tags
become line breaks, entities unescaped, whitespace collapsed."""
m = _BODY_RE.search(raw)
body = m.group(1) if m else raw
body = _SCRIPT_STYLE_RE.sub("", body)
body = _BLOCK_RE.sub("\n", body)
text = html.unescape(_TAG_RE.sub("", body))
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\n\s*\n+", "\n\n", text)
return text.strip()
def extract_text(path: str | Path) -> str:
"""Whole-book plain text, in reading order: spine order when the
OPF resolves, else every ``.xhtml``/``.html`` member sorted by
name. Sections are joined with a blank line; empty sections
(nav/cover files with no text) are dropped."""
with zipfile.ZipFile(path) as zf:
names = spine_paths(zf)
if not names:
names = sorted(
n
for n in zf.namelist()
if n.lower().endswith((".xhtml", ".html", ".htm"))
and not n.startswith("META-INF/")
)
parts = []
for name in names:
try:
raw = zf.read(name).decode("utf-8", errors="replace")
except KeyError:
continue
parts.append(_to_text(raw))
return "\n\n".join(p for p in parts if p.strip())

View File

@@ -45,6 +45,8 @@ def extract_attachment(path: Path, *, ocr_engine=None) -> ExtractResult:
return _extract_docx(path) return _extract_docx(path)
if suffix in (".txt", ".html", ".htm"): if suffix in (".txt", ".html", ".htm"):
return _extract_text_like(path, suffix) return _extract_text_like(path, suffix)
if suffix == ".epub":
return _extract_epub(path)
if suffix in _IMAGE_SUFFIXES: if suffix in _IMAGE_SUFFIXES:
# Real CMS comments include scanned-letter images; route to # Real CMS comments include scanned-letter images; route to
# phase-2 OCR queue, don't drop them as "unsupported". # phase-2 OCR queue, don't drop them as "unsupported".
@@ -110,6 +112,24 @@ def _extract_text_like(path: Path, suffix: str) -> ExtractResult:
return ExtractResult(text=text, status=status, chars=chars) return ExtractResult(text=text, status=status, chars=chars)
def _extract_epub(path: Path) -> ExtractResult:
"""F2: an EPUB attachment (the AMA's CPT codebooks — 2021/2022 and
the CPT Changes annuals ship EPUB-only) is real book text, not an
OCR candidate; ``rex.comments.epub_text`` resolves reading order via
the container/spine, the same mechanism ``pfs.cpt_epub`` uses for
its own content root."""
from rex.comments.epub_text import extract_text
try:
text = extract_text(path)
except Exception as e: # noqa: BLE001 — a malformed zip/xml is not fatal
log.warning("epub extract failed for %s: %s", path, e)
return ExtractResult(text="", status="failed", chars=0)
chars = len(text)
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
return ExtractResult(text=text, status=status, chars=chars)
def _strip_html(s: str) -> str: def _strip_html(s: str) -> str:
"""Cheap HTML→text — break on block tags, drop the rest.""" """Cheap HTML→text — break on block tags, drop the rest."""
s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I) s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I)

View File

@@ -150,6 +150,27 @@ class TestCorpusDocs:
assert docs[key].text.startswith("## letter.txt\n\nAttachment body text") assert docs[key].text.startswith("## letter.txt\n\nAttachment body text")
assert [name for name, _ in docs[key].files] == ["letter.txt"] assert [name for name, _ in docs[key].files] == ["letter.txt"]
def test_epub_attachment_text_included(self, store, tmp_path):
"""F2: an EPUB attachment (the AMA's CPT codebooks) is real book
text for the corpus indexer, same as a PDF/DOCX/txt attachment."""
import zipfile
key = store.create(Item(item_type="source", title="CPT Professional 2024"))
store.add_tag(key, "year:2024")
epub = tmp_path / "CPT Professional 2024.epub"
with zipfile.ZipFile(epub, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr(
"OPS/Chapter01.xhtml",
"<html><body><p>"
+ "Chronic care management services, first 20 minutes. " * 4
+ "</p></body></html>",
)
store.attach_file(key, epub)
docs = {d.key: d for d in iter_corpus_docs(store)}
assert "Chronic care management services" in docs[key].text
assert [name for name, _ in docs[key].files] == [epub.name]
def test_zotero_pdf_fallback_used_when_bib_has_no_attachments( def test_zotero_pdf_fallback_used_when_bib_has_no_attachments(
self, store, tmp_path self, store, tmp_path
): ):

View File

@@ -454,6 +454,65 @@ class TestChapterTitleParenting:
assert sec.path == ("Surgery", "Other Procedures") assert sec.path == ("Surgery", "Other Procedures")
class TestEmptyParseRaisesF4:
"""F4: an empty parse must never reach write_cpt_edition's
delete-then-insert silently — parse_epub raises instead."""
def test_no_chapter_files_at_all_raises(self, tmp_path):
path = tmp_path / "empty.epub"
_build_epub(path, {}) # mimetype only, no OPS/ChapterNN.xhtml
with pytest.raises(ValueError, match="no chapter content found"):
parse_epub(path, year=2024)
def test_files_present_but_none_named_chapter_raises(self, tmp_path):
path = tmp_path / "no_chapters.epub"
_build_epub(path, {"Frontmatter.xhtml": "<p>Not a chapter file.</p>"})
with pytest.raises(ValueError, match="no chapter content found"):
parse_epub(path, year=2024)
class TestContentRootResolution:
"""F2/F4: cpt_epub resolves the OPS/ content-file prefix via
container.xml -> the OPF's own path, not a hard-coded string —
falling back to "OPS/" (every edition checked so far, and what
``_build_epub``'s fixtures use with no container.xml at all)."""
def test_falls_back_to_ops_without_a_container_xml(self, tmp_path):
path = tmp_path / "sample.epub"
_build_epub(path, {"Chapter01.xhtml": _C9_CH2})
edition = parse_epub(path, year=2024)
assert any(c.code == "54321" for c in edition.codes)
def test_resolves_a_non_ops_content_root_via_container_xml(self, tmp_path):
# A differently-templated EPUB whose content root isn't "OPS/" —
# before F2, this silently parsed to nothing.
path = tmp_path / "other_root.epub"
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr(
"META-INF/container.xml",
'<?xml version="1.0"?>\n'
'<container version="1.0" '
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
"<rootfiles>\n"
'<rootfile full-path="EPUB/content.opf" '
'media-type="application/oebps-package+xml"/>\n'
"</rootfiles>\n</container>",
)
zf.writestr(
"EPUB/content.opf",
'<?xml version="1.0"?>\n'
'<package xmlns="http://www.idpf.org/2007/opf" version="3.0">\n'
"<manifest/><spine/>\n</package>",
)
zf.writestr(
"EPUB/Chapter01.xhtml",
f"<html><body>{_C9_CH2}</body></html>",
)
edition = parse_epub(path, year=2024)
assert any(c.code == "54321" for c in edition.codes)
class TestAlternates: class TestAlternates:
def test_ranged_entry_wins_as_canonical(self, c9_edition): def test_ranged_entry_wins_as_canonical(self, c9_edition):
matches = [c for c in c9_edition.codes if c.code == "54321"] matches = [c for c in c9_edition.codes if c.code == "54321"]

View File

@@ -0,0 +1,106 @@
"""rex.comments.epub_text — shared EPUB container/spine resolution."""
from __future__ import annotations
import zipfile
from pathlib import Path
from rex.comments.epub_text import content_root, extract_text, spine_paths
_CONTAINER_XML = (
'<?xml version="1.0"?>\n'
'<container version="1.0" '
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
"<rootfiles>\n"
'<rootfile full-path="OEBPS/content.opf" '
'media-type="application/oebps-package+xml"/>\n'
"</rootfiles>\n</container>"
)
def _opf(spine_idrefs: list[str]) -> str:
spine = "\n".join(f'<itemref idref="{i}"/>' for i in spine_idrefs)
return (
'<?xml version="1.0"?>\n'
'<package xmlns="http://www.idpf.org/2007/opf" version="3.0">\n'
"<manifest>\n"
'<item id="a" href="a.xhtml" media-type="application/xhtml+xml"/>\n'
'<item id="b" href="b.xhtml" media-type="application/xhtml+xml"/>\n'
"</manifest>\n"
f"<spine>\n{spine}\n</spine>\n"
"</package>"
)
def _build_epub_with_container(path: Path, *, spine_order: list[str]) -> None:
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr("META-INF/container.xml", _CONTAINER_XML)
zf.writestr("OEBPS/content.opf", _opf(spine_order))
zf.writestr("OEBPS/a.xhtml", "<html><body><p>Section A text.</p></body></html>")
zf.writestr("OEBPS/b.xhtml", "<html><body><p>Section B text.</p></body></html>")
def _build_epub_without_container(path: Path) -> None:
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
# Deliberately named so name-sort order differs from any
# plausible spine order — proves the fallback really is name-sort.
zf.writestr("OPS/z_first.xhtml", "<html><body><p>Z content.</p></body></html>")
zf.writestr("OPS/a_second.xhtml", "<html><body><p>A content.</p></body></html>")
class TestContentRoot:
def test_resolves_via_container_xml(self, tmp_path):
path = tmp_path / "book.epub"
_build_epub_with_container(path, spine_order=["a", "b"])
with zipfile.ZipFile(path) as zf:
assert content_root(zf) == "OEBPS/"
def test_falls_back_to_ops_without_container_xml(self, tmp_path):
path = tmp_path / "book.epub"
_build_epub_without_container(path)
with zipfile.ZipFile(path) as zf:
assert content_root(zf) == "OPS/"
class TestSpinePaths:
def test_resolves_spine_order_from_the_opf(self, tmp_path):
path = tmp_path / "book.epub"
_build_epub_with_container(path, spine_order=["b", "a"])
with zipfile.ZipFile(path) as zf:
assert spine_paths(zf) == ["OEBPS/b.xhtml", "OEBPS/a.xhtml"]
def test_none_without_container_xml(self, tmp_path):
path = tmp_path / "book.epub"
_build_epub_without_container(path)
with zipfile.ZipFile(path) as zf:
assert spine_paths(zf) is None
class TestExtractText:
def test_extracts_in_spine_order_not_alphabetical(self, tmp_path):
path = tmp_path / "book.epub"
# Spine deliberately reversed from alphabetical (a, b) name order.
_build_epub_with_container(path, spine_order=["b", "a"])
text = extract_text(path)
assert text.index("Section B text") < text.index("Section A text")
def test_falls_back_to_sorted_names_without_container_xml(self, tmp_path):
path = tmp_path / "book.epub"
_build_epub_without_container(path)
text = extract_text(path)
# a_second.xhtml sorts before z_first.xhtml by name.
assert text.index("A content") < text.index("Z content")
def test_strips_tags_and_unescapes_entities(self, tmp_path):
path = tmp_path / "book.epub"
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr(
"OPS/a.xhtml",
"<html><body><p>CPT&#174; codes &amp; values</p></body></html>",
)
text = extract_text(path)
assert "<p>" not in text
assert "CPT® codes & values" in text

View File

@@ -0,0 +1,55 @@
"""F2: extract_attachment's .epub branch."""
from __future__ import annotations
import zipfile
from pathlib import Path
from rex.comments import extract_attachment
def _build_epub(path: Path, body: str) -> None:
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr("OPS/Chapter01.xhtml", f"<html><body>{body}</body></html>")
def test_epub_extracted_ok(tmp_path: Path):
p = tmp_path / "book.epub"
_build_epub(p, "<p>" + "A CPT codebook chapter with real text. " * 5 + "</p>")
result = extract_attachment(p)
assert result.status == "ok"
assert "CPT codebook chapter" in result.text
assert result.chars == len(result.text)
def test_epub_below_threshold_marked_ocr_needed(tmp_path: Path):
p = tmp_path / "book.epub"
_build_epub(p, "<p>short</p>")
result = extract_attachment(p)
assert result.status == "ocr_needed"
def test_epub_malformed_zip_marked_failed(tmp_path: Path):
p = tmp_path / "book.epub"
p.write_bytes(b"not a real zip file")
result = extract_attachment(p)
assert result.status == "failed"
assert result.text == ""
def test_epub_extracted_in_reading_order(tmp_path: Path):
p = tmp_path / "book.epub"
with zipfile.ZipFile(p, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr(
"OPS/Chapter01.xhtml",
"<html><body><p>First chapter content here, plenty of text.</p></body></html>",
)
zf.writestr(
"OPS/Chapter02.xhtml",
"<html><body><p>Second chapter content here, plenty of text.</p></body></html>",
)
result = extract_attachment(p)
assert result.status == "ok"
assert result.text.index("First chapter") < result.text.index("Second chapter")