Files
stack/tests/pfs/test_cpt_epub.py

410 lines
15 KiB
Python

"""Tests for pfs.cpt_epub — the pure CPT EPUB parser.
The synthetic fixture (tests/pfs/fixtures/cpt_sample.xhtml) is written
in the 2024 template's markup style with invented text; it is never
copied from the real AMA codebook. The integration tests at the bottom
run the real parser against the actual EPUB files on disk (read-only,
in data/zotero/data/storage/) and are skipped when a file is absent —
they assert structure and counts only, never descriptor text.
"""
from __future__ import annotations
from pathlib import Path
import pytest
from pfs.cpt_epub import parse_epub, parse_xhtml
FIXTURE = Path(__file__).parent / "fixtures" / "cpt_sample.xhtml"
REPO_ROOT = Path(__file__).resolve().parents[2]
EPUB_2024 = (
REPO_ROOT
/ "data/zotero/data/storage/EIGIRKRK/CPT Professional 2024 - American Medical Association.epub"
)
EPUB_2022 = REPO_ROOT / "data/zotero/data/storage/UG4R55FW/CPT Professional 2022.epub"
EPUB_2021 = (
REPO_ROOT / "data/zotero/data/storage/NM3NJZV5/CPT 2021 Professional Edition.epub"
)
EPUB_2019 = REPO_ROOT / "data/zotero/data/storage/VA34EEIM/CPT 2019.epub"
@pytest.fixture(scope="module")
def parsed():
text = FIXTURE.read_text(encoding="utf-8")
sections, codes, instructions, references = parse_xhtml(
text, year=2024, source="cpt_sample.xhtml"
)
return {
"sections": {s.sec_id: s for s in sections},
"sections_list": sections,
"codes": {c.code: c for c in codes},
"instructions": instructions,
"references": references,
}
class TestSections:
def test_two_top_level_and_two_nested(self, parsed):
levels = sorted(
(s.sec_id, s.level)
for s in parsed["sections_list"]
if s.sec_id in {"sec_1", "sec_2", "sec_3", "sec_4"}
)
assert levels == [
("sec_1", 1),
("sec_2", 2),
("sec_3", 1),
("sec_4", 2),
]
# plus the id-less "No Id Category Services" h1 (TestSecIdFallback)
assert len(parsed["sections_list"]) == 5
def test_paths_include_ancestors(self, parsed):
sec2 = parsed["sections"]["sec_2"]
assert sec2.title == "Sample Chronic Wellness Services"
assert sec2.path == (
"Care Coordination Services",
"Sample Chronic Wellness Services",
)
sec4 = parsed["sections"]["sec_4"]
assert sec4.path == ("Other Sample Services", "Empty Category Services")
def test_lo_hi_from_toc_snippet(self, parsed):
sec1 = parsed["sections"]["sec_1"]
assert (sec1.code_lo, sec1.code_hi) == ("55000", "55010")
sec2 = parsed["sections"]["sec_2"]
assert (sec2.code_lo, sec2.code_hi) == ("55000", "55003")
sec3 = parsed["sections"]["sec_3"]
assert (sec3.code_lo, sec3.code_hi) == ("55100", "55101")
def test_heading_without_range_is_empty(self, parsed):
sec4 = parsed["sections"]["sec_4"]
assert (sec4.code_lo, sec4.code_hi) == ("", "")
def test_section_with_no_codes_still_has_a_path(self, parsed):
sec4 = parsed["sections"]["sec_4"]
assert sec4.path
assert sec4.guideline
def test_entities_and_whitespace_cleaned(self, parsed):
sec1 = parsed["sections"]["sec_1"]
assert "&" not in sec1.guideline
assert "&" in sec1.guideline # entity decoded, not dropped
assert " " not in sec1.guideline
class TestSecIdFallback:
""" "No Id Category Services" (the fixture's last h1) has no ``id``
attribute and no nested anchor carrying one — same as most 2019
headings. It must still get a unique, non-empty ``sec_id`` (never
collapse onto ``""`` alongside every other id-less heading), and
the code under it must carry that same id."""
def test_no_id_heading_gets_a_synthetic_sec_id(self, parsed):
no_id_section = next(
s for s in parsed["sections_list"] if s.title == "No Id Category Services"
)
assert no_id_section.sec_id != ""
assert no_id_section.sec_id.startswith("cpt_sample.xhtml:h")
def test_synthetic_sec_ids_are_unique(self, parsed):
synthetic = [
s.sec_id
for s in parsed["sections_list"]
if s.sec_id.startswith("cpt_sample.xhtml:h")
]
assert len(synthetic) == len(set(synthetic))
assert synthetic # at least the one id-less heading produced one
def test_code_under_id_less_heading_gets_the_synthetic_sec_id(self, parsed):
no_id_section = next(
s for s in parsed["sections_list"] if s.title == "No Id Category Services"
)
code = parsed["codes"]["57000"]
assert code.sec_id == no_id_section.sec_id
assert code.sec_id != ""
class TestPrimaryCode:
def test_three_elements_and_tail(self, parsed):
code = parsed["codes"]["55000"]
assert code.sec_id == "sec_2"
assert len(code.elements) == 3
assert code.elements[0] == "first invented element of the service,"
assert (
code.tail == "first 20 minutes of invented staff time, per calendar month."
)
assert (
code.stem
== "Sample wellness coordination services with the following required elements:"
)
def test_symbols(self, parsed):
code = parsed["codes"]["55000"]
assert code.resequenced is True
assert code.addon is False
assert code.new is False
assert code.revised is False
assert code.telemedicine is False
assert code.parent == ""
def test_category_i(self, parsed):
assert parsed["codes"]["55000"].category == "I"
def test_descriptor_assembled(self, parsed):
code = parsed["codes"]["55000"]
assert code.stem in code.descriptor
assert code.tail in code.descriptor
assert code.elements[0].rstrip(",;") in code.descriptor
def test_descriptor_element_join_has_no_double_punctuation(self, parsed):
code = parsed["codes"]["55000"]
assert ",;" not in code.descriptor
assert ";;" not in code.descriptor
class TestSemicolonRule:
"""25100 "Arthrotomy, wrist joint; with biopsy" / 25105 "with
synovectomy" is the book's own worked example of the indentation
convention (Introduction, "Format of the Terminology"): an
indented child's descriptor is the parent's pre-semicolon stem
plus the child's own fragment — never the parent's *whole* stem
(which would duplicate the parent's own post-semicolon text)."""
def test_parent_stem_excludes_own_continuation(self, parsed):
parent = parsed["codes"]["56000"]
assert parent.stem == "Sample incision, deep fascial plane"
assert "with exploratory biopsy" not in parent.stem
def test_parent_descriptor_has_the_continuation(self, parsed):
parent = parsed["codes"]["56000"]
assert (
parent.descriptor
== "Sample incision, deep fascial plane; with exploratory biopsy"
)
def test_child_inherits_only_the_shared_stem(self, parsed):
child = parsed["codes"]["56005"]
assert child.stem == "Sample incision, deep fascial plane"
assert child.parent == "56000"
def test_child_descriptor_substitutes_its_own_fragment(self, parsed):
child = parsed["codes"]["56005"]
assert (
child.descriptor
== "Sample incision, deep fascial plane; with synovial debridement"
)
# the parent's own fragment must not leak into the child's descriptor
assert "exploratory biopsy" not in child.descriptor
class TestAddonCode:
def test_addon_flag_and_parent(self, parsed):
code = parsed["codes"]["55001"]
assert code.addon is True
assert code.resequenced is True
assert code.parent == "55000"
def test_inherits_stem_and_elements(self, parsed):
addon = parsed["codes"]["55001"]
primary = parsed["codes"]["55000"]
assert addon.stem == primary.stem
assert addon.elements == primary.elements
def test_own_tail(self, parsed):
addon = parsed["codes"]["55001"]
assert "each additional 20 minutes" in addon.tail
assert "primary procedure" in addon.tail
class TestInstructions:
def test_use_with_targets_primary(self, parsed):
use_with = [
i
for i in parsed["instructions"]
if i.code == "55001" and i.kind == "use-with"
]
assert len(use_with) == 1
assert use_with[0].targets == ("55000",)
def test_not_with_targets_expanded_range_plus_listed(self, parsed):
not_with = [
i
for i in parsed["instructions"]
if i.code == "55001" and i.kind == "not-with"
]
assert len(not_with) == 1
targets = not_with[0].targets
expanded_range = tuple(f"909{n}" for n in range(51, 71))
assert len(expanded_range) == 20
for t in expanded_range:
assert t in targets
assert "55555" in targets
def test_plain_parenthetical_is_other(self, parsed):
others = [
i for i in parsed["instructions"] if i.code == "55001" and i.kind == "other"
]
assert len(others) == 3
assert others[0].text.startswith("(Sample wellness services")
def test_five_instructions_on_addon(self, parsed):
addon_instructions = [i for i in parsed["instructions"] if i.code == "55001"]
assert len(addon_instructions) == 5
def test_owner_code_excluded_from_targets(self, parsed):
# "(Do not report 55001 more than twice per calendar month)" —
# 55001 is the instruction's own owning code (a subject), not
# a target of itself.
matches = [
i
for i in parsed["instructions"]
if i.code == "55001" and "more than twice" in i.text
]
assert len(matches) == 1
assert matches[0].kind == "other"
assert matches[0].targets == ()
def test_sibling_code_kept_as_target(self, parsed):
# "(Do not report 55001 in addition to 55002 more than once
# per calendar month)" — 55001 is the owner (excluded), 55002
# is a genuine sibling code and must survive the owner filter.
matches = [
i
for i in parsed["instructions"]
if i.code == "55001" and "in addition to 55002" in i.text
]
assert len(matches) == 1
assert matches[0].targets == ("55002",)
def test_not_with_never_includes_the_owner(self, parsed):
not_with = [
i
for i in parsed["instructions"]
if i.code == "55001" and i.kind == "not-with"
]
assert "55001" not in not_with[0].targets
assert "55555" in not_with[0].targets
class TestReferences:
def test_cpt_changes_years(self, parsed):
refs = [
r
for r in parsed["references"]
if r.code == "55000" and r.kind == "cpt-changes"
]
assert len(refs) == 1
assert refs[0].years == (2015, 2021, 2022)
def test_cpt_assistant_kept_as_text(self, parsed):
refs = [
r
for r in parsed["references"]
if r.code == "55000" and r.kind == "cpt-assistant"
]
assert len(refs) == 1
assert "Jan 21:5" in refs[0].text
def test_addon_reference_year(self, parsed):
refs = [
r
for r in parsed["references"]
if r.code == "55001" and r.kind == "cpt-changes"
]
assert refs[0].years == (2022,)
# --- Real-file integration tests (skipped when the file is absent) ---
@pytest.mark.skipif(not EPUB_2024.exists(), reason="real 2024 CPT EPUB not on disk")
class TestReal2024:
@pytest.fixture(scope="class")
def edition(self):
return parse_epub(EPUB_2024, year=2024)
def test_at_least_8000_codes(self, edition):
assert len(edition.codes) >= 8000
def test_every_section_has_a_path(self, edition):
assert edition.sections
for sec in edition.sections:
assert sec.path
assert sec.path[-1] == sec.title
def test_chronic_care_management_section_has_expected_codes(self, edition):
by_code = {c.code: c for c in edition.codes}
sec = next(
s for s in edition.sections if s.title == "Chronic Care Management Services"
)
codes_in_sec = {c.code for c in edition.codes if c.sec_id == sec.sec_id}
for code in ("99490", "99439", "99491", "99437"):
assert code in codes_in_sec, code
assert code in by_code
def test_99439_is_addon_using_99490(self, edition):
by_code = {c.code: c for c in edition.codes}
code_99439 = by_code["99439"]
assert code_99439.addon is True
use_with = [
i
for i in edition.instructions
if i.code == "99439" and i.kind == "use-with"
]
assert use_with
assert "99490" in use_with[0].targets
def test_appendix_d_addon_codes(self, edition):
d_entries = [e for e in edition.lists if e.appendix == "D"]
assert len(d_entries) >= 500
def test_appendix_m_crosswalk_former_codes_end_in_t(self, edition):
assert edition.crosswalks
assert any(row.former_code.endswith("T") for row in edition.crosswalks)
def test_25100_25105_semicolon_rule(self, edition):
# The book's own worked example (Introduction, "Format of the
# Terminology"): 25100 "Arthrotomy, wrist joint; with biopsy"
# / 25105 "with synovectomy". Structure only — never asserts
# or prints the actual descriptor text.
by_code = {c.code: c for c in edition.codes}
parent = by_code["25100"]
child = by_code["25105"]
assert child.parent == "25100"
assert child.stem == parent.stem
assert child.descriptor.startswith(parent.stem)
parent_own_fragment = parent.descriptor[len(parent.stem) :].lstrip("; ").strip()
assert parent_own_fragment # the parent really did have a post-semicolon part
assert parent_own_fragment not in child.descriptor
def test_no_code_has_an_empty_sec_id(self, edition):
assert all(c.sec_id != "" for c in edition.codes)
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
def test_real_2022_parses():
edition = parse_epub(EPUB_2022, year=2022)
assert len(edition.codes) >= 5000
assert edition.sections
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
def test_real_2021_parses():
edition = parse_epub(EPUB_2021, year=2021)
assert len(edition.codes) >= 5000
assert edition.sections
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
def test_real_2019_parses_without_code_ids():
edition = parse_epub(EPUB_2019, year=2019)
assert len(edition.codes) >= 3000
assert edition.sections
# 2019 headings mostly lack an id in the source markup — every
# code must still land under a non-empty (real or synthetic) sec_id.
assert all(c.sec_id != "" for c in edition.codes)