410 lines
15 KiB
Python
410 lines
15 KiB
Python
"""Tests for pfs.cpt_epub — the pure CPT EPUB parser.
|
|
|
|
The synthetic fixture (tests/pfs/fixtures/cpt_sample.xhtml) is written
|
|
in the 2024 template's markup style with invented text; it is never
|
|
copied from the real AMA codebook. The integration tests at the bottom
|
|
run the real parser against the actual EPUB files on disk (read-only,
|
|
in data/zotero/data/storage/) and are skipped when a file is absent —
|
|
they assert structure and counts only, never descriptor text.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from pfs.cpt_epub import parse_epub, parse_xhtml
|
|
|
|
FIXTURE = Path(__file__).parent / "fixtures" / "cpt_sample.xhtml"
|
|
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
|
|
EPUB_2024 = (
|
|
REPO_ROOT
|
|
/ "data/zotero/data/storage/EIGIRKRK/CPT Professional 2024 - American Medical Association.epub"
|
|
)
|
|
EPUB_2022 = REPO_ROOT / "data/zotero/data/storage/UG4R55FW/CPT Professional 2022.epub"
|
|
EPUB_2021 = (
|
|
REPO_ROOT / "data/zotero/data/storage/NM3NJZV5/CPT 2021 Professional Edition.epub"
|
|
)
|
|
EPUB_2019 = REPO_ROOT / "data/zotero/data/storage/VA34EEIM/CPT 2019.epub"
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def parsed():
|
|
text = FIXTURE.read_text(encoding="utf-8")
|
|
sections, codes, instructions, references = parse_xhtml(
|
|
text, year=2024, source="cpt_sample.xhtml"
|
|
)
|
|
return {
|
|
"sections": {s.sec_id: s for s in sections},
|
|
"sections_list": sections,
|
|
"codes": {c.code: c for c in codes},
|
|
"instructions": instructions,
|
|
"references": references,
|
|
}
|
|
|
|
|
|
class TestSections:
|
|
def test_two_top_level_and_two_nested(self, parsed):
|
|
levels = sorted(
|
|
(s.sec_id, s.level)
|
|
for s in parsed["sections_list"]
|
|
if s.sec_id in {"sec_1", "sec_2", "sec_3", "sec_4"}
|
|
)
|
|
assert levels == [
|
|
("sec_1", 1),
|
|
("sec_2", 2),
|
|
("sec_3", 1),
|
|
("sec_4", 2),
|
|
]
|
|
# plus the id-less "No Id Category Services" h1 (TestSecIdFallback)
|
|
assert len(parsed["sections_list"]) == 5
|
|
|
|
def test_paths_include_ancestors(self, parsed):
|
|
sec2 = parsed["sections"]["sec_2"]
|
|
assert sec2.title == "Sample Chronic Wellness Services"
|
|
assert sec2.path == (
|
|
"Care Coordination Services",
|
|
"Sample Chronic Wellness Services",
|
|
)
|
|
|
|
sec4 = parsed["sections"]["sec_4"]
|
|
assert sec4.path == ("Other Sample Services", "Empty Category Services")
|
|
|
|
def test_lo_hi_from_toc_snippet(self, parsed):
|
|
sec1 = parsed["sections"]["sec_1"]
|
|
assert (sec1.code_lo, sec1.code_hi) == ("55000", "55010")
|
|
sec2 = parsed["sections"]["sec_2"]
|
|
assert (sec2.code_lo, sec2.code_hi) == ("55000", "55003")
|
|
sec3 = parsed["sections"]["sec_3"]
|
|
assert (sec3.code_lo, sec3.code_hi) == ("55100", "55101")
|
|
|
|
def test_heading_without_range_is_empty(self, parsed):
|
|
sec4 = parsed["sections"]["sec_4"]
|
|
assert (sec4.code_lo, sec4.code_hi) == ("", "")
|
|
|
|
def test_section_with_no_codes_still_has_a_path(self, parsed):
|
|
sec4 = parsed["sections"]["sec_4"]
|
|
assert sec4.path
|
|
assert sec4.guideline
|
|
|
|
def test_entities_and_whitespace_cleaned(self, parsed):
|
|
sec1 = parsed["sections"]["sec_1"]
|
|
assert "&" not in sec1.guideline
|
|
assert "&" in sec1.guideline # entity decoded, not dropped
|
|
assert " " not in sec1.guideline
|
|
|
|
|
|
class TestSecIdFallback:
|
|
""" "No Id Category Services" (the fixture's last h1) has no ``id``
|
|
attribute and no nested anchor carrying one — same as most 2019
|
|
headings. It must still get a unique, non-empty ``sec_id`` (never
|
|
collapse onto ``""`` alongside every other id-less heading), and
|
|
the code under it must carry that same id."""
|
|
|
|
def test_no_id_heading_gets_a_synthetic_sec_id(self, parsed):
|
|
no_id_section = next(
|
|
s for s in parsed["sections_list"] if s.title == "No Id Category Services"
|
|
)
|
|
assert no_id_section.sec_id != ""
|
|
assert no_id_section.sec_id.startswith("cpt_sample.xhtml:h")
|
|
|
|
def test_synthetic_sec_ids_are_unique(self, parsed):
|
|
synthetic = [
|
|
s.sec_id
|
|
for s in parsed["sections_list"]
|
|
if s.sec_id.startswith("cpt_sample.xhtml:h")
|
|
]
|
|
assert len(synthetic) == len(set(synthetic))
|
|
assert synthetic # at least the one id-less heading produced one
|
|
|
|
def test_code_under_id_less_heading_gets_the_synthetic_sec_id(self, parsed):
|
|
no_id_section = next(
|
|
s for s in parsed["sections_list"] if s.title == "No Id Category Services"
|
|
)
|
|
code = parsed["codes"]["57000"]
|
|
assert code.sec_id == no_id_section.sec_id
|
|
assert code.sec_id != ""
|
|
|
|
|
|
class TestPrimaryCode:
|
|
def test_three_elements_and_tail(self, parsed):
|
|
code = parsed["codes"]["55000"]
|
|
assert code.sec_id == "sec_2"
|
|
assert len(code.elements) == 3
|
|
assert code.elements[0] == "first invented element of the service,"
|
|
assert (
|
|
code.tail == "first 20 minutes of invented staff time, per calendar month."
|
|
)
|
|
assert (
|
|
code.stem
|
|
== "Sample wellness coordination services with the following required elements:"
|
|
)
|
|
|
|
def test_symbols(self, parsed):
|
|
code = parsed["codes"]["55000"]
|
|
assert code.resequenced is True
|
|
assert code.addon is False
|
|
assert code.new is False
|
|
assert code.revised is False
|
|
assert code.telemedicine is False
|
|
assert code.parent == ""
|
|
|
|
def test_category_i(self, parsed):
|
|
assert parsed["codes"]["55000"].category == "I"
|
|
|
|
def test_descriptor_assembled(self, parsed):
|
|
code = parsed["codes"]["55000"]
|
|
assert code.stem in code.descriptor
|
|
assert code.tail in code.descriptor
|
|
assert code.elements[0].rstrip(",;") in code.descriptor
|
|
|
|
def test_descriptor_element_join_has_no_double_punctuation(self, parsed):
|
|
code = parsed["codes"]["55000"]
|
|
assert ",;" not in code.descriptor
|
|
assert ";;" not in code.descriptor
|
|
|
|
|
|
class TestSemicolonRule:
|
|
"""25100 "Arthrotomy, wrist joint; with biopsy" / 25105 "with
|
|
synovectomy" is the book's own worked example of the indentation
|
|
convention (Introduction, "Format of the Terminology"): an
|
|
indented child's descriptor is the parent's pre-semicolon stem
|
|
plus the child's own fragment — never the parent's *whole* stem
|
|
(which would duplicate the parent's own post-semicolon text)."""
|
|
|
|
def test_parent_stem_excludes_own_continuation(self, parsed):
|
|
parent = parsed["codes"]["56000"]
|
|
assert parent.stem == "Sample incision, deep fascial plane"
|
|
assert "with exploratory biopsy" not in parent.stem
|
|
|
|
def test_parent_descriptor_has_the_continuation(self, parsed):
|
|
parent = parsed["codes"]["56000"]
|
|
assert (
|
|
parent.descriptor
|
|
== "Sample incision, deep fascial plane; with exploratory biopsy"
|
|
)
|
|
|
|
def test_child_inherits_only_the_shared_stem(self, parsed):
|
|
child = parsed["codes"]["56005"]
|
|
assert child.stem == "Sample incision, deep fascial plane"
|
|
assert child.parent == "56000"
|
|
|
|
def test_child_descriptor_substitutes_its_own_fragment(self, parsed):
|
|
child = parsed["codes"]["56005"]
|
|
assert (
|
|
child.descriptor
|
|
== "Sample incision, deep fascial plane; with synovial debridement"
|
|
)
|
|
# the parent's own fragment must not leak into the child's descriptor
|
|
assert "exploratory biopsy" not in child.descriptor
|
|
|
|
|
|
class TestAddonCode:
|
|
def test_addon_flag_and_parent(self, parsed):
|
|
code = parsed["codes"]["55001"]
|
|
assert code.addon is True
|
|
assert code.resequenced is True
|
|
assert code.parent == "55000"
|
|
|
|
def test_inherits_stem_and_elements(self, parsed):
|
|
addon = parsed["codes"]["55001"]
|
|
primary = parsed["codes"]["55000"]
|
|
assert addon.stem == primary.stem
|
|
assert addon.elements == primary.elements
|
|
|
|
def test_own_tail(self, parsed):
|
|
addon = parsed["codes"]["55001"]
|
|
assert "each additional 20 minutes" in addon.tail
|
|
assert "primary procedure" in addon.tail
|
|
|
|
|
|
class TestInstructions:
|
|
def test_use_with_targets_primary(self, parsed):
|
|
use_with = [
|
|
i
|
|
for i in parsed["instructions"]
|
|
if i.code == "55001" and i.kind == "use-with"
|
|
]
|
|
assert len(use_with) == 1
|
|
assert use_with[0].targets == ("55000",)
|
|
|
|
def test_not_with_targets_expanded_range_plus_listed(self, parsed):
|
|
not_with = [
|
|
i
|
|
for i in parsed["instructions"]
|
|
if i.code == "55001" and i.kind == "not-with"
|
|
]
|
|
assert len(not_with) == 1
|
|
targets = not_with[0].targets
|
|
expanded_range = tuple(f"909{n}" for n in range(51, 71))
|
|
assert len(expanded_range) == 20
|
|
for t in expanded_range:
|
|
assert t in targets
|
|
assert "55555" in targets
|
|
|
|
def test_plain_parenthetical_is_other(self, parsed):
|
|
others = [
|
|
i for i in parsed["instructions"] if i.code == "55001" and i.kind == "other"
|
|
]
|
|
assert len(others) == 3
|
|
assert others[0].text.startswith("(Sample wellness services")
|
|
|
|
def test_five_instructions_on_addon(self, parsed):
|
|
addon_instructions = [i for i in parsed["instructions"] if i.code == "55001"]
|
|
assert len(addon_instructions) == 5
|
|
|
|
def test_owner_code_excluded_from_targets(self, parsed):
|
|
# "(Do not report 55001 more than twice per calendar month)" —
|
|
# 55001 is the instruction's own owning code (a subject), not
|
|
# a target of itself.
|
|
matches = [
|
|
i
|
|
for i in parsed["instructions"]
|
|
if i.code == "55001" and "more than twice" in i.text
|
|
]
|
|
assert len(matches) == 1
|
|
assert matches[0].kind == "other"
|
|
assert matches[0].targets == ()
|
|
|
|
def test_sibling_code_kept_as_target(self, parsed):
|
|
# "(Do not report 55001 in addition to 55002 more than once
|
|
# per calendar month)" — 55001 is the owner (excluded), 55002
|
|
# is a genuine sibling code and must survive the owner filter.
|
|
matches = [
|
|
i
|
|
for i in parsed["instructions"]
|
|
if i.code == "55001" and "in addition to 55002" in i.text
|
|
]
|
|
assert len(matches) == 1
|
|
assert matches[0].targets == ("55002",)
|
|
|
|
def test_not_with_never_includes_the_owner(self, parsed):
|
|
not_with = [
|
|
i
|
|
for i in parsed["instructions"]
|
|
if i.code == "55001" and i.kind == "not-with"
|
|
]
|
|
assert "55001" not in not_with[0].targets
|
|
assert "55555" in not_with[0].targets
|
|
|
|
|
|
class TestReferences:
|
|
def test_cpt_changes_years(self, parsed):
|
|
refs = [
|
|
r
|
|
for r in parsed["references"]
|
|
if r.code == "55000" and r.kind == "cpt-changes"
|
|
]
|
|
assert len(refs) == 1
|
|
assert refs[0].years == (2015, 2021, 2022)
|
|
|
|
def test_cpt_assistant_kept_as_text(self, parsed):
|
|
refs = [
|
|
r
|
|
for r in parsed["references"]
|
|
if r.code == "55000" and r.kind == "cpt-assistant"
|
|
]
|
|
assert len(refs) == 1
|
|
assert "Jan 21:5" in refs[0].text
|
|
|
|
def test_addon_reference_year(self, parsed):
|
|
refs = [
|
|
r
|
|
for r in parsed["references"]
|
|
if r.code == "55001" and r.kind == "cpt-changes"
|
|
]
|
|
assert refs[0].years == (2022,)
|
|
|
|
|
|
# --- Real-file integration tests (skipped when the file is absent) ---
|
|
|
|
|
|
@pytest.mark.skipif(not EPUB_2024.exists(), reason="real 2024 CPT EPUB not on disk")
|
|
class TestReal2024:
|
|
@pytest.fixture(scope="class")
|
|
def edition(self):
|
|
return parse_epub(EPUB_2024, year=2024)
|
|
|
|
def test_at_least_8000_codes(self, edition):
|
|
assert len(edition.codes) >= 8000
|
|
|
|
def test_every_section_has_a_path(self, edition):
|
|
assert edition.sections
|
|
for sec in edition.sections:
|
|
assert sec.path
|
|
assert sec.path[-1] == sec.title
|
|
|
|
def test_chronic_care_management_section_has_expected_codes(self, edition):
|
|
by_code = {c.code: c for c in edition.codes}
|
|
sec = next(
|
|
s for s in edition.sections if s.title == "Chronic Care Management Services"
|
|
)
|
|
codes_in_sec = {c.code for c in edition.codes if c.sec_id == sec.sec_id}
|
|
for code in ("99490", "99439", "99491", "99437"):
|
|
assert code in codes_in_sec, code
|
|
assert code in by_code
|
|
|
|
def test_99439_is_addon_using_99490(self, edition):
|
|
by_code = {c.code: c for c in edition.codes}
|
|
code_99439 = by_code["99439"]
|
|
assert code_99439.addon is True
|
|
use_with = [
|
|
i
|
|
for i in edition.instructions
|
|
if i.code == "99439" and i.kind == "use-with"
|
|
]
|
|
assert use_with
|
|
assert "99490" in use_with[0].targets
|
|
|
|
def test_appendix_d_addon_codes(self, edition):
|
|
d_entries = [e for e in edition.lists if e.appendix == "D"]
|
|
assert len(d_entries) >= 500
|
|
|
|
def test_appendix_m_crosswalk_former_codes_end_in_t(self, edition):
|
|
assert edition.crosswalks
|
|
assert any(row.former_code.endswith("T") for row in edition.crosswalks)
|
|
|
|
def test_25100_25105_semicolon_rule(self, edition):
|
|
# The book's own worked example (Introduction, "Format of the
|
|
# Terminology"): 25100 "Arthrotomy, wrist joint; with biopsy"
|
|
# / 25105 "with synovectomy". Structure only — never asserts
|
|
# or prints the actual descriptor text.
|
|
by_code = {c.code: c for c in edition.codes}
|
|
parent = by_code["25100"]
|
|
child = by_code["25105"]
|
|
assert child.parent == "25100"
|
|
assert child.stem == parent.stem
|
|
assert child.descriptor.startswith(parent.stem)
|
|
parent_own_fragment = parent.descriptor[len(parent.stem) :].lstrip("; ").strip()
|
|
assert parent_own_fragment # the parent really did have a post-semicolon part
|
|
assert parent_own_fragment not in child.descriptor
|
|
|
|
def test_no_code_has_an_empty_sec_id(self, edition):
|
|
assert all(c.sec_id != "" for c in edition.codes)
|
|
|
|
|
|
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
|
|
def test_real_2022_parses():
|
|
edition = parse_epub(EPUB_2022, year=2022)
|
|
assert len(edition.codes) >= 5000
|
|
assert edition.sections
|
|
|
|
|
|
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
|
|
def test_real_2021_parses():
|
|
edition = parse_epub(EPUB_2021, year=2021)
|
|
assert len(edition.codes) >= 5000
|
|
assert edition.sections
|
|
|
|
|
|
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
|
|
def test_real_2019_parses_without_code_ids():
|
|
edition = parse_epub(EPUB_2019, year=2019)
|
|
assert len(edition.codes) >= 3000
|
|
assert edition.sections
|
|
# 2019 headings mostly lack an id in the source markup — every
|
|
# code must still land under a non-empty (real or synthetic) sec_id.
|
|
assert all(c.sec_id != "" for c in edition.codes)
|