fix(pfs): CPT parser resolves guideline-reprint duplicates structurally; cpt_code_alt keeps the alternates (refs #687)
C9: some codes print twice with a real descriptor both times — a guideline "Unlisted Service or Procedure" summary table, a "Qualifying Circumstances for Anesthesia" cross-reference reprint, a section-level variant of the same pattern — distinct from C8's placeholder rows (already dropped, never real text). A title-string denylist was tried and rejected: excluding "Unlisted Service" sections silently drops the sole real entries for codes 90749/91299 in the 2022 edition, proving title matching isn't a safe signal on its own. _resolve_alternates (pfs/cpt_epub.py, run at edition level in parse_epub after every chapter is parsed) scores every entry for a duplicated code instead: +2 if the entry's section (or its nearest ancestor with a TOC code range) has a range containing the code, +1 if that section has any range at all, +1 if the entry carries elements or a reference, -3 if any path component ends with "Guidelines". Highest score wins as canonical (ties: first in document order); the rest become CptAlternate(code, sec_id, reason) rows — kept, not dropped. A code that only ever appears in a guideline section has one entry and is untouched. pfs.cpt_model.CptEdition gains `alternates: tuple[CptAlternate, ...] = ()`. pfs.codetables adds pfs.cpt_code_alt (CptCodeAltRow) and write_cpt_edition writes it; the C8 one-row-per-code assertion now passes on all four real editions (verified: zero duplicate codes in 2019/2021/2022/2024).
This commit is contained in:
@@ -8,12 +8,16 @@
|
|||||||
``pfs.cpt_*`` (docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
|
``pfs.cpt_*`` (docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
|
||||||
§3) hold one CPT codebook edition's own organizing structure —
|
§3) hold one CPT codebook edition's own organizing structure —
|
||||||
``cpt_section``/``cpt_code``/``cpt_instruction``/``cpt_reference``/
|
``cpt_section``/``cpt_code``/``cpt_instruction``/``cpt_reference``/
|
||||||
``cpt_crosswalk``/``cpt_list`` — written by ``pfs.cpt_load.ingest`` from
|
``cpt_crosswalk``/``cpt_list``/``cpt_code_alt`` — written by
|
||||||
``pfs.cpt_epub.parse_epub``. ``cpt_section.path_key`` is
|
``pfs.cpt_load.ingest`` from ``pfs.cpt_epub.parse_epub``.
|
||||||
``" > ".join(path)`` (C7): the same heading title can occur as several
|
``cpt_section.path_key`` is ``" > ".join(path)`` (C7): the same heading
|
||||||
``sec_id``s (2019's running-header repeats), so ``cpt_section`` keeps
|
title can occur as several ``sec_id``s (2019's running-header repeats),
|
||||||
every row and consumers group by ``path_key`` to treat those repeats as
|
so ``cpt_section`` keeps every row and consumers group by ``path_key``
|
||||||
one heading.
|
to treat those repeats as one heading. ``cpt_code_alt`` holds the
|
||||||
|
losing entries when a code prints more than once with a real descriptor
|
||||||
|
both times (C9 — the parser picks one canonical ``pfs.cpt_code`` row by
|
||||||
|
structural score and keeps the rest here with a ``reason``, rather than
|
||||||
|
either dropping or duplicating them).
|
||||||
|
|
||||||
Writers are delete-then-insert per code (events, elements) or per
|
Writers are delete-then-insert per code (events, elements) or per
|
||||||
``edition_year`` (cpt_*), or full replace (families), so a re-run is
|
``edition_year`` (cpt_*), or full replace (families), so a re-run is
|
||||||
@@ -157,6 +161,15 @@ class CptListRow:
|
|||||||
code: str
|
code: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CptCodeAltRow:
|
||||||
|
edition_year: int
|
||||||
|
item_key: str
|
||||||
|
code: str
|
||||||
|
sec_id: str
|
||||||
|
reason: str # "guidelines-reprint" | "lower-score" (C9)
|
||||||
|
|
||||||
|
|
||||||
_DDL = """
|
_DDL = """
|
||||||
CREATE SCHEMA IF NOT EXISTS pfs;
|
CREATE SCHEMA IF NOT EXISTS pfs;
|
||||||
CREATE TABLE IF NOT EXISTS pfs.code_element (
|
CREATE TABLE IF NOT EXISTS pfs.code_element (
|
||||||
@@ -190,6 +203,8 @@ CREATE TABLE IF NOT EXISTS pfs.cpt_crosswalk (
|
|||||||
year_deleted VARCHAR, citations VARCHAR);
|
year_deleted VARCHAR, citations VARCHAR);
|
||||||
CREATE TABLE IF NOT EXISTS pfs.cpt_list (
|
CREATE TABLE IF NOT EXISTS pfs.cpt_list (
|
||||||
edition_year INTEGER, item_key VARCHAR, appendix VARCHAR, code VARCHAR);
|
edition_year INTEGER, item_key VARCHAR, appendix VARCHAR, code VARCHAR);
|
||||||
|
CREATE TABLE IF NOT EXISTS pfs.cpt_code_alt (
|
||||||
|
edition_year INTEGER, item_key VARCHAR, code VARCHAR, sec_id VARCHAR, reason VARCHAR);
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
@@ -288,6 +303,7 @@ _CPT_TABLES = (
|
|||||||
"pfs.cpt_reference",
|
"pfs.cpt_reference",
|
||||||
"pfs.cpt_crosswalk",
|
"pfs.cpt_crosswalk",
|
||||||
"pfs.cpt_list",
|
"pfs.cpt_list",
|
||||||
|
"pfs.cpt_code_alt",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -296,7 +312,10 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
|
|||||||
``pfs.cpt_*``, keyed by ``edition.year``. Returns rows written per
|
``pfs.cpt_*``, keyed by ``edition.year``. Returns rows written per
|
||||||
table. C8: asserts one ``pfs.cpt_code`` row per (edition_year, code)
|
table. C8: asserts one ``pfs.cpt_code`` row per (edition_year, code)
|
||||||
before inserting — a duplicate means a resequenced placeholder row
|
before inserting — a duplicate means a resequenced placeholder row
|
||||||
slipped past the parser's own filter (``pfs.cpt_epub``)."""
|
slipped past the parser's own filter (``pfs.cpt_epub``). C9's
|
||||||
|
guideline-reprint/lower-score losers are resolved by the parser
|
||||||
|
before this ever runs (``edition.alternates``, written to
|
||||||
|
``pfs.cpt_code_alt``), so that assertion should now always pass."""
|
||||||
ensure_tables(con)
|
ensure_tables(con)
|
||||||
year = edition.year
|
year = edition.year
|
||||||
|
|
||||||
@@ -388,6 +407,16 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
|
|||||||
)
|
)
|
||||||
for e in edition.lists
|
for e in edition.lists
|
||||||
]
|
]
|
||||||
|
alt_rows = [
|
||||||
|
CptCodeAltRow(
|
||||||
|
edition_year=year,
|
||||||
|
item_key=item_key,
|
||||||
|
code=a.code,
|
||||||
|
sec_id=a.sec_id,
|
||||||
|
reason=a.reason,
|
||||||
|
)
|
||||||
|
for a in edition.alternates
|
||||||
|
]
|
||||||
|
|
||||||
for table in _CPT_TABLES:
|
for table in _CPT_TABLES:
|
||||||
con.execute(f"DELETE FROM {table} WHERE edition_year = ?", [year])
|
con.execute(f"DELETE FROM {table} WHERE edition_year = ?", [year])
|
||||||
@@ -399,6 +428,7 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
|
|||||||
"references": _insert(con, "pfs.cpt_reference", reference_rows),
|
"references": _insert(con, "pfs.cpt_reference", reference_rows),
|
||||||
"crosswalks": _insert(con, "pfs.cpt_crosswalk", crosswalk_rows),
|
"crosswalks": _insert(con, "pfs.cpt_crosswalk", crosswalk_rows),
|
||||||
"lists": _insert(con, "pfs.cpt_list", list_rows),
|
"lists": _insert(con, "pfs.cpt_list", list_rows),
|
||||||
|
"alternates": _insert(con, "pfs.cpt_code_alt", alt_rows),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ from html.parser import HTMLParser
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from pfs.cpt_model import (
|
from pfs.cpt_model import (
|
||||||
|
CptAlternate,
|
||||||
CptCode,
|
CptCode,
|
||||||
CptCrosswalk,
|
CptCrosswalk,
|
||||||
CptEdition,
|
CptEdition,
|
||||||
@@ -676,6 +677,101 @@ def _chapter_groups(names: list[str]) -> list[list[str]]:
|
|||||||
return [sorted(grouped[key]) for key in order]
|
return [sorted(grouped[key]) for key in order]
|
||||||
|
|
||||||
|
|
||||||
|
def _has_range(sec: CptSection) -> bool:
|
||||||
|
return bool(sec.code_lo and sec.code_hi)
|
||||||
|
|
||||||
|
|
||||||
|
def _in_range(code: str, lo: str, hi: str) -> bool:
|
||||||
|
"""5-char string comparison (the brief's own convention) — CPT
|
||||||
|
codes are always 5 characters, and this reads Category II/III
|
||||||
|
(``NNNNF``/``NNNNT``) ranges the same way as Category I ranges since
|
||||||
|
the letter suffix is constant within a range."""
|
||||||
|
if not lo or not hi:
|
||||||
|
return False
|
||||||
|
return lo <= code <= hi
|
||||||
|
|
||||||
|
|
||||||
|
def _is_guideline_path(path: tuple[str, ...]) -> bool:
|
||||||
|
return any(p.strip().lower().endswith("guidelines") for p in path)
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_alternates(
|
||||||
|
sections: list[CptSection],
|
||||||
|
codes: list[CptCode],
|
||||||
|
references: list[CptReference],
|
||||||
|
) -> tuple[list[CptCode], list[CptAlternate]]:
|
||||||
|
"""C9: some codes print twice with a full, real-looking descriptor
|
||||||
|
both times — a guideline "Unlisted Service or Procedure" summary
|
||||||
|
table, a "Qualifying Circumstances for Anesthesia" cross-reference
|
||||||
|
reprint, etc. — structurally distinct from C8's placeholder rows
|
||||||
|
(already dropped upstream, never reaching here: both entries seen
|
||||||
|
here carry real text). For each code with more than one ``CptCode``
|
||||||
|
entry, score every entry and keep the highest (ties: first in
|
||||||
|
document order) as canonical; the rest become ``CptAlternate`` rows
|
||||||
|
instead of silently vanishing or duplicating. A code with only one
|
||||||
|
entry — even one that lives entirely inside a guideline section —
|
||||||
|
is untouched and always canonical."""
|
||||||
|
sections_by_id = {s.sec_id: s for s in sections}
|
||||||
|
sections_by_path: dict[tuple[str, ...], list[CptSection]] = {}
|
||||||
|
for s in sections:
|
||||||
|
sections_by_path.setdefault(s.path, []).append(s)
|
||||||
|
codes_with_references = {r.code for r in references}
|
||||||
|
|
||||||
|
def resolved_range_section(sec: CptSection) -> CptSection | None:
|
||||||
|
if _has_range(sec):
|
||||||
|
return sec
|
||||||
|
for k in range(len(sec.path) - 1, 0, -1):
|
||||||
|
for cand in sections_by_path.get(sec.path[:k], ()):
|
||||||
|
if _has_range(cand):
|
||||||
|
return cand
|
||||||
|
return None
|
||||||
|
|
||||||
|
def score(entry: CptCode) -> int:
|
||||||
|
s = 0
|
||||||
|
sec = sections_by_id.get(entry.sec_id)
|
||||||
|
if sec is not None:
|
||||||
|
resolved = resolved_range_section(sec)
|
||||||
|
if resolved is not None:
|
||||||
|
s += 1
|
||||||
|
if _in_range(entry.code, resolved.code_lo, resolved.code_hi):
|
||||||
|
s += 2
|
||||||
|
if _is_guideline_path(sec.path):
|
||||||
|
s -= 3
|
||||||
|
if entry.elements or entry.code in codes_with_references:
|
||||||
|
s += 1
|
||||||
|
return s
|
||||||
|
|
||||||
|
by_code_idx: dict[str, list[int]] = {}
|
||||||
|
for i, c in enumerate(codes):
|
||||||
|
by_code_idx.setdefault(c.code, []).append(i)
|
||||||
|
|
||||||
|
drop: set[int] = set()
|
||||||
|
alternates: list[CptAlternate] = []
|
||||||
|
for code, idxs in by_code_idx.items():
|
||||||
|
if len(idxs) < 2:
|
||||||
|
continue
|
||||||
|
scored = [(score(codes[i]), i) for i in idxs]
|
||||||
|
best_score = max(s for s, _ in scored)
|
||||||
|
best_i = next(i for s, i in scored if s == best_score)
|
||||||
|
for _s, i in scored:
|
||||||
|
if i == best_i:
|
||||||
|
continue
|
||||||
|
drop.add(i)
|
||||||
|
entry = codes[i]
|
||||||
|
sec = sections_by_id.get(entry.sec_id)
|
||||||
|
reason = (
|
||||||
|
"guidelines-reprint"
|
||||||
|
if sec is not None and _is_guideline_path(sec.path)
|
||||||
|
else "lower-score"
|
||||||
|
)
|
||||||
|
alternates.append(
|
||||||
|
CptAlternate(code=code, sec_id=entry.sec_id, reason=reason)
|
||||||
|
)
|
||||||
|
|
||||||
|
canonical = [c for i, c in enumerate(codes) if i not in drop]
|
||||||
|
return canonical, alternates
|
||||||
|
|
||||||
|
|
||||||
def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
||||||
"""Parse one CPT Professional edition EPUB into a ``CptEdition``.
|
"""Parse one CPT Professional edition EPUB into a ``CptEdition``.
|
||||||
|
|
||||||
@@ -744,12 +840,18 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
|
|||||||
elif letter in _LIST_APPENDICES:
|
elif letter in _LIST_APPENDICES:
|
||||||
lists.extend(_parse_appendix_list(text, letter))
|
lists.extend(_parse_appendix_list(text, letter))
|
||||||
|
|
||||||
|
# C9: resolve any code that printed more than once (real descriptor
|
||||||
|
# both times — never the C8 placeholder, already dropped) down to
|
||||||
|
# one canonical CptCode, keeping the rest as CptAlternate rows.
|
||||||
|
canonical_codes, alternates = _resolve_alternates(fixed_sections, codes, references)
|
||||||
|
|
||||||
return CptEdition(
|
return CptEdition(
|
||||||
year=year,
|
year=year,
|
||||||
sections=tuple(fixed_sections),
|
sections=tuple(fixed_sections),
|
||||||
codes=tuple(codes),
|
codes=tuple(canonical_codes),
|
||||||
instructions=tuple(instructions),
|
instructions=tuple(instructions),
|
||||||
references=tuple(references),
|
references=tuple(references),
|
||||||
crosswalks=tuple(crosswalks),
|
crosswalks=tuple(crosswalks),
|
||||||
lists=tuple(lists),
|
lists=tuple(lists),
|
||||||
|
alternates=tuple(alternates),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -111,6 +111,7 @@ def ingest(
|
|||||||
"references": len(edition.references),
|
"references": len(edition.references),
|
||||||
"crosswalks": len(edition.crosswalks),
|
"crosswalks": len(edition.crosswalks),
|
||||||
"lists": len(edition.lists),
|
"lists": len(edition.lists),
|
||||||
|
"alternates": len(edition.alternates),
|
||||||
}
|
}
|
||||||
else:
|
else:
|
||||||
out[year] = write_cpt_edition(con, edition, item_key)
|
out[year] = write_cpt_edition(con, edition, item_key)
|
||||||
|
|||||||
@@ -106,6 +106,22 @@ class CptListEntry:
|
|||||||
code: str
|
code: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CptAlternate:
|
||||||
|
"""A losing duplicate ``CptCode`` entry for a code the book prints
|
||||||
|
more than once (C9) — a guideline "Unlisted Service or Procedure"
|
||||||
|
summary table's copy, a "Qualifying Circumstances for Anesthesia"
|
||||||
|
cross-reference reprint, etc. — structurally distinct from C8's
|
||||||
|
resequenced-code placeholder (which the parser drops entirely, never
|
||||||
|
reaching this stage): here *both* entries carry a real descriptor,
|
||||||
|
so the loser is kept, not discarded, as a record of what didn't
|
||||||
|
become the canonical ``CptCode`` row and why."""
|
||||||
|
|
||||||
|
code: str
|
||||||
|
sec_id: str
|
||||||
|
reason: str # "guidelines-reprint" | "lower-score"
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
class CptEdition:
|
class CptEdition:
|
||||||
"""Everything parsed from one CPT codebook edition."""
|
"""Everything parsed from one CPT codebook edition."""
|
||||||
@@ -117,3 +133,4 @@ class CptEdition:
|
|||||||
references: tuple[CptReference, ...]
|
references: tuple[CptReference, ...]
|
||||||
crosswalks: tuple[CptCrosswalk, ...]
|
crosswalks: tuple[CptCrosswalk, ...]
|
||||||
lists: tuple[CptListEntry, ...]
|
lists: tuple[CptListEntry, ...]
|
||||||
|
alternates: tuple[CptAlternate, ...] = ()
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ from pfs.codetables import (
|
|||||||
write_families,
|
write_families,
|
||||||
)
|
)
|
||||||
from pfs.cpt_model import (
|
from pfs.cpt_model import (
|
||||||
|
CptAlternate,
|
||||||
CptCode,
|
CptCode,
|
||||||
CptCrosswalk,
|
CptCrosswalk,
|
||||||
CptEdition,
|
CptEdition,
|
||||||
@@ -70,6 +71,7 @@ class TestDDL:
|
|||||||
"cpt_reference",
|
"cpt_reference",
|
||||||
"cpt_crosswalk",
|
"cpt_crosswalk",
|
||||||
"cpt_list",
|
"cpt_list",
|
||||||
|
"cpt_code_alt",
|
||||||
} <= names
|
} <= names
|
||||||
|
|
||||||
|
|
||||||
@@ -201,7 +203,7 @@ class TestFamilies:
|
|||||||
assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on"
|
assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on"
|
||||||
|
|
||||||
|
|
||||||
def _tiny_edition(year=2024, codes=None):
|
def _tiny_edition(year=2024, codes=None, alternates=()):
|
||||||
sections = (
|
sections = (
|
||||||
CptSection(
|
CptSection(
|
||||||
sec_id="sec_1",
|
sec_id="sec_1",
|
||||||
@@ -291,6 +293,7 @@ def _tiny_edition(year=2024, codes=None):
|
|||||||
references=references,
|
references=references,
|
||||||
crosswalks=crosswalks,
|
crosswalks=crosswalks,
|
||||||
lists=lists,
|
lists=lists,
|
||||||
|
alternates=alternates,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -304,8 +307,20 @@ class TestCptEdition:
|
|||||||
"references": 1,
|
"references": 1,
|
||||||
"crosswalks": 1,
|
"crosswalks": 1,
|
||||||
"lists": 1,
|
"lists": 1,
|
||||||
|
"alternates": 0,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def test_alternates_round_trip(self, con):
|
||||||
|
alt = CptAlternate(
|
||||||
|
code="99490", sec_id="sec_guideline", reason="guidelines-reprint"
|
||||||
|
)
|
||||||
|
counts = write_cpt_edition(con, _tiny_edition(alternates=(alt,)), "GQGTPGYV")
|
||||||
|
assert counts["alternates"] == 1
|
||||||
|
row = con.execute(
|
||||||
|
"SELECT edition_year, item_key, code, sec_id, reason FROM pfs.cpt_code_alt"
|
||||||
|
).fetchone()
|
||||||
|
assert row == (2024, "GQGTPGYV", "99490", "sec_guideline", "guidelines-reprint")
|
||||||
|
|
||||||
def test_round_trip_reads(self, con):
|
def test_round_trip_reads(self, con):
|
||||||
write_cpt_edition(con, _tiny_edition(), "GQGTPGYV")
|
write_cpt_edition(con, _tiny_edition(), "GQGTPGYV")
|
||||||
|
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ they assert structure and counts only, never descriptor text.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import zipfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
@@ -336,6 +337,90 @@ class TestReferences:
|
|||||||
assert refs[0].years == (2022,)
|
assert refs[0].years == (2022,)
|
||||||
|
|
||||||
|
|
||||||
|
# --- C9: guideline-reprint / cross-reference duplicates ------------------
|
||||||
|
#
|
||||||
|
# Unlike C8's resequenced placeholder (a bare pointer row with no real
|
||||||
|
# descriptor, dropped entirely by the parser), C9 handles a code that
|
||||||
|
# prints *twice with a real descriptor both times* — once inside a
|
||||||
|
# guideline "convenience" table (no TOC code range of its own) and once
|
||||||
|
# under its real, ranged home section. This can only be exercised at
|
||||||
|
# the ``parse_epub`` (edition) level, since the resolution runs after
|
||||||
|
# every chapter has been parsed — so these tests build a tiny synthetic
|
||||||
|
# EPUB rather than using ``cpt_sample.xhtml``/``parse_xhtml``.
|
||||||
|
|
||||||
|
|
||||||
|
def _build_epub(path: Path, chapter_body: str) -> None:
|
||||||
|
with zipfile.ZipFile(path, "w") as zf:
|
||||||
|
zf.writestr("mimetype", "application/epub+zip")
|
||||||
|
zf.writestr("OPS/Chapter01.xhtml", f"<html><body>{chapter_body}</body></html>")
|
||||||
|
|
||||||
|
|
||||||
|
_C9_CHAPTER = """
|
||||||
|
<div class="h1-toc"><a href="body.xhtml#sec_r"><span class="green">Sample Procedures* (54321-54329)</span></a></div>
|
||||||
|
|
||||||
|
<div class="h1" id="sec_g1">Surgery Guidelines</div>
|
||||||
|
<div class="h2" id="sec_g2">Unlisted Service or Procedure</div>
|
||||||
|
<div class="noindent">The “Unlisted Procedures” and accompanying codes are as follows:</div>
|
||||||
|
<table class="table1"><tbody>
|
||||||
|
<tr>
|
||||||
|
<td class="td-w1"><div class="table-para"><b>54321</b></div></td>
|
||||||
|
<td class="td"><div class="table-para1">Unlisted procedure, sample</div></td>
|
||||||
|
</tr>
|
||||||
|
<tr>
|
||||||
|
<td class="td-w1"><div class="table-para"><b>54322</b></div></td>
|
||||||
|
<td class="td"><div class="table-para1">Unlisted procedure, only ever printed in the guidelines</div></td>
|
||||||
|
</tr>
|
||||||
|
</tbody></table>
|
||||||
|
|
||||||
|
<div class="h1" id="sec_r">Sample Procedures</div>
|
||||||
|
<div class="noindent">Guideline text for the real, ranged section.</div>
|
||||||
|
<table class="table1"><tbody>
|
||||||
|
<tr>
|
||||||
|
<td class="td-w1" id="code_54321"><div class="table-para"><b>54321</b></div></td>
|
||||||
|
<td class="td"><div class="table-para1">Unlisted procedure, sample with the following required elements:</div>
|
||||||
|
<div class="table-slist">first element;</div>
|
||||||
|
<div class="table-RT"><span class="blue"><span class="ama-en">➲</span></span><i>CPT Changes: An Insider's View</i> 2020</div></td>
|
||||||
|
</tr>
|
||||||
|
</tbody></table>
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="module")
|
||||||
|
def c9_edition(tmp_path_factory):
|
||||||
|
path = tmp_path_factory.mktemp("c9") / "sample.epub"
|
||||||
|
_build_epub(path, _C9_CHAPTER)
|
||||||
|
return parse_epub(path, year=2024)
|
||||||
|
|
||||||
|
|
||||||
|
class TestAlternates:
|
||||||
|
def test_ranged_entry_wins_as_canonical(self, c9_edition):
|
||||||
|
matches = [c for c in c9_edition.codes if c.code == "54321"]
|
||||||
|
assert len(matches) == 1
|
||||||
|
assert matches[0].sec_id == "sec_r"
|
||||||
|
assert matches[0].elements # the real entry, not the bare guideline row
|
||||||
|
|
||||||
|
def test_guideline_entry_becomes_an_alternate(self, c9_edition):
|
||||||
|
alts = [a for a in c9_edition.alternates if a.code == "54321"]
|
||||||
|
assert len(alts) == 1
|
||||||
|
assert alts[0].sec_id == "sec_g2"
|
||||||
|
assert alts[0].reason == "guidelines-reprint"
|
||||||
|
|
||||||
|
def test_code_only_in_guidelines_is_kept_as_canonical(self, c9_edition):
|
||||||
|
matches = [c for c in c9_edition.codes if c.code == "54322"]
|
||||||
|
assert len(matches) == 1
|
||||||
|
assert matches[0].sec_id == "sec_g2"
|
||||||
|
alts = [a for a in c9_edition.alternates if a.code == "54322"]
|
||||||
|
assert alts == []
|
||||||
|
|
||||||
|
def test_c8_assertion_would_now_pass(self, c9_edition):
|
||||||
|
# write_cpt_edition's own "one row per (edition_year, code)"
|
||||||
|
# check (C8) — the whole point of C9 is that this always holds.
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
counts = Counter(c.code for c in c9_edition.codes)
|
||||||
|
assert all(n == 1 for n in counts.values())
|
||||||
|
|
||||||
|
|
||||||
# --- Real-file integration tests (skipped when the file is absent) ---
|
# --- Real-file integration tests (skipped when the file is absent) ---
|
||||||
|
|
||||||
|
|
||||||
@@ -402,6 +487,24 @@ class TestReal2024:
|
|||||||
def test_no_code_has_an_empty_sec_id(self, edition):
|
def test_no_code_has_an_empty_sec_id(self, edition):
|
||||||
assert all(c.sec_id != "" for c in edition.codes)
|
assert all(c.sec_id != "" for c in edition.codes)
|
||||||
|
|
||||||
|
def test_c9_zero_codes_with_more_than_one_canonical_row(self, edition):
|
||||||
|
# C9: every code the real book prints more than once (guideline
|
||||||
|
# "Unlisted Service or Procedure" summary tables, "Qualifying
|
||||||
|
# Circumstances for Anesthesia" cross-reference reprints, and
|
||||||
|
# section-specific variants of the same pattern) must resolve
|
||||||
|
# to exactly one canonical pfs.cpt_code row — write_cpt_edition's
|
||||||
|
# own C8 assertion depends on this.
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
counts = Counter(c.code for c in edition.codes)
|
||||||
|
dupes = sorted(code for code, n in counts.items() if n > 1)
|
||||||
|
assert dupes == [], f"duplicate canonical codes: {dupes}"
|
||||||
|
# Not a hard-coded expectation of the real book's content (that
|
||||||
|
# would be fragile) — just proof the mechanism actually engaged
|
||||||
|
# on real data, and a number for the report.
|
||||||
|
assert len(edition.alternates) > 100
|
||||||
|
print(f"2024 alternates: {len(edition.alternates)}")
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
|
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
|
||||||
def test_real_2022_parses():
|
def test_real_2022_parses():
|
||||||
@@ -410,6 +513,16 @@ def test_real_2022_parses():
|
|||||||
assert edition.sections
|
assert edition.sections
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
|
||||||
|
def test_real_2022_c9_zero_duplicate_canonical_codes():
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
edition = parse_epub(EPUB_2022, year=2022)
|
||||||
|
counts = Counter(c.code for c in edition.codes)
|
||||||
|
dupes = sorted(code for code, n in counts.items() if n > 1)
|
||||||
|
assert dupes == [], f"duplicate canonical codes: {dupes}"
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
|
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
|
||||||
def test_real_2021_parses():
|
def test_real_2021_parses():
|
||||||
edition = parse_epub(EPUB_2021, year=2021)
|
edition = parse_epub(EPUB_2021, year=2021)
|
||||||
@@ -417,6 +530,16 @@ def test_real_2021_parses():
|
|||||||
assert edition.sections
|
assert edition.sections
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
|
||||||
|
def test_real_2021_c9_zero_duplicate_canonical_codes():
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
edition = parse_epub(EPUB_2021, year=2021)
|
||||||
|
counts = Counter(c.code for c in edition.codes)
|
||||||
|
dupes = sorted(code for code, n in counts.items() if n > 1)
|
||||||
|
assert dupes == [], f"duplicate canonical codes: {dupes}"
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
|
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
|
||||||
def test_real_2019_parses_without_code_ids():
|
def test_real_2019_parses_without_code_ids():
|
||||||
edition = parse_epub(EPUB_2019, year=2019)
|
edition = parse_epub(EPUB_2019, year=2019)
|
||||||
@@ -425,3 +548,13 @@ def test_real_2019_parses_without_code_ids():
|
|||||||
# 2019 headings mostly lack an id in the source markup — every
|
# 2019 headings mostly lack an id in the source markup — every
|
||||||
# code must still land under a non-empty (real or synthetic) sec_id.
|
# code must still land under a non-empty (real or synthetic) sec_id.
|
||||||
assert all(c.sec_id != "" for c in edition.codes)
|
assert all(c.sec_id != "" for c in edition.codes)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
|
||||||
|
def test_real_2019_c9_zero_duplicate_canonical_codes():
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
edition = parse_epub(EPUB_2019, year=2019)
|
||||||
|
counts = Counter(c.code for c in edition.codes)
|
||||||
|
dupes = sorted(code for code, n in counts.items() if n > 1)
|
||||||
|
assert dupes == [], f"duplicate canonical codes: {dupes}"
|
||||||
|
|||||||
@@ -151,6 +151,7 @@ class TestIngestDryRun:
|
|||||||
"references": 0,
|
"references": 0,
|
||||||
"crosswalks": 0,
|
"crosswalks": 0,
|
||||||
"lists": 0,
|
"lists": 0,
|
||||||
|
"alternates": 0,
|
||||||
}
|
}
|
||||||
assert out[2021]["codes"] == 1
|
assert out[2021]["codes"] == 1
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user