fix(pfs): CPT parser resolves guideline-reprint duplicates structurally; cpt_code_alt keeps the alternates (refs #687)

C9: some codes print twice with a real descriptor both times — a
guideline "Unlisted Service or Procedure" summary table, a "Qualifying
Circumstances for Anesthesia" cross-reference reprint, a section-level
variant of the same pattern — distinct from C8's placeholder rows
(already dropped, never real text). A title-string denylist was tried
and rejected: excluding "Unlisted Service" sections silently drops the
sole real entries for codes 90749/91299 in the 2022 edition, proving
title matching isn't a safe signal on its own.

_resolve_alternates (pfs/cpt_epub.py, run at edition level in
parse_epub after every chapter is parsed) scores every entry for a
duplicated code instead: +2 if the entry's section (or its nearest
ancestor with a TOC code range) has a range containing the code, +1 if
that section has any range at all, +1 if the entry carries elements or
a reference, -3 if any path component ends with "Guidelines". Highest
score wins as canonical (ties: first in document order); the rest
become CptAlternate(code, sec_id, reason) rows — kept, not dropped. A
code that only ever appears in a guideline section has one entry and
is untouched.

pfs.cpt_model.CptEdition gains `alternates: tuple[CptAlternate, ...] =
()`. pfs.codetables adds pfs.cpt_code_alt (CptCodeAltRow) and
write_cpt_edition writes it; the C8 one-row-per-code assertion now
passes on all four real editions (verified: zero duplicate codes in
2019/2021/2022/2024).
This commit is contained in:
kert
2026-09-09 18:34:14 -04:00
parent 412527dcde
commit afec69c933
7 changed files with 308 additions and 9 deletions

View File

@@ -8,12 +8,16 @@
``pfs.cpt_*`` (docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md ``pfs.cpt_*`` (docs/superpowers/specs/2026-09-09-cpt-canonical-schema-design.md
§3) hold one CPT codebook edition's own organizing structure — §3) hold one CPT codebook edition's own organizing structure —
``cpt_section``/``cpt_code``/``cpt_instruction``/``cpt_reference``/ ``cpt_section``/``cpt_code``/``cpt_instruction``/``cpt_reference``/
``cpt_crosswalk``/``cpt_list`` — written by ``pfs.cpt_load.ingest`` from ``cpt_crosswalk``/``cpt_list``/``cpt_code_alt`` — written by
``pfs.cpt_epub.parse_epub``. ``cpt_section.path_key`` is ``pfs.cpt_load.ingest`` from ``pfs.cpt_epub.parse_epub``.
``" > ".join(path)`` (C7): the same heading title can occur as several ``cpt_section.path_key`` is ``" > ".join(path)`` (C7): the same heading
``sec_id``s (2019's running-header repeats), so ``cpt_section`` keeps title can occur as several ``sec_id``s (2019's running-header repeats),
every row and consumers group by ``path_key`` to treat those repeats as so ``cpt_section`` keeps every row and consumers group by ``path_key``
one heading. to treat those repeats as one heading. ``cpt_code_alt`` holds the
losing entries when a code prints more than once with a real descriptor
both times (C9 — the parser picks one canonical ``pfs.cpt_code`` row by
structural score and keeps the rest here with a ``reason``, rather than
either dropping or duplicating them).
Writers are delete-then-insert per code (events, elements) or per Writers are delete-then-insert per code (events, elements) or per
``edition_year`` (cpt_*), or full replace (families), so a re-run is ``edition_year`` (cpt_*), or full replace (families), so a re-run is
@@ -157,6 +161,15 @@ class CptListRow:
code: str code: str
@dataclass(frozen=True)
class CptCodeAltRow:
edition_year: int
item_key: str
code: str
sec_id: str
reason: str # "guidelines-reprint" | "lower-score" (C9)
_DDL = """ _DDL = """
CREATE SCHEMA IF NOT EXISTS pfs; CREATE SCHEMA IF NOT EXISTS pfs;
CREATE TABLE IF NOT EXISTS pfs.code_element ( CREATE TABLE IF NOT EXISTS pfs.code_element (
@@ -190,6 +203,8 @@ CREATE TABLE IF NOT EXISTS pfs.cpt_crosswalk (
year_deleted VARCHAR, citations VARCHAR); year_deleted VARCHAR, citations VARCHAR);
CREATE TABLE IF NOT EXISTS pfs.cpt_list ( CREATE TABLE IF NOT EXISTS pfs.cpt_list (
edition_year INTEGER, item_key VARCHAR, appendix VARCHAR, code VARCHAR); edition_year INTEGER, item_key VARCHAR, appendix VARCHAR, code VARCHAR);
CREATE TABLE IF NOT EXISTS pfs.cpt_code_alt (
edition_year INTEGER, item_key VARCHAR, code VARCHAR, sec_id VARCHAR, reason VARCHAR);
""" """
@@ -288,6 +303,7 @@ _CPT_TABLES = (
"pfs.cpt_reference", "pfs.cpt_reference",
"pfs.cpt_crosswalk", "pfs.cpt_crosswalk",
"pfs.cpt_list", "pfs.cpt_list",
"pfs.cpt_code_alt",
) )
@@ -296,7 +312,10 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
``pfs.cpt_*``, keyed by ``edition.year``. Returns rows written per ``pfs.cpt_*``, keyed by ``edition.year``. Returns rows written per
table. C8: asserts one ``pfs.cpt_code`` row per (edition_year, code) table. C8: asserts one ``pfs.cpt_code`` row per (edition_year, code)
before inserting — a duplicate means a resequenced placeholder row before inserting — a duplicate means a resequenced placeholder row
slipped past the parser's own filter (``pfs.cpt_epub``).""" slipped past the parser's own filter (``pfs.cpt_epub``). C9's
guideline-reprint/lower-score losers are resolved by the parser
before this ever runs (``edition.alternates``, written to
``pfs.cpt_code_alt``), so that assertion should now always pass."""
ensure_tables(con) ensure_tables(con)
year = edition.year year = edition.year
@@ -388,6 +407,16 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
) )
for e in edition.lists for e in edition.lists
] ]
alt_rows = [
CptCodeAltRow(
edition_year=year,
item_key=item_key,
code=a.code,
sec_id=a.sec_id,
reason=a.reason,
)
for a in edition.alternates
]
for table in _CPT_TABLES: for table in _CPT_TABLES:
con.execute(f"DELETE FROM {table} WHERE edition_year = ?", [year]) con.execute(f"DELETE FROM {table} WHERE edition_year = ?", [year])
@@ -399,6 +428,7 @@ def write_cpt_edition(con: Any, edition: "CptEdition", item_key: str) -> dict[st
"references": _insert(con, "pfs.cpt_reference", reference_rows), "references": _insert(con, "pfs.cpt_reference", reference_rows),
"crosswalks": _insert(con, "pfs.cpt_crosswalk", crosswalk_rows), "crosswalks": _insert(con, "pfs.cpt_crosswalk", crosswalk_rows),
"lists": _insert(con, "pfs.cpt_list", list_rows), "lists": _insert(con, "pfs.cpt_list", list_rows),
"alternates": _insert(con, "pfs.cpt_code_alt", alt_rows),
} }

View File

@@ -35,6 +35,7 @@ from html.parser import HTMLParser
from pathlib import Path from pathlib import Path
from pfs.cpt_model import ( from pfs.cpt_model import (
CptAlternate,
CptCode, CptCode,
CptCrosswalk, CptCrosswalk,
CptEdition, CptEdition,
@@ -676,6 +677,101 @@ def _chapter_groups(names: list[str]) -> list[list[str]]:
return [sorted(grouped[key]) for key in order] return [sorted(grouped[key]) for key in order]
def _has_range(sec: CptSection) -> bool:
return bool(sec.code_lo and sec.code_hi)
def _in_range(code: str, lo: str, hi: str) -> bool:
"""5-char string comparison (the brief's own convention) — CPT
codes are always 5 characters, and this reads Category II/III
(``NNNNF``/``NNNNT``) ranges the same way as Category I ranges since
the letter suffix is constant within a range."""
if not lo or not hi:
return False
return lo <= code <= hi
def _is_guideline_path(path: tuple[str, ...]) -> bool:
return any(p.strip().lower().endswith("guidelines") for p in path)
def _resolve_alternates(
sections: list[CptSection],
codes: list[CptCode],
references: list[CptReference],
) -> tuple[list[CptCode], list[CptAlternate]]:
"""C9: some codes print twice with a full, real-looking descriptor
both times — a guideline "Unlisted Service or Procedure" summary
table, a "Qualifying Circumstances for Anesthesia" cross-reference
reprint, etc. — structurally distinct from C8's placeholder rows
(already dropped upstream, never reaching here: both entries seen
here carry real text). For each code with more than one ``CptCode``
entry, score every entry and keep the highest (ties: first in
document order) as canonical; the rest become ``CptAlternate`` rows
instead of silently vanishing or duplicating. A code with only one
entry — even one that lives entirely inside a guideline section —
is untouched and always canonical."""
sections_by_id = {s.sec_id: s for s in sections}
sections_by_path: dict[tuple[str, ...], list[CptSection]] = {}
for s in sections:
sections_by_path.setdefault(s.path, []).append(s)
codes_with_references = {r.code for r in references}
def resolved_range_section(sec: CptSection) -> CptSection | None:
if _has_range(sec):
return sec
for k in range(len(sec.path) - 1, 0, -1):
for cand in sections_by_path.get(sec.path[:k], ()):
if _has_range(cand):
return cand
return None
def score(entry: CptCode) -> int:
s = 0
sec = sections_by_id.get(entry.sec_id)
if sec is not None:
resolved = resolved_range_section(sec)
if resolved is not None:
s += 1
if _in_range(entry.code, resolved.code_lo, resolved.code_hi):
s += 2
if _is_guideline_path(sec.path):
s -= 3
if entry.elements or entry.code in codes_with_references:
s += 1
return s
by_code_idx: dict[str, list[int]] = {}
for i, c in enumerate(codes):
by_code_idx.setdefault(c.code, []).append(i)
drop: set[int] = set()
alternates: list[CptAlternate] = []
for code, idxs in by_code_idx.items():
if len(idxs) < 2:
continue
scored = [(score(codes[i]), i) for i in idxs]
best_score = max(s for s, _ in scored)
best_i = next(i for s, i in scored if s == best_score)
for _s, i in scored:
if i == best_i:
continue
drop.add(i)
entry = codes[i]
sec = sections_by_id.get(entry.sec_id)
reason = (
"guidelines-reprint"
if sec is not None and _is_guideline_path(sec.path)
else "lower-score"
)
alternates.append(
CptAlternate(code=code, sec_id=entry.sec_id, reason=reason)
)
canonical = [c for i, c in enumerate(codes) if i not in drop]
return canonical, alternates
def parse_epub(path: Path, *, year: int | None = None) -> CptEdition: def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
"""Parse one CPT Professional edition EPUB into a ``CptEdition``. """Parse one CPT Professional edition EPUB into a ``CptEdition``.
@@ -744,12 +840,18 @@ def parse_epub(path: Path, *, year: int | None = None) -> CptEdition:
elif letter in _LIST_APPENDICES: elif letter in _LIST_APPENDICES:
lists.extend(_parse_appendix_list(text, letter)) lists.extend(_parse_appendix_list(text, letter))
# C9: resolve any code that printed more than once (real descriptor
# both times — never the C8 placeholder, already dropped) down to
# one canonical CptCode, keeping the rest as CptAlternate rows.
canonical_codes, alternates = _resolve_alternates(fixed_sections, codes, references)
return CptEdition( return CptEdition(
year=year, year=year,
sections=tuple(fixed_sections), sections=tuple(fixed_sections),
codes=tuple(codes), codes=tuple(canonical_codes),
instructions=tuple(instructions), instructions=tuple(instructions),
references=tuple(references), references=tuple(references),
crosswalks=tuple(crosswalks), crosswalks=tuple(crosswalks),
lists=tuple(lists), lists=tuple(lists),
alternates=tuple(alternates),
) )

View File

@@ -111,6 +111,7 @@ def ingest(
"references": len(edition.references), "references": len(edition.references),
"crosswalks": len(edition.crosswalks), "crosswalks": len(edition.crosswalks),
"lists": len(edition.lists), "lists": len(edition.lists),
"alternates": len(edition.alternates),
} }
else: else:
out[year] = write_cpt_edition(con, edition, item_key) out[year] = write_cpt_edition(con, edition, item_key)

View File

@@ -106,6 +106,22 @@ class CptListEntry:
code: str code: str
@dataclass(frozen=True)
class CptAlternate:
"""A losing duplicate ``CptCode`` entry for a code the book prints
more than once (C9) — a guideline "Unlisted Service or Procedure"
summary table's copy, a "Qualifying Circumstances for Anesthesia"
cross-reference reprint, etc. — structurally distinct from C8's
resequenced-code placeholder (which the parser drops entirely, never
reaching this stage): here *both* entries carry a real descriptor,
so the loser is kept, not discarded, as a record of what didn't
become the canonical ``CptCode`` row and why."""
code: str
sec_id: str
reason: str # "guidelines-reprint" | "lower-score"
@dataclass(frozen=True) @dataclass(frozen=True)
class CptEdition: class CptEdition:
"""Everything parsed from one CPT codebook edition.""" """Everything parsed from one CPT codebook edition."""
@@ -117,3 +133,4 @@ class CptEdition:
references: tuple[CptReference, ...] references: tuple[CptReference, ...]
crosswalks: tuple[CptCrosswalk, ...] crosswalks: tuple[CptCrosswalk, ...]
lists: tuple[CptListEntry, ...] lists: tuple[CptListEntry, ...]
alternates: tuple[CptAlternate, ...] = ()

View File

@@ -26,6 +26,7 @@ from pfs.codetables import (
write_families, write_families,
) )
from pfs.cpt_model import ( from pfs.cpt_model import (
CptAlternate,
CptCode, CptCode,
CptCrosswalk, CptCrosswalk,
CptEdition, CptEdition,
@@ -70,6 +71,7 @@ class TestDDL:
"cpt_reference", "cpt_reference",
"cpt_crosswalk", "cpt_crosswalk",
"cpt_list", "cpt_list",
"cpt_code_alt",
} <= names } <= names
@@ -201,7 +203,7 @@ class TestFamilies:
assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on" assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on"
def _tiny_edition(year=2024, codes=None): def _tiny_edition(year=2024, codes=None, alternates=()):
sections = ( sections = (
CptSection( CptSection(
sec_id="sec_1", sec_id="sec_1",
@@ -291,6 +293,7 @@ def _tiny_edition(year=2024, codes=None):
references=references, references=references,
crosswalks=crosswalks, crosswalks=crosswalks,
lists=lists, lists=lists,
alternates=alternates,
) )
@@ -304,8 +307,20 @@ class TestCptEdition:
"references": 1, "references": 1,
"crosswalks": 1, "crosswalks": 1,
"lists": 1, "lists": 1,
"alternates": 0,
} }
def test_alternates_round_trip(self, con):
alt = CptAlternate(
code="99490", sec_id="sec_guideline", reason="guidelines-reprint"
)
counts = write_cpt_edition(con, _tiny_edition(alternates=(alt,)), "GQGTPGYV")
assert counts["alternates"] == 1
row = con.execute(
"SELECT edition_year, item_key, code, sec_id, reason FROM pfs.cpt_code_alt"
).fetchone()
assert row == (2024, "GQGTPGYV", "99490", "sec_guideline", "guidelines-reprint")
def test_round_trip_reads(self, con): def test_round_trip_reads(self, con):
write_cpt_edition(con, _tiny_edition(), "GQGTPGYV") write_cpt_edition(con, _tiny_edition(), "GQGTPGYV")

View File

@@ -10,6 +10,7 @@ they assert structure and counts only, never descriptor text.
from __future__ import annotations from __future__ import annotations
import zipfile
from pathlib import Path from pathlib import Path
import pytest import pytest
@@ -336,6 +337,90 @@ class TestReferences:
assert refs[0].years == (2022,) assert refs[0].years == (2022,)
# --- C9: guideline-reprint / cross-reference duplicates ------------------
#
# Unlike C8's resequenced placeholder (a bare pointer row with no real
# descriptor, dropped entirely by the parser), C9 handles a code that
# prints *twice with a real descriptor both times* — once inside a
# guideline "convenience" table (no TOC code range of its own) and once
# under its real, ranged home section. This can only be exercised at
# the ``parse_epub`` (edition) level, since the resolution runs after
# every chapter has been parsed — so these tests build a tiny synthetic
# EPUB rather than using ``cpt_sample.xhtml``/``parse_xhtml``.
def _build_epub(path: Path, chapter_body: str) -> None:
with zipfile.ZipFile(path, "w") as zf:
zf.writestr("mimetype", "application/epub+zip")
zf.writestr("OPS/Chapter01.xhtml", f"<html><body>{chapter_body}</body></html>")
_C9_CHAPTER = """
<div class="h1-toc"><a href="body.xhtml#sec_r"><span class="green">Sample Procedures* (54321-54329)</span></a></div>
<div class="h1" id="sec_g1">Surgery Guidelines</div>
<div class="h2" id="sec_g2">Unlisted Service or Procedure</div>
<div class="noindent">The &#8220;Unlisted Procedures&#8221; and accompanying codes are as follows:</div>
<table class="table1"><tbody>
<tr>
<td class="td-w1"><div class="table-para"><b>54321</b></div></td>
<td class="td"><div class="table-para1">Unlisted procedure, sample</div></td>
</tr>
<tr>
<td class="td-w1"><div class="table-para"><b>54322</b></div></td>
<td class="td"><div class="table-para1">Unlisted procedure, only ever printed in the guidelines</div></td>
</tr>
</tbody></table>
<div class="h1" id="sec_r">Sample Procedures</div>
<div class="noindent">Guideline text for the real, ranged section.</div>
<table class="table1"><tbody>
<tr>
<td class="td-w1" id="code_54321"><div class="table-para"><b>54321</b></div></td>
<td class="td"><div class="table-para1">Unlisted procedure, sample with the following required elements:</div>
<div class="table-slist">first element;</div>
<div class="table-RT"><span class="blue"><span class="ama-en">➲</span></span><i>CPT Changes: An Insider's View</i> 2020</div></td>
</tr>
</tbody></table>
"""
@pytest.fixture(scope="module")
def c9_edition(tmp_path_factory):
path = tmp_path_factory.mktemp("c9") / "sample.epub"
_build_epub(path, _C9_CHAPTER)
return parse_epub(path, year=2024)
class TestAlternates:
def test_ranged_entry_wins_as_canonical(self, c9_edition):
matches = [c for c in c9_edition.codes if c.code == "54321"]
assert len(matches) == 1
assert matches[0].sec_id == "sec_r"
assert matches[0].elements # the real entry, not the bare guideline row
def test_guideline_entry_becomes_an_alternate(self, c9_edition):
alts = [a for a in c9_edition.alternates if a.code == "54321"]
assert len(alts) == 1
assert alts[0].sec_id == "sec_g2"
assert alts[0].reason == "guidelines-reprint"
def test_code_only_in_guidelines_is_kept_as_canonical(self, c9_edition):
matches = [c for c in c9_edition.codes if c.code == "54322"]
assert len(matches) == 1
assert matches[0].sec_id == "sec_g2"
alts = [a for a in c9_edition.alternates if a.code == "54322"]
assert alts == []
def test_c8_assertion_would_now_pass(self, c9_edition):
# write_cpt_edition's own "one row per (edition_year, code)"
# check (C8) — the whole point of C9 is that this always holds.
from collections import Counter
counts = Counter(c.code for c in c9_edition.codes)
assert all(n == 1 for n in counts.values())
# --- Real-file integration tests (skipped when the file is absent) --- # --- Real-file integration tests (skipped when the file is absent) ---
@@ -402,6 +487,24 @@ class TestReal2024:
def test_no_code_has_an_empty_sec_id(self, edition): def test_no_code_has_an_empty_sec_id(self, edition):
assert all(c.sec_id != "" for c in edition.codes) assert all(c.sec_id != "" for c in edition.codes)
def test_c9_zero_codes_with_more_than_one_canonical_row(self, edition):
# C9: every code the real book prints more than once (guideline
# "Unlisted Service or Procedure" summary tables, "Qualifying
# Circumstances for Anesthesia" cross-reference reprints, and
# section-specific variants of the same pattern) must resolve
# to exactly one canonical pfs.cpt_code row — write_cpt_edition's
# own C8 assertion depends on this.
from collections import Counter
counts = Counter(c.code for c in edition.codes)
dupes = sorted(code for code, n in counts.items() if n > 1)
assert dupes == [], f"duplicate canonical codes: {dupes}"
# Not a hard-coded expectation of the real book's content (that
# would be fragile) — just proof the mechanism actually engaged
# on real data, and a number for the report.
assert len(edition.alternates) > 100
print(f"2024 alternates: {len(edition.alternates)}")
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk") @pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
def test_real_2022_parses(): def test_real_2022_parses():
@@ -410,6 +513,16 @@ def test_real_2022_parses():
assert edition.sections assert edition.sections
@pytest.mark.skipif(not EPUB_2022.exists(), reason="real 2022 CPT EPUB not on disk")
def test_real_2022_c9_zero_duplicate_canonical_codes():
from collections import Counter
edition = parse_epub(EPUB_2022, year=2022)
counts = Counter(c.code for c in edition.codes)
dupes = sorted(code for code, n in counts.items() if n > 1)
assert dupes == [], f"duplicate canonical codes: {dupes}"
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk") @pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
def test_real_2021_parses(): def test_real_2021_parses():
edition = parse_epub(EPUB_2021, year=2021) edition = parse_epub(EPUB_2021, year=2021)
@@ -417,6 +530,16 @@ def test_real_2021_parses():
assert edition.sections assert edition.sections
@pytest.mark.skipif(not EPUB_2021.exists(), reason="real 2021 CPT EPUB not on disk")
def test_real_2021_c9_zero_duplicate_canonical_codes():
from collections import Counter
edition = parse_epub(EPUB_2021, year=2021)
counts = Counter(c.code for c in edition.codes)
dupes = sorted(code for code, n in counts.items() if n > 1)
assert dupes == [], f"duplicate canonical codes: {dupes}"
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk") @pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
def test_real_2019_parses_without_code_ids(): def test_real_2019_parses_without_code_ids():
edition = parse_epub(EPUB_2019, year=2019) edition = parse_epub(EPUB_2019, year=2019)
@@ -425,3 +548,13 @@ def test_real_2019_parses_without_code_ids():
# 2019 headings mostly lack an id in the source markup — every # 2019 headings mostly lack an id in the source markup — every
# code must still land under a non-empty (real or synthetic) sec_id. # code must still land under a non-empty (real or synthetic) sec_id.
assert all(c.sec_id != "" for c in edition.codes) assert all(c.sec_id != "" for c in edition.codes)
@pytest.mark.skipif(not EPUB_2019.exists(), reason="real 2019 CPT EPUB not on disk")
def test_real_2019_c9_zero_duplicate_canonical_codes():
from collections import Counter
edition = parse_epub(EPUB_2019, year=2019)
counts = Counter(c.code for c in edition.codes)
dupes = sorted(code for code, n in counts.items() if n > 1)
assert dupes == [], f"duplicate canonical codes: {dupes}"

View File

@@ -151,6 +151,7 @@ class TestIngestDryRun:
"references": 0, "references": 0,
"crosswalks": 0, "crosswalks": 0,
"lists": 0, "lists": 0,
"alternates": 0,
} }
assert out[2021]["codes"] == 1 assert out[2021]["codes"] == 1