feat(pfs): families from the CPT manual hierarchy — lowest multi-code heading names the family; use-with and parent edges; FamilyRow.note (refs #687)

For CPT codes, the family is now the lowest heading (by section path,
walking up) that groups >= 2 codes in the newest ingested edition —
cpt_groups() builds this as one trie-style rollup pass, not a per-code
rescan. Heading co-membership is itself a union-find edge (so a heading
with no extracted elements still becomes a family), on top of use-with
instructions (single-target only, same #687 discipline as the existing
multi-code-event guard — a live corpus use-with note can name 228
codes) and pfs.cpt_code.parent. Hand families still win when a CPT
group intersects them; FamilyRow.note carries each code's own heading
path so it survives even when the family name comes from elsewhere.
HCPCS codes keep the slice-1 derivation unchanged.

FamilyRow gains note (last field, defaulted); ensure_tables upgrades a
pre-existing pfs.code_family table in place. stack pfs families feeds
the CPT tables when present and prints a summary line.

Fixed two bugs found against the live corpus while verifying: the
walk-up heading match was unioning a code's full descendant rollup
(not just the codes that actually settle at that heading), collapsing
861 real headings into ~76 giant blobs (one was all of Surgery, 5860
codes); and the code-shape filter (CODE_RE) was silently dropping
Category II/III/PLA codes (4 digits + letter) from every family.
This commit is contained in:
kert
2026-09-09 19:34:44 -04:00
parent 969d34c41c
commit d1d4187f95
6 changed files with 881 additions and 27 deletions

View File

@@ -21,9 +21,13 @@ from typing import Any
import typer import typer
from pfs.codetables import ( from pfs.codetables import (
cpt_years,
ensure_tables, ensure_tables,
read_all_elements, read_all_elements,
read_all_events, read_all_events,
read_cpt_codes,
read_cpt_instructions,
read_cpt_sections,
write_elements, write_elements,
write_events, write_events,
write_families, write_families,
@@ -208,6 +212,26 @@ def lineage_cmd(
con.close() con.close()
def _cpt_inputs(con: Any) -> tuple[list[Any], list[Any], list[Any]]:
"""The newest CPT edition's rows for ``derive_families`` — ``()``
empty on any replica shape that doesn't have them yet (I4: a plain
read of an old replica, pre pfs.cpt_*, is not a bug)."""
try:
years = cpt_years(con)
except Exception as exc:
if _missing_table(exc):
return [], [], []
raise
if not years:
return [], [], []
newest = max(years)
return (
read_cpt_codes(con, newest),
read_cpt_sections(con, newest),
read_cpt_instructions(con, newest),
)
def _derive_rows(con: Any) -> list[Any]: def _derive_rows(con: Any) -> list[Any]:
codes = [ codes = [
r[0] r[0]
@@ -228,15 +252,36 @@ def _derive_rows(con: Any) -> list[Any]:
"SELECT hcpcs, arg_max(description, year) FROM pfs.rvu WHERE mod IS NULL OR mod = '' GROUP BY hcpcs" "SELECT hcpcs, arg_max(description, year) FROM pfs.rvu WHERE mod IS NULL OR mod = '' GROUP BY hcpcs"
).fetchall() ).fetchall()
} }
return derive_families(elements_by_code, events_by_code, descriptions) cpt_codes, cpt_sections, cpt_instructions = _cpt_inputs(con)
return derive_families(
elements_by_code,
events_by_code,
descriptions,
cpt_codes=cpt_codes,
cpt_sections=cpt_sections,
cpt_instructions=cpt_instructions,
)
def _print_families(rows: list[Any]) -> None: def _print_families(rows: list[Any]) -> None:
by_key: dict[str, list[str]] = {} by_key: dict[str, list[Any]] = {}
for r in rows: for r in rows:
by_key.setdefault(r.key, []).append(f"{r.code}({r.role})") by_key.setdefault(r.key, []).append(r)
for key, members in sorted(by_key.items()): for key, members in sorted(by_key.items()):
typer.echo(f"{key}: {' '.join(members)}") typer.echo(f"{key}: {' '.join(f'{r.code}({r.role})' for r in members)}")
total = len(by_key)
multi = sum(1 for members in by_key.values() if len(members) >= 2)
hand = sum(1 for key in by_key if key in HAND_FAMILIES)
cpt_named = sum(
1
for key, members in by_key.items()
if key not in HAND_FAMILIES and any(r.note for r in members)
)
other = total - hand - cpt_named
typer.echo(
f"families: total {total}, multi-code {multi}, cpt-named {cpt_named}, "
f"hand {hand}, other {other}"
)
@app.command() @app.command()

View File

@@ -84,6 +84,14 @@ class FamilyRow:
until: int | None until: int | None
item_key: str item_key: str
p_id: int p_id: int
#: The code's own CPT heading path (``" > ".join(path)``, C7) — the
#: heading that actually classified this code, which may differ from
#: whichever heading named the family (a hand family or a slice-1
#: edge can pull codes from more than one heading into one family).
#: ``""`` for a code with no CPT heading (HCPCS, or a rare code alone
#: under every ancestor heading). Last field, defaulted, so callers
#: that build a ``FamilyRow`` without it keep working (Ruling C1).
note: str = ""
@dataclass(frozen=True) @dataclass(frozen=True)
@@ -183,7 +191,7 @@ CREATE TABLE IF NOT EXISTS pfs.code_event (
item_key VARCHAR, p_id INTEGER, page INTEGER, source VARCHAR, anchored BOOLEAN, note VARCHAR); item_key VARCHAR, p_id INTEGER, page INTEGER, source VARCHAR, anchored BOOLEAN, note VARCHAR);
CREATE TABLE IF NOT EXISTS pfs.code_family ( CREATE TABLE IF NOT EXISTS pfs.code_family (
key VARCHAR, name VARCHAR, code VARCHAR, role VARCHAR, since INTEGER, until INTEGER, key VARCHAR, name VARCHAR, code VARCHAR, role VARCHAR, since INTEGER, until INTEGER,
item_key VARCHAR, p_id INTEGER); item_key VARCHAR, p_id INTEGER, note VARCHAR);
CREATE TABLE IF NOT EXISTS pfs.cpt_section ( CREATE TABLE IF NOT EXISTS pfs.cpt_section (
edition_year INTEGER, item_key VARCHAR, sec_id VARCHAR, level INTEGER, title VARCHAR, edition_year INTEGER, item_key VARCHAR, sec_id VARCHAR, level INTEGER, title VARCHAR,
path VARCHAR[], path_key VARCHAR, code_lo VARCHAR, code_hi VARCHAR, guideline VARCHAR); path VARCHAR[], path_key VARCHAR, code_lo VARCHAR, code_hi VARCHAR, guideline VARCHAR);
@@ -212,6 +220,10 @@ def ensure_tables(con: Any) -> None:
for stmt in _DDL.strip().split(";"): for stmt in _DDL.strip().split(";"):
if stmt.strip(): if stmt.strip():
con.execute(stmt) con.execute(stmt)
# Ruling C1: a pre-existing pfs.code_family table (created before
# `note` existed) needs the column added — CREATE TABLE IF NOT EXISTS
# above is a no-op once the table already exists.
con.execute("ALTER TABLE pfs.code_family ADD COLUMN IF NOT EXISTS note VARCHAR")
def _insert(con: Any, table: str, rows: Sequence[Any]) -> int: def _insert(con: Any, table: str, rows: Sequence[Any]) -> int:

View File

@@ -234,14 +234,128 @@ def _unique_key(key: str, rep: str, used_keys: set[str]) -> str:
return key return key
_SLUG_RE = re.compile(r"[^A-Za-z0-9]+")
def _slug(text: str) -> str:
"""UPPER-KEBAB slug of a CPT heading title (``Chronic Care Management
Services`` -> ``CHRONIC-CARE-MANAGEMENT-SERVICES``)."""
return _SLUG_RE.sub("-", text.strip()).strip("-").upper()
def cpt_groups(
cpt_codes: Sequence[Any], cpt_sections: Sequence[Any]
) -> dict[str, tuple[str, str, str, tuple[str, ...]]]:
"""For every code in *cpt_codes*, the lowest CPT heading (by path,
walking up from the code's own section) that groups >= 2 codes.
Ruling C7: several ``sec_id``s can share a ``path_key`` (2019's
running-header repeats), so grouping is by the heading's own path —
not ``sec_id`` — from the start: codes are bucketed by their
section's ``path`` tuple, and two sections with the same path pool
their codes into one heading automatically.
Returns ``code -> (key, title, path_key, group_codes)``. A code that
is alone under its own heading *and* every ancestor heading (rare) is
omitted — it keeps its own stem-derived/singleton family instead.
``key`` is the UPPER-KEBAB slug of the heading title, disambiguated
on a collision between distinct headings sharing a title by
appending ``@`` + the parent heading's title slug (deterministic:
headings are processed in sorted path order; the first occupant of a
slug keeps the bare key).
Built as one pass over every code's path prefixes (a trie-style
rollup), not a per-code rescan of all codes — 12k codes x <= 5 path
levels is a single ~60k-entry pass.
"""
path_by_sec: dict[str, tuple[str, ...]] = {
s.sec_id: tuple(s.path) for s in cpt_sections
}
code_path: dict[str, tuple[str, ...]] = {}
for c in cpt_codes:
path = path_by_sec.get(c.sec_id)
if path:
code_path[c.code] = path
prefix_codes: dict[tuple[str, ...], list[str]] = {}
for code, path in code_path.items():
for i in range(len(path), 0, -1):
prefix_codes.setdefault(path[:i], []).append(code)
# The *qualifying test* at each level is the full rollup (does this
# ancestor have >= 2 codes anywhere beneath it — that's what "walk up
# to the parent" means); but the *membership* of the winning group
# must be only the codes that actually settle there — codes that
# resolve to a deeper, more specific heading (e.g. "Fine Needle
# Aspiration (FNA) Biopsy" under "Surgery") are not members of every
# ancestor they happen to roll up through, or one code walking up to
# a broad top-level chapter (because its own leaf was a singleton)
# would drag the chapter's thousands of already-classified codes
# along with it.
win_path_for: dict[str, tuple[str, ...]] = {}
for code, path in code_path.items():
for i in range(len(path), 0, -1):
prefix = path[:i]
if len(prefix_codes[prefix]) >= 2:
win_path_for[code] = prefix
break
settled: dict[tuple[str, ...], list[str]] = {}
for code, path in win_path_for.items():
settled.setdefault(path, []).append(code)
used_keys: set[str] = set()
key_for_path: dict[tuple[str, ...], str] = {}
for path in sorted(settled):
title = path[-1]
base = _slug(title)
if base not in used_keys:
key = base
else:
parent = path[-2] if len(path) >= 2 else ""
key = f"{base}@{_slug(parent)}" if parent else base
n = 2
while key in used_keys:
key = f"{base}@{_slug(parent)}-{n}" if parent else f"{base}-{n}"
n += 1
used_keys.add(key)
key_for_path[path] = key
out: dict[str, tuple[str, str, str, tuple[str, ...]]] = {}
for path, members in settled.items():
codes = tuple(sorted(members))
key, title = key_for_path[path], path[-1]
note = " > ".join(path)
for code in members:
out[code] = (key, title, note, codes)
return out
def _group_rows( def _group_rows(
key: str, key: str,
name: str, name: str,
members: Sequence[str], members: Sequence[str],
elements: Mapping[str, Sequence[Any]], elements: Mapping[str, Sequence[Any]],
events: Mapping[str, Sequence[Any]], events: Mapping[str, Sequence[Any]],
groups_by_code: Mapping[str, tuple[str, str, str, tuple[str, ...]]],
cpt_code_rows: Mapping[str, Any],
cpt_since: Mapping[str, int],
cpt_last_year: Mapping[str, int],
newest_cpt_year: int | None,
) -> list[Any]: ) -> list[Any]:
"""``FamilyRow`` per code in *members*, labeled ``key``/``name``.""" """``FamilyRow`` per code in *members*, labeled ``key``/``name``.
``note`` is the code's own CPT heading path (``groups_by_code``), not
the family's — a hand family or a slice-1 edge can pull codes from
more than one CPT heading into one component, and the heading that
actually classified *this* code is more informative than whichever
heading happened to win the family's name.
``since``/``until`` come from ``pfs.cpt_code`` (the book's own record
of a code's active years) when the code is a CPT code; otherwise they
fall back to the FR/RVU-derived lineage events, unchanged from
slice-1.
"""
from pfs.codetables import FamilyRow from pfs.codetables import FamilyRow
out = [] out = []
@@ -249,18 +363,32 @@ def _group_rows(
els = elements.get(c, ()) els = elements.get(c, ())
evs = events.get(c, ()) evs = events.get(c, ())
kinds = {ev.kind for ev in evs} kinds = {ev.kind for ev in evs}
cpt_row = cpt_code_rows.get(c)
if "replaced_by" in kinds: if "replaced_by" in kinds:
role = "predecessor" role = "predecessor"
elif any(e.type == "relation" and e.value == "addon-of" for e in els): elif (cpt_row is not None and cpt_row.addon) or any(
e.type == "relation" and e.value == "addon-of" for e in els
):
role = "add-on" role = "add-on"
elif "replaces" in kinds: elif "replaces" in kinds:
role = "successor" role = "successor"
else: else:
role = "base" role = "base"
if c in cpt_since:
since = cpt_since[c]
until = (
None
if newest_cpt_year is not None and cpt_last_year[c] >= newest_cpt_year
else cpt_last_year[c] + 1
)
else:
since = next((ev.year for ev in evs if ev.kind == "appeared"), None) since = next((ev.year for ev in evs if ev.kind == "appeared"), None)
until = next((ev.year for ev in evs if ev.kind == "disappeared"), None) until = next((ev.year for ev in evs if ev.kind == "disappeared"), None)
anchor = next(((ev.item_key, ev.p_id) for ev in evs if ev.item_key), ("", 0)) anchor = next(((ev.item_key, ev.p_id) for ev in evs if ev.item_key), ("", 0))
out.append(FamilyRow(key, name, c, role, since, until, anchor[0], anchor[1])) note = groups_by_code[c][2] if c in groups_by_code else ""
out.append(
FamilyRow(key, name, c, role, since, until, anchor[0], anchor[1], note)
)
return out return out
@@ -291,14 +419,21 @@ def derive_families(
elements: Mapping[str, Sequence[Any]], elements: Mapping[str, Sequence[Any]],
events: Mapping[str, Sequence[Any]], events: Mapping[str, Sequence[Any]],
descriptions: Mapping[str, str], descriptions: Mapping[str, str],
*,
cpt_codes: Sequence[Any] = (),
cpt_sections: Sequence[Any] = (),
cpt_instructions: Sequence[Any] = (),
) -> list[Any]: ) -> list[Any]:
"""Connected components over codes linked by relation elements, """Connected components over codes linked by the CPT manual's own
hierarchy (``cpt_codes``/``cpt_sections`` — the primary organizing
principle, see ``cpt_groups``), ``use-with`` instructions and
``parent`` (the semicolon-inheritance rule), relation elements,
lineage events, or a shared service stem + activity set. lineage events, or a shared service stem + activity set.
Role precedence when a code qualifies for more than one: Role precedence when a code qualifies for more than one:
predecessor (has a ``replaced_by`` event) > add-on (has an predecessor (has a ``replaced_by`` event) > add-on (has an
``addon-of`` relation element) > successor (has a ``replaces`` ``addon-of`` relation element or ``pfs.cpt_code.addon``) > successor
event) > base. (has a ``replaces`` event) > base.
A component that reaches into two or more hand families (#687: the A component that reaches into two or more hand families (#687: the
real corpus's blanket multi-code lineage sentences used to merge ACP, real corpus's blanket multi-code lineage sentences used to merge ACP,
@@ -308,14 +443,39 @@ def derive_families(
order, and every code the BFS reaches joins that hand family. Codes no order, and every code the BFS reaches joins that hand family. Codes no
BFS reaches keep their own stem-keyed family, same as an ungrouped BFS reaches keep their own stem-keyed family, same as an ungrouped
component. component.
HCPCS codes (not in ``cpt_codes``) keep the slice-1 derivation
unchanged — they only join a CPT-headed family through one of the
slice-1 edges (e.g. a ``replaced_by`` event into a CPT code).
""" """
cpt_code_rows: dict[str, Any] = {c.code: c for c in cpt_codes}
groups_by_code = cpt_groups(cpt_codes, cpt_sections)
cpt_since: dict[str, int] = {}
cpt_last_year: dict[str, int] = {}
for c in cpt_codes:
cpt_since[c.code] = min(cpt_since.get(c.code, c.edition_year), c.edition_year)
cpt_last_year[c.code] = max(
cpt_last_year.get(c.code, c.edition_year), c.edition_year
)
newest_cpt_year = max(cpt_last_year.values(), default=None)
# Reject members that aren't code-shaped before deriving anything — # Reject members that aren't code-shaped before deriving anything —
# a corpus parsing artifact (e.g. a bare '\x1a') must never become a # a corpus parsing artifact (e.g. a bare '\x1a') must never become a
# family of its own. # family of its own. `cpt_code_rows` is exempt from the CODE_RE
# shape check: those codes come from the structured `pfs.cpt_code`
# table, not a free-text scan, and CODE_RE (5 digits or a letter +
# 4 digits) doesn't cover Category II/III/PLA's 4-digits-then-letter
# shape (0001U, 0001T, 0001F) — excluding them would silently drop
# ~1,400 real 2024-edition codes from every family (#687-style regression
# found live: those categories vanished from the derivation entirely).
codes = sorted( codes = sorted(
set(cpt_code_rows)
| {
c c
for c in (set(elements) | set(events) | set(descriptions)) for c in (set(elements) | set(events) | set(descriptions))
if CODE_RE.fullmatch(c) if CODE_RE.fullmatch(c)
}
) )
parent = {c: c for c in codes} parent = {c: c for c in codes}
adj: dict[str, set[str]] = {c: set() for c in codes} adj: dict[str, set[str]] = {c: set() for c in codes}
@@ -332,6 +492,36 @@ def derive_families(
adj[b].add(a) adj[b].add(a)
parent[find(a)] = find(b) parent[find(a)] = find(b)
# CPT heading co-membership: every code the lowest qualifying heading
# groups is a family-membership edge by itself — this is what lets a
# heading with no extracted elements at all (no FR paragraph reprints
# it) still become a family. One heading (by path) chained per pass,
# not re-walked per member.
seen_paths: set[str] = set()
for info in groups_by_code.values():
path_key = info[2]
if path_key in seen_paths:
continue
seen_paths.add(path_key)
members = [c for c in info[3] if c in parent]
for a, b in zip(members, members[1:]):
union(a, b)
# Same #687 discipline as _LINK_EVENTS below: a `use-with` note naming
# one code is an unambiguous add-on -> primary edge; a note naming a
# whole range ("Use 0690T in conjunction with 76536, 76604, ...
# [21 codes]") is the same kind of blanket multi-code sentence that
# bridges unrelated families and is evidence, not a merge signal
# (live corpus: 637 use-with instructions, 325 name >1 target, one
# names 228 — narrowing this is what keeps the CPT heading grouping
# from cascading into a single book-spanning component).
for instr in cpt_instructions:
if instr.kind == "use-with" and len(instr.targets) == 1:
union(instr.code, instr.targets[0].upper())
for code, row in cpt_code_rows.items():
if row.parent:
union(code, row.parent.upper())
for code, els in elements.items(): for code, els in elements.items():
for e in els: for e in els:
if e.type == "relation" and e.value in _LINK_RELATIONS and e.detail: if e.type == "relation" and e.value in _LINK_RELATIONS and e.detail:
@@ -383,8 +573,49 @@ def derive_families(
) )
return until - since return until - since
def _rows(key: str, name: str, members: Sequence[str]) -> list[Any]:
return _group_rows(
key,
name,
members,
elements,
events,
groups_by_code,
cpt_code_rows,
cpt_since,
cpt_last_year,
newest_cpt_year,
)
def _cpt_name(members: Sequence[str]) -> tuple[str, str] | None:
"""The CPT-derived ``(key, name)`` for *members*, by majority
heading — a component can span more than one CPT heading once
slice-1/``use-with``/``parent`` edges are layered on (the same
bridging *derive_families* already guards for hand families), so
the heading that classifies the most of this component's
CPT-covered codes names the whole group; ties break on the
heading's path (deterministic). ``None`` when no member has a
CPT heading at all (an HCPCS-only or pre-CPT-book component)."""
counts: dict[str, int] = {}
info_by_path: dict[str, tuple[str, str, str, tuple[str, ...]]] = {}
for c in members:
info = groups_by_code.get(c)
if info is None:
continue
path_key = info[2]
counts[path_key] = counts.get(path_key, 0) + 1
info_by_path[path_key] = info
if not counts:
return None
win = min(counts, key=lambda pk: (-counts[pk], pk))
key, title, _, _ = info_by_path[win]
return key, title
rows: list[Any] = [] rows: list[Any] = []
used_keys: set[str] = set() # Seed with every CPT-derived key so a later stem-derived key
# (_name_group/_unique_key, for a component with no CPT heading) can
# never collide with one cpt_groups already handed out.
used_keys: set[str] = {info[0] for info in groups_by_code.values()}
for members in groups.values(): for members in groups.values():
member_set = set(members) member_set = set(members)
hands = [k for k in HAND_FAMILIES if set(HAND_FAMILIES[k].codes) & member_set] hands = [k for k in HAND_FAMILIES if set(HAND_FAMILIES[k].codes) & member_set]
@@ -413,22 +644,28 @@ def derive_families(
for hk in hands: for hk in hands:
grp = [c for c, v in assigned.items() if v == hk] grp = [c for c, v in assigned.items() if v == hk]
if grp: if grp:
rows.extend( rows.extend(_rows(hk, HAND_FAMILIES[hk].name, grp))
_group_rows(hk, HAND_FAMILIES[hk].name, grp, elements, events)
)
unreached = [c for c in members if c not in assigned] unreached = [c for c in members if c not in assigned]
for sub in _connected(unreached, adj): for sub in _connected(unreached, adj):
cpt_name = _cpt_name(sub)
if cpt_name:
key, name = cpt_name
else:
key, name, rep = _name_group(sub, stems, descriptions, tenure) key, name, rep = _name_group(sub, stems, descriptions, tenure)
key = _unique_key(key, rep, used_keys) key = _unique_key(key, rep, used_keys)
rows.extend(_group_rows(key, name, sub, elements, events)) rows.extend(_rows(key, name, sub))
continue continue
hand = hands[0] if hands else None hand = hands[0] if hands else None
if hand: if hand:
key, name = hand, HAND_FAMILIES[hand].name key, name = hand, HAND_FAMILIES[hand].name
else:
cpt_name = _cpt_name(members)
if cpt_name:
key, name = cpt_name
else: else:
key, name, rep = _name_group(members, stems, descriptions, tenure) key, name, rep = _name_group(members, stems, descriptions, tenure)
key = _unique_key(key, rep, used_keys) key = _unique_key(key, rep, used_keys)
rows.extend(_group_rows(key, name, members, elements, events)) rows.extend(_rows(key, name, members))
return sorted(rows, key=lambda r: (r.key, r.code)) return sorted(rows, key=lambda r: (r.key, r.code))

View File

@@ -15,9 +15,11 @@ from pfs.codetables import (
read_elements, read_elements,
read_events, read_events,
read_families, read_families,
write_cpt_edition,
write_elements, write_elements,
write_events, write_events,
) )
from pfs.cpt_model import CptCode, CptEdition, CptSection
from pfs.extract import Extraction from pfs.extract import Extraction
runner = CliRunner() runner = CliRunner()
@@ -264,6 +266,90 @@ class TestFamilies:
fams = read_families(con) fams = read_families(con)
assert {r.code for r in fams if r.key == "CCM"} >= {"99490", "99439", "G2058"} assert {r.code for r in fams if r.key == "CCM"} >= {"99490", "99439", "G2058"}
assert "CCM" in res.output assert "CCM" in res.output
# No pfs.cpt_* rows seeded in this test — the summary line must
# still print, with cpt-named at 0 (I4: read-only-tolerant path
# also applies to the write path when the CPT tables are empty).
assert "families: total 2, multi-code 1, cpt-named 0, hand 1, other 1" in (
res.output
)
def _ccm_cpt_edition(self):
section = CptSection(
sec_id="sec_ccm",
level=3,
title="Chronic Care Management Services",
path=(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
code_lo="99490",
code_hi="99439",
guideline="",
)
codes = (
CptCode(
code="99490",
sec_id="sec_ccm",
descriptor="Chronic care management services, first 20 minutes.",
stem="Chronic care management services",
elements=(),
tail="first 20 minutes.",
addon=False,
resequenced=False,
new=False,
revised=False,
telemedicine=False,
parent="",
mod51_exempt=False,
audio_only=False,
fda_pending=False,
pla=False,
category="I",
),
CptCode(
code="99439",
sec_id="sec_ccm",
descriptor="each additional 20 minutes.",
stem="Chronic care management services",
elements=(),
tail="each additional 20 minutes.",
addon=True,
resequenced=False,
new=False,
revised=False,
telemedicine=False,
parent="99490",
mod51_exempt=False,
audio_only=False,
fda_pending=False,
pla=False,
category="I",
),
)
return CptEdition(
year=2024,
sections=(section,),
codes=codes,
instructions=(),
references=(),
crosswalks=(),
lists=(),
)
def test_feeds_cpt_inputs_when_present_and_prints_summary_line(self, con):
write_cpt_edition(con, self._ccm_cpt_edition(), "ITEM0001")
res = runner.invoke(app, ["pfs", "families", "--write"])
assert res.exit_code == 0, res.output
fams = read_families(con)
ccm = {r.code: r for r in fams if r.key == "CCM"}
assert {"99490", "99439"} <= set(ccm)
assert ccm["99490"].note == (
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services"
)
assert "families: total" in res.output
assert "cpt-named" in res.output
def test_write_tolerates_null_rvu_description(self, con): def test_write_tolerates_null_rvu_description(self, con):
# #687 regression: a real pfs.rvu row can carry a NULL description # #687 regression: a real pfs.rvu row can carry a NULL description

View File

@@ -202,6 +202,94 @@ class TestFamilies:
fams = read_families(con) fams = read_families(con)
assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on" assert len(fams) == 1 and fams[0].code == "99439" and fams[0].role == "add-on"
def test_note_round_trips(self, con):
# Ruling C1: `note` is the last field, defaulted, so a caller that
# omits it (as `test_full_replace` above does) still works — and
# a caller that sets it gets it back unchanged.
path_key = (
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services"
)
write_families(
con,
[
FamilyRow(
"CCM",
"Chronic Care Management",
"99490",
"base",
2015,
None,
"",
0,
path_key,
)
],
)
fams = read_families(con)
assert len(fams) == 1
assert fams[0].note == path_key
def test_note_defaults_to_empty_string(self, con):
write_families(
con,
[
FamilyRow(
"CCM", "Chronic Care Management", "99490", "base", 2015, None, "", 0
)
],
)
assert read_families(con)[0].note == ""
def test_ensure_tables_upgrades_a_pre_existing_table_without_note(self):
# A replica created before `note` existed has an 8-column
# pfs.code_family — ensure_tables must add the column in place,
# not require a drop/recreate (Ruling C1).
con = duckdb.connect(":memory:")
try:
con.execute("CREATE SCHEMA IF NOT EXISTS pfs;")
con.execute(
"CREATE TABLE pfs.code_family ("
"key VARCHAR, name VARCHAR, code VARCHAR, role VARCHAR, "
"since INTEGER, until INTEGER, item_key VARCHAR, p_id INTEGER)"
)
con.execute(
"INSERT INTO pfs.code_family VALUES "
"('CCM', 'Chronic Care Management', '99490', 'base', 2015, NULL, '', 0)"
)
ensure_tables(con)
cols = {
r[0]
for r in con.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema='pfs' AND table_name='code_family'"
).fetchall()
}
assert "note" in cols
# The pre-existing row survives the upgrade with note = NULL,
# and a fresh write still round-trips.
row = read_families(con)[0]
assert row.code == "99490" and row.note is None
write_families(
con,
[
FamilyRow(
"CCM",
"Chronic Care Management",
"99491",
"base",
2015,
None,
"",
0,
"some path key",
)
],
)
assert read_families(con)[0].note == "some path key"
finally:
con.close()
def _tiny_edition(year=2024, codes=None, alternates=()): def _tiny_edition(year=2024, codes=None, alternates=()):
sections = ( sections = (

View File

@@ -9,6 +9,9 @@ import duckdb
import pytest import pytest
from pfs.codetables import ( from pfs.codetables import (
CptCodeRow,
CptInstructionRow,
CptSectionRow,
ElementRow, ElementRow,
EventRow, EventRow,
FamilyRow, FamilyRow,
@@ -19,6 +22,7 @@ from pfs.families import (
FAMILIES, FAMILIES,
HAND_FAMILIES, HAND_FAMILIES,
Detection, Detection,
cpt_groups,
derive_families, derive_families,
detect_codes, detect_codes,
family_of, family_of,
@@ -396,6 +400,388 @@ class TestDerive:
assert elapsed < 2.0 assert elapsed < 2.0
def _cpt_section(sec_id, path, guideline=""):
path = tuple(path)
return CptSectionRow(
edition_year=2024,
item_key="ITEM0001",
sec_id=sec_id,
level=len(path),
title=path[-1],
path=list(path),
path_key=" > ".join(path),
code_lo="",
code_hi="",
guideline=guideline,
)
def _cpt_code(code, sec_id, *, year=2024, parent="", addon=False, descriptor=""):
return CptCodeRow(
edition_year=year,
item_key="ITEM0001",
code=code,
sec_id=sec_id,
category="I",
descriptor=descriptor,
stem="",
elements=[],
tail="",
parent=parent,
addon=addon,
resequenced=False,
new=False,
revised=False,
telemedicine=False,
mod51_exempt=False,
audio_only=False,
fda_pending=False,
pla=False,
)
class TestCptGroups:
def test_lowest_heading_wins_not_the_parent(self):
# Task 4 ruling: "Care Management Services" (h1) groups >= 2 codes
# too (6, rolled up), but cpt_groups must stop at the h2 leaf
# heading each code actually sits under, not walk up past it.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
),
_cpt_section(
"sec_ccx",
(
"Evaluation and Management",
"Care Management Services",
"Complex Chronic Care Management Services",
),
),
]
codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm"),
_cpt_code("99491", "sec_ccm"),
_cpt_code("99437", "sec_ccm"),
_cpt_code("99487", "sec_ccx"),
_cpt_code("99489", "sec_ccx"),
]
groups = cpt_groups(codes, sections)
assert groups["99490"][:3] == (
"CHRONIC-CARE-MANAGEMENT-SERVICES",
"Chronic Care Management Services",
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services",
)
assert set(groups["99490"][3]) == {"99490", "99439", "99491", "99437"}
assert groups["99487"][:3] == (
"COMPLEX-CHRONIC-CARE-MANAGEMENT-SERVICES",
"Complex Chronic Care Management Services",
"Evaluation and Management > Care Management Services > "
"Complex Chronic Care Management Services",
)
assert set(groups["99487"][3]) == {"99487", "99489"}
# Never the h1 "Care Management Services" rollup:
assert all("Care Management Services)" not in g[1] for g in groups.values())
assert all(g[1] != "Care Management Services" for g in groups.values())
def test_singleton_leaf_walks_up_to_the_grouping_parent(self):
sections = [
_cpt_section("sec_parent", ("Chapter X", "Section A")),
_cpt_section("sec_leaf1", ("Chapter X", "Section A", "Leaf One")),
_cpt_section("sec_leaf2", ("Chapter X", "Section A", "Leaf Two")),
]
codes = [
_cpt_code("10001", "sec_leaf1"),
_cpt_code("10002", "sec_leaf2"),
]
groups = cpt_groups(codes, sections)
assert groups["10001"][:3] == (
"SECTION-A",
"Section A",
"Chapter X > Section A",
)
assert groups["10002"][:3] == groups["10001"][:3]
assert set(groups["10001"][3]) == {"10001", "10002"}
def test_same_title_under_different_parents_disambiguated(self):
sections = [
_cpt_section(
"sec_office",
("Chapter A", "Office Visits", "New or Established Patient"),
),
_cpt_section(
"sec_home",
("Chapter B", "Home Visits", "New or Established Patient"),
),
]
codes = [
_cpt_code("20001", "sec_office"),
_cpt_code("20002", "sec_office"),
_cpt_code("20003", "sec_home"),
_cpt_code("20004", "sec_home"),
]
groups = cpt_groups(codes, sections)
# "Chapter A ..." sorts first, so it keeps the bare slug; the
# colliding "Chapter B ..." heading is disambiguated by its
# parent title (deterministic — sorted path order, first occupant
# wins the bare key).
assert groups["20001"][0] == "NEW-OR-ESTABLISHED-PATIENT"
assert groups["20003"][0] == "NEW-OR-ESTABLISHED-PATIENT@HOME-VISITS"
assert groups["20001"][0] != groups["20003"][0]
class TestDeriveCpt:
def test_ccm_hand_family_wins_merged_component_note_is_own_heading(self):
# The CCM and Complex CCM CPT headings are two separate multi-code
# groups; slice-1's stem+activity edge still joins 99487 into the
# same component as before (unchanged fixture from
# test_ccm_reproduced_with_predecessor_and_addon_roles), so the
# merged component intersects the CCM hand family and the hand
# key/name win for every code in it — but each code's `note`
# stays its own CPT heading's path, not the family's.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
),
_cpt_section(
"sec_ccx",
(
"Evaluation and Management",
"Care Management Services",
"Complex Chronic Care Management Services",
),
),
]
cpt_codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm", parent="99490", addon=True),
_cpt_code("99491", "sec_ccm"),
_cpt_code("99437", "sec_ccm", parent="99491", addon=True),
_cpt_code("99487", "sec_ccx"),
_cpt_code("99489", "sec_ccx", parent="99487", addon=True),
]
elements = {
"99490": [_el("99490", "activity", "comprehensive-care-plan")],
"99487": [_el("99487", "activity", "comprehensive-care-plan")],
"99491": [_el("99491", "activity", "comprehensive-care-plan")],
}
descriptions = {
"99490": "Chrnc care mgmt staff 1st 20",
"99439": "Chrnc care mgmt staf ea addl",
"99487": "Cplx chrnc care 1st 60 min",
"99489": "Cplx chrnc care ea addl 30",
"99491": "Chrnc care mgmt phys 1st 30",
"99437": "Chrnc care mgmt phys ea addl",
}
rows = derive_families(
elements,
{},
descriptions,
cpt_codes=cpt_codes,
cpt_sections=sections,
)
by_code = {r.code: r for r in rows}
assert {c for c in by_code} == {
"99490",
"99439",
"99487",
"99489",
"99491",
"99437",
}
assert all(r.key == "CCM" for r in by_code.values())
assert all(r.name == HAND_FAMILIES["CCM"].name for r in by_code.values())
ccm_path_key = (
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services"
)
ccx_path_key = (
"Evaluation and Management > Care Management Services > "
"Complex Chronic Care Management Services"
)
for c in ("99490", "99439", "99491", "99437"):
assert by_code[c].note == ccm_path_key
for c in ("99487", "99489"):
assert by_code[c].note == ccx_path_key
# cpt_code.addon and parent both feed the add-on role and its edge.
assert by_code["99439"].role == "add-on"
assert by_code["99489"].role == "add-on"
# Book-derived since: earliest edition_year in cpt_codes.
assert by_code["99490"].since == 2024
def test_heading_with_no_elements_still_becomes_a_family(self):
sections = [
_cpt_section(
"sec_rpm",
(
"Medicine",
"Remote Physiologic Monitoring Treatment Management Services",
),
)
]
cpt_codes = [
_cpt_code("99457", "sec_rpm", year=2024),
_cpt_code("99458", "sec_rpm", year=2024, parent="99457", addon=True),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
by_code = {r.code: r for r in rows}
assert set(by_code) == {"99457", "99458"}
assert all(
r.key == "REMOTE-PHYSIOLOGIC-MONITORING-TREATMENT-MANAGEMENT-SERVICES"
for r in by_code.values()
)
assert all(r.since == 2024 for r in by_code.values())
assert all(r.until is None for r in by_code.values())
assert by_code["99458"].role == "add-on"
def test_singleton_leaf_grouped_at_parent_heading(self):
sections = [
_cpt_section("sec_parent", ("Chapter X", "Section A")),
_cpt_section("sec_leaf1", ("Chapter X", "Section A", "Leaf One")),
_cpt_section("sec_leaf2", ("Chapter X", "Section A", "Leaf Two")),
]
cpt_codes = [
_cpt_code("30001", "sec_leaf1"),
_cpt_code("30002", "sec_leaf2"),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
keys = {r.code: r.key for r in rows}
assert keys["30001"] == keys["30002"] == "SECTION-A"
def test_same_title_headings_get_distinct_keys(self):
sections = [
_cpt_section(
"sec_office",
("Chapter A", "Office Visits", "New or Established Patient"),
),
_cpt_section(
"sec_home",
("Chapter B", "Home Visits", "New or Established Patient"),
),
]
cpt_codes = [
_cpt_code("40001", "sec_office"),
_cpt_code("40002", "sec_office"),
_cpt_code("40003", "sec_home"),
_cpt_code("40004", "sec_home"),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
keys = {r.code: r.key for r in rows}
assert keys["40001"] == keys["40002"]
assert keys["40003"] == keys["40004"]
assert keys["40001"] != keys["40003"]
def test_use_with_instruction_and_parent_join_codes_with_no_heading(self):
# No cpt_sections at all — these codes have no CPT heading of
# their own, so the only thing that can join them is the
# use-with instruction (add-on -> primary) and the parent field
# (semicolon-rule child -> parent).
cpt_codes = [
_cpt_code("50000", "sec_missing", descriptor="Alpha widget"),
_cpt_code("50001", "sec_missing", descriptor="Beta gadget"),
_cpt_code(
"50002", "sec_missing", parent="50000", descriptor="Gamma sprocket"
),
]
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="50001",
kind="use-with",
text="(Use 50001 in conjunction with 50000)",
targets=["50000"],
)
]
rows = derive_families(
{},
{},
{
"50000": "Alpha widget",
"50001": "Beta gadget",
"50002": "Gamma sprocket",
},
cpt_codes=cpt_codes,
cpt_sections=[],
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["50000"] == keys["50001"] == keys["50002"]
assert {r.note for r in rows if r.code in ("50000", "50001", "50002")} == {""}
def test_not_with_instruction_never_joins(self):
cpt_codes = [
_cpt_code("60000", "sec_missing", descriptor="Zeta thing"),
_cpt_code("60001", "sec_missing", descriptor="Eta gizmo"),
]
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="60001",
kind="not-with-time",
text="(Do not report 60001 for service time reported with 60000)",
targets=["60000"],
)
]
rows = derive_families(
{},
{},
{"60000": "Zeta thing", "60001": "Eta gizmo"},
cpt_codes=cpt_codes,
cpt_sections=[],
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["60000"] != keys["60001"]
def test_hcpcs_code_keeps_slice1_derivation(self):
# A HCPCS G-code absent from pfs.cpt_code joins a CPT family only
# through a slice-1 edge (here: a replaced_by event), never
# through cpt_groups.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
)
]
cpt_codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm", parent="99490", addon=True),
]
events = {
"G2058": [
_ev("G2058", 2020, "appeared"),
_ev("G2058", 2021, "disappeared"),
_ev("G2058", 2021, "replaced_by", to="99439"),
],
"99439": [_ev("99439", 2021, "replaces", frm="G2058")],
}
rows = derive_families(
{}, events, {}, cpt_codes=cpt_codes, cpt_sections=sections
)
by_code = {r.code: r for r in rows}
assert by_code["G2058"].key == by_code["99490"].key == "CCM"
assert by_code["G2058"].note == ""
assert by_code["G2058"].role == "predecessor"
assert by_code["G2058"].since == 2020 and by_code["G2058"].until == 2021
class TestLoadAndRefresh: class TestLoadAndRefresh:
def test_load_and_refresh_in_place(self, restore_families): def test_load_and_refresh_in_place(self, restore_families):
mod = restore_families mod = restore_families