Files
stack/tests/pfs/test_families.py
kert 589b8338ef fix(pfs): since = earliest evidence; derived families get no synonyms (refs F3 F5)
`since` used to take cpt_since unconditionally, so a code first ingested
in the 2019 CPT edition read `since=2019` even when the RVU/FR record
shows it created years earlier (99490, CY2015). Now `since =
min(cpt_since, appeared-event year)` when an appeared event exists —
cpt_since is only the earliest *ingested* edition, not proof the code
didn't exist before that.

Derived (non-hand) families used to get `(key.lower(), name.lower())`
as synonyms, turning ~1,334 one/two-word CPT heading names ("Repair",
"Excision") into chat triggers and costing detect_codes 0.50s/call at
8.5k families. Derived families now get an empty synonym tuple; hand
families are unaffected. Matching a derived family by name is #699's
job (a precompiled alternation).
2026-09-09 20:56:51 -04:00

1235 lines
46 KiB
Python

"""pfs.families — code-family registry + deterministic code detection."""
from __future__ import annotations
import logging
import time
import duckdb
import pytest
from pfs.codetables import (
CptCodeRow,
CptInstructionRow,
CptSectionRow,
ElementRow,
EventRow,
FamilyRow,
ensure_tables,
write_families,
)
from pfs.families import (
FAMILIES,
HAND_FAMILIES,
Detection,
_cpt_edges,
cpt_groups,
derive_families,
detect_codes,
family_of,
find_codes,
load_families,
refresh_from,
stem_tokens,
)
@pytest.fixture
def restore_families():
"""Snapshot ``pfs.families.FAMILIES`` and restore it in teardown, even
if the test body raises — a test that calls ``refresh_from`` must not
leave the live registry clobbered for the rest of the session."""
from pfs import families as mod
before = dict(mod.FAMILIES)
try:
yield mod
finally:
mod.FAMILIES.clear()
mod.FAMILIES.update(before)
class TestFindCodes:
def test_hcpcs_and_cpt(self):
assert find_codes("Codes G0556 and 99490 apply; see g0557.") == (
"99490",
"G0556",
"G0557",
)
def test_lower_case_codes_are_normalised(self):
assert find_codes("what is g0556?") == ("G0556",)
def test_no_false_hits_on_years_and_fr_pages(self):
# A Federal Register citation is stripped before scanning, so its page
# number (43842 — a well-formed CPT code) never reads as a code.
assert find_codes("91 FR 43842, CY2026") == ()
def test_fr_citation_alone_yields_nothing(self):
assert find_codes("What does 91 FR 43842 say?") == ()
def test_dedupes_and_sorts(self):
assert find_codes("G0558 G0556 G0558") == ("G0556", "G0558")
def test_empty(self):
assert find_codes("") == ()
class TestRegistry:
def test_apcm_family(self):
f = FAMILIES["APCM"]
assert f.codes == ("G0556", "G0557", "G0558")
assert "advanced primary care management" in f.synonyms
assert family_of("G0557") is f
assert family_of("00000") is None
def test_all_families_present(self):
assert set(FAMILIES) == {"ACP", "CCM", "PCM", "TCM", "APCM"}
class TestDetectCodes:
def test_family_name_expands_to_codes(self):
d = detect_codes("What is APCM and how is it valued?")
assert d == Detection(
codes=("G0556", "G0557", "G0558"), families=("APCM",), explicit=()
)
def test_synonym_case_insensitive_word_boundary(self):
d = detect_codes("Tell me about Advanced Primary Care Management.")
assert d.families == ("APCM",)
assert detect_codes("the apcmx code").families == ()
def test_single_member_code_expands_family(self):
d = detect_codes("How much does G0557 pay?")
assert d.explicit == ("G0557",)
assert d.families == ("APCM",)
assert d.codes == ("G0556", "G0557", "G0558")
def test_unregistered_code_stays_alone(self):
d = detect_codes("value of 99213")
assert d == Detection(codes=("99213",), families=(), explicit=("99213",))
def test_multiple_families(self):
d = detect_codes("compare CCM and TCM")
assert d.families == ("CCM", "TCM")
assert d.codes == tuple(sorted(FAMILIES["CCM"].codes + FAMILIES["TCM"].codes))
def test_nothing(self):
assert detect_codes("what did commenters say about telehealth?") == Detection(
(), (), ()
)
def _el(code, type_, value, detail=""):
return ElementRow(code, 2025, type_, value, detail, "", "", 0, 0, "fr")
def _ev(code, year, kind, frm="", to=""):
return EventRow(code, year, kind, frm, to, "", 0, 0, "fr", True, "")
class TestStemTokens:
def test_strips_time_and_actor_noise(self):
assert stem_tokens("Chrnc care mgmt staff 1st 20") == frozenset(
{"chrnc", "care", "mgmt"}
)
assert stem_tokens("Chrnc care mgmt phys ea addl") == frozenset(
{"chrnc", "care", "mgmt"}
)
assert stem_tokens("Adv prim care mgmt lvl 1") == frozenset(
{"adv", "prim", "care", "mgmt"}
)
def test_none_description_yields_empty_set(self):
# A real pfs.rvu row can carry a NULL description (corpus parsing
# artifact, e.g. a garbage hcpcs value) — stem_tokens must not
# crash on it (#687 live-run regression: AttributeError on
# `None.lower()` in derive_families).
assert stem_tokens(None) == frozenset()
class TestDerive:
def test_ccm_reproduced_with_predecessor_and_addon_roles(self):
# #687 ruling 16: `not-with` is an exclusion, not a membership
# signal, and was dropped from _LINK_RELATIONS. The `not-with`
# element below is now inert as a merge edge — 99487 joins CCM
# through the stem-Jaccard + shared "comprehensive-care-plan"
# activity rule instead (its stem overlaps 99490's >= 0.5).
elements = {
"99490": [
_el("99490", "actor", "clinical-staff-directed"),
_el("99490", "activity", "comprehensive-care-plan"),
],
"99439": [
_el("99439", "relation", "addon-of", "99490"),
_el("99439", "relation", "not-with", "99487"),
],
"99487": [_el("99487", "activity", "comprehensive-care-plan")],
"99489": [_el("99489", "relation", "addon-of", "99487")],
"99491": [
_el("99491", "actor", "physician-or-qhp-personally"),
_el("99491", "activity", "comprehensive-care-plan"),
],
"99437": [_el("99437", "relation", "addon-of", "99491")],
"G2058": [_el("G2058", "relation", "addon-of", "99490")],
"99497": [_el("99497", "activity", "advance-directive-discussion")],
"99498": [_el("99498", "relation", "addon-of", "99497")],
}
events = {
"G2058": [
_ev("G2058", 2020, "appeared"),
_ev("G2058", 2021, "disappeared"),
_ev("G2058", 2021, "replaced_by", to="99439"),
],
"99439": [
_ev("99439", 2021, "appeared"),
_ev("99439", 2021, "replaces", frm="G2058"),
],
}
descriptions = {
"99490": "Chrnc care mgmt staff 1st 20",
"99439": "Chrnc care mgmt staf ea addl",
"99487": "Cplx chrnc care 1st 60 min",
"99489": "Cplx chrnc care ea addl 30",
"99491": "Chrnc care mgmt phys 1st 30",
"99437": "Chrnc care mgmt phys ea addl",
"G2058": "Ccm add 20min",
"99497": "Advncd care plan 30 min",
"99498": "Advncd care plan addl 30 min",
}
rows = derive_families(elements, events, descriptions)
ccm = {r.code: r for r in rows if r.key == "CCM"}
assert set(ccm) == {
"99490",
"99439",
"99487",
"99489",
"99491",
"99437",
"G2058",
}
assert ccm["99439"].role == "add-on" and ccm["G2058"].role == "predecessor"
assert ccm["G2058"].since == 2020 and ccm["G2058"].until == 2021
assert ccm["99490"].role == "base"
acp = {r.code for r in rows if r.key == "ACP"}
assert acp == {"99497", "99498"}
assert all(
r.name == HAND_FAMILIES[r.key].name for r in rows if r.key in HAND_FAMILIES
)
def test_multi_code_replaced_by_does_not_merge(self):
# #687 ruling 16: a `replaced_by` event naming more than one
# to_codes is ambiguous (a blanket "these codes are replaced by
# ..." sentence spanning several families in the real corpus) and
# must not become a merge edge.
elements = {"11111": [], "22222": []}
events = {"11111": [_ev("11111", 2021, "replaced_by", to="22222 33333")]}
descriptions = {"11111": "Alpha widget", "22222": "Zeta gadget"}
rows = derive_families(elements, events, descriptions)
keys = {r.code: r.key for r in rows}
assert keys["11111"] != keys["22222"]
def test_not_with_relation_does_not_merge(self):
# #687 ruling 16: `not-with` is an explicit exclusion (cannot be
# billed together), the opposite of family membership.
elements = {
"99490": [_el("99490", "relation", "not-with", "G0556")],
"G0556": [],
}
descriptions = {
"99490": "Chrnc care mgmt staff 1st 20",
"G0556": "Adv prim care mgmt lvl 1",
}
rows = derive_families(elements, {}, descriptions)
keys = {r.code: r.key for r in rows}
assert keys["99490"] != keys["G0556"]
def test_bridging_addon_chain_splits_by_hand_family_and_warns(self, caplog):
# #687 ruling 16 guard: a component that reaches into two or more
# hand families is split by a BFS seeded from each hand family's
# own codes, run in HAND_FAMILIES order (ACP, CCM, PCM, TCM,
# APCM). Here "77777" is addon-of both a CCM seed (99490) and an
# ACP seed (99497), joining all three into one component; ACP
# comes first in HAND_FAMILIES, so its BFS reaches — and claims —
# the bridging code before CCM's BFS runs.
elements = {
"99490": [],
"99497": [],
"77777": [
_el("77777", "relation", "addon-of", "99490"),
_el("77777", "relation", "addon-of", "99497"),
],
}
with caplog.at_level(logging.WARNING, logger="pfs.families"):
rows = derive_families(elements, {}, {})
by_code = {r.code: r for r in rows}
assert by_code["99490"].key == "CCM"
assert by_code["99497"].key == "ACP"
assert by_code["77777"].key == "ACP"
assert any("spans hand families" in rec.message for rec in caplog.records), (
caplog.text
)
def test_unknown_family_gets_stem_key(self):
elements = {
"99453": [_el("99453", "activity", "device-data-review")],
"99454": [_el("99454", "activity", "device-data-review")],
}
rows = derive_families(
elements,
{},
{
"99453": "Rem mntr physiol param setup",
"99454": "Rem mntr physiol param dev supl",
},
)
keys = {r.key for r in rows}
assert len(keys) == 1 and next(iter(keys)).startswith("REM-MNTR")
def test_no_activity_elements_never_merges_on_stem_alone(self):
# Both codes have identical stem tokens (jaccard == 1.0, well past
# the 0.5 threshold) but neither has an activity element — an empty
# activity set must not compare equal-and-qualifying, or every
# activity-less code with a common description would merge with
# every other one. Word order differs so a stem-based merge would
# be visible as a shared key even though the codes never touch.
elements = {"11111": [], "22222": []}
descriptions = {"11111": "Foo bar widget", "22222": "Bar widget foo"}
rows = derive_families(elements, {}, descriptions)
keys = {r.code: r.key for r in rows}
assert keys["11111"] != keys["22222"]
def test_disjoint_stem_never_merges(self):
# 10000/10001 share enough stem tokens (and a matching activity) to
# merge; 20000's stem shares no token with either, so it must never
# even be compared, let alone merged.
elements = {
"10000": [_el("10000", "activity", "act-a")],
"10001": [_el("10001", "activity", "act-a")],
"20000": [_el("20000", "activity", "act-a")],
}
descriptions = {
"10000": "Alpha beta gamma",
"10001": "Alpha beta delta",
"20000": "Zeta eta theta",
}
rows = derive_families(elements, {}, descriptions)
keys = {r.code: r.key for r in rows}
assert keys["10000"] == keys["10001"]
assert keys["20000"] != keys["10000"]
def test_non_code_member_is_dropped(self):
# I3: a corpus parsing artifact like '\x1a' must never become a
# family of its own — reject non-code-shaped members before
# deriving anything.
elements = {
"99453": [_el("99453", "activity", "device-data-review")],
"\x1a": [_el("\x1a", "activity", "device-data-review")],
}
descriptions = {
"99453": "Rem mntr physiol param setup",
"\x1a": "corpus artifact",
}
rows = derive_families(elements, {}, descriptions)
assert {r.code for r in rows} == {"99453"}
def test_duplicate_derived_keys_get_representative_suffix(self):
# I3: 627 stem keys in the live table are shared by >= 2 unrelated
# components (e.g. "GENE" x13); a collision must not silently
# merge them under `load_families`' one-row-per-key model.
elements = {
"10001": [_el("10001", "activity", "act-a")],
"10002": [_el("10002", "activity", "act-a")],
"20001": [_el("20001", "activity", "act-b")],
"20002": [_el("20002", "activity", "act-b")],
}
descriptions = {
"10001": "Rare widget",
"10002": "Rare widget",
"20001": "Rare widget",
"20002": "Rare widget",
}
rows = derive_families(elements, {}, descriptions)
keys = {r.code: r.key for r in rows}
assert keys["10001"] == keys["10002"]
assert keys["20001"] == keys["20002"]
assert keys["10001"] != keys["20001"]
both = {keys["10001"], keys["20001"]}
assert "RARE-WIDGET" in both
other = (both - {"RARE-WIDGET"}).pop()
assert other.startswith("RARE-WIDGET-")
def test_none_description_does_not_raise_and_stands_alone(self):
# #687 live-run regression: a code whose `descriptions` mapping
# holds an explicit None (a real NULL pfs.rvu description, not a
# missing key — dict.get(c, "") only substitutes its default when
# c is absent) must not crash derive_families, and with nothing
# else linking it to another code it lands in its own family.
elements = {"99999": [_el("99999", "activity", "act-a")]}
descriptions = {"99999": None}
rows = derive_families(elements, {}, descriptions)
assert {r.code for r in rows} == {"99999"}
assert rows[0].key == "99999"
assert rows[0].name == "99999"
def test_many_distinct_stems_is_fast(self):
# ~2,000 codes whose descriptions share no stem token with any
# other code's. The token-bucketed merge must not degrade to the
# O(n^2) all-pairs scan this guards against; kept generous (< 2s)
# so CI timing noise doesn't make it flaky.
def _word(i: int) -> str:
letters = []
n = i + 1
while n:
n, r = divmod(n - 1, 26)
letters.append(chr(97 + r))
return "".join(reversed(letters))
n = 2000
elements = {}
descriptions = {}
for i in range(n):
code = f"{10000 + i}"
elements[code] = [_el(code, "activity", "act")]
descriptions[code] = f"stem{_word(i)} term{_word(i + n)}"
start = time.perf_counter()
rows = derive_families(elements, {}, descriptions)
elapsed = time.perf_counter() - start
assert len(rows) == n
assert elapsed < 2.0
def _cpt_section(sec_id, path, guideline=""):
path = tuple(path)
return CptSectionRow(
edition_year=2024,
item_key="ITEM0001",
sec_id=sec_id,
level=len(path),
title=path[-1],
path=list(path),
path_key=" > ".join(path),
code_lo="",
code_hi="",
guideline=guideline,
)
def _cpt_code(code, sec_id, *, year=2024, parent="", addon=False, descriptor=""):
return CptCodeRow(
edition_year=year,
item_key="ITEM0001",
code=code,
sec_id=sec_id,
category="I",
descriptor=descriptor,
stem="",
elements=[],
tail="",
parent=parent,
addon=addon,
resequenced=False,
new=False,
revised=False,
telemedicine=False,
mod51_exempt=False,
audio_only=False,
fda_pending=False,
pla=False,
)
class TestCptGroups:
def test_lowest_heading_wins_not_the_parent(self):
# Task 4 ruling: "Care Management Services" (h1) groups >= 2 codes
# too (6, rolled up), but cpt_groups must stop at the h2 leaf
# heading each code actually sits under, not walk up past it.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
),
_cpt_section(
"sec_ccx",
(
"Evaluation and Management",
"Care Management Services",
"Complex Chronic Care Management Services",
),
),
]
codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm"),
_cpt_code("99491", "sec_ccm"),
_cpt_code("99437", "sec_ccm"),
_cpt_code("99487", "sec_ccx"),
_cpt_code("99489", "sec_ccx"),
]
groups = cpt_groups(codes, sections)
assert groups["99490"][:3] == (
"CHRONIC-CARE-MANAGEMENT-SERVICES",
"Chronic Care Management Services",
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services",
)
assert set(groups["99490"][3]) == {"99490", "99439", "99491", "99437"}
assert groups["99487"][:3] == (
"COMPLEX-CHRONIC-CARE-MANAGEMENT-SERVICES",
"Complex Chronic Care Management Services",
"Evaluation and Management > Care Management Services > "
"Complex Chronic Care Management Services",
)
assert set(groups["99487"][3]) == {"99487", "99489"}
# Never the h1 "Care Management Services" rollup:
assert all("Care Management Services)" not in g[1] for g in groups.values())
assert all(g[1] != "Care Management Services" for g in groups.values())
def test_singleton_leaf_walks_up_to_the_grouping_parent(self):
sections = [
_cpt_section("sec_parent", ("Chapter X", "Section A")),
_cpt_section("sec_leaf1", ("Chapter X", "Section A", "Leaf One")),
_cpt_section("sec_leaf2", ("Chapter X", "Section A", "Leaf Two")),
]
codes = [
_cpt_code("10001", "sec_leaf1"),
_cpt_code("10002", "sec_leaf2"),
]
groups = cpt_groups(codes, sections)
assert groups["10001"][:3] == (
"SECTION-A",
"Section A",
"Chapter X > Section A",
)
assert groups["10002"][:3] == groups["10001"][:3]
assert set(groups["10001"][3]) == {"10001", "10002"}
def test_same_title_under_different_parents_disambiguated(self):
sections = [
_cpt_section(
"sec_office",
("Chapter A", "Office Visits", "New or Established Patient"),
),
_cpt_section(
"sec_home",
("Chapter B", "Home Visits", "New or Established Patient"),
),
]
codes = [
_cpt_code("20001", "sec_office"),
_cpt_code("20002", "sec_office"),
_cpt_code("20003", "sec_home"),
_cpt_code("20004", "sec_home"),
]
groups = cpt_groups(codes, sections)
# "Chapter A ..." sorts first, so it keeps the bare slug; the
# colliding "Chapter B ..." heading is disambiguated by its
# parent title (deterministic — sorted path order, first occupant
# wins the bare key).
assert groups["20001"][0] == "NEW-OR-ESTABLISHED-PATIENT"
assert groups["20003"][0] == "NEW-OR-ESTABLISHED-PATIENT@HOME-VISITS"
assert groups["20001"][0] != groups["20003"][0]
class TestDeriveCpt:
def test_ccm_hand_family_wins_merged_component_note_is_own_heading(self):
# The CCM and Complex CCM CPT headings are two separate multi-code
# groups; slice-1's stem+activity edge still joins 99487 into the
# same component as before (unchanged fixture from
# test_ccm_reproduced_with_predecessor_and_addon_roles), so the
# merged component intersects the CCM hand family and the hand
# key/name win for every code in it — but each code's `note`
# stays its own CPT heading's path, not the family's.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
),
_cpt_section(
"sec_ccx",
(
"Evaluation and Management",
"Care Management Services",
"Complex Chronic Care Management Services",
),
),
]
cpt_codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm", parent="99490", addon=True),
_cpt_code("99491", "sec_ccm"),
_cpt_code("99437", "sec_ccm", parent="99491", addon=True),
_cpt_code("99487", "sec_ccx"),
_cpt_code("99489", "sec_ccx", parent="99487", addon=True),
]
elements = {
"99490": [_el("99490", "activity", "comprehensive-care-plan")],
"99487": [_el("99487", "activity", "comprehensive-care-plan")],
"99491": [_el("99491", "activity", "comprehensive-care-plan")],
}
descriptions = {
"99490": "Chrnc care mgmt staff 1st 20",
"99439": "Chrnc care mgmt staf ea addl",
"99487": "Cplx chrnc care 1st 60 min",
"99489": "Cplx chrnc care ea addl 30",
"99491": "Chrnc care mgmt phys 1st 30",
"99437": "Chrnc care mgmt phys ea addl",
}
rows = derive_families(
elements,
{},
descriptions,
cpt_codes=cpt_codes,
cpt_sections=sections,
)
by_code = {r.code: r for r in rows}
assert {c for c in by_code} == {
"99490",
"99439",
"99487",
"99489",
"99491",
"99437",
}
assert all(r.key == "CCM" for r in by_code.values())
assert all(r.name == HAND_FAMILIES["CCM"].name for r in by_code.values())
ccm_path_key = (
"Evaluation and Management > Care Management Services > "
"Chronic Care Management Services"
)
ccx_path_key = (
"Evaluation and Management > Care Management Services > "
"Complex Chronic Care Management Services"
)
for c in ("99490", "99439", "99491", "99437"):
assert by_code[c].note == ccm_path_key
for c in ("99487", "99489"):
assert by_code[c].note == ccx_path_key
# cpt_code.addon and parent both feed the add-on role and its edge.
assert by_code["99439"].role == "add-on"
assert by_code["99489"].role == "add-on"
# Book-derived since: earliest edition_year in cpt_codes.
assert by_code["99490"].since == 2024
def test_heading_with_no_elements_still_becomes_a_family(self):
sections = [
_cpt_section(
"sec_rpm",
(
"Medicine",
"Remote Physiologic Monitoring Treatment Management Services",
),
)
]
cpt_codes = [
_cpt_code("99457", "sec_rpm", year=2024),
_cpt_code("99458", "sec_rpm", year=2024, parent="99457", addon=True),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
by_code = {r.code: r for r in rows}
assert set(by_code) == {"99457", "99458"}
assert all(
r.key == "REMOTE-PHYSIOLOGIC-MONITORING-TREATMENT-MANAGEMENT-SERVICES"
for r in by_code.values()
)
assert all(r.since == 2024 for r in by_code.values())
assert all(r.until is None for r in by_code.values())
assert by_code["99458"].role == "add-on"
def test_singleton_leaf_grouped_at_parent_heading(self):
sections = [
_cpt_section("sec_parent", ("Chapter X", "Section A")),
_cpt_section("sec_leaf1", ("Chapter X", "Section A", "Leaf One")),
_cpt_section("sec_leaf2", ("Chapter X", "Section A", "Leaf Two")),
]
cpt_codes = [
_cpt_code("30001", "sec_leaf1"),
_cpt_code("30002", "sec_leaf2"),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
keys = {r.code: r.key for r in rows}
assert keys["30001"] == keys["30002"] == "SECTION-A"
def test_same_title_headings_get_distinct_keys(self):
sections = [
_cpt_section(
"sec_office",
("Chapter A", "Office Visits", "New or Established Patient"),
),
_cpt_section(
"sec_home",
("Chapter B", "Home Visits", "New or Established Patient"),
),
]
cpt_codes = [
_cpt_code("40001", "sec_office"),
_cpt_code("40002", "sec_office"),
_cpt_code("40003", "sec_home"),
_cpt_code("40004", "sec_home"),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
keys = {r.code: r.key for r in rows}
assert keys["40001"] == keys["40002"]
assert keys["40003"] == keys["40004"]
assert keys["40001"] != keys["40003"]
def test_use_with_instruction_and_parent_join_codes_with_no_heading(self):
# No cpt_sections at all — these codes have no CPT heading of
# their own, so the only thing that can join them is the
# use-with instruction (add-on -> primary) and the parent field
# (semicolon-rule child -> parent).
cpt_codes = [
_cpt_code("50000", "sec_missing", descriptor="Alpha widget"),
_cpt_code("50001", "sec_missing", descriptor="Beta gadget"),
_cpt_code(
"50002", "sec_missing", parent="50000", descriptor="Gamma sprocket"
),
]
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="50001",
kind="use-with",
text="(Use 50001 in conjunction with 50000)",
targets=["50000"],
)
]
rows = derive_families(
{},
{},
{
"50000": "Alpha widget",
"50001": "Beta gadget",
"50002": "Gamma sprocket",
},
cpt_codes=cpt_codes,
cpt_sections=[],
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["50000"] == keys["50001"] == keys["50002"]
assert {r.note for r in rows if r.code in ("50000", "50001", "50002")} == {""}
def test_not_with_instruction_never_joins(self):
cpt_codes = [
_cpt_code("60000", "sec_missing", descriptor="Zeta thing"),
_cpt_code("60001", "sec_missing", descriptor="Eta gizmo"),
]
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="60001",
kind="not-with-time",
text="(Do not report 60001 for service time reported with 60000)",
targets=["60000"],
)
]
rows = derive_families(
{},
{},
{"60000": "Zeta thing", "60001": "Eta gizmo"},
cpt_codes=cpt_codes,
cpt_sections=[],
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["60000"] != keys["60001"]
def test_hcpcs_code_keeps_slice1_derivation(self):
# A HCPCS G-code absent from pfs.cpt_code joins a CPT family only
# through a slice-1 edge (here: a replaced_by event), never
# through cpt_groups.
sections = [
_cpt_section(
"sec_ccm",
(
"Evaluation and Management",
"Care Management Services",
"Chronic Care Management Services",
),
)
]
cpt_codes = [
_cpt_code("99490", "sec_ccm"),
_cpt_code("99439", "sec_ccm", parent="99490", addon=True),
]
events = {
"G2058": [
_ev("G2058", 2020, "appeared"),
_ev("G2058", 2021, "disappeared"),
_ev("G2058", 2021, "replaced_by", to="99439"),
],
"99439": [_ev("99439", 2021, "replaces", frm="G2058")],
}
rows = derive_families(
{}, events, {}, cpt_codes=cpt_codes, cpt_sections=sections
)
by_code = {r.code: r for r in rows}
assert by_code["G2058"].key == by_code["99490"].key == "CCM"
assert by_code["G2058"].note == ""
assert by_code["G2058"].role == "predecessor"
assert by_code["G2058"].since == 2020 and by_code["G2058"].until == 2021
class TestCptEdgesHelper:
def test_skips_only_edges_between_distinct_headings(self):
# #687 controller review (Ruling C11), tested against the helper
# in isolation: an edge is skipped only when BOTH endpoints
# already resolve to a heading and those headings differ.
groups_by_code = {
"70001": (
"HEADING-A",
"Heading A",
"Chapter > Heading A",
("70001", "70002"),
),
"70002": (
"HEADING-A",
"Heading A",
"Chapter > Heading A",
("70001", "70002"),
),
"70003": (
"HEADING-B",
"Heading B",
"Chapter > Heading B",
("70003", "70004"),
),
}
unioned = []
def fake_union(a, b):
unioned.append((a, b))
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="I",
code="70001",
kind="use-with",
text="",
targets=["70003"], # both headed, distinct -> skipped
),
CptInstructionRow(
edition_year=2024,
item_key="I",
code="70001",
kind="use-with",
text="",
targets=["70005"], # 70005 unheaded -> unioned
),
]
_cpt_edges(fake_union, groups_by_code, instructions, {}, {})
assert unioned == [("70001", "70005")]
class TestCptEdgeGuardC11:
def _two_headings(self):
sections = [
_cpt_section("sec_a", ("Chapter", "Heading A")),
_cpt_section("sec_b", ("Chapter", "Heading B")),
]
cpt_codes = [
_cpt_code("70001", "sec_a"),
_cpt_code("70002", "sec_a"),
_cpt_code("70003", "sec_b"),
_cpt_code("70004", "sec_b"),
]
return sections, cpt_codes
def test_use_with_between_two_headed_codes_does_not_bridge(self):
sections, cpt_codes = self._two_headings()
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="70001",
kind="use-with",
text="(Use 70001 in conjunction with 70003)",
targets=["70003"],
)
]
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["70001"] == keys["70002"]
assert keys["70003"] == keys["70004"]
assert keys["70001"] != keys["70003"]
def test_use_with_from_unheaded_code_still_joins_the_headed_family(self):
sections, cpt_codes = self._two_headings()
# 70005 sits alone under its own chapter — no sibling anywhere it
# rolls up to, so it never qualifies for a heading of its own.
sections = sections + [_cpt_section("sec_x", ("Solo Chapter", "Heading X"))]
cpt_codes = cpt_codes + [_cpt_code("70005", "sec_x")]
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="70005",
kind="use-with",
text="(Use 70005 in conjunction with 70001)",
targets=["70001"],
)
]
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_instructions=instructions,
)
keys = {r.code: r.key for r in rows}
assert keys["70005"] == keys["70001"] == keys["70002"]
def test_parent_between_two_headed_codes_does_not_bridge(self):
sections, cpt_codes = self._two_headings()
# 70003's `parent` names a code in a DIFFERENT heading — the
# guard must hold even though a real semicolon-rule parent is
# always within the same heading in practice.
cpt_codes = [
c if c.code != "70003" else _cpt_code("70003", "sec_b", parent="70001")
for c in cpt_codes
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
keys = {r.code: r.key for r in rows}
assert keys["70003"] == keys["70004"]
assert keys["70003"] != keys["70001"]
def test_addon_of_between_two_headed_codes_does_not_bridge(self):
sections, cpt_codes = self._two_headings()
elements = {"70003": [_el("70003", "relation", "addon-of", "70001")]}
rows = derive_families(
elements, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections
)
keys = {r.code: r.key for r in rows}
assert keys["70003"] == keys["70004"]
assert keys["70003"] != keys["70001"]
def test_addon_of_from_unheaded_code_still_joins(self):
sections, cpt_codes = self._two_headings()
elements = {"77000": [_el("77000", "relation", "addon-of", "70001")]}
rows = derive_families(
elements, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections
)
keys = {r.code: r.key for r in rows}
assert keys["77000"] == keys["70001"]
def test_note_value_is_unaffected_by_the_guard(self):
# The guard only skips a union edge; it must not touch a code's
# own note (still its own heading's path_key).
sections, cpt_codes = self._two_headings()
instructions = [
CptInstructionRow(
edition_year=2024,
item_key="ITEM0001",
code="70001",
kind="use-with",
text="(Use 70001 in conjunction with 70003)",
targets=["70003"],
)
]
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_instructions=instructions,
)
by_code = {r.code: r for r in rows}
assert by_code["70001"].note == "Chapter > Heading A"
assert by_code["70003"].note == "Chapter > Heading B"
class TestCptPresenceC12:
def test_code_absent_from_newest_edition_gets_since_and_until_from_all_editions(
self,
):
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [
_cpt_code("90001", "sec_a", year=2024),
_cpt_code("90002", "sec_a", year=2024),
]
cpt_presence = {
"90001": (2019, 2021, 2022, 2024),
"90002": (2019, 2021, 2022, 2024),
"80001": (2019, 2020, 2021, 2022), # gone by 2024, not in cpt_codes
}
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_presence=cpt_presence,
)
by_code = {r.code: r for r in rows}
assert "80001" in by_code
assert by_code["80001"].since == 2019
assert by_code["80001"].until == 2023
def test_code_new_in_newest_edition_has_no_until(self):
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [
_cpt_code("90001", "sec_a", year=2024),
_cpt_code("90003", "sec_a", year=2024),
]
cpt_presence = {
"90001": (2019, 2021, 2022, 2024),
"90003": (2024,),
}
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_presence=cpt_presence,
)
by_code = {r.code: r for r in rows}
assert by_code["90001"].since == 2019
assert by_code["90003"].since == 2024
assert by_code["90003"].until is None
def test_without_cpt_presence_falls_back_to_cpt_codes_alone(self):
# Pre-C12 behavior preserved when the caller doesn't have a
# multi-edition presence map to give.
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [
_cpt_code("90001", "sec_a", year=2024),
_cpt_code("90002", "sec_a", year=2024),
]
rows = derive_families({}, {}, {}, cpt_codes=cpt_codes, cpt_sections=sections)
by_code = {r.code: r for r in rows}
assert by_code["90001"].since == 2024
assert by_code["90001"].until is None
class TestSinceEarliestEvidenceC13:
"""F5 / Ruling C13: ``since`` = min(cpt_since, appeared-event year)
when an ``appeared`` event exists — cpt_since is only the earliest
*ingested* edition, not evidence the code didn't exist earlier."""
def test_since_is_min_of_cpt_presence_and_appeared_event(self):
# 99490 was created CY2015 but the oldest CPT edition on hand is
# 2019 — since must read 2015, not 2019.
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [_cpt_code("99490", "sec_a", year=2024)]
cpt_presence = {"99490": (2019, 2021, 2022, 2024)}
events = {"99490": [_ev("99490", 2015, "appeared")]}
rows = derive_families(
{},
events,
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_presence=cpt_presence,
)
by_code = {r.code: r for r in rows}
assert by_code["99490"].since == 2015
def test_since_falls_back_to_cpt_presence_when_no_appeared_event(self):
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [_cpt_code("90001", "sec_a", year=2024)]
cpt_presence = {"90001": (2019, 2021, 2022, 2024)}
rows = derive_families(
{},
{},
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_presence=cpt_presence,
)
by_code = {r.code: r for r in rows}
assert by_code["90001"].since == 2019
def test_since_keeps_cpt_presence_when_appeared_event_is_later(self):
# An appeared event newer than the CPT evidence must not win —
# min(), not "prefer the event".
sections = [_cpt_section("sec_a", ("Chapter", "Heading A"))]
cpt_codes = [_cpt_code("90001", "sec_a", year=2024)]
cpt_presence = {"90001": (2019, 2021, 2022, 2024)}
events = {"90001": [_ev("90001", 2021, "appeared")]}
rows = derive_families(
{},
events,
{},
cpt_codes=cpt_codes,
cpt_sections=sections,
cpt_presence=cpt_presence,
)
by_code = {r.code: r for r in rows}
assert by_code["90001"].since == 2019
class TestLoadAndRefresh:
def test_load_and_refresh_in_place(self, restore_families):
mod = restore_families
con = duckdb.connect(":memory:")
try:
ensure_tables(con)
assert load_families(con) == {}
before = dict(mod.FAMILIES)
assert refresh_from(con) == 0 and mod.FAMILIES == before
write_families(
con,
[
FamilyRow(
"CCM",
"Chronic Care Management",
"99490",
"base",
None,
None,
"",
0,
),
FamilyRow(
"CCM",
"Chronic Care Management",
"G2058",
"predecessor",
2020,
2021,
"",
0,
),
],
)
n = refresh_from(con)
# Ruling 17: a hand key the table reaches (CCM) is replaced by
# the derived (superset) codes; every other hand key survives
# untouched since `refresh_from` seeds from HAND_FAMILIES
# first instead of clearing the registry outright.
assert n == len(HAND_FAMILIES)
assert set(mod.FAMILIES) >= set(HAND_FAMILIES)
assert set(mod.FAMILIES["CCM"].codes) == {"99490", "G2058"}
assert mod.FAMILIES["CCM"].name == HAND_FAMILIES["CCM"].name
finally:
con.close()
def test_hand_key_absent_from_table_survives(self, restore_families):
# Ruling 17: only CCM appears in the derived table; the other four
# hand families (ACP, PCM, TCM, APCM) must come through unchanged.
mod = restore_families
con = duckdb.connect(":memory:")
try:
ensure_tables(con)
write_families(
con,
[
FamilyRow(
"CCM",
"Chronic Care Management",
"99490",
"base",
None,
None,
"",
0,
)
],
)
refresh_from(con)
for key in ("ACP", "PCM", "TCM", "APCM"):
assert mod.FAMILIES[key] == HAND_FAMILIES[key]
finally:
con.close()
def test_derived_family_gets_empty_synonyms(self, restore_families):
# F3: a derived (non-hand) family must not become a chat trigger
# off its own one/two-word name — matching by name is #699's job.
mod = restore_families
con = duckdb.connect(":memory:")
try:
ensure_tables(con)
write_families(
con,
[
FamilyRow(
"AORTA-REPAIR",
"Repair",
"99999",
"base",
None,
None,
"",
0,
)
],
)
refresh_from(con)
assert mod.FAMILIES["AORTA-REPAIR"].synonyms == ()
assert detect_codes("repair of the aorta").families == ()
finally:
con.close()
def test_hand_family_keeps_its_synonyms_after_refresh(self, restore_families):
mod = restore_families
con = duckdb.connect(":memory:")
try:
ensure_tables(con)
write_families(
con,
[
FamilyRow(
"CCM",
"Chronic Care Management",
"99490",
"base",
None,
None,
"",
0,
)
],
)
refresh_from(con)
assert mod.FAMILIES["CCM"].synonyms == HAND_FAMILIES["CCM"].synonyms
finally:
con.close()
def test_schema_present_table_absent_returns_empty(self):
# pfs schema created (e.g. by an earlier ensure_tables call for a
# sibling table) but pfs.code_family itself never materialized —
# an old replica shape, not a real error.
con = duckdb.connect(":memory:")
try:
con.execute("CREATE SCHEMA IF NOT EXISTS pfs;")
assert load_families(con) == {}
finally:
con.close()
def test_wrong_shape_table_raises(self):
# A pfs.code_family table exists but not in the expected shape —
# this is a real bug, not an absent table, and must not be
# swallowed.
con = duckdb.connect(":memory:")
try:
con.execute("CREATE SCHEMA IF NOT EXISTS pfs;")
con.execute("CREATE TABLE pfs.code_family (only_col VARCHAR);")
with pytest.raises(Exception):
load_families(con)
finally:
con.close()