fix(pfs,cli): families tolerate NULL RVU descriptions (refs #687)

Claude-Session: https://claude.ai/code/session_01Aum3pEMAM3yQVdFSdVe6Gc
This commit is contained in:
kert
2026-09-09 13:17:29 -04:00
parent aa11fe6a52
commit 5eef63e7b6
4 changed files with 44 additions and 7 deletions

View File

@@ -174,7 +174,7 @@ def families(write: bool = typer.Option(False, "--write")) -> None:
elements_by_code = {c: read_elements(con, c) for c in codes} elements_by_code = {c: read_elements(con, c) for c in codes}
events = {c: read_events(con, c) for c in codes} events = {c: read_events(con, c) for c in codes}
descriptions = { descriptions = {
r[0]: r[1] r[0]: r[1] or ""
for r in con.execute( for r in con.execute(
"SELECT hcpcs, arg_max(description, year) FROM pfs.rvu WHERE mod IS NULL OR mod = '' GROUP BY hcpcs" "SELECT hcpcs, arg_max(description, year) FROM pfs.rvu WHERE mod IS NULL OR mod = '' GROUP BY hcpcs"
).fetchall() ).fetchall()

View File

@@ -154,12 +154,17 @@ _NOISE = frozenset(
_RE_TOKEN = re.compile(r"(?<![a-z0-9])[a-z]+(?:/[a-z]+)?(?![a-z0-9])") _RE_TOKEN = re.compile(r"(?<![a-z0-9])[a-z]+(?:/[a-z]+)?(?![a-z0-9])")
def stem_tokens(description: str) -> frozenset[str]: def stem_tokens(description: str | None) -> frozenset[str]:
"""Short-descriptor tokens minus time/actor/number noise — the part """Short-descriptor tokens minus time/actor/number noise — the part
of a description that names the service.""" of a description that names the service.
``description`` may be ``None`` — a real ``pfs.rvu`` row can carry a
NULL description (a corpus parsing artifact, not a missing key), and
the caller's ``dict.get(c, default)`` only applies its default when
``c`` is absent, not when it maps to ``None``."""
return frozenset( return frozenset(
t t
for t in _RE_TOKEN.findall(description.lower()) for t in _RE_TOKEN.findall((description or "").lower())
if t not in _NOISE and len(t) > 1 if t not in _NOISE and len(t) > 1
) )
@@ -216,7 +221,7 @@ def derive_families(
if ev.kind in _LINK_EVENTS: if ev.kind in _LINK_EVENTS:
for other in (ev.from_codes + " " + ev.to_codes).split(): for other in (ev.from_codes + " " + ev.to_codes).split():
union(code, other.upper()) union(code, other.upper())
stems = {c: stem_tokens(descriptions.get(c, "")) for c in codes} stems = {c: stem_tokens(descriptions.get(c) or "") for c in codes}
acts = { acts = {
c: frozenset(e.value for e in elements.get(c, ()) if e.type == "activity") c: frozenset(e.value for e in elements.get(c, ()) if e.type == "activity")
for c in codes for c in codes
@@ -267,12 +272,12 @@ def derive_families(
key = ( key = (
"-".join( "-".join(
t t
for t in _RE_TOKEN.findall(descriptions.get(rep, "").lower()) for t in _RE_TOKEN.findall((descriptions.get(rep) or "").lower())
if t in stems[rep] if t in stems[rep]
).upper() ).upper()
or rep or rep
) )
name = descriptions.get(rep, rep) name = descriptions.get(rep) or rep
for c in sorted(members): for c in sorted(members):
els = elements.get(c, ()) els = elements.get(c, ())
evs = events.get(c, ()) evs = events.get(c, ())

View File

@@ -208,6 +208,18 @@ class TestFamilies:
assert {r.code for r in fams if r.key == "CCM"} >= {"99490", "99439", "G2058"} assert {r.code for r in fams if r.key == "CCM"} >= {"99490", "99439", "G2058"}
assert "CCM" in res.output assert "CCM" in res.output
def test_write_tolerates_null_rvu_description(self, con):
# #687 regression: a real pfs.rvu row can carry a NULL description
# (a corpus parsing artifact) at the current max year with a
# payable status — that row lands in the derivation scope and used
# to crash `families --write` with an AttributeError.
con.execute(
"INSERT INTO pfs.rvu VALUES (?,?,?,?,?,?)",
("Z9999", None, None, "A", 1.0, 2026),
)
res = runner.invoke(app, ["pfs", "families", "--write"])
assert res.exit_code == 0, res.output
class TestReview: class TestReview:
def test_lists_queue(self, con): def test_lists_queue(self, con):

View File

@@ -134,6 +134,13 @@ class TestStemTokens:
{"adv", "prim", "care", "mgmt"} {"adv", "prim", "care", "mgmt"}
) )
def test_none_description_yields_empty_set(self):
# A real pfs.rvu row can carry a NULL description (corpus parsing
# artifact, e.g. a garbage hcpcs value) — stem_tokens must not
# crash on it (#687 live-run regression: AttributeError on
# `None.lower()` in derive_families).
assert stem_tokens(None) == frozenset()
class TestDerive: class TestDerive:
def test_ccm_reproduced_with_predecessor_and_addon_roles(self): def test_ccm_reproduced_with_predecessor_and_addon_roles(self):
@@ -247,6 +254,19 @@ class TestDerive:
assert keys["10000"] == keys["10001"] assert keys["10000"] == keys["10001"]
assert keys["20000"] != keys["10000"] assert keys["20000"] != keys["10000"]
def test_none_description_does_not_raise_and_stands_alone(self):
# #687 live-run regression: a code whose `descriptions` mapping
# holds an explicit None (a real NULL pfs.rvu description, not a
# missing key — dict.get(c, "") only substitutes its default when
# c is absent) must not crash derive_families, and with nothing
# else linking it to another code it lands in its own family.
elements = {"99999": [_el("99999", "activity", "act-a")]}
descriptions = {"99999": None}
rows = derive_families(elements, {}, descriptions)
assert {r.code for r in rows} == {"99999"}
assert rows[0].key == "99999"
assert rows[0].name == "99999"
def test_many_distinct_stems_is_fast(self): def test_many_distinct_stems_is_fast(self):
# ~2,000 codes whose descriptions share no stem token with any # ~2,000 codes whose descriptions share no stem token with any
# other code's. The token-bucketed merge must not degrade to the # other code's. The token-bucketed merge must not degrade to the