Files
stack/tests/pfs/test_descriptors.py
kert e14f12e18e perf(pfs,cli): elements --all-payable in one inverted pass, not two fr_anchors scans per code (refs #698)
`stack pfs elements --all-payable` targeted ~8.7k A/R/T codes and ran two
full scans of the unindexed 193k-row `fr_anchors` table per code — the
`LIKE '%CODE%'` descriptor-stem query plus a follow-up query per stem —
on top of three per-code replica queries. ~15-25 minutes of SQL before
the first model call, which is why the command had never been run end to
end.

Invert the loop, the way `lineage --all-payable` already does:

- `pfs.descriptors.descriptor_runs_bucketed` streams `fr_anchors` once in
  `(item_key, p_id)` order — the order the per-code follow-up query walks
  — keeping each stem's run open until a paragraph fails
  `is_element_paragraph`, closes with a parenthesis, or hits the element
  cap. Stems are found with one `_stem_pattern_any` regex over all target
  codes instead of one compiled pattern per code.
- `hcpcs_long_descriptions` / `rvu_descriptions_bucketed` /
  `_cpt_elements_bucketed` fetch every target code's row in one query
  each, keeping the same newest-row picks as `QUALIFY row_number()`.
- `pfs.extract._assemble` becomes the single home of the precedence rules
  (FR > CPT > HCPCS/RVU with `confirmed_by` cross-checks), shared by the
  per-code `extract_code` and the new `extract_codes`, so the two paths
  cannot drift.

`uv run stack pfs elements --all-payable --no-llm --dry-run` now finishes
in ~12s for 8,689 codes, and all 17 hand-family codes extract identical
rows and reviews on the live corpus. Drop the "experimental" wording and
the "~20 min of SQL" warning; `_codes_for` keeps its signature and still
logs the target count under `warn_slow`.

No index on `fr_anchors.text`: the inverted pass never does a text LIKE,
so an FTS5/trigram index would be write-path cost for nothing.
2026-09-11 18:16:56 -04:00

461 lines
17 KiB
Python

"""pfs.descriptors — where descriptor text comes from."""
from __future__ import annotations
import sqlite3
import duckdb
import pytest
from pfs.descriptors import (
DescriptorRun,
_stem_pattern_any,
descriptor_runs,
descriptor_runs_bucketed,
hcpcs_long_description,
hcpcs_long_descriptions,
rule_year_of,
rvu_descriptions,
rvu_descriptions_bucketed,
)
class _Store:
"""Minimal stand-in for bib.Store: only ``_con()`` is used."""
def __init__(self) -> None:
self.con = sqlite3.connect(":memory:")
self.con.row_factory = sqlite3.Row
self.con.executescript(
"""
CREATE TABLE items (key TEXT PRIMARY KEY, title TEXT, date_published TEXT);
CREATE TABLE fr_anchors (item_key TEXT, p_id INTEGER, page INTEGER, ordinal INTEGER, text TEXT);
"""
)
def _con(self):
return self.con
def close(self):
self.con.close()
@pytest.fixture
def store():
s = _Store()
s.con.execute(
"INSERT INTO items VALUES ('DE2VH9PD', 'Medicare Program; CY 2015 PFS Final Rule', '2014-11-13')"
)
rows = [
(
"DE2VH9PD",
1243,
67716,
"Comment: commenters noted the CPT panel created a code.",
),
(
"DE2VH9PD",
1244,
67716,
"These commenters suggested that we use the new CPT code 99490 (Chronic care management services, at least 20 minutes of clinical staff time directed by a physician or other qualified health care professional, per calendar month, with the following required elements:",
),
(
"DE2VH9PD",
1245,
67716,
"Multiple (two or more) chronic conditions expected to last at least 12 months, or until the death of the patient;",
),
(
"DE2VH9PD",
1246,
67716,
"Chronic conditions place the patient at significant risk of death, acute exacerbation/decompensation, or functional decline;",
),
(
"DE2VH9PD",
1247,
67716,
"Comprehensive care plan established, implemented, revised, or monitored).",
),
(
"DE2VH9PD",
1248,
67716,
"Many of these commenters expressed a preference for the per calendar month period.",
),
("DE2VH9PD", 1249, 67716, "Response: It is our preference to use CPT codes."),
]
s.con.executemany(
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
[(k, p, pg, p, t) for k, p, pg, t in rows],
)
yield s
s.close()
@pytest.fixture
def con():
c = duckdb.connect(":memory:")
c.execute("CREATE SCHEMA terminology; CREATE SCHEMA pfs")
c.execute(
"CREATE TABLE terminology.hcpcs_level_2 (hcpcs VARCHAR, seqnum VARCHAR, recid VARCHAR, long_description VARCHAR, short_description VARCHAR)"
)
c.execute(
"INSERT INTO terminology.hcpcs_level_2 VALUES ('G0556','10','3','Advanced primary care management services … per calendar month','Adv prim care mgmt lvl 1')"
)
c.execute(
"INSERT INTO terminology.hcpcs_level_2 VALUES ('G0556','5','1','Lower seqnum description for G0556','Short desc')"
)
c.execute(
"CREATE TABLE pfs.rvu (hcpcs VARCHAR, mod VARCHAR, description VARCHAR, status_code VARCHAR, year INTEGER)"
)
c.executemany(
"INSERT INTO pfs.rvu VALUES (?,?,?,?,?)",
[
("99490", None, "Chron care mgmt srvc 20 min", "A", 2015),
("99490", "26", "ignored modifier row", "A", 2015),
("99490", "", "Chrnc care mgmt staff 1st 20", "A", 2022),
],
)
yield c
c.close()
class TestRuleYear:
def test_from_title(self):
assert (
rule_year_of("Medicare Program; CY 2015 PFS Final Rule", "2014-11-13")
== 2015
)
assert (
rule_year_of(
"Medicare and Medicaid Programs; CY 2027 Payment Policies", "2026-07-16"
)
== 2027
)
def test_fallback_to_date_plus_one(self):
assert rule_year_of("Revisions to Payment Policies", "2017-07-21") == 2018
class TestDescriptorRuns:
def test_stem_and_following_elements(self, store):
runs = descriptor_runs(store, "99490")
assert len(runs) == 1
run = runs[0]
assert isinstance(run, DescriptorRun)
assert run.item_key == "DE2VH9PD" and run.rule_year == 2015
assert run.stem.p_id == 1244
assert [p.p_id for p in run.elements] == [1245, 1246, 1247]
assert (
"per calendar month" in run.text and "Comprehensive care plan" in run.text
)
def test_no_stem_no_run(self, store):
assert descriptor_runs(store, "99491") == []
def test_max_elements_cap(self, store):
run = descriptor_runs(store, "99490", max_elements=2)[0]
assert [p.p_id for p in run.elements] == [1245, 1246]
def test_enumeration_list_item_is_not_a_descriptor_stem(self, store):
# C1 live-corpus incident: "( 9) 99439 (code for non-complex
# chronic care management)." reads like a stem ("99439 (") but is
# an enumeration/cross-reference line, not the opening of 99439's
# own descriptor.
store.con.executemany(
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
[
(
"DE2VH9PD",
1300,
67800,
1300,
"( 9) 99439 (code for non-complex chronic care management).",
),
(
"DE2VH9PD",
1301,
67800,
1301,
"( 10) 99457 and 99458 (codes for remote physiologic monitoring).",
),
],
)
assert descriptor_runs(store, "99439") == []
def test_lowercase_stem_still_matches(self):
# `_stem_pattern` stays case-insensitive: some rules print the
# descriptor lowercase.
s = _Store()
s.con.execute(
"INSERT INTO items VALUES ('AAAAAAAA', 'Medicare Program; CY 2021 PFS Final Rule', '2020-11-13')"
)
s.con.executemany(
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
[
(
"AAAAAAAA",
1,
100,
1,
"CPT code 99490 (chronic care management services, at least 20 "
"minutes of clinical staff time directed by a physician or other "
"qualified health care professional, per calendar month, with the "
"following required elements:",
),
("AAAAAAAA", 2, 100, 2, "Consent;"),
],
)
try:
runs = descriptor_runs(s, "99490")
assert len(runs) == 1
finally:
s.close()
class TestDuckDBSources:
def test_long_description(self, con):
assert hcpcs_long_description(con, "G0556").startswith("Advanced primary care")
assert hcpcs_long_description(con, "99490") == ""
def test_long_description_deterministic_seqnum(self, con):
# With multiple rows for same hcpcs, ORDER BY seqnum DESC picks the highest
result = hcpcs_long_description(con, "G0556")
assert (
result == "Advanced primary care management services … per calendar month"
)
def test_rvu_descriptions_base_rows_only(self, con):
assert rvu_descriptions(con, "99490") == [
(2015, "A", "Chron care mgmt srvc 20 min"),
(2022, "A", "Chrnc care mgmt staff 1st 20"),
]
class TestDescriptorSpan:
"""#700: a long background paragraph quoting a descriptor inline must
contribute only the quoted descriptor, not the whole paragraph."""
def test_trims_to_the_codes_parenthetical(self):
from pfs.descriptors import descriptor_span
text = (
"In the CY 2017 PFS final rule we created HCPCS code G0502 (Initial "
"psychiatric collaborative care management, first 70 minutes in the "
"first calendar month of behavioral health care manager activities "
"(in consultation with a psychiatric consultant)). "
+ "Background prose about chronic care management services, comprehensive "
"care plans and multiple chronic conditions expected to last 12 months. "
* 60
)
span = descriptor_span(text, "G0502")
assert span.startswith("G0502 (Initial psychiatric")
assert span.endswith("consultant))")
assert "chronic care management" not in span
assert len(span) < 300
def test_no_close_paren_cuts_at_a_sentence_inside_the_cap(self):
from pfs.descriptors import descriptor_span
text = "HCPCS code G0502 (Initial psychiatric care. " + "More words. " * 400
span = descriptor_span(text, "G0502", max_chars=200)
assert len(span) <= 200
assert span.endswith(".")
def test_no_stem_returns_the_text(self):
from pfs.descriptors import descriptor_span
assert descriptor_span("nothing about the code here", "G0502") == (
"nothing about the code here"
)
def test_descriptor_runs_stem_is_the_span(self, store):
from pfs.descriptors import descriptor_runs
con = store._con()
con.execute(
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
(
"LONGPARA",
9,
1,
1,
"As background, HCPCS code G0502 (Initial psychiatric collaborative "
"care management, first 70 minutes) was created in CY 2017. "
+ "Chronic care management prose. "
* 100,
),
)
con.execute(
"INSERT INTO items (key, title, date_published) VALUES (?,?,?)",
("LONGPARA", "Medicare Program; CY 2027 PFS Proposed Rule", "2026-07-16"),
)
con.commit()
run = [r for r in descriptor_runs(store, "G0502") if r.item_key == "LONGPARA"][
0
]
assert run.stem.text.startswith("G0502 (Initial psychiatric")
assert "Chronic care management prose" not in run.stem.text
@pytest.fixture
def wide_store():
"""Three rule items naming several codes, so the inverted pass has to
reset its open runs at an item boundary, carry two stems through the
same element paragraphs, and order one code's runs across items."""
s = _Store()
s.con.executemany(
"INSERT INTO items VALUES (?,?,?)",
[
("AAAAAAAA", "Medicare Program; CY 2015 PFS Final Rule", "2014-11-13"),
("BBBBBBBB", "Medicare Program; CY 2021 PFS Final Rule", "2020-12-28"),
("CCCCCCCC", "Revisions to Payment Policies", "2016-11-15"),
],
)
rows = [
# AAAAAAAA — a plain stem run, then an unrelated prose paragraph
(
"AAAAAAAA",
10,
100,
"Comment: commenters noted the CPT panel created a code.",
),
(
"AAAAAAAA",
11,
100,
"We use the new CPT code 99490 (Chronic care management services, at least "
"20 minutes of clinical staff time directed by a physician or other "
"qualified health care professional, per calendar month, with the "
"following required elements:",
),
("AAAAAAAA", 12, 100, "Consent;"),
(
"AAAAAAAA",
13,
101,
"Comprehensive care plan established, implemented, revised, or monitored).",
),
("AAAAAAAA", 14, 101, "Response: it is our preference to use CPT codes."),
# BBBBBBBB — two stems back to back, so 99487's run must absorb the
# element paragraphs that follow 99489's stem too
(
"BBBBBBBB",
20,
200,
"CPT code 99487 (Complex chronic care management services, with the "
"following required elements:",
),
(
"BBBBBBBB",
21,
200,
"99489 (Each additional 30 minutes of clinical staff time, per calendar "
"month, with the following required elements:",
),
("BBBBBBBB", 22, 200, "Consent;"),
(
"BBBBBBBB",
23,
201,
"Provide 24/7 access for urgent needs to care team/practitioner;",
),
("BBBBBBBB", 24, 201, "We received many comments on this proposal."),
(
"BBBBBBBB",
25,
201,
"( 9) 99439 (code for non-complex chronic care management).",
),
# CCCCCCCC — an earlier-dated item mentioning 99490 again, so the
# per-code ORDER BY date_published puts it first
(
"CCCCCCCC",
30,
300,
"We finalized 99490 (Chronic care management services, per calendar "
"month, with the following required elements:",
),
("CCCCCCCC", 31, 300, "Consent;"),
# an odd-length target code, exercising the literal branch of
# `_stem_pattern_any`
(
"CCCCCCCC",
32,
300,
"And HCPCS code G0556X (Advanced primary care management services, per "
"calendar month, with the following elements:",
),
("CCCCCCCC", 33, 300, "Consent;"),
]
s.con.executemany(
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
[(k, pid, pg, pid, t) for k, pid, pg, t in rows],
)
yield s
s.close()
class TestDescriptorRunsBucketed:
"""#698: one streaming pass over ``fr_anchors`` must reproduce
``descriptor_runs`` exactly, code for code."""
_CODES = ("99490", "99487", "99489", "99439", "G0556X", "00000")
def test_matches_descriptor_runs_per_code(self, wide_store):
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
assert sorted(bucketed) == sorted(self._CODES)
for c in self._CODES:
assert bucketed[c] == descriptor_runs(wide_store, c), c
def test_finds_the_runs_the_fixture_plants(self, wide_store):
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
# 99490 stems in two items, ordered by the rule's publication date
assert [r.item_key for r in bucketed["99490"]] == ["AAAAAAAA", "CCCCCCCC"]
# 99487's run keeps taking element paragraphs past 99489's stem
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22, 23]
assert [p.p_id for p in bucketed["99489"][0].elements] == [22, 23]
# an enumeration line is not a stem, and an unmentioned code is empty
assert bucketed["99439"] == [] and bucketed["00000"] == []
# the odd-length literal branch of `_stem_pattern_any` still matches
assert [r.stem.p_id for r in bucketed["G0556X"]] == [32]
def test_max_elements_cap_matches(self, wide_store):
bucketed = descriptor_runs_bucketed(wide_store, ["99487"], max_elements=2)
assert bucketed["99487"] == descriptor_runs(wide_store, "99487", max_elements=2)
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22]
def test_no_codes_is_an_empty_result(self, wide_store):
assert descriptor_runs_bucketed(wide_store, []) == {}
def test_generic_token_only_when_every_target_is_five_wide(self):
# The all-five-wide fast path compiles the character class alone;
# an odd-length target adds its own literal alternative.
assert "|" not in _stem_pattern_any(["99490", "G0556"]).pattern
assert "G0556X" in _stem_pattern_any(["99490", "G0556X"]).pattern
class TestBucketedReplicaLookups:
"""The DuckDB half of the inversion: one query per table for every
target code, matching the per-code helpers row for row."""
def test_hcpcs_long_descriptions_matches_per_code(self, con):
codes = ["G0556", "G9999"]
bucketed = hcpcs_long_descriptions(con, codes)
for c in codes:
assert bucketed.get(c, "") == hcpcs_long_description(con, c), c
assert bucketed["G0556"].startswith("Advanced primary care")
assert "G9999" not in bucketed
def test_rvu_descriptions_matches_per_code(self, con):
codes = ["99490", "99999"]
bucketed = rvu_descriptions_bucketed(con, codes)
for c in codes:
assert bucketed.get(c, []) == rvu_descriptions(con, c), c
assert [y for y, _s, _d in bucketed["99490"]] == [2015, 2022]
def test_no_codes_is_an_empty_result(self, con):
assert hcpcs_long_descriptions(con, []) == {}
assert rvu_descriptions_bucketed(con, []) == {}