`stack pfs elements --all-payable` targeted ~8.7k A/R/T codes and ran two full scans of the unindexed 193k-row `fr_anchors` table per code — the `LIKE '%CODE%'` descriptor-stem query plus a follow-up query per stem — on top of three per-code replica queries. ~15-25 minutes of SQL before the first model call, which is why the command had never been run end to end. Invert the loop, the way `lineage --all-payable` already does: - `pfs.descriptors.descriptor_runs_bucketed` streams `fr_anchors` once in `(item_key, p_id)` order — the order the per-code follow-up query walks — keeping each stem's run open until a paragraph fails `is_element_paragraph`, closes with a parenthesis, or hits the element cap. Stems are found with one `_stem_pattern_any` regex over all target codes instead of one compiled pattern per code. - `hcpcs_long_descriptions` / `rvu_descriptions_bucketed` / `_cpt_elements_bucketed` fetch every target code's row in one query each, keeping the same newest-row picks as `QUALIFY row_number()`. - `pfs.extract._assemble` becomes the single home of the precedence rules (FR > CPT > HCPCS/RVU with `confirmed_by` cross-checks), shared by the per-code `extract_code` and the new `extract_codes`, so the two paths cannot drift. `uv run stack pfs elements --all-payable --no-llm --dry-run` now finishes in ~12s for 8,689 codes, and all 17 hand-family codes extract identical rows and reviews on the live corpus. Drop the "experimental" wording and the "~20 min of SQL" warning; `_codes_for` keeps its signature and still logs the target count under `warn_slow`. No index on `fr_anchors.text`: the inverted pass never does a text LIKE, so an FTS5/trigram index would be write-path cost for nothing.
461 lines
17 KiB
Python
461 lines
17 KiB
Python
"""pfs.descriptors — where descriptor text comes from."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
|
|
import duckdb
|
|
import pytest
|
|
|
|
from pfs.descriptors import (
|
|
DescriptorRun,
|
|
_stem_pattern_any,
|
|
descriptor_runs,
|
|
descriptor_runs_bucketed,
|
|
hcpcs_long_description,
|
|
hcpcs_long_descriptions,
|
|
rule_year_of,
|
|
rvu_descriptions,
|
|
rvu_descriptions_bucketed,
|
|
)
|
|
|
|
|
|
class _Store:
|
|
"""Minimal stand-in for bib.Store: only ``_con()`` is used."""
|
|
|
|
def __init__(self) -> None:
|
|
self.con = sqlite3.connect(":memory:")
|
|
self.con.row_factory = sqlite3.Row
|
|
self.con.executescript(
|
|
"""
|
|
CREATE TABLE items (key TEXT PRIMARY KEY, title TEXT, date_published TEXT);
|
|
CREATE TABLE fr_anchors (item_key TEXT, p_id INTEGER, page INTEGER, ordinal INTEGER, text TEXT);
|
|
"""
|
|
)
|
|
|
|
def _con(self):
|
|
return self.con
|
|
|
|
def close(self):
|
|
self.con.close()
|
|
|
|
|
|
@pytest.fixture
|
|
def store():
|
|
s = _Store()
|
|
s.con.execute(
|
|
"INSERT INTO items VALUES ('DE2VH9PD', 'Medicare Program; CY 2015 PFS Final Rule', '2014-11-13')"
|
|
)
|
|
rows = [
|
|
(
|
|
"DE2VH9PD",
|
|
1243,
|
|
67716,
|
|
"Comment: commenters noted the CPT panel created a code.",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1244,
|
|
67716,
|
|
"These commenters suggested that we use the new CPT code 99490 (Chronic care management services, at least 20 minutes of clinical staff time directed by a physician or other qualified health care professional, per calendar month, with the following required elements:",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1245,
|
|
67716,
|
|
"Multiple (two or more) chronic conditions expected to last at least 12 months, or until the death of the patient;",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1246,
|
|
67716,
|
|
"Chronic conditions place the patient at significant risk of death, acute exacerbation/decompensation, or functional decline;",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1247,
|
|
67716,
|
|
"Comprehensive care plan established, implemented, revised, or monitored).",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1248,
|
|
67716,
|
|
"Many of these commenters expressed a preference for the per calendar month period.",
|
|
),
|
|
("DE2VH9PD", 1249, 67716, "Response: It is our preference to use CPT codes."),
|
|
]
|
|
s.con.executemany(
|
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
|
[(k, p, pg, p, t) for k, p, pg, t in rows],
|
|
)
|
|
yield s
|
|
s.close()
|
|
|
|
|
|
@pytest.fixture
|
|
def con():
|
|
c = duckdb.connect(":memory:")
|
|
c.execute("CREATE SCHEMA terminology; CREATE SCHEMA pfs")
|
|
c.execute(
|
|
"CREATE TABLE terminology.hcpcs_level_2 (hcpcs VARCHAR, seqnum VARCHAR, recid VARCHAR, long_description VARCHAR, short_description VARCHAR)"
|
|
)
|
|
c.execute(
|
|
"INSERT INTO terminology.hcpcs_level_2 VALUES ('G0556','10','3','Advanced primary care management services … per calendar month','Adv prim care mgmt lvl 1')"
|
|
)
|
|
c.execute(
|
|
"INSERT INTO terminology.hcpcs_level_2 VALUES ('G0556','5','1','Lower seqnum description for G0556','Short desc')"
|
|
)
|
|
c.execute(
|
|
"CREATE TABLE pfs.rvu (hcpcs VARCHAR, mod VARCHAR, description VARCHAR, status_code VARCHAR, year INTEGER)"
|
|
)
|
|
c.executemany(
|
|
"INSERT INTO pfs.rvu VALUES (?,?,?,?,?)",
|
|
[
|
|
("99490", None, "Chron care mgmt srvc 20 min", "A", 2015),
|
|
("99490", "26", "ignored modifier row", "A", 2015),
|
|
("99490", "", "Chrnc care mgmt staff 1st 20", "A", 2022),
|
|
],
|
|
)
|
|
yield c
|
|
c.close()
|
|
|
|
|
|
class TestRuleYear:
|
|
def test_from_title(self):
|
|
assert (
|
|
rule_year_of("Medicare Program; CY 2015 PFS Final Rule", "2014-11-13")
|
|
== 2015
|
|
)
|
|
assert (
|
|
rule_year_of(
|
|
"Medicare and Medicaid Programs; CY 2027 Payment Policies", "2026-07-16"
|
|
)
|
|
== 2027
|
|
)
|
|
|
|
def test_fallback_to_date_plus_one(self):
|
|
assert rule_year_of("Revisions to Payment Policies", "2017-07-21") == 2018
|
|
|
|
|
|
class TestDescriptorRuns:
|
|
def test_stem_and_following_elements(self, store):
|
|
runs = descriptor_runs(store, "99490")
|
|
assert len(runs) == 1
|
|
run = runs[0]
|
|
assert isinstance(run, DescriptorRun)
|
|
assert run.item_key == "DE2VH9PD" and run.rule_year == 2015
|
|
assert run.stem.p_id == 1244
|
|
assert [p.p_id for p in run.elements] == [1245, 1246, 1247]
|
|
assert (
|
|
"per calendar month" in run.text and "Comprehensive care plan" in run.text
|
|
)
|
|
|
|
def test_no_stem_no_run(self, store):
|
|
assert descriptor_runs(store, "99491") == []
|
|
|
|
def test_max_elements_cap(self, store):
|
|
run = descriptor_runs(store, "99490", max_elements=2)[0]
|
|
assert [p.p_id for p in run.elements] == [1245, 1246]
|
|
|
|
def test_enumeration_list_item_is_not_a_descriptor_stem(self, store):
|
|
# C1 live-corpus incident: "( 9) 99439 (code for non-complex
|
|
# chronic care management)." reads like a stem ("99439 (") but is
|
|
# an enumeration/cross-reference line, not the opening of 99439's
|
|
# own descriptor.
|
|
store.con.executemany(
|
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
|
[
|
|
(
|
|
"DE2VH9PD",
|
|
1300,
|
|
67800,
|
|
1300,
|
|
"( 9) 99439 (code for non-complex chronic care management).",
|
|
),
|
|
(
|
|
"DE2VH9PD",
|
|
1301,
|
|
67800,
|
|
1301,
|
|
"( 10) 99457 and 99458 (codes for remote physiologic monitoring).",
|
|
),
|
|
],
|
|
)
|
|
assert descriptor_runs(store, "99439") == []
|
|
|
|
def test_lowercase_stem_still_matches(self):
|
|
# `_stem_pattern` stays case-insensitive: some rules print the
|
|
# descriptor lowercase.
|
|
s = _Store()
|
|
s.con.execute(
|
|
"INSERT INTO items VALUES ('AAAAAAAA', 'Medicare Program; CY 2021 PFS Final Rule', '2020-11-13')"
|
|
)
|
|
s.con.executemany(
|
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
|
[
|
|
(
|
|
"AAAAAAAA",
|
|
1,
|
|
100,
|
|
1,
|
|
"CPT code 99490 (chronic care management services, at least 20 "
|
|
"minutes of clinical staff time directed by a physician or other "
|
|
"qualified health care professional, per calendar month, with the "
|
|
"following required elements:",
|
|
),
|
|
("AAAAAAAA", 2, 100, 2, "Consent;"),
|
|
],
|
|
)
|
|
try:
|
|
runs = descriptor_runs(s, "99490")
|
|
assert len(runs) == 1
|
|
finally:
|
|
s.close()
|
|
|
|
|
|
class TestDuckDBSources:
|
|
def test_long_description(self, con):
|
|
assert hcpcs_long_description(con, "G0556").startswith("Advanced primary care")
|
|
assert hcpcs_long_description(con, "99490") == ""
|
|
|
|
def test_long_description_deterministic_seqnum(self, con):
|
|
# With multiple rows for same hcpcs, ORDER BY seqnum DESC picks the highest
|
|
result = hcpcs_long_description(con, "G0556")
|
|
assert (
|
|
result == "Advanced primary care management services … per calendar month"
|
|
)
|
|
|
|
def test_rvu_descriptions_base_rows_only(self, con):
|
|
assert rvu_descriptions(con, "99490") == [
|
|
(2015, "A", "Chron care mgmt srvc 20 min"),
|
|
(2022, "A", "Chrnc care mgmt staff 1st 20"),
|
|
]
|
|
|
|
|
|
class TestDescriptorSpan:
|
|
"""#700: a long background paragraph quoting a descriptor inline must
|
|
contribute only the quoted descriptor, not the whole paragraph."""
|
|
|
|
def test_trims_to_the_codes_parenthetical(self):
|
|
from pfs.descriptors import descriptor_span
|
|
|
|
text = (
|
|
"In the CY 2017 PFS final rule we created HCPCS code G0502 (Initial "
|
|
"psychiatric collaborative care management, first 70 minutes in the "
|
|
"first calendar month of behavioral health care manager activities "
|
|
"(in consultation with a psychiatric consultant)). "
|
|
+ "Background prose about chronic care management services, comprehensive "
|
|
"care plans and multiple chronic conditions expected to last 12 months. "
|
|
* 60
|
|
)
|
|
span = descriptor_span(text, "G0502")
|
|
assert span.startswith("G0502 (Initial psychiatric")
|
|
assert span.endswith("consultant))")
|
|
assert "chronic care management" not in span
|
|
assert len(span) < 300
|
|
|
|
def test_no_close_paren_cuts_at_a_sentence_inside_the_cap(self):
|
|
from pfs.descriptors import descriptor_span
|
|
|
|
text = "HCPCS code G0502 (Initial psychiatric care. " + "More words. " * 400
|
|
span = descriptor_span(text, "G0502", max_chars=200)
|
|
assert len(span) <= 200
|
|
assert span.endswith(".")
|
|
|
|
def test_no_stem_returns_the_text(self):
|
|
from pfs.descriptors import descriptor_span
|
|
|
|
assert descriptor_span("nothing about the code here", "G0502") == (
|
|
"nothing about the code here"
|
|
)
|
|
|
|
def test_descriptor_runs_stem_is_the_span(self, store):
|
|
from pfs.descriptors import descriptor_runs
|
|
|
|
con = store._con()
|
|
con.execute(
|
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
|
(
|
|
"LONGPARA",
|
|
9,
|
|
1,
|
|
1,
|
|
"As background, HCPCS code G0502 (Initial psychiatric collaborative "
|
|
"care management, first 70 minutes) was created in CY 2017. "
|
|
+ "Chronic care management prose. "
|
|
* 100,
|
|
),
|
|
)
|
|
con.execute(
|
|
"INSERT INTO items (key, title, date_published) VALUES (?,?,?)",
|
|
("LONGPARA", "Medicare Program; CY 2027 PFS Proposed Rule", "2026-07-16"),
|
|
)
|
|
con.commit()
|
|
run = [r for r in descriptor_runs(store, "G0502") if r.item_key == "LONGPARA"][
|
|
0
|
|
]
|
|
assert run.stem.text.startswith("G0502 (Initial psychiatric")
|
|
assert "Chronic care management prose" not in run.stem.text
|
|
|
|
|
|
@pytest.fixture
|
|
def wide_store():
|
|
"""Three rule items naming several codes, so the inverted pass has to
|
|
reset its open runs at an item boundary, carry two stems through the
|
|
same element paragraphs, and order one code's runs across items."""
|
|
s = _Store()
|
|
s.con.executemany(
|
|
"INSERT INTO items VALUES (?,?,?)",
|
|
[
|
|
("AAAAAAAA", "Medicare Program; CY 2015 PFS Final Rule", "2014-11-13"),
|
|
("BBBBBBBB", "Medicare Program; CY 2021 PFS Final Rule", "2020-12-28"),
|
|
("CCCCCCCC", "Revisions to Payment Policies", "2016-11-15"),
|
|
],
|
|
)
|
|
rows = [
|
|
# AAAAAAAA — a plain stem run, then an unrelated prose paragraph
|
|
(
|
|
"AAAAAAAA",
|
|
10,
|
|
100,
|
|
"Comment: commenters noted the CPT panel created a code.",
|
|
),
|
|
(
|
|
"AAAAAAAA",
|
|
11,
|
|
100,
|
|
"We use the new CPT code 99490 (Chronic care management services, at least "
|
|
"20 minutes of clinical staff time directed by a physician or other "
|
|
"qualified health care professional, per calendar month, with the "
|
|
"following required elements:",
|
|
),
|
|
("AAAAAAAA", 12, 100, "Consent;"),
|
|
(
|
|
"AAAAAAAA",
|
|
13,
|
|
101,
|
|
"Comprehensive care plan established, implemented, revised, or monitored).",
|
|
),
|
|
("AAAAAAAA", 14, 101, "Response: it is our preference to use CPT codes."),
|
|
# BBBBBBBB — two stems back to back, so 99487's run must absorb the
|
|
# element paragraphs that follow 99489's stem too
|
|
(
|
|
"BBBBBBBB",
|
|
20,
|
|
200,
|
|
"CPT code 99487 (Complex chronic care management services, with the "
|
|
"following required elements:",
|
|
),
|
|
(
|
|
"BBBBBBBB",
|
|
21,
|
|
200,
|
|
"99489 (Each additional 30 minutes of clinical staff time, per calendar "
|
|
"month, with the following required elements:",
|
|
),
|
|
("BBBBBBBB", 22, 200, "Consent;"),
|
|
(
|
|
"BBBBBBBB",
|
|
23,
|
|
201,
|
|
"Provide 24/7 access for urgent needs to care team/practitioner;",
|
|
),
|
|
("BBBBBBBB", 24, 201, "We received many comments on this proposal."),
|
|
(
|
|
"BBBBBBBB",
|
|
25,
|
|
201,
|
|
"( 9) 99439 (code for non-complex chronic care management).",
|
|
),
|
|
# CCCCCCCC — an earlier-dated item mentioning 99490 again, so the
|
|
# per-code ORDER BY date_published puts it first
|
|
(
|
|
"CCCCCCCC",
|
|
30,
|
|
300,
|
|
"We finalized 99490 (Chronic care management services, per calendar "
|
|
"month, with the following required elements:",
|
|
),
|
|
("CCCCCCCC", 31, 300, "Consent;"),
|
|
# an odd-length target code, exercising the literal branch of
|
|
# `_stem_pattern_any`
|
|
(
|
|
"CCCCCCCC",
|
|
32,
|
|
300,
|
|
"And HCPCS code G0556X (Advanced primary care management services, per "
|
|
"calendar month, with the following elements:",
|
|
),
|
|
("CCCCCCCC", 33, 300, "Consent;"),
|
|
]
|
|
s.con.executemany(
|
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
|
[(k, pid, pg, pid, t) for k, pid, pg, t in rows],
|
|
)
|
|
yield s
|
|
s.close()
|
|
|
|
|
|
class TestDescriptorRunsBucketed:
|
|
"""#698: one streaming pass over ``fr_anchors`` must reproduce
|
|
``descriptor_runs`` exactly, code for code."""
|
|
|
|
_CODES = ("99490", "99487", "99489", "99439", "G0556X", "00000")
|
|
|
|
def test_matches_descriptor_runs_per_code(self, wide_store):
|
|
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
|
|
assert sorted(bucketed) == sorted(self._CODES)
|
|
for c in self._CODES:
|
|
assert bucketed[c] == descriptor_runs(wide_store, c), c
|
|
|
|
def test_finds_the_runs_the_fixture_plants(self, wide_store):
|
|
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
|
|
# 99490 stems in two items, ordered by the rule's publication date
|
|
assert [r.item_key for r in bucketed["99490"]] == ["AAAAAAAA", "CCCCCCCC"]
|
|
# 99487's run keeps taking element paragraphs past 99489's stem
|
|
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22, 23]
|
|
assert [p.p_id for p in bucketed["99489"][0].elements] == [22, 23]
|
|
# an enumeration line is not a stem, and an unmentioned code is empty
|
|
assert bucketed["99439"] == [] and bucketed["00000"] == []
|
|
# the odd-length literal branch of `_stem_pattern_any` still matches
|
|
assert [r.stem.p_id for r in bucketed["G0556X"]] == [32]
|
|
|
|
def test_max_elements_cap_matches(self, wide_store):
|
|
bucketed = descriptor_runs_bucketed(wide_store, ["99487"], max_elements=2)
|
|
assert bucketed["99487"] == descriptor_runs(wide_store, "99487", max_elements=2)
|
|
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22]
|
|
|
|
def test_no_codes_is_an_empty_result(self, wide_store):
|
|
assert descriptor_runs_bucketed(wide_store, []) == {}
|
|
|
|
def test_generic_token_only_when_every_target_is_five_wide(self):
|
|
# The all-five-wide fast path compiles the character class alone;
|
|
# an odd-length target adds its own literal alternative.
|
|
assert "|" not in _stem_pattern_any(["99490", "G0556"]).pattern
|
|
assert "G0556X" in _stem_pattern_any(["99490", "G0556X"]).pattern
|
|
|
|
|
|
class TestBucketedReplicaLookups:
|
|
"""The DuckDB half of the inversion: one query per table for every
|
|
target code, matching the per-code helpers row for row."""
|
|
|
|
def test_hcpcs_long_descriptions_matches_per_code(self, con):
|
|
codes = ["G0556", "G9999"]
|
|
bucketed = hcpcs_long_descriptions(con, codes)
|
|
for c in codes:
|
|
assert bucketed.get(c, "") == hcpcs_long_description(con, c), c
|
|
assert bucketed["G0556"].startswith("Advanced primary care")
|
|
assert "G9999" not in bucketed
|
|
|
|
def test_rvu_descriptions_matches_per_code(self, con):
|
|
codes = ["99490", "99999"]
|
|
bucketed = rvu_descriptions_bucketed(con, codes)
|
|
for c in codes:
|
|
assert bucketed.get(c, []) == rvu_descriptions(con, c), c
|
|
assert [y for y, _s, _d in bucketed["99490"]] == [2015, 2022]
|
|
|
|
def test_no_codes_is_an_empty_result(self, con):
|
|
assert hcpcs_long_descriptions(con, []) == {}
|
|
assert rvu_descriptions_bucketed(con, []) == {}
|