Files
stack/tests/pfs/test_utilization.py
kert 4c69d5be9b feat(pfs): Part B utilization per HCPCS — pfs.utilization from the CMS by-Geography-and-Service PUFs, stack pfs utilization, notebook 7d/7e (refs #694)
One dataset per calendar year on data.cms.gov (2013–2024, UUIDs in
pfs.utilization.DATASETS), pulled through the Data API filtered to
national rows (~13k a year, one per HCPCS × place of service), cached
under data/cms/utilization/, loaded delete-then-insert per year, each
year cited in bib as a Source (module:pfs, table:pfs.utilization,
source:cms-website, year:N) and logged in cms.ingest_log with the
cache file's sha256. Suppressed cells (<11 beneficiaries) load as NULL.
read_series() sums places of service and service-weights the average
allowed/paid amounts.

stack pfs utilization [--year N]... [--write [--offline]] [--code C]...
Notebook 7d charts a family's services per year with the exposure
dates from 7c as rules; 7e answers the three #694 questions (99490
2015–, G2058→99439 continuity, APCM 2025 vs CCM — no 2025 file yet).
CLI docs regenerated (sidebar renumbered around the new page).
2026-09-22 16:04:41 -04:00

229 lines
7.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""pfs.utilization — Medicare Physician & Other Practitioners by Geography
and Service (national totals per HCPCS × place of service), #694."""
from __future__ import annotations
import json
import duckdb
import httpx
import pytest
from bib.store import Store
from pfs.utilization import (
API,
DATASETS,
UtilizationRow,
cite,
ensure_table,
fetch_year,
read_series,
to_rows,
write,
)
REC = {
"Rndrng_Prvdr_Geo_Lvl": "National",
"Rndrng_Prvdr_Geo_Cd": "",
"Rndrng_Prvdr_Geo_Desc": "National",
"HCPCS_Cd": "99490",
"HCPCS_Desc": "Chronic care management services, first 20 minutes",
"HCPCS_Drug_Ind": "N",
"Place_Of_Srvc": "O",
"Tot_Rndrng_Prvdrs": "31182",
"Tot_Benes": "1158246",
"Tot_Srvcs": "5715325",
"Tot_Bene_Day_Srvcs": "5676766",
"Avg_Sbmtd_Chrg": "106.87513302",
"Avg_Mdcr_Alowd_Amt": "60.722041509",
"Avg_Mdcr_Pymt_Amt": "46.196134143",
"Avg_Mdcr_Stdzd_Amt": "45.634579675",
}
class TestDatasets:
def test_every_year_2013_to_2024_has_a_dataset(self):
assert sorted(DATASETS) == list(range(2013, 2025))
assert all(len(v) == 36 for v in DATASETS.values())
class TestToRows:
def test_parses_numbers(self):
rows = to_rows(2023, [REC], dataset_id="d", item_key="K")
assert len(rows) == 1
r = rows[0]
assert isinstance(r, UtilizationRow)
assert (r.year, r.hcpcs, r.place_of_service, r.drug_ind) == (
2023,
"99490",
"O",
"N",
)
assert r.n_providers == 31182 and r.n_beneficiaries == 1158246
assert r.n_services == 5715325.0 and r.n_bene_day_services == 5676766.0
assert r.avg_allowed == pytest.approx(60.722041509)
assert r.avg_paid == pytest.approx(46.196134143)
assert r.avg_submitted == pytest.approx(106.87513302)
assert r.avg_standardized == pytest.approx(45.634579675)
assert r.dataset_id == "d" and r.item_key == "K"
def test_blanks_become_none_and_non_national_rows_are_skipped(self):
blank = dict(REC, Tot_Benes="", Avg_Mdcr_Pymt_Amt="")
state = dict(REC, Rndrng_Prvdr_Geo_Lvl="State", Rndrng_Prvdr_Geo_Desc="Ohio")
rows = to_rows(2023, [blank, state], dataset_id="d", item_key="K")
assert len(rows) == 1
assert rows[0].n_beneficiaries is None and rows[0].avg_paid is None
def _paged_client(pages: list[list[dict]], seen: list[str]):
def handler(request):
seen.append(str(request.url))
offset = int(request.url.params.get("offset", "0"))
size = int(request.url.params.get("size", "0"))
idx = offset // size if size else 0
body = pages[idx] if idx < len(pages) else []
return httpx.Response(200, json=body)
return httpx.Client(transport=httpx.MockTransport(handler))
class TestFetchYear:
def test_pages_until_a_short_page(self):
seen: list[str] = []
pages = [[REC] * 3, [REC] * 3, [REC]] # size 3 → stop after the short page
client = _paged_client(pages, seen)
out = fetch_year(client, 2023, page_size=3)
assert len(out) == 7
assert len(seen) == 3
assert seen[0].startswith(API.format(id=DATASETS[2023]))
assert "Rndrng_Prvdr_Geo_Lvl%5D=National" in seen[0]
assert "offset=3" in seen[1] and "offset=6" in seen[2]
def test_exact_multiple_stops_on_empty_page(self):
seen: list[str] = []
client = _paged_client([[REC] * 2, [REC] * 2, []], seen)
assert len(fetch_year(client, 2023, page_size=2)) == 4
assert len(seen) == 3
def test_unknown_year(self):
with pytest.raises(KeyError):
fetch_year(httpx.Client(), 1999)
def test_http_error_raises(self):
client = httpx.Client(
transport=httpx.MockTransport(lambda r: httpx.Response(503))
)
with pytest.raises(httpx.HTTPStatusError):
fetch_year(client, 2023)
@pytest.fixture
def con():
c = duckdb.connect(":memory:")
yield c
c.close()
class TestWriteAndRead:
def test_write_is_idempotent_per_year(self, con):
rows23 = to_rows(
2023,
[REC, dict(REC, Place_Of_Srvc="F", Tot_Srvcs="100")],
dataset_id="d",
item_key="K",
)
rows22 = to_rows(
2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J"
)
assert write(con, 2023, rows23) == 2
assert write(con, 2022, rows22) == 1
assert write(con, 2023, rows23) == 2 # replaces, never appends
assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 3
def test_read_series_sums_places_of_service(self, con):
write(
con,
2023,
to_rows(
2023,
[
REC,
dict(
REC,
Place_Of_Srvc="F",
Tot_Srvcs="100",
Tot_Benes="10",
Tot_Rndrng_Prvdrs="5",
),
],
dataset_id="d",
item_key="K",
),
)
write(
con,
2022,
to_rows(
2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J"
),
)
write(
con,
2022,
to_rows(
2022,
[
dict(REC, Tot_Srvcs="4000000"),
dict(REC, HCPCS_Cd="99439", Tot_Srvcs="7"),
],
dataset_id="e",
item_key="J",
),
)
s = read_series(con, ["99490"])
assert [(r.year, r.n_services) for r in s] == [
(2022, 4000000.0),
(2023, 5715425.0),
]
assert s[1].n_beneficiaries == 1158256 and s[1].n_providers == 31187
# allowed is the service-weighted mean across places of service
assert s[1].avg_allowed == pytest.approx(60.722041509, rel=1e-4)
assert [
r.hcpcs for r in read_series(con, ["99439", "99490"], years=[2022])
] == ["99439", "99490"]
def test_ensure_table_is_safe_to_repeat(self, con):
ensure_table(con)
ensure_table(con)
assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 0
class TestCite:
def test_one_source_per_year_with_tags(self, tmp_path):
s = Store(":memory:", storage_dir=tmp_path / "storage")
k1 = cite(s, 2023)
k2 = cite(s, 2023)
k3 = cite(s, 2022)
assert k1 == k2 and k1 != k3
item = s.get(k1)
assert "CY2023" in item.title and DATASETS[2023] in item.url
tags = set(item.tags)
assert {
"module:pfs",
"table:pfs.utilization",
"source:cms-website",
"year:2023",
} <= tags
s.close()
class TestCachePayload:
def test_records_round_trip_json(self, tmp_path):
from pfs.utilization import cache_path, load_cache, save_cache
p = cache_path(2023, root=tmp_path)
save_cache(p, [REC])
assert p.exists() and p.name == "mup-phy-geo-2023.json"
assert load_cache(p) == [REC]
assert json.loads(p.read_text())[0]["HCPCS_Cd"] == "99490"