One dataset per calendar year on data.cms.gov (2013–2024, UUIDs in pfs.utilization.DATASETS), pulled through the Data API filtered to national rows (~13k a year, one per HCPCS × place of service), cached under data/cms/utilization/, loaded delete-then-insert per year, each year cited in bib as a Source (module:pfs, table:pfs.utilization, source:cms-website, year:N) and logged in cms.ingest_log with the cache file's sha256. Suppressed cells (<11 beneficiaries) load as NULL. read_series() sums places of service and service-weights the average allowed/paid amounts. stack pfs utilization [--year N]... [--write [--offline]] [--code C]... Notebook 7d charts a family's services per year with the exposure dates from 7c as rules; 7e answers the three #694 questions (99490 2015–, G2058→99439 continuity, APCM 2025 vs CCM — no 2025 file yet). CLI docs regenerated (sidebar renumbered around the new page).
229 lines
7.3 KiB
Python
229 lines
7.3 KiB
Python
"""pfs.utilization — Medicare Physician & Other Practitioners by Geography
|
||
and Service (national totals per HCPCS × place of service), #694."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
|
||
import duckdb
|
||
import httpx
|
||
import pytest
|
||
|
||
from bib.store import Store
|
||
from pfs.utilization import (
|
||
API,
|
||
DATASETS,
|
||
UtilizationRow,
|
||
cite,
|
||
ensure_table,
|
||
fetch_year,
|
||
read_series,
|
||
to_rows,
|
||
write,
|
||
)
|
||
|
||
REC = {
|
||
"Rndrng_Prvdr_Geo_Lvl": "National",
|
||
"Rndrng_Prvdr_Geo_Cd": "",
|
||
"Rndrng_Prvdr_Geo_Desc": "National",
|
||
"HCPCS_Cd": "99490",
|
||
"HCPCS_Desc": "Chronic care management services, first 20 minutes",
|
||
"HCPCS_Drug_Ind": "N",
|
||
"Place_Of_Srvc": "O",
|
||
"Tot_Rndrng_Prvdrs": "31182",
|
||
"Tot_Benes": "1158246",
|
||
"Tot_Srvcs": "5715325",
|
||
"Tot_Bene_Day_Srvcs": "5676766",
|
||
"Avg_Sbmtd_Chrg": "106.87513302",
|
||
"Avg_Mdcr_Alowd_Amt": "60.722041509",
|
||
"Avg_Mdcr_Pymt_Amt": "46.196134143",
|
||
"Avg_Mdcr_Stdzd_Amt": "45.634579675",
|
||
}
|
||
|
||
|
||
class TestDatasets:
|
||
def test_every_year_2013_to_2024_has_a_dataset(self):
|
||
assert sorted(DATASETS) == list(range(2013, 2025))
|
||
assert all(len(v) == 36 for v in DATASETS.values())
|
||
|
||
|
||
class TestToRows:
|
||
def test_parses_numbers(self):
|
||
rows = to_rows(2023, [REC], dataset_id="d", item_key="K")
|
||
assert len(rows) == 1
|
||
r = rows[0]
|
||
assert isinstance(r, UtilizationRow)
|
||
assert (r.year, r.hcpcs, r.place_of_service, r.drug_ind) == (
|
||
2023,
|
||
"99490",
|
||
"O",
|
||
"N",
|
||
)
|
||
assert r.n_providers == 31182 and r.n_beneficiaries == 1158246
|
||
assert r.n_services == 5715325.0 and r.n_bene_day_services == 5676766.0
|
||
assert r.avg_allowed == pytest.approx(60.722041509)
|
||
assert r.avg_paid == pytest.approx(46.196134143)
|
||
assert r.avg_submitted == pytest.approx(106.87513302)
|
||
assert r.avg_standardized == pytest.approx(45.634579675)
|
||
assert r.dataset_id == "d" and r.item_key == "K"
|
||
|
||
def test_blanks_become_none_and_non_national_rows_are_skipped(self):
|
||
blank = dict(REC, Tot_Benes="", Avg_Mdcr_Pymt_Amt="")
|
||
state = dict(REC, Rndrng_Prvdr_Geo_Lvl="State", Rndrng_Prvdr_Geo_Desc="Ohio")
|
||
rows = to_rows(2023, [blank, state], dataset_id="d", item_key="K")
|
||
assert len(rows) == 1
|
||
assert rows[0].n_beneficiaries is None and rows[0].avg_paid is None
|
||
|
||
|
||
def _paged_client(pages: list[list[dict]], seen: list[str]):
|
||
def handler(request):
|
||
seen.append(str(request.url))
|
||
offset = int(request.url.params.get("offset", "0"))
|
||
size = int(request.url.params.get("size", "0"))
|
||
idx = offset // size if size else 0
|
||
body = pages[idx] if idx < len(pages) else []
|
||
return httpx.Response(200, json=body)
|
||
|
||
return httpx.Client(transport=httpx.MockTransport(handler))
|
||
|
||
|
||
class TestFetchYear:
|
||
def test_pages_until_a_short_page(self):
|
||
seen: list[str] = []
|
||
pages = [[REC] * 3, [REC] * 3, [REC]] # size 3 → stop after the short page
|
||
client = _paged_client(pages, seen)
|
||
out = fetch_year(client, 2023, page_size=3)
|
||
assert len(out) == 7
|
||
assert len(seen) == 3
|
||
assert seen[0].startswith(API.format(id=DATASETS[2023]))
|
||
assert "Rndrng_Prvdr_Geo_Lvl%5D=National" in seen[0]
|
||
assert "offset=3" in seen[1] and "offset=6" in seen[2]
|
||
|
||
def test_exact_multiple_stops_on_empty_page(self):
|
||
seen: list[str] = []
|
||
client = _paged_client([[REC] * 2, [REC] * 2, []], seen)
|
||
assert len(fetch_year(client, 2023, page_size=2)) == 4
|
||
assert len(seen) == 3
|
||
|
||
def test_unknown_year(self):
|
||
with pytest.raises(KeyError):
|
||
fetch_year(httpx.Client(), 1999)
|
||
|
||
def test_http_error_raises(self):
|
||
client = httpx.Client(
|
||
transport=httpx.MockTransport(lambda r: httpx.Response(503))
|
||
)
|
||
with pytest.raises(httpx.HTTPStatusError):
|
||
fetch_year(client, 2023)
|
||
|
||
|
||
@pytest.fixture
|
||
def con():
|
||
c = duckdb.connect(":memory:")
|
||
yield c
|
||
c.close()
|
||
|
||
|
||
class TestWriteAndRead:
|
||
def test_write_is_idempotent_per_year(self, con):
|
||
rows23 = to_rows(
|
||
2023,
|
||
[REC, dict(REC, Place_Of_Srvc="F", Tot_Srvcs="100")],
|
||
dataset_id="d",
|
||
item_key="K",
|
||
)
|
||
rows22 = to_rows(
|
||
2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J"
|
||
)
|
||
assert write(con, 2023, rows23) == 2
|
||
assert write(con, 2022, rows22) == 1
|
||
assert write(con, 2023, rows23) == 2 # replaces, never appends
|
||
assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 3
|
||
|
||
def test_read_series_sums_places_of_service(self, con):
|
||
write(
|
||
con,
|
||
2023,
|
||
to_rows(
|
||
2023,
|
||
[
|
||
REC,
|
||
dict(
|
||
REC,
|
||
Place_Of_Srvc="F",
|
||
Tot_Srvcs="100",
|
||
Tot_Benes="10",
|
||
Tot_Rndrng_Prvdrs="5",
|
||
),
|
||
],
|
||
dataset_id="d",
|
||
item_key="K",
|
||
),
|
||
)
|
||
write(
|
||
con,
|
||
2022,
|
||
to_rows(
|
||
2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J"
|
||
),
|
||
)
|
||
write(
|
||
con,
|
||
2022,
|
||
to_rows(
|
||
2022,
|
||
[
|
||
dict(REC, Tot_Srvcs="4000000"),
|
||
dict(REC, HCPCS_Cd="99439", Tot_Srvcs="7"),
|
||
],
|
||
dataset_id="e",
|
||
item_key="J",
|
||
),
|
||
)
|
||
s = read_series(con, ["99490"])
|
||
assert [(r.year, r.n_services) for r in s] == [
|
||
(2022, 4000000.0),
|
||
(2023, 5715425.0),
|
||
]
|
||
assert s[1].n_beneficiaries == 1158256 and s[1].n_providers == 31187
|
||
# allowed is the service-weighted mean across places of service
|
||
assert s[1].avg_allowed == pytest.approx(60.722041509, rel=1e-4)
|
||
assert [
|
||
r.hcpcs for r in read_series(con, ["99439", "99490"], years=[2022])
|
||
] == ["99439", "99490"]
|
||
|
||
def test_ensure_table_is_safe_to_repeat(self, con):
|
||
ensure_table(con)
|
||
ensure_table(con)
|
||
assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 0
|
||
|
||
|
||
class TestCite:
|
||
def test_one_source_per_year_with_tags(self, tmp_path):
|
||
s = Store(":memory:", storage_dir=tmp_path / "storage")
|
||
k1 = cite(s, 2023)
|
||
k2 = cite(s, 2023)
|
||
k3 = cite(s, 2022)
|
||
assert k1 == k2 and k1 != k3
|
||
item = s.get(k1)
|
||
assert "CY2023" in item.title and DATASETS[2023] in item.url
|
||
tags = set(item.tags)
|
||
assert {
|
||
"module:pfs",
|
||
"table:pfs.utilization",
|
||
"source:cms-website",
|
||
"year:2023",
|
||
} <= tags
|
||
s.close()
|
||
|
||
|
||
class TestCachePayload:
|
||
def test_records_round_trip_json(self, tmp_path):
|
||
from pfs.utilization import cache_path, load_cache, save_cache
|
||
|
||
p = cache_path(2023, root=tmp_path)
|
||
save_cache(p, [REC])
|
||
assert p.exists() and p.name == "mup-phy-geo-2023.json"
|
||
assert load_cache(p) == [REC]
|
||
assert json.loads(p.read_text())[0]["HCPCS_Cd"] == "99490"
|