From 4c69d5be9ba55664a6f40c63d3be5be83c329958 Mon Sep 17 00:00:00 2001 From: kert Date: Tue, 22 Sep 2026 16:04:41 -0400 Subject: [PATCH] =?UTF-8?q?feat(pfs):=20Part=20B=20utilization=20per=20HCP?= =?UTF-8?q?CS=20=E2=80=94=20pfs.utilization=20from=20the=20CMS=20by-Geogra?= =?UTF-8?q?phy-and-Service=20PUFs,=20stack=20pfs=20utilization,=20notebook?= =?UTF-8?q?=207d/7e=20(refs=20#694)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One dataset per calendar year on data.cms.gov (2013–2024, UUIDs in pfs.utilization.DATASETS), pulled through the Data API filtered to national rows (~13k a year, one per HCPCS × place of service), cached under data/cms/utilization/, loaded delete-then-insert per year, each year cited in bib as a Source (module:pfs, table:pfs.utilization, source:cms-website, year:N) and logged in cms.ingest_log with the cache file's sha256. Suppressed cells (<11 beneficiaries) load as NULL. read_series() sums places of service and service-weights the average allowed/paid amounts. stack pfs utilization [--year N]... [--write [--offline]] [--code C]... Notebook 7d charts a family's services per year with the exposure dates from 7c as rules; 7e answers the three #694 questions (99490 2015–, G2058→99439 continuity, APCM 2025 vs CCM — no 2025 file yet). CLI docs regenerated (sidebar renumbered around the new page). --- docs/docs/cli/mail-attach-smarthost.md | 2 +- docs/docs/cli/mail-dkim-export.md | 2 +- docs/docs/cli/mail-dns.md | 2 +- docs/docs/cli/mail-down.md | 2 +- docs/docs/cli/mail-provision.md | 2 +- docs/docs/cli/mail-rotate-creds.md | 2 +- docs/docs/cli/mail-seed-mailboxes.md | 2 +- docs/docs/cli/mail-status.md | 2 +- docs/docs/cli/mail-up.md | 2 +- docs/docs/cli/mail-wire-git.md | 2 +- docs/docs/cli/mail.md | 2 +- docs/docs/cli/pfs-cpt-ingest.md | 2 +- docs/docs/cli/pfs-review.md | 2 +- docs/docs/cli/pfs-utilization.md | 31 +++ docs/docs/cli/pfs.md | 75 +++--- docs/docs/cli/prisma-eligible.md | 2 +- docs/docs/cli/prisma-export.md | 2 +- docs/docs/cli/prisma-extract.md | 2 +- docs/docs/cli/prisma-fetch.md | 2 +- docs/docs/cli/prisma-flow.md | 2 +- docs/docs/cli/prisma-init.md | 2 +- docs/docs/cli/prisma-ping-llm.md | 2 +- docs/docs/cli/prisma-run.md | 2 +- docs/docs/cli/prisma-screen.md | 2 +- docs/docs/cli/prisma-vpn.md | 2 +- docs/docs/cli/prisma.md | 2 +- docs/docs/cli/zot-dump-schema.md | 2 +- docs/docs/cli/zot-fix-dates.md | 2 +- docs/docs/cli/zot-fix-fields.md | 2 +- docs/docs/cli/zot-fix-keys.md | 2 +- docs/docs/cli/zot-verify-parity.md | 2 +- docs/docs/cli/zot.md | 2 +- notebooks/code_families.py | 179 +++++++++++++- src/cli/pfs.py | 121 +++++++++ src/pfs/table/__init__.py | 1 + src/pfs/table/utilization.py | 65 +++++ src/pfs/utilization.py | 299 +++++++++++++++++++++++ tests/cli/test_pfs_cli.py | 103 ++++++++ tests/notebooks/test_code_families_nb.py | 16 +- tests/pfs/test_utilization.py | 228 +++++++++++++++++ 40 files changed, 1113 insertions(+), 65 deletions(-) create mode 100644 docs/docs/cli/pfs-utilization.md create mode 100644 src/pfs/table/utilization.py create mode 100644 src/pfs/utilization.py create mode 100644 tests/pfs/test_utilization.py diff --git a/docs/docs/cli/mail-attach-smarthost.md b/docs/docs/cli/mail-attach-smarthost.md index 62e0fb1..d751453 100644 --- a/docs/docs/cli/mail-attach-smarthost.md +++ b/docs/docs/cli/mail-attach-smarthost.md @@ -1,6 +1,6 @@ --- title: stack mail attach-smarthost -sidebar_position: 98 +sidebar_position: 99 --- # `stack mail attach-smarthost` diff --git a/docs/docs/cli/mail-dkim-export.md b/docs/docs/cli/mail-dkim-export.md index 57a8e61..4a14cd7 100644 --- a/docs/docs/cli/mail-dkim-export.md +++ b/docs/docs/cli/mail-dkim-export.md @@ -1,6 +1,6 @@ --- title: stack mail dkim-export -sidebar_position: 97 +sidebar_position: 98 --- # `stack mail dkim-export` diff --git a/docs/docs/cli/mail-dns.md b/docs/docs/cli/mail-dns.md index 4d1fa73..f0d7133 100644 --- a/docs/docs/cli/mail-dns.md +++ b/docs/docs/cli/mail-dns.md @@ -1,6 +1,6 @@ --- title: stack mail dns -sidebar_position: 96 +sidebar_position: 97 --- # `stack mail dns` diff --git a/docs/docs/cli/mail-down.md b/docs/docs/cli/mail-down.md index 5eb6aad..625a7d7 100644 --- a/docs/docs/cli/mail-down.md +++ b/docs/docs/cli/mail-down.md @@ -1,6 +1,6 @@ --- title: stack mail down -sidebar_position: 94 +sidebar_position: 95 --- # `stack mail down` diff --git a/docs/docs/cli/mail-provision.md b/docs/docs/cli/mail-provision.md index 9b1f22e..6584634 100644 --- a/docs/docs/cli/mail-provision.md +++ b/docs/docs/cli/mail-provision.md @@ -1,6 +1,6 @@ --- title: stack mail provision -sidebar_position: 92 +sidebar_position: 93 --- # `stack mail provision` diff --git a/docs/docs/cli/mail-rotate-creds.md b/docs/docs/cli/mail-rotate-creds.md index e40a094..9cb669d 100644 --- a/docs/docs/cli/mail-rotate-creds.md +++ b/docs/docs/cli/mail-rotate-creds.md @@ -1,6 +1,6 @@ --- title: stack mail rotate-creds -sidebar_position: 99 +sidebar_position: 100 --- # `stack mail rotate-creds` diff --git a/docs/docs/cli/mail-seed-mailboxes.md b/docs/docs/cli/mail-seed-mailboxes.md index aaaa92c..4d97b10 100644 --- a/docs/docs/cli/mail-seed-mailboxes.md +++ b/docs/docs/cli/mail-seed-mailboxes.md @@ -1,6 +1,6 @@ --- title: stack mail seed-mailboxes -sidebar_position: 100 +sidebar_position: 101 --- # `stack mail seed-mailboxes` diff --git a/docs/docs/cli/mail-status.md b/docs/docs/cli/mail-status.md index ddc85a2..f1f5da7 100644 --- a/docs/docs/cli/mail-status.md +++ b/docs/docs/cli/mail-status.md @@ -1,6 +1,6 @@ --- title: stack mail status -sidebar_position: 95 +sidebar_position: 96 --- # `stack mail status` diff --git a/docs/docs/cli/mail-up.md b/docs/docs/cli/mail-up.md index 725f56a..ee80616 100644 --- a/docs/docs/cli/mail-up.md +++ b/docs/docs/cli/mail-up.md @@ -1,6 +1,6 @@ --- title: stack mail up -sidebar_position: 93 +sidebar_position: 94 --- # `stack mail up` diff --git a/docs/docs/cli/mail-wire-git.md b/docs/docs/cli/mail-wire-git.md index bafcbe0..1283b5f 100644 --- a/docs/docs/cli/mail-wire-git.md +++ b/docs/docs/cli/mail-wire-git.md @@ -1,6 +1,6 @@ --- title: stack mail wire-git -sidebar_position: 101 +sidebar_position: 102 --- # `stack mail wire-git` diff --git a/docs/docs/cli/mail.md b/docs/docs/cli/mail.md index 79af001..937ef89 100644 --- a/docs/docs/cli/mail.md +++ b/docs/docs/cli/mail.md @@ -1,6 +1,6 @@ --- title: stack mail -sidebar_position: 91 +sidebar_position: 92 --- # `stack mail` diff --git a/docs/docs/cli/pfs-cpt-ingest.md b/docs/docs/cli/pfs-cpt-ingest.md index 1c8cffd..4a64917 100644 --- a/docs/docs/cli/pfs-cpt-ingest.md +++ b/docs/docs/cli/pfs-cpt-ingest.md @@ -1,6 +1,6 @@ --- title: stack pfs cpt-ingest -sidebar_position: 73 +sidebar_position: 74 --- # `stack pfs cpt-ingest` diff --git a/docs/docs/cli/pfs-review.md b/docs/docs/cli/pfs-review.md index 0e5a1b0..b3af1c9 100644 --- a/docs/docs/cli/pfs-review.md +++ b/docs/docs/cli/pfs-review.md @@ -1,6 +1,6 @@ --- title: stack pfs review -sidebar_position: 72 +sidebar_position: 73 --- # `stack pfs review` diff --git a/docs/docs/cli/pfs-utilization.md b/docs/docs/cli/pfs-utilization.md new file mode 100644 index 0000000..a43adef --- /dev/null +++ b/docs/docs/cli/pfs-utilization.md @@ -0,0 +1,31 @@ +--- +title: stack pfs utilization +sidebar_position: 72 +--- + +# `stack pfs utilization` + +``` +Usage: stack pfs utilization [OPTIONS] + + Part B utilization per HCPCS (pfs.utilization, #694): the CMS "Medicare + Physician & Other Practitioners — by Geography and Service" national totals, + one row per (year, code, place of service), with each year's dataset cited in + bib and logged in cms.ingest_log. + + --write pulls every year (or --year) through the Data API, caches the + raw pages under data/cms/utilization/, replaces that year's rows and + republishes the replica; --code prints series from the replica. + +╭─ Options ────────────────────────────────────────────────────────────────────╮ +│ --year INTEGER Calendar year(s) to ingest with --write; default │ +│ all 2013–. │ +│ --code TEXT Print the per-year series for these codes │ +│ (repeatable). │ +│ --write Fetch from data.cms.gov, load pfs.utilization, │ +│ cite, republish. │ +│ --offline With --write: load from the cached JSON instead of │ +│ the API. │ +│ --help Show this message and exit. │ +╰──────────────────────────────────────────────────────────────────────────────╯ +``` diff --git a/docs/docs/cli/pfs.md b/docs/docs/cli/pfs.md index 59badba..40f8511 100644 --- a/docs/docs/cli/pfs.md +++ b/docs/docs/cli/pfs.md @@ -14,38 +14,47 @@ Usage: stack pfs [OPTIONS] COMMAND [ARGS]... │ --help Show this message and exit. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ╭─ Commands ───────────────────────────────────────────────────────────────────╮ -│ elements Extract typed elements for codes into pfs.code_element (+ review │ -│ queue). │ -│ lineage Timeline of a code: RVU-file diffs and FR paragraphs, │ -│ cross-checked. │ -│ families Derive families from pfs.code_element / pfs.code_event / │ -│ pfs.rvu. │ -│ guidance CFR/IOM/MLN sub-regulatory crosswalk per code family │ -│ (pfs.code_guidance): which CFR sections and IOM manual chapters │ -│ the │ -│ family's FR paragraphs and the CPT manual cite, with FR │ -│ provenance. │ -│ reaction Public-reaction series per code family (pfs.code_reaction): │ -│ comment │ -│ counts per docket (share of the docket's commenters mentioning │ -│ the │ -│ family), an optional stance sample, and the FR │ -│ Comment:/Response: │ -│ pair counts per rule as the pre-2017 proxy. Always recomputed │ -│ live │ -│ from pgvector/bib — there is no persisted-table fallback to read │ -│ without --write, so --write only adds persisting the result. │ -│ exposure The exposure calendar (pfs.code_exposure, #693): when each code │ -│ became payable, was revalued, changed payable status, ended, or │ -│ was │ -│ listed for telehealth — dated 1 January of the rule year (or the │ -│ correction notice's date), anchored to the FR paragraph from its │ -│ lineage, with the status and RVU-decile band a control set │ -│ matches on. │ -│ review Element lines the classifier could not place. │ -│ cpt-ingest Parse CPT EPUB editions into │ -│ pfs.cpt_section/cpt_code/cpt_instruction/ │ -│ cpt_reference/cpt_crosswalk/cpt_list (docs/superpowers/specs/ │ -│ 2026-09-09-cpt-canonical-schema-design.md §3). │ +│ elements Extract typed elements for codes into pfs.code_element (+ │ +│ review queue). │ +│ lineage Timeline of a code: RVU-file diffs and FR paragraphs, │ +│ cross-checked. │ +│ families Derive families from pfs.code_element / pfs.code_event / │ +│ pfs.rvu. │ +│ guidance CFR/IOM/MLN sub-regulatory crosswalk per code family │ +│ (pfs.code_guidance): which CFR sections and IOM manual chapters │ +│ the │ +│ family's FR paragraphs and the CPT manual cite, with FR │ +│ provenance. │ +│ reaction Public-reaction series per code family (pfs.code_reaction): │ +│ comment │ +│ counts per docket (share of the docket's commenters mentioning │ +│ the │ +│ family), an optional stance sample, and the FR │ +│ Comment:/Response: │ +│ pair counts per rule as the pre-2017 proxy. Always recomputed │ +│ live │ +│ from pgvector/bib — there is no persisted-table fallback to │ +│ read │ +│ without --write, so --write only adds persisting the result. │ +│ exposure The exposure calendar (pfs.code_exposure, #693): when each code │ +│ became payable, was revalued, changed payable status, ended, or │ +│ was │ +│ listed for telehealth — dated 1 January of the rule year (or │ +│ the │ +│ correction notice's date), anchored to the FR paragraph from │ +│ its │ +│ lineage, with the status and RVU-decile band a control set │ +│ matches on. │ +│ utilization Part B utilization per HCPCS (pfs.utilization, #694): the CMS │ +│ "Medicare Physician & Other Practitioners — by Geography and │ +│ Service" │ +│ national totals, one row per (year, code, place of service), │ +│ with │ +│ each year's dataset cited in bib and logged in cms.ingest_log. │ +│ review Element lines the classifier could not place. │ +│ cpt-ingest Parse CPT EPUB editions into │ +│ pfs.cpt_section/cpt_code/cpt_instruction/ │ +│ cpt_reference/cpt_crosswalk/cpt_list (docs/superpowers/specs/ │ +│ 2026-09-09-cpt-canonical-schema-design.md §3). │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/docs/docs/cli/prisma-eligible.md b/docs/docs/cli/prisma-eligible.md index 86a021b..5124633 100644 --- a/docs/docs/cli/prisma-eligible.md +++ b/docs/docs/cli/prisma-eligible.md @@ -1,6 +1,6 @@ --- title: stack prisma eligible -sidebar_position: 85 +sidebar_position: 86 --- # `stack prisma eligible` diff --git a/docs/docs/cli/prisma-export.md b/docs/docs/cli/prisma-export.md index 53bdabe..afc5cba 100644 --- a/docs/docs/cli/prisma-export.md +++ b/docs/docs/cli/prisma-export.md @@ -1,6 +1,6 @@ --- title: stack prisma export -sidebar_position: 82 +sidebar_position: 83 --- # `stack prisma export` diff --git a/docs/docs/cli/prisma-extract.md b/docs/docs/cli/prisma-extract.md index 16142f2..634b2d1 100644 --- a/docs/docs/cli/prisma-extract.md +++ b/docs/docs/cli/prisma-extract.md @@ -1,6 +1,6 @@ --- title: stack prisma extract -sidebar_position: 86 +sidebar_position: 87 --- # `stack prisma extract` diff --git a/docs/docs/cli/prisma-fetch.md b/docs/docs/cli/prisma-fetch.md index 5585803..34a0082 100644 --- a/docs/docs/cli/prisma-fetch.md +++ b/docs/docs/cli/prisma-fetch.md @@ -1,6 +1,6 @@ --- title: stack prisma fetch -sidebar_position: 88 +sidebar_position: 89 --- # `stack prisma fetch` diff --git a/docs/docs/cli/prisma-flow.md b/docs/docs/cli/prisma-flow.md index a1bac12..bf44d50 100644 --- a/docs/docs/cli/prisma-flow.md +++ b/docs/docs/cli/prisma-flow.md @@ -1,6 +1,6 @@ --- title: stack prisma flow -sidebar_position: 87 +sidebar_position: 88 --- # `stack prisma flow` diff --git a/docs/docs/cli/prisma-init.md b/docs/docs/cli/prisma-init.md index d0f9feb..5e3bf18 100644 --- a/docs/docs/cli/prisma-init.md +++ b/docs/docs/cli/prisma-init.md @@ -1,6 +1,6 @@ --- title: stack prisma init -sidebar_position: 81 +sidebar_position: 82 --- # `stack prisma init` diff --git a/docs/docs/cli/prisma-ping-llm.md b/docs/docs/cli/prisma-ping-llm.md index 0e8f655..1d4f026 100644 --- a/docs/docs/cli/prisma-ping-llm.md +++ b/docs/docs/cli/prisma-ping-llm.md @@ -1,6 +1,6 @@ --- title: stack prisma ping-llm -sidebar_position: 83 +sidebar_position: 84 --- # `stack prisma ping-llm` diff --git a/docs/docs/cli/prisma-run.md b/docs/docs/cli/prisma-run.md index 1c9ba9d..93ec9c6 100644 --- a/docs/docs/cli/prisma-run.md +++ b/docs/docs/cli/prisma-run.md @@ -1,6 +1,6 @@ --- title: stack prisma run -sidebar_position: 89 +sidebar_position: 90 --- # `stack prisma run` diff --git a/docs/docs/cli/prisma-screen.md b/docs/docs/cli/prisma-screen.md index 206f6ce..caa02d3 100644 --- a/docs/docs/cli/prisma-screen.md +++ b/docs/docs/cli/prisma-screen.md @@ -1,6 +1,6 @@ --- title: stack prisma screen -sidebar_position: 84 +sidebar_position: 85 --- # `stack prisma screen` diff --git a/docs/docs/cli/prisma-vpn.md b/docs/docs/cli/prisma-vpn.md index c6edb1e..d399839 100644 --- a/docs/docs/cli/prisma-vpn.md +++ b/docs/docs/cli/prisma-vpn.md @@ -1,6 +1,6 @@ --- title: stack prisma vpn -sidebar_position: 90 +sidebar_position: 91 --- # `stack prisma vpn` diff --git a/docs/docs/cli/prisma.md b/docs/docs/cli/prisma.md index 98ac801..cffa0c6 100644 --- a/docs/docs/cli/prisma.md +++ b/docs/docs/cli/prisma.md @@ -1,6 +1,6 @@ --- title: stack prisma -sidebar_position: 80 +sidebar_position: 81 --- # `stack prisma` diff --git a/docs/docs/cli/zot-dump-schema.md b/docs/docs/cli/zot-dump-schema.md index 5d872a8..0c96835 100644 --- a/docs/docs/cli/zot-dump-schema.md +++ b/docs/docs/cli/zot-dump-schema.md @@ -1,6 +1,6 @@ --- title: stack zot dump-schema -sidebar_position: 75 +sidebar_position: 76 --- # `stack zot dump-schema` diff --git a/docs/docs/cli/zot-fix-dates.md b/docs/docs/cli/zot-fix-dates.md index 1bb7067..e9dfa33 100644 --- a/docs/docs/cli/zot-fix-dates.md +++ b/docs/docs/cli/zot-fix-dates.md @@ -1,6 +1,6 @@ --- title: stack zot fix-dates -sidebar_position: 76 +sidebar_position: 77 --- # `stack zot fix-dates` diff --git a/docs/docs/cli/zot-fix-fields.md b/docs/docs/cli/zot-fix-fields.md index b55b051..ac02721 100644 --- a/docs/docs/cli/zot-fix-fields.md +++ b/docs/docs/cli/zot-fix-fields.md @@ -1,6 +1,6 @@ --- title: stack zot fix-fields -sidebar_position: 78 +sidebar_position: 79 --- # `stack zot fix-fields` diff --git a/docs/docs/cli/zot-fix-keys.md b/docs/docs/cli/zot-fix-keys.md index cd10bc5..8367c95 100644 --- a/docs/docs/cli/zot-fix-keys.md +++ b/docs/docs/cli/zot-fix-keys.md @@ -1,6 +1,6 @@ --- title: stack zot fix-keys -sidebar_position: 77 +sidebar_position: 78 --- # `stack zot fix-keys` diff --git a/docs/docs/cli/zot-verify-parity.md b/docs/docs/cli/zot-verify-parity.md index 0924b84..4bf6034 100644 --- a/docs/docs/cli/zot-verify-parity.md +++ b/docs/docs/cli/zot-verify-parity.md @@ -1,6 +1,6 @@ --- title: stack zot verify-parity -sidebar_position: 79 +sidebar_position: 80 --- # `stack zot verify-parity` diff --git a/docs/docs/cli/zot.md b/docs/docs/cli/zot.md index 4ca8eb0..99a4258 100644 --- a/docs/docs/cli/zot.md +++ b/docs/docs/cli/zot.md @@ -1,6 +1,6 @@ --- title: stack zot -sidebar_position: 74 +sidebar_position: 75 --- # `stack zot` diff --git a/notebooks/code_families.py b/notebooks/code_families.py index 2cc9338..eab32b2 100644 --- a/notebooks/code_families.py +++ b/notebooks/code_families.py @@ -1127,6 +1127,182 @@ def _(code, con, fr_md, mo, not_built, pl): return +@app.cell(hide_code=True) +def _(alt, code, con, mo, not_built, pl): + # ── 7d. Utilization ── + from pfs.codetables import read_exposures as _read_exposures_u + from pfs.families import family_of as _family_of_u + from pfs.utilization import read_series as _read_series + + _fam = _family_of_u(code) + _key = _fam.key if _fam else code + _codes = list(_fam.codes) if _fam else [code] + + _series = [] + if con is not None: + try: + _series = _read_series(con, _codes) + except Exception: # noqa: BLE001 — a missing table is "not built yet" + _series = [] + + _intro = ( + "## 7d. Utilization\n\n" + "The first outcome series P50 joins to the exposure calendar: services " + "billed, beneficiaries and rendering providers per code and year from the " + "CMS *Medicare Physician & Other Practitioners — by Geography and Service* " + "public use files (national totals, 2013 on; `pfs.utilization`, one bib " + "Source item per year). Counts are summed across facility and office " + "settings; a provider billing in both is counted twice, because the file " + "gives no cross-setting distinct count. Beneficiary counts under 11 are " + "suppressed by CMS and appear as gaps, not zeros. Dashed rules mark the " + "family's exposure dates from section 7c." + ) + if not _series: + _view = mo.md(_intro + "\n\n" + not_built("stack pfs utilization --write")) + else: + _df = pl.DataFrame( + { + "code": [r.hcpcs for r in _series], + "year": [r.year for r in _series], + "services": [r.n_services for r in _series], + "beneficiaries": [r.n_beneficiaries for r in _series], + "providers": [r.n_providers for r in _series], + "avg allowed": [r.avg_allowed for r in _series], + } + ) + try: + _exp = ( + _read_exposures_u(con, family=_key) + if _fam + else _read_exposures_u(con, code=code) + ) + except Exception: # noqa: BLE001 + _exp = [] + _rules = pl.DataFrame( + { + "year": [r.year for r in _exp if r.kind in ("becomes-payable", "ends")], + "exposure": [ + f"{r.code} {r.kind}" + for r in _exp + if r.kind in ("becomes-payable", "ends") + ], + } + ) + _chart = ( + alt.Chart(_df.to_pandas()) + .mark_line(point=True) + .encode( + x=alt.X("year:O", title="Calendar year"), + y=alt.Y("services:Q", title="Services billed"), + color=alt.Color("code:N", title="Code"), + tooltip=[ + "code", + "year", + "services", + "beneficiaries", + "providers", + "avg allowed", + ], + ) + .properties(width=640, height=300, title=f"{_key}: services per year") + ) + if len(_rules): + _chart = _chart + ( + alt.Chart(_rules.to_pandas()) + .mark_rule(strokeDash=[4, 4], color="gray") + .encode(x="year:O", tooltip=["exposure"]) + ) + _latest = max(r.year for r in _series) + _view = mo.vstack( + [ + mo.md( + _intro + f"\n\n**{_key}**: {len(_df)} (code, year) rows, " + f"latest year on file {_latest}." + ), + mo.ui.altair_chart(_chart), + mo.ui.table(_df, label=f"pfs.utilization — {_key}"), + ] + ) + _view + return + + +@app.cell(hide_code=True) +def _(con, mo, pl): + # ── 7e. The three P50 questions ── + from pfs.utilization import read_series as _read_series_q + + _q = [] + if con is not None: + try: + _q = _read_series_q( + con, ["99490", "G2058", "99439", "G0556", "G0557", "G0558"] + ) + except Exception: # noqa: BLE001 + _q = [] + _intro = mo.md( + "## 7e. The three P50 questions\n\n" + "#694 asks the utilization table three things before anything else: " + "how many CCM services (99490) were billed each year from 2015; whether " + "the G2058 → 99439 add-on carried over across the 2021 code change; and " + "how APCM (G0556–G0558) uptake in 2025 compares with CCM. The first two " + "read straight off the table. The third waits on CMS: the newest " + "Geography-and-Service file covers services through calendar 2024, so " + "APCM — payable from 2025-01-01 — has no rows yet; the panel below " + "states that rather than plotting zeros." + ) + if not _q: + _view = mo.vstack( + [_intro, mo.md("_Not built yet — run `stack pfs utilization --write`._")] + ) + else: + _by = {} + for r in _q: + _by.setdefault(r.hcpcs, {})[r.year] = r + _ccm = _by.get("99490", {}) + _t1 = pl.DataFrame( + { + "year": sorted(_ccm), + "99490 services": [_ccm[y].n_services for y in sorted(_ccm)], + "beneficiaries": [_ccm[y].n_beneficiaries for y in sorted(_ccm)], + } + ) + _g, _n = _by.get("G2058", {}), _by.get("99439", {}) + _t2 = pl.DataFrame( + { + "year": [2020, 2021], + "G2058 services": [ + _g.get(2020) and _g[2020].n_services, + _g.get(2021) and _g[2021].n_services, + ], + "99439 services": [ + _n.get(2020) and _n[2020].n_services, + _n.get(2021) and _n[2021].n_services, + ], + } + ) + _apcm_years = sorted( + {y for c in ("G0556", "G0557", "G0558") for y in _by.get(c, {})} + ) + _t3 = ( + f"APCM rows on file: {', '.join(map(str, _apcm_years))}." + if _apcm_years + else "No APCM (G0556–G0558) rows yet — the 2025 file has not been published." + ) + _view = mo.vstack( + [ + _intro, + mo.md("**1. CCM (99490) services per year**"), + mo.ui.table(_t1, label="99490 by year"), + mo.md("**2. G2058 → 99439 continuity, 2020–2021**"), + mo.ui.table(_t2, label="add-on continuity"), + mo.md(f"**3. APCM uptake vs CCM** — {_t3}"), + ] + ) + _view + return + + @app.cell(hide_code=True) def _(NOTES, REPLICA_PATH, con, mo, pl, q, store): # ── 8. Provenance ── @@ -1144,7 +1320,8 @@ def _(NOTES, REPLICA_PATH, con, mo, pl, q, store): "UNION ALL SELECT 'code_element_review', count(*) FROM pfs.code_element_review " "UNION ALL SELECT 'code_event', count(*) FROM pfs.code_event " "UNION ALL SELECT 'code_family', count(*) FROM pfs.code_family " - "UNION ALL SELECT 'code_exposure', count(*) FROM pfs.code_exposure" + "UNION ALL SELECT 'code_exposure', count(*) FROM pfs.code_exposure " + "UNION ALL SELECT 'utilization', count(*) FROM pfs.utilization" ) _log = q( "SELECT run_id, ingested_at, module, table_name, rule_id, source_file, sha256, rows, " diff --git a/src/cli/pfs.py b/src/cli/pfs.py index 8c8b7f0..09c9534 100644 --- a/src/cli/pfs.py +++ b/src/cli/pfs.py @@ -56,6 +56,15 @@ from pfs.families import ( from pfs.guidance import build as build_guidance from pfs.lineage import lineage, lineage_all from pfs.reaction import series as reaction_series +from pfs.utilization import DATASETS as UTILIZATION_YEARS +from pfs.utilization import cache_path as utilization_cache_path +from pfs.utilization import cite as cite_utilization +from pfs.utilization import fetch_year as fetch_utilization_year +from pfs.utilization import load_cache as load_utilization_cache +from pfs.utilization import read_series as read_utilization_series +from pfs.utilization import save_cache as save_utilization_cache +from pfs.utilization import to_rows as utilization_rows +from pfs.utilization import write as write_utilization log = logging.getLogger(__name__) @@ -771,6 +780,118 @@ def exposure( con.close() +def _http() -> Any: + import httpx + + return httpx.Client(headers={"User-Agent": "stack-pfs-utilization/1.0"}) + + +def _print_series(rows: list[Any]) -> None: + for r in rows: + svc = f"{r.n_services:,.0f}" if r.n_services is not None else "—" + ben = f"{r.n_beneficiaries:,}" if r.n_beneficiaries is not None else "—" + prv = f"{r.n_providers:,}" if r.n_providers is not None else "—" + allowed = f"${r.avg_allowed:,.2f}" if r.avg_allowed is not None else "—" + typer.echo( + f"{r.hcpcs} {r.year} services {svc:>12} benes {ben:>10} " + f"providers {prv:>8} avg allowed {allowed}" + ) + typer.echo(f"utilization: {len(rows)} (code, year) rows") + + +@app.command() +def utilization( + year: list[int] = typer.Option( + [], "--year", help="Calendar year(s) to ingest with --write; default all 2013–." + ), + code: list[str] = typer.Option( + [], "--code", help="Print the per-year series for these codes (repeatable)." + ), + write: bool = typer.Option( + False, + "--write", + help="Fetch from data.cms.gov, load pfs.utilization, cite, republish.", + ), + offline: bool = typer.Option( + False, + "--offline", + help="With --write: load from the cached JSON instead of the API.", + ), +) -> None: + """Part B utilization per HCPCS (pfs.utilization, #694): the CMS + "Medicare Physician & Other Practitioners — by Geography and Service" + national totals, one row per (year, code, place of service), with + each year's dataset cited in bib and logged in cms.ingest_log. + + --write pulls every year (or --year) through the Data API, caches the + raw pages under data/cms/utilization/, replaces that year's rows and + republishes the replica; --code prints series from the replica.""" + codes = [c.upper() for c in code] + if write: + years = sorted(year) if year else sorted(UTILIZATION_YEARS) + unknown = [y for y in years if y not in UTILIZATION_YEARS] + if unknown: + raise typer.BadParameter( + f"no dataset for {unknown}; known years " + f"{min(UTILIZATION_YEARS)}–{max(UTILIZATION_YEARS)}" + ) + # Ruling A13: all network + bib work first, then one short batch + # that only writes and publishes — never the write lock a + # notebook may be holding (#508-#514) while the API is paged. + store = _store() + staged: list[tuple[int, list[Any], str, str]] = [] + client = None if offline else _http() + try: + for y in years: + cache = utilization_cache_path(y) + if offline: + records = load_utilization_cache(cache) + else: + records = fetch_utilization_year(client, y) + save_utilization_cache(cache, records) + key = cite_utilization(store, y) + rows = utilization_rows( + y, records, dataset_id=UTILIZATION_YEARS[y], item_key=key + ) + staged.append((y, rows, key, str(cache))) + typer.echo( + f"{y}: {len(records)} records → {len(rows)} national rows ← {key}" + ) + finally: + if client is not None: + client.close() + from cms.ingest_log import log_ingest + + with _batch() as con: + for y, rows, key, cache in staged: + n = write_utilization(con, y, rows) + log_ingest( + con, + module="pfs.utilization", + table_name="pfs.utilization", + rows=n, + source_file=cache, + pincite_key=key, + ) + _publish() + if not codes: + return + con = _read() + try: + if not codes: + raise typer.BadParameter("pass --code (repeatable) or --write") + try: + rows = read_utilization_series(con, codes) + except Exception as exc: + if _missing_table(exc): + typer.echo("no utilization table yet — run with --write") + return + raise + _print_series(rows) + finally: + con.close() + + @app.command() def review(code: str = typer.Option("", "--code")) -> None: """Element lines the classifier could not place.""" diff --git a/src/pfs/table/__init__.py b/src/pfs/table/__init__.py index 972f4a3..74eaabf 100644 --- a/src/pfs/table/__init__.py +++ b/src/pfs/table/__init__.py @@ -11,4 +11,5 @@ from .gpci import Gpci as Gpci from .labor import ClinicalLabor as ClinicalLabor from .rvu import Rvu as Rvu from .supply import MedicalSupply as MedicalSupply +from .utilization import Utilization as Utilization from .work_time import PhysicianWorkTime as PhysicianWorkTime diff --git a/src/pfs/table/utilization.py b/src/pfs/table/utilization.py new file mode 100644 index 0000000..f5798d0 --- /dev/null +++ b/src/pfs/table/utilization.py @@ -0,0 +1,65 @@ +"""Medicare Physician & Other Practitioners — by Geography and Service, +national totals per HCPCS × place of service (#694). + +One row per (year, hcpcs, place_of_service). Source: the CMS Provider +Summary by Type of Service datasets on data.cms.gov (annual, 2013–), +pulled through the Data API filtered to ``Rndrng_Prvdr_Geo_Lvl = +National`` by ``pfs.utilization``. +""" + +from __future__ import annotations + +from conf.table_base import SQLTable + + +class Utilization(SQLTable): + """National Part B utilization and payment per HCPCS code and place + of service, one calendar year per row set.""" + + __schema__ = "pfs" + __tablename__ = "utilization" + + year: int | None = None + """Calendar year of service.""" + + hcpcs: str | None = None + """HCPCS/CPT code.""" + + hcpcs_desc: str | None = None + """CMS's long descriptor for the code in that year's file.""" + + drug_ind: str | None = None + """``Y`` when the code is a drug (ASP-priced), else ``N``.""" + + place_of_service: str | None = None + """``F`` facility or ``O`` non-facility (office).""" + + n_providers: int | None = None + """Distinct rendering providers billing the code (``Tot_Rndrng_Prvdrs``).""" + + n_beneficiaries: int | None = None + """Distinct beneficiaries (``Tot_Benes``); suppressed as NULL when < 11.""" + + n_services: float | None = None + """Services billed (``Tot_Srvcs``) — fractional for some drug units.""" + + n_bene_day_services: float | None = None + """Distinct beneficiary-days (``Tot_Bene_Day_Srvcs``).""" + + avg_submitted: float | None = None + """Average submitted charge (``Avg_Sbmtd_Chrg``).""" + + avg_allowed: float | None = None + """Average Medicare allowed amount (``Avg_Mdcr_Alowd_Amt``).""" + + avg_paid: float | None = None + """Average Medicare payment (``Avg_Mdcr_Pymt_Amt``).""" + + avg_standardized: float | None = None + """Average standardized payment (``Avg_Mdcr_Stdzd_Amt``).""" + + dataset_id: str | None = None + """data.cms.gov dataset UUID the row came from (one per year).""" + + item_key: str | None = None + """bib key of the Source item citing that year's dataset.""" diff --git a/src/pfs/utilization.py b/src/pfs/utilization.py new file mode 100644 index 0000000..88cebee --- /dev/null +++ b/src/pfs/utilization.py @@ -0,0 +1,299 @@ +"""Part B utilization per HCPCS code (#694): the CMS "Medicare Physician & +Other Practitioners — by Geography and Service" public use files, national +totals only, 2013 on. + +The direct outcome of a code becoming payable is whether anyone bills it, +so this is the first outcome series P50 joins to the exposure calendar +(``pfs.code_exposure``). Each calendar year is its own dataset on +data.cms.gov (``DATASETS`` maps year → UUID, from the data.json +catalogue); the Data API is pulled with ``filter[Rndrng_Prvdr_Geo_Lvl]= +National`` — ~13k rows a year, one per (HCPCS, place of service) — so +the whole 2013–2024 series is ~160k rows and refetches in minutes. + + fetch_year(client, year) → the raw API records for one year + to_rows(year, records, ...) → UtilizationRow per national row + cite(store, year) → the bib Source item for that year's dataset + write(con, year, rows) → delete + insert that year (idempotent) + read_series(con, codes) → per (code, year) totals across places of service + +Raw pages are cached under ``data/cms/utilization/mup-phy-geo-.json`` +so ``cms.ingest_log`` can carry the file's sha256 and a re-run needs no +network. Suppressed cells (``Tot_Benes`` < 11) arrive as ``""`` and are +stored as NULL, never 0. +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Sequence + +import httpx + +from bib.item import Source +from bib.tag import Tag + +#: Calendar year → data.cms.gov dataset UUID ("by Geography and Service"), +#: from https://data.cms.gov/data.json (catalogue read 2026-09-22). +DATASETS: dict[int, str] = { + 2013: "3c2a4756-0a8c-4e4d-845a-6ad169cb13d3", + 2014: "28181bd2-b377-4003-b73a-4bd92d1db4a9", + 2015: "dbee9609-2c90-43ca-b1b8-161bd9cfcdb2", + 2016: "c7d3f18c-2f00-4553-8cd1-871b727d5cdd", + 2017: "8e96a9f2-ce6e-46fd-b30d-8c695c756bfd", + 2018: "05a85700-052f-4509-af43-7042b9b35868", + 2019: "673030ae-ceed-4561-8fca-b1275395a86a", + 2020: "31eb3018-43e0-4259-a0d8-e7a1112ffd08", + 2021: "f00f462f-bb61-48b1-a321-1c50d78ce175", + 2022: "87304f15-9ed0-41dc-a141-6141a0327453", + 2023: "ddee9e22-7889-4bef-975a-7853e4cd0fbb", + 2024: "0c75b0b3-b40f-4007-a5ac-f9f2fed95862", +} +API = "https://data.cms.gov/data-api/v1/dataset/{id}/data" +LANDING = ( + "https://data.cms.gov/provider-summary-by-type-of-service/" + "medicare-physician-other-practitioners/" + "medicare-physician-other-practitioners-by-geography-and-service" +) +TITLE = "Medicare Physician & Other Practitioners - by Geography and Service" +TABLE = "pfs.utilization" +_PAGE = 5000 + + +@dataclass(frozen=True) +class UtilizationRow: + year: int + hcpcs: str + hcpcs_desc: str + drug_ind: str + place_of_service: str + n_providers: int | None + n_beneficiaries: int | None + n_services: float | None + n_bene_day_services: float | None + avg_submitted: float | None + avg_allowed: float | None + avg_paid: float | None + avg_standardized: float | None + dataset_id: str + item_key: str + + +@dataclass(frozen=True) +class SeriesRow: + """One (code, year) across places of service: counts summed, + ``avg_allowed``/``avg_paid`` weighted by services.""" + + hcpcs: str + year: int + n_providers: int | None + n_beneficiaries: int | None + n_services: float | None + avg_allowed: float | None + avg_paid: float | None + + +# ── fetch ───────────────────────────────────────────────────────────── + + +def fetch_year( + client: httpx.Client, year: int, *, page_size: int = _PAGE +) -> list[dict]: + """Every national row of *year*'s dataset, paged ``page_size`` at a + time until a short (or empty) page. Raises ``KeyError`` for a year + with no dataset and ``httpx.HTTPStatusError`` on a non-200.""" + url = API.format(id=DATASETS[year]) + out: list[dict] = [] + offset = 0 + while True: + r = client.get( + url, + params={ + "filter[Rndrng_Prvdr_Geo_Lvl]": "National", + "size": page_size, + "offset": offset, + }, + timeout=120, + ) + r.raise_for_status() + page = r.json() + out.extend(page) + if len(page) < page_size: + return out + offset += page_size + + +def _int(v: Any) -> int | None: + s = str(v or "").strip() + if not s: + return None + try: + return int(float(s)) + except ValueError: + return None + + +def _float(v: Any) -> float | None: + s = str(v or "").strip() + if not s: + return None + try: + return float(s) + except ValueError: + return None + + +def to_rows( + year: int, records: Sequence[dict], *, dataset_id: str, item_key: str +) -> list[UtilizationRow]: + """National rows only (the filter is re-checked here so a cached + payload from an unfiltered pull still loads cleanly).""" + out: list[UtilizationRow] = [] + for rec in records: + if (rec.get("Rndrng_Prvdr_Geo_Lvl") or "") != "National": + continue + out.append( + UtilizationRow( + year=year, + hcpcs=(rec.get("HCPCS_Cd") or "").strip().upper(), + hcpcs_desc=(rec.get("HCPCS_Desc") or "").strip(), + drug_ind=(rec.get("HCPCS_Drug_Ind") or "").strip(), + place_of_service=(rec.get("Place_Of_Srvc") or "").strip(), + n_providers=_int(rec.get("Tot_Rndrng_Prvdrs")), + n_beneficiaries=_int(rec.get("Tot_Benes")), + n_services=_float(rec.get("Tot_Srvcs")), + n_bene_day_services=_float(rec.get("Tot_Bene_Day_Srvcs")), + avg_submitted=_float(rec.get("Avg_Sbmtd_Chrg")), + avg_allowed=_float(rec.get("Avg_Mdcr_Alowd_Amt")), + avg_paid=_float(rec.get("Avg_Mdcr_Pymt_Amt")), + avg_standardized=_float(rec.get("Avg_Mdcr_Stdzd_Amt")), + dataset_id=dataset_id, + item_key=item_key, + ) + ) + return out + + +# ── cache ───────────────────────────────────────────────────────────── + + +def cache_path(year: int, *, root: Path | None = None) -> Path: + """``data/cms/utilization/mup-phy-geo-.json`` (next to the + primary DuckDB, so the sha256 in ``cms.ingest_log`` points at a file + that lives with the data it produced).""" + if root is None: + from conf import path + + root = path("db.aco").parent / "cms" / "utilization" + return Path(root) / f"mup-phy-geo-{year}.json" + + +def save_cache(p: Path, records: Sequence[dict]) -> None: + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text(json.dumps(list(records), separators=(",", ":"))) + + +def load_cache(p: Path) -> list[dict]: + return json.loads(p.read_text()) + + +# ── bib ─────────────────────────────────────────────────────────────── + + +def cite(store: Any, year: int) -> str: + """The bib Source item for *year*'s dataset — one per year, keyed by + the dataset's own API URL so re-runs return the same key. Tagged + ``module:pfs``, ``table:pfs.utilization``, ``source:cms-website``, + ``year:``.""" + item = Source( + title=f"{TITLE}, CY{year} (national totals per HCPCS and place of service)", + url=API.format(id=DATASETS[year]), + doc_type="dataset", + ) + tags = [ + Tag.module("pfs"), + Tag.table(TABLE), + Tag.source("cms-website"), + Tag.year(year), + ] + return store.upsert(item, tags=tags) + + +# ── DuckDB ──────────────────────────────────────────────────────────── + + +def ensure_table(con: Any) -> None: + from pfs.table.utilization import Utilization + + con.execute("CREATE SCHEMA IF NOT EXISTS pfs") + con.execute( + Utilization.to_ddl().replace("CREATE TABLE ", "CREATE TABLE IF NOT EXISTS ", 1) + ) + + +def write(con: Any, year: int, rows: Sequence[UtilizationRow]) -> int: + """Replace *year*'s rows.""" + ensure_table(con) + con.execute(f"DELETE FROM {TABLE} WHERE year = ?", [year]) + if not rows: + return 0 + con.executemany( + f"INSERT INTO {TABLE} VALUES ({','.join('?' * 15)})", + [ + ( + r.year, + r.hcpcs, + r.hcpcs_desc, + r.drug_ind, + r.place_of_service, + r.n_providers, + r.n_beneficiaries, + r.n_services, + r.n_bene_day_services, + r.avg_submitted, + r.avg_allowed, + r.avg_paid, + r.avg_standardized, + r.dataset_id, + r.item_key, + ) + for r in rows + ], + ) + return len(rows) + + +def read_series( + con: Any, codes: Sequence[str], *, years: Sequence[int] | None = None +) -> list[SeriesRow]: + """Per (code, year) totals across places of service, ordered by code + then year. Counts are summed (a provider billing both settings is + counted twice — the file gives no cross-setting distinct count); + ``avg_allowed``/``avg_paid`` are service-weighted means.""" + if not codes: + return [] + marks = ",".join("?" * len(codes)) + sql = ( + "SELECT hcpcs, year, sum(n_providers), sum(n_beneficiaries), sum(n_services), " + "sum(avg_allowed * n_services) / nullif(sum(n_services), 0), " + "sum(avg_paid * n_services) / nullif(sum(n_services), 0) " + f"FROM {TABLE} WHERE hcpcs IN ({marks})" + ) + params: list[Any] = [c.upper() for c in codes] + if years: + sql += f" AND year IN ({','.join('?' * len(years))})" + params.extend(int(y) for y in years) + sql += " GROUP BY hcpcs, year ORDER BY hcpcs, year" + return [ + SeriesRow( + hcpcs=h, + year=int(y), + n_providers=int(p) if p is not None else None, + n_beneficiaries=int(b) if b is not None else None, + n_services=float(s) if s is not None else None, + avg_allowed=float(a) if a is not None else None, + avg_paid=float(pd) if pd is not None else None, + ) + for h, y, p, b, s, a, pd in con.execute(sql, params).fetchall() + ] diff --git a/tests/cli/test_pfs_cli.py b/tests/cli/test_pfs_cli.py index b0a8cac..b6cceb9 100644 --- a/tests/cli/test_pfs_cli.py +++ b/tests/cli/test_pfs_cli.py @@ -2,6 +2,8 @@ from __future__ import annotations +from unittest.mock import MagicMock + import duckdb import pytest from typer.testing import CliRunner @@ -1096,3 +1098,104 @@ class TestExposure: assert "no exposure calendar yet" in res.output finally: bare.close() + + +class TestUtilization: + _REC = { + "Rndrng_Prvdr_Geo_Lvl": "National", + "HCPCS_Cd": "99490", + "HCPCS_Desc": "CCM first 20 min", + "HCPCS_Drug_Ind": "N", + "Place_Of_Srvc": "O", + "Tot_Rndrng_Prvdrs": "10", + "Tot_Benes": "100", + "Tot_Srvcs": "1000", + "Tot_Bene_Day_Srvcs": "990", + "Avg_Sbmtd_Chrg": "100", + "Avg_Mdcr_Alowd_Amt": "60", + "Avg_Mdcr_Pymt_Amt": "45", + "Avg_Mdcr_Stdzd_Amt": "44", + } + + @pytest.fixture + def env(self, con, monkeypatch, tmp_path): + from bib.store import Store + + store = Store(":memory:", storage_dir=tmp_path / "storage") + monkeypatch.setattr(pfs_cli, "_store", lambda: store) + monkeypatch.setattr( + pfs_cli, + "utilization_cache_path", + lambda y, root=None: tmp_path / f"mup-phy-geo-{y}.json", + ) + fetched: list[int] = [] + + def fake_fetch(client, y, **kw): + fetched.append(y) + return [self._REC, dict(self._REC, Place_Of_Srvc="F", Tot_Srvcs="5")] + + monkeypatch.setattr(pfs_cli, "fetch_utilization_year", fake_fetch) + monkeypatch.setattr(pfs_cli, "_http", lambda: MagicMock()) + try: + yield store, fetched, tmp_path + finally: + store.close() + + def test_write_loads_cites_logs_and_publishes(self, con, env): + store, fetched, tmp_path = env + res = runner.invoke(app, ["pfs", "utilization", "--write", "--year", "2023"]) + assert res.exit_code == 0, res.output + assert fetched == [2023] + assert con.execute( + "SELECT count(*), sum(n_services) FROM pfs.utilization WHERE year = 2023" + ).fetchone() == (2, 1005.0) + key = con.execute("SELECT DISTINCT item_key FROM pfs.utilization").fetchone()[0] + item = store.get(key) + assert "CY2023" in item.title and "year:2023" in item.tags + log = con.execute( + "SELECT module, table_name, rows, pincite_key, source_file FROM cms.ingest_log" + ).fetchone() + assert log[:4] == ("pfs.utilization", "pfs.utilization", 2, key) + assert log[4].endswith("mup-phy-geo-2023.json") + assert (tmp_path / "mup-phy-geo-2023.json").exists() + assert con.published == [True] + assert "2023: 2 records → 2 national rows" in res.output + + def test_write_is_idempotent_and_offline_reads_the_cache(self, con, env): + store, fetched, tmp_path = env + runner.invoke(app, ["pfs", "utilization", "--write", "--year", "2023"]) + res = runner.invoke( + app, ["pfs", "utilization", "--write", "--year", "2023", "--offline"] + ) + assert res.exit_code == 0, res.output + assert fetched == [2023] # the offline run never fetched + assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 2 + + def test_unknown_year_is_rejected_before_any_fetch(self, con, env): + store, fetched, _ = env + res = runner.invoke(app, ["pfs", "utilization", "--write", "--year", "1999"]) + assert res.exit_code != 0 + assert fetched == [] + + def test_code_reads_series_without_the_write_lock(self, con, env, monkeypatch): + runner.invoke(app, ["pfs", "utilization", "--write", "--year", "2023"]) + con.published.clear() + + def fail_batch(): + raise AssertionError("must not open a RW duckdb_batch connection") + + monkeypatch.setattr(pfs_cli, "_batch", fail_batch) + res = runner.invoke(app, ["pfs", "utilization", "--code", "99490"]) + assert res.exit_code == 0, res.output + assert "99490 2023 services 1,005" in res.output + assert con.published == [] + + def test_no_table_yet_is_a_friendly_no_op(self, monkeypatch): + bare = duckdb.connect(":memory:") + monkeypatch.setattr(pfs_cli, "_read", lambda: bare) + try: + res = runner.invoke(app, ["pfs", "utilization", "--code", "99490"]) + assert res.exit_code == 0, res.output + assert "no utilization table yet" in res.output + finally: + bare.close() diff --git a/tests/notebooks/test_code_families_nb.py b/tests/notebooks/test_code_families_nb.py index 62c79d5..7f048ec 100644 --- a/tests/notebooks/test_code_families_nb.py +++ b/tests/notebooks/test_code_families_nb.py @@ -39,7 +39,21 @@ def test_cells_are_anonymous_and_banners_present(): names = [n.name for n in tree.body if isinstance(n, ast.FunctionDef)] assert names and set(names) == {"_"} banners = re.findall(r"# ── (\S+)\. ", src) - assert banners == ["0", "1", "2", "3", "4", "5", "6", "7", "7b", "7c", "8"] + assert banners == [ + "0", + "1", + "2", + "3", + "4", + "5", + "6", + "7", + "7b", + "7c", + "7d", + "7e", + "8", + ] def test_headless_run_degrades_without_data(monkeypatch, tmp_path): diff --git a/tests/pfs/test_utilization.py b/tests/pfs/test_utilization.py new file mode 100644 index 0000000..0d858ad --- /dev/null +++ b/tests/pfs/test_utilization.py @@ -0,0 +1,228 @@ +"""pfs.utilization — Medicare Physician & Other Practitioners by Geography +and Service (national totals per HCPCS × place of service), #694.""" + +from __future__ import annotations + +import json + +import duckdb +import httpx +import pytest + +from bib.store import Store +from pfs.utilization import ( + API, + DATASETS, + UtilizationRow, + cite, + ensure_table, + fetch_year, + read_series, + to_rows, + write, +) + +REC = { + "Rndrng_Prvdr_Geo_Lvl": "National", + "Rndrng_Prvdr_Geo_Cd": "", + "Rndrng_Prvdr_Geo_Desc": "National", + "HCPCS_Cd": "99490", + "HCPCS_Desc": "Chronic care management services, first 20 minutes", + "HCPCS_Drug_Ind": "N", + "Place_Of_Srvc": "O", + "Tot_Rndrng_Prvdrs": "31182", + "Tot_Benes": "1158246", + "Tot_Srvcs": "5715325", + "Tot_Bene_Day_Srvcs": "5676766", + "Avg_Sbmtd_Chrg": "106.87513302", + "Avg_Mdcr_Alowd_Amt": "60.722041509", + "Avg_Mdcr_Pymt_Amt": "46.196134143", + "Avg_Mdcr_Stdzd_Amt": "45.634579675", +} + + +class TestDatasets: + def test_every_year_2013_to_2024_has_a_dataset(self): + assert sorted(DATASETS) == list(range(2013, 2025)) + assert all(len(v) == 36 for v in DATASETS.values()) + + +class TestToRows: + def test_parses_numbers(self): + rows = to_rows(2023, [REC], dataset_id="d", item_key="K") + assert len(rows) == 1 + r = rows[0] + assert isinstance(r, UtilizationRow) + assert (r.year, r.hcpcs, r.place_of_service, r.drug_ind) == ( + 2023, + "99490", + "O", + "N", + ) + assert r.n_providers == 31182 and r.n_beneficiaries == 1158246 + assert r.n_services == 5715325.0 and r.n_bene_day_services == 5676766.0 + assert r.avg_allowed == pytest.approx(60.722041509) + assert r.avg_paid == pytest.approx(46.196134143) + assert r.avg_submitted == pytest.approx(106.87513302) + assert r.avg_standardized == pytest.approx(45.634579675) + assert r.dataset_id == "d" and r.item_key == "K" + + def test_blanks_become_none_and_non_national_rows_are_skipped(self): + blank = dict(REC, Tot_Benes="", Avg_Mdcr_Pymt_Amt="") + state = dict(REC, Rndrng_Prvdr_Geo_Lvl="State", Rndrng_Prvdr_Geo_Desc="Ohio") + rows = to_rows(2023, [blank, state], dataset_id="d", item_key="K") + assert len(rows) == 1 + assert rows[0].n_beneficiaries is None and rows[0].avg_paid is None + + +def _paged_client(pages: list[list[dict]], seen: list[str]): + def handler(request): + seen.append(str(request.url)) + offset = int(request.url.params.get("offset", "0")) + size = int(request.url.params.get("size", "0")) + idx = offset // size if size else 0 + body = pages[idx] if idx < len(pages) else [] + return httpx.Response(200, json=body) + + return httpx.Client(transport=httpx.MockTransport(handler)) + + +class TestFetchYear: + def test_pages_until_a_short_page(self): + seen: list[str] = [] + pages = [[REC] * 3, [REC] * 3, [REC]] # size 3 → stop after the short page + client = _paged_client(pages, seen) + out = fetch_year(client, 2023, page_size=3) + assert len(out) == 7 + assert len(seen) == 3 + assert seen[0].startswith(API.format(id=DATASETS[2023])) + assert "Rndrng_Prvdr_Geo_Lvl%5D=National" in seen[0] + assert "offset=3" in seen[1] and "offset=6" in seen[2] + + def test_exact_multiple_stops_on_empty_page(self): + seen: list[str] = [] + client = _paged_client([[REC] * 2, [REC] * 2, []], seen) + assert len(fetch_year(client, 2023, page_size=2)) == 4 + assert len(seen) == 3 + + def test_unknown_year(self): + with pytest.raises(KeyError): + fetch_year(httpx.Client(), 1999) + + def test_http_error_raises(self): + client = httpx.Client( + transport=httpx.MockTransport(lambda r: httpx.Response(503)) + ) + with pytest.raises(httpx.HTTPStatusError): + fetch_year(client, 2023) + + +@pytest.fixture +def con(): + c = duckdb.connect(":memory:") + yield c + c.close() + + +class TestWriteAndRead: + def test_write_is_idempotent_per_year(self, con): + rows23 = to_rows( + 2023, + [REC, dict(REC, Place_Of_Srvc="F", Tot_Srvcs="100")], + dataset_id="d", + item_key="K", + ) + rows22 = to_rows( + 2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J" + ) + assert write(con, 2023, rows23) == 2 + assert write(con, 2022, rows22) == 1 + assert write(con, 2023, rows23) == 2 # replaces, never appends + assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 3 + + def test_read_series_sums_places_of_service(self, con): + write( + con, + 2023, + to_rows( + 2023, + [ + REC, + dict( + REC, + Place_Of_Srvc="F", + Tot_Srvcs="100", + Tot_Benes="10", + Tot_Rndrng_Prvdrs="5", + ), + ], + dataset_id="d", + item_key="K", + ), + ) + write( + con, + 2022, + to_rows( + 2022, [dict(REC, Tot_Srvcs="4000000")], dataset_id="e", item_key="J" + ), + ) + write( + con, + 2022, + to_rows( + 2022, + [ + dict(REC, Tot_Srvcs="4000000"), + dict(REC, HCPCS_Cd="99439", Tot_Srvcs="7"), + ], + dataset_id="e", + item_key="J", + ), + ) + s = read_series(con, ["99490"]) + assert [(r.year, r.n_services) for r in s] == [ + (2022, 4000000.0), + (2023, 5715425.0), + ] + assert s[1].n_beneficiaries == 1158256 and s[1].n_providers == 31187 + # allowed is the service-weighted mean across places of service + assert s[1].avg_allowed == pytest.approx(60.722041509, rel=1e-4) + assert [ + r.hcpcs for r in read_series(con, ["99439", "99490"], years=[2022]) + ] == ["99439", "99490"] + + def test_ensure_table_is_safe_to_repeat(self, con): + ensure_table(con) + ensure_table(con) + assert con.execute("SELECT count(*) FROM pfs.utilization").fetchone()[0] == 0 + + +class TestCite: + def test_one_source_per_year_with_tags(self, tmp_path): + s = Store(":memory:", storage_dir=tmp_path / "storage") + k1 = cite(s, 2023) + k2 = cite(s, 2023) + k3 = cite(s, 2022) + assert k1 == k2 and k1 != k3 + item = s.get(k1) + assert "CY2023" in item.title and DATASETS[2023] in item.url + tags = set(item.tags) + assert { + "module:pfs", + "table:pfs.utilization", + "source:cms-website", + "year:2023", + } <= tags + s.close() + + +class TestCachePayload: + def test_records_round_trip_json(self, tmp_path): + from pfs.utilization import cache_path, load_cache, save_cache + + p = cache_path(2023, root=tmp_path) + save_cache(p, [REC]) + assert p.exists() and p.name == "mup-phy-geo-2023.json" + assert load_cache(p) == [REC] + assert json.loads(p.read_text())[0]["HCPCS_Cd"] == "99490"