Files
stack/notebooks/code_families.py
kert b1e06b1b9e feat(bls): OEWS occupation wages + CES health-care employment — bls.oews / bls.ces, stack bls oews|ces, notebook 7f (refs #695)
OEWS: the BLS API serves only the latest OEWS year, so history comes
from the annual oesm{yy}nat.zip files (2013–2024), which bls.gov only
serves to a full browser header set; each zip's national_M{year}_dl
sheet (xls through 2013, xlsx after; OCC_GROUP through 2018, O_GROUP
plus area/industry columns from 2019) is normalised to one row per SOC
occupation with employment and the hourly/annual wage distribution,
suppressed cells (* # **) as NULL, and the BLS series-id prefix kept
per row. One bib Source per release (module:bls, table:bls.oews,
source:bls-website, year:N) and a cms.ingest_log row with the zip's
sha256.

CES: a closed registry of employment (01) and average hourly earnings
(03) series for health care, ambulatory care, offices of physicians,
home health, hospitals and nursing/residential care (CES65<industry>
<type>), pulled through the public API in 10-year × 25-series chunks
(20 × 50 with BLS_API_KEY), monthly rows with the series id on every
row, one bib Source per series.

stack bls oews [--year]... [--write [--offline]] [--occ]...;
stack bls ces [--write] [--series]... [--annual]. Notebook 7f overlays
clinical-staff wages (31-9092, 29-2061, 29-1141, 29-1171) and
physician-office / ambulatory employment on the family's exposure
dates. CLI docs regenerated.
2026-09-22 16:25:46 -04:00

1565 lines
63 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import marimo
__generated_with = "0.23.13"
app = marimo.App(width="medium")
@app.cell(hide_code=True)
def _():
import marimo as mo
return (mo,)
@app.cell(hide_code=True)
def _(mo):
mo.md("""
# Code families as first-class objects
A physician fee schedule code is not a number — it is a **bundle of logical elements**:
who furnishes the service, for how long, per what period, to which patients, doing which
activities, by which modality. This notebook walks through how the stack turns that idea
into tables you can query and cite: elements → extraction → lineage → families → anchors
→ guidance → public reaction. Every number on this page is read live from the replica and
the bibliography; every claim links to the Federal Register paragraph it came from.
""")
return
@app.cell(hide_code=True)
def _():
# ── Setup ──
import altair as alt
import polars as pl
from conf import connect, path
from conf.display import plain_years
connect.theme()
NOTES = {}
# data/replica/<name>.ro.duckdb — same layout connect._replica_path resolves,
# computed here (not imported) because it's a private helper.
_primary = path("db.aco")
REPLICA_PATH = _primary.parent / "replica" / f"{_primary.stem}.ro.duckdb"
def _open_replica():
try:
return connect.duckdb("aco", read_only=True)
except Exception as e: # noqa: BLE001 — degrade, never crash the page
NOTES["replica"] = f"replica unavailable: {e}"
return None
def _open_bib():
try:
return connect.bib()
except Exception as e: # noqa: BLE001
NOTES["bib"] = f"bibliography unavailable: {e}"
return None
con = _open_replica()
store = _open_bib()
def q(sql, params=()):
if con is None:
return pl.DataFrame()
try:
return con.execute(sql, list(params)).pl()
except Exception as e: # noqa: BLE001 — a missing table is a "not built yet"
NOTES[sql[:40]] = str(e)
return pl.DataFrame()
def fr_md(item_key, p_id):
if store is None:
return f"{item_key}{p_id}"
try:
from bib.frlink import md_link
return md_link(
f"p-{p_id}", store=store, item_key=item_key, text=f"{item_key}{p_id}"
)
except Exception: # noqa: BLE001
return f"{item_key}{p_id}"
def not_built(cmd):
return f"_Not built yet — run `{cmd}` and republish the replica._"
return (
NOTES,
REPLICA_PATH,
alt,
con,
fr_md,
not_built,
pl,
plain_years,
q,
store,
)
@app.cell(hide_code=True)
def _(fr_md, mo, store):
# ── 0. Why a code is a bundle of elements ──
_steps = []
if store is not None:
try:
_con = store._con() # noqa: SLF001
for _p in (394, 396, 398):
_row = _con.execute(
"SELECT text FROM fr_anchors WHERE item_key = ? AND p_id = ?",
("2KVJ2HKX", _p),
).fetchone()
if _row:
_steps.append(f"> {_row[0][:400]}… — {fr_md('2KVJ2HKX', _p)}")
except Exception: # noqa: BLE001
pass
mo.md(
"## 0. Why a code is a bundle of elements\n\n"
"CMS says so itself. When it decides whether a service can be furnished by telehealth it "
"walks three steps, and the third is literally *review the elements of the service as "
"described by the HCPCS code* (CY2026 proposed rule, 90 FR 32389):\n\n"
+ (
"\n\n".join(_steps)
if _steps
else "_(bibliography unavailable — quotes omitted)_"
)
+ "\n\nThat one sentence is the whole design: a code is not an opaque five-character "
"string CMS prices as a unit. It is a bundle of *who* furnishes it, *how long* it takes, "
"*how often* it can be billed, *which patients* qualify, *which activities* it covers, "
"and *by what modality* — and CMS itself reasons about codes at that level of detail, one "
"element at a time. Everything below builds machine-readable tables out of that same "
"bundle: parse the descriptor into typed elements, extract them with their FR anchor, "
"trace how a code's identity changes over the years (lineage), and group codes that share "
"one clinical program into a family."
)
return
@app.cell(hide_code=True)
def _(mo, not_built, q):
# Two organizing principles behind every table on this page
_editions = q(
"SELECT DISTINCT edition_year FROM pfs.cpt_code ORDER BY edition_year"
)
if _editions.is_empty():
_view = mo.md(not_built("stack pfs cpt-ingest --all"))
else:
_years = ", ".join(str(y) for y in _editions["edition_year"].to_list())
_view = mo.md(
"**Two organizing principles, in that order.** The CPT codebook is the **first**: "
"the American Medical Association's own hierarchy — Section → subsection → category "
"→ subcategory → codes, with guideline text at every section, symbols marking new, "
"revised, add-on and telemedicine codes, and parenthetical instructions "
"cross-referencing related codes. That hierarchy is what groups codes into clinical "
"families below, and it is what a descriptor's elements are drawn from. The Federal "
"Register is the **second**: it is where CMS decides whether Medicare pays for a code "
"the AMA has already defined, and how much — the FR reprices codes, it does not "
f"reorganize the code set. CPT editions on this replica: {_years}."
)
_view
return
@app.cell(hide_code=True)
def _(mo):
# ── 1. Reading a descriptor ──
from pfs.families import HAND_FAMILIES
_codes = sorted({c for f in HAND_FAMILIES.values() for c in f.codes})
code_picker = mo.ui.dropdown(options=_codes, value="99490", label="Code")
mo.vstack(
[
mo.md(
"## 1. Reading a descriptor\n\n"
"The Federal Register prints a code's descriptor as a *stem* paragraph — the "
"sentence that opens with the code number and a parenthesis — followed by one "
"paragraph per required element, each ending in a semicolon or closing "
"parenthesis. `descriptor_runs` finds every place a rule prints that pattern for "
"a code and pairs the stem with the element paragraphs that immediately follow "
"it; `parse_descriptor` then reads whatever a regex can read out of that text — "
"minutes, billing periods, populations, activities, modalities — against a "
"**closed vocabulary** that only grows by human review. Pick a code from any of "
"the five hand-registered families below."
),
code_picker,
]
)
return (code_picker,)
@app.cell(hide_code=True)
def _(code_picker, fr_md, mo, pl, store):
from pfs.descriptors import descriptor_runs
from pfs.elements import VOCAB, parse_descriptor
code = code_picker.value
runs = descriptor_runs(store, code) if store is not None else []
if not runs:
_view = mo.md(
f"_No Federal Register descriptor run found for {code} (bibliography unavailable "
"or code never printed as a stem)._"
)
elements = ()
else:
# The original codification, preferring the earliest run that carries its own
# element paragraphs (a later rule often just cites the code inline mid-sentence,
# with no paragraph break) — for 99490 this is the CY2015 final rule, stem ¶1244.
run = next((r for r in runs if r.elements), runs[0])
_paras = pl.DataFrame(
{
"p_id": [run.stem.p_id, *[p.p_id for p in run.elements]],
"role": ["stem", *["element"] * len(run.elements)],
"text": [run.stem.text[:300], *[p.text[:300] for p in run.elements]],
}
)
elements = parse_descriptor(run.text)
_els = pl.DataFrame(
{
"type": [e.type.value for e in elements],
"value": [e.value for e in elements],
"detail": [e.detail for e in elements],
}
)
_vocab = pl.DataFrame(
{
"type": [t.value for t in VOCAB for _ in VOCAB[t]],
"value": [v for t in VOCAB for v in VOCAB[t]],
}
)
_view = mo.vstack(
[
mo.md(
f"**{code}** as printed in {run.item_key} (CY{run.rule_year}), stem "
f"{fr_md(run.item_key, run.stem.p_id)}: the stem paragraph opens the "
"descriptor and each following paragraph is one element."
),
mo.ui.table(_paras, label="Descriptor paragraphs"),
mo.md(
"The deterministic parser reads what a regex can read — minutes, periods, "
"code references, and the recurring phrases:"
),
mo.ui.table(_els, label="Typed elements"),
mo.accordion(
{
"The closed vocabulary (values grow only by review)": mo.ui.table(
_vocab
)
}
),
]
)
_view
return (code,)
@app.cell(hide_code=True)
def _(code, mo, not_built, pl, q):
# Where the manual files this code
_ed = q("SELECT max(edition_year) y FROM pfs.cpt_code")
if _ed.is_empty():
_view = mo.md(
"**Where the manual files this code.**\n\n"
+ not_built("stack pfs cpt-ingest --all")
)
else:
_year = _ed.item(0, "y")
_cpt = q(
"SELECT c.stem, c.elements, c.tail, c.addon, c.resequenced, c.new, c.revised, "
"c.telemedicine, c.mod51_exempt, c.audio_only, c.fda_pending, c.pla, s.path_key "
"FROM pfs.cpt_code c JOIN pfs.cpt_section s USING (edition_year, item_key, sec_id) "
"WHERE c.edition_year = ? AND c.code = ?",
(_year, code),
)
if _cpt.is_empty():
_view = mo.md(
f"**Where the manual files this code.** `{code}` is a HCPCS Level II code — "
"CMS's own coding system, used when Medicare has a programmatic need CPT does "
f"not cover — and is **not in the CPT {_year} book**; the manual's hierarchy "
"below does not apply to it."
)
else:
_row = _cpt.row(0, named=True)
_crumb = " → ".join(_row["path_key"].split(" > "))
_flags = [
label
for label, on in (
("add-on ✚", _row["addon"]),
("resequenced #", _row["resequenced"]),
("new ●", _row["new"]),
("revised ▲", _row["revised"]),
("telemedicine ★", _row["telemedicine"]),
("modifier-51 exempt ⦸", _row["mod51_exempt"]),
("audio-only", _row["audio_only"]),
("FDA-pending ⚡", _row["fda_pending"]),
("PLA", _row["pla"]),
)
if on
]
_instr = q(
"SELECT kind, text, targets FROM pfs.cpt_instruction WHERE edition_year = ? "
"AND code = ? ORDER BY kind, text",
(_year, code),
)
_view = mo.vstack(
[
mo.md(
"**Where the manual files this code.** The CPT hierarchy is the first "
f"organizing principle: in the CY{_year} book `{code}` sits under "
f"**{_crumb}**."
+ (
f" Symbols: {', '.join(_flags)}."
if _flags
else " No symbols set."
)
),
mo.ui.table(
pl.DataFrame(
{
"part": [
"stem",
*["element"] * len(_row["elements"]),
"tail",
],
"text": [_row["stem"], *_row["elements"], _row["tail"]],
}
),
label="The manual's own stem / elements / tail",
),
(
mo.ui.table(_instr, label="Parenthetical instructions")
if not _instr.is_empty()
else mo.md(
"_No parenthetical instructions for this code in this edition._"
)
),
]
)
_view
return
@app.cell(hide_code=True)
def _(code, mo, not_built, q):
# ── 2. What the extractor wrote ──
_els = q(
"SELECT type, value, detail, source, confirmed_by, item_key, p_id, page "
"FROM pfs.code_element WHERE code = ? ORDER BY type, value",
(code,),
)
_rev = q(
"SELECT text, proposed_value, source, item_key, p_id FROM pfs.code_element_review "
"WHERE code = ? ORDER BY p_id",
(code,),
)
if _els.is_empty():
_view = mo.md(
"## 2. What the extractor wrote\n\n"
+ not_built(f"stack pfs elements --code {code}")
)
else:
_view = mo.vstack(
[
mo.md(
"## 2. What the extractor wrote\n\n"
"Three passes, in order: the regex parser above finds what it can; a local "
"model then reads every remaining candidate line and chooses **one slug from "
"the closed list, or `none`**; whatever neither pass can place is queued for "
"human review rather than guessed. Nothing enters `pfs.code_element` unless "
"it is a member of the closed vocabulary, and every row keeps the exact "
"Federal Register paragraph it came from. When a second source (the CPT "
"manual, HCPCS, RVU) independently transcribes the same element, that source "
"isn't a second row — it's recorded in `confirmed_by` on the winning row."
),
mo.ui.table(
_els, label=f"pfs.code_element — {code} ({_els.height} rows)"
),
(
mo.ui.table(
_rev,
label=f"pfs.code_element_review — {code} ({_rev.height} lines)",
)
if not _rev.is_empty()
else mo.md("_Review queue empty for this code._")
),
]
)
_view
return
@app.cell(hide_code=True)
def _(alt, code, fr_md, mo, not_built, pl, plain_years, q):
# ── 3. Lineage ──
from pfs.families import family_of as _family_of
_fam = _family_of(code)
_codes = list(_fam.codes) if _fam else [code]
_ev = q(
"SELECT code, year, kind, from_codes, to_codes, source, anchored, item_key, p_id, note "
"FROM pfs.code_event WHERE code IN (" + ",".join("?" * len(_codes)) + ") "
"ORDER BY year, code, kind",
_codes,
)
if _ev.is_empty():
_view = mo.md(
"## 3. Lineage\n\n" + not_built(f"stack pfs lineage --code {code} --write")
)
else:
_chart = (
alt.Chart(_ev.to_pandas())
.mark_circle(size=90)
.encode(
x=alt.X("year:O", title="Rule year"),
y=alt.Y("kind:N", title=None),
color=alt.Color("source:N", title="Source"),
shape=alt.Shape("anchored:N", title="Anchored"),
tooltip=[
"code",
"year",
"kind",
"from_codes",
"to_codes",
"item_key",
"p_id",
"note",
],
)
.properties(height=260, width=640)
)
# Ruling A2: build the anchor column with a plain list comprehension, not
# DataFrame.map_rows.
_anchor_col = [
fr_md(item_key, p_id) if item_key else "rvu"
for item_key, p_id in zip(_ev["item_key"].to_list(), _ev["p_id"].to_list())
]
_links = _ev.with_columns(pl.Series("anchor", _anchor_col))
_view = mo.vstack(
[
mo.md(
"## 3. Lineage\n\n"
f"Every dated event for the **{_fam.name if _fam else code}** codes. "
"RVU-file events (`source=rvu`) are dated by the fee-schedule year; Federal "
"Register events are dated by the **rule that mentions them** — a later rule "
"recounting a code's creation adds a later `created` row, which is why the "
"earliest anchored event is the origin, not the latest one. An RVU event is "
"`anchored` when a Federal Register event for the same code lies within one "
"rule year of it; an unanchored RVU event is evidence CMS never wrote a "
"sentence about, and is weaker to cite. CPT-source events (`source=cpt`) are "
"dated a third way: `cpt_changed` rows use the AMA's own CPT Changes edition "
"year — the year the AMA revised the code, independent of whether or when any "
"FR rule mentions it — and both an FR event and a CPT event for the same code "
"can anchor the same RVU change from two independent directions."
),
mo.ui.altair_chart(_chart),
mo.ui.table(
plain_years(_links.drop("item_key", "p_id")), label="pfs.code_event"
),
]
)
_view
return
@app.cell(hide_code=True)
def _(code, mo, not_built, pl, plain_years, q):
# ── 4. Families ──
from pfs.families import family_of as _family_of
_hand = _family_of(code)
_key = _hand.key if _hand else ""
_rows = (
q(
"SELECT key, name, code, role, since, until, item_key, p_id, note "
"FROM pfs.code_family WHERE key = ? ORDER BY code",
(_key,),
)
if _key
else None
)
if _rows is None or _rows.is_empty():
_view = mo.md("## 4. Families\n\n" + not_built("stack pfs families --write"))
else:
# The family's CPT heading — the organizing principle that classified its members
# (Task 6 context: "note" is per-code, so different members can carry different
# headings when the hand list or a derived edge pulls in a neighboring heading).
_notes = [n for n in _rows["note"].to_list() if n]
_note = _notes[0] if _notes else ""
_siblings = pl.DataFrame()
if _note:
_parts = _note.split(" > ")
_parent_prefix = " > ".join(_parts[:-1])
_yr = q("SELECT max(edition_year) y FROM pfs.cpt_section")
_year = _yr.item(0, "y") if not _yr.is_empty() else None
if _year is not None:
_lvl_row = q(
"SELECT min(level) lvl FROM pfs.cpt_section WHERE edition_year = ? "
"AND path_key = ?",
(_year, _note),
)
_lvl = (
_lvl_row.item(0, "lvl")
if not _lvl_row.is_empty() and _lvl_row.item(0, "lvl") is not None
else None
)
if _lvl is not None:
_siblings = q(
"SELECT DISTINCT title FROM pfs.cpt_section WHERE edition_year = ? "
"AND path_key LIKE ? AND level = ? ORDER BY title",
(_year, _parent_prefix + " > %", _lvl),
)
_summary = q(
"WITH fam AS (SELECT key, count(*) n, bool_or(note <> '') has_note "
"FROM pfs.code_family GROUP BY key) "
"SELECT count(*) total, count(*) FILTER (WHERE n > 1) multi, "
"count(*) FILTER (WHERE has_note) cpt_named FROM fam"
)
_by_chapter = q(
"SELECT split_part(note, ' > ', 1) AS chapter, count(DISTINCT key) AS families "
"FROM pfs.code_family WHERE note <> '' GROUP BY 1 ORDER BY families DESC"
)
_view = mo.vstack(
[
mo.md(
"## 4. Families\n\n"
"A family is a connected component over four kinds of edge: an **add-on** "
"relation element (`in conjunction with 99490`), a **defined-by-reference** "
"relation (`with the elements included in 99490`), a **single-target "
"replacement** lineage event (one code's `replaced_by` names exactly one "
"successor), and **stem similarity with an identical activity set** (two "
"descriptors share half their service-naming words and every activity "
"element). The hand-written registry is a floor, never a ceiling: "
f"**{_key}** lists {len(_hand.codes)} hand codes; the derived table below "
f"shows {_rows.height}, because the connected-component search also reaches "
"codes the hand list never named."
),
mo.md(
f"**Where this family sits in the manual.** {_note}"
if _note
else "_No member of this family carries a CPT heading (HCPCS-only family)._"
),
mo.ui.table(plain_years(_rows), label=f"pfs.code_family — {_key}"),
(
mo.ui.table(
_siblings, label="Sibling headings under the same parent"
)
if not _siblings.is_empty()
else mo.md("_No sibling headings found._")
),
mo.md(
"**Example (single-target replacement):** HCPCS G2058, billable only in "
"2020, was replaced the following year by CPT 99439 — an identical "
"descriptor crosswalked at the same value. That `replaced_by` event names "
"exactly one target code, so it is unambiguous evidence, and G2058 joins CCM "
"as a *predecessor* rather than becoming a one-code family of its own."
),
(
mo.md(
f"**Across all families:** {_summary.item(0, 'total')} total, "
f"{_summary.item(0, 'multi')} multi-code, "
f"{_summary.item(0, 'cpt_named')} carry a CPT heading (`note <> ''`)."
)
if not _summary.is_empty()
else mo.md("")
),
(
mo.ui.table(
_by_chapter, label="CPT-named families by top-level chapter"
)
if not _by_chapter.is_empty()
else mo.md("")
),
]
)
_view
return
@app.cell(hide_code=True)
def _(code, mo, not_built, pl, store):
# ── 5. Anchors in the corpus ──
import os as _os
from pfs.families import family_of as _family_of
_fam = _family_of(code)
_key = (_fam.key if _fam else code).upper()
_fam_codes = set(_fam.codes) if _fam else {code}
_intro = mo.md(
"## 5. Anchors in the corpus\n\n"
"How many chunks in the RAG index — comments, guidance, Federal Register "
"text — carry this family's codes as metadata, and how many bibliography "
"items carry a `family:` or `code:` tag (`stack bib code-tags`). Chunk "
f"counts live in pgvector (**{_key}**, family-membership match), not the "
"DuckDB replica this notebook otherwise reads; item tags live in the "
"bibliography SQLite."
)
_panels = [_intro]
if not _os.environ.get("LLM_DB_PASSWORD"):
_panels.append(
mo.md(
"_`LLM_DB_PASSWORD` not set — pgvector is unreachable from this "
"process, so chunk-per-collection counts are skipped; item-tag "
"counts from the bibliography follow instead._"
)
)
else:
try:
# sqlalchemy + psycopg only (the notebooks image, #720) — not
# llm.index, which pulls in langchain.
from sqlalchemy import create_engine as _sa_engine
from sqlalchemy import text as _sa_text
from llm.config import load as _load_llm_cfg
from llm.config import pg_url as _pg_url
from pfs.anchors import families_array_sql as _families_array_sql
_eng = _sa_engine(_pg_url(_load_llm_cfg()))
with _eng.begin() as _conn:
_chunk_rows = _conn.execute(
_sa_text(
"SELECT c.name, count(*) AS n "
"FROM langchain_pg_embedding e "
"JOIN langchain_pg_collection c ON c.uuid = e.collection_id "
f"WHERE {_families_array_sql('e')} && ARRAY[:key] "
"GROUP BY c.name ORDER BY c.name"
),
{"key": _key},
).fetchall()
except ImportError as e:
_panels.append(
mo.md(
"_pgvector client not installed in this notebook environment "
f"(`sqlalchemy`/`psycopg` — {e}); chunk-per-collection counts "
"are skipped._"
)
)
except Exception as e: # noqa: BLE001 — degrade, never crash the page
_panels.append(mo.md(f"_pgvector unavailable: {e}_"))
else:
_chunks = pl.DataFrame(
{
"collection": [r[0] for r in _chunk_rows],
"chunks": [r[1] for r in _chunk_rows],
}
)
_panels.append(
mo.ui.table(_chunks, label=f"pgvector chunks tagged families ∋ {_key}")
if not _chunks.is_empty()
else mo.md(f"_No pgvector chunks tagged `families ∋ {_key}` yet._")
)
if store is None:
_panels.append(mo.md("_bibliography unavailable — item-tag counts omitted._"))
else:
try:
_fam_tags = pl.DataFrame(store.list_tags(namespace="family"))
_code_tags = pl.DataFrame(
[
row
for row in store.list_tags(namespace="code")
if row["name"].removeprefix("code:") in _fam_codes
]
)
except Exception as e: # noqa: BLE001
_panels.append(mo.md(f"_item-tag counts unavailable: {e}_"))
else:
_panels.append(
mo.ui.table(_fam_tags, label="bib item tags — family:*")
if not _fam_tags.is_empty()
else mo.md(not_built("stack bib code-tags"))
)
_panels.append(
mo.ui.table(_code_tags, label=f"bib item tags — code:* for {_key}")
if not _code_tags.is_empty()
else mo.md(f"_No `code:` tags for {_key}'s member codes yet._")
)
mo.vstack(_panels)
return
@app.cell(hide_code=True)
def _(code, con, fr_md, mo, not_built, pl, store):
# ── 6. Guidance ──
from bib.cfrlink import md_link as _md_link
from pfs.codetables import read_guidance as _read_guidance
from pfs.families import family_of as _family_of
_fam = _family_of(code)
_key = _fam.key if _fam else code
_rows = []
if con is not None:
try:
_rows = _read_guidance(con, _key)
except Exception: # noqa: BLE001 — a missing table is "not built yet"
_rows = []
if not _rows:
_view = mo.md(
"## 6. Guidance\n\n"
"Sub-regulatory guidance — Medicare Learning Network articles, "
"Internet-Only Manual sections, MACs' local coverage determinations — "
"that cites a family's codes, with CFR cross-references where the "
"guidance implements a rule.\n\n"
+ not_built(f"stack pfs guidance --family {_key} --write")
)
else:
def _citation(r):
# CFR: a live eCFR link built straight from the stored locator.
if r.kind == "cfr":
try:
return _md_link(r.locator)
except ValueError:
return r.locator
# IOM: the manual chapter's own bib title when the citation
# resolved to a library item, else fall back to the locator.
if r.kind == "iom" and r.item_key and store is not None:
try:
title = store.get(r.item_key).title
if title:
return title
except KeyError:
pass
return r.locator # iom (unresolved) and mln both show the locator
def _provenance(r):
# CPT-manual guideline citations carry no FR paragraph
# (p_id_src=0, page_src=0 — module docstring); everything else
# anchors to the exact FR paragraph that named the code.
if r.p_id_src:
return fr_md(r.item_key_src, r.p_id_src)
return f"{r.item_key_src} (CPT manual)" if r.item_key_src else ""
_tbl = pl.DataFrame(
{
"code": [r.code for r in _rows],
"kind": [r.kind for r in _rows],
"citation": [_citation(r) for r in _rows],
"cited from": [_provenance(r) for r in _rows],
}
)
_n_cfr = sum(1 for r in _rows if r.kind == "cfr")
_n_iom = sum(1 for r in _rows if r.kind == "iom")
_n_mln = sum(1 for r in _rows if r.kind == "mln")
_view = mo.vstack(
[
mo.md(
"## 6. Guidance\n\n"
"Sub-regulatory guidance — Medicare Learning Network articles, "
"Internet-Only Manual sections, MACs' local coverage "
"determinations — that cites a family's codes, with CFR "
"cross-references where the guidance implements a rule. Every "
"row is a reference actually found in text that also names one "
"of the family's codes, resolved against the bibliography where "
"possible and always anchored to where it was cited. "
f"**{_key}**: {_n_cfr} CFR, {_n_iom} IOM, {_n_mln} MLN "
f"reference{'s' if len(_rows) != 1 else ''}."
),
mo.ui.table(
_tbl, label=f"pfs.code_guidance — {_key} ({len(_rows)} rows)"
),
]
)
_view
return
@app.cell(hide_code=True)
def _(alt, code, con, mo, not_built, pl):
# ── 7. Reaction ──
from pfs.codetables import read_reaction as _read_reaction
from pfs.families import family_of as _family_of
_fam = _family_of(code)
_key = _fam.key if _fam else code
_rows = []
if con is not None:
try:
_rows = _read_reaction(con, _key)
except Exception: # noqa: BLE001 — a missing table is "not built yet"
_rows = []
if not _rows:
_view = mo.md(
"## 7. Reaction\n\n"
"Public comment volume on a family's codes over time — how many "
"comment letters mention them, in which rule years, and (on a sample) "
"whether commenters supported or opposed the proposal.\n\n"
+ not_built(f"stack pfs reaction --family {_key} --write")
)
else:
_dockets = sorted(
(r for r in _rows if r.period_kind == "docket"),
key=lambda r: (r.year, r.period),
)
_fr_rows = sorted(
(r for r in _rows if r.period_kind == "fr-pairs"), key=lambda r: r.year
)
_has_stance = any(
r.stance_support or r.stance_oppose or r.stance_modify or r.stance_unclear
for r in _dockets
)
_panels = [
mo.md(
"## 7. Reaction\n\n"
"Two independent proxies for public reaction, both counted from "
"chunk/paragraph metadata rather than hand-read. **Dockets** "
"(regulations.gov, 2017 on) count the distinct commenters "
"(`item_key`) whose comment text names one of the family's codes "
"— `n_items` out of the docket's `n_total` distinct commenters, "
"read straight from the `comments` pgvector collection; a stance "
"breakdown (support/oppose/modify/unclear) is filled in only for "
"dockets built with `--stance-sample N`, which classifies the "
"newest *N* matching commenters' first chunk with the self-hosted "
"closed-vocabulary classifier — an answer outside the four "
"stances counts as `unclear` — so it is a sample, not a census. "
"**FR pairs** (back to 2001) count `Comment:`/`Response:` "
"paragraph pairs CMS itself printed in a rule's preamble that "
"name a family code — the pre-2017 proxy, since regulations.gov "
"comment text isn't indexed that far back; `n_items` is the "
"qualifying-pair count naming the family and `n_total` the "
"rule's total `Comment:` paragraph count (family or not) — the "
"pool `n_items` is drawn out of."
+ (
""
if _has_stance
else " No docket in this family was built with "
"`--stance-sample`, so the stance columns below are all zero."
)
)
]
if _dockets:
_dk = pl.DataFrame(
{
"label": [f"{r.year} {r.period}" for r in _dockets],
"n_items": [r.n_items for r in _dockets],
"n_total": [r.n_total for r in _dockets],
"support": [r.stance_support for r in _dockets],
"oppose": [r.stance_oppose for r in _dockets],
"modify": [r.stance_modify for r in _dockets],
"unclear": [r.stance_unclear for r in _dockets],
}
)
_chart1 = (
alt.Chart(_dk.to_pandas())
.mark_bar()
.encode(
x=alt.X(
"label:N", sort=_dk["label"].to_list(), title="Docket (year)"
),
y=alt.Y("n_items:Q", title="Commenters naming the family"),
tooltip=[
"label",
"n_items",
"n_total",
"support",
"oppose",
"modify",
"unclear",
],
)
.properties(height=240, width=640)
)
_panels.append(mo.ui.altair_chart(_chart1))
else:
_panels.append(mo.md("_No docket has comment chunks for this family._"))
if _fr_rows:
_fp = pl.DataFrame(
{
"year": [r.year for r in _fr_rows],
"item_key": [r.period for r in _fr_rows],
"n_items": [r.n_items for r in _fr_rows],
}
)
_chart2 = (
alt.Chart(_fp.to_pandas())
.mark_bar()
.encode(
x=alt.X("year:O", title="Rule year"),
y=alt.Y("n_items:Q", title="Comment/Response pairs"),
tooltip=["year", "item_key", "n_items"],
)
.properties(height=220, width=640)
)
_panels.append(mo.ui.altair_chart(_chart2))
else:
_panels.append(
mo.md("_No pre-2017 Comment:/Response: pairs found for this family._")
)
_tbl = pl.DataFrame(
{
"period_kind": [r.period_kind for r in _rows],
"period": [r.period for r in _rows],
"year": [r.year for r in _rows],
"n_items": [r.n_items for r in _rows],
"n_total": [r.n_total for r in _rows],
"support": [r.stance_support for r in _rows],
"oppose": [r.stance_oppose for r in _rows],
"modify": [r.stance_modify for r in _rows],
"unclear": [r.stance_unclear for r in _rows],
}
)
_panels.append(
mo.ui.table(_tbl, label=f"pfs.code_reaction — {_key} ({len(_rows)} rows)")
)
_view = mo.vstack(_panels)
_view
return
@app.cell(hide_code=True)
def _(code, mo, pl):
# ── 7b. What the chat sees ──
_import_err = ""
try:
# Replica-only reads — no LLM_DB_PASSWORD, no pool, no langchain
# (llm's pool exports resolve lazily; #720).
from llm.config import load as _load_llm_cfg
from llm.lineage import lineage_evidence as _lineage_evidence
except ImportError as _e: # the notebooks image lacks the `llm` extra
_import_err = str(_e)
_ev = None
else:
_cfg = _load_llm_cfg()
_ev = _lineage_evidence(f"history of {code}", _cfg)
_intro = mo.md(
"## 7b. What the chat sees\n\n"
"This is the same timeline the chat itself builds when it answers a "
"question about this code. `llm.lineage.lineage_evidence` reads "
"`pfs.code_event` — populated for every PFS-payable code, not just "
"the codes with hand-built element extractions, by "
"`stack pfs lineage --all-payable --write` — collapses repeated "
"mentions of the same event down to one representative row, labels "
"each surviving row with the rule paragraph (or RVU-file year, or "
"CPT Changes edition) it comes from, and hands the model exactly "
"the bracketed labels shown below in its prompt, so any dated claim "
"in a chat answer can be traced back to the paragraph it came from."
)
if _ev is None:
_view = mo.vstack(
[
_intro,
mo.md(
f"_The chat's lineage module could not be imported here ({_import_err}); "
"this section needs the `llm` package's base dependencies._"
)
if _import_err
else mo.md(
"_No lineage events for this code — run "
"`stack pfs lineage --all-payable --write`._"
),
]
)
else:
_payload = _ev.payload()
_events = _payload["events"]
def _codes_cell(e):
if e["from_codes"] or e["to_codes"]:
return (
f"{e['code']} ({', '.join(e['from_codes'])} → "
f"{', '.join(e['to_codes'])})"
)
return e["code"]
def _label_cell(e):
return f"[{e['label']}]({e['url']})" if e["url"] else e["label"]
_tbl = pl.DataFrame(
{
"Year": [e["year"] for e in _events],
"Event": [e["kind"] for e in _events],
"Codes": [_codes_cell(e) for e in _events],
"Label": [_label_cell(e) for e in _events],
}
)
_panels = [
_intro,
mo.ui.table(_tbl, label=f"lineage_evidence events — {len(_events)} rows"),
mo.accordion(
{
"Prompt block the model sees": mo.md(
f"```\n{_ev.prompt_block()}\n```"
)
}
),
]
if _payload["element_diffs"]:
_panels.append(
mo.md(
"**Element differences**\n\n"
+ "\n".join(
f"- `{d['type']}={d['value']}`: in {', '.join(d['in_codes'])}; "
f"not in {', '.join(d['not_in_codes'])} ({d['label']})"
for d in _payload["element_diffs"]
)
)
)
if _payload.get("elements_note"):
_panels.append(
mo.md(
f"_{_payload['elements_note']} — element diffs cover only the "
"elements-extracted codes until #698 lands._"
)
)
if _payload["guidance"]:
_panels.append(
mo.md(
"**Guidance references**\n\n"
+ "\n".join(
f"- {g['locator']} — {g['kind'].upper()} ({g['label']})"
for g in _payload["guidance"]
)
)
)
_view = mo.vstack(_panels)
_view
return
@app.cell(hide_code=True)
def _(code, con, fr_md, mo, not_built, pl):
# ── 7c. Exposure calendar ──
from pfs.codetables import read_exposures as _read_exposures
from pfs.exposure import control_codes as _control_codes
from pfs.families import family_of as _family_of_x
_fam = _family_of_x(code)
_key = _fam.key if _fam else ""
_rows = []
if con is not None:
try:
_rows = (
_read_exposures(con, family=_key)
if _key
else _read_exposures(con, code=code)
)
except Exception: # noqa: BLE001 — a missing table is "not built yet"
_rows = []
_intro = (
"## 7c. Exposure calendar\n\n"
"P50 reads the fee schedule as law with effects: every row here is a date "
"on which a code's legal status under the PFS changed — it became payable, "
"was revalued by more than ten percent, moved between payable statuses, "
"ended, or was listed for telehealth — derived from the RVU files and the "
"lineage events above, never hand-entered. The effective date is 1 January "
"of the rule year unless the only Federal Register anchor is a correction "
"notice, in which case the notice's own date is used. Each row keeps the "
"code's RVU status and decile band that year, which is what a "
"never-treated control set is matched on."
)
if not _rows:
_view = mo.md(_intro + "\n\n" + not_built("stack pfs exposure --write"))
else:
_tbl = pl.DataFrame(
{
"code": [r.code for r in _rows],
"effective": [r.effective_date.isoformat() for r in _rows],
"exposure": [r.kind for r in _rows],
"status": [
f"{r.prior_status or '—'}→{r.status or '—'}"
if r.kind in ("becomes-payable", "ends", "status-change")
else (f"{r.delta_pct:+.1%}" if r.kind == "revalued" else r.status)
for r in _rows
],
"band": [r.rvu_band for r in _rows],
"anchor": [
(
fr_md(r.item_key, r.p_id)
+ (" (correction)" if r.correction else "")
)
if r.item_key
else "_unanchored_"
for r in _rows
],
}
)
# A control-set preview for the family's most recent becomes-payable
# exposure (or its newest row): same status + band, quiet for ±2 years.
_pick = max(
_rows,
key=lambda r: (r.kind == "becomes-payable", r.year),
)
try:
_ctrl = _control_codes(con, _pick.code, _pick.year, window=2, limit=12)
except Exception: # noqa: BLE001
_ctrl = []
_ctrl_md = (
f"**Controls for {_pick.code} {_pick.kind} ({_pick.year})** — "
f"{len(_ctrl)} shown, cheapest first: "
+ ", ".join(f"{c}" + (f" [{f}]" if f else "") for c, f, _s, _b in _ctrl)
if _ctrl
else f"_No never-treated controls found for {_pick.code} in {_pick.year}._"
)
_n_anchored = sum(1 for r in _rows if r.item_key)
_view = mo.vstack(
[
mo.md(
_intro + f"\n\n**{_key or code}**: {len(_rows)} exposure"
f"{'s' if len(_rows) != 1 else ''}, {_n_anchored} anchored to an "
"FR paragraph."
),
mo.ui.table(
_tbl,
label=f"pfs.code_exposure — {_key or code} ({len(_rows)} rows)",
),
mo.md(_ctrl_md),
]
)
_view
return
@app.cell(hide_code=True)
def _(alt, code, con, mo, not_built, pl):
# ── 7d. Utilization ──
from pfs.codetables import read_exposures as _read_exposures_u
from pfs.families import family_of as _family_of_u
from pfs.utilization import read_series as _read_series
_fam = _family_of_u(code)
_key = _fam.key if _fam else code
_codes = list(_fam.codes) if _fam else [code]
_series = []
if con is not None:
try:
_series = _read_series(con, _codes)
except Exception: # noqa: BLE001 — a missing table is "not built yet"
_series = []
_intro = (
"## 7d. Utilization\n\n"
"The first outcome series P50 joins to the exposure calendar: services "
"billed, beneficiaries and rendering providers per code and year from the "
"CMS *Medicare Physician & Other Practitioners — by Geography and Service* "
"public use files (national totals, 2013 on; `pfs.utilization`, one bib "
"Source item per year). Counts are summed across facility and office "
"settings; a provider billing in both is counted twice, because the file "
"gives no cross-setting distinct count. Beneficiary counts under 11 are "
"suppressed by CMS and appear as gaps, not zeros. Dashed rules mark the "
"family's exposure dates from section 7c."
)
if not _series:
_view = mo.md(_intro + "\n\n" + not_built("stack pfs utilization --write"))
else:
_df = pl.DataFrame(
{
"code": [r.hcpcs for r in _series],
"year": [r.year for r in _series],
"services": [r.n_services for r in _series],
"beneficiaries": [r.n_beneficiaries for r in _series],
"providers": [r.n_providers for r in _series],
"avg allowed": [r.avg_allowed for r in _series],
}
)
try:
_exp = (
_read_exposures_u(con, family=_key)
if _fam
else _read_exposures_u(con, code=code)
)
except Exception: # noqa: BLE001
_exp = []
_rules = pl.DataFrame(
{
"year": [r.year for r in _exp if r.kind in ("becomes-payable", "ends")],
"exposure": [
f"{r.code} {r.kind}"
for r in _exp
if r.kind in ("becomes-payable", "ends")
],
}
)
_chart = (
alt.Chart(_df.to_pandas())
.mark_line(point=True)
.encode(
x=alt.X("year:O", title="Calendar year"),
y=alt.Y("services:Q", title="Services billed"),
color=alt.Color("code:N", title="Code"),
tooltip=[
"code",
"year",
"services",
"beneficiaries",
"providers",
"avg allowed",
],
)
.properties(width=640, height=300, title=f"{_key}: services per year")
)
if len(_rules):
_chart = _chart + (
alt.Chart(_rules.to_pandas())
.mark_rule(strokeDash=[4, 4], color="gray")
.encode(x="year:O", tooltip=["exposure"])
)
_latest = max(r.year for r in _series)
_view = mo.vstack(
[
mo.md(
_intro + f"\n\n**{_key}**: {len(_df)} (code, year) rows, "
f"latest year on file {_latest}."
),
mo.ui.altair_chart(_chart),
mo.ui.table(_df, label=f"pfs.utilization — {_key}"),
]
)
_view
return
@app.cell(hide_code=True)
def _(con, mo, pl):
# ── 7e. The three P50 questions ──
from pfs.utilization import read_series as _read_series_q
_q = []
if con is not None:
try:
_q = _read_series_q(
con, ["99490", "G2058", "99439", "G0556", "G0557", "G0558"]
)
except Exception: # noqa: BLE001
_q = []
_intro = mo.md(
"## 7e. The three P50 questions\n\n"
"#694 asks the utilization table three things before anything else: "
"how many CCM services (99490) were billed each year from 2015; whether "
"the G2058 → 99439 add-on carried over across the 2021 code change; and "
"how APCM (G0556–G0558) uptake in 2025 compares with CCM. The first two "
"read straight off the table. The third waits on CMS: the newest "
"Geography-and-Service file covers services through calendar 2024, so "
"APCM — payable from 2025-01-01 — has no rows yet; the panel below "
"states that rather than plotting zeros."
)
if not _q:
_view = mo.vstack(
[_intro, mo.md("_Not built yet — run `stack pfs utilization --write`._")]
)
else:
_by = {}
for r in _q:
_by.setdefault(r.hcpcs, {})[r.year] = r
_ccm = _by.get("99490", {})
_t1 = pl.DataFrame(
{
"year": sorted(_ccm),
"99490 services": [_ccm[y].n_services for y in sorted(_ccm)],
"beneficiaries": [_ccm[y].n_beneficiaries for y in sorted(_ccm)],
}
)
_g, _n = _by.get("G2058", {}), _by.get("99439", {})
_t2 = pl.DataFrame(
{
"year": [2020, 2021],
"G2058 services": [
_g.get(2020) and _g[2020].n_services,
_g.get(2021) and _g[2021].n_services,
],
"99439 services": [
_n.get(2020) and _n[2020].n_services,
_n.get(2021) and _n[2021].n_services,
],
}
)
_apcm_years = sorted(
{y for c in ("G0556", "G0557", "G0558") for y in _by.get(c, {})}
)
_t3 = (
f"APCM rows on file: {', '.join(map(str, _apcm_years))}."
if _apcm_years
else "No APCM (G0556–G0558) rows yet — the 2025 file has not been published."
)
_view = mo.vstack(
[
_intro,
mo.md("**1. CCM (99490) services per year**"),
mo.ui.table(_t1, label="99490 by year"),
mo.md("**2. G2058 → 99439 continuity, 2020–2021**"),
mo.ui.table(_t2, label="add-on continuity"),
mo.md(f"**3. APCM uptake vs CCM** — {_t3}"),
]
)
_view
return
@app.cell(hide_code=True)
def _(alt, code, con, mo, not_built, pl):
# ── 7f. Clinical-staff wages and employment ──
from bls.ces import read_series as _read_ces
from bls.oews import read_occupations as _read_oews
from pfs.codetables import read_exposures as _read_exposures_w
from pfs.families import family_of as _family_of_w
_fam = _family_of_w(code)
_key = _fam.key if _fam else code
_occs = {
"31-9092": "Medical assistants",
"29-2061": "Licensed practical / vocational nurses",
"29-1141": "Registered nurses",
"29-1171": "Nurse practitioners",
}
_wages, _ces = [], []
if con is not None:
try:
_wages = _read_oews(con, list(_occs))
except Exception: # noqa: BLE001 — not built yet
_wages = []
try:
_ces = _read_ces(
con, ["CES6562110001", "CES6562100001", "CES6562110003"], annual=True
)
except Exception: # noqa: BLE001
_ces = []
_intro = (
"## 7f. Clinical-staff wages and employment\n\n"
"Care-management codes pay for clinical-staff time — 99490 is twenty "
"minutes of clinical staff directed by the practitioner, APCM's `actor` "
"element is the same staff — so the second outcome series is what that "
"time costs and how many people are employed to supply it. Wages are the "
"BLS OEWS national estimates (`bls.oews`, one May release a year, each "
"cited as its own bib Source); employment is the BLS CES monthly series "
"for offices of physicians and ambulatory care (`bls.ces`, annual means "
"of the seasonally adjusted months, series ids kept on every row). Dashed "
"rules are the family's `becomes-payable` / `ends` dates from section 7c. "
"The SOC revision between the 2018 and 2019 releases can move an "
"occupation's boundary; the four occupations shown kept their codes."
)
if not _wages and not _ces:
_view = mo.md(
_intro
+ "\n\n"
+ not_built("stack bls oews --write` and `stack bls ces --write")
)
else:
try:
_exp = (
_read_exposures_w(con, family=_key)
if _fam
else _read_exposures_w(con, code=code)
)
except Exception: # noqa: BLE001
_exp = []
_rules = pl.DataFrame(
{
"year": [r.year for r in _exp if r.kind in ("becomes-payable", "ends")],
"exposure": [
f"{r.code} {r.kind}"
for r in _exp
if r.kind in ("becomes-payable", "ends")
],
}
)
_panels = [mo.md(_intro)]
if _wages:
_wdf = pl.DataFrame(
{
"occupation": [
f"{r.occ_code} {_occs.get(r.occ_code, r.occ_title)}"
for r in _wages
],
"year": [r.year for r in _wages],
"mean annual wage": [r.a_mean for r in _wages],
"median annual wage": [r.a_median for r in _wages],
"employment": [r.tot_emp for r in _wages],
}
)
_wchart = (
alt.Chart(_wdf.to_pandas())
.mark_line(point=True)
.encode(
x=alt.X("year:O", title="OEWS release year"),
y=alt.Y("mean annual wage:Q", title="Mean annual wage ($)"),
color=alt.Color("occupation:N", title="Occupation"),
tooltip=[
"occupation",
"year",
"mean annual wage",
"median annual wage",
"employment",
],
)
.properties(
width=640,
height=300,
title="Clinical-staff mean annual wage (OEWS)",
)
)
if len(_rules):
_wchart = _wchart + (
alt.Chart(_rules.to_pandas())
.mark_rule(strokeDash=[4, 4], color="gray")
.encode(x="year:O", tooltip=["exposure"])
)
_panels += [
mo.ui.altair_chart(_wchart),
mo.ui.table(_wdf, label="bls.oews — clinical staff"),
]
else:
_panels.append(mo.md(not_built("stack bls oews --write")))
if _ces:
_cdf = pl.DataFrame(
{
"series": [
f"{r.series_id} {r.industry_name} {r.data_type}" for r in _ces
],
"year": [r.year for r in _ces],
"annual mean": [r.value for r in _ces],
}
)
_emp = _cdf.filter(pl.col("series").str.contains("employment"))
_cchart = (
alt.Chart(_emp.to_pandas())
.mark_line(point=True)
.encode(
x=alt.X("year:O", title="Year"),
y=alt.Y(
"annual mean:Q", title="Employees (thousands, SA annual mean)"
),
color=alt.Color("series:N", title="CES series"),
tooltip=["series", "year", "annual mean"],
)
.properties(width=640, height=260, title="Health-care employment (CES)")
)
if len(_rules):
_cchart = _cchart + (
alt.Chart(_rules.to_pandas())
.mark_rule(strokeDash=[4, 4], color="gray")
.encode(x="year:O", tooltip=["exposure"])
)
_panels += [
mo.ui.altair_chart(_cchart),
mo.ui.table(_cdf, label="bls.ces — annual means"),
]
else:
_panels.append(mo.md(not_built("stack bls ces --write")))
_view = mo.vstack(_panels)
_view
return
@app.cell(hide_code=True)
def _(NOTES, REPLICA_PATH, con, mo, pl, q, store):
# ── 8. Provenance ──
import datetime as _dt
if REPLICA_PATH.exists():
_mtime = _dt.datetime.fromtimestamp(REPLICA_PATH.stat().st_mtime).isoformat(
timespec="seconds"
)
_replica_line = f"`{REPLICA_PATH}` — last published {_mtime}"
else:
_replica_line = f"`{REPLICA_PATH}` — not found"
_counts = q(
"SELECT 'code_element' t, count(*) n FROM pfs.code_element "
"UNION ALL SELECT 'code_element_review', count(*) FROM pfs.code_element_review "
"UNION ALL SELECT 'code_event', count(*) FROM pfs.code_event "
"UNION ALL SELECT 'code_family', count(*) FROM pfs.code_family "
"UNION ALL SELECT 'code_exposure', count(*) FROM pfs.code_exposure "
"UNION ALL SELECT 'utilization', count(*) FROM pfs.utilization "
"UNION ALL SELECT 'bls.oews', count(*) FROM bls.oews "
"UNION ALL SELECT 'bls.ces', count(*) FROM bls.ces"
)
_log = q(
"SELECT run_id, ingested_at, module, table_name, rule_id, source_file, sha256, rows, "
"fr_citation, pincite_key FROM cms.ingest_log WHERE table_name LIKE 'pfs.%' "
"ORDER BY ingested_at DESC LIMIT 20"
)
# pfs.cpt_* comes from a separate ingest path (stack pfs cpt-ingest) with its own
# per-edition grain, so it gets its own counts table and its own bib provenance —
# one row per CPT edition on the replica, with the edition's bib title.
_cpt_counts = q(
"SELECT 'cpt_section' t, edition_year, count(*) n FROM pfs.cpt_section GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_code', edition_year, count(*) FROM pfs.cpt_code GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_instruction', edition_year, count(*) "
"FROM pfs.cpt_instruction GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_reference', edition_year, count(*) "
"FROM pfs.cpt_reference GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_crosswalk', edition_year, count(*) "
"FROM pfs.cpt_crosswalk GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_list', edition_year, count(*) FROM pfs.cpt_list GROUP BY 1, 2 "
"UNION ALL SELECT 'cpt_code_alt', edition_year, count(*) "
"FROM pfs.cpt_code_alt GROUP BY 1, 2 "
"ORDER BY edition_year, t"
)
_cpt_editions = q(
"SELECT DISTINCT edition_year, item_key FROM pfs.cpt_code ORDER BY edition_year"
)
_bib_rows = []
for _yr, _key in zip(
[] if _cpt_editions.is_empty() else _cpt_editions["edition_year"].to_list(),
[] if _cpt_editions.is_empty() else _cpt_editions["item_key"].to_list(),
):
if store is None:
_title = "(bibliography unavailable)"
else:
try:
_title = store.get(_key).title
except KeyError:
_title = "(item not found)"
_bib_rows.append({"edition_year": _yr, "item_key": _key, "title": _title})
_cpt_bib = pl.DataFrame(_bib_rows) if _bib_rows else pl.DataFrame()
mo.vstack(
[
mo.md(
"## 8. Provenance\n\n"
f"Replica: {_replica_line}"
+ (
" (connection open)"
if con is not None
else " (connection unavailable)"
)
+ ". Row counts and the most recent ingest-log entries for every `pfs.code_*` "
"table this notebook reads follow, so a reader can tell how fresh the family "
"data is without leaving the page. The CPT codebook tables (`pfs.cpt_*`) are "
"ingested separately, one edition at a time, so their counts and bib "
"provenance are broken out by edition below."
),
mo.ui.table(_counts, label="pfs.code_* row counts")
if not _counts.is_empty()
else mo.md("_replica unavailable_"),
mo.ui.table(_log, label="Ingest log — pfs.*")
if not _log.is_empty()
else mo.md("_no ingest log rows_"),
mo.ui.table(_cpt_counts, label="pfs.cpt_* row counts by edition")
if not _cpt_counts.is_empty()
else mo.md(
"_no CPT codebook tables built yet — run `stack pfs cpt-ingest --all`_"
),
mo.ui.table(_cpt_bib, label="CPT edition bib items")
if not _cpt_bib.is_empty()
else mo.md("_no CPT editions ingested_"),
mo.md(
"\n".join(f"- {k}: {v}" for k, v in NOTES.items())
if NOTES
else "_All sources available._"
),
]
)
return
if __name__ == "__main__":
app.run()