perf(pfs,cli): elements --all-payable in one inverted pass, not two fr_anchors scans per code (refs #698)
`stack pfs elements --all-payable` targeted ~8.7k A/R/T codes and ran two full scans of the unindexed 193k-row `fr_anchors` table per code — the `LIKE '%CODE%'` descriptor-stem query plus a follow-up query per stem — on top of three per-code replica queries. ~15-25 minutes of SQL before the first model call, which is why the command had never been run end to end. Invert the loop, the way `lineage --all-payable` already does: - `pfs.descriptors.descriptor_runs_bucketed` streams `fr_anchors` once in `(item_key, p_id)` order — the order the per-code follow-up query walks — keeping each stem's run open until a paragraph fails `is_element_paragraph`, closes with a parenthesis, or hits the element cap. Stems are found with one `_stem_pattern_any` regex over all target codes instead of one compiled pattern per code. - `hcpcs_long_descriptions` / `rvu_descriptions_bucketed` / `_cpt_elements_bucketed` fetch every target code's row in one query each, keeping the same newest-row picks as `QUALIFY row_number()`. - `pfs.extract._assemble` becomes the single home of the precedence rules (FR > CPT > HCPCS/RVU with `confirmed_by` cross-checks), shared by the per-code `extract_code` and the new `extract_codes`, so the two paths cannot drift. `uv run stack pfs elements --all-payable --no-llm --dry-run` now finishes in ~12s for 8,689 codes, and all 17 hand-family codes extract identical rows and reviews on the live corpus. Drop the "experimental" wording and the "~20 min of SQL" warning; `_codes_for` keeps its signature and still logs the target count under `warn_slow`. No index on `fr_anchors.text`: the inverted pass never does a text LIKE, so an FTS5/trigram index would be write-path cost for nothing.
This commit is contained in:
@@ -41,7 +41,7 @@ from pfs.codetables import (
|
|||||||
write_reaction,
|
write_reaction,
|
||||||
)
|
)
|
||||||
from pfs.cpt_load import ingest as cpt_ingest
|
from pfs.cpt_load import ingest as cpt_ingest
|
||||||
from pfs.extract import extract_code
|
from pfs.extract import extract_codes
|
||||||
from pfs.families import (
|
from pfs.families import (
|
||||||
FAMILIES,
|
FAMILIES,
|
||||||
HAND_FAMILIES,
|
HAND_FAMILIES,
|
||||||
@@ -117,8 +117,13 @@ def _engine() -> Any:
|
|||||||
def _run_elements(
|
def _run_elements(
|
||||||
con: Any, store: Any, targets: list[str], classify: Any, *, write: bool
|
con: Any, store: Any, targets: list[str], classify: Any, *, write: bool
|
||||||
) -> None:
|
) -> None:
|
||||||
|
# #698: one inverted pass over fr_anchors (plus one query each over
|
||||||
|
# pfs.cpt_code / terminology.hcpcs_level_2 / pfs.rvu) for the whole
|
||||||
|
# target list — the per-code form re-scanned the 193k-row, unindexed
|
||||||
|
# fr_anchors table twice per code, ~15-25 min for --all-payable.
|
||||||
|
by_code = extract_codes(store, con, targets, classify=classify)
|
||||||
for c in targets:
|
for c in targets:
|
||||||
x = extract_code(store, con, c, classify=classify)
|
x = by_code[c]
|
||||||
if write:
|
if write:
|
||||||
write_elements(con, c, x.rows, x.reviews)
|
write_elements(con, c, x.rows, x.reviews)
|
||||||
confirmed = sum(1 for r in x.rows if r.confirmed_by)
|
confirmed = sum(1 for r in x.rows if r.confirmed_by)
|
||||||
@@ -134,10 +139,13 @@ def _codes_for(
|
|||||||
*,
|
*,
|
||||||
warn_slow: bool = True,
|
warn_slow: bool = True,
|
||||||
) -> list[str]:
|
) -> list[str]:
|
||||||
"""*warn_slow*: the "~20 min of SQL before any model call" warning
|
"""*warn_slow*: log how many codes ``--all-payable`` resolved to, so a
|
||||||
describes ``elements``' per-code LLM classification pass — it does
|
run that is about to make one model call per unplaced line for
|
||||||
NOT apply to ``lineage --all-payable`` (one inverted pass over
|
thousands of codes says so up front. ``lineage --all-payable`` is
|
||||||
``fr_anchors``, seconds not minutes), which passes ``warn_slow=False``."""
|
pure SQL/regex and needs no such note, so it passes
|
||||||
|
``warn_slow=False``. (Both commands now sweep ``fr_anchors`` in one
|
||||||
|
inverted pass — #698 — so neither is the ~20 min of per-code SQL the
|
||||||
|
warning used to describe.)"""
|
||||||
out: list[str] = [c.upper() for c in codes]
|
out: list[str] = [c.upper() for c in codes]
|
||||||
if families:
|
if families:
|
||||||
refresh_from(con)
|
refresh_from(con)
|
||||||
@@ -157,11 +165,7 @@ def _codes_for(
|
|||||||
"AND year = (SELECT max(year) FROM pfs.rvu) ORDER BY hcpcs"
|
"AND year = (SELECT max(year) FROM pfs.rvu) ORDER BY hcpcs"
|
||||||
).fetchall()
|
).fetchall()
|
||||||
if warn_slow:
|
if warn_slow:
|
||||||
log.warning(
|
log.warning("--all-payable: targeting %d codes", len(rows))
|
||||||
"--all-payable is experimental: targeting %d codes (~20 min of SQL "
|
|
||||||
"before any model call)",
|
|
||||||
len(rows),
|
|
||||||
)
|
|
||||||
out.extend(r[0] for r in rows)
|
out.extend(r[0] for r in rows)
|
||||||
if not out:
|
if not out:
|
||||||
raise typer.BadParameter("pass --code, --family or --all-payable")
|
raise typer.BadParameter("pass --code, --family or --all-payable")
|
||||||
@@ -178,8 +182,8 @@ def elements(
|
|||||||
False,
|
False,
|
||||||
"--all-payable",
|
"--all-payable",
|
||||||
help=(
|
help=(
|
||||||
"Every A/R/T code in the newest RVU year (experimental: ~20 min "
|
"Every A/R/T code in the newest RVU year, one inverted pass over "
|
||||||
"of SQL before any model call)."
|
"fr_anchors."
|
||||||
),
|
),
|
||||||
),
|
),
|
||||||
no_llm: bool = typer.Option(
|
no_llm: bool = typer.Option(
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import re
|
import re
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Any
|
from typing import Any, Sequence
|
||||||
|
|
||||||
from pfs.elements import is_element_paragraph
|
from pfs.elements import is_element_paragraph
|
||||||
|
|
||||||
@@ -157,6 +157,118 @@ def descriptor_runs(
|
|||||||
return runs
|
return runs
|
||||||
|
|
||||||
|
|
||||||
|
#: A code token as the codes we target are shaped: five alphanumerics
|
||||||
|
#: ("99490", "G0556", "0075T"). ``_stem_pattern_any`` uses it as the
|
||||||
|
#: generic stand-in for "any target code" so one regex finds every code's
|
||||||
|
#: stem in a paragraph; a target of any other length falls back to its
|
||||||
|
#: own literal in the same alternation.
|
||||||
|
_CODE_TOKEN = re.compile(r"[0-9A-Za-z]{5}\Z")
|
||||||
|
|
||||||
|
|
||||||
|
def _stem_pattern_any(codes: Sequence[str]) -> re.Pattern[str]:
|
||||||
|
"""One pattern matching *any* of *codes* in descriptor-stem position.
|
||||||
|
|
||||||
|
``_stem_pattern`` compiled per code turns an N-code sweep into N
|
||||||
|
passes over every paragraph. The per-code pattern differs between
|
||||||
|
codes only in the code literal, and every target is a fixed-width
|
||||||
|
token, so substituting the token's character class for the literal
|
||||||
|
matches at exactly the positions the per-code patterns do — the class
|
||||||
|
can only stand for the whole five characters that precede the opening
|
||||||
|
parenthesis, so no other code can slip in. The caller re-checks
|
||||||
|
``group(1)`` against its target set. Odd-length targets, if any, are
|
||||||
|
added as explicit literals so they keep matching too."""
|
||||||
|
odd = sorted(c for c in codes if not _CODE_TOKEN.match(c))
|
||||||
|
token = "[0-9A-Z]{5}"
|
||||||
|
if odd:
|
||||||
|
token = "(?:" + token + "|" + "|".join(re.escape(c) for c in odd) + ")"
|
||||||
|
return re.compile(rf"\b({token})\s*\(\s*(?!codes?\b)[A-Za-z]", re.I)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class _OpenRun:
|
||||||
|
"""A stem whose element paragraphs are still arriving in the stream."""
|
||||||
|
|
||||||
|
item_key: str
|
||||||
|
rule_year: int
|
||||||
|
stem: Para
|
||||||
|
sort_key: tuple[str, int]
|
||||||
|
elements: list[Para]
|
||||||
|
|
||||||
|
|
||||||
|
def descriptor_runs_bucketed(
|
||||||
|
store: Any, codes: Sequence[str], *, max_elements: int = 40
|
||||||
|
) -> dict[str, list[DescriptorRun]]:
|
||||||
|
"""``descriptor_runs`` for every one of *codes* in ONE streaming pass
|
||||||
|
over ``fr_anchors`` instead of two unindexed scans per code (a
|
||||||
|
``LIKE '%CODE%'`` stem query plus a follow-up query per stem found) —
|
||||||
|
the whole point for an 8.7k-code universe over ~193k paragraphs
|
||||||
|
(#698).
|
||||||
|
|
||||||
|
Rows arrive ordered by ``(item_key, p_id)``, which is exactly the
|
||||||
|
order the per-code follow-up query walks, so element paragraphs are
|
||||||
|
collected by keeping each stem's run open until a paragraph fails
|
||||||
|
``is_element_paragraph``, closes with a parenthesis, or *max_elements*
|
||||||
|
have been taken. Each code's runs are then ordered by
|
||||||
|
``(date_published, p_id)`` to match ``descriptor_runs``' own
|
||||||
|
``ORDER BY``, so ``descriptor_runs_bucketed(store, [c])[c] ==
|
||||||
|
descriptor_runs(store, c)`` for any single code."""
|
||||||
|
targets = {c.upper() for c in codes}
|
||||||
|
if not targets:
|
||||||
|
return {}
|
||||||
|
pat = _stem_pattern_any(targets)
|
||||||
|
pending: dict[str, list[_OpenRun]] = {c: [] for c in targets}
|
||||||
|
open_runs: list[_OpenRun] = []
|
||||||
|
item = None
|
||||||
|
for r in store._con().execute(
|
||||||
|
"SELECT a.item_key, a.p_id, a.page, a.text, i.title, i.date_published "
|
||||||
|
"FROM fr_anchors a JOIN items i ON i.key = a.item_key "
|
||||||
|
"ORDER BY a.item_key, a.p_id"
|
||||||
|
):
|
||||||
|
text = r["text"]
|
||||||
|
if r["item_key"] != item:
|
||||||
|
item, open_runs = r["item_key"], []
|
||||||
|
if open_runs:
|
||||||
|
if is_element_paragraph(text):
|
||||||
|
para = Para(r["item_key"], r["p_id"], r["page"], text)
|
||||||
|
closes = text.rstrip().endswith((")", ")."))
|
||||||
|
still: list[_OpenRun] = []
|
||||||
|
for run in open_runs:
|
||||||
|
run.elements.append(para)
|
||||||
|
if not closes and len(run.elements) < max_elements:
|
||||||
|
still.append(run)
|
||||||
|
open_runs = still
|
||||||
|
else:
|
||||||
|
open_runs = []
|
||||||
|
hits = {
|
||||||
|
m.group(1).upper()
|
||||||
|
for m in pat.finditer(text)
|
||||||
|
if m.group(1).upper() in targets
|
||||||
|
}
|
||||||
|
for code in sorted(hits):
|
||||||
|
run = _OpenRun(
|
||||||
|
r["item_key"],
|
||||||
|
rule_year_of(r["title"], r["date_published"]),
|
||||||
|
Para(r["item_key"], r["p_id"], r["page"], descriptor_span(text, code)),
|
||||||
|
(r["date_published"] or "", r["p_id"]),
|
||||||
|
[],
|
||||||
|
)
|
||||||
|
pending[code].append(run)
|
||||||
|
open_runs.append(run)
|
||||||
|
return {
|
||||||
|
code: [
|
||||||
|
DescriptorRun(
|
||||||
|
code=code,
|
||||||
|
item_key=run.item_key,
|
||||||
|
rule_year=run.rule_year,
|
||||||
|
stem=run.stem,
|
||||||
|
elements=tuple(run.elements),
|
||||||
|
)
|
||||||
|
for run in sorted(runs, key=lambda run: run.sort_key)
|
||||||
|
]
|
||||||
|
for code, runs in pending.items()
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def hcpcs_long_description(con: Any, code: str) -> str:
|
def hcpcs_long_description(con: Any, code: str) -> str:
|
||||||
row = con.execute(
|
row = con.execute(
|
||||||
"SELECT long_description FROM terminology.hcpcs_level_2 WHERE hcpcs = ? ORDER BY CAST(seqnum AS INTEGER) DESC, recid DESC LIMIT 1",
|
"SELECT long_description FROM terminology.hcpcs_level_2 WHERE hcpcs = ? ORDER BY CAST(seqnum AS INTEGER) DESC, recid DESC LIMIT 1",
|
||||||
@@ -176,3 +288,49 @@ def rvu_descriptions(con: Any, code: str) -> list[tuple[int, str, str]]:
|
|||||||
[code.upper()],
|
[code.upper()],
|
||||||
).fetchall()
|
).fetchall()
|
||||||
return [(int(y), s or "", d or "") for y, s, d in rows]
|
return [(int(y), s or "", d or "") for y, s, d in rows]
|
||||||
|
|
||||||
|
|
||||||
|
#: DuckDB has no bound-list ``IN`` — unnesting a ``VARCHAR[]`` parameter
|
||||||
|
#: is the supported way to filter one query by every target code at once,
|
||||||
|
#: and it keeps the code list out of the SQL text (no 8.7k-literal string
|
||||||
|
#: to build or re-plan).
|
||||||
|
IN_TARGETS = "IN (SELECT unnest(CAST(? AS VARCHAR[])))"
|
||||||
|
|
||||||
|
|
||||||
|
def hcpcs_long_descriptions(con: Any, codes: Sequence[str]) -> dict[str, str]:
|
||||||
|
"""``hcpcs_long_description`` for every one of *codes* in one query.
|
||||||
|
Codes with no ``terminology.hcpcs_level_2`` row are simply absent —
|
||||||
|
callers read them back with ``.get(code, "")``, the same empty string
|
||||||
|
the per-code helper returns."""
|
||||||
|
targets = sorted({c.upper() for c in codes})
|
||||||
|
if not targets:
|
||||||
|
return {}
|
||||||
|
rows = con.execute(
|
||||||
|
"SELECT hcpcs, long_description FROM terminology.hcpcs_level_2 "
|
||||||
|
f"WHERE hcpcs {IN_TARGETS} QUALIFY row_number() OVER "
|
||||||
|
"(PARTITION BY hcpcs ORDER BY CAST(seqnum AS INTEGER) DESC, recid DESC) = 1",
|
||||||
|
[targets],
|
||||||
|
).fetchall()
|
||||||
|
return {h: (d or "") for h, d in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def rvu_descriptions_bucketed(
|
||||||
|
con: Any, codes: Sequence[str]
|
||||||
|
) -> dict[str, list[tuple[int, str, str]]]:
|
||||||
|
"""``rvu_descriptions`` for every one of *codes* in one query —
|
||||||
|
partitioning by ``(hcpcs, year)`` is the same base-row pick the
|
||||||
|
per-code helper makes with ``PARTITION BY year`` inside its own
|
||||||
|
``WHERE hcpcs = ?``."""
|
||||||
|
targets = sorted({c.upper() for c in codes})
|
||||||
|
if not targets:
|
||||||
|
return {}
|
||||||
|
out: dict[str, list[tuple[int, str, str]]] = {}
|
||||||
|
for h, y, st, d in con.execute(
|
||||||
|
"SELECT hcpcs, year, status_code, description FROM pfs.rvu "
|
||||||
|
f"WHERE hcpcs {IN_TARGETS} AND (mod IS NULL OR mod = '') "
|
||||||
|
"QUALIFY row_number() OVER (PARTITION BY hcpcs, year ORDER BY mod NULLS FIRST) = 1 "
|
||||||
|
"ORDER BY hcpcs, year",
|
||||||
|
[targets],
|
||||||
|
).fetchall():
|
||||||
|
out.setdefault(h, []).append((int(y), st or "", d or ""))
|
||||||
|
return out
|
||||||
|
|||||||
@@ -11,7 +11,10 @@ winning over HCPCS/RVU on a duplicate ``(type, value, detail)``. The
|
|||||||
losing source isn't just dropped: it's recorded on the winning row's
|
losing source isn't just dropped: it's recorded on the winning row's
|
||||||
``confirmed_by`` as a ``source:item_key`` token (#703) — an independent
|
``confirmed_by`` as a ``source:item_key`` token (#703) — an independent
|
||||||
transcription of the same element cross-checks the winner without
|
transcription of the same element cross-checks the winner without
|
||||||
doubling the row count. No I/O here beyond what the caller hands in.
|
doubling the row count. ``extract_codes`` is ``extract_code`` for a whole
|
||||||
|
target list with the source lookups inverted — one pass over the corpus
|
||||||
|
rather than one per code (#698). No I/O here beyond what the caller hands
|
||||||
|
in.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -21,11 +24,15 @@ from typing import Any, Callable, Sequence
|
|||||||
|
|
||||||
from pfs.codetables import ElementRow, ReviewRow, is_missing_table_error
|
from pfs.codetables import ElementRow, ReviewRow, is_missing_table_error
|
||||||
from pfs.descriptors import (
|
from pfs.descriptors import (
|
||||||
|
IN_TARGETS,
|
||||||
DescriptorRun,
|
DescriptorRun,
|
||||||
Para,
|
Para,
|
||||||
descriptor_runs,
|
descriptor_runs,
|
||||||
|
descriptor_runs_bucketed,
|
||||||
hcpcs_long_description,
|
hcpcs_long_description,
|
||||||
|
hcpcs_long_descriptions,
|
||||||
rvu_descriptions,
|
rvu_descriptions,
|
||||||
|
rvu_descriptions_bucketed,
|
||||||
)
|
)
|
||||||
from pfs.elements import VOCAB, Element, ElementType, parse_descriptor
|
from pfs.elements import VOCAB, Element, ElementType, parse_descriptor
|
||||||
|
|
||||||
@@ -158,6 +165,18 @@ def _cpt_elements(
|
|||||||
raise
|
raise
|
||||||
if row is None:
|
if row is None:
|
||||||
return [], []
|
return [], []
|
||||||
|
return _cpt_elements_of_row(code, row, classify=classify)
|
||||||
|
|
||||||
|
|
||||||
|
def _cpt_elements_of_row(
|
||||||
|
code: str,
|
||||||
|
row: Sequence[Any],
|
||||||
|
*,
|
||||||
|
classify: Classifier | None = None,
|
||||||
|
) -> tuple[list[ElementRow], list[ReviewRow]]:
|
||||||
|
"""The parsing half of ``_cpt_elements``, over one already-fetched
|
||||||
|
``(edition_year, item_key, stem, elements, tail)`` row, so the
|
||||||
|
per-code query and the bucketed one share it exactly."""
|
||||||
edition_year, item_key, stem, elements, tail = row
|
edition_year, item_key, stem, elements, tail = row
|
||||||
stem_text = (stem or "").strip()
|
stem_text = (stem or "").strip()
|
||||||
tail_text = (tail or "").strip()
|
tail_text = (tail or "").strip()
|
||||||
@@ -185,26 +204,56 @@ def _cpt_elements(
|
|||||||
return rows, reviews
|
return rows, reviews
|
||||||
|
|
||||||
|
|
||||||
def extract_code(
|
def _cpt_elements_bucketed(
|
||||||
store: Any, con: Any, code: str, *, classify: Classifier | None = None
|
con: Any, codes: Sequence[str], *, classify: Classifier | None = None
|
||||||
|
) -> dict[str, tuple[list[ElementRow], list[ReviewRow]]]:
|
||||||
|
"""``_cpt_elements`` for every one of *codes* in one query — the same
|
||||||
|
newest-edition pick (the per-code ``ORDER BY edition_year DESC LIMIT
|
||||||
|
1``, expressed as a ``QUALIFY row_number()``) and the same
|
||||||
|
missing-``pfs.cpt_code`` tolerance (I4)."""
|
||||||
|
targets = sorted({c.upper() for c in codes})
|
||||||
|
if not targets:
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
rows = con.execute(
|
||||||
|
"SELECT code, edition_year, item_key, stem, elements, tail "
|
||||||
|
f"FROM pfs.cpt_code WHERE code {IN_TARGETS} QUALIFY row_number() OVER "
|
||||||
|
"(PARTITION BY code ORDER BY edition_year DESC) = 1",
|
||||||
|
[targets],
|
||||||
|
).fetchall()
|
||||||
|
except Exception as exc:
|
||||||
|
if is_missing_table_error(exc):
|
||||||
|
return {}
|
||||||
|
raise
|
||||||
|
return {r[0]: _cpt_elements_of_row(r[0], r[1:], classify=classify) for r in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def _assemble(
|
||||||
|
code: str,
|
||||||
|
runs: Sequence[DescriptorRun],
|
||||||
|
cpt: tuple[list[ElementRow], list[ReviewRow]],
|
||||||
|
long_desc: str,
|
||||||
|
years: Sequence[tuple[int, str, str]],
|
||||||
|
*,
|
||||||
|
classify: Classifier | None = None,
|
||||||
) -> Extraction:
|
) -> Extraction:
|
||||||
"""All sources for *code*; FR rows win over CPT, and CPT wins over
|
"""Merge one code's already-fetched sources into an ``Extraction``.
|
||||||
HCPCS/RVU rows for the same element — each source contributes its own
|
|
||||||
year."""
|
The single home of the precedence rules (FR > CPT > HCPCS/RVU, the
|
||||||
code = code.upper()
|
loser recorded on the winner's ``confirmed_by``), shared by the
|
||||||
|
per-code ``extract_code`` and the bucketed ``extract_codes`` so the
|
||||||
|
two paths cannot drift apart (#698)."""
|
||||||
merged: dict[tuple[str, str, str], ElementRow] = {}
|
merged: dict[tuple[str, str, str], ElementRow] = {}
|
||||||
reviews: list[ReviewRow] = []
|
reviews: list[ReviewRow] = []
|
||||||
for run in descriptor_runs(store, code):
|
for run in runs:
|
||||||
x = extract_run(run, classify=classify)
|
x = extract_run(run, classify=classify)
|
||||||
for r in x.rows:
|
for r in x.rows:
|
||||||
_merge_or_confirm(merged, r)
|
_merge_or_confirm(merged, r)
|
||||||
reviews.extend(x.reviews)
|
reviews.extend(x.reviews)
|
||||||
cpt_rows, cpt_reviews = _cpt_elements(con, code, classify=classify)
|
cpt_rows, cpt_reviews = cpt
|
||||||
for r in cpt_rows:
|
for r in cpt_rows:
|
||||||
_merge_or_confirm(merged, r)
|
_merge_or_confirm(merged, r)
|
||||||
reviews.extend(cpt_reviews)
|
reviews.extend(cpt_reviews)
|
||||||
long_desc = hcpcs_long_description(con, code)
|
|
||||||
years = rvu_descriptions(con, code)
|
|
||||||
if long_desc:
|
if long_desc:
|
||||||
y = years[-1][0] if years else 0
|
y = years[-1][0] if years else 0
|
||||||
for r in extract_text(code, long_desc, year=y, source="hcpcs").rows:
|
for r in extract_text(code, long_desc, year=y, source="hcpcs").rows:
|
||||||
@@ -214,3 +263,51 @@ def extract_code(
|
|||||||
_merge_or_confirm(merged, r)
|
_merge_or_confirm(merged, r)
|
||||||
ordered = sorted(merged.values(), key=lambda r: (r.type, r.value, r.detail))
|
ordered = sorted(merged.values(), key=lambda r: (r.type, r.value, r.detail))
|
||||||
return Extraction(code, tuple(ordered), tuple(reviews))
|
return Extraction(code, tuple(ordered), tuple(reviews))
|
||||||
|
|
||||||
|
|
||||||
|
def extract_code(
|
||||||
|
store: Any, con: Any, code: str, *, classify: Classifier | None = None
|
||||||
|
) -> Extraction:
|
||||||
|
"""All sources for *code*; FR rows win over CPT, and CPT wins over
|
||||||
|
HCPCS/RVU rows for the same element — each source contributes its own
|
||||||
|
year."""
|
||||||
|
code = code.upper()
|
||||||
|
return _assemble(
|
||||||
|
code,
|
||||||
|
descriptor_runs(store, code),
|
||||||
|
_cpt_elements(con, code, classify=classify),
|
||||||
|
hcpcs_long_description(con, code),
|
||||||
|
rvu_descriptions(con, code),
|
||||||
|
classify=classify,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_codes(
|
||||||
|
store: Any, con: Any, codes: Sequence[str], *, classify: Classifier | None = None
|
||||||
|
) -> dict[str, Extraction]:
|
||||||
|
"""``extract_code`` for every one of *codes*, with the source lookups
|
||||||
|
inverted: ONE streaming pass over ``fr_anchors`` and ONE query each
|
||||||
|
over ``pfs.cpt_code``, ``terminology.hcpcs_level_2`` and ``pfs.rvu``,
|
||||||
|
instead of two unindexed ``fr_anchors`` scans plus three replica
|
||||||
|
queries per code (#698). ``extract_codes(store, con, [c])[c] ==
|
||||||
|
extract_code(store, con, c)``: the bucketed lookups select the same
|
||||||
|
rows per code, and ``_assemble`` does the merging for both paths.
|
||||||
|
|
||||||
|
Keyed in ``sorted(set(...))`` order, one entry per distinct code — a
|
||||||
|
code no source mentions still gets an empty ``Extraction``."""
|
||||||
|
targets = sorted({c.upper() for c in codes})
|
||||||
|
runs = descriptor_runs_bucketed(store, targets)
|
||||||
|
cpt = _cpt_elements_bucketed(con, targets, classify=classify)
|
||||||
|
long_descs = hcpcs_long_descriptions(con, targets)
|
||||||
|
years = rvu_descriptions_bucketed(con, targets)
|
||||||
|
return {
|
||||||
|
code: _assemble(
|
||||||
|
code,
|
||||||
|
runs.get(code, ()),
|
||||||
|
cpt.get(code, ([], [])),
|
||||||
|
long_descs.get(code, ""),
|
||||||
|
years.get(code, ()),
|
||||||
|
classify=classify,
|
||||||
|
)
|
||||||
|
for code in targets
|
||||||
|
}
|
||||||
|
|||||||
@@ -99,10 +99,17 @@ def con(monkeypatch, restore_families):
|
|||||||
c.close()
|
c.close()
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_extractions(store, con, codes, *, classify=None):
|
||||||
|
"""``pfs.extract.extract_codes``'s shape with nothing extracted — the
|
||||||
|
stand-in for tests that only care which codes the CLI targeted."""
|
||||||
|
return {code: Extraction(code, (), ()) for code in codes}
|
||||||
|
|
||||||
|
|
||||||
class TestElements:
|
class TestElements:
|
||||||
def test_writes_rows_and_publishes(self, con, monkeypatch):
|
def test_writes_rows_and_publishes(self, con, monkeypatch):
|
||||||
def fake_extract(store, con_, code, *, classify=None):
|
def fake_extract(store, con_, codes, *, classify=None):
|
||||||
return Extraction(
|
return {
|
||||||
|
code: Extraction(
|
||||||
code,
|
code,
|
||||||
(
|
(
|
||||||
ElementRow(
|
ElementRow(
|
||||||
@@ -120,8 +127,10 @@ class TestElements:
|
|||||||
),
|
),
|
||||||
(),
|
(),
|
||||||
)
|
)
|
||||||
|
for code in codes
|
||||||
|
}
|
||||||
|
|
||||||
monkeypatch.setattr(pfs_cli, "extract_code", fake_extract)
|
monkeypatch.setattr(pfs_cli, "extract_codes", fake_extract)
|
||||||
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
||||||
assert res.exit_code == 0, res.output
|
assert res.exit_code == 0, res.output
|
||||||
assert "G0556: 1 elements, 0 for review" in res.output
|
assert "G0556: 1 elements, 0 for review" in res.output
|
||||||
@@ -131,8 +140,9 @@ class TestElements:
|
|||||||
def test_summary_line_reports_confirmed_count_when_nonzero(self, con, monkeypatch):
|
def test_summary_line_reports_confirmed_count_when_nonzero(self, con, monkeypatch):
|
||||||
# #703: the elements summary surfaces confirmed_by, but only when
|
# #703: the elements summary surfaces confirmed_by, but only when
|
||||||
# there's something to report.
|
# there's something to report.
|
||||||
def fake_extract(store, con_, code, *, classify=None):
|
def fake_extract(store, con_, codes, *, classify=None):
|
||||||
return Extraction(
|
return {
|
||||||
|
code: Extraction(
|
||||||
code,
|
code,
|
||||||
(
|
(
|
||||||
ElementRow(
|
ElementRow(
|
||||||
@@ -163,8 +173,10 @@ class TestElements:
|
|||||||
),
|
),
|
||||||
(),
|
(),
|
||||||
)
|
)
|
||||||
|
for code in codes
|
||||||
|
}
|
||||||
|
|
||||||
monkeypatch.setattr(pfs_cli, "extract_code", fake_extract)
|
monkeypatch.setattr(pfs_cli, "extract_codes", fake_extract)
|
||||||
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
||||||
assert res.exit_code == 0, res.output
|
assert res.exit_code == 0, res.output
|
||||||
assert "G0556: 2 elements, 0 for review, 1 confirmed by a second source" in (
|
assert "G0556: 2 elements, 0 for review, 1 confirmed by a second source" in (
|
||||||
@@ -172,21 +184,13 @@ class TestElements:
|
|||||||
)
|
)
|
||||||
|
|
||||||
def test_summary_line_omits_confirmed_when_none(self, con, monkeypatch):
|
def test_summary_line_omits_confirmed_when_none(self, con, monkeypatch):
|
||||||
monkeypatch.setattr(
|
monkeypatch.setattr(pfs_cli, "extract_codes", _empty_extractions)
|
||||||
pfs_cli,
|
|
||||||
"extract_code",
|
|
||||||
lambda s, c, code, *, classify=None: Extraction(code, (), ()),
|
|
||||||
)
|
|
||||||
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
res = runner.invoke(app, ["pfs", "elements", "--code", "g0556", "--no-llm"])
|
||||||
assert res.exit_code == 0, res.output
|
assert res.exit_code == 0, res.output
|
||||||
assert "confirmed" not in res.output
|
assert "confirmed" not in res.output
|
||||||
|
|
||||||
def test_dry_run_does_not_write(self, con, monkeypatch):
|
def test_dry_run_does_not_write(self, con, monkeypatch):
|
||||||
monkeypatch.setattr(
|
monkeypatch.setattr(pfs_cli, "extract_codes", _empty_extractions)
|
||||||
pfs_cli,
|
|
||||||
"extract_code",
|
|
||||||
lambda s, c, code, *, classify=None: Extraction(code, (), ()),
|
|
||||||
)
|
|
||||||
|
|
||||||
def fail_batch():
|
def fail_batch():
|
||||||
raise AssertionError("--dry-run must not open a RW duckdb_batch connection")
|
raise AssertionError("--dry-run must not open a RW duckdb_batch connection")
|
||||||
@@ -213,9 +217,9 @@ class TestElements:
|
|||||||
seen = []
|
seen = []
|
||||||
monkeypatch.setattr(
|
monkeypatch.setattr(
|
||||||
pfs_cli,
|
pfs_cli,
|
||||||
"extract_code",
|
"extract_codes",
|
||||||
lambda s, c, code, *, classify=None: (
|
lambda s, c, codes, *, classify=None: (
|
||||||
seen.append(code) or Extraction(code, (), ())
|
seen.extend(codes) or _empty_extractions(s, c, codes)
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
res = runner.invoke(
|
res = runner.invoke(
|
||||||
@@ -230,9 +234,9 @@ class TestElements:
|
|||||||
seen = []
|
seen = []
|
||||||
monkeypatch.setattr(
|
monkeypatch.setattr(
|
||||||
pfs_cli,
|
pfs_cli,
|
||||||
"extract_code",
|
"extract_codes",
|
||||||
lambda s, c, code, *, classify=None: (
|
lambda s, c, codes, *, classify=None: (
|
||||||
seen.append(code) or Extraction(code, (), ())
|
seen.extend(codes) or _empty_extractions(s, c, codes)
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
res = runner.invoke(
|
res = runner.invoke(
|
||||||
@@ -256,17 +260,17 @@ class TestElements:
|
|||||||
|
|
||||||
|
|
||||||
class TestCodesForWarnSlow:
|
class TestCodesForWarnSlow:
|
||||||
"""``_codes_for``'s "~20 min of SQL" warning describes the
|
"""``_codes_for``'s target-count note is for ``elements``, whose
|
||||||
per-code LLM classification pass ``elements`` runs — it does not
|
``--all-payable`` run makes one model call per unplaced line — not for
|
||||||
apply to ``lineage --all-payable`` (one inverted pass, seconds)."""
|
``lineage --all-payable``, which is pure SQL/regex."""
|
||||||
|
|
||||||
def test_warns_by_default(self, con, caplog):
|
def test_warns_by_default(self, con, caplog):
|
||||||
pfs_cli._codes_for(con, [], [], True)
|
pfs_cli._codes_for(con, [], [], True)
|
||||||
assert "20 min" in caplog.text
|
assert "targeting 1 codes" in caplog.text
|
||||||
|
|
||||||
def test_silent_when_warn_slow_is_false(self, con, caplog):
|
def test_silent_when_warn_slow_is_false(self, con, caplog):
|
||||||
pfs_cli._codes_for(con, [], [], True, warn_slow=False)
|
pfs_cli._codes_for(con, [], [], True, warn_slow=False)
|
||||||
assert "20 min" not in caplog.text
|
assert "targeting" not in caplog.text
|
||||||
|
|
||||||
|
|
||||||
class TestLineage:
|
class TestLineage:
|
||||||
|
|||||||
@@ -9,10 +9,14 @@ import pytest
|
|||||||
|
|
||||||
from pfs.descriptors import (
|
from pfs.descriptors import (
|
||||||
DescriptorRun,
|
DescriptorRun,
|
||||||
|
_stem_pattern_any,
|
||||||
descriptor_runs,
|
descriptor_runs,
|
||||||
|
descriptor_runs_bucketed,
|
||||||
hcpcs_long_description,
|
hcpcs_long_description,
|
||||||
|
hcpcs_long_descriptions,
|
||||||
rule_year_of,
|
rule_year_of,
|
||||||
rvu_descriptions,
|
rvu_descriptions,
|
||||||
|
rvu_descriptions_bucketed,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -293,3 +297,164 @@ class TestDescriptorSpan:
|
|||||||
]
|
]
|
||||||
assert run.stem.text.startswith("G0502 (Initial psychiatric")
|
assert run.stem.text.startswith("G0502 (Initial psychiatric")
|
||||||
assert "Chronic care management prose" not in run.stem.text
|
assert "Chronic care management prose" not in run.stem.text
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def wide_store():
|
||||||
|
"""Three rule items naming several codes, so the inverted pass has to
|
||||||
|
reset its open runs at an item boundary, carry two stems through the
|
||||||
|
same element paragraphs, and order one code's runs across items."""
|
||||||
|
s = _Store()
|
||||||
|
s.con.executemany(
|
||||||
|
"INSERT INTO items VALUES (?,?,?)",
|
||||||
|
[
|
||||||
|
("AAAAAAAA", "Medicare Program; CY 2015 PFS Final Rule", "2014-11-13"),
|
||||||
|
("BBBBBBBB", "Medicare Program; CY 2021 PFS Final Rule", "2020-12-28"),
|
||||||
|
("CCCCCCCC", "Revisions to Payment Policies", "2016-11-15"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
rows = [
|
||||||
|
# AAAAAAAA — a plain stem run, then an unrelated prose paragraph
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
10,
|
||||||
|
100,
|
||||||
|
"Comment: commenters noted the CPT panel created a code.",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
11,
|
||||||
|
100,
|
||||||
|
"We use the new CPT code 99490 (Chronic care management services, at least "
|
||||||
|
"20 minutes of clinical staff time directed by a physician or other "
|
||||||
|
"qualified health care professional, per calendar month, with the "
|
||||||
|
"following required elements:",
|
||||||
|
),
|
||||||
|
("AAAAAAAA", 12, 100, "Consent;"),
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
13,
|
||||||
|
101,
|
||||||
|
"Comprehensive care plan established, implemented, revised, or monitored).",
|
||||||
|
),
|
||||||
|
("AAAAAAAA", 14, 101, "Response: it is our preference to use CPT codes."),
|
||||||
|
# BBBBBBBB — two stems back to back, so 99487's run must absorb the
|
||||||
|
# element paragraphs that follow 99489's stem too
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
20,
|
||||||
|
200,
|
||||||
|
"CPT code 99487 (Complex chronic care management services, with the "
|
||||||
|
"following required elements:",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
21,
|
||||||
|
200,
|
||||||
|
"99489 (Each additional 30 minutes of clinical staff time, per calendar "
|
||||||
|
"month, with the following required elements:",
|
||||||
|
),
|
||||||
|
("BBBBBBBB", 22, 200, "Consent;"),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
23,
|
||||||
|
201,
|
||||||
|
"Provide 24/7 access for urgent needs to care team/practitioner;",
|
||||||
|
),
|
||||||
|
("BBBBBBBB", 24, 201, "We received many comments on this proposal."),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
25,
|
||||||
|
201,
|
||||||
|
"( 9) 99439 (code for non-complex chronic care management).",
|
||||||
|
),
|
||||||
|
# CCCCCCCC — an earlier-dated item mentioning 99490 again, so the
|
||||||
|
# per-code ORDER BY date_published puts it first
|
||||||
|
(
|
||||||
|
"CCCCCCCC",
|
||||||
|
30,
|
||||||
|
300,
|
||||||
|
"We finalized 99490 (Chronic care management services, per calendar "
|
||||||
|
"month, with the following required elements:",
|
||||||
|
),
|
||||||
|
("CCCCCCCC", 31, 300, "Consent;"),
|
||||||
|
# an odd-length target code, exercising the literal branch of
|
||||||
|
# `_stem_pattern_any`
|
||||||
|
(
|
||||||
|
"CCCCCCCC",
|
||||||
|
32,
|
||||||
|
300,
|
||||||
|
"And HCPCS code G0556X (Advanced primary care management services, per "
|
||||||
|
"calendar month, with the following elements:",
|
||||||
|
),
|
||||||
|
("CCCCCCCC", 33, 300, "Consent;"),
|
||||||
|
]
|
||||||
|
s.con.executemany(
|
||||||
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
||||||
|
[(k, pid, pg, pid, t) for k, pid, pg, t in rows],
|
||||||
|
)
|
||||||
|
yield s
|
||||||
|
s.close()
|
||||||
|
|
||||||
|
|
||||||
|
class TestDescriptorRunsBucketed:
|
||||||
|
"""#698: one streaming pass over ``fr_anchors`` must reproduce
|
||||||
|
``descriptor_runs`` exactly, code for code."""
|
||||||
|
|
||||||
|
_CODES = ("99490", "99487", "99489", "99439", "G0556X", "00000")
|
||||||
|
|
||||||
|
def test_matches_descriptor_runs_per_code(self, wide_store):
|
||||||
|
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
|
||||||
|
assert sorted(bucketed) == sorted(self._CODES)
|
||||||
|
for c in self._CODES:
|
||||||
|
assert bucketed[c] == descriptor_runs(wide_store, c), c
|
||||||
|
|
||||||
|
def test_finds_the_runs_the_fixture_plants(self, wide_store):
|
||||||
|
bucketed = descriptor_runs_bucketed(wide_store, self._CODES)
|
||||||
|
# 99490 stems in two items, ordered by the rule's publication date
|
||||||
|
assert [r.item_key for r in bucketed["99490"]] == ["AAAAAAAA", "CCCCCCCC"]
|
||||||
|
# 99487's run keeps taking element paragraphs past 99489's stem
|
||||||
|
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22, 23]
|
||||||
|
assert [p.p_id for p in bucketed["99489"][0].elements] == [22, 23]
|
||||||
|
# an enumeration line is not a stem, and an unmentioned code is empty
|
||||||
|
assert bucketed["99439"] == [] and bucketed["00000"] == []
|
||||||
|
# the odd-length literal branch of `_stem_pattern_any` still matches
|
||||||
|
assert [r.stem.p_id for r in bucketed["G0556X"]] == [32]
|
||||||
|
|
||||||
|
def test_max_elements_cap_matches(self, wide_store):
|
||||||
|
bucketed = descriptor_runs_bucketed(wide_store, ["99487"], max_elements=2)
|
||||||
|
assert bucketed["99487"] == descriptor_runs(wide_store, "99487", max_elements=2)
|
||||||
|
assert [p.p_id for p in bucketed["99487"][0].elements] == [21, 22]
|
||||||
|
|
||||||
|
def test_no_codes_is_an_empty_result(self, wide_store):
|
||||||
|
assert descriptor_runs_bucketed(wide_store, []) == {}
|
||||||
|
|
||||||
|
def test_generic_token_only_when_every_target_is_five_wide(self):
|
||||||
|
# The all-five-wide fast path compiles the character class alone;
|
||||||
|
# an odd-length target adds its own literal alternative.
|
||||||
|
assert "|" not in _stem_pattern_any(["99490", "G0556"]).pattern
|
||||||
|
assert "G0556X" in _stem_pattern_any(["99490", "G0556X"]).pattern
|
||||||
|
|
||||||
|
|
||||||
|
class TestBucketedReplicaLookups:
|
||||||
|
"""The DuckDB half of the inversion: one query per table for every
|
||||||
|
target code, matching the per-code helpers row for row."""
|
||||||
|
|
||||||
|
def test_hcpcs_long_descriptions_matches_per_code(self, con):
|
||||||
|
codes = ["G0556", "G9999"]
|
||||||
|
bucketed = hcpcs_long_descriptions(con, codes)
|
||||||
|
for c in codes:
|
||||||
|
assert bucketed.get(c, "") == hcpcs_long_description(con, c), c
|
||||||
|
assert bucketed["G0556"].startswith("Advanced primary care")
|
||||||
|
assert "G9999" not in bucketed
|
||||||
|
|
||||||
|
def test_rvu_descriptions_matches_per_code(self, con):
|
||||||
|
codes = ["99490", "99999"]
|
||||||
|
bucketed = rvu_descriptions_bucketed(con, codes)
|
||||||
|
for c in codes:
|
||||||
|
assert bucketed.get(c, []) == rvu_descriptions(con, c), c
|
||||||
|
assert [y for y, _s, _d in bucketed["99490"]] == [2015, 2022]
|
||||||
|
|
||||||
|
def test_no_codes_is_an_empty_result(self, con):
|
||||||
|
assert hcpcs_long_descriptions(con, []) == {}
|
||||||
|
assert rvu_descriptions_bucketed(con, []) == {}
|
||||||
|
|||||||
@@ -12,11 +12,14 @@ from pfs.descriptors import DescriptorRun, Para, descriptor_runs
|
|||||||
from pfs.extract import (
|
from pfs.extract import (
|
||||||
Extraction,
|
Extraction,
|
||||||
_cpt_elements,
|
_cpt_elements,
|
||||||
|
_cpt_elements_bucketed,
|
||||||
_merge_or_confirm,
|
_merge_or_confirm,
|
||||||
extract_code,
|
extract_code,
|
||||||
|
extract_codes,
|
||||||
extract_run,
|
extract_run,
|
||||||
extract_text,
|
extract_text,
|
||||||
)
|
)
|
||||||
|
from pfs.families import HAND_FAMILIES
|
||||||
|
|
||||||
STEM = Para(
|
STEM = Para(
|
||||||
"JJ6AM5HJ",
|
"JJ6AM5HJ",
|
||||||
@@ -534,3 +537,192 @@ class TestMergeOrConfirm:
|
|||||||
_merge_or_confirm(merged, _el(source="hcpcs"))
|
_merge_or_confirm(merged, _el(source="hcpcs"))
|
||||||
(row,) = merged.values()
|
(row,) = merged.values()
|
||||||
assert row.confirmed_by == "cpt:GQGTPGYV,hcpcs:"
|
assert row.confirmed_by == "cpt:GQGTPGYV,hcpcs:"
|
||||||
|
|
||||||
|
|
||||||
|
#: The 17 hand-family codes, the fixture universe #698 is measured
|
||||||
|
#: against: whatever `extract_codes` does in one inverted pass must match
|
||||||
|
#: `extract_code`'s per-code path for every one of them.
|
||||||
|
FIXTURE_CODES = tuple(c for fam in HAND_FAMILIES.values() for c in fam.codes)
|
||||||
|
|
||||||
|
|
||||||
|
def _wide_store_con():
|
||||||
|
con = _sqlite_store_con()
|
||||||
|
con.executemany(
|
||||||
|
"INSERT INTO items VALUES (?,?,?)",
|
||||||
|
[
|
||||||
|
("AAAAAAAA", "Medicare Program; CY 2015 PFS Final Rule", "2014-11-13"),
|
||||||
|
("BBBBBBBB", "Medicare Program; CY 2025 PFS Final Rule", "2024-11-01"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
rows = [
|
||||||
|
("AAAAAAAA", 10, 100, "Comment: commenters noted the CPT panel created codes."),
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
11,
|
||||||
|
100,
|
||||||
|
"We adopted CPT code 99490 (Chronic care management services, at least 20 "
|
||||||
|
"minutes of clinical staff time directed by a physician or other "
|
||||||
|
"qualified health care professional, per calendar month, with the "
|
||||||
|
"following required elements:",
|
||||||
|
),
|
||||||
|
("AAAAAAAA", 12, 100, "Consent;"),
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
13,
|
||||||
|
101,
|
||||||
|
"Chronic conditions place the patient at significant risk of death, acute "
|
||||||
|
"exacerbation/decompensation, or functional decline;",
|
||||||
|
),
|
||||||
|
("AAAAAAAA", 14, 101, "A line the vocabulary does not know about at all;"),
|
||||||
|
(
|
||||||
|
"AAAAAAAA",
|
||||||
|
15,
|
||||||
|
101,
|
||||||
|
"Comprehensive care plan established, implemented, revised, or monitored).",
|
||||||
|
),
|
||||||
|
("AAAAAAAA", 16, 101, "Response: we agree with the commenters."),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
20,
|
||||||
|
200,
|
||||||
|
"HCPCS code G0556 ( Advanced primary care management services provided by "
|
||||||
|
"clinical staff and directed by a physician, per calendar month, with the "
|
||||||
|
"following elements, as appropriate:",
|
||||||
|
),
|
||||||
|
("BBBBBBBB", 21, 200, "Consent;"),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
22,
|
||||||
|
200,
|
||||||
|
"Provide 24/7 access for urgent needs to care team/practitioner;",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"BBBBBBBB",
|
||||||
|
23,
|
||||||
|
201,
|
||||||
|
"CPT code 99495 (Transitional care management services, with the following "
|
||||||
|
"required elements:",
|
||||||
|
),
|
||||||
|
("BBBBBBBB", 24, 201, "Consent;"),
|
||||||
|
("BBBBBBBB", 25, 201, "( 9) 99439 (code for non-complex chronic care)."),
|
||||||
|
]
|
||||||
|
con.executemany(
|
||||||
|
"INSERT INTO fr_anchors VALUES (?,?,?,?,?)",
|
||||||
|
[(k, pid, pg, pid, t) for k, pid, pg, t in rows],
|
||||||
|
)
|
||||||
|
return con
|
||||||
|
|
||||||
|
|
||||||
|
class TestExtractCodesBucketed:
|
||||||
|
"""#698: `extract_codes` inverts the per-code loop — one streaming
|
||||||
|
pass over `fr_anchors` plus one query each over `pfs.cpt_code`,
|
||||||
|
`terminology.hcpcs_level_2` and `pfs.rvu` — and must yield exactly
|
||||||
|
what `extract_code` yields, code for code."""
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def store(self):
|
||||||
|
con = _wide_store_con()
|
||||||
|
yield _Store(con)
|
||||||
|
con.close()
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def con(self):
|
||||||
|
c = _duckdb_con()
|
||||||
|
c.executemany(
|
||||||
|
"INSERT INTO pfs.rvu VALUES (?,?,?,?,?,?)",
|
||||||
|
[
|
||||||
|
("99490", "", "Chron care mgmt srvc 20 min", "A", 1.0, 2015),
|
||||||
|
("99490", None, "Chrnc care mgmt staff 1st 20", "A", 1.2, 2026),
|
||||||
|
("99490", "26", "ignored modifier row", "A", 1.0, 2026),
|
||||||
|
("G0556", "", "Adv prim care mgmt lvl 1, consent", "A", 2.0, 2026),
|
||||||
|
("99497", "", "Advncd care plan, per calendar month", "A", 1.0, 2026),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
c.executemany(
|
||||||
|
"INSERT INTO terminology.hcpcs_level_2 VALUES (?,?,?,?)",
|
||||||
|
[
|
||||||
|
("G0556", "… per calendar month, consent …", "10", "3"),
|
||||||
|
("G0556", "a lower-seqnum transcription", "5", "1"),
|
||||||
|
("G0557", "… 24/7 access, per calendar month …", "1", "1"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
_insert_cpt_code(
|
||||||
|
c,
|
||||||
|
2022,
|
||||||
|
"OLDEDITN",
|
||||||
|
"99490",
|
||||||
|
stem="An older edition's stem",
|
||||||
|
elements=["Consent;"],
|
||||||
|
tail="per calendar month.",
|
||||||
|
)
|
||||||
|
_insert_cpt_code(
|
||||||
|
c,
|
||||||
|
2024,
|
||||||
|
"GQGTPGYV",
|
||||||
|
"99490",
|
||||||
|
stem="Chronic care management services",
|
||||||
|
elements=["Consent;", "A line the vocabulary does not know at all;"],
|
||||||
|
tail="first 20 minutes, per calendar month.",
|
||||||
|
)
|
||||||
|
_insert_cpt_code(
|
||||||
|
c,
|
||||||
|
2024,
|
||||||
|
"TCMENTRY",
|
||||||
|
"99495",
|
||||||
|
stem="Transitional care management services",
|
||||||
|
elements=["Consent;"],
|
||||||
|
tail="within 14 days of discharge.",
|
||||||
|
)
|
||||||
|
yield c
|
||||||
|
c.close()
|
||||||
|
|
||||||
|
def test_matches_extract_code_for_every_fixture_code(self, store, con):
|
||||||
|
bucketed = extract_codes(store, con, FIXTURE_CODES)
|
||||||
|
assert sorted(bucketed) == sorted(set(FIXTURE_CODES))
|
||||||
|
for c in FIXTURE_CODES:
|
||||||
|
assert bucketed[c] == extract_code(store, con, c), c
|
||||||
|
# the fixture really exercises all four sources, so the equality
|
||||||
|
# above is not a comparison of 17 empty extractions
|
||||||
|
sources = {r.source for x in bucketed.values() for r in x.rows}
|
||||||
|
assert sources == {"fr", "cpt", "hcpcs", "rvu"}
|
||||||
|
assert any(x.reviews for x in bucketed.values())
|
||||||
|
|
||||||
|
def test_matches_extract_code_with_a_classifier(self, store, con):
|
||||||
|
classify = lambda text, choices: ( # noqa: E731
|
||||||
|
"community-coordination" if "vocabulary" in text else None
|
||||||
|
)
|
||||||
|
bucketed = extract_codes(store, con, FIXTURE_CODES, classify=classify)
|
||||||
|
for c in FIXTURE_CODES:
|
||||||
|
assert bucketed[c] == extract_code(store, con, c, classify=classify), c
|
||||||
|
assert any(
|
||||||
|
r.value == "community-coordination"
|
||||||
|
for x in bucketed.values()
|
||||||
|
for r in x.rows
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_lowercase_input_is_normalised_and_deduped(self, store, con):
|
||||||
|
assert list(extract_codes(store, con, ["g0556", "G0556", "99490"])) == [
|
||||||
|
"99490",
|
||||||
|
"G0556",
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_no_codes_is_an_empty_result(self, store, con):
|
||||||
|
assert extract_codes(store, con, []) == {}
|
||||||
|
|
||||||
|
def test_replica_with_no_cpt_tables_at_all_does_not_raise(self, store):
|
||||||
|
# Same I4 tolerance the per-code path has: `elements --dry-run`
|
||||||
|
# never calls ensure_tables, so a pre-cpt-ingest replica must
|
||||||
|
# degrade to no CPT rows, not a CatalogException.
|
||||||
|
con = _duckdb_con_no_cpt()
|
||||||
|
try:
|
||||||
|
bucketed = extract_codes(store, con, ["99490"])
|
||||||
|
finally:
|
||||||
|
con.close()
|
||||||
|
assert not any(r.source == "cpt" for r in bucketed["99490"].rows)
|
||||||
|
|
||||||
|
def test_non_missing_table_error_propagates(self):
|
||||||
|
with pytest.raises(RuntimeError, match="disk I/O error"):
|
||||||
|
_cpt_elements_bucketed(TestCptElementsErrors._RaisingCon(), ["99490"])
|
||||||
|
|
||||||
|
def test_no_codes_never_touches_the_replica(self):
|
||||||
|
assert _cpt_elements_bucketed(TestCptElementsErrors._RaisingCon(), []) == {}
|
||||||
|
|||||||
Reference in New Issue
Block a user