Files
stack/src/pfs/guidance.py
kert 6bff63b499
Some checks failed
CI / lint (push) Successful in 45s
CI / notebooks-smoke (push) Has been cancelled
Deploy / notebooks (push) Has been cancelled
Deploy / zotero (push) Has been cancelled
Deploy / docs (push) Has been cancelled
Deploy / api (push) Has been cancelled
Deploy / llm (push) Has been cancelled
Deploy / mc (push) Has been cancelled
Deploy / report (push) Has been cancelled
CI / test (push) Has been cancelled
Infra CI / docs (push) Has been cancelled
Infra CI / api (push) Has been cancelled
Infra CI / llm (push) Has been cancelled
Infra CI / mc (push) Has been cancelled
Infra CI / notebooks (push) Has been cancelled
Infra CI / zotero (push) Has been cancelled
feat(pfs,llm): resolve MLN products/articles to bib items; locate IOM section pages in chapter PDFs (refs #705)
- pfs.guidance: MLN_RE accepts ICN/MLN product numbers ("ICN MLN909188",
  "ICN 909289"), MLN_MATTERS_RE captures MM/SE article numbers; mln_refs
  normalises both to "MLN <n>" / "MLN Matters MM<n>"; resolve_mln finds
  the bib item whose URL or title carries the identifier (word-bounded).
- pfs.guidance.iom_section_page: locates an IOM section heading inside the
  chapter PDF attached to the resolved item (llm.pages.locate_section —
  the body heading is the one followed by its "(Rev." line, which the
  table of contents lacks); cached per item/section and per PDF.
- pfs.codetables.GuidanceRow.page (+ column, added in place by
  ensure_tables; legacy NULL reads 0).
- llm.lineage: resolved IOM/MLN guidance rows link to the bib item's URL,
  with #page=N when the section was located.
2026-09-11 17:51:20 -04:00

477 lines
20 KiB
Python

"""Sub-regulatory crosswalk: CFR sections and IOM manual chapters/sections
a code family's rules (and the CPT manual) cite, with FR paragraph
provenance (#689, task 6 of the 2026-09-09 code-family-anchors plan).
Two sources, one extraction pass over each hit's text:
1. ``harvest`` — FR paragraphs (``fr_anchors`` in the bib SQLite) that
name any of a family's codes. Cheap and needs no environment (the
bib sqlite has no ``.env`` dependency), unlike the pgvector rule
chunks the controller notes as an equally-valid but heavier
alternative source.
2. ``harvest_cpt`` — the CPT codebook's own guideline text
(``pfs.cpt_section.guideline``, the heading that owns the family's
codes) and ``see``/``other`` instructions (``pfs.cpt_instruction``)
that mention Medicare policy, from the newest ingested edition.
Both funnel through ``_extract``, which runs three regex families
(CFR, IOM, MLN) over one paragraph/guideline's text and resolves each
hit against the bib (``resolve_cfr``/``resolve_iom``/``resolve_mln``) — unresolved
references keep an empty ``item_key`` (the canonical locator is kept
either way, Resolutions §fr_anchors). Rows are written by
``pfs.codetables.write_guidance``/``read_guidance`` into
``pfs.code_guidance``, deduped on ``(family, code, kind, locator,
item_key_src, p_id_src)`` — the same citation named in the same
paragraph for two codes in one family (a common shape: "99490 and
99439 ... see § 410.78") legitimately yields two rows, but a citation
matched twice by overlapping regex branches (e.g. both the titled and
bare CFR forms firing on the same span) collapses to one.
"""
from __future__ import annotations
import re
from pathlib import Path
from typing import Any, Sequence
from bib.cfrlink import canonical, item_for, parse_cite
from pfs.codetables import GuidanceRow, is_missing_table_error
from pfs.families import code_pattern, codes_in
# manual display name (lower-cased, as it appears in FR prose) -> CMS
# publication number. Controller ruling (task-6-context.md): only these
# six IOM manuals are in scope for #689.
_MANUAL_PUB = {
"claims processing": "100-04",
"benefit policy": "100-02",
"program integrity": "100-08",
"national coverage determinations": "100-03",
"managed care": "100-16",
"financial management": "100-06",
}
# "42 CFR 410.78(a)(3)" / "42 C.F.R. § 425.400" / "42 CFR Part 425" /
# "42 CFR parts 405, 414, and 426" — title-bearing form. The paragraph
# suffix and part-only form are left to bib.cfrlink.parse_cite (fed
# "<title> CFR <section>") to interpret, so this regex just has to
# isolate title/partword/part/dec/paras. Named groups (Ruling A10): the
# "parts" keyword capture lets ``_cfr_refs`` tell a plural part-list
# head ("parts 405, ...") from a singular one-off part cite ("Part
# 425") — only the plural head triggers the bare-part-number
# continuation grammar below.
CFR_RE = re.compile(
r"\b(?P<title>\d{1,2})\s*C\.?F\.?R\.?\s*"
r"(?:(?P<partword>[Pp]arts?)\s*)?(?P<part>\d{2,4})"
r"(?:\.(?P<dec>\d+[a-z]?)(?P<paras>(?:\([^)\s]+\))*))?"
)
#: Bare "§ 410.78(a)(3)" — no title present. Every paragraph this module
#: scans is a PFS rule (or the CPT manual's own Medicare-policy prose),
#: so the title is inferred as 42 (Resolutions: kept in memory only —
#: the stored ``locator`` is always the canonical "42 CFR ..." form,
#: never a flag).
BARE_CFR_RE = re.compile(r"§\s*(\d{2,4})\.(\d+[a-z]?)((?:\([^)\s]+\))*)")
#: "§§ 410.26 and 410.32" — the two-section plural form (Ruling A10:
#: the only bare-§ list shape handled; a longer comma-separated bare
#: list is not, since "§" alone doesn't carry the same list-continuation
#: convention FR drafting uses after a titled "CFR" cite).
_BARE_CFR_PLURAL_RE = re.compile(
r"§§\s*(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)\s+and\s+"
r"(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)"
)
#: Separator between list items in a CFR citation run: ", ", ", and ",
#: " and ", " or ", " through ", etc.
_CFR_LIST_SEP = r"(?:\s*,\s*(?:and\s+|or\s+|through\s+)?|\s+(?:and|or|through)\s+)"
#: A continuation token after a titled section-form head ("410.26",
#: "410.26(a)(3)") — no "CFR" keyword, so it must immediately follow a
#: list separator (Ruling A10) or it is prose, not a citation.
# No "^" anchor: Pattern.match(text, pos) already requires the match to
# start exactly at pos — "^" would instead (and wrongly) require pos to
# be the real start of the whole string (or MULTILINE-after-newline),
# so every continuation past the first character would silently fail.
_CFR_CONT_SECTION_RE = re.compile(
_CFR_LIST_SEP + r"(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)"
)
#: A continuation token after a plural "parts" head ("405, 414, and
#: 426") — bare part numbers, no decimal (the negative lookahead keeps
#: "414.5" from being misread as a bare part).
_CFR_CONT_PART_RE = re.compile(_CFR_LIST_SEP + r"(\d{2,4})(?!\.\d)")
#: "Pub. 100-04, chapter 12" / "Publication 100-04 ... Chapter 12" (the
#: pub-number form) or "Medicare Claims Processing Manual ... Chapter
#: 12, Section 30.6.4" (the manual-name form) — either may open the
#: match; both close on an optional trailing "Section N.N".
IOM_RE = re.compile(
r"(?:Pub(?:lication)?\.?\s*100-0(\d)"
r"|(Claims Processing|Benefit Policy|Program Integrity|"
r"National Coverage Determinations|Managed Care|Financial Management)"
r"\s+Manual)"
r"[^.]{0,120}?[Cc]hapter\s*(\d+[A-Z]?)"
r"(?:[^.]{0,120}?[Ss]ection\s*([\d.]+))?"
)
#: "MLN907166" / "MLN 907166" / "ICN 909289" / "ICN MLN909188" — an MLN
#: product number (booklet, fact sheet, web-based training). CMS renamed
#: the "ICN" (Internet Content Number) prefix to "MLN" in 2020 without
#: renumbering, so both prefixes name the same product series and the
#: stored locator is always the "MLN <number>" form (#705 item 3).
MLN_RE = re.compile(r"\b(?:ICN\s*)?(?:MLN|ICN)\s*(\d{6,7})\b", re.IGNORECASE)
#: "MLN Matters® Number MM9603" / "MLN Matters article SE1316" / "MLN
#: Matters SE 1316" / "MLN Matters article 11268" (a bare number is a
#: change-request article, prefix "MM"). Product numbers (``MLN_RE``)
#: never follow the "Matters" keyword, so the two patterns are disjoint.
MLN_MATTERS_RE = re.compile(
r"\bMLN\s+Matters\W{0,3}(?:(?:article|number|no\.?|#)\s*)*"
r"(MM|SE)?\s?(\d{4,5})\b",
re.IGNORECASE,
)
def mln_refs(text: str) -> list[str]:
"""Every MLN locator in *text* — ``"MLN 909188"`` for product numbers,
``"MLN Matters MM9603"`` / ``"MLN Matters SE1316"`` for articles —
in order of appearance, first occurrence wins."""
out: list[str] = []
for m in MLN_RE.finditer(text):
loc = f"MLN {m.group(1)}"
if loc not in out:
out.append(loc)
for m in MLN_MATTERS_RE.finditer(text):
prefix = (m.group(1) or "MM").upper()
loc = f"MLN Matters {prefix}{m.group(2)}"
if loc not in out:
out.append(loc)
return out
def resolve_cfr(store: Any, title: str, section: str) -> str:
"""The bib ``regulation`` item covering "*title* CFR *section*", or
``""`` when the library has no covering item (699 eCFR sections on
hand; most of the U.S. Code of Federal Regulations is not)."""
cite = parse_cite(f"{title} CFR {section}")
return item_for(cite, store)
def resolve_iom(store: Any, pub: str, chapter: str) -> str:
"""The bib item tagged ``pub:<pub>`` whose title names ``Chapter
<chapter>:`` (``bib.iom`` stamps chapter titles as "<manual> —
Chapter <N>: <title>"), or ``""`` when unresolved."""
marker = f"Chapter {chapter}:"
for item in store.list_items(tag=f"pub:{pub}"):
if marker in item.title:
return item.key
return ""
def resolve_mln(store: Any, locator: str) -> str:
"""The bib item whose URL or title carries the MLN identifier in
*locator* (``"MLN 909188"`` -> ``MLN909188`` or the pre-2020
``ICN909188`` spelling; ``"MLN Matters MM9603"`` -> ``MM9603``), or
``""`` when unresolved. CMS files its products under the number
(``.../MLNProducts/Downloads/eval-mgmt-serv-guide-ICN006764.pdf``,
``.../MLNMattersArticles/downloads/MM9603.pdf``) and ``bib`` titles
curated booklets with it ("... (MLN909188, June 2025)"), so a
word-bounded match on either column is the whole resolver."""
tail = locator.split()[-1] if locator else ""
if not tail:
return ""
if tail.isdigit():
idents = (f"MLN{tail}", f"ICN{tail}")
else:
idents = (tail,)
con = store._con() # noqa: SLF001 — same pattern as ``harvest``
like = " OR ".join("url LIKE ? OR title LIKE ?" for _ in idents)
params = [p for i in idents for p in (f"%{i}%", f"%{i}%")]
rows = con.execute(
f"SELECT key, title, url FROM items WHERE {like} ORDER BY id", params
).fetchall()
bounded = [
re.compile(rf"(?<![A-Za-z0-9]){i}(?![0-9])", re.IGNORECASE) for i in idents
]
for key, title, url in rows:
for rx in bounded:
if rx.search(url or "") or rx.search(title or ""):
return key
return ""
#: (item_key, section) -> located page; storage path -> normalized page
#: texts. Both live for the process — one ``stack pfs guidance --write``
#: run locates the same chapter's sections for several families.
_PAGE_CACHE: dict[tuple[str, str], int] = {}
_PDF_CACHE: dict[str, list[str]] = {}
def iom_section_page(store: Any, item_key: str, section: str) -> int:
"""1-based page of IOM *section* ("30.6.4") inside the chapter PDF
attached to bib item *item_key* (``bib.iom.download_attachments``
stores one PDF per chapter), or 0 when the item has no PDF on hand,
the section has no body heading, or the PDF toolkit is missing
(#705 item 2). Located at build time on the host so the chat, whose
image has neither the storage tree nor pymupdf, only reads the
stored number."""
if not item_key or not section:
return 0
key = (item_key, section)
if key in _PAGE_CACHE:
return _PAGE_CACHE[key]
try:
from llm.pages import locate_section, pdf_pages
except ImportError: # pragma: no cover — llm package always ships with pfs
return 0
con = store._con() # noqa: SLF001 — same pattern as ``harvest``
rows = con.execute(
"SELECT a.storage_path FROM attachments a JOIN items i ON i.id = a.item_id "
"WHERE i.key = ? ORDER BY a.id",
(item_key,),
).fetchall()
page = 0
for (path,) in rows:
if not path or not path.lower().endswith(".pdf"):
continue
if path not in _PDF_CACHE:
_PDF_CACHE[path] = pdf_pages(Path(path))
page = locate_section(_PDF_CACHE[path], section)
if page:
break
_PAGE_CACHE[key] = page
return page
def _guidance_page(store: Any, kind: str, locator: str, item_key: str) -> int:
"""``iom_section_page`` for a resolved IOM locator that names a
section ("100-04 ch.12 §30.6.4"); 0 for everything else."""
if kind != "iom" or not item_key or "§" not in locator:
return 0
return iom_section_page(store, item_key, locator.rsplit("§", 1)[1])
def _cfr_continuations(text: str, pos: int, pattern: re.Pattern[str]) -> list[str]:
"""Every list-continuation token starting at *pos* (Ruling A10) —
``pattern`` is ``_CFR_CONT_SECTION_RE`` or ``_CFR_CONT_PART_RE``.
Stops at the first position that isn't a separator + token (a
sentence continuing in prose, e.g. "... 410.78 and the physician",
never captures "the")."""
out: list[str] = []
while True:
m = pattern.match(text, pos)
if m is None:
break
out.append(m.group(1))
pos = m.end()
return out
def _cfr_refs(text: str) -> list[tuple[str, str]]:
"""(title, section) pairs — *section* already shaped for
``bib.cfrlink.parse_cite``'s ``"<title> CFR <section>"`` grammar
("410.78(a)(3)" or "Part 425") — for every CFR reference in *text*,
including every section/part named in a list continuation after a
titled head (Ruling A10: "42 CFR 410.20, 410.26, and 410.32" / "...
410.26(a)(3) and 410.26(b) ..." / "42 CFR parts 405, 414, and
426")."""
out: list[tuple[str, str]] = []
for m in CFR_RE.finditer(text):
title = m.group("title")
part, dec, paras = m.group("part"), m.group("dec"), m.group("paras") or ""
if dec:
out.append((title, f"{part}.{dec}{paras}"))
for token in _cfr_continuations(text, m.end(), _CFR_CONT_SECTION_RE):
out.append((title, token))
else:
out.append((title, f"Part {part}"))
if (m.group("partword") or "").lower() == "parts":
for token in _cfr_continuations(text, m.end(), _CFR_CONT_PART_RE):
out.append((title, f"Part {token}"))
consumed: list[tuple[int, int]] = []
for m in _BARE_CFR_PLURAL_RE.finditer(text):
out.append(("42", m.group(1)))
out.append(("42", m.group(2)))
consumed.append(m.span())
for m in BARE_CFR_RE.finditer(text):
if any(start <= m.start() < end for start, end in consumed):
continue # already captured by the §§ plural form
part, dec, paras = m.group(1), m.group(2), m.group(3) or ""
out.append(("42", f"{part}.{dec}{paras}"))
return out
def _iom_refs(text: str) -> list[tuple[str, str, str]]:
"""(pub, chapter, section) for every IOM reference in *text* —
*section* is ``""`` when the sentence names a chapter but no
section. A manual name the pub map doesn't cover is dropped rather
than resolved with a guessed number."""
out: list[tuple[str, str, str]] = []
for m in IOM_RE.finditer(text):
pub_digit, manual_name, chapter, section = m.groups()
pub = (
f"100-0{pub_digit}"
if pub_digit
else _MANUAL_PUB.get((manual_name or "").lower(), "")
)
if not pub:
continue
# A sentence-final "Section 10.2.5.2." swallows the closing
# period into [\d.]+ — trim it; a real section number never
# ends in ".".
out.append((pub, chapter.upper(), (section or "").rstrip(".")))
return out
def _extract(text: str, store: Any) -> list[tuple[str, str, str]]:
"""(kind, locator, item_key) for every CFR/IOM/MLN reference in
*text*, resolved against *store* — ``item_key`` is ``""`` when
unresolved."""
out: list[tuple[str, str, str]] = []
for title, section in _cfr_refs(text):
cite = parse_cite(f"{title} CFR {section}")
out.append(("cfr", canonical(cite), resolve_cfr(store, title, section)))
for pub, chapter, section in _iom_refs(text):
locator = f"{pub} ch.{chapter}" + (f" §{section}" if section else "")
out.append(("iom", locator, resolve_iom(store, pub, chapter)))
for locator in mln_refs(text):
out.append(("mln", locator, resolve_mln(store, locator)))
return out
def dedupe(rows: Sequence[GuidanceRow]) -> list[GuidanceRow]:
"""First-seen-wins de-duplication on ``(family, code, kind,
locator, item_key_src, p_id_src)`` — see the module docstring."""
seen: set[tuple[str, str, str, str, str, int]] = set()
out: list[GuidanceRow] = []
for r in rows:
key = (r.family, r.code, r.kind, r.locator, r.item_key_src, r.p_id_src)
if key in seen:
continue
seen.add(key)
out.append(r)
return out
def harvest(store: Any, codes: Sequence[str], *, family: str) -> list[GuidanceRow]:
"""FR paragraphs (``fr_anchors``) naming any of *codes* -> the
CFR/IOM/MLN references they cite, anchored to that paragraph
(``item_key_src``/``p_id_src``/``page_src``)."""
if not codes:
return []
con = store._con() # noqa: SLF001 — same pattern as pfs.descriptors
pattern = code_pattern(codes)
like_clause = " OR ".join("text LIKE ?" for _ in codes)
params = [f"%{c.upper()}%" for c in codes]
rows = con.execute(
f"SELECT item_key, p_id, page, text FROM fr_anchors WHERE {like_clause}",
params,
).fetchall()
out: list[GuidanceRow] = []
for item_key, p_id, page, text in rows:
present = codes_in(text, pattern)
if not present:
continue
for kind, locator, resolved in _extract(text, store):
located = _guidance_page(store, kind, locator, resolved)
for code in present:
out.append(
GuidanceRow(
family,
code,
kind,
locator,
resolved,
item_key,
p_id,
page,
located,
)
)
return dedupe(out)
def harvest_cpt(
con: Any, store: Any, codes: Sequence[str], *, family: str
) -> list[GuidanceRow]:
"""The newest ingested CPT edition's guideline text and ``see``/
``other`` instructions for *codes* -> the CFR/IOM/MLN references
they cite, anchored to the CPT edition item (``p_id_src=0``,
``page_src=0`` — no FR paragraph). A resolved IOM section carries
the chapter-PDF page it was located on (``page``, #705 item 2). ``[]`` on a replica with no
``pfs.cpt_*`` tables yet (I4: not a bug, just nothing ingested)."""
from pfs.codetables import (
cpt_years,
read_cpt_codes,
read_cpt_instructions,
read_cpt_sections,
)
if not codes:
return []
try:
years = cpt_years(con)
except Exception as exc:
if is_missing_table_error(exc):
return []
raise
if not years:
return []
year = max(years)
codes_u = {c.upper() for c in codes}
sections = {s.sec_id: s for s in read_cpt_sections(con, year)}
codes_by_sec: dict[str, set[str]] = {}
for c in read_cpt_codes(con, year):
if c.code in codes_u:
codes_by_sec.setdefault(c.sec_id, set()).add(c.code)
out: list[GuidanceRow] = []
for sec_id, fam_codes in codes_by_sec.items():
section = sections.get(sec_id)
if section is None or not section.guideline:
continue
for kind, locator, resolved in _extract(section.guideline, store):
located = _guidance_page(store, kind, locator, resolved)
for code in sorted(fam_codes):
out.append(
GuidanceRow(
family,
code,
kind,
locator,
resolved,
section.item_key,
0,
0,
located,
)
)
for instr in read_cpt_instructions(con, year):
if instr.code not in codes_u or instr.kind not in ("see", "other"):
continue
if not any(k in instr.text for k in ("CFR", "Medicare", "Chapter", "chapter")):
continue
for kind, locator, resolved in _extract(instr.text, store):
located = _guidance_page(store, kind, locator, resolved)
out.append(
GuidanceRow(
family,
instr.code,
kind,
locator,
resolved,
instr.item_key,
0,
0,
located,
)
)
return dedupe(out)
def build(
con: Any, store: Any, codes: Sequence[str], *, family: str
) -> list[GuidanceRow]:
"""Every guidance row for one family — FR (``harvest``) + CPT
(``harvest_cpt``), deduped across both sources."""
rows = harvest(store, codes, family=family)
rows.extend(harvest_cpt(con, store, codes, family=family))
return dedupe(rows)