fix(pfs): CFR citation lists — capture every section and part in a run (refs #689)

This commit is contained in:
kert
2026-09-09 23:32:48 -04:00
parent 383df0933d
commit 839c5bfc2c
2 changed files with 216 additions and 12 deletions

View File

@@ -49,13 +49,19 @@ _MANUAL_PUB = {
"financial management": "100-06", "financial management": "100-06",
} }
# "42 CFR 410.78(a)(3)" / "42 C.F.R. § 425.400" / "42 CFR Part 425" — # "42 CFR 410.78(a)(3)" / "42 C.F.R. § 425.400" / "42 CFR Part 425" /
# title-bearing form. The paragraph suffix and part-only form are left # "42 CFR parts 405, 414, and 426" — title-bearing form. The paragraph
# to bib.cfrlink.parse_cite (fed "<title> CFR <section>") to interpret, # suffix and part-only form are left to bib.cfrlink.parse_cite (fed
# so this regex just has to isolate title/part/section/paras. # "<title> CFR <section>") to interpret, so this regex just has to
# isolate title/partword/part/dec/paras. Named groups (Ruling A10): the
# "parts" keyword capture lets ``_cfr_refs`` tell a plural part-list
# head ("parts 405, ...") from a singular one-off part cite ("Part
# 425") — only the plural head triggers the bare-part-number
# continuation grammar below.
CFR_RE = re.compile( CFR_RE = re.compile(
r"\b(\d{1,2})\s*C\.?F\.?R\.?\s*(?:[Pp]art\s*)?(\d{2,4})" r"\b(?P<title>\d{1,2})\s*C\.?F\.?R\.?\s*"
r"(?:\.(\d+[a-z]?)((?:\([^)\s]+\))*))?" r"(?:(?P<partword>[Pp]arts?)\s*)?(?P<part>\d{2,4})"
r"(?:\.(?P<dec>\d+[a-z]?)(?P<paras>(?:\([^)\s]+\))*))?"
) )
#: Bare "§ 410.78(a)(3)" — no title present. Every paragraph this module #: Bare "§ 410.78(a)(3)" — no title present. Every paragraph this module
#: scans is a PFS rule (or the CPT manual's own Medicare-policy prose), #: scans is a PFS rule (or the CPT manual's own Medicare-policy prose),
@@ -63,6 +69,32 @@ CFR_RE = re.compile(
#: the stored ``locator`` is always the canonical "42 CFR ..." form, #: the stored ``locator`` is always the canonical "42 CFR ..." form,
#: never a flag). #: never a flag).
BARE_CFR_RE = re.compile(r"§\s*(\d{2,4})\.(\d+[a-z]?)((?:\([^)\s]+\))*)") BARE_CFR_RE = re.compile(r"§\s*(\d{2,4})\.(\d+[a-z]?)((?:\([^)\s]+\))*)")
#: "§§ 410.26 and 410.32" — the two-section plural form (Ruling A10:
#: the only bare-§ list shape handled; a longer comma-separated bare
#: list is not, since "§" alone doesn't carry the same list-continuation
#: convention FR drafting uses after a titled "CFR" cite).
_BARE_CFR_PLURAL_RE = re.compile(
r"§§\s*(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)\s+and\s+"
r"(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)"
)
#: Separator between list items in a CFR citation run: ", ", ", and ",
#: " and ", " or ", " through ", etc.
_CFR_LIST_SEP = r"(?:\s*,\s*(?:and\s+|or\s+|through\s+)?|\s+(?:and|or|through)\s+)"
#: A continuation token after a titled section-form head ("410.26",
#: "410.26(a)(3)") — no "CFR" keyword, so it must immediately follow a
#: list separator (Ruling A10) or it is prose, not a citation.
# No "^" anchor: Pattern.match(text, pos) already requires the match to
# start exactly at pos — "^" would instead (and wrongly) require pos to
# be the real start of the whole string (or MULTILINE-after-newline),
# so every continuation past the first character would silently fail.
_CFR_CONT_SECTION_RE = re.compile(
_CFR_LIST_SEP + r"(\d{2,4}\.\d+[a-z]?(?:\([^)\s]+\))*)"
)
#: A continuation token after a plural "parts" head ("405, 414, and
#: 426") — bare part numbers, no decimal (the negative lookahead keeps
#: "414.5" from being misread as a bare part).
_CFR_CONT_PART_RE = re.compile(_CFR_LIST_SEP + r"(\d{2,4})(?!\.\d)")
#: "Pub. 100-04, chapter 12" / "Publication 100-04 ... Chapter 12" (the #: "Pub. 100-04, chapter 12" / "Publication 100-04 ... Chapter 12" (the
#: pub-number form) or "Medicare Claims Processing Manual ... Chapter #: pub-number form) or "Medicare Claims Processing Manual ... Chapter
@@ -102,16 +134,51 @@ def resolve_iom(store: Any, pub: str, chapter: str) -> str:
return "" return ""
def _cfr_continuations(text: str, pos: int, pattern: re.Pattern[str]) -> list[str]:
"""Every list-continuation token starting at *pos* (Ruling A10) —
``pattern`` is ``_CFR_CONT_SECTION_RE`` or ``_CFR_CONT_PART_RE``.
Stops at the first position that isn't a separator + token (a
sentence continuing in prose, e.g. "... 410.78 and the physician",
never captures "the")."""
out: list[str] = []
while True:
m = pattern.match(text, pos)
if m is None:
break
out.append(m.group(1))
pos = m.end()
return out
def _cfr_refs(text: str) -> list[tuple[str, str]]: def _cfr_refs(text: str) -> list[tuple[str, str]]:
"""(title, section) pairs — *section* already shaped for """(title, section) pairs — *section* already shaped for
``bib.cfrlink.parse_cite``'s ``"<title> CFR <section>"`` grammar ``bib.cfrlink.parse_cite``'s ``"<title> CFR <section>"`` grammar
("410.78(a)(3)" or "Part 425") — for every CFR reference in *text*.""" ("410.78(a)(3)" or "Part 425") — for every CFR reference in *text*,
including every section/part named in a list continuation after a
titled head (Ruling A10: "42 CFR 410.20, 410.26, and 410.32" / "...
410.26(a)(3) and 410.26(b) ..." / "42 CFR parts 405, 414, and
426")."""
out: list[tuple[str, str]] = [] out: list[tuple[str, str]] = []
for m in CFR_RE.finditer(text): for m in CFR_RE.finditer(text):
title, part, dec, paras = m.group(1), m.group(2), m.group(3), m.group(4) or "" title = m.group("title")
section = f"{part}.{dec}{paras}" if dec else f"Part {part}" part, dec, paras = m.group("part"), m.group("dec"), m.group("paras") or ""
out.append((title, section)) if dec:
out.append((title, f"{part}.{dec}{paras}"))
for token in _cfr_continuations(text, m.end(), _CFR_CONT_SECTION_RE):
out.append((title, token))
else:
out.append((title, f"Part {part}"))
if (m.group("partword") or "").lower() == "parts":
for token in _cfr_continuations(text, m.end(), _CFR_CONT_PART_RE):
out.append((title, f"Part {token}"))
consumed: list[tuple[int, int]] = []
for m in _BARE_CFR_PLURAL_RE.finditer(text):
out.append(("42", m.group(1)))
out.append(("42", m.group(2)))
consumed.append(m.span())
for m in BARE_CFR_RE.finditer(text): for m in BARE_CFR_RE.finditer(text):
if any(start <= m.start() < end for start, end in consumed):
continue # already captured by the §§ plural form
part, dec, paras = m.group(1), m.group(2), m.group(3) or "" part, dec, paras = m.group(1), m.group(2), m.group(3) or ""
out.append(("42", f"{part}.{dec}{paras}")) out.append(("42", f"{part}.{dec}{paras}"))
return out return out

View File

@@ -13,8 +13,10 @@ from pfs.codetables import (
GuidanceRow, GuidanceRow,
ensure_tables, ensure_tables,
read_guidance, read_guidance,
write_cpt_edition,
write_guidance, write_guidance,
) )
from pfs.cpt_model import CptCode, CptEdition, CptSection
from pfs.guidance import ( from pfs.guidance import (
BARE_CFR_RE, BARE_CFR_RE,
CFR_RE, CFR_RE,
@@ -71,12 +73,37 @@ class TestCfrRegex:
def test_titled_cfr_captures_title_part_section_paras(self): def test_titled_cfr_captures_title_part_section_paras(self):
m = CFR_RE.search("under 42 CFR 410.78(a)(3) and elsewhere") m = CFR_RE.search("under 42 CFR 410.78(a)(3) and elsewhere")
assert m is not None assert m is not None
assert m.groups() == ("42", "410", "78", "(a)(3)") assert (
m.group("title"),
m.group("part"),
m.group("dec"),
m.group("paras"),
) == (
"42",
"410",
"78",
"(a)(3)",
)
assert m.group("partword") is None
def test_titled_cfr_part_only(self): def test_titled_cfr_part_only(self):
m = CFR_RE.search("see 42 CFR Part 425 for details") m = CFR_RE.search("see 42 CFR Part 425 for details")
assert m is not None assert m is not None
assert m.groups() == ("42", "425", None, None) assert (m.group("title"), m.group("partword"), m.group("part")) == (
"42",
"Part",
"425",
)
assert m.group("dec") is None and m.group("paras") is None
def test_titled_cfr_plural_parts_only(self):
m = CFR_RE.search("see 42 CFR parts 405 for details")
assert m is not None
assert (m.group("title"), m.group("partword"), m.group("part")) == (
"42",
"parts",
"405",
)
def test_dotted_cfr_with_section_sign_resolves_via_bare_fallback(self): def test_dotted_cfr_with_section_sign_resolves_via_bare_fallback(self):
# CFR_RE (titled) has no "§" between "C.F.R." and the number — # CFR_RE (titled) has no "§" between "C.F.R." and the number —
@@ -86,6 +113,47 @@ class TestCfrRegex:
assert _cfr_refs("42 C.F.R. § 414.1425 governs this") == [("42", "414.1425")] assert _cfr_refs("42 C.F.R. § 414.1425 governs this") == [("42", "414.1425")]
class TestCfrListContinuation:
"""Ruling A10: a titled CFR head followed by a list continuation
(comma/and/or/through-separated bare tokens) captures every
section/part in the run, not just the head."""
def test_comma_list_of_sections(self):
assert _cfr_refs(
"the current regulations at 42 CFR 410.20, 410.26, and 410.32 for CCM"
) == [("42", "410.20"), ("42", "410.26"), ("42", "410.32")]
def test_and_joined_paragraph_cites(self):
assert _cfr_refs("… 42 CFR 410.26(a)(3) and 410.26(b) to better define …") == [
("42", "410.26(a)(3)"),
("42", "410.26(b)"),
]
def test_plural_parts_list(self):
assert _cfr_refs("… 42 CFR parts 405, 414, and 426 …") == [
("42", "Part 405"),
("42", "Part 414"),
("42", "Part 426"),
]
def test_bare_double_section_plural(self):
assert _cfr_refs("§§ 410.26 and 410.32 govern this") == [
("42", "410.26"),
("42", "410.32"),
]
def test_no_continuation_across_prose(self):
# "the" is not a citation token — the continuation grammar must
# not fire on ordinary prose following "and".
assert _cfr_refs("42 CFR 410.78 and the physician") == [("42", "410.78")]
def test_singular_part_does_not_trigger_continuation(self):
# Only a plural "parts" head triggers the bare-part-number
# continuation grammar (Ruling A10) — a singular "Part 425"
# followed by an unrelated number must not absorb it.
assert _cfr_refs("42 CFR Part 425 and 42 other things") == [("42", "Part 425")]
class TestIomRegex: class TestIomRegex:
def test_manual_name_form(self): def test_manual_name_form(self):
m = IOM_RE.search(IOM_MANUAL_TEXT) m = IOM_RE.search(IOM_MANUAL_TEXT)
@@ -302,6 +370,75 @@ class TestGuidanceTable:
assert len(read_guidance(con, "APCM")) == 1 assert len(read_guidance(con, "APCM")) == 1
# ── harvest_cpt (non-empty: a real CFR cite in the guideline text) ───
def _cpt_code(code: str, sec_id: str) -> CptCode:
return CptCode(
code=code,
sec_id=sec_id,
descriptor="d",
stem="s",
elements=(),
tail="",
addon=False,
resequenced=False,
new=False,
revised=False,
telemedicine=False,
parent="",
mod51_exempt=False,
audio_only=False,
fda_pending=False,
pla=False,
category="I",
)
class TestHarvestCpt:
def test_guideline_cfr_cite_produces_a_row(self, store, con):
s, sec_key, _manual_key = store
edition = CptEdition(
year=2024,
sections=(
CptSection(
sec_id="sec_1",
level=2,
title="Chronic Care Management Services",
path=(
"Evaluation and Management",
"Chronic Care Management Services",
),
code_lo="99490",
code_hi="99490",
guideline=(
"See 42 CFR 410.78(a)(3) for telehealth conditions of payment."
),
),
),
codes=(_cpt_code("99490", "sec_1"),),
instructions=(),
references=(),
crosswalks=(),
lists=(),
alternates=(),
)
write_cpt_edition(con, edition, "CPTED2024")
rows = harvest_cpt(con, s, ("99490",), family="CCM")
assert len(rows) == 1
r = rows[0]
assert (r.family, r.code, r.kind, r.locator) == (
"CCM",
"99490",
"cfr",
"42 CFR 410.78(a)(3)",
)
assert r.item_key == sec_key
assert r.item_key_src == "CPTED2024"
assert r.p_id_src == 0
assert r.page_src == 0
# ── build (FR + CPT, no CPT tables ingested) ───────────────────────── # ── build (FR + CPT, no CPT tables ingested) ─────────────────────────