feat(llm): rule discussions ↔ manual sections — cited IOM sections join the chat as companion sources; search links them; rules theme-stamped
Some checks failed
CI / lint (push) Successful in 39s
CI / notebooks-smoke (push) Successful in 1m42s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
CI / test (push) Failing after 2m30s
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 15s
Infra CI / notebooks (push) Successful in 52s
Deploy / report (push) Has been cancelled
Infra CI / api (push) Has been cancelled
Infra CI / llm (push) Has been cancelled
Infra CI / mc (push) Has been cancelled
Infra CI / docs (push) Has started running

A retrieved FR paragraph that cites a CMS manual ('Pub. 100-04,
Chapter 12, Section 30.6.4') now carries those citations on its source
(as_source iom_refs, parsed by pfs.guidance.iom_refs over the full
chunk text). llm.iomlink resolves each cited section to its indexed
manual chunk (exact section, else the first finer section under it,
else the chapter's first section) and the chat merges up to
iom_companion_max (4) of them in after the CPT manual block, labelled
'IOM 100-04 ch.12 §30.6.4', linked to the chapter PDF page, marked
'cited by [91 FR … ¶n]' in the prompt and the sources drawer; most-cited
sections first, nothing already among the sources repeated, no query
when no rule cites a manual. The search page shows 'cites IOM …' links
on rule hits that open the cited section under the manual filters, and
gains manual / chapter / section / theme fields. Rule paragraphs are
theme-stamped as well (stack llm stamp-themes --collection rules
--doctype rule: 67,340 chunks).
This commit is contained in:
kert
2026-09-24 20:25:33 -04:00
parent 26f01bd0ba
commit 531fca9655
10 changed files with 417 additions and 2 deletions

View File

@@ -57,6 +57,9 @@ class LlmConfig:
lineage_on_demand_max: int = 3
lineage_sources_max: int = 8
chat_codes_max: int = 24
#: IOM sections cited by retrieved rule paragraphs pulled in as
#: companion sources per turn (llm.iomlink)
iom_companion_max: int = 4
valuation_rows_max: int = 24
@@ -133,6 +136,7 @@ def load() -> LlmConfig:
_opt(section, "code_cited_collections", ["rules", "comments", "corpus"])
),
code_cited_max=int(_opt(section, "code_cited_max", 12)),
iom_companion_max=int(_opt(section, "iom_companion_max", 4)),
lineage_max_rows=int(_opt(section, "lineage_max_rows", 25)),
lineage_on_demand_max=int(_opt(section, "lineage_on_demand_max", 3)),
lineage_sources_max=int(_opt(section, "lineage_sources_max", 8)),

121
src/llm/iomlink.py Normal file
View File

@@ -0,0 +1,121 @@
"""Rule discussions ↔ manual sections: the IOM sections a retrieved rule
paragraph cites, pulled in as companion sources.
A Federal Register paragraph that says "see Pub. 100-04, Chapter 12,
Section 30.6.4" is pointing the reader at the operational rule; the
manual chapters are indexed by section (``llm.manuals``), so the cited
section's own text can sit beside the paragraph in the chat's sources —
labelled ``IOM 100-04 ch.12 §30.6.4``, linked to the chapter PDF page,
and marked with the FR paragraph(s) that cite it. Chapter-only citations
("Chapter 12") resolve to the chapter's first section; a section the
manual splits finer than the citation ("30.6" → 30.6.1) resolves to the
first section under it.
companions(engine, sources, max_total=…) → companion source dicts
"""
from __future__ import annotations
import logging
from typing import Any, Sequence
from sqlalchemy import text
from llm.links import as_source
log = logging.getLogger(__name__)
_EXACT = """
SELECT e.document, e.cmetadata FROM langchain_pg_embedding e
WHERE e.cmetadata->>'doctype' = 'manual' AND e.cmetadata->>'manual' = :pub
AND e.cmetadata->>'chapter' = :chapter AND e.cmetadata->>'iom_section' = :section
ORDER BY (e.cmetadata->>'seq')::int LIMIT 1
"""
_PREFIX = """
SELECT e.document, e.cmetadata FROM langchain_pg_embedding e
WHERE e.cmetadata->>'doctype' = 'manual' AND e.cmetadata->>'manual' = :pub
AND e.cmetadata->>'chapter' = :chapter AND e.cmetadata->>'iom_section' LIKE :prefix
ORDER BY (e.cmetadata->>'seq')::int LIMIT 1
"""
_CHAPTER = """
SELECT e.document, e.cmetadata FROM langchain_pg_embedding e
WHERE e.cmetadata->>'doctype' = 'manual' AND e.cmetadata->>'manual' = :pub
AND e.cmetadata->>'chapter' = :chapter AND coalesce(e.cmetadata->>'iom_section', '') <> ''
ORDER BY (e.cmetadata->>'seq')::int LIMIT 1
"""
def lookup(conn: Any, pub: str, chapter: str, section: str) -> tuple[str, dict] | None:
"""The first chunk of the cited section (exact, then finer sections
under it, then the chapter's first section for a chapter-only cite)."""
if section:
row = conn.execute(
text(_EXACT), {"pub": pub, "chapter": chapter, "section": section}
).fetchone()
if row is None:
row = conn.execute(
text(_PREFIX),
{"pub": pub, "chapter": chapter, "prefix": section + ".%"},
).fetchone()
else:
row = conn.execute(text(_CHAPTER), {"pub": pub, "chapter": chapter}).fetchone()
if row is None:
return None
return row[0] or "", dict(row[1] or {})
def companions(
engine: Any, sources: Sequence[dict], *, max_total: int = 4
) -> list[dict]:
"""Manual-section sources for the IOM citations carried by the rule
sources in *sources* (``as_source``'s ``iom_refs``), most-cited
first, each with ``cited_by`` (the citing FR labels) and
``companion = "iom"``; a section already present in *sources* is not
repeated. Never raises — a lookup failure logs and yields nothing."""
order: list[tuple[str, str, str]] = []
cited_by: dict[tuple[str, str, str], list[str]] = {}
for s in sources:
if s.get("kind") != "rule":
continue
for ref in s.get("iom_refs") or []:
key = (ref.get("pub", ""), ref.get("chapter", ""), ref.get("section", ""))
if not key[0] or not key[1]:
continue
if key not in cited_by:
cited_by[key] = []
order.append(key)
if s.get("label") and s["label"] not in cited_by[key]:
cited_by[key].append(s["label"])
if not order:
return []
# most-cited first, ties in order of appearance; sectioned cites
# before chapter-only ones
ranked = sorted(
order, key=lambda k: (-len(cited_by[k]), 0 if k[2] else 1, order.index(k))
)
present = {
(s.get("manual"), s.get("chapter"), s.get("iom_section")) for s in sources
}
out: list[dict] = []
seen: set[tuple[str, str, str]] = set()
try:
with engine.begin() as conn:
for pub, chapter, section in ranked:
if len(out) >= max_total:
break
hit = lookup(conn, pub, chapter, section)
if hit is None:
continue
doc, md = hit
found = (md.get("manual"), md.get("chapter"), md.get("iom_section"))
if found in seen or found in present:
continue
seen.add(found)
src = as_source({k: str(v) for k, v in md.items()}, doc, 0.0)
src["cited_by"] = list(cited_by[(pub, chapter, section)])
src["companion"] = "iom"
out.append(src)
except Exception as exc: # noqa: BLE001 — a companion lookup must never break the chat
log.warning("iom companions skipped: %s", exc)
return out
return out

View File

@@ -95,6 +95,33 @@ def for_source(md: dict[str, str], snippet: str) -> tuple[str, str]:
return _corpus(md)
def iom_refs_of(text: str) -> list[dict[str, str]]:
"""``[{"pub", "chapter", "section", "label"}]`` for every CMS manual
citation in *text* ("Pub. 100-04, Chapter 12, Section 30.6.4"),
de-duplicated in order of appearance."""
from llm.manuals import label as manual_label
from pfs.guidance import iom_refs
out: list[dict[str, str]] = []
seen: set[tuple[str, str, str]] = set()
for pub, chapter, section in iom_refs(text or ""):
key = (pub, chapter, section)
if key in seen:
continue
seen.add(key)
out.append(
{
"pub": pub,
"chapter": chapter,
"section": section,
"label": manual_label(
{"manual": pub, "chapter": chapter, "iom_section": section}
),
}
)
return out
def as_source(md: dict[str, str], text: str, score: float) -> dict:
"""The source dict the chat sends to the prompt and the UI.
@@ -121,6 +148,9 @@ def as_source(md: dict[str, str], text: str, score: float) -> dict:
"iom_section": md.get("iom_section", ""),
# theme slugs stamped by llm.themes (comma-joined), for filters and badges
"themes": md.get("themes", ""),
# IOM sections a rule paragraph cites (llm.iomlink pulls them in as
# companion sources; the search page links them)
"iom_refs": iom_refs_of(text) if kind == "rule" else [],
"url": url,
"title": md.get("title", ""),
"date": md.get("date", ""),

View File

@@ -417,8 +417,13 @@ def _user_message(
sources = retrieved + cited + lineage_src + manual + rest
if sources:
context = "\n\n".join(
f"[{s['label']}] ({source_kind(s)}, {s.get('date', '') or 'undated'}) "
f"{s['snippet']}"
f"[{s['label']}] ({source_kind(s)}, {s.get('date', '') or 'undated'}"
+ (
f"; cited by {', '.join('[' + c + ']' for c in s['cited_by'])}"
if s.get("cited_by")
else ""
)
+ f") {s['snippet']}"
for s in sources
)
else:
@@ -591,6 +596,21 @@ def build_messages(
]
def _iom_companions(
sources: list[dict], cfg: LlmConfig, *, max_total: int
) -> list[dict]:
"""``llm.iomlink.companions`` over the current sources; never raises."""
if not any(s.get("kind") == "rule" and s.get("iom_refs") for s in sources):
return [] # nothing cites a manual: no engine, no query
try:
from llm.iomlink import companions
return companions(_engine(cfg), sources, max_total=max_total)
except Exception as e: # noqa: BLE001 — a companion bug must not break the chat
log.warning("iom companions skipped: %s", e)
return []
def _manual_sources(families: Sequence[str], cfg: LlmConfig) -> list[dict]:
"""Cursor-per-call wrapper around ``evidence.manual_sources`` — the
same cached-replica-cursor pattern ``valuation_evidence`` uses
@@ -747,6 +767,15 @@ def _stream_answer(
"dropped %d excerpt(s) using a family acronym for something else",
n_conflict,
)
# Rule discussions ↔ manual sections: the IOM sections the retrieved
# and cited FR paragraphs point at, as companion sources right after
# the CPT manual block (they share its budget slot, llm.iomlink).
if cfg.iom_companion_max > 0:
before = len(sources)
sources = merge_sources(
sources, _iom_companions(sources, cfg, max_total=cfg.iom_companion_max)
)
n_manual += len(sources) - before
if lineage is not None:
yield lineage.payload()
if evidence is not None:

View File

@@ -205,6 +205,7 @@
background: color-mix(in srgb, var(--primary) 12%, transparent); margin-right: 6px;
}
.src .date { color: var(--muted-fg); margin-left: 6px; }
.src .citedby { font-family: var(--font-mono); font-size: 11px; color: var(--accent); margin-left: 4px; }
.src a.viewer {
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; margin-left: 6px;
@@ -660,6 +661,13 @@
if (src.title && src.kind !== 'comment') {
el.appendChild(document.createTextNode(' — ' + src.title));
}
// an IOM section pulled in because a retrieved rule paragraph cites it
if (src.cited_by && src.cited_by.length) {
const c = document.createElement('span');
c.className = 'citedby';
c.textContent = ' cited by ' + src.cited_by.map(l => '[' + l + ']').join(', ');
el.appendChild(c);
}
// A comment or article attachment on file opens in the PDF viewer
// on the page the passage was located on (rules link to the FR).
if (src.attachment && src.item_key && src.kind !== 'rule') {

View File

@@ -56,6 +56,8 @@
}
form#search input#q { flex: 1 1 320px; font-size: 16px; }
form#search input#docket, form#search input#year { width: 150px; }
form#search input#manual { width: 90px; } form#search input#chapter { width: 56px; }
form#search input#section { width: 90px; } form#search input#theme { width: 150px; }
form#search label {
display: flex; align-items: center; gap: 6px; font-size: 13px;
color: var(--muted-fg); white-space: nowrap;
@@ -95,6 +97,9 @@
text-decoration: underline dotted;
}
.hit .date, .hit .meta { color: var(--muted-fg); font-size: 12px; font-family: var(--font-mono); }
.hit a.iom {
font-family: var(--font-mono); font-size: 11px; color: var(--accent); text-decoration: underline dotted;
}
.hit a.viewer {
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; white-space: nowrap;
@@ -144,6 +149,10 @@
</label>
<label>docket <input type="text" id="docket" placeholder="CMS-2023-0121"></label>
<label>year <input type="text" id="year" placeholder="2024"></label>
<label>manual <input type="text" id="manual" placeholder="100-04" title="IOM publication number"></label>
<label>ch. <input type="text" id="chapter" placeholder="12" title="IOM chapter"></label>
<label>§ <input type="text" id="section" placeholder="30.6.4" title="IOM section"></label>
<label>theme <input type="text" id="theme" placeholder="telehealth" title="theme slug (stack llm vocab)"></label>
<!-- An "llm:" tag filter belongs here once the tagging chain (#575)
lands and indexed chunks carry llm: tags; there is nothing to
filter on yet, so no control is drawn. -->
@@ -190,6 +199,11 @@
if (val('kind')) p.set('kind', val('kind'));
if (val('docket')) p.set('docket', val('docket'));
if (val('year')) p.set('year', val('year'));
if (val('manual')) p.set('manual', val('manual'));
if (val('chapter')) p.set('chapter', val('chapter'));
if (val('section')) p.set('section', val('section'));
if (val('manual') || val('chapter') || val('section')) p.set('doctype', 'manual');
if (val('theme')) p.set('theme', val('theme'));
if (offset) p.set('offset', String(offset));
p.set('limit', String(LIMIT));
return p;
@@ -202,6 +216,10 @@
document.getElementById('kind').value = p.get('kind') || '';
document.getElementById('docket').value = p.get('docket') || '';
document.getElementById('year').value = p.get('year') || '';
document.getElementById('manual').value = p.get('manual') || '';
document.getElementById('chapter').value = p.get('chapter') || '';
document.getElementById('section').value = p.get('section') || '';
document.getElementById('theme').value = p.get('theme') || '';
offset = Math.max(0, parseInt(p.get('offset') || '0', 10) || 0);
return p.get('q') || '';
}
@@ -249,6 +267,18 @@
line.appendChild(view);
}
// a rule paragraph's CMS manual citations → the cited section, filtered
(r.iom_refs || []).forEach(function (ref) {
if (!ref.label) return;
var a = document.createElement('a');
a.className = 'iom';
var p = new URLSearchParams({ q: (document.getElementById('q') || {}).value || ref.label, doctype: 'manual', manual: ref.pub, chapter: ref.chapter });
if (ref.section) p.set('section', ref.section);
a.href = '/ui/search?' + p.toString();
a.textContent = 'cites ' + ref.label;
line.appendChild(a);
});
if (r.date) {
var date = document.createElement('span');
date.className = 'date';

View File

@@ -296,6 +296,12 @@ def _cfr_refs(text: str) -> list[tuple[str, str]]:
return out
def iom_refs(text: str) -> list[tuple[str, str, str]]:
"""Public name for :func:`_iom_refs` — ``llm.links`` uses it to carry a
rule paragraph's IOM citations on its source dict."""
return _iom_refs(text)
def _iom_refs(text: str) -> list[tuple[str, str, str]]:
"""(pub, chapter, section) for every IOM reference in *text* —
*section* is ``""`` when the sentence names a chapter but no

View File

@@ -125,3 +125,20 @@ class TestStrTuple:
raw["llm"]["code_cited_collections"] = "rules"
monkeypatch.setattr(conf, "cfg", _Cfg(raw))
assert llm_config.load().code_cited_collections == ("rules",)
def test_iom_companion_max_defaults_to_four():
from llm.config import LlmConfig
cfg = LlmConfig(
ollama_hosts=("http://h",),
embed_model="e",
instruct_model="i",
embed_dim=1,
build_ann_index=False,
pg_host="h",
pg_port=1,
pg_db="d",
pg_user="u",
)
assert cfg.iom_companion_max == 4

169
tests/llm/test_iomlink.py Normal file
View File

@@ -0,0 +1,169 @@
"""llm.iomlink — cited IOM sections as companion sources."""
from __future__ import annotations
from unittest.mock import MagicMock
from llm.iomlink import companions, lookup
from llm.links import as_source, iom_refs_of
RULE_TEXT = (
"We refer readers to Pub. 100-04, Chapter 12, Section 30.6.4 for incident-to "
"billing, and to the Medicare Benefit Policy Manual, Chapter 15, for coverage. "
"See also Pub. 100-04, Chapter 12, Section 30.6.4 again."
)
class TestIomRefsOnSources:
def test_rule_source_carries_deduplicated_refs(self):
refs = iom_refs_of(RULE_TEXT)
assert refs == [
{
"pub": "100-04",
"chapter": "12",
"section": "30.6.4",
"label": "IOM 100-04 ch.12 §30.6.4",
},
{
"pub": "100-02",
"chapter": "15",
"section": "",
"label": "IOM 100-02 ch.15",
},
]
s = as_source({"kind": "rule", "item_key": "R", "title": "t"}, RULE_TEXT, 0.4)
assert [r["label"] for r in s["iom_refs"]] == [
"IOM 100-04 ch.12 §30.6.4",
"IOM 100-02 ch.15",
]
assert (
as_source({"kind": "comment", "comment_id": "C-1"}, RULE_TEXT, 0.4)[
"iom_refs"
]
== []
)
def _engine(rows_by_sql):
"""rows_by_sql: substring of SQL → row (document, cmetadata) or None."""
engine = MagicMock()
conn = engine.begin.return_value.__enter__.return_value
calls = []
def execute(sql, params=None):
calls.append((str(sql), params))
r = MagicMock()
for needle, row in rows_by_sql.items():
if needle in str(sql):
r.fetchone.return_value = row
return r
r.fetchone.return_value = None
return r
conn.execute.side_effect = execute
return engine, conn, calls
MD = {
"doctype": "manual",
"kind": "corpus",
"manual": "100-04",
"chapter": "12",
"iom_section": "30.6.4",
"iom_title": "E/M Services Furnished Incident to",
"page": "39",
"item_key": "SPJB4FX2",
"seq": "73",
"url": "https://www.cms.gov/x/clm104c12.pdf",
"title": "Medicare Claims Processing Manual — Chapter 12",
}
class TestCompanions:
def _rule(self, label, text=RULE_TEXT):
return as_source(
{
"kind": "rule",
"item_key": "R",
"p_id": "5",
"page": "43949",
"ordinal": "4",
"fr_volume": "91",
"title": "t",
"html_url": "https://fr/x",
},
text,
0.5,
) | {"label": label}
def test_cited_section_becomes_a_companion_with_cited_by(self):
engine, conn, calls = _engine(
{"iom_section' = :section": ("## 30.6.4 - E/M ... incident to", MD)}
)
out = companions(
engine,
[self._rule("91 FR 43949 ¶4"), self._rule("91 FR 43950 ¶2")],
max_total=4,
)
assert len(out) == 1
c = out[0]
assert c["label"] == "IOM 100-04 ch.12 §30.6.4" and c["companion"] == "iom"
assert c["cited_by"] == ["91 FR 43949 ¶4", "91 FR 43950 ¶2"]
assert (
c["url"] == "https://www.cms.gov/x/clm104c12.pdf#page=39"
and c["doctype"] == "manual"
)
assert c["snippet"].startswith("## 30.6.4")
def test_prefix_and_chapter_fallbacks(self):
finer = dict(MD, iom_section="30.6.4.1")
engine, conn, calls = _engine({"LIKE :prefix": ("finer text", finer)})
assert lookup(conn, "100-04", "12", "30.6.4") == ("finer text", finer)
assert [p["prefix"] for _, p in calls if p and "prefix" in p] == ["30.6.4.%"]
first = dict(MD, iom_section="10")
engine, conn, calls = _engine({"<> ''": ("chapter first", first)})
assert lookup(conn, "100-02", "15", "") == ("chapter first", first)
engine, conn, calls = _engine({})
assert lookup(conn, "100-04", "99", "1") is None
def test_ranking_dedupe_cap_and_present_sections(self):
by = {
"iom_section' = :section": ("sec text", MD),
"<> ''": (
"chapter text",
dict(MD, chapter="15", manual="100-02", iom_section="10"),
),
}
engine, conn, calls = _engine(by)
rules = [self._rule("A"), self._rule("B")]
out = companions(engine, rules, max_total=1)
assert [c["label"] for c in out] == [
"IOM 100-04 ch.12 §30.6.4"
] # cited twice, sectioned → first; cap 1
# already among the sources → not repeated
present = as_source({k: str(v) for k, v in MD.items()}, "sec text", 0.9)
assert companions(engine, rules + [present], max_total=4) == [
c
for c in companions(engine, rules + [present], max_total=4)
if c["label"] != "IOM 100-04 ch.12 §30.6.4"
]
def test_no_rule_sources_no_queries_and_failures_swallowed(self):
engine, conn, calls = _engine({})
assert (
companions(
engine,
[
as_source(
{"kind": "comment", "comment_id": "C"},
"Pub. 100-04, Chapter 12",
0.1,
)
],
)
== []
)
assert calls == []
engine = MagicMock()
engine.begin.side_effect = RuntimeError("db down")
assert companions(engine, [self._rule("A")]) == []

View File

@@ -156,6 +156,7 @@ class TestAsSource:
"chapter",
"iom_section",
"themes",
"iom_refs",
"id",
"label",
"kind",