Files
stack/tests/llm/test_links.py
kert 6eb4f2cb8b
Some checks failed
CI / lint (push) Successful in 40s
CI / notebooks-smoke (push) Successful in 1m52s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
CI / test (push) Failing after 3m8s
Infra CI / zotero (push) Successful in 31s
Infra CI / notebooks (push) Successful in 1m4s
Infra CI / api (push) Successful in 1m29s
Infra CI / docs (push) Successful in 1m55s
Infra CI / mc (push) Failing after 58s
Infra CI / llm (push) Successful in 1m20s
Deploy / report (push) Successful in 22s
fix(llm): rules never double as corpus; citations say proposed/final/correction; IOM chapters indexed by section
Every Federal Register rule item was indexed twice: as anchored FR
paragraphs in the rules collection (67k chunks, federalregister.gov
links) and again from its PDF attachment into corpus (161k chunks,
kind 'corpus', item-URL links). The PDF copies outnumbered the anchored
ones 2:1 in retrieval, so proposed rules and corrections came back as
'corpus' citations. iter_corpus_refs now skips rule-typed and
fr_anchors-bearing items; stack llm prune-rules deletes the corpus
copies (and their index_state rows) and stamps rule_kind
(proposed/final/correction, bib.frlink.rule_kind) on the rules chunks;
new rule docs carry rule_kind from the start. as_source exposes it, the
prompt's source line reads '(proposed rule, 2026-07-16)', and the chat
and search badges show it instead of 'rule'.

CMS Internet-Only Manual chapters: llm.manuals.sectionize turns each
body section heading (the ones followed by their '(Rev.' line; the
table of contents stays plain) into a markdown heading, so chunks split
at section boundaries and carry section / iom_section / iom_title plus
the chapter's publication number and chapter; enrich_pdf_pages lets
chunks under a sub-heading inherit the enclosing file so they keep
their page. Manual citations read 'IOM 100-04 ch.18 §10.1.2' and link
to the chapter PDF page; /search takes doctype, manual, chapter and
section filters.
2026-09-24 19:38:17 -04:00

260 lines
8.1 KiB
Python

"""llm.links — kind-specific evidence deep links (pure)."""
from llm.links import for_source
class TestComment:
def test_plain_comment_links_to_comment_page(self):
url, label = for_source(
{"kind": "comment", "comment_id": "CMS-2026-2377-3438"}, "snippet"
)
assert url == "https://www.regulations.gov/comment/CMS-2026-2377-3438"
assert label == "CMS-2026-2377-3438"
def test_pdf_attachment_chunk_links_to_page(self):
url, label = for_source(
{
"kind": "comment",
"comment_id": "CMS-2026-2377-3438",
"attachment": "attachment_2.pdf",
"page": "4",
},
"s",
)
assert url == (
"https://downloads.regulations.gov/CMS-2026-2377-3438/attachment_2.pdf#page=4"
)
assert label == "CMS-2026-2377-3438 p.4"
def test_non_pdf_attachment_no_page(self):
url, label = for_source(
{"kind": "comment", "comment_id": "C-1", "attachment": "attachment_1.docx"},
"s",
)
assert url == "https://downloads.regulations.gov/C-1/attachment_1.docx"
assert label == "C-1"
def test_missing_comment_id_falls_back_to_item_key(self):
url, label = for_source({"kind": "comment", "item_key": "ABCD1234"}, "s")
assert label == "ABCD1234"
assert url == ""
class TestRule:
MD = {
"kind": "rule",
"html_url": "https://www.federalregister.gov/documents/2026/07/16/2026-14327/x",
"p_id": "938",
"page": "43949",
"ordinal": "4",
"fr_volume": "91",
}
def test_paragraph_link_with_highlight(self):
url, label = for_source(
self.MD,
"In the FY 2027 Hospice proposed rule, CMS solicited comment. More.",
)
assert url == (
"https://www.federalregister.gov/documents/2026/07/16/2026-14327/x#p-938"
":~:text=In%20the%20FY%202027%20Hospice%20proposed%20rule%2C%20CMS%20solicited%20comment."
)
assert label == "91 FR 43949 ¶4"
def test_no_p_id_degrades_to_page_link(self):
md = {**self.MD, "p_id": "", "ordinal": ""}
url, label = for_source(md, "s")
assert url.endswith("/x#page-43949")
assert label == "91 FR 43949"
def test_no_page_degrades_to_document(self):
md = {
"kind": "rule",
"html_url": "https://fr.test/doc",
"title": "CY2027 PFS NPRM",
}
url, label = for_source(md, "s")
assert url == "https://fr.test/doc"
assert label == "CY2027 PFS NPRM"
class TestCorpus:
def test_item_url_and_short_title_with_year(self):
url, label = for_source(
{
"kind": "corpus",
"url": "https://pubmed.ncbi.nlm.nih.gov/19922199/",
"title": "A consensus on palliative care quality metrics for hospital programs",
"year": "2009",
},
"s",
)
assert url == "https://pubmed.ncbi.nlm.nih.gov/19922199/"
assert (
label
== "A consensus on palliative care quality metrics for hospital… (2009)"
)
def test_pdf_url_gets_page(self):
url, _ = for_source(
{
"kind": "corpus",
"url": "https://x.test/report.pdf",
"title": "R",
"page": "7",
},
"s",
)
assert url == "https://x.test/report.pdf#page=7"
def test_non_pdf_url_ignores_page(self):
url, _ = for_source(
{
"kind": "corpus",
"url": "https://x.test/report",
"title": "R",
"page": "7",
},
"s",
)
assert url == "https://x.test/report"
def test_untitled_falls_back_to_item_key(self):
_, label = for_source({"kind": "corpus", "item_key": "K1", "url": ""}, "s")
assert label == "K1"
def test_unknown_kind_treated_as_corpus():
url, label = for_source({"url": "https://u.test", "title": "T"}, "s")
assert (url, label) == ("https://u.test", "T")
class TestAsSource:
def test_shape_matches_rag_source(self):
from llm.links import as_source
s = as_source(
{
"kind": "comment",
"comment_id": "CMS-2026-2377-1",
"date": "2026-08-19",
"title": "t",
"docket": "CMS-2026-2377",
"item_key": "ABCD1234",
"seq": "3",
"section": "Some Heading",
},
"body " * 200,
0.123456,
)
assert set(s) == {
"attachment",
"page",
"rule_kind",
"doctype",
"manual",
"chapter",
"iom_section",
"id",
"label",
"kind",
"url",
"title",
"date",
"docket",
"comment_id",
"snippet",
"score",
"item_key",
"p_id",
"seq",
"section",
}
assert s["label"] == "CMS-2026-2377-1" and s["kind"] == "comment"
assert len(s["snippet"]) <= 500 and s["score"] == 0.1235
# non-rule kind: seq/section carried through, p_id blanked
assert s["item_key"] == "ABCD1234" and s["seq"] == "3"
assert s["section"] == "Some Heading" and s["p_id"] == ""
def test_rule_kind_carries_p_id_not_seq(self):
from llm.links import as_source
s = as_source({**TestRule.MD, "item_key": "WXYZ5678", "seq": "9"}, "text", 0.0)
assert s["item_key"] == "WXYZ5678"
assert s["p_id"] == "938" and s["seq"] == ""
def test_missing_keys_default_to_empty(self):
from llm.links import as_source
s = as_source({"kind": "corpus"}, "text", 0.0)
assert s["item_key"] == "" and s["p_id"] == ""
assert s["seq"] == "" and s["section"] == ""
class TestRuleKindAndManual:
def test_rule_source_carries_rule_kind(self):
from llm.links import as_source
s = as_source(
{
"kind": "rule",
"item_key": "R1",
"p_id": "12",
"page": "43949",
"ordinal": "4",
"fr_volume": "91",
"html_url": "https://www.federalregister.gov/d/2026-14327",
"rule_kind": "proposed",
"title": "t",
},
"text",
0.5,
)
assert s["rule_kind"] == "proposed" and s["label"] == "91 FR 43949 ¶4"
assert s["url"].startswith("https://www.federalregister.gov/d/2026-14327#p-12")
assert (
as_source({"kind": "corpus", "rule_kind": "final"}, "t", 0.1)["rule_kind"]
== ""
)
def test_manual_section_label_and_link(self):
from llm.links import as_source
s = as_source(
{
"kind": "corpus",
"doctype": "manual",
"manual": "100-04",
"chapter": "18",
"iom_section": "10.1.2",
"iom_title": "Influenza Virus Vaccine",
"page": "7",
"title": "Medicare Claims Processing Manual — Chapter 18",
"url": "https://www.cms.gov/x/clm104c18.pdf",
},
"text",
0.5,
)
assert s["label"] == "IOM 100-04 ch.18 §10.1.2"
assert s["url"] == "https://www.cms.gov/x/clm104c18.pdf#page=7"
assert (s["doctype"], s["manual"], s["chapter"], s["iom_section"]) == (
"manual",
"100-04",
"18",
"10.1.2",
)
# a manual chunk before the first section (front matter) keeps the title label
s2 = as_source(
{
"kind": "corpus",
"doctype": "manual",
"manual": "",
"chapter": "",
"title": "Chapter 18",
"year": "2024",
},
"t",
0.5,
)
assert s2["label"] == "Chapter 18 (2024)"