Files
stack/tests/llm/test_pages.py
kert 6eb4f2cb8b
Some checks failed
CI / lint (push) Successful in 40s
CI / notebooks-smoke (push) Successful in 1m52s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
CI / test (push) Failing after 3m8s
Infra CI / zotero (push) Successful in 31s
Infra CI / notebooks (push) Successful in 1m4s
Infra CI / api (push) Successful in 1m29s
Infra CI / docs (push) Successful in 1m55s
Infra CI / mc (push) Failing after 58s
Infra CI / llm (push) Successful in 1m20s
Deploy / report (push) Successful in 22s
fix(llm): rules never double as corpus; citations say proposed/final/correction; IOM chapters indexed by section
Every Federal Register rule item was indexed twice: as anchored FR
paragraphs in the rules collection (67k chunks, federalregister.gov
links) and again from its PDF attachment into corpus (161k chunks,
kind 'corpus', item-URL links). The PDF copies outnumbered the anchored
ones 2:1 in retrieval, so proposed rules and corrections came back as
'corpus' citations. iter_corpus_refs now skips rule-typed and
fr_anchors-bearing items; stack llm prune-rules deletes the corpus
copies (and their index_state rows) and stamps rule_kind
(proposed/final/correction, bib.frlink.rule_kind) on the rules chunks;
new rule docs carry rule_kind from the start. as_source exposes it, the
prompt's source line reads '(proposed rule, 2026-07-16)', and the chat
and search badges show it instead of 'rule'.

CMS Internet-Only Manual chapters: llm.manuals.sectionize turns each
body section heading (the ones followed by their '(Rev.' line; the
table of contents stays plain) into a markdown heading, so chunks split
at section boundaries and carry section / iom_section / iom_title plus
the chapter's publication number and chapter; enrich_pdf_pages lets
chunks under a sub-heading inherit the enclosing file so they keep
their page. Manual citations read 'IOM 100-04 ch.18 §10.1.2' and link
to the chapter PDF page; /search takes doctype, manual, chapter and
section filters.
2026-09-24 19:38:17 -04:00

178 lines
6.4 KiB
Python

"""llm.pages — chunk → PDF page location via PyMuPDF."""
import fitz
import pytest
from llm import pages as pages_mod
from llm.chunk import Chunk, Doc
from llm.pages import enrich_pdf_pages, locate, pdf_pages
@pytest.fixture
def pdf(tmp_path):
path = tmp_path / "attachment_1.pdf"
doc = fitz.open()
for i, body in enumerate(
["Page one talks about telehealth originating sites.", "Page two covers E/M."]
):
page = doc.new_page()
page.insert_text((72, 72), f"Header {i + 1}\n{body}")
doc.save(path)
doc.close()
return path
class TestPdfPages:
def test_one_normalized_string_per_page(self, pdf):
pages = pdf_pages(pdf)
assert len(pages) == 2
assert "telehealth originating sites" in pages[0]
assert "\n" not in pages[0]
def test_unreadable_file_is_empty(self, tmp_path):
bad = tmp_path / "x.pdf"
bad.write_bytes(b"not a pdf")
assert pdf_pages(bad) == []
class TestLocate:
def test_finds_page_by_probe(self):
assert locate(["alpha beta gamma", "delta epsilon"], "delta epsilon") == 2
def test_probe_normalized_before_search(self):
assert locate(["alpha beta\ngamma"], "alpha beta gamma") == 1
def test_not_found_is_zero(self):
assert locate(["alpha"], "zeta") == 0
def test_short_probe_is_zero(self):
assert locate(["ab cd"], "ab") == 0
class TestEnrich:
def _chunk(self, text, section):
return Chunk(id="c", text=text, metadata={"section": section, "seq": "0"})
def test_sets_attachment_and_page_for_pdf_sections(self, pdf):
doc = Doc(
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
)
chunks = [
self._chunk("Page two covers E/M.", "attachment_1.pdf"),
self._chunk("Inline abstract text", ""),
]
out = enrich_pdf_pages(doc, chunks)
assert out[0].metadata["attachment"] == "attachment_1.pdf"
assert out[0].metadata["page"] == "2"
assert "attachment" not in out[1].metadata
assert out[0].id == "c" and out[0].text == chunks[0].text
def test_leading_section_heading_ignored_when_probing(self, pdf):
doc = Doc(
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
)
c = self._chunk(
"## attachment_1.pdf\n\nPage one talks about telehealth originating sites.",
"attachment_1.pdf",
)
(out,) = enrich_pdf_pages(doc, [c])
assert out.metadata["page"] == "1"
def test_unlocated_chunk_keeps_attachment_without_page(self, pdf):
doc = Doc(
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
)
(out,) = enrich_pdf_pages(
doc, [self._chunk("nothing matches here", "attachment_1.pdf")]
)
assert out.metadata["attachment"] == "attachment_1.pdf"
assert out.metadata["page"] == ""
def test_non_pdf_section_gets_attachment_only(self, tmp_path):
docx = tmp_path / "attachment_1.docx"
docx.write_bytes(b"x")
doc = Doc(
key="K", text="", metadata={}, files=(("attachment_1.docx", str(docx)),)
)
(out,) = enrich_pdf_pages(doc, [self._chunk("body", "attachment_1.docx")])
assert out.metadata["attachment"] == "attachment_1.docx"
assert "page" not in out.metadata
def test_no_files_is_identity(self):
doc = Doc(key="K", text="", metadata={})
chunks = [self._chunk("body", "attachment_1.pdf")]
assert enrich_pdf_pages(doc, chunks) == chunks
def test_does_not_override_existing_page(self, pdf):
doc = Doc(
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
)
c = Chunk(
id="c",
text="Page two covers E/M.",
metadata={"section": "attachment_1.pdf", "page": "9"},
)
(out,) = enrich_pdf_pages(doc, [c])
assert out.metadata["page"] == "9"
# ── locate_section (#705 item 2: IOM section headings) ──────────────
class TestLocateSection:
TOC = (
"Table of Contents (Rev. 12780) 30.6.3 - Payment for Immunosuppressive "
"Therapy Management 30.6.4 - Evaluation and Management (E/M) Services "
"Furnished Incident to Physician's Service 30.6.5 - Physicians in Group"
)
BODY = (
"visit is for immunosuppressive therapy. 30.6.4 - Evaluation and "
"Management (E/M) Services Furnished Incident to Physician's Service by "
"Nonphysician Practitioners (Rev. 11288; Issued: 03-31-22) A. General"
)
def test_body_heading_beats_the_table_of_contents(self):
assert pages_mod.locate_section([self.TOC, "filler", self.BODY], "30.6.4") == 3
def test_toc_only_is_not_a_location(self):
assert pages_mod.locate_section([self.TOC], "30.6.4") == 0
def test_prefix_numbers_do_not_match(self):
# "30.6.4" must not fire on "130.6.4" or "30.6.4.1"
body = "130.6.4 - Other (Rev. 1) 30.6.4.1 - Sub (Rev. 2)"
assert pages_mod.locate_section([body], "30.6.4") == 0
def test_trailing_period_and_blank_section(self):
assert pages_mod.locate_section([self.BODY], "30.6.4.") == 1
assert pages_mod.locate_section([self.BODY], "") == 0
class TestEnrichInheritsFile:
def test_sub_heading_chunks_inherit_the_enclosing_file(self, pdf):
"""IOM sections (llm.manuals) sit under the file heading: their
chunks inherit the attachment and get a page, and a chunk before
any file heading stays untouched."""
doc = Doc(key="K", text="", metadata={}, files=(("clm104c18.pdf", str(pdf)),))
def mk(section, text):
return Chunk(id="c", text=text, metadata={"section": section, "seq": "0"})
out = enrich_pdf_pages(
doc,
[
mk("", "front matter"),
mk("clm104c18.pdf", "telehealth originating sites"),
mk("10.1 - Coverage", "telehealth originating sites"),
],
)
assert "attachment" not in out[0].metadata
assert (
out[1].metadata["attachment"] == "clm104c18.pdf"
and out[1].metadata["page"] == "1"
)
assert (
out[2].metadata["attachment"] == "clm104c18.pdf"
and out[2].metadata["page"] == "1"
)
assert out[2].metadata["section"] == "10.1 - Coverage"