Some checks failed
CI / lint (push) Successful in 40s
CI / notebooks-smoke (push) Successful in 1m52s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
CI / test (push) Failing after 3m8s
Infra CI / zotero (push) Successful in 31s
Infra CI / notebooks (push) Successful in 1m4s
Infra CI / api (push) Successful in 1m29s
Infra CI / docs (push) Successful in 1m55s
Infra CI / mc (push) Failing after 58s
Infra CI / llm (push) Successful in 1m20s
Deploy / report (push) Successful in 22s
Every Federal Register rule item was indexed twice: as anchored FR paragraphs in the rules collection (67k chunks, federalregister.gov links) and again from its PDF attachment into corpus (161k chunks, kind 'corpus', item-URL links). The PDF copies outnumbered the anchored ones 2:1 in retrieval, so proposed rules and corrections came back as 'corpus' citations. iter_corpus_refs now skips rule-typed and fr_anchors-bearing items; stack llm prune-rules deletes the corpus copies (and their index_state rows) and stamps rule_kind (proposed/final/correction, bib.frlink.rule_kind) on the rules chunks; new rule docs carry rule_kind from the start. as_source exposes it, the prompt's source line reads '(proposed rule, 2026-07-16)', and the chat and search badges show it instead of 'rule'. CMS Internet-Only Manual chapters: llm.manuals.sectionize turns each body section heading (the ones followed by their '(Rev.' line; the table of contents stays plain) into a markdown heading, so chunks split at section boundaries and carry section / iom_section / iom_title plus the chapter's publication number and chapter; enrich_pdf_pages lets chunks under a sub-heading inherit the enclosing file so they keep their page. Manual citations read 'IOM 100-04 ch.18 §10.1.2' and link to the chapter PDF page; /search takes doctype, manual, chapter and section filters.
178 lines
6.4 KiB
Python
178 lines
6.4 KiB
Python
"""llm.pages — chunk → PDF page location via PyMuPDF."""
|
|
|
|
import fitz
|
|
import pytest
|
|
|
|
from llm import pages as pages_mod
|
|
from llm.chunk import Chunk, Doc
|
|
from llm.pages import enrich_pdf_pages, locate, pdf_pages
|
|
|
|
|
|
@pytest.fixture
|
|
def pdf(tmp_path):
|
|
path = tmp_path / "attachment_1.pdf"
|
|
doc = fitz.open()
|
|
for i, body in enumerate(
|
|
["Page one talks about telehealth originating sites.", "Page two covers E/M."]
|
|
):
|
|
page = doc.new_page()
|
|
page.insert_text((72, 72), f"Header {i + 1}\n{body}")
|
|
doc.save(path)
|
|
doc.close()
|
|
return path
|
|
|
|
|
|
class TestPdfPages:
|
|
def test_one_normalized_string_per_page(self, pdf):
|
|
pages = pdf_pages(pdf)
|
|
assert len(pages) == 2
|
|
assert "telehealth originating sites" in pages[0]
|
|
assert "\n" not in pages[0]
|
|
|
|
def test_unreadable_file_is_empty(self, tmp_path):
|
|
bad = tmp_path / "x.pdf"
|
|
bad.write_bytes(b"not a pdf")
|
|
assert pdf_pages(bad) == []
|
|
|
|
|
|
class TestLocate:
|
|
def test_finds_page_by_probe(self):
|
|
assert locate(["alpha beta gamma", "delta epsilon"], "delta epsilon") == 2
|
|
|
|
def test_probe_normalized_before_search(self):
|
|
assert locate(["alpha beta\ngamma"], "alpha beta gamma") == 1
|
|
|
|
def test_not_found_is_zero(self):
|
|
assert locate(["alpha"], "zeta") == 0
|
|
|
|
def test_short_probe_is_zero(self):
|
|
assert locate(["ab cd"], "ab") == 0
|
|
|
|
|
|
class TestEnrich:
|
|
def _chunk(self, text, section):
|
|
return Chunk(id="c", text=text, metadata={"section": section, "seq": "0"})
|
|
|
|
def test_sets_attachment_and_page_for_pdf_sections(self, pdf):
|
|
doc = Doc(
|
|
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
|
|
)
|
|
chunks = [
|
|
self._chunk("Page two covers E/M.", "attachment_1.pdf"),
|
|
self._chunk("Inline abstract text", ""),
|
|
]
|
|
out = enrich_pdf_pages(doc, chunks)
|
|
assert out[0].metadata["attachment"] == "attachment_1.pdf"
|
|
assert out[0].metadata["page"] == "2"
|
|
assert "attachment" not in out[1].metadata
|
|
assert out[0].id == "c" and out[0].text == chunks[0].text
|
|
|
|
def test_leading_section_heading_ignored_when_probing(self, pdf):
|
|
doc = Doc(
|
|
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
|
|
)
|
|
c = self._chunk(
|
|
"## attachment_1.pdf\n\nPage one talks about telehealth originating sites.",
|
|
"attachment_1.pdf",
|
|
)
|
|
(out,) = enrich_pdf_pages(doc, [c])
|
|
assert out.metadata["page"] == "1"
|
|
|
|
def test_unlocated_chunk_keeps_attachment_without_page(self, pdf):
|
|
doc = Doc(
|
|
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
|
|
)
|
|
(out,) = enrich_pdf_pages(
|
|
doc, [self._chunk("nothing matches here", "attachment_1.pdf")]
|
|
)
|
|
assert out.metadata["attachment"] == "attachment_1.pdf"
|
|
assert out.metadata["page"] == ""
|
|
|
|
def test_non_pdf_section_gets_attachment_only(self, tmp_path):
|
|
docx = tmp_path / "attachment_1.docx"
|
|
docx.write_bytes(b"x")
|
|
doc = Doc(
|
|
key="K", text="", metadata={}, files=(("attachment_1.docx", str(docx)),)
|
|
)
|
|
(out,) = enrich_pdf_pages(doc, [self._chunk("body", "attachment_1.docx")])
|
|
assert out.metadata["attachment"] == "attachment_1.docx"
|
|
assert "page" not in out.metadata
|
|
|
|
def test_no_files_is_identity(self):
|
|
doc = Doc(key="K", text="", metadata={})
|
|
chunks = [self._chunk("body", "attachment_1.pdf")]
|
|
assert enrich_pdf_pages(doc, chunks) == chunks
|
|
|
|
def test_does_not_override_existing_page(self, pdf):
|
|
doc = Doc(
|
|
key="K", text="", metadata={}, files=(("attachment_1.pdf", str(pdf)),)
|
|
)
|
|
c = Chunk(
|
|
id="c",
|
|
text="Page two covers E/M.",
|
|
metadata={"section": "attachment_1.pdf", "page": "9"},
|
|
)
|
|
(out,) = enrich_pdf_pages(doc, [c])
|
|
assert out.metadata["page"] == "9"
|
|
|
|
|
|
# ── locate_section (#705 item 2: IOM section headings) ──────────────
|
|
|
|
|
|
class TestLocateSection:
|
|
TOC = (
|
|
"Table of Contents (Rev. 12780) 30.6.3 - Payment for Immunosuppressive "
|
|
"Therapy Management 30.6.4 - Evaluation and Management (E/M) Services "
|
|
"Furnished Incident to Physician's Service 30.6.5 - Physicians in Group"
|
|
)
|
|
BODY = (
|
|
"visit is for immunosuppressive therapy. 30.6.4 - Evaluation and "
|
|
"Management (E/M) Services Furnished Incident to Physician's Service by "
|
|
"Nonphysician Practitioners (Rev. 11288; Issued: 03-31-22) A. General"
|
|
)
|
|
|
|
def test_body_heading_beats_the_table_of_contents(self):
|
|
assert pages_mod.locate_section([self.TOC, "filler", self.BODY], "30.6.4") == 3
|
|
|
|
def test_toc_only_is_not_a_location(self):
|
|
assert pages_mod.locate_section([self.TOC], "30.6.4") == 0
|
|
|
|
def test_prefix_numbers_do_not_match(self):
|
|
# "30.6.4" must not fire on "130.6.4" or "30.6.4.1"
|
|
body = "130.6.4 - Other (Rev. 1) 30.6.4.1 - Sub (Rev. 2)"
|
|
assert pages_mod.locate_section([body], "30.6.4") == 0
|
|
|
|
def test_trailing_period_and_blank_section(self):
|
|
assert pages_mod.locate_section([self.BODY], "30.6.4.") == 1
|
|
assert pages_mod.locate_section([self.BODY], "") == 0
|
|
|
|
|
|
class TestEnrichInheritsFile:
|
|
def test_sub_heading_chunks_inherit_the_enclosing_file(self, pdf):
|
|
"""IOM sections (llm.manuals) sit under the file heading: their
|
|
chunks inherit the attachment and get a page, and a chunk before
|
|
any file heading stays untouched."""
|
|
doc = Doc(key="K", text="", metadata={}, files=(("clm104c18.pdf", str(pdf)),))
|
|
|
|
def mk(section, text):
|
|
return Chunk(id="c", text=text, metadata={"section": section, "seq": "0"})
|
|
|
|
out = enrich_pdf_pages(
|
|
doc,
|
|
[
|
|
mk("", "front matter"),
|
|
mk("clm104c18.pdf", "telehealth originating sites"),
|
|
mk("10.1 - Coverage", "telehealth originating sites"),
|
|
],
|
|
)
|
|
assert "attachment" not in out[0].metadata
|
|
assert (
|
|
out[1].metadata["attachment"] == "clm104c18.pdf"
|
|
and out[1].metadata["page"] == "1"
|
|
)
|
|
assert (
|
|
out[2].metadata["attachment"] == "clm104c18.pdf"
|
|
and out[2].metadata["page"] == "1"
|
|
)
|
|
assert out[2].metadata["section"] == "10.1 - Coverage"
|