Some checks failed
CI / lint (push) Successful in 39s
CI / notebooks-smoke (push) Successful in 1m39s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m29s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 15s
Infra CI / docs (push) Successful in 20s
Infra CI / notebooks (push) Successful in 52s
Infra CI / api (push) Successful in 1m2s
Infra CI / llm (push) Successful in 47s
Deploy / report (push) Successful in 14s
Infra CI / mc (push) Failing after 37s
Rules keep their federalregister.gov links; this is for the files the library holds itself: a regulations.gov comment's downloaded attachments (.state/comments/<docket>/<id>/attachment_N.*), an item's bib attachments (data/storage/<att>/<name>) and Zotero-only storage PDFs (zotero.sqlite read immutable, one indexed lookup per request). llm.pdfs lists an item's attachments from those three places (comment first, de-duplicated by name) and resolves a file by *name* only, refusing anything outside the storage roots. GET /pdf/<key> lists them (name, size, source, renderable, media type); GET /pdf/<key>/file?name= serves a PDF inline with range requests and other formats as downloads. /ui/vendor/<name> serves the vendored pdf.js 4.10.38 (same-origin worker; no CDN dependency). /ui/pdf/<key>?file=&page=&q=: continuous scroll, pages rendered lazily (IntersectionObserver, ~1.5 screens ahead) at device pixel ratio, re-fit on resize/orientation (ResizeObserver, debounced), fit-to-width and zoom, page input with hash tracking, a text layer for selection, whole-document find with highlights (seeded from the cited snippet), keyboard shortcuts, a file selector when an item has several, and a download fallback when a file is not a PDF or cannot be rendered. as_source carries attachment + page from the chunk metadata; the chat sources and search hits show a 'PDF p.N' link into the viewer for comment/corpus sources. compose: the llm service mounts ./.state/comments read-only.
187 lines
5.9 KiB
Python
187 lines
5.9 KiB
Python
"""llm.links — kind-specific evidence deep links (pure)."""
|
|
|
|
from llm.links import for_source
|
|
|
|
|
|
class TestComment:
|
|
def test_plain_comment_links_to_comment_page(self):
|
|
url, label = for_source(
|
|
{"kind": "comment", "comment_id": "CMS-2026-2377-3438"}, "snippet"
|
|
)
|
|
assert url == "https://www.regulations.gov/comment/CMS-2026-2377-3438"
|
|
assert label == "CMS-2026-2377-3438"
|
|
|
|
def test_pdf_attachment_chunk_links_to_page(self):
|
|
url, label = for_source(
|
|
{
|
|
"kind": "comment",
|
|
"comment_id": "CMS-2026-2377-3438",
|
|
"attachment": "attachment_2.pdf",
|
|
"page": "4",
|
|
},
|
|
"s",
|
|
)
|
|
assert url == (
|
|
"https://downloads.regulations.gov/CMS-2026-2377-3438/attachment_2.pdf#page=4"
|
|
)
|
|
assert label == "CMS-2026-2377-3438 p.4"
|
|
|
|
def test_non_pdf_attachment_no_page(self):
|
|
url, label = for_source(
|
|
{"kind": "comment", "comment_id": "C-1", "attachment": "attachment_1.docx"},
|
|
"s",
|
|
)
|
|
assert url == "https://downloads.regulations.gov/C-1/attachment_1.docx"
|
|
assert label == "C-1"
|
|
|
|
def test_missing_comment_id_falls_back_to_item_key(self):
|
|
url, label = for_source({"kind": "comment", "item_key": "ABCD1234"}, "s")
|
|
assert label == "ABCD1234"
|
|
assert url == ""
|
|
|
|
|
|
class TestRule:
|
|
MD = {
|
|
"kind": "rule",
|
|
"html_url": "https://www.federalregister.gov/documents/2026/07/16/2026-14327/x",
|
|
"p_id": "938",
|
|
"page": "43949",
|
|
"ordinal": "4",
|
|
"fr_volume": "91",
|
|
}
|
|
|
|
def test_paragraph_link_with_highlight(self):
|
|
url, label = for_source(
|
|
self.MD,
|
|
"In the FY 2027 Hospice proposed rule, CMS solicited comment. More.",
|
|
)
|
|
assert url == (
|
|
"https://www.federalregister.gov/documents/2026/07/16/2026-14327/x#p-938"
|
|
":~:text=In%20the%20FY%202027%20Hospice%20proposed%20rule%2C%20CMS%20solicited%20comment."
|
|
)
|
|
assert label == "91 FR 43949 ¶4"
|
|
|
|
def test_no_p_id_degrades_to_page_link(self):
|
|
md = {**self.MD, "p_id": "", "ordinal": ""}
|
|
url, label = for_source(md, "s")
|
|
assert url.endswith("/x#page-43949")
|
|
assert label == "91 FR 43949"
|
|
|
|
def test_no_page_degrades_to_document(self):
|
|
md = {
|
|
"kind": "rule",
|
|
"html_url": "https://fr.test/doc",
|
|
"title": "CY2027 PFS NPRM",
|
|
}
|
|
url, label = for_source(md, "s")
|
|
assert url == "https://fr.test/doc"
|
|
assert label == "CY2027 PFS NPRM"
|
|
|
|
|
|
class TestCorpus:
|
|
def test_item_url_and_short_title_with_year(self):
|
|
url, label = for_source(
|
|
{
|
|
"kind": "corpus",
|
|
"url": "https://pubmed.ncbi.nlm.nih.gov/19922199/",
|
|
"title": "A consensus on palliative care quality metrics for hospital programs",
|
|
"year": "2009",
|
|
},
|
|
"s",
|
|
)
|
|
assert url == "https://pubmed.ncbi.nlm.nih.gov/19922199/"
|
|
assert (
|
|
label
|
|
== "A consensus on palliative care quality metrics for hospital… (2009)"
|
|
)
|
|
|
|
def test_pdf_url_gets_page(self):
|
|
url, _ = for_source(
|
|
{
|
|
"kind": "corpus",
|
|
"url": "https://x.test/report.pdf",
|
|
"title": "R",
|
|
"page": "7",
|
|
},
|
|
"s",
|
|
)
|
|
assert url == "https://x.test/report.pdf#page=7"
|
|
|
|
def test_non_pdf_url_ignores_page(self):
|
|
url, _ = for_source(
|
|
{
|
|
"kind": "corpus",
|
|
"url": "https://x.test/report",
|
|
"title": "R",
|
|
"page": "7",
|
|
},
|
|
"s",
|
|
)
|
|
assert url == "https://x.test/report"
|
|
|
|
def test_untitled_falls_back_to_item_key(self):
|
|
_, label = for_source({"kind": "corpus", "item_key": "K1", "url": ""}, "s")
|
|
assert label == "K1"
|
|
|
|
|
|
def test_unknown_kind_treated_as_corpus():
|
|
url, label = for_source({"url": "https://u.test", "title": "T"}, "s")
|
|
assert (url, label) == ("https://u.test", "T")
|
|
|
|
|
|
class TestAsSource:
|
|
def test_shape_matches_rag_source(self):
|
|
from llm.links import as_source
|
|
|
|
s = as_source(
|
|
{
|
|
"kind": "comment",
|
|
"comment_id": "CMS-2026-2377-1",
|
|
"date": "2026-08-19",
|
|
"title": "t",
|
|
"docket": "CMS-2026-2377",
|
|
"item_key": "ABCD1234",
|
|
"seq": "3",
|
|
"section": "Some Heading",
|
|
},
|
|
"body " * 200,
|
|
0.123456,
|
|
)
|
|
assert set(s) == {
|
|
"attachment",
|
|
"page",
|
|
"id",
|
|
"label",
|
|
"kind",
|
|
"url",
|
|
"title",
|
|
"date",
|
|
"docket",
|
|
"comment_id",
|
|
"snippet",
|
|
"score",
|
|
"item_key",
|
|
"p_id",
|
|
"seq",
|
|
"section",
|
|
}
|
|
assert s["label"] == "CMS-2026-2377-1" and s["kind"] == "comment"
|
|
assert len(s["snippet"]) <= 500 and s["score"] == 0.1235
|
|
# non-rule kind: seq/section carried through, p_id blanked
|
|
assert s["item_key"] == "ABCD1234" and s["seq"] == "3"
|
|
assert s["section"] == "Some Heading" and s["p_id"] == ""
|
|
|
|
def test_rule_kind_carries_p_id_not_seq(self):
|
|
from llm.links import as_source
|
|
|
|
s = as_source({**TestRule.MD, "item_key": "WXYZ5678", "seq": "9"}, "text", 0.0)
|
|
assert s["item_key"] == "WXYZ5678"
|
|
assert s["p_id"] == "938" and s["seq"] == ""
|
|
|
|
def test_missing_keys_default_to_empty(self):
|
|
from llm.links import as_source
|
|
|
|
s = as_source({"kind": "corpus"}, "text", 0.0)
|
|
assert s["item_key"] == "" and s["p_id"] == ""
|
|
assert s["seq"] == "" and s["section"] == ""
|