feat(llm): in-page PDF viewer for comment and article attachments — /ui/pdf/<key>, /pdf/<key>[/file], self-hosted PDF.js, viewer links on chat sources and search hits
Some checks failed
CI / lint (push) Successful in 39s
CI / notebooks-smoke (push) Successful in 1m39s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m29s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 15s
Infra CI / docs (push) Successful in 20s
Infra CI / notebooks (push) Successful in 52s
Infra CI / api (push) Successful in 1m2s
Infra CI / llm (push) Successful in 47s
Deploy / report (push) Successful in 14s
Infra CI / mc (push) Failing after 37s

Rules keep their federalregister.gov links; this is for the files the
library holds itself: a regulations.gov comment's downloaded
attachments (.state/comments/<docket>/<id>/attachment_N.*), an item's
bib attachments (data/storage/<att>/<name>) and Zotero-only storage
PDFs (zotero.sqlite read immutable, one indexed lookup per request).

llm.pdfs lists an item's attachments from those three places (comment
first, de-duplicated by name) and resolves a file by *name* only,
refusing anything outside the storage roots. GET /pdf/<key> lists
them (name, size, source, renderable, media type); GET
/pdf/<key>/file?name= serves a PDF inline with range requests and
other formats as downloads. /ui/vendor/<name> serves the vendored
pdf.js 4.10.38 (same-origin worker; no CDN dependency).

/ui/pdf/<key>?file=&page=&q=: continuous scroll, pages rendered lazily
(IntersectionObserver, ~1.5 screens ahead) at device pixel ratio,
re-fit on resize/orientation (ResizeObserver, debounced), fit-to-width
and zoom, page input with hash tracking, a text layer for selection,
whole-document find with highlights (seeded from the cited snippet),
keyboard shortcuts, a file selector when an item has several, and a
download fallback when a file is not a PDF or cannot be rendered.

as_source carries attachment + page from the chunk metadata; the chat
sources and search hits show a 'PDF p.N' link into the viewer for
comment/corpus sources. compose: the llm service mounts
./.state/comments read-only.
This commit is contained in:
kert
2026-09-24 18:56:10 -04:00
parent 47e0dba7d0
commit 1b81aeac92
12 changed files with 1067 additions and 1 deletions

View File

@@ -691,6 +691,7 @@ services:
# mount shares the WAL/SHM siblings like the api service does.
- ./data:/app/data
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
environment:
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
- LLM_PG_HOST=postgres

View File

@@ -15,10 +15,11 @@ import re
import time
from contextlib import asynccontextmanager
from importlib import resources
from pathlib import Path
from typing import AsyncIterator, Iterator
from fastapi import FastAPI, Header, HTTPException
from fastapi.responses import HTMLResponse, StreamingResponse
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
from pydantic import BaseModel
log = logging.getLogger(__name__)
@@ -107,6 +108,129 @@ def search_page() -> str:
return _page("search.html")
_VENDOR = {"pdf.min.mjs", "pdf.worker.min.mjs"}
@app.get("/ui/vendor/{name}")
def vendor(name: str) -> Response:
"""Self-hosted browser libraries (PDF.js) — a fixed allow-list, served
from the package so the viewer works without a CDN and its worker is
same-origin (a cross-origin worker script is refused by browsers)."""
if name not in _VENDOR:
raise HTTPException(404)
body = resources.files("llm").joinpath(f"web/vendor/{name}").read_bytes()
return Response(
body,
media_type="text/javascript",
headers={"Cache-Control": "public, max-age=86400"},
)
@app.get("/ui/pdf/{key}", response_class=HTMLResponse)
def pdf_page(key: str) -> str:
"""The PDF viewer; the page reads its item key, file, page and search
text from its own URL and fetches /pdf/<key> for the file list."""
if not _KEY_RE.match(key):
raise HTTPException(404)
return _page("pdf.html")
_KEY_RE = re.compile(r"^[A-Za-z0-9]{6,12}$")
def _pdf_paths() -> tuple[Path, Path, Path]:
"""(zotero.sqlite, zotero storage, comments root) — the comments root
is the indexer's (``llm.source._default_root``): ``.state/comments``."""
from conf import ROOT
from conf import path as cpath
return (
Path(cpath("db.zotero")),
Path(cpath("storage.zotero")),
Path(ROOT) / ".state" / "comments",
)
def _pdf_files(key: str) -> list:
from conf.connect import bib
from llm.pdfs import pdf_files
store = bib()
try:
zdb, zstore, croot = _pdf_paths()
return pdf_files(
store, key, zotero_sqlite=zdb, zotero_storage=zstore, comments_root=croot
)
finally:
store.close()
@app.get("/pdf/{key}")
def pdf_list(key: str) -> dict:
"""The attachments on file for a comment or library item: name, size,
where it came from (comment download, bib attachment, Zotero storage)
and whether the viewer can render it (PDF) or only offer it for
download. Rules are not served here — they link to federalregister.gov."""
if not _KEY_RE.match(key):
raise HTTPException(404)
files = _pdf_files(key)
title = ""
try:
from conf.connect import bib
store = bib()
try:
title = store.get(key).title or ""
finally:
store.close()
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
title = ""
return {
"key": key,
"title": title,
"files": [
{
"name": f.name,
"size": f.size,
"source": f.source,
"renderable": f.renderable,
"media_type": f.media_type,
}
for f in files
],
}
@app.get("/pdf/{key}/file")
def pdf_file(key: str, name: str = "") -> FileResponse:
"""The attachment bytes — a PDF inline with range requests (PDF.js
streams pages from a large file instead of waiting for all of it),
anything else as a download."""
from llm.pdfs import resolve
if not _KEY_RE.match(key):
raise HTTPException(404)
files = _pdf_files(key)
from conf.connect import bib
store = bib()
try:
bib_root = Path(getattr(store, "_storage", "storage"))
finally:
store.close()
_zdb, zstore, croot = _pdf_paths()
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
if chosen is None:
raise HTTPException(404, "no such attachment on file for this item")
return FileResponse(
chosen.path,
media_type=chosen.media_type,
filename=chosen.name,
content_disposition_type="inline" if chosen.renderable else "attachment",
headers={"Accept-Ranges": "bytes", "Cache-Control": "private, max-age=3600"},
)
@app.get("/whoami")
def whoami(x_auth_request_user: str = Header(default="")) -> dict:
"""The Gitea username Traefik forwarded (empty if unset)."""

View File

@@ -114,4 +114,8 @@ def as_source(md: dict[str, str], text: str, score: float) -> dict:
"p_id": md.get("p_id", "") if kind == "rule" else "",
"seq": md.get("seq", "") if kind != "rule" else "",
"section": md.get("section", ""),
# the PDF the chunk came from and the page it was located on
# (``llm.pages.enrich_pdf_pages``) — the UI links them to the viewer
"attachment": md.get("attachment", ""),
"page": md.get("page", ""),
}

174
src/llm/pdfs.py Normal file
View File

@@ -0,0 +1,174 @@
"""Attachments on file for a library item, for the viewer (``/ui/pdf/<key>``).
Comment attachments and article attachments — never rules, which link
straight to federalregister.gov. The same places the indexer reads
(``llm.source``): a regulations.gov comment's downloaded attachments
(``.state/comments/<docket>/<comment id>/attachment_N.pdf`` …), the
item's bib attachments (``data/storage/<attachment key>/<filename>``)
and, for Zotero-only material, the storage PDFs of the item's Zotero
attachments (``data/zotero/data/storage/<attachment key>/<filename>``),
read from the live ``zotero.sqlite`` in immutable mode (a cheap indexed
lookup per item; the indexer's 2 GB snapshot is for whole-library walks,
not one request).
PDFs render in the page; other formats (docx, doc, txt) are listed for
download. A file is only ever served after it was found through this
lookup by its *name* — the API never joins a client-supplied path — and
only when it sits inside one of the storage roots.
"""
from __future__ import annotations
import logging
import sqlite3
from dataclasses import dataclass
from pathlib import Path
from typing import Any
log = logging.getLogger(__name__)
_MEDIA = {
".pdf": "application/pdf",
".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
".doc": "application/msword",
".txt": "text/plain",
}
_ATTACHMENT_EXT = tuple(_MEDIA)
@dataclass(frozen=True)
class PdfFile:
name: str
path: Path
size: int
source: str # "comment" | "bib" | "zotero"
@property
def media_type(self) -> str:
return _MEDIA.get(Path(self.name).suffix.lower(), "application/octet-stream")
@property
def renderable(self) -> bool:
return self.name.lower().endswith(".pdf")
def _comment_files(store: Any, key: str, root: Path) -> list[PdfFile]:
"""A regulations.gov comment's downloaded attachments, in file order."""
row = store._con().execute("SELECT url FROM items WHERE key = ?", (key,)).fetchone() # noqa: SLF001
url = (row[0] if row else "") or ""
prefix = "https://www.regulations.gov/comment/"
if not url.startswith(prefix):
return []
comment_id = url[len(prefix) :].strip("/")
if not comment_id or "/" in comment_id or comment_id.startswith("."):
return []
docket = comment_id.rsplit("-", 1)[0]
d = Path(root) / docket / comment_id
if not d.is_dir():
return []
return [
PdfFile(p.name, p, p.stat().st_size, "comment")
for p in sorted(d.iterdir())
if p.is_file() and p.suffix.lower() in _ATTACHMENT_EXT
]
def _bib_pdfs(store: Any, key: str) -> list[PdfFile]:
root = Path(getattr(store, "_storage", Path("storage")))
rows = (
store._con() # noqa: SLF001
.execute(
"SELECT a.key, a.filename, a.storage_path, a.content_type FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ? ORDER BY a.id",
(key,),
)
.fetchall()
)
out: list[PdfFile] = []
for att_key, filename, storage_path, ctype in rows:
name = filename or (storage_path or "").split(":", 1)[-1]
if not name or Path(name).suffix.lower() not in _ATTACHMENT_EXT:
continue
p = root / att_key / Path(name).name
if p.is_file():
out.append(PdfFile(Path(name).name, p, p.stat().st_size, "bib"))
return out
def _zotero_pdfs(key: str, *, sqlite_path: Path, storage_dir: Path) -> list[PdfFile]:
if not Path(sqlite_path).exists():
return []
try:
con = sqlite3.connect(f"file:{sqlite_path}?mode=ro&immutable=1", uri=True)
try:
rows = con.execute(
"SELECT a.key, ia.path FROM itemAttachments ia "
"JOIN items a ON a.itemID = ia.itemID "
"JOIN items p ON p.itemID = ia.parentItemID "
"WHERE p.key = ? AND ia.path LIKE 'storage:%.pdf' "
"AND a.itemID NOT IN (SELECT itemID FROM deletedItems) ORDER BY a.itemID",
(key,),
).fetchall()
finally:
con.close()
except sqlite3.Error as exc:
log.warning("zotero lookup failed for %s: %s", key, exc)
return []
out: list[PdfFile] = []
for att_key, path in rows:
name = Path(path[len("storage:") :]).name
p = Path(storage_dir) / att_key / name
if p.is_file():
out.append(PdfFile(name, p, p.stat().st_size, "zotero"))
return out
def pdf_files(
store: Any,
key: str,
*,
zotero_sqlite: Path | None = None,
zotero_storage: Path | None = None,
comments_root: Path | None = None,
) -> list[PdfFile]:
"""Every attachment on file for *key*: a comment's downloaded files
first, then bib attachments, then Zotero storage PDFs, de-duplicated
by file name (a Zotero copy of a bib attachment is the same document)."""
files: list[PdfFile] = []
if comments_root is not None:
files += _comment_files(store, key, comments_root)
files += _bib_pdfs(store, key)
if zotero_sqlite is not None and zotero_storage is not None:
files += _zotero_pdfs(
key, sqlite_path=zotero_sqlite, storage_dir=zotero_storage
)
seen: set[str] = set()
out: list[PdfFile] = []
for f in files:
if f.name in seen:
continue
seen.add(f.name)
out.append(f)
return out
def resolve(
files: list[PdfFile], name: str, *, roots: tuple[Path, ...]
) -> PdfFile | None:
"""The listed file called *name* (an empty name means the first),
provided its real path lies under one of *roots*."""
if not files:
return None
chosen = files[0] if not name else next((f for f in files if f.name == name), None)
if chosen is None:
return None
real = chosen.path.resolve()
for root in roots:
try:
real.relative_to(Path(root).resolve())
return chosen
except ValueError:
continue
log.warning("refusing %s: outside storage roots", chosen.path)
return None

View File

@@ -205,6 +205,12 @@
background: color-mix(in srgb, var(--primary) 12%, transparent); margin-right: 6px;
}
.src .date { color: var(--muted-fg); margin-left: 6px; }
.src a.viewer {
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; margin-left: 6px;
white-space: nowrap;
}
.src a.viewer:hover { background: color-mix(in srgb, var(--accent) 10%, transparent); }
.meta { font-size: 12px; color: var(--muted-fg); margin-top: 4px; font-family: var(--font-mono); }
.hint {
margin: auto; max-width: 500px; text-align: center; color: var(--muted-fg);
@@ -652,6 +658,21 @@
if (src.title && src.kind !== 'comment') {
el.appendChild(document.createTextNode(' — ' + src.title));
}
// A comment or article attachment on file opens in the PDF viewer
// on the page the passage was located on (rules link to the FR).
if (src.attachment && src.item_key && src.kind !== 'rule') {
const v = document.createElement('a');
v.className = 'viewer';
v.href = '/ui/pdf/' + encodeURIComponent(src.item_key) +
'?file=' + encodeURIComponent(src.attachment) +
(src.page ? '&page=' + encodeURIComponent(src.page) : '') +
'&q=' + encodeURIComponent((src.snippet || '').slice(0, 80));
v.target = '_blank'; v.rel = 'noopener';
v.textContent = (src.attachment.toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
(src.page ? ' p.' + src.page : '');
el.appendChild(document.createTextNode(' '));
el.appendChild(v);
}
el.appendChild(document.createTextNode(' ' + src.snippet));
d.appendChild(el);
}

468
src/llm/web/pdf.html Normal file
View File

@@ -0,0 +1,468 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Library PDF</title>
<link rel="icon" type="image/png" sizes="32x32" href="//dashboard.fhirworx.io/fav32.png">
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link rel="stylesheet"
href="https://fonts.googleapis.com/css2?family=Playfair+Display:wght@600;700&family=Source+Serif+4:wght@400;600&family=JetBrains+Mono:wght@500&display=swap">
<style>
:root {
--background: #F7F5F0; --foreground: #1A1A18; --card: #FAFAF7;
--primary: #1C2B3A; --primary-fg: #F7F5F0; --secondary: #EDEBE6;
--muted-fg: #6B6B68; --border: #D4D0C8; --accent: #2E3D8F;
--destructive: #C0392B; --highlight: rgba(255, 214, 0, .45);
--font-display: "Playfair Display", Georgia, serif;
--font-body: "Source Serif 4", Georgia, serif;
--font-mono: "JetBrains Mono", "Fira Code", monospace;
}
* { box-sizing: border-box; }
html, body { height: 100%; }
body {
margin: 0; display: flex; flex-direction: column; overflow: hidden;
background: var(--background); color: var(--foreground);
font-family: var(--font-body); font-size: 15px;
}
header {
flex: 0 0 auto; background: var(--primary); color: var(--primary-fg);
border-bottom: 3px solid var(--foreground); padding: 10px 16px;
display: flex; align-items: center; gap: 12px; flex-wrap: wrap; min-width: 0;
}
header h1 {
font-family: var(--font-display); font-weight: 700; font-size: 16px; margin: 0;
white-space: nowrap;
}
header .title {
flex: 1 1 240px; min-width: 0; font-size: 13px; opacity: .85;
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
}
header a.nav {
font-family: var(--font-mono); font-size: 11px; text-transform: uppercase;
letter-spacing: .08em; color: var(--primary-fg); text-decoration: underline dotted;
}
.toolbar {
flex: 0 0 auto; background: var(--secondary); border-bottom: 1px solid var(--border);
padding: 6px 12px; display: flex; align-items: center; gap: 8px; flex-wrap: wrap;
font-family: var(--font-mono); font-size: 12px;
}
.toolbar .group { display: flex; align-items: center; gap: 4px; }
.toolbar button, .toolbar select, .toolbar input {
font-family: var(--font-mono); font-size: 12px; height: 28px;
border: 1px solid var(--border); border-radius: 3px; background: var(--card); color: var(--foreground);
}
.toolbar button { padding: 0 10px; cursor: pointer; min-width: 30px; }
.toolbar button:hover:not(:disabled) { background: #E2DFD8; }
.toolbar button:disabled { opacity: .4; cursor: default; }
.toolbar button.primary { background: var(--primary); color: var(--primary-fg); border-color: var(--primary); }
.toolbar input#pageno { width: 58px; text-align: right; padding: 0 6px; }
.toolbar input#find { width: 200px; padding: 0 8px; font-family: var(--font-body); font-size: 13px; }
.toolbar .count, .toolbar .zoom { color: var(--muted-fg); min-width: 44px; text-align: center; }
.toolbar select#file { max-width: 260px; text-overflow: ellipsis; }
.toolbar a.dl { color: var(--accent); text-decoration: underline dotted; font-size: 11px; }
.toolbar .spacer { flex: 1 1 auto; }
#progress { height: 3px; background: transparent; flex: 0 0 auto; }
#progress > div { height: 100%; width: 0; background: var(--accent); transition: width .2s ease; }
#viewer {
flex: 1 1 auto; overflow: auto; padding: 18px 12px 40px;
display: flex; flex-direction: column; align-items: center; gap: 14px;
-webkit-overflow-scrolling: touch; overscroll-behavior: contain;
}
.page {
position: relative; background: #fff; box-shadow: 0 1px 3px rgba(0,0,0,.25);
flex: 0 0 auto; max-width: 100%;
}
.page canvas { display: block; width: 100%; height: 100%; }
.page .num {
position: absolute; right: 6px; bottom: 4px; font-family: var(--font-mono); font-size: 10px;
color: var(--muted-fg); background: rgba(255,255,255,.7); padding: 1px 4px; border-radius: 2px;
pointer-events: none;
}
.page .textLayer {
position: absolute; inset: 0; overflow: hidden; line-height: 1; opacity: 1;
text-size-adjust: none; forced-color-adjust: none; transform-origin: 0 0;
}
.page .textLayer span, .page .textLayer br {
color: transparent; position: absolute; white-space: pre; cursor: text; transform-origin: 0 0;
}
.page .textLayer ::selection { background: rgba(46, 61, 143, .35); }
.page .textLayer .endOfContent { display: block; position: absolute; inset: 100% 0 0; z-index: -1; cursor: default; user-select: none; }
.page .textLayer mark { background: var(--highlight); color: transparent; border-radius: 2px; }
.page .textLayer mark.current { background: rgba(255, 140, 0, .55); }
.page.pending { min-height: 300px; }
.page .placeholder {
position: absolute; inset: 0; display: flex; align-items: center; justify-content: center;
color: var(--muted-fg); font-family: var(--font-mono); font-size: 12px;
}
.notice {
margin: 40px auto; max-width: 560px; padding: 16px 20px; border: 1px solid var(--border);
background: var(--card); border-radius: 3px; color: var(--muted-fg); font-style: italic;
}
.notice.err { color: var(--destructive); border-color: var(--destructive); font-style: normal; }
.notice a { color: var(--accent); }
@media (max-width: 640px) {
header { padding: 8px 10px; gap: 8px; }
header .title { flex-basis: 100%; }
.toolbar { padding: 6px 8px; gap: 6px; }
.toolbar input#find { width: 130px; }
.toolbar select#file { max-width: 160px; }
#viewer { padding: 10px 6px 30px; gap: 10px; }
}
</style>
</head>
<body>
<header>
<h1>Library PDF</h1>
<span class="title" id="title"></span>
<a class="nav" href="/">Chat</a>
<a class="nav" href="/ui/search">Search</a>
</header>
<div class="toolbar">
<div class="group">
<select id="file" title="Which PDF on file for this item" hidden></select>
</div>
<div class="group">
<button id="prev" title="Previous page (←)">‹</button>
<input id="pageno" type="number" min="1" value="1" title="Page">
<span class="count" id="count">/ –</span>
<button id="next" title="Next page (→)">›</button>
</div>
<div class="group">
<button id="zoomout" title="Zoom out (−)">−</button>
<span class="zoom" id="zoomlabel">fit</span>
<button id="zoomin" title="Zoom in (+)">+</button>
<button id="fit" title="Fit to width (0)">fit</button>
</div>
<div class="group">
<input id="find" type="search" placeholder="find in document" title="Find (Enter: next, Shift+Enter: previous)">
<span class="count" id="findcount"></span>
</div>
<span class="spacer"></span>
<a class="dl" id="download" href="#" download>download</a>
</div>
<div id="progress"><div id="bar"></div></div>
<div id="viewer" tabindex="0"></div>
<script type="module">
// Everything below reads only this page's own URL and the JSON from
// /pdf/<key>; nothing from those ever reaches innerHTML.
const $ = id => document.getElementById(id);
const viewer = $('viewer'), bar = $('bar');
const params = new URLSearchParams(location.search);
const key = location.pathname.split('/').filter(Boolean).pop();
const wantFile = params.get('file') || '';
const wantPage = Math.max(1, parseInt(params.get('page') || '1', 10) || 1);
const wantQuery = (params.get('q') || '').trim();
function notice(text, err, downloadUrl) {
const d = document.createElement('div');
d.className = 'notice' + (err ? ' err' : '');
d.appendChild(document.createTextNode(text));
if (downloadUrl) {
d.appendChild(document.createTextNode(' '));
const a = document.createElement('a'); a.href = downloadUrl; a.textContent = 'Download the file instead.';
d.appendChild(a);
}
viewer.replaceChildren(d);
}
let pdfjs = null;
try {
pdfjs = await import('/ui/vendor/pdf.min.mjs');
pdfjs.GlobalWorkerOptions.workerSrc = '/ui/vendor/pdf.worker.min.mjs';
} catch (e) {
pdfjs = null;
}
// ── file list ──
let files = [];
try {
const r = await fetch('/pdf/' + encodeURIComponent(key));
if (!r.ok) throw new Error('lookup failed (' + r.status + ')');
const d = await r.json();
files = d.files || [];
$('title').textContent = d.title || key;
document.title = (d.title ? d.title + ' — ' : '') + 'Library PDF';
} catch (e) {
notice('Could not look up the item: ' + e.message, true);
throw e;
}
if (!files.length) {
notice('No PDF is on file for this item.', false);
throw new Error('no files');
}
const sel = $('file');
for (const f of files) {
const o = document.createElement('option');
o.value = f.name; o.textContent = f.name + ' (' + Math.round(f.size / 1024) + ' KB' + (f.renderable === false ? ', download only' : '') + ')';
sel.appendChild(o);
}
if (files.length > 1) sel.hidden = false;
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
sel.value = current.name;
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
$('download').href = fileUrl;
sel.addEventListener('change', () => {
const p = new URLSearchParams(location.search);
p.set('file', sel.value); p.delete('page');
location.search = p.toString();
});
if (current.renderable === false) {
notice('"' + current.name + '" is not a PDF (' + (current.media_type || 'file') + '), so it cannot be rendered here.', false, fileUrl);
throw new Error('not renderable');
}
if (!pdfjs) {
notice('The in-page renderer could not be loaded.', true, fileUrl);
throw new Error('pdfjs missing');
}
// ── document ──
let doc;
try {
const task = pdfjs.getDocument({ url: fileUrl, rangeChunkSize: 1 << 18, disableAutoFetch: false });
task.onProgress = ({ loaded, total }) => { if (total) bar.style.width = Math.min(100, 100 * loaded / total) + '%'; };
doc = await task.promise;
bar.style.width = '0';
} catch (e) {
notice('This file could not be rendered as a PDF: ' + (e && e.message || e), true, fileUrl);
throw e;
}
const total = doc.numPages;
$('count').textContent = '/ ' + total;
$('pageno').max = String(total);
// ── layout: one placeholder per page, sized from page 1's ratio,
// rendered lazily as they scroll into view, re-rendered on resize ──
const pages = []; // {el, canvas, textLayer, num, viewportW, rendered, task, text}
let scale = 1; // CSS px per PDF unit at the current zoom
let fitMode = true; // fit to width until the user zooms
let zoomFactor = 1;
const first = await doc.getPage(1);
const baseW = first.getViewport({ scale: 1 }).width;
const baseH = first.getViewport({ scale: 1 }).height;
function availableWidth() {
const cs = getComputedStyle(viewer);
return viewer.clientWidth - parseFloat(cs.paddingLeft) - parseFloat(cs.paddingRight);
}
function computeScale() {
const fit = Math.max(0.2, availableWidth() / baseW);
scale = fitMode ? fit : fit * zoomFactor;
$('zoomlabel').textContent = fitMode ? 'fit' : Math.round(zoomFactor * 100) + '%';
}
for (let n = 1; n <= total; n++) {
const el = document.createElement('div');
el.className = 'page pending'; el.dataset.page = String(n);
const ph = document.createElement('div'); ph.className = 'placeholder'; ph.textContent = 'page ' + n;
const num = document.createElement('span'); num.className = 'num'; num.textContent = String(n);
el.appendChild(ph); el.appendChild(num);
viewer.appendChild(el);
pages.push({ el, num: n, canvas: null, textLayer: null, rendered: 0, task: null, text: null, page: null });
}
function sizeAll() {
computeScale();
for (const p of pages) {
const pg = p.page;
const w = Math.floor((pg ? pg.getViewport({ scale: 1 }).width : baseW) * scale);
const h = Math.floor((pg ? pg.getViewport({ scale: 1 }).height : baseH) * scale);
p.el.style.width = w + 'px'; p.el.style.height = h + 'px';
}
}
async function render(p) {
if (p.rendered === scale) return;
if (p.task) { try { p.task.cancel(); } catch (e) {} p.task = null; }
if (!p.page) p.page = await doc.getPage(p.num);
const vp = p.page.getViewport({ scale });
const dpr = Math.min(window.devicePixelRatio || 1, 3);
let canvas = p.canvas;
if (!canvas) {
canvas = document.createElement('canvas'); p.canvas = canvas;
p.el.classList.remove('pending');
p.el.replaceChildren(canvas);
const num = document.createElement('span'); num.className = 'num'; num.textContent = String(p.num);
p.el.appendChild(num);
}
canvas.width = Math.floor(vp.width * dpr); canvas.height = Math.floor(vp.height * dpr);
p.el.style.width = Math.floor(vp.width) + 'px'; p.el.style.height = Math.floor(vp.height) + 'px';
const ctx = canvas.getContext('2d', { alpha: false });
const task = p.page.render({ canvasContext: ctx, viewport: vp, transform: dpr !== 1 ? [dpr, 0, 0, dpr, 0, 0] : null });
p.task = task;
const mine = scale;
try { await task.promise; } catch (e) { if (e && e.name === 'RenderingCancelledException') return; throw e; }
if (p.task !== task) return;
p.task = null; p.rendered = mine;
await paintText(p, vp);
}
async function paintText(p, vp) {
if (p.textLayer) p.textLayer.remove();
const layer = document.createElement('div');
layer.className = 'textLayer';
p.el.appendChild(layer); p.textLayer = layer;
try {
const content = await p.page.getTextContent();
p.text = content;
const tl = new pdfjs.TextLayer({ textContentSource: content, container: layer, viewport: vp });
await tl.render();
const end = document.createElement('div'); end.className = 'endOfContent'; layer.appendChild(end);
} catch (e) {
// no text layer (scanned page) — the canvas alone is fine
}
if (findState.term) highlight(p);
}
// Lazy rendering: pages within ~1.5 screens of the viewport render,
// everything else stays a placeholder (and drops its canvas if far off).
const io = new IntersectionObserver(entries => {
for (const en of entries) {
const p = pages[parseInt(en.target.dataset.page, 10) - 1];
if (en.isIntersecting) render(p).catch(() => {});
}
}, { root: viewer, rootMargin: '150% 0px' });
for (const p of pages) io.observe(p.el);
// Current page = the page whose top is nearest the viewport's upper third.
let currentPage = 1;
function updateCurrent() {
const top = viewer.scrollTop + viewer.clientHeight / 3;
let best = 1, bestDist = Infinity;
for (const p of pages) {
const d = Math.abs(p.el.offsetTop - top);
if (d < bestDist) { bestDist = d; best = p.num; }
}
if (best !== currentPage) {
currentPage = best;
$('pageno').value = String(best);
history.replaceState(null, '', '#page=' + best);
}
}
viewer.addEventListener('scroll', () => requestAnimationFrame(updateCurrent), { passive: true });
function goTo(n, smooth) {
n = Math.min(total, Math.max(1, n | 0));
const p = pages[n - 1];
viewer.scrollTo({ top: p.el.offsetTop - 12, behavior: smooth ? 'smooth' : 'auto' });
currentPage = n; $('pageno').value = String(n);
history.replaceState(null, '', '#page=' + n);
}
$('prev').addEventListener('click', () => goTo(currentPage - 1, true));
$('next').addEventListener('click', () => goTo(currentPage + 1, true));
$('pageno').addEventListener('change', () => goTo(parseInt($('pageno').value, 10) || 1, false));
// Zoom: re-size every page now, re-render the visible ones; the
// scroll position keeps the same page in view.
function rezoom(fit, factor) {
const keep = currentPage;
fitMode = fit; zoomFactor = factor;
sizeAll();
for (const p of pages) if (p.canvas) p.rendered = 0;
goTo(keep, false);
for (const p of pages) {
const r = p.el.getBoundingClientRect(), v = viewer.getBoundingClientRect();
if (r.bottom > v.top - v.height && r.top < v.bottom + v.height) render(p).catch(() => {});
}
}
$('zoomin').addEventListener('click', () => rezoom(false, Math.min(4, (fitMode ? 1 : zoomFactor) * 1.25)));
$('zoomout').addEventListener('click', () => rezoom(false, Math.max(0.3, (fitMode ? 1 : zoomFactor) / 1.25)));
$('fit').addEventListener('click', () => rezoom(true, 1));
// Resize (window, orientation, sidebars): debounce, then re-fit.
let rt = null;
new ResizeObserver(() => {
clearTimeout(rt);
rt = setTimeout(() => rezoom(fitMode, zoomFactor), 120);
}).observe(viewer);
// Keyboard.
document.addEventListener('keydown', e => {
if (e.target.tagName === 'INPUT' || e.target.tagName === 'SELECT') return;
if (e.key === 'ArrowLeft' || e.key === 'PageUp') { e.preventDefault(); goTo(currentPage - 1, true); }
else if (e.key === 'ArrowRight' || e.key === 'PageDown') { e.preventDefault(); goTo(currentPage + 1, true); }
else if (e.key === '+' || e.key === '=') { e.preventDefault(); $('zoomin').click(); }
else if (e.key === '-') { e.preventDefault(); $('zoomout').click(); }
else if (e.key === '0') { e.preventDefault(); $('fit').click(); }
else if ((e.ctrlKey || e.metaKey) && e.key.toLowerCase() === 'f') { e.preventDefault(); $('find').focus(); $('find').select(); }
});
// ── find: whole-document text search over the text layers, page by page ──
const findState = { term: '', hits: [], idx: -1 };
function norm(s) { return s.replace(/\s+/g, ' ').toLowerCase(); }
async function pageText(p) {
if (!p.page) p.page = await doc.getPage(p.num);
if (!p.text) p.text = await p.page.getTextContent();
return p.text;
}
async function runFind(term) {
findState.term = norm(term); findState.hits = []; findState.idx = -1;
for (const p of pages) if (p.textLayer) unhighlight(p);
if (!findState.term) { $('findcount').textContent = ''; return; }
for (const p of pages) {
const content = await pageText(p);
const joined = norm(content.items.map(i => i.str).join(' '));
let at = joined.indexOf(findState.term);
while (at !== -1) { findState.hits.push({ page: p.num }); at = joined.indexOf(findState.term, at + findState.term.length); }
if (p.textLayer) highlight(p);
}
$('findcount').textContent = findState.hits.length ? '0/' + findState.hits.length : 'no matches';
if (findState.hits.length) stepFind(1);
}
function stepFind(dir) {
if (!findState.hits.length) return;
findState.idx = (findState.idx + dir + findState.hits.length) % findState.hits.length;
$('findcount').textContent = (findState.idx + 1) + '/' + findState.hits.length;
goTo(findState.hits[findState.idx].page, true);
}
function unhighlight(p) {
for (const m of p.textLayer.querySelectorAll('mark')) {
const t = document.createTextNode(m.textContent); m.replaceWith(t);
}
for (const span of p.textLayer.querySelectorAll('span')) span.normalize();
}
function highlight(p) {
// mark the term inside each text span it occurs in (cross-span
// matches are counted by runFind but only same-span runs light up)
const term = findState.term;
if (!term || !p.textLayer) return;
for (const span of p.textLayer.querySelectorAll('span')) {
if (span.querySelector('mark')) continue;
const text = span.textContent, low = norm(text);
let at = low.indexOf(term);
if (at === -1) continue;
const frag = document.createDocumentFragment();
let last = 0;
while (at !== -1) {
frag.appendChild(document.createTextNode(text.slice(last, at)));
const m = document.createElement('mark'); m.textContent = text.slice(at, at + term.length); frag.appendChild(m);
last = at + term.length; at = low.indexOf(term, last);
}
frag.appendChild(document.createTextNode(text.slice(last)));
span.replaceChildren(frag);
}
}
let ft = null;
$('find').addEventListener('input', () => { clearTimeout(ft); ft = setTimeout(() => runFind($('find').value), 250); });
$('find').addEventListener('keydown', e => {
if (e.key === 'Enter') { e.preventDefault(); if (norm($('find').value) !== findState.term) runFind($('find').value); else stepFind(e.shiftKey ? -1 : 1); }
if (e.key === 'Escape') { $('find').value = ''; runFind(''); viewer.focus(); }
});
// ── start: size, jump to the requested page, seed the find box ──
sizeAll();
const hashPage = parseInt((location.hash.match(/page=(\d+)/) || [])[1] || '0', 10);
goTo(hashPage || wantPage, false);
updateCurrent();
if (wantQuery) {
// the snippet's first words are enough to land on the passage
const probe = wantQuery.split(/\s+/).slice(0, 6).join(' ');
$('find').value = probe;
runFind(probe);
}
</script>
</body>
</html>

View File

@@ -95,6 +95,11 @@
text-decoration: underline dotted;
}
.hit .date, .hit .meta { color: var(--muted-fg); font-size: 12px; font-family: var(--font-mono); }
.hit a.viewer {
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; white-space: nowrap;
}
.hit a.viewer:hover { background: color-mix(in srgb, var(--accent) 10%, transparent); }
.hit .title { margin-top: 3px; font-weight: 600; }
.hit .snippet { margin-top: 6px; }
.hit .score { margin-left: auto; color: var(--muted-fg); font-size: 11px; font-family: var(--font-mono); }
@@ -230,6 +235,20 @@
label.textContent = '[' + (r.label || r.id || '') + ']';
line.appendChild(label);
if (r.attachment && r.item_key && r.kind !== 'rule') {
// comment / article attachment on file → the in-page PDF viewer
var view = document.createElement('a');
view.className = 'viewer';
view.href = '/ui/pdf/' + encodeURIComponent(r.item_key) +
'?file=' + encodeURIComponent(r.attachment) +
(r.page ? '&page=' + encodeURIComponent(r.page) : '') +
'&q=' + encodeURIComponent((r.snippet || '').slice(0, 80));
view.target = '_blank'; view.rel = 'noopener';
view.textContent = (String(r.attachment).toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
(r.page ? ' p.' + r.page : '');
line.appendChild(view);
}
if (r.date) {
var date = document.createElement('span');
date.className = 'date';

21
src/llm/web/vendor/pdf.min.mjs vendored Normal file

File diff suppressed because one or more lines are too long

21
src/llm/web/vendor/pdf.worker.min.mjs vendored Normal file

File diff suppressed because one or more lines are too long

View File

@@ -439,3 +439,86 @@ class TestSimilarEndpoint:
kwargs = mock_similar.call_args.kwargs
assert kwargs["collection"] == "all"
assert kwargs["limit"] == 5
class TestPdfViewer:
"""/ui/pdf/<key>, /pdf/<key>, /pdf/<key>/file and the vendored PDF.js."""
def test_viewer_page_and_vendor_assets(self):
html = client.get("/ui/pdf/ABCD1234").text
assert "import('/ui/vendor/pdf.min.mjs')" in html
assert "workerSrc = '/ui/vendor/pdf.worker.min.mjs'" in html
assert "IntersectionObserver" in html and "ResizeObserver" in html
assert "TextLayer" in html and "runFind" in html
assert client.get("/ui/pdf/not a key").status_code == 404
js = client.get("/ui/vendor/pdf.min.mjs")
assert js.status_code == 200 and js.headers["content-type"].startswith(
"text/javascript"
)
assert client.get("/ui/vendor/pdf.worker.min.mjs").status_code == 200
assert client.get("/ui/vendor/../api.py").status_code in (404, 400)
assert client.get("/ui/vendor/evil.mjs").status_code == 404
def test_list_and_serve(self, tmp_path, monkeypatch):
import llm.api as api
from llm.pdfs import PdfFile
root = tmp_path / "root"
(root / "K").mkdir(parents=True)
pdf = root / "K" / "a.pdf"
pdf.write_bytes(b"%PDF-1.4\n" + b"x" * 5000)
docx = root / "K" / "b.docx"
docx.write_bytes(b"PK")
files = [
PdfFile("a.pdf", pdf, pdf.stat().st_size, "comment"),
PdfFile("b.docx", docx, 2, "comment"),
]
monkeypatch.setattr(
api, "_pdf_files", lambda key: files if key == "ABCD1234" else []
)
monkeypatch.setattr(
api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root)
)
class _Store:
_storage = tmp_path / "bibstorage"
def get(self, key):
class _I:
title = "Comment on CMS-2026-2377-1"
return _I()
def close(self):
pass
monkeypatch.setattr("conf.connect.bib", lambda: _Store())
r = client.get("/pdf/ABCD1234")
assert r.status_code == 200
body = r.json()
assert body["title"] == "Comment on CMS-2026-2377-1"
assert [(f["name"], f["renderable"]) for f in body["files"]] == [
("a.pdf", True),
("b.docx", False),
]
assert client.get("/pdf/NOPE9999").json()["files"] == []
r = client.get("/pdf/ABCD1234/file")
assert r.status_code == 200 and r.headers["content-type"] == "application/pdf"
assert r.headers["content-disposition"].startswith("inline")
assert r.headers["accept-ranges"] == "bytes" and r.content.startswith(b"%PDF")
r = client.get("/pdf/ABCD1234/file", headers={"Range": "bytes=0-3"})
assert r.status_code == 206 and r.content == b"%PDF"
r = client.get("/pdf/ABCD1234/file?name=b.docx")
assert r.status_code == 200 and r.headers["content-disposition"].startswith(
"attachment"
)
assert client.get("/pdf/ABCD1234/file?name=nope.pdf").status_code == 404
assert client.get("/pdf/ABCD1234/file?name=../a.pdf").status_code == 404
def test_chat_and_search_pages_link_the_viewer(self):
chat = client.get("/").text
assert "'/ui/pdf/' + encodeURIComponent(src.item_key)" in chat
assert "src.kind !== 'rule'" in chat
search = client.get("/ui/search").text
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search

View File

@@ -148,6 +148,8 @@ class TestAsSource:
0.123456,
)
assert set(s) == {
"attachment",
"page",
"id",
"label",
"kind",

128
tests/llm/test_pdfs.py Normal file
View File

@@ -0,0 +1,128 @@
"""llm.pdfs — attachments on file for the viewer: comment downloads, bib
attachments, Zotero storage PDFs; name-only resolution inside the roots."""
from __future__ import annotations
import sqlite3
from pathlib import Path
from bib.item import Source
from bib.store import Store
from llm.pdfs import PdfFile, pdf_files, resolve
def _zotero(
tmp_path: Path, parent_key: str, att_key: str, name: str
) -> tuple[Path, Path]:
db = tmp_path / "zotero.sqlite"
con = sqlite3.connect(db)
con.executescript(
"CREATE TABLE items (itemID INTEGER PRIMARY KEY, key TEXT);"
"CREATE TABLE itemAttachments (itemID INTEGER, parentItemID INTEGER, path TEXT);"
"CREATE TABLE deletedItems (itemID INTEGER);"
)
con.execute("INSERT INTO items VALUES (1, ?)", (parent_key,))
con.execute("INSERT INTO items VALUES (2, ?)", (att_key,))
con.execute("INSERT INTO items VALUES (3, 'GONE1234')")
con.execute("INSERT INTO itemAttachments VALUES (2, 1, ?)", (f"storage:{name}",))
con.execute("INSERT INTO itemAttachments VALUES (3, 1, 'storage:deleted.pdf')")
con.execute("INSERT INTO deletedItems VALUES (3)")
con.commit()
con.close()
storage = tmp_path / "zstorage"
(storage / att_key).mkdir(parents=True)
(storage / att_key / name).write_bytes(b"%PDF-1.4 z")
(storage / "GONE1234").mkdir()
(storage / "GONE1234" / "deleted.pdf").write_bytes(b"%PDF-1.4 d")
return db, storage
def test_all_three_sources_in_order_and_deduped(tmp_path):
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
key = store.upsert(Source(title="Paper", url="https://doi.org/10.1/x"))
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF-1.4 b")
store.attach_file(key, pdf)
(tmp_path / "notes.docx").write_bytes(b"PK docx")
store.attach_file(key, tmp_path / "notes.docx")
zdb, zstore = _zotero(
tmp_path, key, "ATT00001", "paper.pdf"
) # same name as the bib copy → deduped
files = pdf_files(
store,
key,
zotero_sqlite=zdb,
zotero_storage=zstore,
comments_root=tmp_path / "comments",
)
assert [(f.name, f.source, f.renderable) for f in files] == [
("paper.pdf", "bib", True),
("notes.docx", "bib", False),
]
assert files[1].media_type.startswith("application/vnd.openxmlformats")
assert all(isinstance(f, PdfFile) and f.size > 0 for f in files)
store.close()
def test_zotero_only_item_and_deleted_attachments_skipped(tmp_path):
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
key = store.upsert(Source(title="Zot only", url="https://example.org/z"))
zdb, zstore = _zotero(tmp_path, key, "ATT00002", "Cassidy - 2025 - RFI.pdf")
files = pdf_files(store, key, zotero_sqlite=zdb, zotero_storage=zstore)
assert [(f.name, f.source) for f in files] == [
("Cassidy - 2025 - RFI.pdf", "zotero")
]
assert pdf_files(store, "NOPE1234", zotero_sqlite=zdb, zotero_storage=zstore) == []
assert (
pdf_files(
store, key, zotero_sqlite=tmp_path / "missing.sqlite", zotero_storage=zstore
)
== []
)
store.close()
def test_comment_attachments_from_the_state_tree(tmp_path):
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
key = store.upsert(
Source(title="c", url="https://www.regulations.gov/comment/CMS-2026-2377-40314")
)
d = tmp_path / "comments" / "CMS-2026-2377" / "CMS-2026-2377-40314"
d.mkdir(parents=True)
(d / "attachment_1.pdf").write_bytes(b"%PDF-1.4 a1")
(d / "attachment_2.docx").write_bytes(b"PK")
(d / "combined.md").write_text("# not an attachment")
(d / "attachment_1.png").write_bytes(b"png")
files = pdf_files(store, key, comments_root=tmp_path / "comments")
assert [(f.name, f.source, f.renderable) for f in files] == [
("attachment_1.pdf", "comment", True),
("attachment_2.docx", "comment", False),
]
# a non-comment url, or a comment with no downloaded dir, yields nothing
other = store.upsert(Source(title="p", url="https://doi.org/10.1/y"))
assert pdf_files(store, other, comments_root=tmp_path / "comments") == []
gone = store.upsert(
Source(
title="c2", url="https://www.regulations.gov/comment/CMS-2026-2377-99999"
)
)
assert pdf_files(store, gone, comments_root=tmp_path / "comments") == []
store.close()
def test_resolve_by_name_inside_roots_only(tmp_path):
inside = tmp_path / "root" / "K" / "a.pdf"
inside.parent.mkdir(parents=True)
inside.write_bytes(b"%PDF")
outside = tmp_path / "elsewhere" / "b.pdf"
outside.parent.mkdir()
outside.write_bytes(b"%PDF")
files = [PdfFile("a.pdf", inside, 4, "bib"), PdfFile("b.pdf", outside, 4, "bib")]
roots = (tmp_path / "root",)
assert resolve(files, "", roots=roots).name == "a.pdf"
assert resolve(files, "a.pdf", roots=roots).name == "a.pdf"
assert (
resolve(files, "b.pdf", roots=roots) is None
) # listed, but outside every root
assert resolve(files, "../a.pdf", roots=roots) is None
assert resolve([], "", roots=roots) is None