feat(llm): in-page PDF viewer for comment and article attachments — /ui/pdf/<key>, /pdf/<key>[/file], self-hosted PDF.js, viewer links on chat sources and search hits
Some checks failed
CI / lint (push) Successful in 39s
CI / notebooks-smoke (push) Successful in 1m39s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m29s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 15s
Infra CI / docs (push) Successful in 20s
Infra CI / notebooks (push) Successful in 52s
Infra CI / api (push) Successful in 1m2s
Infra CI / llm (push) Successful in 47s
Deploy / report (push) Successful in 14s
Infra CI / mc (push) Failing after 37s
Some checks failed
CI / lint (push) Successful in 39s
CI / notebooks-smoke (push) Successful in 1m39s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m29s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 15s
Infra CI / docs (push) Successful in 20s
Infra CI / notebooks (push) Successful in 52s
Infra CI / api (push) Successful in 1m2s
Infra CI / llm (push) Successful in 47s
Deploy / report (push) Successful in 14s
Infra CI / mc (push) Failing after 37s
Rules keep their federalregister.gov links; this is for the files the library holds itself: a regulations.gov comment's downloaded attachments (.state/comments/<docket>/<id>/attachment_N.*), an item's bib attachments (data/storage/<att>/<name>) and Zotero-only storage PDFs (zotero.sqlite read immutable, one indexed lookup per request). llm.pdfs lists an item's attachments from those three places (comment first, de-duplicated by name) and resolves a file by *name* only, refusing anything outside the storage roots. GET /pdf/<key> lists them (name, size, source, renderable, media type); GET /pdf/<key>/file?name= serves a PDF inline with range requests and other formats as downloads. /ui/vendor/<name> serves the vendored pdf.js 4.10.38 (same-origin worker; no CDN dependency). /ui/pdf/<key>?file=&page=&q=: continuous scroll, pages rendered lazily (IntersectionObserver, ~1.5 screens ahead) at device pixel ratio, re-fit on resize/orientation (ResizeObserver, debounced), fit-to-width and zoom, page input with hash tracking, a text layer for selection, whole-document find with highlights (seeded from the cited snippet), keyboard shortcuts, a file selector when an item has several, and a download fallback when a file is not a PDF or cannot be rendered. as_source carries attachment + page from the chunk metadata; the chat sources and search hits show a 'PDF p.N' link into the viewer for comment/corpus sources. compose: the llm service mounts ./.state/comments read-only.
This commit is contained in:
@@ -691,6 +691,7 @@ services:
|
|||||||
# mount shares the WAL/SHM siblings like the api service does.
|
# mount shares the WAL/SHM siblings like the api service does.
|
||||||
- ./data:/app/data
|
- ./data:/app/data
|
||||||
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
|
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
|
||||||
|
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
|
||||||
environment:
|
environment:
|
||||||
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
|
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
|
||||||
- LLM_PG_HOST=postgres
|
- LLM_PG_HOST=postgres
|
||||||
|
|||||||
126
src/llm/api.py
126
src/llm/api.py
@@ -15,10 +15,11 @@ import re
|
|||||||
import time
|
import time
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
from importlib import resources
|
from importlib import resources
|
||||||
|
from pathlib import Path
|
||||||
from typing import AsyncIterator, Iterator
|
from typing import AsyncIterator, Iterator
|
||||||
|
|
||||||
from fastapi import FastAPI, Header, HTTPException
|
from fastapi import FastAPI, Header, HTTPException
|
||||||
from fastapi.responses import HTMLResponse, StreamingResponse
|
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
|
||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
@@ -107,6 +108,129 @@ def search_page() -> str:
|
|||||||
return _page("search.html")
|
return _page("search.html")
|
||||||
|
|
||||||
|
|
||||||
|
_VENDOR = {"pdf.min.mjs", "pdf.worker.min.mjs"}
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/ui/vendor/{name}")
|
||||||
|
def vendor(name: str) -> Response:
|
||||||
|
"""Self-hosted browser libraries (PDF.js) — a fixed allow-list, served
|
||||||
|
from the package so the viewer works without a CDN and its worker is
|
||||||
|
same-origin (a cross-origin worker script is refused by browsers)."""
|
||||||
|
if name not in _VENDOR:
|
||||||
|
raise HTTPException(404)
|
||||||
|
body = resources.files("llm").joinpath(f"web/vendor/{name}").read_bytes()
|
||||||
|
return Response(
|
||||||
|
body,
|
||||||
|
media_type="text/javascript",
|
||||||
|
headers={"Cache-Control": "public, max-age=86400"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/ui/pdf/{key}", response_class=HTMLResponse)
|
||||||
|
def pdf_page(key: str) -> str:
|
||||||
|
"""The PDF viewer; the page reads its item key, file, page and search
|
||||||
|
text from its own URL and fetches /pdf/<key> for the file list."""
|
||||||
|
if not _KEY_RE.match(key):
|
||||||
|
raise HTTPException(404)
|
||||||
|
return _page("pdf.html")
|
||||||
|
|
||||||
|
|
||||||
|
_KEY_RE = re.compile(r"^[A-Za-z0-9]{6,12}$")
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_paths() -> tuple[Path, Path, Path]:
|
||||||
|
"""(zotero.sqlite, zotero storage, comments root) — the comments root
|
||||||
|
is the indexer's (``llm.source._default_root``): ``.state/comments``."""
|
||||||
|
from conf import ROOT
|
||||||
|
from conf import path as cpath
|
||||||
|
|
||||||
|
return (
|
||||||
|
Path(cpath("db.zotero")),
|
||||||
|
Path(cpath("storage.zotero")),
|
||||||
|
Path(ROOT) / ".state" / "comments",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_files(key: str) -> list:
|
||||||
|
from conf.connect import bib
|
||||||
|
from llm.pdfs import pdf_files
|
||||||
|
|
||||||
|
store = bib()
|
||||||
|
try:
|
||||||
|
zdb, zstore, croot = _pdf_paths()
|
||||||
|
return pdf_files(
|
||||||
|
store, key, zotero_sqlite=zdb, zotero_storage=zstore, comments_root=croot
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
store.close()
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/pdf/{key}")
|
||||||
|
def pdf_list(key: str) -> dict:
|
||||||
|
"""The attachments on file for a comment or library item: name, size,
|
||||||
|
where it came from (comment download, bib attachment, Zotero storage)
|
||||||
|
and whether the viewer can render it (PDF) or only offer it for
|
||||||
|
download. Rules are not served here — they link to federalregister.gov."""
|
||||||
|
if not _KEY_RE.match(key):
|
||||||
|
raise HTTPException(404)
|
||||||
|
files = _pdf_files(key)
|
||||||
|
title = ""
|
||||||
|
try:
|
||||||
|
from conf.connect import bib
|
||||||
|
|
||||||
|
store = bib()
|
||||||
|
try:
|
||||||
|
title = store.get(key).title or ""
|
||||||
|
finally:
|
||||||
|
store.close()
|
||||||
|
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
|
||||||
|
title = ""
|
||||||
|
return {
|
||||||
|
"key": key,
|
||||||
|
"title": title,
|
||||||
|
"files": [
|
||||||
|
{
|
||||||
|
"name": f.name,
|
||||||
|
"size": f.size,
|
||||||
|
"source": f.source,
|
||||||
|
"renderable": f.renderable,
|
||||||
|
"media_type": f.media_type,
|
||||||
|
}
|
||||||
|
for f in files
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/pdf/{key}/file")
|
||||||
|
def pdf_file(key: str, name: str = "") -> FileResponse:
|
||||||
|
"""The attachment bytes — a PDF inline with range requests (PDF.js
|
||||||
|
streams pages from a large file instead of waiting for all of it),
|
||||||
|
anything else as a download."""
|
||||||
|
from llm.pdfs import resolve
|
||||||
|
|
||||||
|
if not _KEY_RE.match(key):
|
||||||
|
raise HTTPException(404)
|
||||||
|
files = _pdf_files(key)
|
||||||
|
from conf.connect import bib
|
||||||
|
|
||||||
|
store = bib()
|
||||||
|
try:
|
||||||
|
bib_root = Path(getattr(store, "_storage", "storage"))
|
||||||
|
finally:
|
||||||
|
store.close()
|
||||||
|
_zdb, zstore, croot = _pdf_paths()
|
||||||
|
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
|
||||||
|
if chosen is None:
|
||||||
|
raise HTTPException(404, "no such attachment on file for this item")
|
||||||
|
return FileResponse(
|
||||||
|
chosen.path,
|
||||||
|
media_type=chosen.media_type,
|
||||||
|
filename=chosen.name,
|
||||||
|
content_disposition_type="inline" if chosen.renderable else "attachment",
|
||||||
|
headers={"Accept-Ranges": "bytes", "Cache-Control": "private, max-age=3600"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@app.get("/whoami")
|
@app.get("/whoami")
|
||||||
def whoami(x_auth_request_user: str = Header(default="")) -> dict:
|
def whoami(x_auth_request_user: str = Header(default="")) -> dict:
|
||||||
"""The Gitea username Traefik forwarded (empty if unset)."""
|
"""The Gitea username Traefik forwarded (empty if unset)."""
|
||||||
|
|||||||
@@ -114,4 +114,8 @@ def as_source(md: dict[str, str], text: str, score: float) -> dict:
|
|||||||
"p_id": md.get("p_id", "") if kind == "rule" else "",
|
"p_id": md.get("p_id", "") if kind == "rule" else "",
|
||||||
"seq": md.get("seq", "") if kind != "rule" else "",
|
"seq": md.get("seq", "") if kind != "rule" else "",
|
||||||
"section": md.get("section", ""),
|
"section": md.get("section", ""),
|
||||||
|
# the PDF the chunk came from and the page it was located on
|
||||||
|
# (``llm.pages.enrich_pdf_pages``) — the UI links them to the viewer
|
||||||
|
"attachment": md.get("attachment", ""),
|
||||||
|
"page": md.get("page", ""),
|
||||||
}
|
}
|
||||||
|
|||||||
174
src/llm/pdfs.py
Normal file
174
src/llm/pdfs.py
Normal file
@@ -0,0 +1,174 @@
|
|||||||
|
"""Attachments on file for a library item, for the viewer (``/ui/pdf/<key>``).
|
||||||
|
|
||||||
|
Comment attachments and article attachments — never rules, which link
|
||||||
|
straight to federalregister.gov. The same places the indexer reads
|
||||||
|
(``llm.source``): a regulations.gov comment's downloaded attachments
|
||||||
|
(``.state/comments/<docket>/<comment id>/attachment_N.pdf`` …), the
|
||||||
|
item's bib attachments (``data/storage/<attachment key>/<filename>``)
|
||||||
|
and, for Zotero-only material, the storage PDFs of the item's Zotero
|
||||||
|
attachments (``data/zotero/data/storage/<attachment key>/<filename>``),
|
||||||
|
read from the live ``zotero.sqlite`` in immutable mode (a cheap indexed
|
||||||
|
lookup per item; the indexer's 2 GB snapshot is for whole-library walks,
|
||||||
|
not one request).
|
||||||
|
|
||||||
|
PDFs render in the page; other formats (docx, doc, txt) are listed for
|
||||||
|
download. A file is only ever served after it was found through this
|
||||||
|
lookup by its *name* — the API never joins a client-supplied path — and
|
||||||
|
only when it sits inside one of the storage roots.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import sqlite3
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
_MEDIA = {
|
||||||
|
".pdf": "application/pdf",
|
||||||
|
".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||||
|
".doc": "application/msword",
|
||||||
|
".txt": "text/plain",
|
||||||
|
}
|
||||||
|
_ATTACHMENT_EXT = tuple(_MEDIA)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class PdfFile:
|
||||||
|
name: str
|
||||||
|
path: Path
|
||||||
|
size: int
|
||||||
|
source: str # "comment" | "bib" | "zotero"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def media_type(self) -> str:
|
||||||
|
return _MEDIA.get(Path(self.name).suffix.lower(), "application/octet-stream")
|
||||||
|
|
||||||
|
@property
|
||||||
|
def renderable(self) -> bool:
|
||||||
|
return self.name.lower().endswith(".pdf")
|
||||||
|
|
||||||
|
|
||||||
|
def _comment_files(store: Any, key: str, root: Path) -> list[PdfFile]:
|
||||||
|
"""A regulations.gov comment's downloaded attachments, in file order."""
|
||||||
|
row = store._con().execute("SELECT url FROM items WHERE key = ?", (key,)).fetchone() # noqa: SLF001
|
||||||
|
url = (row[0] if row else "") or ""
|
||||||
|
prefix = "https://www.regulations.gov/comment/"
|
||||||
|
if not url.startswith(prefix):
|
||||||
|
return []
|
||||||
|
comment_id = url[len(prefix) :].strip("/")
|
||||||
|
if not comment_id or "/" in comment_id or comment_id.startswith("."):
|
||||||
|
return []
|
||||||
|
docket = comment_id.rsplit("-", 1)[0]
|
||||||
|
d = Path(root) / docket / comment_id
|
||||||
|
if not d.is_dir():
|
||||||
|
return []
|
||||||
|
return [
|
||||||
|
PdfFile(p.name, p, p.stat().st_size, "comment")
|
||||||
|
for p in sorted(d.iterdir())
|
||||||
|
if p.is_file() and p.suffix.lower() in _ATTACHMENT_EXT
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _bib_pdfs(store: Any, key: str) -> list[PdfFile]:
|
||||||
|
root = Path(getattr(store, "_storage", Path("storage")))
|
||||||
|
rows = (
|
||||||
|
store._con() # noqa: SLF001
|
||||||
|
.execute(
|
||||||
|
"SELECT a.key, a.filename, a.storage_path, a.content_type FROM attachments a "
|
||||||
|
"JOIN items i ON i.id = a.item_id WHERE i.key = ? ORDER BY a.id",
|
||||||
|
(key,),
|
||||||
|
)
|
||||||
|
.fetchall()
|
||||||
|
)
|
||||||
|
out: list[PdfFile] = []
|
||||||
|
for att_key, filename, storage_path, ctype in rows:
|
||||||
|
name = filename or (storage_path or "").split(":", 1)[-1]
|
||||||
|
if not name or Path(name).suffix.lower() not in _ATTACHMENT_EXT:
|
||||||
|
continue
|
||||||
|
p = root / att_key / Path(name).name
|
||||||
|
if p.is_file():
|
||||||
|
out.append(PdfFile(Path(name).name, p, p.stat().st_size, "bib"))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _zotero_pdfs(key: str, *, sqlite_path: Path, storage_dir: Path) -> list[PdfFile]:
|
||||||
|
if not Path(sqlite_path).exists():
|
||||||
|
return []
|
||||||
|
try:
|
||||||
|
con = sqlite3.connect(f"file:{sqlite_path}?mode=ro&immutable=1", uri=True)
|
||||||
|
try:
|
||||||
|
rows = con.execute(
|
||||||
|
"SELECT a.key, ia.path FROM itemAttachments ia "
|
||||||
|
"JOIN items a ON a.itemID = ia.itemID "
|
||||||
|
"JOIN items p ON p.itemID = ia.parentItemID "
|
||||||
|
"WHERE p.key = ? AND ia.path LIKE 'storage:%.pdf' "
|
||||||
|
"AND a.itemID NOT IN (SELECT itemID FROM deletedItems) ORDER BY a.itemID",
|
||||||
|
(key,),
|
||||||
|
).fetchall()
|
||||||
|
finally:
|
||||||
|
con.close()
|
||||||
|
except sqlite3.Error as exc:
|
||||||
|
log.warning("zotero lookup failed for %s: %s", key, exc)
|
||||||
|
return []
|
||||||
|
out: list[PdfFile] = []
|
||||||
|
for att_key, path in rows:
|
||||||
|
name = Path(path[len("storage:") :]).name
|
||||||
|
p = Path(storage_dir) / att_key / name
|
||||||
|
if p.is_file():
|
||||||
|
out.append(PdfFile(name, p, p.stat().st_size, "zotero"))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def pdf_files(
|
||||||
|
store: Any,
|
||||||
|
key: str,
|
||||||
|
*,
|
||||||
|
zotero_sqlite: Path | None = None,
|
||||||
|
zotero_storage: Path | None = None,
|
||||||
|
comments_root: Path | None = None,
|
||||||
|
) -> list[PdfFile]:
|
||||||
|
"""Every attachment on file for *key*: a comment's downloaded files
|
||||||
|
first, then bib attachments, then Zotero storage PDFs, de-duplicated
|
||||||
|
by file name (a Zotero copy of a bib attachment is the same document)."""
|
||||||
|
files: list[PdfFile] = []
|
||||||
|
if comments_root is not None:
|
||||||
|
files += _comment_files(store, key, comments_root)
|
||||||
|
files += _bib_pdfs(store, key)
|
||||||
|
if zotero_sqlite is not None and zotero_storage is not None:
|
||||||
|
files += _zotero_pdfs(
|
||||||
|
key, sqlite_path=zotero_sqlite, storage_dir=zotero_storage
|
||||||
|
)
|
||||||
|
seen: set[str] = set()
|
||||||
|
out: list[PdfFile] = []
|
||||||
|
for f in files:
|
||||||
|
if f.name in seen:
|
||||||
|
continue
|
||||||
|
seen.add(f.name)
|
||||||
|
out.append(f)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def resolve(
|
||||||
|
files: list[PdfFile], name: str, *, roots: tuple[Path, ...]
|
||||||
|
) -> PdfFile | None:
|
||||||
|
"""The listed file called *name* (an empty name means the first),
|
||||||
|
provided its real path lies under one of *roots*."""
|
||||||
|
if not files:
|
||||||
|
return None
|
||||||
|
chosen = files[0] if not name else next((f for f in files if f.name == name), None)
|
||||||
|
if chosen is None:
|
||||||
|
return None
|
||||||
|
real = chosen.path.resolve()
|
||||||
|
for root in roots:
|
||||||
|
try:
|
||||||
|
real.relative_to(Path(root).resolve())
|
||||||
|
return chosen
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
log.warning("refusing %s: outside storage roots", chosen.path)
|
||||||
|
return None
|
||||||
@@ -205,6 +205,12 @@
|
|||||||
background: color-mix(in srgb, var(--primary) 12%, transparent); margin-right: 6px;
|
background: color-mix(in srgb, var(--primary) 12%, transparent); margin-right: 6px;
|
||||||
}
|
}
|
||||||
.src .date { color: var(--muted-fg); margin-left: 6px; }
|
.src .date { color: var(--muted-fg); margin-left: 6px; }
|
||||||
|
.src a.viewer {
|
||||||
|
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
|
||||||
|
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; margin-left: 6px;
|
||||||
|
white-space: nowrap;
|
||||||
|
}
|
||||||
|
.src a.viewer:hover { background: color-mix(in srgb, var(--accent) 10%, transparent); }
|
||||||
.meta { font-size: 12px; color: var(--muted-fg); margin-top: 4px; font-family: var(--font-mono); }
|
.meta { font-size: 12px; color: var(--muted-fg); margin-top: 4px; font-family: var(--font-mono); }
|
||||||
.hint {
|
.hint {
|
||||||
margin: auto; max-width: 500px; text-align: center; color: var(--muted-fg);
|
margin: auto; max-width: 500px; text-align: center; color: var(--muted-fg);
|
||||||
@@ -652,6 +658,21 @@
|
|||||||
if (src.title && src.kind !== 'comment') {
|
if (src.title && src.kind !== 'comment') {
|
||||||
el.appendChild(document.createTextNode(' — ' + src.title));
|
el.appendChild(document.createTextNode(' — ' + src.title));
|
||||||
}
|
}
|
||||||
|
// A comment or article attachment on file opens in the PDF viewer
|
||||||
|
// on the page the passage was located on (rules link to the FR).
|
||||||
|
if (src.attachment && src.item_key && src.kind !== 'rule') {
|
||||||
|
const v = document.createElement('a');
|
||||||
|
v.className = 'viewer';
|
||||||
|
v.href = '/ui/pdf/' + encodeURIComponent(src.item_key) +
|
||||||
|
'?file=' + encodeURIComponent(src.attachment) +
|
||||||
|
(src.page ? '&page=' + encodeURIComponent(src.page) : '') +
|
||||||
|
'&q=' + encodeURIComponent((src.snippet || '').slice(0, 80));
|
||||||
|
v.target = '_blank'; v.rel = 'noopener';
|
||||||
|
v.textContent = (src.attachment.toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
|
||||||
|
(src.page ? ' p.' + src.page : '');
|
||||||
|
el.appendChild(document.createTextNode(' '));
|
||||||
|
el.appendChild(v);
|
||||||
|
}
|
||||||
el.appendChild(document.createTextNode(' ' + src.snippet));
|
el.appendChild(document.createTextNode(' ' + src.snippet));
|
||||||
d.appendChild(el);
|
d.appendChild(el);
|
||||||
}
|
}
|
||||||
|
|||||||
468
src/llm/web/pdf.html
Normal file
468
src/llm/web/pdf.html
Normal file
@@ -0,0 +1,468 @@
|
|||||||
|
<!DOCTYPE html>
|
||||||
|
<html lang="en">
|
||||||
|
<head>
|
||||||
|
<meta charset="utf-8">
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||||
|
<title>Library PDF</title>
|
||||||
|
<link rel="icon" type="image/png" sizes="32x32" href="//dashboard.fhirworx.io/fav32.png">
|
||||||
|
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||||
|
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||||
|
<link rel="stylesheet"
|
||||||
|
href="https://fonts.googleapis.com/css2?family=Playfair+Display:wght@600;700&family=Source+Serif+4:wght@400;600&family=JetBrains+Mono:wght@500&display=swap">
|
||||||
|
<style>
|
||||||
|
:root {
|
||||||
|
--background: #F7F5F0; --foreground: #1A1A18; --card: #FAFAF7;
|
||||||
|
--primary: #1C2B3A; --primary-fg: #F7F5F0; --secondary: #EDEBE6;
|
||||||
|
--muted-fg: #6B6B68; --border: #D4D0C8; --accent: #2E3D8F;
|
||||||
|
--destructive: #C0392B; --highlight: rgba(255, 214, 0, .45);
|
||||||
|
--font-display: "Playfair Display", Georgia, serif;
|
||||||
|
--font-body: "Source Serif 4", Georgia, serif;
|
||||||
|
--font-mono: "JetBrains Mono", "Fira Code", monospace;
|
||||||
|
}
|
||||||
|
* { box-sizing: border-box; }
|
||||||
|
html, body { height: 100%; }
|
||||||
|
body {
|
||||||
|
margin: 0; display: flex; flex-direction: column; overflow: hidden;
|
||||||
|
background: var(--background); color: var(--foreground);
|
||||||
|
font-family: var(--font-body); font-size: 15px;
|
||||||
|
}
|
||||||
|
header {
|
||||||
|
flex: 0 0 auto; background: var(--primary); color: var(--primary-fg);
|
||||||
|
border-bottom: 3px solid var(--foreground); padding: 10px 16px;
|
||||||
|
display: flex; align-items: center; gap: 12px; flex-wrap: wrap; min-width: 0;
|
||||||
|
}
|
||||||
|
header h1 {
|
||||||
|
font-family: var(--font-display); font-weight: 700; font-size: 16px; margin: 0;
|
||||||
|
white-space: nowrap;
|
||||||
|
}
|
||||||
|
header .title {
|
||||||
|
flex: 1 1 240px; min-width: 0; font-size: 13px; opacity: .85;
|
||||||
|
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
|
||||||
|
}
|
||||||
|
header a.nav {
|
||||||
|
font-family: var(--font-mono); font-size: 11px; text-transform: uppercase;
|
||||||
|
letter-spacing: .08em; color: var(--primary-fg); text-decoration: underline dotted;
|
||||||
|
}
|
||||||
|
.toolbar {
|
||||||
|
flex: 0 0 auto; background: var(--secondary); border-bottom: 1px solid var(--border);
|
||||||
|
padding: 6px 12px; display: flex; align-items: center; gap: 8px; flex-wrap: wrap;
|
||||||
|
font-family: var(--font-mono); font-size: 12px;
|
||||||
|
}
|
||||||
|
.toolbar .group { display: flex; align-items: center; gap: 4px; }
|
||||||
|
.toolbar button, .toolbar select, .toolbar input {
|
||||||
|
font-family: var(--font-mono); font-size: 12px; height: 28px;
|
||||||
|
border: 1px solid var(--border); border-radius: 3px; background: var(--card); color: var(--foreground);
|
||||||
|
}
|
||||||
|
.toolbar button { padding: 0 10px; cursor: pointer; min-width: 30px; }
|
||||||
|
.toolbar button:hover:not(:disabled) { background: #E2DFD8; }
|
||||||
|
.toolbar button:disabled { opacity: .4; cursor: default; }
|
||||||
|
.toolbar button.primary { background: var(--primary); color: var(--primary-fg); border-color: var(--primary); }
|
||||||
|
.toolbar input#pageno { width: 58px; text-align: right; padding: 0 6px; }
|
||||||
|
.toolbar input#find { width: 200px; padding: 0 8px; font-family: var(--font-body); font-size: 13px; }
|
||||||
|
.toolbar .count, .toolbar .zoom { color: var(--muted-fg); min-width: 44px; text-align: center; }
|
||||||
|
.toolbar select#file { max-width: 260px; text-overflow: ellipsis; }
|
||||||
|
.toolbar a.dl { color: var(--accent); text-decoration: underline dotted; font-size: 11px; }
|
||||||
|
.toolbar .spacer { flex: 1 1 auto; }
|
||||||
|
#progress { height: 3px; background: transparent; flex: 0 0 auto; }
|
||||||
|
#progress > div { height: 100%; width: 0; background: var(--accent); transition: width .2s ease; }
|
||||||
|
#viewer {
|
||||||
|
flex: 1 1 auto; overflow: auto; padding: 18px 12px 40px;
|
||||||
|
display: flex; flex-direction: column; align-items: center; gap: 14px;
|
||||||
|
-webkit-overflow-scrolling: touch; overscroll-behavior: contain;
|
||||||
|
}
|
||||||
|
.page {
|
||||||
|
position: relative; background: #fff; box-shadow: 0 1px 3px rgba(0,0,0,.25);
|
||||||
|
flex: 0 0 auto; max-width: 100%;
|
||||||
|
}
|
||||||
|
.page canvas { display: block; width: 100%; height: 100%; }
|
||||||
|
.page .num {
|
||||||
|
position: absolute; right: 6px; bottom: 4px; font-family: var(--font-mono); font-size: 10px;
|
||||||
|
color: var(--muted-fg); background: rgba(255,255,255,.7); padding: 1px 4px; border-radius: 2px;
|
||||||
|
pointer-events: none;
|
||||||
|
}
|
||||||
|
.page .textLayer {
|
||||||
|
position: absolute; inset: 0; overflow: hidden; line-height: 1; opacity: 1;
|
||||||
|
text-size-adjust: none; forced-color-adjust: none; transform-origin: 0 0;
|
||||||
|
}
|
||||||
|
.page .textLayer span, .page .textLayer br {
|
||||||
|
color: transparent; position: absolute; white-space: pre; cursor: text; transform-origin: 0 0;
|
||||||
|
}
|
||||||
|
.page .textLayer ::selection { background: rgba(46, 61, 143, .35); }
|
||||||
|
.page .textLayer .endOfContent { display: block; position: absolute; inset: 100% 0 0; z-index: -1; cursor: default; user-select: none; }
|
||||||
|
.page .textLayer mark { background: var(--highlight); color: transparent; border-radius: 2px; }
|
||||||
|
.page .textLayer mark.current { background: rgba(255, 140, 0, .55); }
|
||||||
|
.page.pending { min-height: 300px; }
|
||||||
|
.page .placeholder {
|
||||||
|
position: absolute; inset: 0; display: flex; align-items: center; justify-content: center;
|
||||||
|
color: var(--muted-fg); font-family: var(--font-mono); font-size: 12px;
|
||||||
|
}
|
||||||
|
.notice {
|
||||||
|
margin: 40px auto; max-width: 560px; padding: 16px 20px; border: 1px solid var(--border);
|
||||||
|
background: var(--card); border-radius: 3px; color: var(--muted-fg); font-style: italic;
|
||||||
|
}
|
||||||
|
.notice.err { color: var(--destructive); border-color: var(--destructive); font-style: normal; }
|
||||||
|
.notice a { color: var(--accent); }
|
||||||
|
@media (max-width: 640px) {
|
||||||
|
header { padding: 8px 10px; gap: 8px; }
|
||||||
|
header .title { flex-basis: 100%; }
|
||||||
|
.toolbar { padding: 6px 8px; gap: 6px; }
|
||||||
|
.toolbar input#find { width: 130px; }
|
||||||
|
.toolbar select#file { max-width: 160px; }
|
||||||
|
#viewer { padding: 10px 6px 30px; gap: 10px; }
|
||||||
|
}
|
||||||
|
</style>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<header>
|
||||||
|
<h1>Library PDF</h1>
|
||||||
|
<span class="title" id="title"></span>
|
||||||
|
<a class="nav" href="/">Chat</a>
|
||||||
|
<a class="nav" href="/ui/search">Search</a>
|
||||||
|
</header>
|
||||||
|
<div class="toolbar">
|
||||||
|
<div class="group">
|
||||||
|
<select id="file" title="Which PDF on file for this item" hidden></select>
|
||||||
|
</div>
|
||||||
|
<div class="group">
|
||||||
|
<button id="prev" title="Previous page (←)">‹</button>
|
||||||
|
<input id="pageno" type="number" min="1" value="1" title="Page">
|
||||||
|
<span class="count" id="count">/ –</span>
|
||||||
|
<button id="next" title="Next page (→)">›</button>
|
||||||
|
</div>
|
||||||
|
<div class="group">
|
||||||
|
<button id="zoomout" title="Zoom out (−)">−</button>
|
||||||
|
<span class="zoom" id="zoomlabel">fit</span>
|
||||||
|
<button id="zoomin" title="Zoom in (+)">+</button>
|
||||||
|
<button id="fit" title="Fit to width (0)">fit</button>
|
||||||
|
</div>
|
||||||
|
<div class="group">
|
||||||
|
<input id="find" type="search" placeholder="find in document" title="Find (Enter: next, Shift+Enter: previous)">
|
||||||
|
<span class="count" id="findcount"></span>
|
||||||
|
</div>
|
||||||
|
<span class="spacer"></span>
|
||||||
|
<a class="dl" id="download" href="#" download>download</a>
|
||||||
|
</div>
|
||||||
|
<div id="progress"><div id="bar"></div></div>
|
||||||
|
<div id="viewer" tabindex="0"></div>
|
||||||
|
|
||||||
|
<script type="module">
|
||||||
|
// Everything below reads only this page's own URL and the JSON from
|
||||||
|
// /pdf/<key>; nothing from those ever reaches innerHTML.
|
||||||
|
const $ = id => document.getElementById(id);
|
||||||
|
const viewer = $('viewer'), bar = $('bar');
|
||||||
|
const params = new URLSearchParams(location.search);
|
||||||
|
const key = location.pathname.split('/').filter(Boolean).pop();
|
||||||
|
const wantFile = params.get('file') || '';
|
||||||
|
const wantPage = Math.max(1, parseInt(params.get('page') || '1', 10) || 1);
|
||||||
|
const wantQuery = (params.get('q') || '').trim();
|
||||||
|
|
||||||
|
function notice(text, err, downloadUrl) {
|
||||||
|
const d = document.createElement('div');
|
||||||
|
d.className = 'notice' + (err ? ' err' : '');
|
||||||
|
d.appendChild(document.createTextNode(text));
|
||||||
|
if (downloadUrl) {
|
||||||
|
d.appendChild(document.createTextNode(' '));
|
||||||
|
const a = document.createElement('a'); a.href = downloadUrl; a.textContent = 'Download the file instead.';
|
||||||
|
d.appendChild(a);
|
||||||
|
}
|
||||||
|
viewer.replaceChildren(d);
|
||||||
|
}
|
||||||
|
|
||||||
|
let pdfjs = null;
|
||||||
|
try {
|
||||||
|
pdfjs = await import('/ui/vendor/pdf.min.mjs');
|
||||||
|
pdfjs.GlobalWorkerOptions.workerSrc = '/ui/vendor/pdf.worker.min.mjs';
|
||||||
|
} catch (e) {
|
||||||
|
pdfjs = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── file list ──
|
||||||
|
let files = [];
|
||||||
|
try {
|
||||||
|
const r = await fetch('/pdf/' + encodeURIComponent(key));
|
||||||
|
if (!r.ok) throw new Error('lookup failed (' + r.status + ')');
|
||||||
|
const d = await r.json();
|
||||||
|
files = d.files || [];
|
||||||
|
$('title').textContent = d.title || key;
|
||||||
|
document.title = (d.title ? d.title + ' — ' : '') + 'Library PDF';
|
||||||
|
} catch (e) {
|
||||||
|
notice('Could not look up the item: ' + e.message, true);
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
if (!files.length) {
|
||||||
|
notice('No PDF is on file for this item.', false);
|
||||||
|
throw new Error('no files');
|
||||||
|
}
|
||||||
|
const sel = $('file');
|
||||||
|
for (const f of files) {
|
||||||
|
const o = document.createElement('option');
|
||||||
|
o.value = f.name; o.textContent = f.name + ' (' + Math.round(f.size / 1024) + ' KB' + (f.renderable === false ? ', download only' : '') + ')';
|
||||||
|
sel.appendChild(o);
|
||||||
|
}
|
||||||
|
if (files.length > 1) sel.hidden = false;
|
||||||
|
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
|
||||||
|
sel.value = current.name;
|
||||||
|
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
|
||||||
|
$('download').href = fileUrl;
|
||||||
|
sel.addEventListener('change', () => {
|
||||||
|
const p = new URLSearchParams(location.search);
|
||||||
|
p.set('file', sel.value); p.delete('page');
|
||||||
|
location.search = p.toString();
|
||||||
|
});
|
||||||
|
|
||||||
|
if (current.renderable === false) {
|
||||||
|
notice('"' + current.name + '" is not a PDF (' + (current.media_type || 'file') + '), so it cannot be rendered here.', false, fileUrl);
|
||||||
|
throw new Error('not renderable');
|
||||||
|
}
|
||||||
|
if (!pdfjs) {
|
||||||
|
notice('The in-page renderer could not be loaded.', true, fileUrl);
|
||||||
|
throw new Error('pdfjs missing');
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── document ──
|
||||||
|
let doc;
|
||||||
|
try {
|
||||||
|
const task = pdfjs.getDocument({ url: fileUrl, rangeChunkSize: 1 << 18, disableAutoFetch: false });
|
||||||
|
task.onProgress = ({ loaded, total }) => { if (total) bar.style.width = Math.min(100, 100 * loaded / total) + '%'; };
|
||||||
|
doc = await task.promise;
|
||||||
|
bar.style.width = '0';
|
||||||
|
} catch (e) {
|
||||||
|
notice('This file could not be rendered as a PDF: ' + (e && e.message || e), true, fileUrl);
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
const total = doc.numPages;
|
||||||
|
$('count').textContent = '/ ' + total;
|
||||||
|
$('pageno').max = String(total);
|
||||||
|
|
||||||
|
// ── layout: one placeholder per page, sized from page 1's ratio,
|
||||||
|
// rendered lazily as they scroll into view, re-rendered on resize ──
|
||||||
|
const pages = []; // {el, canvas, textLayer, num, viewportW, rendered, task, text}
|
||||||
|
let scale = 1; // CSS px per PDF unit at the current zoom
|
||||||
|
let fitMode = true; // fit to width until the user zooms
|
||||||
|
let zoomFactor = 1;
|
||||||
|
const first = await doc.getPage(1);
|
||||||
|
const baseW = first.getViewport({ scale: 1 }).width;
|
||||||
|
const baseH = first.getViewport({ scale: 1 }).height;
|
||||||
|
|
||||||
|
function availableWidth() {
|
||||||
|
const cs = getComputedStyle(viewer);
|
||||||
|
return viewer.clientWidth - parseFloat(cs.paddingLeft) - parseFloat(cs.paddingRight);
|
||||||
|
}
|
||||||
|
function computeScale() {
|
||||||
|
const fit = Math.max(0.2, availableWidth() / baseW);
|
||||||
|
scale = fitMode ? fit : fit * zoomFactor;
|
||||||
|
$('zoomlabel').textContent = fitMode ? 'fit' : Math.round(zoomFactor * 100) + '%';
|
||||||
|
}
|
||||||
|
|
||||||
|
for (let n = 1; n <= total; n++) {
|
||||||
|
const el = document.createElement('div');
|
||||||
|
el.className = 'page pending'; el.dataset.page = String(n);
|
||||||
|
const ph = document.createElement('div'); ph.className = 'placeholder'; ph.textContent = 'page ' + n;
|
||||||
|
const num = document.createElement('span'); num.className = 'num'; num.textContent = String(n);
|
||||||
|
el.appendChild(ph); el.appendChild(num);
|
||||||
|
viewer.appendChild(el);
|
||||||
|
pages.push({ el, num: n, canvas: null, textLayer: null, rendered: 0, task: null, text: null, page: null });
|
||||||
|
}
|
||||||
|
|
||||||
|
function sizeAll() {
|
||||||
|
computeScale();
|
||||||
|
for (const p of pages) {
|
||||||
|
const pg = p.page;
|
||||||
|
const w = Math.floor((pg ? pg.getViewport({ scale: 1 }).width : baseW) * scale);
|
||||||
|
const h = Math.floor((pg ? pg.getViewport({ scale: 1 }).height : baseH) * scale);
|
||||||
|
p.el.style.width = w + 'px'; p.el.style.height = h + 'px';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function render(p) {
|
||||||
|
if (p.rendered === scale) return;
|
||||||
|
if (p.task) { try { p.task.cancel(); } catch (e) {} p.task = null; }
|
||||||
|
if (!p.page) p.page = await doc.getPage(p.num);
|
||||||
|
const vp = p.page.getViewport({ scale });
|
||||||
|
const dpr = Math.min(window.devicePixelRatio || 1, 3);
|
||||||
|
let canvas = p.canvas;
|
||||||
|
if (!canvas) {
|
||||||
|
canvas = document.createElement('canvas'); p.canvas = canvas;
|
||||||
|
p.el.classList.remove('pending');
|
||||||
|
p.el.replaceChildren(canvas);
|
||||||
|
const num = document.createElement('span'); num.className = 'num'; num.textContent = String(p.num);
|
||||||
|
p.el.appendChild(num);
|
||||||
|
}
|
||||||
|
canvas.width = Math.floor(vp.width * dpr); canvas.height = Math.floor(vp.height * dpr);
|
||||||
|
p.el.style.width = Math.floor(vp.width) + 'px'; p.el.style.height = Math.floor(vp.height) + 'px';
|
||||||
|
const ctx = canvas.getContext('2d', { alpha: false });
|
||||||
|
const task = p.page.render({ canvasContext: ctx, viewport: vp, transform: dpr !== 1 ? [dpr, 0, 0, dpr, 0, 0] : null });
|
||||||
|
p.task = task;
|
||||||
|
const mine = scale;
|
||||||
|
try { await task.promise; } catch (e) { if (e && e.name === 'RenderingCancelledException') return; throw e; }
|
||||||
|
if (p.task !== task) return;
|
||||||
|
p.task = null; p.rendered = mine;
|
||||||
|
await paintText(p, vp);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function paintText(p, vp) {
|
||||||
|
if (p.textLayer) p.textLayer.remove();
|
||||||
|
const layer = document.createElement('div');
|
||||||
|
layer.className = 'textLayer';
|
||||||
|
p.el.appendChild(layer); p.textLayer = layer;
|
||||||
|
try {
|
||||||
|
const content = await p.page.getTextContent();
|
||||||
|
p.text = content;
|
||||||
|
const tl = new pdfjs.TextLayer({ textContentSource: content, container: layer, viewport: vp });
|
||||||
|
await tl.render();
|
||||||
|
const end = document.createElement('div'); end.className = 'endOfContent'; layer.appendChild(end);
|
||||||
|
} catch (e) {
|
||||||
|
// no text layer (scanned page) — the canvas alone is fine
|
||||||
|
}
|
||||||
|
if (findState.term) highlight(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lazy rendering: pages within ~1.5 screens of the viewport render,
|
||||||
|
// everything else stays a placeholder (and drops its canvas if far off).
|
||||||
|
const io = new IntersectionObserver(entries => {
|
||||||
|
for (const en of entries) {
|
||||||
|
const p = pages[parseInt(en.target.dataset.page, 10) - 1];
|
||||||
|
if (en.isIntersecting) render(p).catch(() => {});
|
||||||
|
}
|
||||||
|
}, { root: viewer, rootMargin: '150% 0px' });
|
||||||
|
for (const p of pages) io.observe(p.el);
|
||||||
|
|
||||||
|
// Current page = the page whose top is nearest the viewport's upper third.
|
||||||
|
let currentPage = 1;
|
||||||
|
function updateCurrent() {
|
||||||
|
const top = viewer.scrollTop + viewer.clientHeight / 3;
|
||||||
|
let best = 1, bestDist = Infinity;
|
||||||
|
for (const p of pages) {
|
||||||
|
const d = Math.abs(p.el.offsetTop - top);
|
||||||
|
if (d < bestDist) { bestDist = d; best = p.num; }
|
||||||
|
}
|
||||||
|
if (best !== currentPage) {
|
||||||
|
currentPage = best;
|
||||||
|
$('pageno').value = String(best);
|
||||||
|
history.replaceState(null, '', '#page=' + best);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
viewer.addEventListener('scroll', () => requestAnimationFrame(updateCurrent), { passive: true });
|
||||||
|
|
||||||
|
function goTo(n, smooth) {
|
||||||
|
n = Math.min(total, Math.max(1, n | 0));
|
||||||
|
const p = pages[n - 1];
|
||||||
|
viewer.scrollTo({ top: p.el.offsetTop - 12, behavior: smooth ? 'smooth' : 'auto' });
|
||||||
|
currentPage = n; $('pageno').value = String(n);
|
||||||
|
history.replaceState(null, '', '#page=' + n);
|
||||||
|
}
|
||||||
|
$('prev').addEventListener('click', () => goTo(currentPage - 1, true));
|
||||||
|
$('next').addEventListener('click', () => goTo(currentPage + 1, true));
|
||||||
|
$('pageno').addEventListener('change', () => goTo(parseInt($('pageno').value, 10) || 1, false));
|
||||||
|
|
||||||
|
// Zoom: re-size every page now, re-render the visible ones; the
|
||||||
|
// scroll position keeps the same page in view.
|
||||||
|
function rezoom(fit, factor) {
|
||||||
|
const keep = currentPage;
|
||||||
|
fitMode = fit; zoomFactor = factor;
|
||||||
|
sizeAll();
|
||||||
|
for (const p of pages) if (p.canvas) p.rendered = 0;
|
||||||
|
goTo(keep, false);
|
||||||
|
for (const p of pages) {
|
||||||
|
const r = p.el.getBoundingClientRect(), v = viewer.getBoundingClientRect();
|
||||||
|
if (r.bottom > v.top - v.height && r.top < v.bottom + v.height) render(p).catch(() => {});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
$('zoomin').addEventListener('click', () => rezoom(false, Math.min(4, (fitMode ? 1 : zoomFactor) * 1.25)));
|
||||||
|
$('zoomout').addEventListener('click', () => rezoom(false, Math.max(0.3, (fitMode ? 1 : zoomFactor) / 1.25)));
|
||||||
|
$('fit').addEventListener('click', () => rezoom(true, 1));
|
||||||
|
|
||||||
|
// Resize (window, orientation, sidebars): debounce, then re-fit.
|
||||||
|
let rt = null;
|
||||||
|
new ResizeObserver(() => {
|
||||||
|
clearTimeout(rt);
|
||||||
|
rt = setTimeout(() => rezoom(fitMode, zoomFactor), 120);
|
||||||
|
}).observe(viewer);
|
||||||
|
|
||||||
|
// Keyboard.
|
||||||
|
document.addEventListener('keydown', e => {
|
||||||
|
if (e.target.tagName === 'INPUT' || e.target.tagName === 'SELECT') return;
|
||||||
|
if (e.key === 'ArrowLeft' || e.key === 'PageUp') { e.preventDefault(); goTo(currentPage - 1, true); }
|
||||||
|
else if (e.key === 'ArrowRight' || e.key === 'PageDown') { e.preventDefault(); goTo(currentPage + 1, true); }
|
||||||
|
else if (e.key === '+' || e.key === '=') { e.preventDefault(); $('zoomin').click(); }
|
||||||
|
else if (e.key === '-') { e.preventDefault(); $('zoomout').click(); }
|
||||||
|
else if (e.key === '0') { e.preventDefault(); $('fit').click(); }
|
||||||
|
else if ((e.ctrlKey || e.metaKey) && e.key.toLowerCase() === 'f') { e.preventDefault(); $('find').focus(); $('find').select(); }
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── find: whole-document text search over the text layers, page by page ──
|
||||||
|
const findState = { term: '', hits: [], idx: -1 };
|
||||||
|
function norm(s) { return s.replace(/\s+/g, ' ').toLowerCase(); }
|
||||||
|
async function pageText(p) {
|
||||||
|
if (!p.page) p.page = await doc.getPage(p.num);
|
||||||
|
if (!p.text) p.text = await p.page.getTextContent();
|
||||||
|
return p.text;
|
||||||
|
}
|
||||||
|
async function runFind(term) {
|
||||||
|
findState.term = norm(term); findState.hits = []; findState.idx = -1;
|
||||||
|
for (const p of pages) if (p.textLayer) unhighlight(p);
|
||||||
|
if (!findState.term) { $('findcount').textContent = ''; return; }
|
||||||
|
for (const p of pages) {
|
||||||
|
const content = await pageText(p);
|
||||||
|
const joined = norm(content.items.map(i => i.str).join(' '));
|
||||||
|
let at = joined.indexOf(findState.term);
|
||||||
|
while (at !== -1) { findState.hits.push({ page: p.num }); at = joined.indexOf(findState.term, at + findState.term.length); }
|
||||||
|
if (p.textLayer) highlight(p);
|
||||||
|
}
|
||||||
|
$('findcount').textContent = findState.hits.length ? '0/' + findState.hits.length : 'no matches';
|
||||||
|
if (findState.hits.length) stepFind(1);
|
||||||
|
}
|
||||||
|
function stepFind(dir) {
|
||||||
|
if (!findState.hits.length) return;
|
||||||
|
findState.idx = (findState.idx + dir + findState.hits.length) % findState.hits.length;
|
||||||
|
$('findcount').textContent = (findState.idx + 1) + '/' + findState.hits.length;
|
||||||
|
goTo(findState.hits[findState.idx].page, true);
|
||||||
|
}
|
||||||
|
function unhighlight(p) {
|
||||||
|
for (const m of p.textLayer.querySelectorAll('mark')) {
|
||||||
|
const t = document.createTextNode(m.textContent); m.replaceWith(t);
|
||||||
|
}
|
||||||
|
for (const span of p.textLayer.querySelectorAll('span')) span.normalize();
|
||||||
|
}
|
||||||
|
function highlight(p) {
|
||||||
|
// mark the term inside each text span it occurs in (cross-span
|
||||||
|
// matches are counted by runFind but only same-span runs light up)
|
||||||
|
const term = findState.term;
|
||||||
|
if (!term || !p.textLayer) return;
|
||||||
|
for (const span of p.textLayer.querySelectorAll('span')) {
|
||||||
|
if (span.querySelector('mark')) continue;
|
||||||
|
const text = span.textContent, low = norm(text);
|
||||||
|
let at = low.indexOf(term);
|
||||||
|
if (at === -1) continue;
|
||||||
|
const frag = document.createDocumentFragment();
|
||||||
|
let last = 0;
|
||||||
|
while (at !== -1) {
|
||||||
|
frag.appendChild(document.createTextNode(text.slice(last, at)));
|
||||||
|
const m = document.createElement('mark'); m.textContent = text.slice(at, at + term.length); frag.appendChild(m);
|
||||||
|
last = at + term.length; at = low.indexOf(term, last);
|
||||||
|
}
|
||||||
|
frag.appendChild(document.createTextNode(text.slice(last)));
|
||||||
|
span.replaceChildren(frag);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let ft = null;
|
||||||
|
$('find').addEventListener('input', () => { clearTimeout(ft); ft = setTimeout(() => runFind($('find').value), 250); });
|
||||||
|
$('find').addEventListener('keydown', e => {
|
||||||
|
if (e.key === 'Enter') { e.preventDefault(); if (norm($('find').value) !== findState.term) runFind($('find').value); else stepFind(e.shiftKey ? -1 : 1); }
|
||||||
|
if (e.key === 'Escape') { $('find').value = ''; runFind(''); viewer.focus(); }
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── start: size, jump to the requested page, seed the find box ──
|
||||||
|
sizeAll();
|
||||||
|
const hashPage = parseInt((location.hash.match(/page=(\d+)/) || [])[1] || '0', 10);
|
||||||
|
goTo(hashPage || wantPage, false);
|
||||||
|
updateCurrent();
|
||||||
|
if (wantQuery) {
|
||||||
|
// the snippet's first words are enough to land on the passage
|
||||||
|
const probe = wantQuery.split(/\s+/).slice(0, 6).join(' ');
|
||||||
|
$('find').value = probe;
|
||||||
|
runFind(probe);
|
||||||
|
}
|
||||||
|
</script>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
@@ -95,6 +95,11 @@
|
|||||||
text-decoration: underline dotted;
|
text-decoration: underline dotted;
|
||||||
}
|
}
|
||||||
.hit .date, .hit .meta { color: var(--muted-fg); font-size: 12px; font-family: var(--font-mono); }
|
.hit .date, .hit .meta { color: var(--muted-fg); font-size: 12px; font-family: var(--font-mono); }
|
||||||
|
.hit a.viewer {
|
||||||
|
font-family: var(--font-mono); font-size: 11px; padding: 1px 6px; border-radius: 3px;
|
||||||
|
border: 1px solid var(--accent); color: var(--accent); text-decoration: none; white-space: nowrap;
|
||||||
|
}
|
||||||
|
.hit a.viewer:hover { background: color-mix(in srgb, var(--accent) 10%, transparent); }
|
||||||
.hit .title { margin-top: 3px; font-weight: 600; }
|
.hit .title { margin-top: 3px; font-weight: 600; }
|
||||||
.hit .snippet { margin-top: 6px; }
|
.hit .snippet { margin-top: 6px; }
|
||||||
.hit .score { margin-left: auto; color: var(--muted-fg); font-size: 11px; font-family: var(--font-mono); }
|
.hit .score { margin-left: auto; color: var(--muted-fg); font-size: 11px; font-family: var(--font-mono); }
|
||||||
@@ -230,6 +235,20 @@
|
|||||||
label.textContent = '[' + (r.label || r.id || '') + ']';
|
label.textContent = '[' + (r.label || r.id || '') + ']';
|
||||||
line.appendChild(label);
|
line.appendChild(label);
|
||||||
|
|
||||||
|
if (r.attachment && r.item_key && r.kind !== 'rule') {
|
||||||
|
// comment / article attachment on file → the in-page PDF viewer
|
||||||
|
var view = document.createElement('a');
|
||||||
|
view.className = 'viewer';
|
||||||
|
view.href = '/ui/pdf/' + encodeURIComponent(r.item_key) +
|
||||||
|
'?file=' + encodeURIComponent(r.attachment) +
|
||||||
|
(r.page ? '&page=' + encodeURIComponent(r.page) : '') +
|
||||||
|
'&q=' + encodeURIComponent((r.snippet || '').slice(0, 80));
|
||||||
|
view.target = '_blank'; view.rel = 'noopener';
|
||||||
|
view.textContent = (String(r.attachment).toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
|
||||||
|
(r.page ? ' p.' + r.page : '');
|
||||||
|
line.appendChild(view);
|
||||||
|
}
|
||||||
|
|
||||||
if (r.date) {
|
if (r.date) {
|
||||||
var date = document.createElement('span');
|
var date = document.createElement('span');
|
||||||
date.className = 'date';
|
date.className = 'date';
|
||||||
|
|||||||
21
src/llm/web/vendor/pdf.min.mjs
vendored
Normal file
21
src/llm/web/vendor/pdf.min.mjs
vendored
Normal file
File diff suppressed because one or more lines are too long
21
src/llm/web/vendor/pdf.worker.min.mjs
vendored
Normal file
21
src/llm/web/vendor/pdf.worker.min.mjs
vendored
Normal file
File diff suppressed because one or more lines are too long
@@ -439,3 +439,86 @@ class TestSimilarEndpoint:
|
|||||||
kwargs = mock_similar.call_args.kwargs
|
kwargs = mock_similar.call_args.kwargs
|
||||||
assert kwargs["collection"] == "all"
|
assert kwargs["collection"] == "all"
|
||||||
assert kwargs["limit"] == 5
|
assert kwargs["limit"] == 5
|
||||||
|
|
||||||
|
|
||||||
|
class TestPdfViewer:
|
||||||
|
"""/ui/pdf/<key>, /pdf/<key>, /pdf/<key>/file and the vendored PDF.js."""
|
||||||
|
|
||||||
|
def test_viewer_page_and_vendor_assets(self):
|
||||||
|
html = client.get("/ui/pdf/ABCD1234").text
|
||||||
|
assert "import('/ui/vendor/pdf.min.mjs')" in html
|
||||||
|
assert "workerSrc = '/ui/vendor/pdf.worker.min.mjs'" in html
|
||||||
|
assert "IntersectionObserver" in html and "ResizeObserver" in html
|
||||||
|
assert "TextLayer" in html and "runFind" in html
|
||||||
|
assert client.get("/ui/pdf/not a key").status_code == 404
|
||||||
|
js = client.get("/ui/vendor/pdf.min.mjs")
|
||||||
|
assert js.status_code == 200 and js.headers["content-type"].startswith(
|
||||||
|
"text/javascript"
|
||||||
|
)
|
||||||
|
assert client.get("/ui/vendor/pdf.worker.min.mjs").status_code == 200
|
||||||
|
assert client.get("/ui/vendor/../api.py").status_code in (404, 400)
|
||||||
|
assert client.get("/ui/vendor/evil.mjs").status_code == 404
|
||||||
|
|
||||||
|
def test_list_and_serve(self, tmp_path, monkeypatch):
|
||||||
|
import llm.api as api
|
||||||
|
from llm.pdfs import PdfFile
|
||||||
|
|
||||||
|
root = tmp_path / "root"
|
||||||
|
(root / "K").mkdir(parents=True)
|
||||||
|
pdf = root / "K" / "a.pdf"
|
||||||
|
pdf.write_bytes(b"%PDF-1.4\n" + b"x" * 5000)
|
||||||
|
docx = root / "K" / "b.docx"
|
||||||
|
docx.write_bytes(b"PK")
|
||||||
|
files = [
|
||||||
|
PdfFile("a.pdf", pdf, pdf.stat().st_size, "comment"),
|
||||||
|
PdfFile("b.docx", docx, 2, "comment"),
|
||||||
|
]
|
||||||
|
monkeypatch.setattr(
|
||||||
|
api, "_pdf_files", lambda key: files if key == "ABCD1234" else []
|
||||||
|
)
|
||||||
|
monkeypatch.setattr(
|
||||||
|
api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root)
|
||||||
|
)
|
||||||
|
|
||||||
|
class _Store:
|
||||||
|
_storage = tmp_path / "bibstorage"
|
||||||
|
|
||||||
|
def get(self, key):
|
||||||
|
class _I:
|
||||||
|
title = "Comment on CMS-2026-2377-1"
|
||||||
|
|
||||||
|
return _I()
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
monkeypatch.setattr("conf.connect.bib", lambda: _Store())
|
||||||
|
r = client.get("/pdf/ABCD1234")
|
||||||
|
assert r.status_code == 200
|
||||||
|
body = r.json()
|
||||||
|
assert body["title"] == "Comment on CMS-2026-2377-1"
|
||||||
|
assert [(f["name"], f["renderable"]) for f in body["files"]] == [
|
||||||
|
("a.pdf", True),
|
||||||
|
("b.docx", False),
|
||||||
|
]
|
||||||
|
assert client.get("/pdf/NOPE9999").json()["files"] == []
|
||||||
|
|
||||||
|
r = client.get("/pdf/ABCD1234/file")
|
||||||
|
assert r.status_code == 200 and r.headers["content-type"] == "application/pdf"
|
||||||
|
assert r.headers["content-disposition"].startswith("inline")
|
||||||
|
assert r.headers["accept-ranges"] == "bytes" and r.content.startswith(b"%PDF")
|
||||||
|
r = client.get("/pdf/ABCD1234/file", headers={"Range": "bytes=0-3"})
|
||||||
|
assert r.status_code == 206 and r.content == b"%PDF"
|
||||||
|
r = client.get("/pdf/ABCD1234/file?name=b.docx")
|
||||||
|
assert r.status_code == 200 and r.headers["content-disposition"].startswith(
|
||||||
|
"attachment"
|
||||||
|
)
|
||||||
|
assert client.get("/pdf/ABCD1234/file?name=nope.pdf").status_code == 404
|
||||||
|
assert client.get("/pdf/ABCD1234/file?name=../a.pdf").status_code == 404
|
||||||
|
|
||||||
|
def test_chat_and_search_pages_link_the_viewer(self):
|
||||||
|
chat = client.get("/").text
|
||||||
|
assert "'/ui/pdf/' + encodeURIComponent(src.item_key)" in chat
|
||||||
|
assert "src.kind !== 'rule'" in chat
|
||||||
|
search = client.get("/ui/search").text
|
||||||
|
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search
|
||||||
|
|||||||
@@ -148,6 +148,8 @@ class TestAsSource:
|
|||||||
0.123456,
|
0.123456,
|
||||||
)
|
)
|
||||||
assert set(s) == {
|
assert set(s) == {
|
||||||
|
"attachment",
|
||||||
|
"page",
|
||||||
"id",
|
"id",
|
||||||
"label",
|
"label",
|
||||||
"kind",
|
"kind",
|
||||||
|
|||||||
128
tests/llm/test_pdfs.py
Normal file
128
tests/llm/test_pdfs.py
Normal file
@@ -0,0 +1,128 @@
|
|||||||
|
"""llm.pdfs — attachments on file for the viewer: comment downloads, bib
|
||||||
|
attachments, Zotero storage PDFs; name-only resolution inside the roots."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sqlite3
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from bib.item import Source
|
||||||
|
from bib.store import Store
|
||||||
|
from llm.pdfs import PdfFile, pdf_files, resolve
|
||||||
|
|
||||||
|
|
||||||
|
def _zotero(
|
||||||
|
tmp_path: Path, parent_key: str, att_key: str, name: str
|
||||||
|
) -> tuple[Path, Path]:
|
||||||
|
db = tmp_path / "zotero.sqlite"
|
||||||
|
con = sqlite3.connect(db)
|
||||||
|
con.executescript(
|
||||||
|
"CREATE TABLE items (itemID INTEGER PRIMARY KEY, key TEXT);"
|
||||||
|
"CREATE TABLE itemAttachments (itemID INTEGER, parentItemID INTEGER, path TEXT);"
|
||||||
|
"CREATE TABLE deletedItems (itemID INTEGER);"
|
||||||
|
)
|
||||||
|
con.execute("INSERT INTO items VALUES (1, ?)", (parent_key,))
|
||||||
|
con.execute("INSERT INTO items VALUES (2, ?)", (att_key,))
|
||||||
|
con.execute("INSERT INTO items VALUES (3, 'GONE1234')")
|
||||||
|
con.execute("INSERT INTO itemAttachments VALUES (2, 1, ?)", (f"storage:{name}",))
|
||||||
|
con.execute("INSERT INTO itemAttachments VALUES (3, 1, 'storage:deleted.pdf')")
|
||||||
|
con.execute("INSERT INTO deletedItems VALUES (3)")
|
||||||
|
con.commit()
|
||||||
|
con.close()
|
||||||
|
storage = tmp_path / "zstorage"
|
||||||
|
(storage / att_key).mkdir(parents=True)
|
||||||
|
(storage / att_key / name).write_bytes(b"%PDF-1.4 z")
|
||||||
|
(storage / "GONE1234").mkdir()
|
||||||
|
(storage / "GONE1234" / "deleted.pdf").write_bytes(b"%PDF-1.4 d")
|
||||||
|
return db, storage
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_three_sources_in_order_and_deduped(tmp_path):
|
||||||
|
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
|
||||||
|
key = store.upsert(Source(title="Paper", url="https://doi.org/10.1/x"))
|
||||||
|
pdf = tmp_path / "paper.pdf"
|
||||||
|
pdf.write_bytes(b"%PDF-1.4 b")
|
||||||
|
store.attach_file(key, pdf)
|
||||||
|
(tmp_path / "notes.docx").write_bytes(b"PK docx")
|
||||||
|
store.attach_file(key, tmp_path / "notes.docx")
|
||||||
|
zdb, zstore = _zotero(
|
||||||
|
tmp_path, key, "ATT00001", "paper.pdf"
|
||||||
|
) # same name as the bib copy → deduped
|
||||||
|
files = pdf_files(
|
||||||
|
store,
|
||||||
|
key,
|
||||||
|
zotero_sqlite=zdb,
|
||||||
|
zotero_storage=zstore,
|
||||||
|
comments_root=tmp_path / "comments",
|
||||||
|
)
|
||||||
|
assert [(f.name, f.source, f.renderable) for f in files] == [
|
||||||
|
("paper.pdf", "bib", True),
|
||||||
|
("notes.docx", "bib", False),
|
||||||
|
]
|
||||||
|
assert files[1].media_type.startswith("application/vnd.openxmlformats")
|
||||||
|
assert all(isinstance(f, PdfFile) and f.size > 0 for f in files)
|
||||||
|
store.close()
|
||||||
|
|
||||||
|
|
||||||
|
def test_zotero_only_item_and_deleted_attachments_skipped(tmp_path):
|
||||||
|
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
|
||||||
|
key = store.upsert(Source(title="Zot only", url="https://example.org/z"))
|
||||||
|
zdb, zstore = _zotero(tmp_path, key, "ATT00002", "Cassidy - 2025 - RFI.pdf")
|
||||||
|
files = pdf_files(store, key, zotero_sqlite=zdb, zotero_storage=zstore)
|
||||||
|
assert [(f.name, f.source) for f in files] == [
|
||||||
|
("Cassidy - 2025 - RFI.pdf", "zotero")
|
||||||
|
]
|
||||||
|
assert pdf_files(store, "NOPE1234", zotero_sqlite=zdb, zotero_storage=zstore) == []
|
||||||
|
assert (
|
||||||
|
pdf_files(
|
||||||
|
store, key, zotero_sqlite=tmp_path / "missing.sqlite", zotero_storage=zstore
|
||||||
|
)
|
||||||
|
== []
|
||||||
|
)
|
||||||
|
store.close()
|
||||||
|
|
||||||
|
|
||||||
|
def test_comment_attachments_from_the_state_tree(tmp_path):
|
||||||
|
store = Store(":memory:", storage_dir=tmp_path / "bibstorage")
|
||||||
|
key = store.upsert(
|
||||||
|
Source(title="c", url="https://www.regulations.gov/comment/CMS-2026-2377-40314")
|
||||||
|
)
|
||||||
|
d = tmp_path / "comments" / "CMS-2026-2377" / "CMS-2026-2377-40314"
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF-1.4 a1")
|
||||||
|
(d / "attachment_2.docx").write_bytes(b"PK")
|
||||||
|
(d / "combined.md").write_text("# not an attachment")
|
||||||
|
(d / "attachment_1.png").write_bytes(b"png")
|
||||||
|
files = pdf_files(store, key, comments_root=tmp_path / "comments")
|
||||||
|
assert [(f.name, f.source, f.renderable) for f in files] == [
|
||||||
|
("attachment_1.pdf", "comment", True),
|
||||||
|
("attachment_2.docx", "comment", False),
|
||||||
|
]
|
||||||
|
# a non-comment url, or a comment with no downloaded dir, yields nothing
|
||||||
|
other = store.upsert(Source(title="p", url="https://doi.org/10.1/y"))
|
||||||
|
assert pdf_files(store, other, comments_root=tmp_path / "comments") == []
|
||||||
|
gone = store.upsert(
|
||||||
|
Source(
|
||||||
|
title="c2", url="https://www.regulations.gov/comment/CMS-2026-2377-99999"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
assert pdf_files(store, gone, comments_root=tmp_path / "comments") == []
|
||||||
|
store.close()
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_by_name_inside_roots_only(tmp_path):
|
||||||
|
inside = tmp_path / "root" / "K" / "a.pdf"
|
||||||
|
inside.parent.mkdir(parents=True)
|
||||||
|
inside.write_bytes(b"%PDF")
|
||||||
|
outside = tmp_path / "elsewhere" / "b.pdf"
|
||||||
|
outside.parent.mkdir()
|
||||||
|
outside.write_bytes(b"%PDF")
|
||||||
|
files = [PdfFile("a.pdf", inside, 4, "bib"), PdfFile("b.pdf", outside, 4, "bib")]
|
||||||
|
roots = (tmp_path / "root",)
|
||||||
|
assert resolve(files, "", roots=roots).name == "a.pdf"
|
||||||
|
assert resolve(files, "a.pdf", roots=roots).name == "a.pdf"
|
||||||
|
assert (
|
||||||
|
resolve(files, "b.pdf", roots=roots) is None
|
||||||
|
) # listed, but outside every root
|
||||||
|
assert resolve(files, "../a.pdf", roots=roots) is None
|
||||||
|
assert resolve([], "", roots=roots) is None
|
||||||
Reference in New Issue
Block a user