feat(llm): the document viewer renders Word (.docx) and plain-text attachments in-page, not just PDFs
Some checks failed
CI / lint (push) Successful in 38s
CI / notebooks-smoke (push) Successful in 1m38s
Deploy / notebooks (push) Has been skipped
CI / test (push) Failing after 2m22s
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 13s
Infra CI / docs (push) Successful in 21s
Infra CI / notebooks (push) Successful in 51s
Deploy / report (push) Has been cancelled
Infra CI / mc (push) Has been cancelled
Infra CI / api (push) Has been cancelled
Infra CI / llm (push) Has been cancelled

.docx: vendored docx-preview 0.3.5 (+ JSZip 3.10.1, both served from
/ui/vendor/) renders the file in the browser as pages (sections split
on page breaks, headers/footers/footnotes, embedded images), with the
same toolbar: fit-to-width / zoom (CSS zoom, transform fallback),
page input with hash tracking, arrow keys, and find-in-document over
the rendered DOM with highlighted, steppable matches seeded from the
cited snippet. .txt renders as a text pane with the same find. Legacy
.doc stays download-only (no comment attachment on file is one).

/pdf/<key> now says how each file renders (renderer: pdf | docx |
text | null); the file endpoint serves PDFs inline and everything else
as a download so the viewer fetches it as a blob. Chat and search
links read 'PDF' / 'Word' / 'file' by format.
This commit is contained in:
kert
2026-09-24 19:02:57 -04:00
parent 1b81aeac92
commit 6324960f28
9 changed files with 266 additions and 24 deletions

View File

@@ -108,12 +108,12 @@ def search_page() -> str:
return _page("search.html")
_VENDOR = {"pdf.min.mjs", "pdf.worker.min.mjs"}
_VENDOR = {"pdf.min.mjs", "pdf.worker.min.mjs", "jszip.min.js", "docx-preview.min.js"}
@app.get("/ui/vendor/{name}")
def vendor(name: str) -> Response:
"""Self-hosted browser libraries (PDF.js) — a fixed allow-list, served
"""Self-hosted browser libraries (PDF.js, JSZip + docx-preview) — a fixed allow-list, served
from the package so the viewer works without a CDN and its worker is
same-origin (a cross-origin worker script is refused by browsers)."""
if name not in _VENDOR:
@@ -169,8 +169,9 @@ def _pdf_files(key: str) -> list:
def pdf_list(key: str) -> dict:
"""The attachments on file for a comment or library item: name, size,
where it came from (comment download, bib attachment, Zotero storage)
and whether the viewer can render it (PDF) or only offer it for
download. Rules are not served here — they link to federalregister.gov."""
and how the viewer renders it (``pdf`` / ``docx`` / ``text``, or null
for download only). Rules are not served here — they link to
federalregister.gov."""
if not _KEY_RE.match(key):
raise HTTPException(404)
files = _pdf_files(key)
@@ -194,6 +195,7 @@ def pdf_list(key: str) -> dict:
"size": f.size,
"source": f.source,
"renderable": f.renderable,
"renderer": f.renderer,
"media_type": f.media_type,
}
for f in files
@@ -226,7 +228,7 @@ def pdf_file(key: str, name: str = "") -> FileResponse:
chosen.path,
media_type=chosen.media_type,
filename=chosen.name,
content_disposition_type="inline" if chosen.renderable else "attachment",
content_disposition_type="inline" if chosen.renderer == "pdf" else "attachment",
headers={"Accept-Ranges": "bytes", "Cache-Control": "private, max-age=3600"},
)

View File

@@ -11,8 +11,8 @@ read from the live ``zotero.sqlite`` in immutable mode (a cheap indexed
lookup per item; the indexer's 2 GB snapshot is for whole-library walks,
not one request).
PDFs render in the page; other formats (docx, doc, txt) are listed for
download. A file is only ever served after it was found through this
PDFs, Word (.docx) and plain text render in the page; legacy .doc is
listed for download. A file is only ever served after it was found through this
lookup by its *name* — the API never joins a client-supplied path — and
only when it sits inside one of the storage roots.
"""
@@ -48,9 +48,17 @@ class PdfFile:
def media_type(self) -> str:
return _MEDIA.get(Path(self.name).suffix.lower(), "application/octet-stream")
@property
def renderer(self) -> str | None:
"""How the viewer shows it: ``pdf`` (PDF.js), ``docx``
(docx-preview, page-like layout in the browser), ``text`` (a
plain-text pane), or ``None`` — download only (legacy ``.doc``)."""
ext = Path(self.name).suffix.lower()
return {".pdf": "pdf", ".docx": "docx", ".txt": "text"}.get(ext)
@property
def renderable(self) -> bool:
return self.name.lower().endswith(".pdf")
return self.renderer is not None
def _comment_files(store: Any, key: str, root: Path) -> list[PdfFile]:

View File

@@ -668,7 +668,7 @@
(src.page ? '&page=' + encodeURIComponent(src.page) : '') +
'&q=' + encodeURIComponent((src.snippet || '').slice(0, 80));
v.target = '_blank'; v.rel = 'noopener';
v.textContent = (src.attachment.toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
v.textContent = (/\.pdf$/i.test(src.attachment) ? 'PDF' : /\.docx$/i.test(src.attachment) ? 'Word' : 'file') +
(src.page ? ' p.' + src.page : '');
el.appendChild(document.createTextNode(' '));
el.appendChild(v);

View File

@@ -3,7 +3,7 @@
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Library PDF</title>
<title>Library Document</title>
<link rel="icon" type="image/png" sizes="32x32" href="//dashboard.fhirworx.io/fav32.png">
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
@@ -96,6 +96,16 @@
position: absolute; inset: 0; display: flex; align-items: center; justify-content: center;
color: var(--muted-fg); font-family: var(--font-mono); font-size: 12px;
}
/* docx-preview output: each section is a page; zoom scales the wrapper */
#docxwrap { transform-origin: top center; }
#docxwrap .docx-wrapper { background: transparent; padding: 0; display: flex; flex-direction: column; align-items: center; gap: 14px; }
#docxwrap section.docx { box-shadow: 0 1px 3px rgba(0,0,0,.25); margin: 0 !important; background: #fff; overflow: hidden; }
#docxwrap mark, #textdoc mark { background: var(--highlight); border-radius: 2px; color: inherit; }
#docxwrap mark.current, #textdoc mark.current { background: rgba(255, 140, 0, .55); }
#textdoc {
background: #fff; box-shadow: 0 1px 3px rgba(0,0,0,.25); padding: 28px 32px; width: 100%; max-width: 860px;
font-family: var(--font-mono); font-size: 13px; line-height: 1.5; white-space: pre-wrap; overflow-wrap: anywhere; margin: 0;
}
.notice {
margin: 40px auto; max-width: 560px; padding: 16px 20px; border: 1px solid var(--border);
background: var(--card); border-radius: 3px; color: var(--muted-fg); font-style: italic;
@@ -114,7 +124,7 @@
</head>
<body>
<header>
<h1>Library PDF</h1>
<h1>Library Document</h1>
<span class="title" id="title"></span>
<a class="nav" href="/">Chat</a>
<a class="nav" href="/ui/search">Search</a>
@@ -184,7 +194,7 @@
const d = await r.json();
files = d.files || [];
$('title').textContent = d.title || key;
document.title = (d.title ? d.title + ' — ' : '') + 'Library PDF';
document.title = (d.title ? d.title + ' — ' : '') + 'Library Document';
} catch (e) {
notice('Could not look up the item: ' + e.message, true);
throw e;
@@ -201,6 +211,11 @@
}
if (files.length > 1) sel.hidden = false;
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
// the "done:" throws below end this module on purpose once a non-PDF
// pane is up; keep the console quiet about them
window.addEventListener('unhandledrejection', ev => {
if (ev.reason && /^done: /.test(String(ev.reason.message || ev.reason))) ev.preventDefault();
});
sel.value = current.name;
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
$('download').href = fileUrl;
@@ -210,10 +225,198 @@
location.search = p.toString();
});
if (current.renderable === false) {
notice('"' + current.name + '" is not a PDF (' + (current.media_type || 'file') + '), so it cannot be rendered here.', false, fileUrl);
const renderer = current.renderer || (current.renderable === false ? null : 'pdf');
if (!renderer) {
notice('"' + current.name + '" (' + (current.media_type || 'file') + ') has no in-page renderer — legacy Word .doc files open in Word or LibreOffice.', false, fileUrl);
throw new Error('not renderable');
}
// ── DOM-based find (docx and text panes): wrap matches in <mark>,
// step with Enter / Shift+Enter, the current match scrolls into view ──
function domFind(root) {
const state = { term: '', marks: [], idx: -1 };
const norm = s => s.replace(/\s+/g, ' ').toLowerCase();
function clear() {
for (const m of state.marks) { const t = document.createTextNode(m.textContent); m.replaceWith(t); }
state.marks = []; state.idx = -1; root.normalize();
}
function run(term) {
clear();
state.term = norm(term).trim();
$('findcount').textContent = '';
if (!state.term) return;
const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT);
const nodes = [];
for (let n = walker.nextNode(); n; n = walker.nextNode()) if (n.nodeValue.trim()) nodes.push(n);
for (const n of nodes) {
const text = n.nodeValue, low = norm(text);
let at = low.indexOf(state.term);
if (at === -1) continue;
const frag = document.createDocumentFragment();
let last = 0;
while (at !== -1) {
frag.appendChild(document.createTextNode(text.slice(last, at)));
const m = document.createElement('mark'); m.textContent = text.slice(at, at + state.term.length);
frag.appendChild(m); state.marks.push(m);
last = at + state.term.length; at = low.indexOf(state.term, last);
}
frag.appendChild(document.createTextNode(text.slice(last)));
n.parentNode.replaceChild(frag, n);
}
$('findcount').textContent = state.marks.length ? '0/' + state.marks.length : 'no matches';
if (state.marks.length) step(1);
}
function step(dir) {
if (!state.marks.length) return;
if (state.idx >= 0) state.marks[state.idx].classList.remove('current');
state.idx = (state.idx + dir + state.marks.length) % state.marks.length;
const m = state.marks[state.idx];
m.classList.add('current');
m.scrollIntoView({ block: 'center', behavior: 'smooth' });
$('findcount').textContent = (state.idx + 1) + '/' + state.marks.length;
}
let t = null;
$('find').addEventListener('input', () => { clearTimeout(t); t = setTimeout(() => run($('find').value), 250); });
$('find').addEventListener('keydown', e => {
if (e.key === 'Enter') { e.preventDefault(); if (norm($('find').value).trim() !== state.term) run($('find').value); else step(e.shiftKey ? -1 : 1); }
if (e.key === 'Escape') { $('find').value = ''; run(''); viewer.focus(); }
});
return { run, step };
}
function seedFind(find) {
if (!wantQuery) return;
const probe = wantQuery.split(/\s+/).slice(0, 6).join(' ');
$('find').value = probe;
find.run(probe);
}
// Page navigation over a list of page elements (docx sections, or the
// single text pane): prev/next/input, hash tracking, arrow keys.
function domPages(getPages) {
let cur = 1;
const count = () => getPages().length;
function goTo(n, smooth) {
const els = getPages();
n = Math.min(els.length, Math.max(1, n | 0));
if (!els.length) return;
viewer.scrollTo({ top: els[n - 1].offsetTop - 12, behavior: smooth ? 'smooth' : 'auto' });
cur = n; $('pageno').value = String(n); history.replaceState(null, '', '#page=' + n);
}
function update() {
const els = getPages();
const top = viewer.scrollTop + viewer.clientHeight / 3;
let best = 1, bestDist = Infinity;
els.forEach((el, i) => { const d = Math.abs(el.offsetTop - top); if (d < bestDist) { bestDist = d; best = i + 1; } });
if (best !== cur) { cur = best; $('pageno').value = String(best); history.replaceState(null, '', '#page=' + best); }
}
$('count').textContent = '/ ' + count();
$('pageno').max = String(count());
$('prev').addEventListener('click', () => goTo(cur - 1, true));
$('next').addEventListener('click', () => goTo(cur + 1, true));
$('pageno').addEventListener('change', () => goTo(parseInt($('pageno').value, 10) || 1, false));
viewer.addEventListener('scroll', () => requestAnimationFrame(update), { passive: true });
document.addEventListener('keydown', e => {
if (e.target.tagName === 'INPUT' || e.target.tagName === 'SELECT') return;
if (e.key === 'ArrowLeft' || e.key === 'PageUp') { e.preventDefault(); goTo(cur - 1, true); }
else if (e.key === 'ArrowRight' || e.key === 'PageDown') { e.preventDefault(); goTo(cur + 1, true); }
else if ((e.ctrlKey || e.metaKey) && e.key.toLowerCase() === 'f') { e.preventDefault(); $('find').focus(); $('find').select(); }
});
return { goTo, update, refresh() { $('count').textContent = '/ ' + count(); $('pageno').max = String(count()); } };
}
function loadScript(src) {
return new Promise((resolve, reject) => {
const sc = document.createElement('script');
sc.src = src; sc.onload = resolve; sc.onerror = () => reject(new Error('failed to load ' + src));
document.head.appendChild(sc);
});
}
// ── plain text ──
if (renderer === 'text') {
let text;
try {
const r = await fetch(fileUrl);
if (!r.ok) throw new Error('fetch failed (' + r.status + ')');
text = await r.text();
} catch (e) {
notice('Could not load the file: ' + e.message, true, fileUrl);
throw e;
}
const pre = document.createElement('pre'); pre.id = 'textdoc'; pre.textContent = text;
viewer.replaceChildren(pre);
for (const id of ['zoomout', 'zoomin', 'fit', 'zoomlabel']) $(id).hidden = true;
const nav = domPages(() => [pre]);
const find = domFind(pre);
nav.goTo(1, false);
seedFind(find);
throw new Error('done: text'); // the module's PDF path below must not run
}
// ── Word (.docx) via docx-preview: sections become pages; zoom scales
// the wrapper to the viewer width (fit) or a factor ──
if (renderer === 'docx') {
let blob;
try {
await loadScript('/ui/vendor/jszip.min.js');
await loadScript('/ui/vendor/docx-preview.min.js');
const r = await fetch(fileUrl);
if (!r.ok) throw new Error('fetch failed (' + r.status + ')');
blob = await r.blob();
} catch (e) {
notice('The Word renderer could not be loaded: ' + e.message, true, fileUrl);
throw e;
}
const wrap = document.createElement('div'); wrap.id = 'docxwrap';
viewer.replaceChildren(wrap);
try {
await window.docx.renderAsync(blob, wrap, null, {
className: 'docx', inWrapper: true, breakPages: true, ignoreLastRenderedPageBreak: false,
renderHeaders: true, renderFooters: true, renderFootnotes: true, renderEndnotes: true,
useBase64URL: true, experimental: true, trimXmlDeclaration: true,
});
} catch (e) {
notice('This Word file could not be rendered: ' + (e && e.message || e), true, fileUrl);
throw e;
}
const pagesOf = () => Array.from(wrap.querySelectorAll('section.docx'));
if (!pagesOf().length) { notice('The Word file rendered no pages.', true, fileUrl); throw new Error('no pages'); }
// fit / zoom: scale the wrapper; CSS zoom keeps layout metrics honest,
// transform is the fallback for engines without it
let fit = true, factor = 1;
const useZoom = CSS.supports('zoom', '1');
function applyZoom() {
const page = pagesOf()[0];
const natural = page.offsetWidth / (useZoom ? 1 : 1);
const cs = getComputedStyle(viewer);
const avail = viewer.clientWidth - parseFloat(cs.paddingLeft) - parseFloat(cs.paddingRight);
const base = useZoom ? page.getBoundingClientRect().width / (parseFloat(wrap.style.zoom) || 1) : natural;
const s = fit ? Math.min(1.6, avail / base) : (avail / base) * factor;
if (useZoom) wrap.style.zoom = String(s);
else { wrap.style.transform = 'scale(' + s + ')'; wrap.style.width = base + 'px'; wrap.style.height = (wrap.scrollHeight * s) + 'px'; }
$('zoomlabel').textContent = fit ? 'fit' : Math.round(factor * 100) + '%';
}
$('zoomin').addEventListener('click', () => { factor = Math.min(4, (fit ? 1 : factor) * 1.25); fit = false; applyZoom(); });
$('zoomout').addEventListener('click', () => { factor = Math.max(0.3, (fit ? 1 : factor) / 1.25); fit = false; applyZoom(); });
$('fit').addEventListener('click', () => { fit = true; factor = 1; applyZoom(); });
document.addEventListener('keydown', e => {
if (e.target.tagName === 'INPUT' || e.target.tagName === 'SELECT') return;
if (e.key === '+' || e.key === '=') { e.preventDefault(); $('zoomin').click(); }
else if (e.key === '-') { e.preventDefault(); $('zoomout').click(); }
else if (e.key === '0') { e.preventDefault(); $('fit').click(); }
});
let rt2 = null;
new ResizeObserver(() => { clearTimeout(rt2); rt2 = setTimeout(applyZoom, 120); }).observe(viewer);
applyZoom();
const nav = domPages(pagesOf);
const find = domFind(wrap);
const hashPage = parseInt((location.hash.match(/page=(\d+)/) || [])[1] || '0', 10);
nav.goTo(hashPage || wantPage, false);
seedFind(find);
throw new Error('done: docx'); // stop before the PDF path
}
if (!pdfjs) {
notice('The in-page renderer could not be loaded.', true, fileUrl);
throw new Error('pdfjs missing');

View File

@@ -244,7 +244,7 @@
(r.page ? '&page=' + encodeURIComponent(r.page) : '') +
'&q=' + encodeURIComponent((r.snippet || '').slice(0, 80));
view.target = '_blank'; view.rel = 'noopener';
view.textContent = (String(r.attachment).toLowerCase().endsWith('.pdf') ? 'PDF' : 'file') +
view.textContent = (/\.pdf$/i.test(r.attachment) ? 'PDF' : /\.docx$/i.test(r.attachment) ? 'Word' : 'file') +
(r.page ? ' p.' + r.page : '');
line.appendChild(view);
}

File diff suppressed because one or more lines are too long

13
src/llm/web/vendor/jszip.min.js vendored Normal file

File diff suppressed because one or more lines are too long

View File

@@ -456,6 +456,13 @@ class TestPdfViewer:
"text/javascript"
)
assert client.get("/ui/vendor/pdf.worker.min.mjs").status_code == 200
assert client.get("/ui/vendor/jszip.min.js").status_code == 200
assert client.get("/ui/vendor/docx-preview.min.js").status_code == 200
assert (
"renderAsync" in html
and "docx-preview.min.js" in html
and "renderer === 'text'" in html
)
assert client.get("/ui/vendor/../api.py").status_code in (404, 400)
assert client.get("/ui/vendor/evil.mjs").status_code == 404
@@ -497,9 +504,9 @@ class TestPdfViewer:
assert r.status_code == 200
body = r.json()
assert body["title"] == "Comment on CMS-2026-2377-1"
assert [(f["name"], f["renderable"]) for f in body["files"]] == [
("a.pdf", True),
("b.docx", False),
assert [(f["name"], f["renderable"], f["renderer"]) for f in body["files"]] == [
("a.pdf", True, "pdf"),
("b.docx", True, "docx"),
]
assert client.get("/pdf/NOPE9999").json()["files"] == []

View File

@@ -55,10 +55,11 @@ def test_all_three_sources_in_order_and_deduped(tmp_path):
zotero_storage=zstore,
comments_root=tmp_path / "comments",
)
assert [(f.name, f.source, f.renderable) for f in files] == [
("paper.pdf", "bib", True),
("notes.docx", "bib", False),
assert [(f.name, f.source, f.renderer) for f in files] == [
("paper.pdf", "bib", "pdf"),
("notes.docx", "bib", "docx"),
]
assert all(f.renderable for f in files)
assert files[1].media_type.startswith("application/vnd.openxmlformats")
assert all(isinstance(f, PdfFile) and f.size > 0 for f in files)
store.close()
@@ -94,10 +95,16 @@ def test_comment_attachments_from_the_state_tree(tmp_path):
(d / "combined.md").write_text("# not an attachment")
(d / "attachment_1.png").write_bytes(b"png")
files = pdf_files(store, key, comments_root=tmp_path / "comments")
assert [(f.name, f.source, f.renderable) for f in files] == [
("attachment_1.pdf", "comment", True),
("attachment_2.docx", "comment", False),
(d / "attachment_3.doc").write_bytes(b"\xd0\xcf")
(d / "attachment_4.txt").write_text("plain")
files = pdf_files(store, key, comments_root=tmp_path / "comments")
assert [(f.name, f.source, f.renderer) for f in files] == [
("attachment_1.pdf", "comment", "pdf"),
("attachment_2.docx", "comment", "docx"),
("attachment_3.doc", "comment", None),
("attachment_4.txt", "comment", "text"),
]
assert [f.renderable for f in files] == [True, True, False, True]
# a non-comment url, or a comment with no downloaded dir, yields nothing
other = store.upsert(Source(title="p", url="https://doi.org/10.1/y"))
assert pdf_files(store, other, comments_root=tmp_path / "comments") == []