feat(llm): Word-exact pages — LibreOffice in the llm image converts .docx to PDF on demand (cached) for the document viewer
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s

llm.convert.to_pdf runs headless Writer (soffice --convert-to
pdf:writer_pdf_Export, its own profile dir, one at a time) and caches
the result by the source's path/size/mtime under .state/llm/pdfcache
(compose mounts it read-write). The image adds
libreoffice-writer-nogui plus Carlito/Caladea/Liberation/DejaVu so
Calibri/Cambria/Times/Arial letters paginate like Word.

/pdf/<key> now returns, per file, the url the viewer should load and
the download url of the original; a Word file lists as renderer pdf
with url …&as=pdf when the converter is present, so the PDF.js path
(real pages, text layer, find, page links) shows it; /pdf/<key>/file
?as=pdf serves the conversion inline (502 on failure, 404 without the
converter). Browser-side docx-preview remains the fallback.
This commit is contained in:
kert
2026-09-24 19:12:14 -04:00
parent 068963e8ca
commit 523be76fc6
7 changed files with 353 additions and 14 deletions

View File

@@ -692,6 +692,7 @@ services:
- ./data:/app/data
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
- ./.state/llm/pdfcache:/app/.state/llm/pdfcache # LibreOffice Word→PDF cache for the viewer
environment:
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
- LLM_PG_HOST=postgres

View File

@@ -9,6 +9,14 @@ ARG PYPI_INDEX_URL=""
# Patch base image CVEs + install curl for the healthcheck.
RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends curl && rm -rf /var/lib/apt/lists/*
# Word → PDF for the document viewer (llm.convert): headless Writer plus
# metric-compatible fonts so Calibri/Cambria/Times/Arial letters paginate
# like Word (Carlito, Caladea, Liberation) — no X, no Java.
RUN apt-get update && apt-get install -y --no-install-recommends \
libreoffice-writer-nogui fonts-crosextra-carlito fonts-crosextra-caladea \
fonts-liberation2 fonts-dejavu-core \
&& rm -rf /var/lib/apt/lists/*
COPY pyproject.toml uv.lock README.md ./
COPY src/ src/

View File

@@ -17,8 +17,9 @@ from contextlib import asynccontextmanager
from importlib import resources
from pathlib import Path
from typing import AsyncIterator, Iterator
from urllib.parse import quote
from fastapi import FastAPI, Header, HTTPException
from fastapi import FastAPI, Header, HTTPException, Query
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
from pydantic import BaseModel
@@ -186,28 +187,42 @@ def pdf_list(key: str) -> dict:
store.close()
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
title = ""
return {
"key": key,
"title": title,
"files": [
from llm import convert
convertible = convert.available()
base = f"/pdf/{key}/file?name="
out = []
for f in files:
ext = Path(f.name).suffix.lower()
converted = convertible and ext in convert.CONVERTIBLE
out.append(
{
"name": f.name,
"size": f.size,
"source": f.source,
"renderable": f.renderable,
"renderer": f.renderer,
"renderable": f.renderable or converted,
# Word files render as the PDF LibreOffice lays out (Word-exact
# pages) when the converter is in the image; docx-preview in
# the browser is the fallback
"renderer": "pdf" if converted else f.renderer,
"converted": converted,
"media_type": f.media_type,
"url": base + quote(f.name) + ("&as=pdf" if converted else ""),
"download_url": base + quote(f.name),
}
for f in files
],
}
)
return {"key": key, "title": title, "files": out}
@app.get("/pdf/{key}/file")
def pdf_file(key: str, name: str = "") -> FileResponse:
def pdf_file(
key: str, name: str = "", as_: str = Query("", alias="as")
) -> FileResponse:
"""The attachment bytes — a PDF inline with range requests (PDF.js
streams pages from a large file instead of waiting for all of it),
anything else as a download."""
anything else as a download. ``as=pdf`` on a Word file returns the
LibreOffice rendering (cached under .state/llm/pdfcache); 502 when
the conversion fails, 404 when the converter is not in the image."""
from llm.pdfs import resolve
if not _KEY_RE.match(key):
@@ -224,6 +239,31 @@ def pdf_file(key: str, name: str = "") -> FileResponse:
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
if chosen is None:
raise HTTPException(404, "no such attachment on file for this item")
if as_ == "pdf" and chosen.renderer != "pdf":
from conf import ROOT
from llm import convert
if (
Path(chosen.name).suffix.lower() not in convert.CONVERTIBLE
or not convert.available()
):
raise HTTPException(404, "no PDF rendering available for this file")
try:
rendered = convert.to_pdf(
chosen.path, cache_dir=Path(ROOT) / ".state" / "llm" / "pdfcache"
)
except convert.ConversionError as exc:
raise HTTPException(502, str(exc)) from exc
return FileResponse(
rendered,
media_type="application/pdf",
filename=Path(chosen.name).stem + ".pdf",
content_disposition_type="inline",
headers={
"Accept-Ranges": "bytes",
"Cache-Control": "private, max-age=3600",
},
)
return FileResponse(
chosen.path,
media_type=chosen.media_type,

104
src/llm/convert.py Normal file
View File

@@ -0,0 +1,104 @@
"""Word → PDF with LibreOffice, for Word-exact pages in the viewer.
``soffice --headless --convert-to pdf`` lays the document out the way
Writer does (real pagination, styles, metric-compatible fonts for
Calibri/Cambria/Times/Arial via Carlito/Caladea/Liberation), which the
browser-side docx-preview cannot. Conversions are cached by the source
file's identity (path, size, mtime) under ``cache_dir`` so a document is
converted once, and serialised behind one lock because a shared
LibreOffice profile cannot run two instances at a time.
available() → soffice is on PATH
to_pdf(src, cache_dir=…) → cached PDF path (converts on a miss)
"""
from __future__ import annotations
import hashlib
import logging
import os
import shutil
import subprocess
import tempfile
import threading
from pathlib import Path
log = logging.getLogger(__name__)
_LOCK = threading.Lock()
_TIMEOUT = 180
CONVERTIBLE = (".docx", ".doc", ".rtf", ".odt")
class ConversionError(RuntimeError):
pass
def soffice() -> str | None:
return shutil.which("soffice") or shutil.which("libreoffice")
def available() -> bool:
return soffice() is not None
def cache_key(src: Path) -> str:
st = src.stat()
raw = f"{src.resolve()}|{st.st_size}|{st.st_mtime_ns}"
return hashlib.sha1(raw.encode()).hexdigest() # noqa: S324 — cache identity
def to_pdf(src: Path, *, cache_dir: Path, runner=None) -> Path:
"""The PDF rendering of *src*, converting with LibreOffice when the
cache has no current copy. *runner* (tests) replaces ``subprocess.run``.
Raises ``ConversionError`` when soffice is missing, times out, fails or
produces nothing."""
src = Path(src)
if not src.is_file():
raise ConversionError(f"{src.name}: not on file")
cache_dir = Path(cache_dir)
cache_dir.mkdir(parents=True, exist_ok=True)
out = cache_dir / f"{cache_key(src)}.pdf"
if out.is_file() and out.stat().st_size > 0:
return out
exe = soffice()
if exe is None:
raise ConversionError("LibreOffice (soffice) is not installed in this image")
run = runner or subprocess.run
profile = cache_dir / ".lo-profile"
profile.mkdir(exist_ok=True)
with _LOCK, tempfile.TemporaryDirectory(prefix="lo-", dir=cache_dir) as tmp:
cmd = [
exe,
f"-env:UserInstallation=file://{profile}",
"--headless",
"--norestore",
"--nolockcheck",
"--convert-to",
"pdf:writer_pdf_Export",
"--outdir",
tmp,
str(src),
]
env = {**os.environ, "HOME": str(profile)}
try:
proc = run(cmd, capture_output=True, text=True, timeout=_TIMEOUT, env=env)
except subprocess.TimeoutExpired as exc:
raise ConversionError(
f"{src.name}: conversion timed out after {_TIMEOUT}s"
) from exc
produced = Path(tmp) / (src.stem + ".pdf")
if (
proc.returncode != 0
or not produced.is_file()
or produced.stat().st_size == 0
):
detail = (proc.stderr or proc.stdout or "").strip()[-300:]
raise ConversionError(
f"{src.name}: LibreOffice failed ({proc.returncode}) {detail}"
)
tmp_out = out.with_suffix(".pdf.part")
shutil.move(str(produced), tmp_out)
os.replace(tmp_out, out)
log.info("converted %s → %s (%d bytes)", src.name, out.name, out.stat().st_size)
return out

View File

@@ -214,8 +214,10 @@
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
sel.value = current.name;
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
$('download').href = fileUrl;
const rawUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
const fileUrl = current.url || rawUrl; // a Word file's LibreOffice rendering when the image has the converter
$('download').href = current.download_url || rawUrl; // the original bytes
if (current.converted) $('download').textContent = 'download original';
sel.addEventListener('change', () => {
const p = new URLSearchParams(location.search);
p.set('file', sel.value); p.delete('page');

View File

@@ -468,8 +468,13 @@ class TestPdfViewer:
def test_list_and_serve(self, tmp_path, monkeypatch):
import llm.api as api
import llm.convert as conv
from llm.pdfs import PdfFile
monkeypatch.setattr(
conv, "available", lambda: False
) # the host may have soffice
root = tmp_path / "root"
(root / "K").mkdir(parents=True)
pdf = root / "K" / "a.pdf"
@@ -529,3 +534,98 @@ class TestPdfViewer:
assert "src.kind !== 'rule'" in chat
search = client.get("/ui/search").text
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search
class TestWordExactPages:
"""A Word attachment lists as renderer 'pdf' with an as=pdf url when
LibreOffice is in the image, and /pdf/<key>/file?as=pdf serves the
cached conversion; without the converter it falls back to docx-preview."""
def _setup(self, tmp_path, monkeypatch, available):
import llm.api as api
import llm.convert as conv
from llm.pdfs import PdfFile
root = tmp_path / "root"
(root / "K").mkdir(parents=True)
docx = root / "K" / "letter.docx"
docx.write_bytes(b"PK docx")
files = [PdfFile("letter.docx", docx, 7, "comment")]
monkeypatch.setattr(api, "_pdf_files", lambda key: files)
monkeypatch.setattr(
api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root)
)
monkeypatch.setattr(conv, "available", lambda: available)
class _Store:
_storage = tmp_path / "bibstorage"
def get(self, key):
class _I:
title = "t"
return _I()
def close(self):
pass
monkeypatch.setattr("conf.connect.bib", lambda: _Store())
return docx
def test_with_converter(self, tmp_path, monkeypatch):
import llm.convert as conv
docx = self._setup(tmp_path, monkeypatch, True)
converted = tmp_path / "letter.pdf"
converted.write_bytes(b"%PDF-1.7 word-exact")
calls = []
def fake_to_pdf(src, *, cache_dir):
calls.append((src, cache_dir))
return converted
monkeypatch.setattr(conv, "to_pdf", fake_to_pdf)
f = client.get("/pdf/ABCD1234").json()["files"][0]
assert (
f["renderer"] == "pdf"
and f["converted"] is True
and f["renderable"] is True
)
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx&as=pdf"
assert f["download_url"] == "/pdf/ABCD1234/file?name=letter.docx"
r = client.get(f["url"])
assert r.status_code == 200 and r.headers["content-type"] == "application/pdf"
assert r.content == b"%PDF-1.7 word-exact" and r.headers[
"content-disposition"
].startswith("inline")
assert (
calls
and calls[0][0] == docx
and str(calls[0][1]).endswith(".state/llm/pdfcache")
)
# the original is still a download
r = client.get(f["download_url"])
assert r.status_code == 200 and r.headers["content-disposition"].startswith(
"attachment"
)
def test_conversion_failure_is_a_502(self, tmp_path, monkeypatch):
import llm.convert as conv
self._setup(tmp_path, monkeypatch, True)
def boom(src, *, cache_dir):
raise conv.ConversionError("letter.docx: LibreOffice failed (1)")
monkeypatch.setattr(conv, "to_pdf", boom)
r = client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf")
assert r.status_code == 502 and "LibreOffice failed" in r.json()["detail"]
def test_without_converter_falls_back_to_docx_preview(self, tmp_path, monkeypatch):
self._setup(tmp_path, monkeypatch, False)
f = client.get("/pdf/ABCD1234").json()["files"][0]
assert f["renderer"] == "docx" and f["converted"] is False
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx"
assert (
client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf").status_code == 404
)

84
tests/llm/test_convert.py Normal file
View File

@@ -0,0 +1,84 @@
"""llm.convert — Word → PDF through LibreOffice, cached (viewer)."""
from __future__ import annotations
import subprocess
from pathlib import Path
import pytest
import llm.convert as conv
from llm.convert import ConversionError, cache_key, to_pdf
def _fake_runner(behaviour="ok"):
calls = []
def run(cmd, capture_output, text, timeout, env):
calls.append(cmd)
outdir = Path(cmd[cmd.index("--outdir") + 1])
src = Path(cmd[-1])
if behaviour == "ok":
(outdir / (src.stem + ".pdf")).write_bytes(b"%PDF-1.7 converted")
return subprocess.CompletedProcess(cmd, 0, "convert ok", "")
if behaviour == "empty":
return subprocess.CompletedProcess(cmd, 0, "", "")
if behaviour == "fail":
return subprocess.CompletedProcess(
cmd, 1, "", "Error: source file could not be loaded"
)
raise subprocess.TimeoutExpired(cmd, timeout)
run.calls = calls # type: ignore[attr-defined]
return run
@pytest.fixture
def docx(tmp_path):
p = tmp_path / "letter.docx"
p.write_bytes(b"PK docx bytes")
return p
def test_converts_once_then_serves_the_cache(tmp_path, docx, monkeypatch):
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
run = _fake_runner()
cache = tmp_path / "cache"
out = to_pdf(docx, cache_dir=cache, runner=run)
assert out.read_bytes().startswith(b"%PDF") and out.parent == cache
assert out.name == cache_key(docx) + ".pdf"
cmd = run.calls[0]
assert (
cmd[0] == "/usr/bin/soffice" and "--headless" in cmd and "--convert-to" in cmd
)
assert cmd[-1] == str(docx) and any(
a.startswith("-env:UserInstallation=file://") for a in cmd
)
assert to_pdf(docx, cache_dir=cache, runner=run) == out and len(run.calls) == 1
# a changed source (new size) is a cache miss
docx.write_bytes(b"PK docx bytes v2!")
assert to_pdf(docx, cache_dir=cache, runner=run) != out and len(run.calls) == 2
@pytest.mark.parametrize(
"behaviour, msg",
[
("empty", "LibreOffice failed"),
("fail", "could not be loaded"),
("timeout", "timed out"),
],
)
def test_failures_raise(tmp_path, docx, monkeypatch, behaviour, msg):
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
with pytest.raises(ConversionError, match=msg):
to_pdf(docx, cache_dir=tmp_path / "c", runner=_fake_runner(behaviour))
assert not list((tmp_path / "c").glob("*.pdf"))
def test_missing_soffice_and_missing_source(tmp_path, docx, monkeypatch):
monkeypatch.setattr(conv, "soffice", lambda: None)
assert conv.available() is False
with pytest.raises(ConversionError, match="not installed"):
to_pdf(docx, cache_dir=tmp_path / "c")
with pytest.raises(ConversionError, match="not on file"):
to_pdf(tmp_path / "nope.docx", cache_dir=tmp_path / "c")