feat(llm): Word-exact pages — LibreOffice in the llm image converts .docx to PDF on demand (cached) for the document viewer
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s
llm.convert.to_pdf runs headless Writer (soffice --convert-to pdf:writer_pdf_Export, its own profile dir, one at a time) and caches the result by the source's path/size/mtime under .state/llm/pdfcache (compose mounts it read-write). The image adds libreoffice-writer-nogui plus Carlito/Caladea/Liberation/DejaVu so Calibri/Cambria/Times/Arial letters paginate like Word. /pdf/<key> now returns, per file, the url the viewer should load and the download url of the original; a Word file lists as renderer pdf with url …&as=pdf when the converter is present, so the PDF.js path (real pages, text layer, find, page links) shows it; /pdf/<key>/file ?as=pdf serves the conversion inline (502 on failure, 404 without the converter). Browser-side docx-preview remains the fallback.
This commit is contained in:
@@ -692,6 +692,7 @@ services:
|
||||
- ./data:/app/data
|
||||
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
|
||||
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
|
||||
- ./.state/llm/pdfcache:/app/.state/llm/pdfcache # LibreOffice Word→PDF cache for the viewer
|
||||
environment:
|
||||
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
|
||||
- LLM_PG_HOST=postgres
|
||||
|
||||
@@ -9,6 +9,14 @@ ARG PYPI_INDEX_URL=""
|
||||
# Patch base image CVEs + install curl for the healthcheck.
|
||||
RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends curl && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Word → PDF for the document viewer (llm.convert): headless Writer plus
|
||||
# metric-compatible fonts so Calibri/Cambria/Times/Arial letters paginate
|
||||
# like Word (Carlito, Caladea, Liberation) — no X, no Java.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libreoffice-writer-nogui fonts-crosextra-carlito fonts-crosextra-caladea \
|
||||
fonts-liberation2 fonts-dejavu-core \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY pyproject.toml uv.lock README.md ./
|
||||
COPY src/ src/
|
||||
|
||||
|
||||
@@ -17,8 +17,9 @@ from contextlib import asynccontextmanager
|
||||
from importlib import resources
|
||||
from pathlib import Path
|
||||
from typing import AsyncIterator, Iterator
|
||||
from urllib.parse import quote
|
||||
|
||||
from fastapi import FastAPI, Header, HTTPException
|
||||
from fastapi import FastAPI, Header, HTTPException, Query
|
||||
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
@@ -186,28 +187,42 @@ def pdf_list(key: str) -> dict:
|
||||
store.close()
|
||||
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
|
||||
title = ""
|
||||
return {
|
||||
"key": key,
|
||||
"title": title,
|
||||
"files": [
|
||||
from llm import convert
|
||||
|
||||
convertible = convert.available()
|
||||
base = f"/pdf/{key}/file?name="
|
||||
out = []
|
||||
for f in files:
|
||||
ext = Path(f.name).suffix.lower()
|
||||
converted = convertible and ext in convert.CONVERTIBLE
|
||||
out.append(
|
||||
{
|
||||
"name": f.name,
|
||||
"size": f.size,
|
||||
"source": f.source,
|
||||
"renderable": f.renderable,
|
||||
"renderer": f.renderer,
|
||||
"renderable": f.renderable or converted,
|
||||
# Word files render as the PDF LibreOffice lays out (Word-exact
|
||||
# pages) when the converter is in the image; docx-preview in
|
||||
# the browser is the fallback
|
||||
"renderer": "pdf" if converted else f.renderer,
|
||||
"converted": converted,
|
||||
"media_type": f.media_type,
|
||||
"url": base + quote(f.name) + ("&as=pdf" if converted else ""),
|
||||
"download_url": base + quote(f.name),
|
||||
}
|
||||
for f in files
|
||||
],
|
||||
}
|
||||
)
|
||||
return {"key": key, "title": title, "files": out}
|
||||
|
||||
|
||||
@app.get("/pdf/{key}/file")
|
||||
def pdf_file(key: str, name: str = "") -> FileResponse:
|
||||
def pdf_file(
|
||||
key: str, name: str = "", as_: str = Query("", alias="as")
|
||||
) -> FileResponse:
|
||||
"""The attachment bytes — a PDF inline with range requests (PDF.js
|
||||
streams pages from a large file instead of waiting for all of it),
|
||||
anything else as a download."""
|
||||
anything else as a download. ``as=pdf`` on a Word file returns the
|
||||
LibreOffice rendering (cached under .state/llm/pdfcache); 502 when
|
||||
the conversion fails, 404 when the converter is not in the image."""
|
||||
from llm.pdfs import resolve
|
||||
|
||||
if not _KEY_RE.match(key):
|
||||
@@ -224,6 +239,31 @@ def pdf_file(key: str, name: str = "") -> FileResponse:
|
||||
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
|
||||
if chosen is None:
|
||||
raise HTTPException(404, "no such attachment on file for this item")
|
||||
if as_ == "pdf" and chosen.renderer != "pdf":
|
||||
from conf import ROOT
|
||||
from llm import convert
|
||||
|
||||
if (
|
||||
Path(chosen.name).suffix.lower() not in convert.CONVERTIBLE
|
||||
or not convert.available()
|
||||
):
|
||||
raise HTTPException(404, "no PDF rendering available for this file")
|
||||
try:
|
||||
rendered = convert.to_pdf(
|
||||
chosen.path, cache_dir=Path(ROOT) / ".state" / "llm" / "pdfcache"
|
||||
)
|
||||
except convert.ConversionError as exc:
|
||||
raise HTTPException(502, str(exc)) from exc
|
||||
return FileResponse(
|
||||
rendered,
|
||||
media_type="application/pdf",
|
||||
filename=Path(chosen.name).stem + ".pdf",
|
||||
content_disposition_type="inline",
|
||||
headers={
|
||||
"Accept-Ranges": "bytes",
|
||||
"Cache-Control": "private, max-age=3600",
|
||||
},
|
||||
)
|
||||
return FileResponse(
|
||||
chosen.path,
|
||||
media_type=chosen.media_type,
|
||||
|
||||
104
src/llm/convert.py
Normal file
104
src/llm/convert.py
Normal file
@@ -0,0 +1,104 @@
|
||||
"""Word → PDF with LibreOffice, for Word-exact pages in the viewer.
|
||||
|
||||
``soffice --headless --convert-to pdf`` lays the document out the way
|
||||
Writer does (real pagination, styles, metric-compatible fonts for
|
||||
Calibri/Cambria/Times/Arial via Carlito/Caladea/Liberation), which the
|
||||
browser-side docx-preview cannot. Conversions are cached by the source
|
||||
file's identity (path, size, mtime) under ``cache_dir`` so a document is
|
||||
converted once, and serialised behind one lock because a shared
|
||||
LibreOffice profile cannot run two instances at a time.
|
||||
|
||||
available() → soffice is on PATH
|
||||
to_pdf(src, cache_dir=…) → cached PDF path (converts on a miss)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import threading
|
||||
from pathlib import Path
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
_TIMEOUT = 180
|
||||
CONVERTIBLE = (".docx", ".doc", ".rtf", ".odt")
|
||||
|
||||
|
||||
class ConversionError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def soffice() -> str | None:
|
||||
return shutil.which("soffice") or shutil.which("libreoffice")
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
return soffice() is not None
|
||||
|
||||
|
||||
def cache_key(src: Path) -> str:
|
||||
st = src.stat()
|
||||
raw = f"{src.resolve()}|{st.st_size}|{st.st_mtime_ns}"
|
||||
return hashlib.sha1(raw.encode()).hexdigest() # noqa: S324 — cache identity
|
||||
|
||||
|
||||
def to_pdf(src: Path, *, cache_dir: Path, runner=None) -> Path:
|
||||
"""The PDF rendering of *src*, converting with LibreOffice when the
|
||||
cache has no current copy. *runner* (tests) replaces ``subprocess.run``.
|
||||
Raises ``ConversionError`` when soffice is missing, times out, fails or
|
||||
produces nothing."""
|
||||
src = Path(src)
|
||||
if not src.is_file():
|
||||
raise ConversionError(f"{src.name}: not on file")
|
||||
cache_dir = Path(cache_dir)
|
||||
cache_dir.mkdir(parents=True, exist_ok=True)
|
||||
out = cache_dir / f"{cache_key(src)}.pdf"
|
||||
if out.is_file() and out.stat().st_size > 0:
|
||||
return out
|
||||
exe = soffice()
|
||||
if exe is None:
|
||||
raise ConversionError("LibreOffice (soffice) is not installed in this image")
|
||||
run = runner or subprocess.run
|
||||
profile = cache_dir / ".lo-profile"
|
||||
profile.mkdir(exist_ok=True)
|
||||
with _LOCK, tempfile.TemporaryDirectory(prefix="lo-", dir=cache_dir) as tmp:
|
||||
cmd = [
|
||||
exe,
|
||||
f"-env:UserInstallation=file://{profile}",
|
||||
"--headless",
|
||||
"--norestore",
|
||||
"--nolockcheck",
|
||||
"--convert-to",
|
||||
"pdf:writer_pdf_Export",
|
||||
"--outdir",
|
||||
tmp,
|
||||
str(src),
|
||||
]
|
||||
env = {**os.environ, "HOME": str(profile)}
|
||||
try:
|
||||
proc = run(cmd, capture_output=True, text=True, timeout=_TIMEOUT, env=env)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise ConversionError(
|
||||
f"{src.name}: conversion timed out after {_TIMEOUT}s"
|
||||
) from exc
|
||||
produced = Path(tmp) / (src.stem + ".pdf")
|
||||
if (
|
||||
proc.returncode != 0
|
||||
or not produced.is_file()
|
||||
or produced.stat().st_size == 0
|
||||
):
|
||||
detail = (proc.stderr or proc.stdout or "").strip()[-300:]
|
||||
raise ConversionError(
|
||||
f"{src.name}: LibreOffice failed ({proc.returncode}) {detail}"
|
||||
)
|
||||
tmp_out = out.with_suffix(".pdf.part")
|
||||
shutil.move(str(produced), tmp_out)
|
||||
os.replace(tmp_out, out)
|
||||
log.info("converted %s → %s (%d bytes)", src.name, out.name, out.stat().st_size)
|
||||
return out
|
||||
@@ -214,8 +214,10 @@
|
||||
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
|
||||
|
||||
sel.value = current.name;
|
||||
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
|
||||
$('download').href = fileUrl;
|
||||
const rawUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
|
||||
const fileUrl = current.url || rawUrl; // a Word file's LibreOffice rendering when the image has the converter
|
||||
$('download').href = current.download_url || rawUrl; // the original bytes
|
||||
if (current.converted) $('download').textContent = 'download original';
|
||||
sel.addEventListener('change', () => {
|
||||
const p = new URLSearchParams(location.search);
|
||||
p.set('file', sel.value); p.delete('page');
|
||||
|
||||
@@ -468,8 +468,13 @@ class TestPdfViewer:
|
||||
|
||||
def test_list_and_serve(self, tmp_path, monkeypatch):
|
||||
import llm.api as api
|
||||
import llm.convert as conv
|
||||
from llm.pdfs import PdfFile
|
||||
|
||||
monkeypatch.setattr(
|
||||
conv, "available", lambda: False
|
||||
) # the host may have soffice
|
||||
|
||||
root = tmp_path / "root"
|
||||
(root / "K").mkdir(parents=True)
|
||||
pdf = root / "K" / "a.pdf"
|
||||
@@ -529,3 +534,98 @@ class TestPdfViewer:
|
||||
assert "src.kind !== 'rule'" in chat
|
||||
search = client.get("/ui/search").text
|
||||
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search
|
||||
|
||||
|
||||
class TestWordExactPages:
|
||||
"""A Word attachment lists as renderer 'pdf' with an as=pdf url when
|
||||
LibreOffice is in the image, and /pdf/<key>/file?as=pdf serves the
|
||||
cached conversion; without the converter it falls back to docx-preview."""
|
||||
|
||||
def _setup(self, tmp_path, monkeypatch, available):
|
||||
import llm.api as api
|
||||
import llm.convert as conv
|
||||
from llm.pdfs import PdfFile
|
||||
|
||||
root = tmp_path / "root"
|
||||
(root / "K").mkdir(parents=True)
|
||||
docx = root / "K" / "letter.docx"
|
||||
docx.write_bytes(b"PK docx")
|
||||
files = [PdfFile("letter.docx", docx, 7, "comment")]
|
||||
monkeypatch.setattr(api, "_pdf_files", lambda key: files)
|
||||
monkeypatch.setattr(
|
||||
api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root)
|
||||
)
|
||||
monkeypatch.setattr(conv, "available", lambda: available)
|
||||
|
||||
class _Store:
|
||||
_storage = tmp_path / "bibstorage"
|
||||
|
||||
def get(self, key):
|
||||
class _I:
|
||||
title = "t"
|
||||
|
||||
return _I()
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr("conf.connect.bib", lambda: _Store())
|
||||
return docx
|
||||
|
||||
def test_with_converter(self, tmp_path, monkeypatch):
|
||||
import llm.convert as conv
|
||||
|
||||
docx = self._setup(tmp_path, monkeypatch, True)
|
||||
converted = tmp_path / "letter.pdf"
|
||||
converted.write_bytes(b"%PDF-1.7 word-exact")
|
||||
calls = []
|
||||
|
||||
def fake_to_pdf(src, *, cache_dir):
|
||||
calls.append((src, cache_dir))
|
||||
return converted
|
||||
|
||||
monkeypatch.setattr(conv, "to_pdf", fake_to_pdf)
|
||||
f = client.get("/pdf/ABCD1234").json()["files"][0]
|
||||
assert (
|
||||
f["renderer"] == "pdf"
|
||||
and f["converted"] is True
|
||||
and f["renderable"] is True
|
||||
)
|
||||
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx&as=pdf"
|
||||
assert f["download_url"] == "/pdf/ABCD1234/file?name=letter.docx"
|
||||
r = client.get(f["url"])
|
||||
assert r.status_code == 200 and r.headers["content-type"] == "application/pdf"
|
||||
assert r.content == b"%PDF-1.7 word-exact" and r.headers[
|
||||
"content-disposition"
|
||||
].startswith("inline")
|
||||
assert (
|
||||
calls
|
||||
and calls[0][0] == docx
|
||||
and str(calls[0][1]).endswith(".state/llm/pdfcache")
|
||||
)
|
||||
# the original is still a download
|
||||
r = client.get(f["download_url"])
|
||||
assert r.status_code == 200 and r.headers["content-disposition"].startswith(
|
||||
"attachment"
|
||||
)
|
||||
|
||||
def test_conversion_failure_is_a_502(self, tmp_path, monkeypatch):
|
||||
import llm.convert as conv
|
||||
|
||||
self._setup(tmp_path, monkeypatch, True)
|
||||
|
||||
def boom(src, *, cache_dir):
|
||||
raise conv.ConversionError("letter.docx: LibreOffice failed (1)")
|
||||
|
||||
monkeypatch.setattr(conv, "to_pdf", boom)
|
||||
r = client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf")
|
||||
assert r.status_code == 502 and "LibreOffice failed" in r.json()["detail"]
|
||||
|
||||
def test_without_converter_falls_back_to_docx_preview(self, tmp_path, monkeypatch):
|
||||
self._setup(tmp_path, monkeypatch, False)
|
||||
f = client.get("/pdf/ABCD1234").json()["files"][0]
|
||||
assert f["renderer"] == "docx" and f["converted"] is False
|
||||
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx"
|
||||
assert (
|
||||
client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf").status_code == 404
|
||||
)
|
||||
|
||||
84
tests/llm/test_convert.py
Normal file
84
tests/llm/test_convert.py
Normal file
@@ -0,0 +1,84 @@
|
||||
"""llm.convert — Word → PDF through LibreOffice, cached (viewer)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import llm.convert as conv
|
||||
from llm.convert import ConversionError, cache_key, to_pdf
|
||||
|
||||
|
||||
def _fake_runner(behaviour="ok"):
|
||||
calls = []
|
||||
|
||||
def run(cmd, capture_output, text, timeout, env):
|
||||
calls.append(cmd)
|
||||
outdir = Path(cmd[cmd.index("--outdir") + 1])
|
||||
src = Path(cmd[-1])
|
||||
if behaviour == "ok":
|
||||
(outdir / (src.stem + ".pdf")).write_bytes(b"%PDF-1.7 converted")
|
||||
return subprocess.CompletedProcess(cmd, 0, "convert ok", "")
|
||||
if behaviour == "empty":
|
||||
return subprocess.CompletedProcess(cmd, 0, "", "")
|
||||
if behaviour == "fail":
|
||||
return subprocess.CompletedProcess(
|
||||
cmd, 1, "", "Error: source file could not be loaded"
|
||||
)
|
||||
raise subprocess.TimeoutExpired(cmd, timeout)
|
||||
|
||||
run.calls = calls # type: ignore[attr-defined]
|
||||
return run
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def docx(tmp_path):
|
||||
p = tmp_path / "letter.docx"
|
||||
p.write_bytes(b"PK docx bytes")
|
||||
return p
|
||||
|
||||
|
||||
def test_converts_once_then_serves_the_cache(tmp_path, docx, monkeypatch):
|
||||
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
|
||||
run = _fake_runner()
|
||||
cache = tmp_path / "cache"
|
||||
out = to_pdf(docx, cache_dir=cache, runner=run)
|
||||
assert out.read_bytes().startswith(b"%PDF") and out.parent == cache
|
||||
assert out.name == cache_key(docx) + ".pdf"
|
||||
cmd = run.calls[0]
|
||||
assert (
|
||||
cmd[0] == "/usr/bin/soffice" and "--headless" in cmd and "--convert-to" in cmd
|
||||
)
|
||||
assert cmd[-1] == str(docx) and any(
|
||||
a.startswith("-env:UserInstallation=file://") for a in cmd
|
||||
)
|
||||
assert to_pdf(docx, cache_dir=cache, runner=run) == out and len(run.calls) == 1
|
||||
# a changed source (new size) is a cache miss
|
||||
docx.write_bytes(b"PK docx bytes v2!")
|
||||
assert to_pdf(docx, cache_dir=cache, runner=run) != out and len(run.calls) == 2
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"behaviour, msg",
|
||||
[
|
||||
("empty", "LibreOffice failed"),
|
||||
("fail", "could not be loaded"),
|
||||
("timeout", "timed out"),
|
||||
],
|
||||
)
|
||||
def test_failures_raise(tmp_path, docx, monkeypatch, behaviour, msg):
|
||||
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
|
||||
with pytest.raises(ConversionError, match=msg):
|
||||
to_pdf(docx, cache_dir=tmp_path / "c", runner=_fake_runner(behaviour))
|
||||
assert not list((tmp_path / "c").glob("*.pdf"))
|
||||
|
||||
|
||||
def test_missing_soffice_and_missing_source(tmp_path, docx, monkeypatch):
|
||||
monkeypatch.setattr(conv, "soffice", lambda: None)
|
||||
assert conv.available() is False
|
||||
with pytest.raises(ConversionError, match="not installed"):
|
||||
to_pdf(docx, cache_dir=tmp_path / "c")
|
||||
with pytest.raises(ConversionError, match="not on file"):
|
||||
to_pdf(tmp_path / "nope.docx", cache_dir=tmp_path / "c")
|
||||
Reference in New Issue
Block a user