From 523be76fc68d1fd25ee3e8fc19b981916646d39c Mon Sep 17 00:00:00 2001 From: kert Date: Thu, 24 Sep 2026 19:12:14 -0400 Subject: [PATCH] =?UTF-8?q?feat(llm):=20Word-exact=20pages=20=E2=80=94=20L?= =?UTF-8?q?ibreOffice=20in=20the=20llm=20image=20converts=20.docx=20to=20P?= =?UTF-8?q?DF=20on=20demand=20(cached)=20for=20the=20document=20viewer?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit llm.convert.to_pdf runs headless Writer (soffice --convert-to pdf:writer_pdf_Export, its own profile dir, one at a time) and caches the result by the source's path/size/mtime under .state/llm/pdfcache (compose mounts it read-write). The image adds libreoffice-writer-nogui plus Carlito/Caladea/Liberation/DejaVu so Calibri/Cambria/Times/Arial letters paginate like Word. /pdf/ now returns, per file, the url the viewer should load and the download url of the original; a Word file lists as renderer pdf with url …&as=pdf when the converter is present, so the PDF.js path (real pages, text layer, find, page links) shows it; /pdf//file ?as=pdf serves the conversion inline (502 on failure, 404 without the converter). Browser-side docx-preview remains the fallback. --- compose.yml | 1 + infra/images/llm.Dockerfile | 8 +++ src/llm/api.py | 64 +++++++++++++++++----- src/llm/convert.py | 104 ++++++++++++++++++++++++++++++++++++ src/llm/web/pdf.html | 6 ++- tests/llm/test_api.py | 100 ++++++++++++++++++++++++++++++++++ tests/llm/test_convert.py | 84 +++++++++++++++++++++++++++++ 7 files changed, 353 insertions(+), 14 deletions(-) create mode 100644 src/llm/convert.py create mode 100644 tests/llm/test_convert.py diff --git a/compose.yml b/compose.yml index 7bc2547..b5ca748 100644 --- a/compose.yml +++ b/compose.yml @@ -692,6 +692,7 @@ services: - ./data:/app/data - ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish) - ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF) + - ./.state/llm/pdfcache:/app/.state/llm/pdfcache # LibreOffice Word→PDF cache for the viewer environment: - LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434} - LLM_PG_HOST=postgres diff --git a/infra/images/llm.Dockerfile b/infra/images/llm.Dockerfile index f074730..81c1b26 100644 --- a/infra/images/llm.Dockerfile +++ b/infra/images/llm.Dockerfile @@ -9,6 +9,14 @@ ARG PYPI_INDEX_URL="" # Patch base image CVEs + install curl for the healthcheck. RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends curl && rm -rf /var/lib/apt/lists/* +# Word → PDF for the document viewer (llm.convert): headless Writer plus +# metric-compatible fonts so Calibri/Cambria/Times/Arial letters paginate +# like Word (Carlito, Caladea, Liberation) — no X, no Java. +RUN apt-get update && apt-get install -y --no-install-recommends \ + libreoffice-writer-nogui fonts-crosextra-carlito fonts-crosextra-caladea \ + fonts-liberation2 fonts-dejavu-core \ + && rm -rf /var/lib/apt/lists/* + COPY pyproject.toml uv.lock README.md ./ COPY src/ src/ diff --git a/src/llm/api.py b/src/llm/api.py index 5621ee0..614ae5e 100644 --- a/src/llm/api.py +++ b/src/llm/api.py @@ -17,8 +17,9 @@ from contextlib import asynccontextmanager from importlib import resources from pathlib import Path from typing import AsyncIterator, Iterator +from urllib.parse import quote -from fastapi import FastAPI, Header, HTTPException +from fastapi import FastAPI, Header, HTTPException, Query from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse from pydantic import BaseModel @@ -186,28 +187,42 @@ def pdf_list(key: str) -> dict: store.close() except Exception: # noqa: BLE001 — a Zotero-only key has no bib row title = "" - return { - "key": key, - "title": title, - "files": [ + from llm import convert + + convertible = convert.available() + base = f"/pdf/{key}/file?name=" + out = [] + for f in files: + ext = Path(f.name).suffix.lower() + converted = convertible and ext in convert.CONVERTIBLE + out.append( { "name": f.name, "size": f.size, "source": f.source, - "renderable": f.renderable, - "renderer": f.renderer, + "renderable": f.renderable or converted, + # Word files render as the PDF LibreOffice lays out (Word-exact + # pages) when the converter is in the image; docx-preview in + # the browser is the fallback + "renderer": "pdf" if converted else f.renderer, + "converted": converted, "media_type": f.media_type, + "url": base + quote(f.name) + ("&as=pdf" if converted else ""), + "download_url": base + quote(f.name), } - for f in files - ], - } + ) + return {"key": key, "title": title, "files": out} @app.get("/pdf/{key}/file") -def pdf_file(key: str, name: str = "") -> FileResponse: +def pdf_file( + key: str, name: str = "", as_: str = Query("", alias="as") +) -> FileResponse: """The attachment bytes — a PDF inline with range requests (PDF.js streams pages from a large file instead of waiting for all of it), - anything else as a download.""" + anything else as a download. ``as=pdf`` on a Word file returns the + LibreOffice rendering (cached under .state/llm/pdfcache); 502 when + the conversion fails, 404 when the converter is not in the image.""" from llm.pdfs import resolve if not _KEY_RE.match(key): @@ -224,6 +239,31 @@ def pdf_file(key: str, name: str = "") -> FileResponse: chosen = resolve(files, name, roots=(bib_root, zstore, croot)) if chosen is None: raise HTTPException(404, "no such attachment on file for this item") + if as_ == "pdf" and chosen.renderer != "pdf": + from conf import ROOT + from llm import convert + + if ( + Path(chosen.name).suffix.lower() not in convert.CONVERTIBLE + or not convert.available() + ): + raise HTTPException(404, "no PDF rendering available for this file") + try: + rendered = convert.to_pdf( + chosen.path, cache_dir=Path(ROOT) / ".state" / "llm" / "pdfcache" + ) + except convert.ConversionError as exc: + raise HTTPException(502, str(exc)) from exc + return FileResponse( + rendered, + media_type="application/pdf", + filename=Path(chosen.name).stem + ".pdf", + content_disposition_type="inline", + headers={ + "Accept-Ranges": "bytes", + "Cache-Control": "private, max-age=3600", + }, + ) return FileResponse( chosen.path, media_type=chosen.media_type, diff --git a/src/llm/convert.py b/src/llm/convert.py new file mode 100644 index 0000000..1ecef86 --- /dev/null +++ b/src/llm/convert.py @@ -0,0 +1,104 @@ +"""Word → PDF with LibreOffice, for Word-exact pages in the viewer. + +``soffice --headless --convert-to pdf`` lays the document out the way +Writer does (real pagination, styles, metric-compatible fonts for +Calibri/Cambria/Times/Arial via Carlito/Caladea/Liberation), which the +browser-side docx-preview cannot. Conversions are cached by the source +file's identity (path, size, mtime) under ``cache_dir`` so a document is +converted once, and serialised behind one lock because a shared +LibreOffice profile cannot run two instances at a time. + + available() → soffice is on PATH + to_pdf(src, cache_dir=…) → cached PDF path (converts on a miss) +""" + +from __future__ import annotations + +import hashlib +import logging +import os +import shutil +import subprocess +import tempfile +import threading +from pathlib import Path + +log = logging.getLogger(__name__) + +_LOCK = threading.Lock() +_TIMEOUT = 180 +CONVERTIBLE = (".docx", ".doc", ".rtf", ".odt") + + +class ConversionError(RuntimeError): + pass + + +def soffice() -> str | None: + return shutil.which("soffice") or shutil.which("libreoffice") + + +def available() -> bool: + return soffice() is not None + + +def cache_key(src: Path) -> str: + st = src.stat() + raw = f"{src.resolve()}|{st.st_size}|{st.st_mtime_ns}" + return hashlib.sha1(raw.encode()).hexdigest() # noqa: S324 — cache identity + + +def to_pdf(src: Path, *, cache_dir: Path, runner=None) -> Path: + """The PDF rendering of *src*, converting with LibreOffice when the + cache has no current copy. *runner* (tests) replaces ``subprocess.run``. + Raises ``ConversionError`` when soffice is missing, times out, fails or + produces nothing.""" + src = Path(src) + if not src.is_file(): + raise ConversionError(f"{src.name}: not on file") + cache_dir = Path(cache_dir) + cache_dir.mkdir(parents=True, exist_ok=True) + out = cache_dir / f"{cache_key(src)}.pdf" + if out.is_file() and out.stat().st_size > 0: + return out + exe = soffice() + if exe is None: + raise ConversionError("LibreOffice (soffice) is not installed in this image") + run = runner or subprocess.run + profile = cache_dir / ".lo-profile" + profile.mkdir(exist_ok=True) + with _LOCK, tempfile.TemporaryDirectory(prefix="lo-", dir=cache_dir) as tmp: + cmd = [ + exe, + f"-env:UserInstallation=file://{profile}", + "--headless", + "--norestore", + "--nolockcheck", + "--convert-to", + "pdf:writer_pdf_Export", + "--outdir", + tmp, + str(src), + ] + env = {**os.environ, "HOME": str(profile)} + try: + proc = run(cmd, capture_output=True, text=True, timeout=_TIMEOUT, env=env) + except subprocess.TimeoutExpired as exc: + raise ConversionError( + f"{src.name}: conversion timed out after {_TIMEOUT}s" + ) from exc + produced = Path(tmp) / (src.stem + ".pdf") + if ( + proc.returncode != 0 + or not produced.is_file() + or produced.stat().st_size == 0 + ): + detail = (proc.stderr or proc.stdout or "").strip()[-300:] + raise ConversionError( + f"{src.name}: LibreOffice failed ({proc.returncode}) {detail}" + ) + tmp_out = out.with_suffix(".pdf.part") + shutil.move(str(produced), tmp_out) + os.replace(tmp_out, out) + log.info("converted %s → %s (%d bytes)", src.name, out.name, out.stat().st_size) + return out diff --git a/src/llm/web/pdf.html b/src/llm/web/pdf.html index 77c2f07..b378f5f 100644 --- a/src/llm/web/pdf.html +++ b/src/llm/web/pdf.html @@ -214,8 +214,10 @@ const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0]; sel.value = current.name; - const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name); - $('download').href = fileUrl; + const rawUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name); + const fileUrl = current.url || rawUrl; // a Word file's LibreOffice rendering when the image has the converter + $('download').href = current.download_url || rawUrl; // the original bytes + if (current.converted) $('download').textContent = 'download original'; sel.addEventListener('change', () => { const p = new URLSearchParams(location.search); p.set('file', sel.value); p.delete('page'); diff --git a/tests/llm/test_api.py b/tests/llm/test_api.py index 01358ce..a72b4fa 100644 --- a/tests/llm/test_api.py +++ b/tests/llm/test_api.py @@ -468,8 +468,13 @@ class TestPdfViewer: def test_list_and_serve(self, tmp_path, monkeypatch): import llm.api as api + import llm.convert as conv from llm.pdfs import PdfFile + monkeypatch.setattr( + conv, "available", lambda: False + ) # the host may have soffice + root = tmp_path / "root" (root / "K").mkdir(parents=True) pdf = root / "K" / "a.pdf" @@ -529,3 +534,98 @@ class TestPdfViewer: assert "src.kind !== 'rule'" in chat search = client.get("/ui/search").text assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search + + +class TestWordExactPages: + """A Word attachment lists as renderer 'pdf' with an as=pdf url when + LibreOffice is in the image, and /pdf//file?as=pdf serves the + cached conversion; without the converter it falls back to docx-preview.""" + + def _setup(self, tmp_path, monkeypatch, available): + import llm.api as api + import llm.convert as conv + from llm.pdfs import PdfFile + + root = tmp_path / "root" + (root / "K").mkdir(parents=True) + docx = root / "K" / "letter.docx" + docx.write_bytes(b"PK docx") + files = [PdfFile("letter.docx", docx, 7, "comment")] + monkeypatch.setattr(api, "_pdf_files", lambda key: files) + monkeypatch.setattr( + api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root) + ) + monkeypatch.setattr(conv, "available", lambda: available) + + class _Store: + _storage = tmp_path / "bibstorage" + + def get(self, key): + class _I: + title = "t" + + return _I() + + def close(self): + pass + + monkeypatch.setattr("conf.connect.bib", lambda: _Store()) + return docx + + def test_with_converter(self, tmp_path, monkeypatch): + import llm.convert as conv + + docx = self._setup(tmp_path, monkeypatch, True) + converted = tmp_path / "letter.pdf" + converted.write_bytes(b"%PDF-1.7 word-exact") + calls = [] + + def fake_to_pdf(src, *, cache_dir): + calls.append((src, cache_dir)) + return converted + + monkeypatch.setattr(conv, "to_pdf", fake_to_pdf) + f = client.get("/pdf/ABCD1234").json()["files"][0] + assert ( + f["renderer"] == "pdf" + and f["converted"] is True + and f["renderable"] is True + ) + assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx&as=pdf" + assert f["download_url"] == "/pdf/ABCD1234/file?name=letter.docx" + r = client.get(f["url"]) + assert r.status_code == 200 and r.headers["content-type"] == "application/pdf" + assert r.content == b"%PDF-1.7 word-exact" and r.headers[ + "content-disposition" + ].startswith("inline") + assert ( + calls + and calls[0][0] == docx + and str(calls[0][1]).endswith(".state/llm/pdfcache") + ) + # the original is still a download + r = client.get(f["download_url"]) + assert r.status_code == 200 and r.headers["content-disposition"].startswith( + "attachment" + ) + + def test_conversion_failure_is_a_502(self, tmp_path, monkeypatch): + import llm.convert as conv + + self._setup(tmp_path, monkeypatch, True) + + def boom(src, *, cache_dir): + raise conv.ConversionError("letter.docx: LibreOffice failed (1)") + + monkeypatch.setattr(conv, "to_pdf", boom) + r = client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf") + assert r.status_code == 502 and "LibreOffice failed" in r.json()["detail"] + + def test_without_converter_falls_back_to_docx_preview(self, tmp_path, monkeypatch): + self._setup(tmp_path, monkeypatch, False) + f = client.get("/pdf/ABCD1234").json()["files"][0] + assert f["renderer"] == "docx" and f["converted"] is False + assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx" + assert ( + client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf").status_code == 404 + ) diff --git a/tests/llm/test_convert.py b/tests/llm/test_convert.py new file mode 100644 index 0000000..f92d0b9 --- /dev/null +++ b/tests/llm/test_convert.py @@ -0,0 +1,84 @@ +"""llm.convert — Word → PDF through LibreOffice, cached (viewer).""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + +import pytest + +import llm.convert as conv +from llm.convert import ConversionError, cache_key, to_pdf + + +def _fake_runner(behaviour="ok"): + calls = [] + + def run(cmd, capture_output, text, timeout, env): + calls.append(cmd) + outdir = Path(cmd[cmd.index("--outdir") + 1]) + src = Path(cmd[-1]) + if behaviour == "ok": + (outdir / (src.stem + ".pdf")).write_bytes(b"%PDF-1.7 converted") + return subprocess.CompletedProcess(cmd, 0, "convert ok", "") + if behaviour == "empty": + return subprocess.CompletedProcess(cmd, 0, "", "") + if behaviour == "fail": + return subprocess.CompletedProcess( + cmd, 1, "", "Error: source file could not be loaded" + ) + raise subprocess.TimeoutExpired(cmd, timeout) + + run.calls = calls # type: ignore[attr-defined] + return run + + +@pytest.fixture +def docx(tmp_path): + p = tmp_path / "letter.docx" + p.write_bytes(b"PK docx bytes") + return p + + +def test_converts_once_then_serves_the_cache(tmp_path, docx, monkeypatch): + monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice") + run = _fake_runner() + cache = tmp_path / "cache" + out = to_pdf(docx, cache_dir=cache, runner=run) + assert out.read_bytes().startswith(b"%PDF") and out.parent == cache + assert out.name == cache_key(docx) + ".pdf" + cmd = run.calls[0] + assert ( + cmd[0] == "/usr/bin/soffice" and "--headless" in cmd and "--convert-to" in cmd + ) + assert cmd[-1] == str(docx) and any( + a.startswith("-env:UserInstallation=file://") for a in cmd + ) + assert to_pdf(docx, cache_dir=cache, runner=run) == out and len(run.calls) == 1 + # a changed source (new size) is a cache miss + docx.write_bytes(b"PK docx bytes v2!") + assert to_pdf(docx, cache_dir=cache, runner=run) != out and len(run.calls) == 2 + + +@pytest.mark.parametrize( + "behaviour, msg", + [ + ("empty", "LibreOffice failed"), + ("fail", "could not be loaded"), + ("timeout", "timed out"), + ], +) +def test_failures_raise(tmp_path, docx, monkeypatch, behaviour, msg): + monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice") + with pytest.raises(ConversionError, match=msg): + to_pdf(docx, cache_dir=tmp_path / "c", runner=_fake_runner(behaviour)) + assert not list((tmp_path / "c").glob("*.pdf")) + + +def test_missing_soffice_and_missing_source(tmp_path, docx, monkeypatch): + monkeypatch.setattr(conv, "soffice", lambda: None) + assert conv.available() is False + with pytest.raises(ConversionError, match="not installed"): + to_pdf(docx, cache_dir=tmp_path / "c") + with pytest.raises(ConversionError, match="not on file"): + to_pdf(tmp_path / "nope.docx", cache_dir=tmp_path / "c")