Files
stack/tests/llm/test_convert.py
kert 523be76fc6
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s
feat(llm): Word-exact pages — LibreOffice in the llm image converts .docx to PDF on demand (cached) for the document viewer
llm.convert.to_pdf runs headless Writer (soffice --convert-to
pdf:writer_pdf_Export, its own profile dir, one at a time) and caches
the result by the source's path/size/mtime under .state/llm/pdfcache
(compose mounts it read-write). The image adds
libreoffice-writer-nogui plus Carlito/Caladea/Liberation/DejaVu so
Calibri/Cambria/Times/Arial letters paginate like Word.

/pdf/<key> now returns, per file, the url the viewer should load and
the download url of the original; a Word file lists as renderer pdf
with url …&as=pdf when the converter is present, so the PDF.js path
(real pages, text layer, find, page links) shows it; /pdf/<key>/file
?as=pdf serves the conversion inline (502 on failure, 404 without the
converter). Browser-side docx-preview remains the fallback.
2026-09-24 19:12:14 -04:00

85 lines
2.9 KiB
Python

"""llm.convert — Word → PDF through LibreOffice, cached (viewer)."""
from __future__ import annotations
import subprocess
from pathlib import Path
import pytest
import llm.convert as conv
from llm.convert import ConversionError, cache_key, to_pdf
def _fake_runner(behaviour="ok"):
calls = []
def run(cmd, capture_output, text, timeout, env):
calls.append(cmd)
outdir = Path(cmd[cmd.index("--outdir") + 1])
src = Path(cmd[-1])
if behaviour == "ok":
(outdir / (src.stem + ".pdf")).write_bytes(b"%PDF-1.7 converted")
return subprocess.CompletedProcess(cmd, 0, "convert ok", "")
if behaviour == "empty":
return subprocess.CompletedProcess(cmd, 0, "", "")
if behaviour == "fail":
return subprocess.CompletedProcess(
cmd, 1, "", "Error: source file could not be loaded"
)
raise subprocess.TimeoutExpired(cmd, timeout)
run.calls = calls # type: ignore[attr-defined]
return run
@pytest.fixture
def docx(tmp_path):
p = tmp_path / "letter.docx"
p.write_bytes(b"PK docx bytes")
return p
def test_converts_once_then_serves_the_cache(tmp_path, docx, monkeypatch):
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
run = _fake_runner()
cache = tmp_path / "cache"
out = to_pdf(docx, cache_dir=cache, runner=run)
assert out.read_bytes().startswith(b"%PDF") and out.parent == cache
assert out.name == cache_key(docx) + ".pdf"
cmd = run.calls[0]
assert (
cmd[0] == "/usr/bin/soffice" and "--headless" in cmd and "--convert-to" in cmd
)
assert cmd[-1] == str(docx) and any(
a.startswith("-env:UserInstallation=file://") for a in cmd
)
assert to_pdf(docx, cache_dir=cache, runner=run) == out and len(run.calls) == 1
# a changed source (new size) is a cache miss
docx.write_bytes(b"PK docx bytes v2!")
assert to_pdf(docx, cache_dir=cache, runner=run) != out and len(run.calls) == 2
@pytest.mark.parametrize(
"behaviour, msg",
[
("empty", "LibreOffice failed"),
("fail", "could not be loaded"),
("timeout", "timed out"),
],
)
def test_failures_raise(tmp_path, docx, monkeypatch, behaviour, msg):
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
with pytest.raises(ConversionError, match=msg):
to_pdf(docx, cache_dir=tmp_path / "c", runner=_fake_runner(behaviour))
assert not list((tmp_path / "c").glob("*.pdf"))
def test_missing_soffice_and_missing_source(tmp_path, docx, monkeypatch):
monkeypatch.setattr(conv, "soffice", lambda: None)
assert conv.available() is False
with pytest.raises(ConversionError, match="not installed"):
to_pdf(docx, cache_dir=tmp_path / "c")
with pytest.raises(ConversionError, match="not on file"):
to_pdf(tmp_path / "nope.docx", cache_dir=tmp_path / "c")