feat(llm): Word-exact pages — LibreOffice in the llm image converts .docx to PDF on demand (cached) for the document viewer
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s
Some checks failed
CI / notebooks-smoke (push) Successful in 1m31s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Failing after 2m23s
Deploy / mc (push) Has been skipped
Infra CI / notebooks (push) Successful in 54s
Deploy / llm (push) Successful in 1m32s
Infra CI / llm (push) Successful in 14s
Infra CI / mc (push) Failing after 34s
Deploy / report (push) Successful in 14s
Deploy / api (push) Has been skipped
Infra CI / zotero (push) Successful in 19s
Infra CI / docs (push) Successful in 18s
Infra CI / api (push) Successful in 1m6s
CI / lint (push) Successful in 36s
llm.convert.to_pdf runs headless Writer (soffice --convert-to pdf:writer_pdf_Export, its own profile dir, one at a time) and caches the result by the source's path/size/mtime under .state/llm/pdfcache (compose mounts it read-write). The image adds libreoffice-writer-nogui plus Carlito/Caladea/Liberation/DejaVu so Calibri/Cambria/Times/Arial letters paginate like Word. /pdf/<key> now returns, per file, the url the viewer should load and the download url of the original; a Word file lists as renderer pdf with url …&as=pdf when the converter is present, so the PDF.js path (real pages, text layer, find, page links) shows it; /pdf/<key>/file ?as=pdf serves the conversion inline (502 on failure, 404 without the converter). Browser-side docx-preview remains the fallback.
This commit is contained in:
@@ -692,6 +692,7 @@ services:
|
|||||||
- ./data:/app/data
|
- ./data:/app/data
|
||||||
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
|
- ./data/replica:/app/data/replica:ro # DuckDB read replica only (directory mount survives replica re-publish)
|
||||||
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
|
- ./.state/comments:/app/.state/comments:ro # comment attachments for the /ui/pdf viewer (#PDF)
|
||||||
|
- ./.state/llm/pdfcache:/app/.state/llm/pdfcache # LibreOffice Word→PDF cache for the viewer
|
||||||
environment:
|
environment:
|
||||||
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
|
- LLM_OLLAMA_HOSTS=${LLM_OLLAMA_HOSTS_IN_CONTAINER:-http://ollama:11434}
|
||||||
- LLM_PG_HOST=postgres
|
- LLM_PG_HOST=postgres
|
||||||
|
|||||||
@@ -9,6 +9,14 @@ ARG PYPI_INDEX_URL=""
|
|||||||
# Patch base image CVEs + install curl for the healthcheck.
|
# Patch base image CVEs + install curl for the healthcheck.
|
||||||
RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends curl && rm -rf /var/lib/apt/lists/*
|
RUN apt-get update && apt-get upgrade -y && apt-get install -y --no-install-recommends curl && rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
# Word → PDF for the document viewer (llm.convert): headless Writer plus
|
||||||
|
# metric-compatible fonts so Calibri/Cambria/Times/Arial letters paginate
|
||||||
|
# like Word (Carlito, Caladea, Liberation) — no X, no Java.
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
libreoffice-writer-nogui fonts-crosextra-carlito fonts-crosextra-caladea \
|
||||||
|
fonts-liberation2 fonts-dejavu-core \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
COPY pyproject.toml uv.lock README.md ./
|
COPY pyproject.toml uv.lock README.md ./
|
||||||
COPY src/ src/
|
COPY src/ src/
|
||||||
|
|
||||||
|
|||||||
@@ -17,8 +17,9 @@ from contextlib import asynccontextmanager
|
|||||||
from importlib import resources
|
from importlib import resources
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AsyncIterator, Iterator
|
from typing import AsyncIterator, Iterator
|
||||||
|
from urllib.parse import quote
|
||||||
|
|
||||||
from fastapi import FastAPI, Header, HTTPException
|
from fastapi import FastAPI, Header, HTTPException, Query
|
||||||
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
|
from fastapi.responses import FileResponse, HTMLResponse, Response, StreamingResponse
|
||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
|
|
||||||
@@ -186,28 +187,42 @@ def pdf_list(key: str) -> dict:
|
|||||||
store.close()
|
store.close()
|
||||||
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
|
except Exception: # noqa: BLE001 — a Zotero-only key has no bib row
|
||||||
title = ""
|
title = ""
|
||||||
return {
|
from llm import convert
|
||||||
"key": key,
|
|
||||||
"title": title,
|
convertible = convert.available()
|
||||||
"files": [
|
base = f"/pdf/{key}/file?name="
|
||||||
|
out = []
|
||||||
|
for f in files:
|
||||||
|
ext = Path(f.name).suffix.lower()
|
||||||
|
converted = convertible and ext in convert.CONVERTIBLE
|
||||||
|
out.append(
|
||||||
{
|
{
|
||||||
"name": f.name,
|
"name": f.name,
|
||||||
"size": f.size,
|
"size": f.size,
|
||||||
"source": f.source,
|
"source": f.source,
|
||||||
"renderable": f.renderable,
|
"renderable": f.renderable or converted,
|
||||||
"renderer": f.renderer,
|
# Word files render as the PDF LibreOffice lays out (Word-exact
|
||||||
|
# pages) when the converter is in the image; docx-preview in
|
||||||
|
# the browser is the fallback
|
||||||
|
"renderer": "pdf" if converted else f.renderer,
|
||||||
|
"converted": converted,
|
||||||
"media_type": f.media_type,
|
"media_type": f.media_type,
|
||||||
|
"url": base + quote(f.name) + ("&as=pdf" if converted else ""),
|
||||||
|
"download_url": base + quote(f.name),
|
||||||
}
|
}
|
||||||
for f in files
|
)
|
||||||
],
|
return {"key": key, "title": title, "files": out}
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/pdf/{key}/file")
|
@app.get("/pdf/{key}/file")
|
||||||
def pdf_file(key: str, name: str = "") -> FileResponse:
|
def pdf_file(
|
||||||
|
key: str, name: str = "", as_: str = Query("", alias="as")
|
||||||
|
) -> FileResponse:
|
||||||
"""The attachment bytes — a PDF inline with range requests (PDF.js
|
"""The attachment bytes — a PDF inline with range requests (PDF.js
|
||||||
streams pages from a large file instead of waiting for all of it),
|
streams pages from a large file instead of waiting for all of it),
|
||||||
anything else as a download."""
|
anything else as a download. ``as=pdf`` on a Word file returns the
|
||||||
|
LibreOffice rendering (cached under .state/llm/pdfcache); 502 when
|
||||||
|
the conversion fails, 404 when the converter is not in the image."""
|
||||||
from llm.pdfs import resolve
|
from llm.pdfs import resolve
|
||||||
|
|
||||||
if not _KEY_RE.match(key):
|
if not _KEY_RE.match(key):
|
||||||
@@ -224,6 +239,31 @@ def pdf_file(key: str, name: str = "") -> FileResponse:
|
|||||||
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
|
chosen = resolve(files, name, roots=(bib_root, zstore, croot))
|
||||||
if chosen is None:
|
if chosen is None:
|
||||||
raise HTTPException(404, "no such attachment on file for this item")
|
raise HTTPException(404, "no such attachment on file for this item")
|
||||||
|
if as_ == "pdf" and chosen.renderer != "pdf":
|
||||||
|
from conf import ROOT
|
||||||
|
from llm import convert
|
||||||
|
|
||||||
|
if (
|
||||||
|
Path(chosen.name).suffix.lower() not in convert.CONVERTIBLE
|
||||||
|
or not convert.available()
|
||||||
|
):
|
||||||
|
raise HTTPException(404, "no PDF rendering available for this file")
|
||||||
|
try:
|
||||||
|
rendered = convert.to_pdf(
|
||||||
|
chosen.path, cache_dir=Path(ROOT) / ".state" / "llm" / "pdfcache"
|
||||||
|
)
|
||||||
|
except convert.ConversionError as exc:
|
||||||
|
raise HTTPException(502, str(exc)) from exc
|
||||||
|
return FileResponse(
|
||||||
|
rendered,
|
||||||
|
media_type="application/pdf",
|
||||||
|
filename=Path(chosen.name).stem + ".pdf",
|
||||||
|
content_disposition_type="inline",
|
||||||
|
headers={
|
||||||
|
"Accept-Ranges": "bytes",
|
||||||
|
"Cache-Control": "private, max-age=3600",
|
||||||
|
},
|
||||||
|
)
|
||||||
return FileResponse(
|
return FileResponse(
|
||||||
chosen.path,
|
chosen.path,
|
||||||
media_type=chosen.media_type,
|
media_type=chosen.media_type,
|
||||||
|
|||||||
104
src/llm/convert.py
Normal file
104
src/llm/convert.py
Normal file
@@ -0,0 +1,104 @@
|
|||||||
|
"""Word → PDF with LibreOffice, for Word-exact pages in the viewer.
|
||||||
|
|
||||||
|
``soffice --headless --convert-to pdf`` lays the document out the way
|
||||||
|
Writer does (real pagination, styles, metric-compatible fonts for
|
||||||
|
Calibri/Cambria/Times/Arial via Carlito/Caladea/Liberation), which the
|
||||||
|
browser-side docx-preview cannot. Conversions are cached by the source
|
||||||
|
file's identity (path, size, mtime) under ``cache_dir`` so a document is
|
||||||
|
converted once, and serialised behind one lock because a shared
|
||||||
|
LibreOffice profile cannot run two instances at a time.
|
||||||
|
|
||||||
|
available() → soffice is on PATH
|
||||||
|
to_pdf(src, cache_dir=…) → cached PDF path (converts on a miss)
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import threading
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
_LOCK = threading.Lock()
|
||||||
|
_TIMEOUT = 180
|
||||||
|
CONVERTIBLE = (".docx", ".doc", ".rtf", ".odt")
|
||||||
|
|
||||||
|
|
||||||
|
class ConversionError(RuntimeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def soffice() -> str | None:
|
||||||
|
return shutil.which("soffice") or shutil.which("libreoffice")
|
||||||
|
|
||||||
|
|
||||||
|
def available() -> bool:
|
||||||
|
return soffice() is not None
|
||||||
|
|
||||||
|
|
||||||
|
def cache_key(src: Path) -> str:
|
||||||
|
st = src.stat()
|
||||||
|
raw = f"{src.resolve()}|{st.st_size}|{st.st_mtime_ns}"
|
||||||
|
return hashlib.sha1(raw.encode()).hexdigest() # noqa: S324 — cache identity
|
||||||
|
|
||||||
|
|
||||||
|
def to_pdf(src: Path, *, cache_dir: Path, runner=None) -> Path:
|
||||||
|
"""The PDF rendering of *src*, converting with LibreOffice when the
|
||||||
|
cache has no current copy. *runner* (tests) replaces ``subprocess.run``.
|
||||||
|
Raises ``ConversionError`` when soffice is missing, times out, fails or
|
||||||
|
produces nothing."""
|
||||||
|
src = Path(src)
|
||||||
|
if not src.is_file():
|
||||||
|
raise ConversionError(f"{src.name}: not on file")
|
||||||
|
cache_dir = Path(cache_dir)
|
||||||
|
cache_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
out = cache_dir / f"{cache_key(src)}.pdf"
|
||||||
|
if out.is_file() and out.stat().st_size > 0:
|
||||||
|
return out
|
||||||
|
exe = soffice()
|
||||||
|
if exe is None:
|
||||||
|
raise ConversionError("LibreOffice (soffice) is not installed in this image")
|
||||||
|
run = runner or subprocess.run
|
||||||
|
profile = cache_dir / ".lo-profile"
|
||||||
|
profile.mkdir(exist_ok=True)
|
||||||
|
with _LOCK, tempfile.TemporaryDirectory(prefix="lo-", dir=cache_dir) as tmp:
|
||||||
|
cmd = [
|
||||||
|
exe,
|
||||||
|
f"-env:UserInstallation=file://{profile}",
|
||||||
|
"--headless",
|
||||||
|
"--norestore",
|
||||||
|
"--nolockcheck",
|
||||||
|
"--convert-to",
|
||||||
|
"pdf:writer_pdf_Export",
|
||||||
|
"--outdir",
|
||||||
|
tmp,
|
||||||
|
str(src),
|
||||||
|
]
|
||||||
|
env = {**os.environ, "HOME": str(profile)}
|
||||||
|
try:
|
||||||
|
proc = run(cmd, capture_output=True, text=True, timeout=_TIMEOUT, env=env)
|
||||||
|
except subprocess.TimeoutExpired as exc:
|
||||||
|
raise ConversionError(
|
||||||
|
f"{src.name}: conversion timed out after {_TIMEOUT}s"
|
||||||
|
) from exc
|
||||||
|
produced = Path(tmp) / (src.stem + ".pdf")
|
||||||
|
if (
|
||||||
|
proc.returncode != 0
|
||||||
|
or not produced.is_file()
|
||||||
|
or produced.stat().st_size == 0
|
||||||
|
):
|
||||||
|
detail = (proc.stderr or proc.stdout or "").strip()[-300:]
|
||||||
|
raise ConversionError(
|
||||||
|
f"{src.name}: LibreOffice failed ({proc.returncode}) {detail}"
|
||||||
|
)
|
||||||
|
tmp_out = out.with_suffix(".pdf.part")
|
||||||
|
shutil.move(str(produced), tmp_out)
|
||||||
|
os.replace(tmp_out, out)
|
||||||
|
log.info("converted %s → %s (%d bytes)", src.name, out.name, out.stat().st_size)
|
||||||
|
return out
|
||||||
@@ -214,8 +214,10 @@
|
|||||||
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
|
const current = files.find(f => f.name === wantFile) || files.find(f => f.renderable) || files[0];
|
||||||
|
|
||||||
sel.value = current.name;
|
sel.value = current.name;
|
||||||
const fileUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
|
const rawUrl = '/pdf/' + encodeURIComponent(key) + '/file?name=' + encodeURIComponent(current.name);
|
||||||
$('download').href = fileUrl;
|
const fileUrl = current.url || rawUrl; // a Word file's LibreOffice rendering when the image has the converter
|
||||||
|
$('download').href = current.download_url || rawUrl; // the original bytes
|
||||||
|
if (current.converted) $('download').textContent = 'download original';
|
||||||
sel.addEventListener('change', () => {
|
sel.addEventListener('change', () => {
|
||||||
const p = new URLSearchParams(location.search);
|
const p = new URLSearchParams(location.search);
|
||||||
p.set('file', sel.value); p.delete('page');
|
p.set('file', sel.value); p.delete('page');
|
||||||
|
|||||||
@@ -468,8 +468,13 @@ class TestPdfViewer:
|
|||||||
|
|
||||||
def test_list_and_serve(self, tmp_path, monkeypatch):
|
def test_list_and_serve(self, tmp_path, monkeypatch):
|
||||||
import llm.api as api
|
import llm.api as api
|
||||||
|
import llm.convert as conv
|
||||||
from llm.pdfs import PdfFile
|
from llm.pdfs import PdfFile
|
||||||
|
|
||||||
|
monkeypatch.setattr(
|
||||||
|
conv, "available", lambda: False
|
||||||
|
) # the host may have soffice
|
||||||
|
|
||||||
root = tmp_path / "root"
|
root = tmp_path / "root"
|
||||||
(root / "K").mkdir(parents=True)
|
(root / "K").mkdir(parents=True)
|
||||||
pdf = root / "K" / "a.pdf"
|
pdf = root / "K" / "a.pdf"
|
||||||
@@ -529,3 +534,98 @@ class TestPdfViewer:
|
|||||||
assert "src.kind !== 'rule'" in chat
|
assert "src.kind !== 'rule'" in chat
|
||||||
search = client.get("/ui/search").text
|
search = client.get("/ui/search").text
|
||||||
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search
|
assert "'/ui/pdf/' + encodeURIComponent(r.item_key)" in search
|
||||||
|
|
||||||
|
|
||||||
|
class TestWordExactPages:
|
||||||
|
"""A Word attachment lists as renderer 'pdf' with an as=pdf url when
|
||||||
|
LibreOffice is in the image, and /pdf/<key>/file?as=pdf serves the
|
||||||
|
cached conversion; without the converter it falls back to docx-preview."""
|
||||||
|
|
||||||
|
def _setup(self, tmp_path, monkeypatch, available):
|
||||||
|
import llm.api as api
|
||||||
|
import llm.convert as conv
|
||||||
|
from llm.pdfs import PdfFile
|
||||||
|
|
||||||
|
root = tmp_path / "root"
|
||||||
|
(root / "K").mkdir(parents=True)
|
||||||
|
docx = root / "K" / "letter.docx"
|
||||||
|
docx.write_bytes(b"PK docx")
|
||||||
|
files = [PdfFile("letter.docx", docx, 7, "comment")]
|
||||||
|
monkeypatch.setattr(api, "_pdf_files", lambda key: files)
|
||||||
|
monkeypatch.setattr(
|
||||||
|
api, "_pdf_paths", lambda: (tmp_path / "z.sqlite", tmp_path / "zs", root)
|
||||||
|
)
|
||||||
|
monkeypatch.setattr(conv, "available", lambda: available)
|
||||||
|
|
||||||
|
class _Store:
|
||||||
|
_storage = tmp_path / "bibstorage"
|
||||||
|
|
||||||
|
def get(self, key):
|
||||||
|
class _I:
|
||||||
|
title = "t"
|
||||||
|
|
||||||
|
return _I()
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
monkeypatch.setattr("conf.connect.bib", lambda: _Store())
|
||||||
|
return docx
|
||||||
|
|
||||||
|
def test_with_converter(self, tmp_path, monkeypatch):
|
||||||
|
import llm.convert as conv
|
||||||
|
|
||||||
|
docx = self._setup(tmp_path, monkeypatch, True)
|
||||||
|
converted = tmp_path / "letter.pdf"
|
||||||
|
converted.write_bytes(b"%PDF-1.7 word-exact")
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
def fake_to_pdf(src, *, cache_dir):
|
||||||
|
calls.append((src, cache_dir))
|
||||||
|
return converted
|
||||||
|
|
||||||
|
monkeypatch.setattr(conv, "to_pdf", fake_to_pdf)
|
||||||
|
f = client.get("/pdf/ABCD1234").json()["files"][0]
|
||||||
|
assert (
|
||||||
|
f["renderer"] == "pdf"
|
||||||
|
and f["converted"] is True
|
||||||
|
and f["renderable"] is True
|
||||||
|
)
|
||||||
|
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx&as=pdf"
|
||||||
|
assert f["download_url"] == "/pdf/ABCD1234/file?name=letter.docx"
|
||||||
|
r = client.get(f["url"])
|
||||||
|
assert r.status_code == 200 and r.headers["content-type"] == "application/pdf"
|
||||||
|
assert r.content == b"%PDF-1.7 word-exact" and r.headers[
|
||||||
|
"content-disposition"
|
||||||
|
].startswith("inline")
|
||||||
|
assert (
|
||||||
|
calls
|
||||||
|
and calls[0][0] == docx
|
||||||
|
and str(calls[0][1]).endswith(".state/llm/pdfcache")
|
||||||
|
)
|
||||||
|
# the original is still a download
|
||||||
|
r = client.get(f["download_url"])
|
||||||
|
assert r.status_code == 200 and r.headers["content-disposition"].startswith(
|
||||||
|
"attachment"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_conversion_failure_is_a_502(self, tmp_path, monkeypatch):
|
||||||
|
import llm.convert as conv
|
||||||
|
|
||||||
|
self._setup(tmp_path, monkeypatch, True)
|
||||||
|
|
||||||
|
def boom(src, *, cache_dir):
|
||||||
|
raise conv.ConversionError("letter.docx: LibreOffice failed (1)")
|
||||||
|
|
||||||
|
monkeypatch.setattr(conv, "to_pdf", boom)
|
||||||
|
r = client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf")
|
||||||
|
assert r.status_code == 502 and "LibreOffice failed" in r.json()["detail"]
|
||||||
|
|
||||||
|
def test_without_converter_falls_back_to_docx_preview(self, tmp_path, monkeypatch):
|
||||||
|
self._setup(tmp_path, monkeypatch, False)
|
||||||
|
f = client.get("/pdf/ABCD1234").json()["files"][0]
|
||||||
|
assert f["renderer"] == "docx" and f["converted"] is False
|
||||||
|
assert f["url"] == "/pdf/ABCD1234/file?name=letter.docx"
|
||||||
|
assert (
|
||||||
|
client.get("/pdf/ABCD1234/file?name=letter.docx&as=pdf").status_code == 404
|
||||||
|
)
|
||||||
|
|||||||
84
tests/llm/test_convert.py
Normal file
84
tests/llm/test_convert.py
Normal file
@@ -0,0 +1,84 @@
|
|||||||
|
"""llm.convert — Word → PDF through LibreOffice, cached (viewer)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
import llm.convert as conv
|
||||||
|
from llm.convert import ConversionError, cache_key, to_pdf
|
||||||
|
|
||||||
|
|
||||||
|
def _fake_runner(behaviour="ok"):
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
def run(cmd, capture_output, text, timeout, env):
|
||||||
|
calls.append(cmd)
|
||||||
|
outdir = Path(cmd[cmd.index("--outdir") + 1])
|
||||||
|
src = Path(cmd[-1])
|
||||||
|
if behaviour == "ok":
|
||||||
|
(outdir / (src.stem + ".pdf")).write_bytes(b"%PDF-1.7 converted")
|
||||||
|
return subprocess.CompletedProcess(cmd, 0, "convert ok", "")
|
||||||
|
if behaviour == "empty":
|
||||||
|
return subprocess.CompletedProcess(cmd, 0, "", "")
|
||||||
|
if behaviour == "fail":
|
||||||
|
return subprocess.CompletedProcess(
|
||||||
|
cmd, 1, "", "Error: source file could not be loaded"
|
||||||
|
)
|
||||||
|
raise subprocess.TimeoutExpired(cmd, timeout)
|
||||||
|
|
||||||
|
run.calls = calls # type: ignore[attr-defined]
|
||||||
|
return run
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def docx(tmp_path):
|
||||||
|
p = tmp_path / "letter.docx"
|
||||||
|
p.write_bytes(b"PK docx bytes")
|
||||||
|
return p
|
||||||
|
|
||||||
|
|
||||||
|
def test_converts_once_then_serves_the_cache(tmp_path, docx, monkeypatch):
|
||||||
|
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
|
||||||
|
run = _fake_runner()
|
||||||
|
cache = tmp_path / "cache"
|
||||||
|
out = to_pdf(docx, cache_dir=cache, runner=run)
|
||||||
|
assert out.read_bytes().startswith(b"%PDF") and out.parent == cache
|
||||||
|
assert out.name == cache_key(docx) + ".pdf"
|
||||||
|
cmd = run.calls[0]
|
||||||
|
assert (
|
||||||
|
cmd[0] == "/usr/bin/soffice" and "--headless" in cmd and "--convert-to" in cmd
|
||||||
|
)
|
||||||
|
assert cmd[-1] == str(docx) and any(
|
||||||
|
a.startswith("-env:UserInstallation=file://") for a in cmd
|
||||||
|
)
|
||||||
|
assert to_pdf(docx, cache_dir=cache, runner=run) == out and len(run.calls) == 1
|
||||||
|
# a changed source (new size) is a cache miss
|
||||||
|
docx.write_bytes(b"PK docx bytes v2!")
|
||||||
|
assert to_pdf(docx, cache_dir=cache, runner=run) != out and len(run.calls) == 2
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"behaviour, msg",
|
||||||
|
[
|
||||||
|
("empty", "LibreOffice failed"),
|
||||||
|
("fail", "could not be loaded"),
|
||||||
|
("timeout", "timed out"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_failures_raise(tmp_path, docx, monkeypatch, behaviour, msg):
|
||||||
|
monkeypatch.setattr(conv, "soffice", lambda: "/usr/bin/soffice")
|
||||||
|
with pytest.raises(ConversionError, match=msg):
|
||||||
|
to_pdf(docx, cache_dir=tmp_path / "c", runner=_fake_runner(behaviour))
|
||||||
|
assert not list((tmp_path / "c").glob("*.pdf"))
|
||||||
|
|
||||||
|
|
||||||
|
def test_missing_soffice_and_missing_source(tmp_path, docx, monkeypatch):
|
||||||
|
monkeypatch.setattr(conv, "soffice", lambda: None)
|
||||||
|
assert conv.available() is False
|
||||||
|
with pytest.raises(ConversionError, match="not installed"):
|
||||||
|
to_pdf(docx, cache_dir=tmp_path / "c")
|
||||||
|
with pytest.raises(ConversionError, match="not on file"):
|
||||||
|
to_pdf(tmp_path / "nope.docx", cache_dir=tmp_path / "c")
|
||||||
Reference in New Issue
Block a user