Some checks failed
CI / lint (push) Successful in 45s
CI / notebooks-smoke (push) Has been cancelled
Deploy / notebooks (push) Has been cancelled
Deploy / zotero (push) Has been cancelled
Deploy / docs (push) Has been cancelled
Deploy / api (push) Has been cancelled
Deploy / llm (push) Has been cancelled
Deploy / mc (push) Has been cancelled
Deploy / report (push) Has been cancelled
CI / test (push) Has been cancelled
Infra CI / docs (push) Has been cancelled
Infra CI / api (push) Has been cancelled
Infra CI / llm (push) Has been cancelled
Infra CI / mc (push) Has been cancelled
Infra CI / notebooks (push) Has been cancelled
Infra CI / zotero (push) Has been cancelled
- pfs.guidance: MLN_RE accepts ICN/MLN product numbers ("ICN MLN909188",
"ICN 909289"), MLN_MATTERS_RE captures MM/SE article numbers; mln_refs
normalises both to "MLN <n>" / "MLN Matters MM<n>"; resolve_mln finds
the bib item whose URL or title carries the identifier (word-bounded).
- pfs.guidance.iom_section_page: locates an IOM section heading inside the
chapter PDF attached to the resolved item (llm.pages.locate_section —
the body heading is the one followed by its "(Rev." line, which the
table of contents lacks); cached per item/section and per PDF.
- pfs.codetables.GuidanceRow.page (+ column, added in place by
ensure_tables; legacy NULL reads 0).
- llm.lineage: resolved IOM/MLN guidance rows link to the bib item's URL,
with #page=N when the section was located.
110 lines
3.6 KiB
Python
110 lines
3.6 KiB
Python
"""Locate chunks inside local PDFs so evidence links can carry ``#page=N``.
|
|
|
|
Extraction joins PDF pages with blank lines and loses the boundaries
|
|
(``rex.comments.extract._extract_pdf``); rather than re-extract 28k
|
|
comments, the indexer re-opens the PDF that a chunk's markdown section
|
|
came from and finds the page containing the chunk's opening words.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import replace
|
|
from pathlib import Path
|
|
|
|
from llm.chunk import Chunk, Doc
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
_PROBE_CHARS = 60
|
|
_MIN_PROBE = 12
|
|
_HEADING = re.compile(r"\A#{1,6}[^\n]*\n+")
|
|
|
|
|
|
def _norm(text: str) -> str:
|
|
return " ".join(text.split())
|
|
|
|
|
|
def _probe_text(text: str) -> str:
|
|
"""Chunk text minus a leading markdown heading — the first chunk of a
|
|
``## attachment.pdf`` section starts with that heading, which is not
|
|
in the PDF."""
|
|
return _HEADING.sub("", text, count=1)
|
|
|
|
|
|
def pdf_pages(path: Path) -> list[str]:
|
|
"""Whitespace-normalized text per page; ``[]`` when unreadable."""
|
|
try:
|
|
import fitz # pymupdf — imported lazily, the chat image has no PDFs
|
|
except ImportError: # pragma: no cover
|
|
return []
|
|
try:
|
|
with fitz.open(path) as doc:
|
|
return [_norm(page.get_text()) for page in doc]
|
|
except Exception as e: # noqa: BLE001 — pymupdf raises several types
|
|
log.warning("pdf pages failed for %s: %s", path, e)
|
|
return []
|
|
|
|
|
|
def locate(pages: list[str], probe: str) -> int:
|
|
"""1-based page whose text contains ``probe`` (normalized); 0 if none."""
|
|
probe = _norm(probe)[:_PROBE_CHARS]
|
|
if len(probe) < _MIN_PROBE:
|
|
return 0
|
|
for i, page in enumerate(pages, start=1):
|
|
if probe in _norm(page):
|
|
return i
|
|
return 0
|
|
|
|
|
|
#: An IOM chapter heading in the body — "30.6.4 - Evaluation and
|
|
#: Management ... (Rev. 12345; Issued: ...)" — is always followed by its
|
|
#: revision line; the table of contents on the first pages lists the
|
|
#: same "30.6.4 - ..." entry without one, which is how the two are told
|
|
#: apart (#705 item 2).
|
|
_REV_WINDOW = 240
|
|
|
|
|
|
def locate_section(pages: list[str], section: str) -> int:
|
|
"""1-based page whose text carries IOM ``section`` ("30.6.4") as a
|
|
body heading — the number, a dash, the title, then "(Rev." within
|
|
``_REV_WINDOW`` characters; 0 when no page does (a TOC-only hit is
|
|
not a location)."""
|
|
section = section.strip().rstrip(".")
|
|
if not section:
|
|
return 0
|
|
rx = re.compile(
|
|
rf"(?<![\d.]){re.escape(section)}\s*[-–—]\s.{{0,{_REV_WINDOW}}}?\(Rev\.",
|
|
re.DOTALL,
|
|
)
|
|
for i, page in enumerate(pages, start=1):
|
|
if rx.search(_norm(page)):
|
|
return i
|
|
return 0
|
|
|
|
|
|
def enrich_pdf_pages(doc: Doc, chunks: list[Chunk]) -> list[Chunk]:
|
|
"""Stamp ``attachment`` (+ ``page`` for PDFs) on chunks whose
|
|
``section`` names one of ``doc.files``. Identity when ``doc.files``
|
|
is empty; never overrides a ``page`` already set."""
|
|
if not doc.files:
|
|
return chunks
|
|
files = dict(doc.files)
|
|
cache: dict[str, list[str]] = {}
|
|
out: list[Chunk] = []
|
|
for c in chunks:
|
|
section = c.metadata.get("section", "")
|
|
if section not in files:
|
|
out.append(c)
|
|
continue
|
|
md = {**c.metadata, "attachment": section}
|
|
path = files[section]
|
|
if path.lower().endswith(".pdf") and not md.get("page"):
|
|
if path not in cache:
|
|
cache[path] = pdf_pages(Path(path))
|
|
n = locate(cache[path], _probe_text(c.text))
|
|
md["page"] = str(n) if n else ""
|
|
out.append(replace(c, metadata=md))
|
|
return out
|