feat(comments): DOCX/text extract + combined.md writer (refs #253)

- extract_attachment now dispatches PDF/DOCX/TXT/HTML; DOCX via
  python-docx, text-like via plain read with cheap HTML strip.
- combine.py writes one combined.md per comment dir with YAML
  frontmatter (per-attachment status + chars) and per-section body.
- extract_comment is idempotent (existing combined.md is a no-op
  unless force=True).
- parse_combined() splits frontmatter ↔ body for downstream consumers.

Walker + DuckDB view land in the next batch.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
kert
2026-04-23 15:42:32 -04:00
parent 76c7288d28
commit b2ad400764
6 changed files with 320 additions and 18 deletions

View File

@@ -2,11 +2,17 @@
Public API: Public API:
extract_attachment(path) -> ExtractResult — single file extract_attachment(path) -> ExtractResult — single file
extract_comment(comment_dir) -> Path | None — write combined.md for a dir extract_comment(comment_dir, *, inline_body, force) -> Path — write combined.md
""" """
from __future__ import annotations from __future__ import annotations
from rex.comments.extract import ExtractResult, extract_attachment, extract_comment from rex.comments.combine import extract_comment, parse_combined
from rex.comments.extract import ExtractResult, extract_attachment
__all__ = ["ExtractResult", "extract_attachment", "extract_comment"] __all__ = [
"ExtractResult",
"extract_attachment",
"extract_comment",
"parse_combined",
]

106
src/rex/comments/combine.py Normal file
View File

@@ -0,0 +1,106 @@
"""Aggregate inline comment + attachment text into combined.md.
One comment dir → one combined.md with YAML frontmatter (status of each
attachment, char counts) and a markdown body (inline body + per-attachment
sections). Resumable: existing combined.md is a no-op unless force=True.
"""
from __future__ import annotations
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
import yaml
from rex.comments.extract import ExtractResult, extract_attachment
_FILENAME = "combined.md"
def extract_comment(
comment_dir: Path,
*,
inline_body: str = "",
force: bool = False,
) -> Path:
"""Extract every attachment in *comment_dir* and write *combined.md*.
Returns the path to combined.md. If it already exists and ``force``
is False, returns it without doing any work (resume rule).
*comment_dir* must be ``.../{docket_id}/{comment_id}/``.
"""
out = comment_dir / _FILENAME
if out.is_file() and not force:
return out
docket_id = comment_dir.parent.name
comment_id = comment_dir.name
attachments_meta: list[dict[str, Any]] = []
body_sections: list[str] = []
for path in sorted(_attachment_paths(comment_dir)):
result = extract_attachment(path)
attachments_meta.append(
{"file": path.name, "status": result.status, "chars": result.chars}
)
body_sections.append(_render_section(path.name, result))
inline_section = (
f"## Inline comment\n\n{inline_body.strip()}\n" if inline_body.strip() else ""
)
body_text = inline_section + "\n".join(body_sections)
total_chars = len(inline_body) + sum(a["chars"] for a in attachments_meta)
frontmatter = {
"comment_id": comment_id,
"docket_id": docket_id,
"extracted_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
"chars": total_chars,
"attachments": attachments_meta,
}
out.write_text(_emit(frontmatter, body_text))
return out
def parse_combined(text: str) -> tuple[dict[str, Any], str]:
"""Split a combined.md into (frontmatter dict, body markdown)."""
if not text.startswith("---\n"):
raise ValueError("missing frontmatter")
end = text.find("\n---\n", 4)
if end < 0:
raise ValueError("unterminated frontmatter")
fm = yaml.safe_load(text[4:end]) or {}
body = text[end + len("\n---\n") :].lstrip("\n")
return fm, body
# ── internals ─────────────────────────────────────────────────────
def _attachment_paths(comment_dir: Path) -> list[Path]:
return [
p
for p in comment_dir.iterdir()
if p.is_file() and p.name != _FILENAME and not p.name.startswith(".")
]
def _render_section(filename: str, result: ExtractResult) -> str:
if result.status == "ok":
body = result.text.strip()
elif result.status == "ocr_needed":
body = f"_(ocr_needed — {result.chars} chars extracted)_"
elif result.status == "failed":
body = "_(failed — extraction error, see logs)_"
else: # unsupported
body = "_(unsupported file type)_"
return f"## {filename}\n\n{body}\n"
def _emit(frontmatter: dict[str, Any], body: str) -> str:
yaml_text = yaml.safe_dump(frontmatter, sort_keys=False, default_flow_style=False)
return f"---\n{yaml_text}---\n\n{body}"

View File

@@ -6,6 +6,7 @@ python-docx for DOCX, plain-read fallback for txt/html.
""" """
import logging import logging
import re
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path from pathlib import Path
@@ -26,18 +27,14 @@ class ExtractResult:
def extract_attachment(path: Path) -> ExtractResult: def extract_attachment(path: Path) -> ExtractResult:
"""Extract text from one attachment. """Extract text from one attachment. See module docstring for status values."""
Status values:
ok — text extracted, > 50 chars
ocr_needed — file parsed but ≤ 50 chars (likely scanned image PDF)
failed — exception during parsing
unsupported — file extension we don't handle
"""
suffix = path.suffix.lower() suffix = path.suffix.lower()
if suffix == ".pdf": if suffix == ".pdf":
return _extract_pdf(path) return _extract_pdf(path)
# DOCX and text-like handlers land in Batch B; default to unsupported. if suffix == ".docx":
return _extract_docx(path)
if suffix in (".txt", ".html", ".htm"):
return _extract_text_like(path, suffix)
return ExtractResult(text="", status="unsupported", chars=0) return ExtractResult(text="", status="unsupported", chars=0)
@@ -53,10 +50,35 @@ def _extract_pdf(path: Path) -> ExtractResult:
return ExtractResult(text=text, status=status, chars=chars) return ExtractResult(text=text, status=status, chars=chars)
def extract_comment(comment_dir: Path) -> Path: def _extract_docx(path: Path) -> ExtractResult:
"""Write combined.md for one comment dir. try:
import docx # local import — only loaded when needed
Implementation lands in the next batch (combine.py); this stub d = docx.Document(str(path))
exists so the public API surface is stable from day one. text = "\n\n".join(p.text for p in d.paragraphs if p.text)
""" except Exception as e: # noqa: BLE001
raise NotImplementedError("extract_comment lands in Batch B (combine.py)") log.warning("docx extract failed for %s: %s", path, e)
return ExtractResult(text="", status="failed", chars=0)
chars = len(text)
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
return ExtractResult(text=text, status=status, chars=chars)
def _extract_text_like(path: Path, suffix: str) -> ExtractResult:
try:
raw = path.read_text(encoding="utf-8", errors="ignore")
except OSError as e:
log.warning("text-like extract failed for %s: %s", path, e)
return ExtractResult(text="", status="failed", chars=0)
text = _strip_html(raw) if suffix in (".html", ".htm") else raw
chars = len(text)
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
return ExtractResult(text=text, status=status, chars=chars)
def _strip_html(s: str) -> str:
"""Cheap HTML→text — break on block tags, drop the rest."""
s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I)
s = re.sub(r"<(p|div|br|li|h[1-6])[^>]*>", "\n", s, flags=re.I)
s = re.sub(r"<[^>]+>", "", s)
return re.sub(r"\n\s*\n+", "\n\n", s).strip()

View File

@@ -0,0 +1,89 @@
"""Per-comment combine — write combined.md with YAML frontmatter."""
from __future__ import annotations
from pathlib import Path
import fitz
from rex.comments import extract_comment
from rex.comments.combine import parse_combined
def _make_pdf(path: Path, text: str) -> None:
doc = fitz.open()
doc.new_page().insert_text((50, 72), text)
doc.save(str(path))
doc.close()
def _setup_comment_dir(tmp_path: Path, cid: str = "CMS-2024-0001-0042") -> Path:
"""Layout matches .state/comments/{docket}/{cid}/."""
docket = "-".join(cid.split("-")[:3])
d = tmp_path / docket / cid
d.mkdir(parents=True)
_make_pdf(
d / "attachment_1.pdf",
"Position: oppose the proposed rule. This is a substantive comment.",
)
return d
def test_extract_comment_writes_combined_md(tmp_path: Path):
d = _setup_comment_dir(tmp_path)
out = extract_comment(d, inline_body="Inline body text from items.abstract.")
assert out == d / "combined.md"
assert out.is_file()
def test_combined_md_has_frontmatter_and_body(tmp_path: Path):
d = _setup_comment_dir(tmp_path)
extract_comment(d, inline_body="Inline body.")
text = (d / "combined.md").read_text()
fm, body = parse_combined(text)
assert fm["comment_id"] == "CMS-2024-0001-0042"
assert fm["docket_id"] == "CMS-2024-0001"
assert fm["chars"] > 0
assert len(fm["attachments"]) == 1
assert fm["attachments"][0]["file"] == "attachment_1.pdf"
assert fm["attachments"][0]["status"] == "ok"
assert "## Inline comment" in body
assert "Inline body." in body
assert "## attachment_1.pdf" in body
assert "Position: oppose" in body
def test_extract_comment_is_idempotent(tmp_path: Path):
d = _setup_comment_dir(tmp_path)
first = extract_comment(d, inline_body="x")
mtime1 = first.stat().st_mtime
second = extract_comment(d, inline_body="x")
assert second == first
assert second.stat().st_mtime == mtime1 # no rewrite on second call
def test_extract_comment_force_rewrites(tmp_path: Path):
d = _setup_comment_dir(tmp_path)
extract_comment(d, inline_body="first")
extract_comment(d, inline_body="second", force=True)
body = (d / "combined.md").read_text()
assert "second" in body
assert "first" not in body
def test_extract_comment_no_attachments(tmp_path: Path):
d = tmp_path / "CMS-2024-0001" / "CMS-2024-0001-0099"
d.mkdir(parents=True)
out = extract_comment(d, inline_body="Inline only.")
fm, body = parse_combined(out.read_text())
assert fm["attachments"] == []
assert "Inline only." in body
def test_extract_comment_handles_failed_attachment(tmp_path: Path):
d = tmp_path / "CMS-2024-0001" / "CMS-2024-0001-0100"
d.mkdir(parents=True)
(d / "attachment_1.pdf").write_bytes(b"not a real pdf")
out = extract_comment(d, inline_body="x")
fm, _body = parse_combined(out.read_text())
assert fm["attachments"][0]["status"] == "failed"

View File

@@ -0,0 +1,44 @@
"""DOCX extraction via python-docx."""
from __future__ import annotations
from pathlib import Path
import docx
from rex.comments import extract_attachment
def _make_docx(path: Path, paragraphs: list[str]) -> Path:
d = docx.Document()
for p in paragraphs:
d.add_paragraph(p)
d.save(str(path))
return path
def test_extract_docx_with_text(tmp_path: Path):
p = _make_docx(
tmp_path / "a.docx",
["First paragraph of comment letter.", "Second paragraph with details."],
)
result = extract_attachment(p)
assert result.status == "ok"
assert "First paragraph" in result.text
assert "Second paragraph" in result.text
def test_extract_docx_empty_marked_ocr_needed(tmp_path: Path):
p = _make_docx(tmp_path / "empty.docx", [])
result = extract_attachment(p)
# No paragraphs ⇒ 0 chars ⇒ ocr_needed (we treat empty docx the same as
# an image-only pdf — body lives somewhere we can't read it).
assert result.status == "ocr_needed"
assert result.chars <= 50
def test_extract_docx_corrupted_marked_failed(tmp_path: Path):
bad = tmp_path / "bad.docx"
bad.write_bytes(b"this is not a docx zip")
result = extract_attachment(bad)
assert result.status == "failed"

View File

@@ -0,0 +1,35 @@
"""Plain-text fallback + unsupported extensions."""
from __future__ import annotations
from pathlib import Path
from rex.comments import extract_attachment
def test_extract_txt(tmp_path: Path):
p = tmp_path / "note.txt"
p.write_text("This is a plain text comment longer than fifty characters total.")
result = extract_attachment(p)
assert result.status == "ok"
assert "plain text comment" in result.text
def test_extract_html_strips_tags(tmp_path: Path):
p = tmp_path / "page.html"
p.write_text(
"<html><body><p>This is a comment letter</p>"
"<p>with two paragraphs of substance.</p></body></html>"
)
result = extract_attachment(p)
assert result.status == "ok"
assert "This is a comment letter" in result.text
assert "<p>" not in result.text
def test_extract_unknown_extension(tmp_path: Path):
p = tmp_path / "weird.xyz"
p.write_bytes(b"some bytes")
result = extract_attachment(p)
assert result.status == "unsupported"
assert result.chars == 0