feat(comments): DOCX/text extract + combined.md writer (refs #253)
- extract_attachment now dispatches PDF/DOCX/TXT/HTML; DOCX via python-docx, text-like via plain read with cheap HTML strip. - combine.py writes one combined.md per comment dir with YAML frontmatter (per-attachment status + chars) and per-section body. - extract_comment is idempotent (existing combined.md is a no-op unless force=True). - parse_combined() splits frontmatter ↔ body for downstream consumers. Walker + DuckDB view land in the next batch. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -2,11 +2,17 @@
|
|||||||
|
|
||||||
Public API:
|
Public API:
|
||||||
extract_attachment(path) -> ExtractResult — single file
|
extract_attachment(path) -> ExtractResult — single file
|
||||||
extract_comment(comment_dir) -> Path | None — write combined.md for a dir
|
extract_comment(comment_dir, *, inline_body, force) -> Path — write combined.md
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from rex.comments.extract import ExtractResult, extract_attachment, extract_comment
|
from rex.comments.combine import extract_comment, parse_combined
|
||||||
|
from rex.comments.extract import ExtractResult, extract_attachment
|
||||||
|
|
||||||
__all__ = ["ExtractResult", "extract_attachment", "extract_comment"]
|
__all__ = [
|
||||||
|
"ExtractResult",
|
||||||
|
"extract_attachment",
|
||||||
|
"extract_comment",
|
||||||
|
"parse_combined",
|
||||||
|
]
|
||||||
|
|||||||
106
src/rex/comments/combine.py
Normal file
106
src/rex/comments/combine.py
Normal file
@@ -0,0 +1,106 @@
|
|||||||
|
"""Aggregate inline comment + attachment text into combined.md.
|
||||||
|
|
||||||
|
One comment dir → one combined.md with YAML frontmatter (status of each
|
||||||
|
attachment, char counts) and a markdown body (inline body + per-attachment
|
||||||
|
sections). Resumable: existing combined.md is a no-op unless force=True.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
from rex.comments.extract import ExtractResult, extract_attachment
|
||||||
|
|
||||||
|
_FILENAME = "combined.md"
|
||||||
|
|
||||||
|
|
||||||
|
def extract_comment(
|
||||||
|
comment_dir: Path,
|
||||||
|
*,
|
||||||
|
inline_body: str = "",
|
||||||
|
force: bool = False,
|
||||||
|
) -> Path:
|
||||||
|
"""Extract every attachment in *comment_dir* and write *combined.md*.
|
||||||
|
|
||||||
|
Returns the path to combined.md. If it already exists and ``force``
|
||||||
|
is False, returns it without doing any work (resume rule).
|
||||||
|
|
||||||
|
*comment_dir* must be ``.../{docket_id}/{comment_id}/``.
|
||||||
|
"""
|
||||||
|
out = comment_dir / _FILENAME
|
||||||
|
if out.is_file() and not force:
|
||||||
|
return out
|
||||||
|
|
||||||
|
docket_id = comment_dir.parent.name
|
||||||
|
comment_id = comment_dir.name
|
||||||
|
|
||||||
|
attachments_meta: list[dict[str, Any]] = []
|
||||||
|
body_sections: list[str] = []
|
||||||
|
|
||||||
|
for path in sorted(_attachment_paths(comment_dir)):
|
||||||
|
result = extract_attachment(path)
|
||||||
|
attachments_meta.append(
|
||||||
|
{"file": path.name, "status": result.status, "chars": result.chars}
|
||||||
|
)
|
||||||
|
body_sections.append(_render_section(path.name, result))
|
||||||
|
|
||||||
|
inline_section = (
|
||||||
|
f"## Inline comment\n\n{inline_body.strip()}\n" if inline_body.strip() else ""
|
||||||
|
)
|
||||||
|
body_text = inline_section + "\n".join(body_sections)
|
||||||
|
total_chars = len(inline_body) + sum(a["chars"] for a in attachments_meta)
|
||||||
|
|
||||||
|
frontmatter = {
|
||||||
|
"comment_id": comment_id,
|
||||||
|
"docket_id": docket_id,
|
||||||
|
"extracted_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||||
|
"chars": total_chars,
|
||||||
|
"attachments": attachments_meta,
|
||||||
|
}
|
||||||
|
|
||||||
|
out.write_text(_emit(frontmatter, body_text))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def parse_combined(text: str) -> tuple[dict[str, Any], str]:
|
||||||
|
"""Split a combined.md into (frontmatter dict, body markdown)."""
|
||||||
|
if not text.startswith("---\n"):
|
||||||
|
raise ValueError("missing frontmatter")
|
||||||
|
end = text.find("\n---\n", 4)
|
||||||
|
if end < 0:
|
||||||
|
raise ValueError("unterminated frontmatter")
|
||||||
|
fm = yaml.safe_load(text[4:end]) or {}
|
||||||
|
body = text[end + len("\n---\n") :].lstrip("\n")
|
||||||
|
return fm, body
|
||||||
|
|
||||||
|
|
||||||
|
# ── internals ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _attachment_paths(comment_dir: Path) -> list[Path]:
|
||||||
|
return [
|
||||||
|
p
|
||||||
|
for p in comment_dir.iterdir()
|
||||||
|
if p.is_file() and p.name != _FILENAME and not p.name.startswith(".")
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _render_section(filename: str, result: ExtractResult) -> str:
|
||||||
|
if result.status == "ok":
|
||||||
|
body = result.text.strip()
|
||||||
|
elif result.status == "ocr_needed":
|
||||||
|
body = f"_(ocr_needed — {result.chars} chars extracted)_"
|
||||||
|
elif result.status == "failed":
|
||||||
|
body = "_(failed — extraction error, see logs)_"
|
||||||
|
else: # unsupported
|
||||||
|
body = "_(unsupported file type)_"
|
||||||
|
return f"## {filename}\n\n{body}\n"
|
||||||
|
|
||||||
|
|
||||||
|
def _emit(frontmatter: dict[str, Any], body: str) -> str:
|
||||||
|
yaml_text = yaml.safe_dump(frontmatter, sort_keys=False, default_flow_style=False)
|
||||||
|
return f"---\n{yaml_text}---\n\n{body}"
|
||||||
@@ -6,6 +6,7 @@ python-docx for DOCX, plain-read fallback for txt/html.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import re
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -26,18 +27,14 @@ class ExtractResult:
|
|||||||
|
|
||||||
|
|
||||||
def extract_attachment(path: Path) -> ExtractResult:
|
def extract_attachment(path: Path) -> ExtractResult:
|
||||||
"""Extract text from one attachment.
|
"""Extract text from one attachment. See module docstring for status values."""
|
||||||
|
|
||||||
Status values:
|
|
||||||
ok — text extracted, > 50 chars
|
|
||||||
ocr_needed — file parsed but ≤ 50 chars (likely scanned image PDF)
|
|
||||||
failed — exception during parsing
|
|
||||||
unsupported — file extension we don't handle
|
|
||||||
"""
|
|
||||||
suffix = path.suffix.lower()
|
suffix = path.suffix.lower()
|
||||||
if suffix == ".pdf":
|
if suffix == ".pdf":
|
||||||
return _extract_pdf(path)
|
return _extract_pdf(path)
|
||||||
# DOCX and text-like handlers land in Batch B; default to unsupported.
|
if suffix == ".docx":
|
||||||
|
return _extract_docx(path)
|
||||||
|
if suffix in (".txt", ".html", ".htm"):
|
||||||
|
return _extract_text_like(path, suffix)
|
||||||
return ExtractResult(text="", status="unsupported", chars=0)
|
return ExtractResult(text="", status="unsupported", chars=0)
|
||||||
|
|
||||||
|
|
||||||
@@ -53,10 +50,35 @@ def _extract_pdf(path: Path) -> ExtractResult:
|
|||||||
return ExtractResult(text=text, status=status, chars=chars)
|
return ExtractResult(text=text, status=status, chars=chars)
|
||||||
|
|
||||||
|
|
||||||
def extract_comment(comment_dir: Path) -> Path:
|
def _extract_docx(path: Path) -> ExtractResult:
|
||||||
"""Write combined.md for one comment dir.
|
try:
|
||||||
|
import docx # local import — only loaded when needed
|
||||||
|
|
||||||
Implementation lands in the next batch (combine.py); this stub
|
d = docx.Document(str(path))
|
||||||
exists so the public API surface is stable from day one.
|
text = "\n\n".join(p.text for p in d.paragraphs if p.text)
|
||||||
"""
|
except Exception as e: # noqa: BLE001
|
||||||
raise NotImplementedError("extract_comment lands in Batch B (combine.py)")
|
log.warning("docx extract failed for %s: %s", path, e)
|
||||||
|
return ExtractResult(text="", status="failed", chars=0)
|
||||||
|
chars = len(text)
|
||||||
|
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
|
||||||
|
return ExtractResult(text=text, status=status, chars=chars)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_text_like(path: Path, suffix: str) -> ExtractResult:
|
||||||
|
try:
|
||||||
|
raw = path.read_text(encoding="utf-8", errors="ignore")
|
||||||
|
except OSError as e:
|
||||||
|
log.warning("text-like extract failed for %s: %s", path, e)
|
||||||
|
return ExtractResult(text="", status="failed", chars=0)
|
||||||
|
text = _strip_html(raw) if suffix in (".html", ".htm") else raw
|
||||||
|
chars = len(text)
|
||||||
|
status = "ok" if chars > _OCR_THRESHOLD else "ocr_needed"
|
||||||
|
return ExtractResult(text=text, status=status, chars=chars)
|
||||||
|
|
||||||
|
|
||||||
|
def _strip_html(s: str) -> str:
|
||||||
|
"""Cheap HTML→text — break on block tags, drop the rest."""
|
||||||
|
s = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", s, flags=re.S | re.I)
|
||||||
|
s = re.sub(r"<(p|div|br|li|h[1-6])[^>]*>", "\n", s, flags=re.I)
|
||||||
|
s = re.sub(r"<[^>]+>", "", s)
|
||||||
|
return re.sub(r"\n\s*\n+", "\n\n", s).strip()
|
||||||
|
|||||||
89
tests/rex/comments/test_combine.py
Normal file
89
tests/rex/comments/test_combine.py
Normal file
@@ -0,0 +1,89 @@
|
|||||||
|
"""Per-comment combine — write combined.md with YAML frontmatter."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import fitz
|
||||||
|
|
||||||
|
from rex.comments import extract_comment
|
||||||
|
from rex.comments.combine import parse_combined
|
||||||
|
|
||||||
|
|
||||||
|
def _make_pdf(path: Path, text: str) -> None:
|
||||||
|
doc = fitz.open()
|
||||||
|
doc.new_page().insert_text((50, 72), text)
|
||||||
|
doc.save(str(path))
|
||||||
|
doc.close()
|
||||||
|
|
||||||
|
|
||||||
|
def _setup_comment_dir(tmp_path: Path, cid: str = "CMS-2024-0001-0042") -> Path:
|
||||||
|
"""Layout matches .state/comments/{docket}/{cid}/."""
|
||||||
|
docket = "-".join(cid.split("-")[:3])
|
||||||
|
d = tmp_path / docket / cid
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
_make_pdf(
|
||||||
|
d / "attachment_1.pdf",
|
||||||
|
"Position: oppose the proposed rule. This is a substantive comment.",
|
||||||
|
)
|
||||||
|
return d
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_comment_writes_combined_md(tmp_path: Path):
|
||||||
|
d = _setup_comment_dir(tmp_path)
|
||||||
|
out = extract_comment(d, inline_body="Inline body text from items.abstract.")
|
||||||
|
assert out == d / "combined.md"
|
||||||
|
assert out.is_file()
|
||||||
|
|
||||||
|
|
||||||
|
def test_combined_md_has_frontmatter_and_body(tmp_path: Path):
|
||||||
|
d = _setup_comment_dir(tmp_path)
|
||||||
|
extract_comment(d, inline_body="Inline body.")
|
||||||
|
text = (d / "combined.md").read_text()
|
||||||
|
fm, body = parse_combined(text)
|
||||||
|
assert fm["comment_id"] == "CMS-2024-0001-0042"
|
||||||
|
assert fm["docket_id"] == "CMS-2024-0001"
|
||||||
|
assert fm["chars"] > 0
|
||||||
|
assert len(fm["attachments"]) == 1
|
||||||
|
assert fm["attachments"][0]["file"] == "attachment_1.pdf"
|
||||||
|
assert fm["attachments"][0]["status"] == "ok"
|
||||||
|
assert "## Inline comment" in body
|
||||||
|
assert "Inline body." in body
|
||||||
|
assert "## attachment_1.pdf" in body
|
||||||
|
assert "Position: oppose" in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_comment_is_idempotent(tmp_path: Path):
|
||||||
|
d = _setup_comment_dir(tmp_path)
|
||||||
|
first = extract_comment(d, inline_body="x")
|
||||||
|
mtime1 = first.stat().st_mtime
|
||||||
|
second = extract_comment(d, inline_body="x")
|
||||||
|
assert second == first
|
||||||
|
assert second.stat().st_mtime == mtime1 # no rewrite on second call
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_comment_force_rewrites(tmp_path: Path):
|
||||||
|
d = _setup_comment_dir(tmp_path)
|
||||||
|
extract_comment(d, inline_body="first")
|
||||||
|
extract_comment(d, inline_body="second", force=True)
|
||||||
|
body = (d / "combined.md").read_text()
|
||||||
|
assert "second" in body
|
||||||
|
assert "first" not in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_comment_no_attachments(tmp_path: Path):
|
||||||
|
d = tmp_path / "CMS-2024-0001" / "CMS-2024-0001-0099"
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
out = extract_comment(d, inline_body="Inline only.")
|
||||||
|
fm, body = parse_combined(out.read_text())
|
||||||
|
assert fm["attachments"] == []
|
||||||
|
assert "Inline only." in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_comment_handles_failed_attachment(tmp_path: Path):
|
||||||
|
d = tmp_path / "CMS-2024-0001" / "CMS-2024-0001-0100"
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"not a real pdf")
|
||||||
|
out = extract_comment(d, inline_body="x")
|
||||||
|
fm, _body = parse_combined(out.read_text())
|
||||||
|
assert fm["attachments"][0]["status"] == "failed"
|
||||||
44
tests/rex/comments/test_extract_docx.py
Normal file
44
tests/rex/comments/test_extract_docx.py
Normal file
@@ -0,0 +1,44 @@
|
|||||||
|
"""DOCX extraction via python-docx."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import docx
|
||||||
|
|
||||||
|
from rex.comments import extract_attachment
|
||||||
|
|
||||||
|
|
||||||
|
def _make_docx(path: Path, paragraphs: list[str]) -> Path:
|
||||||
|
d = docx.Document()
|
||||||
|
for p in paragraphs:
|
||||||
|
d.add_paragraph(p)
|
||||||
|
d.save(str(path))
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_docx_with_text(tmp_path: Path):
|
||||||
|
p = _make_docx(
|
||||||
|
tmp_path / "a.docx",
|
||||||
|
["First paragraph of comment letter.", "Second paragraph with details."],
|
||||||
|
)
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ok"
|
||||||
|
assert "First paragraph" in result.text
|
||||||
|
assert "Second paragraph" in result.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_docx_empty_marked_ocr_needed(tmp_path: Path):
|
||||||
|
p = _make_docx(tmp_path / "empty.docx", [])
|
||||||
|
result = extract_attachment(p)
|
||||||
|
# No paragraphs ⇒ 0 chars ⇒ ocr_needed (we treat empty docx the same as
|
||||||
|
# an image-only pdf — body lives somewhere we can't read it).
|
||||||
|
assert result.status == "ocr_needed"
|
||||||
|
assert result.chars <= 50
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_docx_corrupted_marked_failed(tmp_path: Path):
|
||||||
|
bad = tmp_path / "bad.docx"
|
||||||
|
bad.write_bytes(b"this is not a docx zip")
|
||||||
|
result = extract_attachment(bad)
|
||||||
|
assert result.status == "failed"
|
||||||
35
tests/rex/comments/test_extract_other.py
Normal file
35
tests/rex/comments/test_extract_other.py
Normal file
@@ -0,0 +1,35 @@
|
|||||||
|
"""Plain-text fallback + unsupported extensions."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from rex.comments import extract_attachment
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_txt(tmp_path: Path):
|
||||||
|
p = tmp_path / "note.txt"
|
||||||
|
p.write_text("This is a plain text comment longer than fifty characters total.")
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ok"
|
||||||
|
assert "plain text comment" in result.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_html_strips_tags(tmp_path: Path):
|
||||||
|
p = tmp_path / "page.html"
|
||||||
|
p.write_text(
|
||||||
|
"<html><body><p>This is a comment letter</p>"
|
||||||
|
"<p>with two paragraphs of substance.</p></body></html>"
|
||||||
|
)
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "ok"
|
||||||
|
assert "This is a comment letter" in result.text
|
||||||
|
assert "<p>" not in result.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_unknown_extension(tmp_path: Path):
|
||||||
|
p = tmp_path / "weird.xyz"
|
||||||
|
p.write_bytes(b"some bytes")
|
||||||
|
result = extract_attachment(p)
|
||||||
|
assert result.status == "unsupported"
|
||||||
|
assert result.chars == 0
|
||||||
Reference in New Issue
Block a user