merge: P47 comment pipeline docket seals + fingerprint-first skips — never redo known-complete work (refs #615 #680)

This commit is contained in:
kert
2026-09-08 16:06:38 -04:00
39 changed files with 4069 additions and 800 deletions

183
dev/scripts/dedupe_attachments.py Executable file
View File

@@ -0,0 +1,183 @@
#!/usr/bin/env python3
"""One-time cleanup of duplicate bib attachments (refs #615).
``Store.attach_file`` used to mint a new row + storage copy on every
call, so re-farms produced thousands of ``(item_id, filename)``
duplicates (33,379 groups / 66,917 of 78,017 rows on 2026-09-08). This
keeps the oldest row of each group, deletes the others' rows and their
storage copies, and prints a report. Dry-run by default.
Zotero is not touched: ``bib.sync`` already dedupes child attachments by
filename, so a duplicate that was synced once is a single Zotero child
attachment and stays valid.
Usage::
uv run python dev/scripts/dedupe_attachments.py # report only
uv run python dev/scripts/dedupe_attachments.py --apply # delete
"""
from __future__ import annotations
import argparse
import os
import sqlite3
import sys
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Group:
item_id: int
filename: str
keep: str
remove: list[str] = field(default_factory=list)
remove_paths: list[str] = field(default_factory=list)
conflict: bool = False # sizes differ → do not touch
@dataclass
class Report:
groups: int = 0
rows_removed: int = 0
files_removed: int = 0
bytes_freed: int = 0
skipped_conflicts: int = 0
removed_keys: list[str] = field(default_factory=list)
def plan(con: sqlite3.Connection) -> list[Group]:
con.row_factory = sqlite3.Row
rows = con.execute(
"""
SELECT a.item_id, a.filename, a.key, a.storage_path, a.rowid AS rid
FROM attachments a
WHERE (a.item_id, a.filename) IN (
SELECT item_id, filename FROM attachments
GROUP BY item_id, filename HAVING count(*) > 1
)
ORDER BY a.item_id, a.filename, a.rowid
"""
).fetchall()
groups: dict[tuple[int, str], Group] = {}
for r in rows:
gkey = (r["item_id"], r["filename"])
g = groups.get(gkey)
if g is None:
groups[gkey] = Group(
item_id=r["item_id"], filename=r["filename"], keep=r["key"]
)
continue
g.remove.append(r["key"])
g.remove_paths.append(r["storage_path"])
# conflict check: every copy must have the same size as the kept one
for g in groups.values():
keep_path = con.execute(
"SELECT storage_path FROM attachments WHERE key = ?", (g.keep,)
).fetchone()["storage_path"]
keep_size = _size(keep_path)
for p in g.remove_paths:
if _size(p) not in (keep_size, -1):
g.conflict = True
break
return list(groups.values())
def _size(path: str) -> int:
try:
return os.stat(path).st_size
except OSError:
return -1
def apply(con: sqlite3.Connection, groups: list[Group]) -> Report:
"""Delete the duplicate rows, then their storage copies.
Rows first, files second, with the COMMIT in between: a failure
mid-transaction rolls the rows back, and rows that still exist must
still have their files. Unlinking inside the transaction would leave
surviving rows pointing at nothing — the one outcome this cleanup
must never produce.
"""
rep = Report(groups=len(groups))
doomed: list[Path] = []
con.execute("BEGIN")
try:
for g in groups:
if g.conflict:
rep.skipped_conflicts += 1
continue
for key, path in zip(g.remove, g.remove_paths):
still_referenced = con.execute(
"SELECT count(*) FROM attachments WHERE storage_path = ? AND key <> ?",
(path, key),
).fetchone()[0]
con.execute("DELETE FROM attachments WHERE key = ?", (key,))
rep.rows_removed += 1
rep.removed_keys.append(key)
if not still_referenced:
doomed.append(Path(path))
con.execute("COMMIT")
except Exception:
con.execute("ROLLBACK")
raise
for p in doomed:
if p.is_file():
rep.bytes_freed += p.stat().st_size
p.unlink()
rep.files_removed += 1
try:
p.parent.rmdir() # the per-key dir, if now empty
except OSError:
pass
return rep
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
ap.add_argument(
"--db", default="", help="bib.sqlite path (default: stack.toml db.bib)"
)
ap.add_argument(
"--apply", action="store_true", help="delete duplicates (default: report only)"
)
args = ap.parse_args(argv)
if args.db:
db = args.db
else:
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "src"))
from conf import path
db = str(path("db.bib"))
con = sqlite3.connect(db, isolation_level=None)
try:
groups = plan(con)
conflicts = sum(1 for g in groups if g.conflict)
rows = sum(len(g.remove) for g in groups)
print(
f"{db}: {len(groups)} duplicate groups, {rows} rows to remove, "
f"{conflicts} conflicts (size mismatch, skipped)"
)
if not args.apply:
print("dry run — pass --apply to delete")
return 0
rep = apply(con, groups)
finally:
con.close()
print(
f"removed rows={rep.rows_removed} files={rep.files_removed} "
f"freed={rep.bytes_freed / 1e6:.1f} MB skipped_conflicts={rep.skipped_conflicts}"
)
if rep.removed_keys:
head = ", ".join(rep.removed_keys[:20])
more = " …" if len(rep.removed_keys) > 20 else ""
print(f"removed keys ({len(rep.removed_keys)}): {head}{more}")
print("reopen the store once (any `stack bib` command) to create the unique index")
return 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -22,24 +22,15 @@ step() { # step <label> <cmd...>: run, count failures, keep going
"$@" || { echo "WARN: $label rc=$?"; FAILURES=$((FAILURES+1)); } "$@" || { echo "WARN: $label rc=$?"; FAILURES=$((FAILURES+1)); }
} }
# Fast path first: the Mirrulations S3 mirror (#662) is a superset of the # Every step is incremental: sealed dockets are skipped outright, open
# live docket and has no rate cap (~1,500 comments/min vs the API's ~970/hr). # dockets pull from the stored watermark, extract/index compare cheap
# It creates comments the farm never listed, enriches, attaches and tags # fingerprints before doing any work. Mirror first (bulk, no cap), then
# them identically (reg-docket:/rule:/enriched:ok), so the index is usable # the API for anything the mirror lags on.
# minutes in rather than the next day.
step "mirror backfill" uv run stack bib backfill-comments --docket CMS-2026-2377 --mirror step "mirror backfill" uv run stack bib backfill-comments --docket CMS-2026-2377 --mirror
step "extract" uv run stack comments extract --docket CMS-2026-2377
# extract-ocr is a phase-2 stub (always exits 2, no --docket) — see #253.
# ocr_needed attachments stay flagged in combined.md frontmatter for now.
step "index" uv run stack llm index --collection comments --docket CMS-2026-2377
# Gap-fill from the reg.gov API for anything the mirror lags on. All
# resumable: the list walk upserts, the backfill skips enriched:ok, and
# extract/index only touch new or changed items.
step "api fetch" uv run stack bib fetch-pfs-comments --docket CMS-2026-2377 step "api fetch" uv run stack bib fetch-pfs-comments --docket CMS-2026-2377
step "api backfill" uv run stack bib backfill-comments --docket CMS-2026-2377 step "api backfill" uv run stack bib backfill-comments --docket CMS-2026-2377
step "extract (gap)" uv run stack comments extract --docket CMS-2026-2377 step "extract" uv run stack comments extract --docket CMS-2026-2377
step "index (gap)" uv run stack llm index --collection comments --docket CMS-2026-2377 step "index" uv run stack llm index --collection comments --docket CMS-2026-2377
COUNTS=$(uv run python - <<'EOF' COUNTS=$(uv run python - <<'EOF'
from conf import connect from conf import connect

View File

@@ -668,6 +668,82 @@ git commit -m "feat(bib): Store.upsert_status — identical re-upsert writes not
--- ---
### Task 3b: Data-preserving upsert — never blank a stored value
**Files:**
- Modify: `src/bib/store.py` (`upsert_status`)
- Test: `tests/bib/test_store_upsert_noop.py` (extend)
**Why (incident 2026-09-08):** the fetch list walk builds a `Source` whose `abstract` is empty (the list endpoint has no body) and re-upserts every comment it sees. With the old upsert that overwrote 17,607 enriched bodies in CMS-2026-2377 and 1,115 in CMS-2017-0092. Task 3's comparison alone would still call that "updated" and write the empty abstract. Upsert must merge, not replace: an empty incoming value carries no information.
**Interfaces:**
- Consumes: `Store.upsert_status` from Task 3.
- Produces: same signature; for every column in `_COMPARE_COLS` (except `extra_json`) an empty incoming string keeps the stored value. `extra_json` is compared/written as today (it is always a full serialization).
- [ ] **Step 1: Write the failing tests** (append to `tests/bib/test_store_upsert_noop.py`)
```python
def test_empty_incoming_abstract_keeps_stored_body():
"""A list-walk row (no body) must not blank an enriched comment."""
s = _store()
key, _ = s.upsert_status(_item(abstract="enriched body"))
_, status = s.upsert_status(_item(abstract=""))
assert status == "unchanged"
assert s.get(key).abstract == "enriched body"
def test_empty_incoming_title_keeps_stored_title():
s = _store()
it = _item(); it.title = "Org: CMS-2026-2377-1"
key, _ = s.upsert_status(it)
it2 = _item(); it2.title = ""
_, status = s.upsert_status(it2)
assert status == "unchanged"
assert s.get(key).title == "Org: CMS-2026-2377-1"
def test_non_empty_incoming_still_updates():
s = _store()
key, _ = s.upsert_status(_item(abstract="v1"))
_, status = s.upsert_status(_item(abstract="v2"))
assert status == "updated" and s.get(key).abstract == "v2"
```
- [ ] **Step 2: Run to verify failure**
Run: `uv run --no-sync pytest tests/bib/test_store_upsert_noop.py -q -p no:cacheprovider`
Expected: the first two new tests FAIL (`status == "updated"`, abstract/title blanked).
- [ ] **Step 3: Implement**
In `upsert_status`, after `cur_row = current.to_row()` and before the `same_cols` check:
```python
# Merge, don't replace: an empty incoming value carries no
# information (a list-walk row has no body), so the stored value
# wins. extra_json is always a full serialization and is exempt.
for c in self._COMPARE_COLS:
if c != "extra_json" and not new_row.get(c) and cur_row.get(c):
new_row[c] = cur_row[c]
setattr(item, c, cur_row[c])
```
`setattr(item, c, …)` keeps the later `item.to_row()` (used to build the written row) consistent with the merged values. `Item` is a pydantic model; plain attribute assignment is allowed on these fields.
- [ ] **Step 4: Run to verify pass**
Run: `uv run --no-sync pytest tests/bib -q -p no:cacheprovider`
Expected: pass.
- [ ] **Step 5: Commit**
```bash
git add src/bib/store.py tests/bib/test_store_upsert_noop.py
git commit -m "fix(bib): upsert never blanks a stored value — empty incoming columns keep the stored ones (refs #615)"
```
---
### Task 4: Idempotent `attach_file` ### Task 4: Idempotent `attach_file`
**Files:** **Files:**
@@ -3409,6 +3485,19 @@ uv run stack comments dockets # reopens the store
sqlite3 data/bib.sqlite "SELECT name FROM sqlite_master WHERE name='idx_attachments_item_filename'" sqlite3 data/bib.sqlite "SELECT name FROM sqlite_master WHERE name='idx_attachments_item_filename'"
``` ```
- [ ] **Step 2b: Recover the abstracts blanked on 2026-09-08**
The old upsert overwrote 17,607 enriched bodies in CMS-2026-2377 and 1,115 in CMS-2017-0092 with '' (incident in the SDD ledger). They still carry `enriched:ok`, so backfill skips them. Clear the tag on those rows and re-enrich from the mirror (idempotent attach + data-preserving upsert are in place by now):
```bash
for d in CMS-2026-2377 CMS-2017-0092; do
sqlite3 data/bib.sqlite "DELETE FROM item_tags WHERE tag_id=(SELECT id FROM tags WHERE name='enriched:ok') AND item_id IN (SELECT id FROM items WHERE url LIKE 'https://www.regulations.gov/comment/$d-%' AND coalesce(abstract,'')='')"
uv run stack bib backfill-comments --docket $d --mirror
done
sqlite3 data/bib.sqlite "SELECT substr(url,37,13), count(*), sum(coalesce(abstract,'')<>'') FROM items WHERE url LIKE 'https://www.regulations.gov/comment/CMS-2026-2377-%' OR url LIKE 'https://www.regulations.gov/comment/CMS-2017-0092-%' GROUP BY 1"
```
Expected: non-empty abstracts ≈ total for both dockets (a few genuinely empty bodies are fine). Then re-run `stack comments extract` and `stack llm index --collection comments` so any comment whose text was lost from combined.md/index is restored (the index still holds pre-wipe text; fingerprints change because `updated_at` moved, hash-skip absorbs unchanged text).
- [ ] **Step 3: Seal the historical dockets** - [ ] **Step 3: Seal the historical dockets**
```bash ```bash

86
src/bib/dockets.py Normal file
View File

@@ -0,0 +1,86 @@
"""Docket-level state for the regulations.gov comment pipeline.
A *docket* row (``dockets`` table in bib.sqlite) remembers what every
stage would otherwise re-derive from the network: the reg.gov document
that carries the comments, the comment close date, the pull watermark,
and — once the docket is known complete — a *seal*. Sealed dockets are
skipped by fetch, backfill, extract and index unless ``--force``.
Pure helpers only; the store owns persistence (``Store.docket_*``).
"""
from __future__ import annotations
import hashlib
from dataclasses import dataclass
from datetime import date, timedelta
from pathlib import Path
from typing import Iterable
DEFAULT_QUIET_DAYS = 30
@dataclass(frozen=True)
class Docket:
id: str
rule_cms_id: str = ""
fr_document_id: str = ""
fr_object_id: str = ""
comment_end_date: str = "" # YYYY-MM-DD
pull_watermark: str = "" # max lastModifiedDate seen on a clean walk
last_pull_at: str = "" # ISO-8601 UTC
last_pull_new: int | None = None
sealed_at: str = ""
seal_reason: str = ""
counts_json: str = "{}"
@property
def sealed(self) -> bool:
return bool(self.sealed_at)
def should_seal(docket: Docket, today: date, quiet_days: int) -> bool:
"""True when the comment period closed ≥ *quiet_days* ago and the
most recent completed pull, itself after that quiet boundary, found
nothing new."""
if docket.sealed or not docket.comment_end_date or not docket.last_pull_at:
return False
if docket.last_pull_new is None or docket.last_pull_new > 0:
return False
try:
end = date.fromisoformat(docket.comment_end_date[:10])
pulled = date.fromisoformat(docket.last_pull_at[:10])
except ValueError:
return False
boundary = end + timedelta(days=quiet_days)
return today >= boundary and pulled >= boundary
def fingerprint_files(paths: Iterable[Path]) -> str:
"""sha256 over sorted ``(name, size, mtime_ns)`` — cheap change
detection without reading contents. Missing files contribute their
name with size -1 so a deletion changes the fingerprint too."""
rows: list[str] = []
for p in paths:
p = Path(p)
try:
st = p.stat()
rows.append(f"{p.name}\x00{st.st_size}\x00{st.st_mtime_ns}")
except FileNotFoundError:
rows.append(f"{p.name}\x00-1\x000")
rows.sort()
return hashlib.sha256("\n".join(rows).encode()).hexdigest()
def _cfg():
from conf import cfg
return cfg
def quiet_days() -> int:
"""``[comments] seal_quiet_days`` from stack.toml, default 30."""
c = _cfg()
if "comments" in c and "seal_quiet_days" in c.comments:
return int(c.comments.seal_quiet_days)
return DEFAULT_QUIET_DAYS

View File

@@ -29,12 +29,14 @@ import os
import re import re
import time import time
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import dataclass, field from dataclasses import dataclass, field, replace
from datetime import date
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING, Iterator from typing import TYPE_CHECKING, Callable, Iterator
import httpx import httpx
from bib.dockets import Docket
from bib.item import Source from bib.item import Source
from bib.tag import Tag from bib.tag import Tag
@@ -195,7 +197,13 @@ class Client:
# ── Comment iteration ────────────────────────────────────── # ── Comment iteration ──────────────────────────────────────
def iter_comments(self, object_id: str) -> Iterator[Comment]: def iter_comments(
self,
object_id: str,
*,
since: str = "",
on_error: "Callable[[Exception], None] | None" = None,
) -> Iterator[Comment]:
"""Yield every comment against a single FR document. """Yield every comment against a single FR document.
The ``commentOnId`` filter on ``/comments`` takes a document's The ``commentOnId`` filter on ``/comments`` takes a document's
@@ -211,8 +219,14 @@ class Client:
2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` — 2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` —
returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space- returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space-
separated, no timezone suffix). Normalize before sending. separated, no timezone suffix). Normalize before sending.
*since* seeds the ``lastModifiedDate`` cursor so an incremental
walk starts where the last clean one ended (the boundary row is
re-yielded; the store's no-op upsert absorbs it). *on_error* is
invoked with the ``HTTPStatusError`` before the walk stops, so a
caller can tell "walked to the end" from "gave up".
""" """
cursor: str | None = None cursor: str | None = _reg_date(since) if since else None
page = 1 page = 1
while True: while True:
params: dict[str, str] = { params: dict[str, str] = {
@@ -233,6 +247,8 @@ class Client:
page, page,
cursor, cursor,
) )
if on_error is not None:
on_error(e)
break break
rows = data.get("data", []) rows = data.get("data", [])
if not rows: if not rows:
@@ -380,6 +396,7 @@ def backfill_details(
client: Client, client: Client,
*, *,
docket: str = "", docket: str = "",
force: bool = False,
limit: int | None = None, limit: int | None = None,
log_path: Path | None = None, log_path: Path | None = None,
commit_every: int = 25, commit_every: int = 25,
@@ -401,6 +418,27 @@ def backfill_details(
so we stop hammering it. Commits land every ``commit_every`` items so we stop hammering it. Commits land every ``commit_every`` items
so a crash loses at most that many items of work. so a crash loses at most that many items of work.
""" """
if docket and not force:
d = store.docket_get(docket)
if d is not None and d.sealed:
log.info(
"%s sealed %s (%s); backfill skipped",
docket,
d.sealed_at[:10],
d.seal_reason,
)
print(
f"{docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipped — use --force",
flush=True,
)
return {
"skipped_sealed": 1,
"seen": 0,
"enriched": 0,
"created": 0,
"attached": 0,
"errors": 0,
}
con = store._con() # noqa: SLF001 con = store._con() # noqa: SLF001
# Resume rule: treat ``enriched:ok`` as the completion marker. # Resume rule: treat ``enriched:ok`` as the completion marker.
# Using a tag (not just the abstract) lets us handle "see attached" # Using a tag (not just the abstract) lets us handle "see attached"
@@ -659,6 +697,7 @@ def backfill_from_mirror(
mirror: Mirror, mirror: Mirror,
*, *,
docket: str, docket: str,
force: bool = False,
limit: int | None = None, limit: int | None = None,
log_path: Path | None = None, log_path: Path | None = None,
commit_every: int = 200, commit_every: int = 200,
@@ -672,6 +711,27 @@ def backfill_from_mirror(
S3 fetches fan out over a thread pool; every Store/sqlite call stays S3 fetches fan out over a thread pool; every Store/sqlite call stays
on this thread. Idempotent: ``enriched:ok`` items are skipped. on this thread. Idempotent: ``enriched:ok`` items are skipped.
""" """
if docket and not force:
d = store.docket_get(docket)
if d is not None and d.sealed:
log.info(
"%s sealed %s (%s); backfill skipped",
docket,
d.sealed_at[:10],
d.seal_reason,
)
print(
f"{docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipped — use --force",
flush=True,
)
return {
"skipped_sealed": 1,
"seen": 0,
"enriched": 0,
"created": 0,
"attached": 0,
"errors": 0,
}
con = store._con() # noqa: SLF001 con = store._con() # noqa: SLF001
existing: dict[str, tuple[str, bool]] = {} # cid -> (key, enriched) existing: dict[str, tuple[str, bool]] = {} # cid -> (key, enriched)
for key, url, enriched in con.execute( for key, url, enriched in con.execute(
@@ -784,6 +844,19 @@ def upsert_comment(
extra_tags: list[str] | None = None, extra_tags: list[str] | None = None,
) -> str: ) -> str:
"""Upsert the comment as a Source item, return bib key.""" """Upsert the comment as a Source item, return bib key."""
return upsert_comment_status(store, comment, cms_id=cms_id, extra_tags=extra_tags)[
0
]
def upsert_comment_status(
store: Store,
comment: Comment,
*,
cms_id: str = "",
extra_tags: list[str] | None = None,
) -> tuple[str, str]:
"""Like :func:`upsert_comment`, also returning created/updated/unchanged."""
# Title = the comment's own CMS-DOCKET-XXXX-NNNN id, optionally # Title = the comment's own CMS-DOCKET-XXXX-NNNN id, optionally
# prefixed with the byline (org or first/last name) when the # prefixed with the byline (org or first/last name) when the
# commenter filled those fields in. The previous "Comment on # commenter filled those fields in. The previous "Comment on
@@ -816,7 +889,146 @@ def upsert_comment(
for t in tags: for t in tags:
item.add_tag(t) item.add_tag(t)
return store.upsert(item) return store.upsert_status(item)
# ── Docket walks ───────────────────────────────────────────────
def discover_docket(client: Client, docket_id: str, *, rule_cms_id: str = "") -> Docket:
"""One ``/documents`` listing → the docket's commentable document.
Picks the document that has an ``objectId`` and a ``commentEndDate``
(latest close date wins when several qualify). Called once per
docket; the result is persisted so later runs make no API call.
"""
best: dict | None = None
for fr_doc in client.find_documents_in_docket(docket_id):
attrs = fr_doc.get("attributes") or {}
if not attrs.get("objectId") or not attrs.get("commentEndDate"):
continue
if (
best is None
or attrs["commentEndDate"]
> (best.get("attributes") or {})["commentEndDate"]
):
best = fr_doc
attrs = (best or {}).get("attributes") or {}
return Docket(
id=docket_id,
rule_cms_id=rule_cms_id,
fr_document_id=(best or {}).get("id", ""),
fr_object_id=attrs.get("objectId", ""),
comment_end_date=(attrs.get("commentEndDate") or "")[:10],
)
@dataclass
class WalkResult:
created: int = 0
updated: int = 0
unchanged: int = 0
watermark: str = ""
clean: bool = True
sealed: bool = False
def walk_docket(
store: Store,
client: Client,
docket: Docket,
*,
attachments: bool = False,
limit: int = 0,
force: bool = False,
quiet_days: int = 30,
today: date | None = None,
scratch_root: Path = Path(".state/comments"),
echo: Callable[[str], None] | None = None,
) -> WalkResult:
"""Pull *docket*'s comments from the stored watermark, upsert them,
advance the watermark on a clean walk, and auto-seal when
:func:`should_seal` says so. ``force`` walks from page 1 and never
seals. Returns per-status counts.
Attachments are fetched for created/updated comments, and for every
comment under ``force`` (an unchanged row can still be missing its
files). A docket is only auto-sealed when bib actually holds at
least one of its comments.
"""
from bib.dockets import should_seal
res = WalkResult()
since = "" if force else docket.pull_watermark
max_lm = docket.pull_watermark if not force else ""
n = 0
def _err(_e: Exception) -> None:
res.clean = False
scratch = scratch_root / docket.id
for c in client.iter_comments(docket.fr_object_id, since=since, on_error=_err):
if limit and n >= limit:
res.clean = False # a capped walk is not a complete one
break
key, status = upsert_comment_status(
store, c, cms_id=docket.rule_cms_id, extra_tags=[f"reg-docket:{docket.id}"]
)
setattr(res, status, getattr(res, status) + 1)
if attachments and c.attachment_count and (force or status != "unchanged"):
for att in client.attachments_for(c.id):
path = client.download_attachment(att.url, scratch / c.id)
if path:
store.attach_file(key, path, title=att.filename)
lm = (c.raw.get("attributes") or {}).get("lastModifiedDate") or ""
if lm > max_lm:
max_lm = lm
n += 1
if n % 50 == 0:
store._con().commit() # noqa: SLF001
if echo:
echo(f" {n} comments")
store._con().commit() # noqa: SLF001
res.watermark = max_lm
if res.clean and not force:
from datetime import datetime, timezone
# ``today`` doubles as a deterministic override of "now" for
# tests exercising the auto-seal quiet-window boundary; real
# callers leave it unset and get the actual wall-clock time.
if today is not None:
now = f"{today.isoformat()}T00:00:00Z"
else:
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
updated = replace(
docket, pull_watermark=max_lm, last_pull_at=now, last_pull_new=res.created
)
store.docket_upsert(updated)
counts = docket_counts(store, docket.id)
# A docket with nothing in bib is a farm that never landed, not a
# finished one: an empty walk over it must not seal it shut.
if counts["comments"] and should_seal(
updated, today or date.today(), quiet_days
):
store.docket_seal(docket.id, reason="auto", counts=counts)
res.sealed = True
return res
def docket_counts(store: Store, docket_id: str) -> dict[str, int]:
"""``{"comments": n, "enriched": n}`` from SQL only (no filesystem)."""
con = store._con() # noqa: SLF001
like = f"https://www.regulations.gov/comment/{docket_id}-%"
comments = con.execute(
"SELECT count(*) FROM items WHERE url LIKE ?", (like,)
).fetchone()[0]
enriched = con.execute(
"""SELECT count(*) FROM items i WHERE i.url LIKE ? AND i.id IN (
SELECT item_id FROM item_tags WHERE tag_id IN (SELECT id FROM tags WHERE name='enriched:ok'))""",
(like,),
).fetchone()[0]
return {"comments": comments, "enriched": enriched}
# ── Internals ────────────────────────────────────────────────── # ── Internals ──────────────────────────────────────────────────

View File

@@ -137,3 +137,25 @@ CREATE TABLE IF NOT EXISTS fr_links (
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')), created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')),
UNIQUE (item_key, url) UNIQUE (item_key, url)
); );
-- ── Dockets (P47) ────────────────────────────────────────────────
-- One row per reg.gov docket: what every stage would otherwise re-fetch
-- (document/object ids, close date), the pull watermark, and the seal
-- that marks the docket complete. See bib/dockets.py.
CREATE TABLE IF NOT EXISTS dockets (
id TEXT PRIMARY KEY,
rule_cms_id TEXT NOT NULL DEFAULT '',
fr_document_id TEXT NOT NULL DEFAULT '',
fr_object_id TEXT NOT NULL DEFAULT '',
comment_end_date TEXT NOT NULL DEFAULT '',
pull_watermark TEXT NOT NULL DEFAULT '',
last_pull_at TEXT NOT NULL DEFAULT '',
last_pull_new INTEGER,
sealed_at TEXT NOT NULL DEFAULT '',
seal_reason TEXT NOT NULL DEFAULT '',
counts_json TEXT NOT NULL DEFAULT '{}'
);
-- upsert() dedupes by URL on every comment; without this every upsert
-- was a full scan of items.
CREATE INDEX IF NOT EXISTS idx_items_url ON items(url);

View File

@@ -17,6 +17,7 @@ Usage::
from __future__ import annotations from __future__ import annotations
import json
import mimetypes import mimetypes
import random import random
import shutil import shutil
@@ -25,6 +26,7 @@ from importlib import resources
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from bib.dockets import Docket
from bib.item import Item from bib.item import Item
@@ -81,6 +83,19 @@ class Store:
def _init_schema(self) -> None: def _init_schema(self) -> None:
ddl = resources.files("bib").joinpath("schema.sql").read_text() ddl = resources.files("bib").joinpath("schema.sql").read_text()
self._con().executescript(ddl) self._con().executescript(ddl)
# The (item_id, filename) unique index can only exist once the
# dedupe migration has run (dev/scripts/dedupe_attachments.py);
# until then we leave it off rather than fail to open the store.
con = self._con()
dup = con.execute(
"SELECT 1 FROM attachments GROUP BY item_id, filename HAVING count(*) > 1 LIMIT 1"
).fetchone()
if dup is None:
con.execute(
"CREATE UNIQUE INDEX IF NOT EXISTS idx_attachments_item_filename "
"ON attachments(item_id, filename)"
)
con.commit()
def close(self) -> None: def close(self) -> None:
if self._connection is not None: if self._connection is not None:
@@ -195,6 +210,11 @@ class Store:
self._sync_tags(item_id, new_tags) self._sync_tags(item_id, new_tags)
if new_collections is not None: if new_collections is not None:
self._sync_collections(item_id, new_collections) self._sync_collections(item_id, new_collections)
if not fields:
# A tags/collections-only rewrite still changes the item
# as the indexer sees it (year:/project:/cms-rule: land in
# chunk metadata), and nothing else stamped the row.
self._stamp(key)
con.commit() con.commit()
@@ -204,6 +224,17 @@ class Store:
con.execute("DELETE FROM items WHERE key = ?", (key,)) con.execute("DELETE FROM items WHERE key = ?", (key,))
con.commit() con.commit()
_COMPARE_COLS = (
"item_type",
"title",
"url",
"date_published",
"abstract",
"institution",
"extra",
"extra_json",
)
def upsert( def upsert(
self, self,
item: Item, item: Item,
@@ -212,37 +243,86 @@ class Store:
collection: str = "", collection: str = "",
) -> str: ) -> str:
"""Create or update an item, deduplicating by URL.""" """Create or update an item, deduplicating by URL."""
if item.url: return self.upsert_status(item, tags=tags, collection=collection)[0]
con = self._con()
existing = con.execute(
"SELECT key FROM items WHERE url = ?", (item.url,)
).fetchone()
if existing:
ekey = existing["key"]
if tags:
for tag in tags:
label = tag.label if hasattr(tag, "label") else str(tag)
item.add_tag(label)
if collection and collection not in item.collections:
item.collections.append(collection)
item.stamp_access()
# Merge with what's already stored — a re-ingest must
# never clobber tags/collections curated on the row
# since the last ingest (#624: the CY2027 NPRM lost its
# sup: registration tag to this and silently vanished
# from the tag-scoped Zotero sync). Deliberate removal
# goes through remove_tag, not upsert.
current = self.get(ekey)
row = item.to_row()
row.pop("key", None)
row["tags"] = list(dict.fromkeys([*current.tags, *item.tags]))
row["collections"] = list(
dict.fromkeys([*current.collections, *item.collections])
)
self.update(ekey, **row)
return ekey
return self.create(item, tags=tags, collection=collection) def upsert_status(
self,
item: Item,
*,
tags: list[Any] | None = None,
collection: str = "",
) -> tuple[str, str]:
"""Like :meth:`upsert` but also report what happened:
``"created"``, ``"updated"`` or ``"unchanged"``.
``unchanged`` means every compared column, the merged tag set and
the merged collection set already match the stored row — nothing
is written, no ``access_date``/``updated_at`` stamp, no tag row
churn. A re-farm over a complete docket is therefore a read-only
pass over ``items``.
Merge rules, all of which make ``unchanged`` a comparison against
the *merged* row rather than the incoming one:
* columns: an empty incoming value keeps the stored one (a
list-walk row carries no body, and must not blank an enriched
abstract). ``extra_json`` is exempt — it is always a full
serialization, so an incoming ``{}`` is a real value and
replaces what is stored.
* tags and collections: union-merged with what is stored, never
replaced. Removal goes through :meth:`remove_tag`.
"""
if tags:
for tag in tags:
label = tag.label if hasattr(tag, "label") else str(tag)
item.add_tag(label)
if collection and collection not in item.collections:
item.collections.append(collection)
if not item.url:
return self.create(item), "created"
con = self._con()
existing = con.execute(
"SELECT key FROM items WHERE url = ?", (item.url,)
).fetchone()
if not existing:
return self.create(item), "created"
ekey = existing["key"]
# Merge with what's already stored — a re-ingest must never
# clobber tags/collections curated on the row since the last
# ingest (#624). Deliberate removal goes through remove_tag.
current = self.get(ekey)
merged_tags = list(dict.fromkeys([*current.tags, *item.tags]))
merged_cols = list(dict.fromkeys([*current.collections, *item.collections]))
new_row = item.to_row()
cur_row = current.to_row()
# Merge, don't replace: an empty incoming value carries no
# information (a list-walk row has no body), so the stored value
# wins. extra_json is always a full serialization and is exempt.
for c in self._COMPARE_COLS:
if c != "extra_json" and not new_row.get(c) and cur_row.get(c):
new_row[c] = cur_row[c]
setattr(item, c, cur_row[c])
same_cols = all(
new_row.get(c, "") == cur_row.get(c, "") for c in self._COMPARE_COLS
)
if (
same_cols
and set(merged_tags) == set(current.tags)
and set(merged_cols) == set(current.collections)
):
return ekey, "unchanged"
item.stamp_access()
row = item.to_row()
row.pop("key", None)
row["tags"] = merged_tags
row["collections"] = merged_cols
self.update(ekey, **row)
return ekey, "updated"
# ── Query ──────────────────────────────────────────────────── # ── Query ────────────────────────────────────────────────────
@@ -452,6 +532,19 @@ class Store:
(row["id"], item_id), (row["id"], item_id),
) )
def _stamp(self, item_key: str) -> None:
"""Mark the item row as changed *now*.
Tags carry indexed metadata (``year:``, ``project:``,
``cms-rule:``), so a tag-only edit has to move ``updated_at`` or
the llm indexer's fingerprint can never see it (#615).
"""
self._con().execute(
"UPDATE items SET updated_at = strftime('%Y-%m-%dT%H:%M:%SZ','now') "
"WHERE key = ?",
(item_key,),
)
def add_tag(self, item_key: str, tag: str) -> None: def add_tag(self, item_key: str, tag: str) -> None:
"""Add a single tag to an item.""" """Add a single tag to an item."""
con = self._con() con = self._con()
@@ -459,21 +552,25 @@ class Store:
if row is None: if row is None:
raise KeyError(f"Item not found: {item_key}") raise KeyError(f"Item not found: {item_key}")
tag_id = self._ensure_tag(tag) tag_id = self._ensure_tag(tag)
con.execute( cur = con.execute(
"INSERT OR IGNORE INTO item_tags (item_id, tag_id) VALUES (?, ?)", "INSERT OR IGNORE INTO item_tags (item_id, tag_id) VALUES (?, ?)",
(row["id"], tag_id), (row["id"], tag_id),
) )
if cur.rowcount: # re-tagging an already-tagged item changes nothing
self._stamp(item_key)
con.commit() con.commit()
def remove_tag(self, item_key: str, tag: str) -> None: def remove_tag(self, item_key: str, tag: str) -> None:
"""Remove a tag from an item.""" """Remove a tag from an item."""
con = self._con() con = self._con()
con.execute( cur = con.execute(
"""DELETE FROM item_tags """DELETE FROM item_tags
WHERE item_id = (SELECT id FROM items WHERE key = ?) WHERE item_id = (SELECT id FROM items WHERE key = ?)
AND tag_id = (SELECT id FROM tags WHERE name = ?)""", AND tag_id = (SELECT id FROM tags WHERE name = ?)""",
(item_key, tag), (item_key, tag),
) )
if cur.rowcount: # the tag wasn't there — nothing changed
self._stamp(item_key)
con.commit() con.commit()
def list_tags(self, *, namespace: str = "") -> list[dict[str, int]]: def list_tags(self, *, namespace: str = "") -> list[dict[str, int]]:
@@ -497,6 +594,93 @@ class Store:
).fetchall() ).fetchall()
return [{"name": r["name"], "count": r["cnt"]} for r in rows] return [{"name": r["name"], "count": r["cnt"]} for r in rows]
# ── Dockets ──────────────────────────────────────────────────
_DOCKET_COLS = (
"id",
"rule_cms_id",
"fr_document_id",
"fr_object_id",
"comment_end_date",
"pull_watermark",
"last_pull_at",
"last_pull_new",
"sealed_at",
"seal_reason",
"counts_json",
)
def _docket_from_row(self, row: sqlite3.Row | None) -> Docket | None:
if row is None:
return None
d = {c: row[c] for c in self._DOCKET_COLS}
return Docket(**d)
def docket_get(self, docket_id: str) -> Docket | None:
row = (
self._con()
.execute("SELECT * FROM dockets WHERE id = ?", (docket_id,))
.fetchone()
)
return self._docket_from_row(row)
def docket_for_rule(self, cms_rule_id: str) -> Docket | None:
row = (
self._con()
.execute(
"SELECT * FROM dockets WHERE rule_cms_id = ? ORDER BY id LIMIT 1",
(cms_rule_id,),
)
.fetchone()
)
return self._docket_from_row(row)
def docket_upsert(self, docket: Docket) -> None:
cols = ", ".join(self._DOCKET_COLS)
marks = ", ".join(f":{c}" for c in self._DOCKET_COLS)
sets = ", ".join(f"{c} = excluded.{c}" for c in self._DOCKET_COLS if c != "id")
con = self._con()
con.execute(
f"INSERT INTO dockets ({cols}) VALUES ({marks}) " # noqa: S608
f"ON CONFLICT(id) DO UPDATE SET {sets}",
{c: getattr(docket, c) for c in self._DOCKET_COLS},
)
con.commit()
def dockets(self) -> list[Docket]:
rows = self._con().execute("SELECT * FROM dockets ORDER BY id").fetchall()
return [self._docket_from_row(r) for r in rows]
def sealed_dockets(self) -> dict[str, str]:
"""docket id → sealed_at for every sealed docket."""
rows = (
self._con()
.execute("SELECT id, sealed_at FROM dockets WHERE sealed_at <> ''")
.fetchall()
)
return {r["id"]: r["sealed_at"] for r in rows}
def docket_seal(
self, docket_id: str, *, reason: str, counts: dict[str, int]
) -> None:
from datetime import datetime, timezone
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
con = self._con()
con.execute(
"UPDATE dockets SET sealed_at = ?, seal_reason = ?, counts_json = ? WHERE id = ?",
(now, reason, json.dumps(counts, sort_keys=True), docket_id),
)
con.commit()
def docket_unseal(self, docket_id: str) -> None:
con = self._con()
con.execute(
"UPDATE dockets SET sealed_at = '', seal_reason = '' WHERE id = ?",
(docket_id,),
)
con.commit()
# ── Attachments & Notes ────────────────────────────────────── # ── Attachments & Notes ──────────────────────────────────────
def attach_file(self, item_key: str, path: Path, *, title: str = "") -> str: def attach_file(self, item_key: str, path: Path, *, title: str = "") -> str:
@@ -506,6 +690,14 @@ class Store:
if row is None: if row is None:
raise KeyError(f"Item not found: {item_key}") raise KeyError(f"Item not found: {item_key}")
filename = title or path.name
dup = con.execute(
"SELECT key FROM attachments WHERE item_id = ? AND filename = ?",
(row["id"], filename),
).fetchone()
if dup:
return dup["key"]
att_key = _generate_key() att_key = _generate_key()
dest_dir = self._storage / att_key dest_dir = self._storage / att_key
dest_dir.mkdir(parents=True, exist_ok=True) dest_dir.mkdir(parents=True, exist_ok=True)
@@ -521,7 +713,7 @@ class Store:
( (
row["id"], row["id"],
att_key, att_key,
title or path.name, filename,
content_type, content_type,
# resolve(): through a symlinked storage dir (a git # resolve(): through a symlinked storage dir (a git
# worktree's data/ pointing at the real data dir) the # worktree's data/ pointing at the real data dir) the

View File

@@ -126,9 +126,29 @@ def discover_pfs_rules(
store.upsert(rule) store.upsert(rule)
def _walk_and_report(store, api, docket, *, attachments, limit, force, echo) -> None:
from bib.dockets import quiet_days
from bib.regulations_gov import walk_docket
r = walk_docket(
store,
api,
docket,
attachments=attachments,
limit=limit,
force=force,
quiet_days=quiet_days(),
echo=echo,
)
echo(
f" {docket.id}: created={r.created} updated={r.updated} "
f"unchanged={r.unchanged} clean={r.clean}" + (" — sealed" if r.sealed else "")
)
@app.command(name="fetch-docket-comments") @app.command(name="fetch-docket-comments")
def fetch_docket_comments( def fetch_docket_comments(
docket: str = typer.Argument(..., help="e.g. CMS-1676-P"), docket: str = typer.Argument(..., help="reg.gov docket id, e.g. CMS-2026-2377"),
cms_id: str = typer.Option( cms_id: str = typer.Option(
"", "--cms-id", help="Associate comments with a specific CMS rule tag." "", "--cms-id", help="Associate comments with a specific CMS rule tag."
), ),
@@ -138,52 +158,54 @@ def fetch_docket_comments(
attachments: bool = typer.Option( attachments: bool = typer.Option(
False, False,
"--attachments", "--attachments",
help="Also download each comment's PDF/DOCX attachments " help="Also download each comment's PDF/DOCX attachments.",
"(expensive — skim first without).",
), ),
sleep: float = typer.Option( sleep: float = typer.Option(
1.3, "--sleep", help="Seconds between API calls (rate budget)." 1.3, "--sleep", help="Seconds between API calls (rate budget)."
), ),
force: bool = typer.Option(
False,
"--force",
help="Walk from page 1 even if sealed / watermarked. Never unseals.",
),
) -> None: ) -> None:
"""Walk every comment on one docket; upsert as Source items. """Walk one docket's comments from its stored watermark; upsert as Source items.
Policy: we iterate every FR document indexed under the docket and The first run discovers the docket's commentable FR document (one
pull their comments. For one PFS rule that usually means one parent listing call) and stores it; later runs make no discovery calls.
document with a few thousand to tens of thousands of comments. Sealed dockets are skipped unless --force.
""" """
from pathlib import Path
from bib import connect from bib import connect
from bib.regulations_gov import Client, upsert_comment from bib.regulations_gov import Client, discover_docket
store = connect() store = connect()
scratch = Path(f".state/comments/{docket}")
with Client(sleep=sleep) as api: with Client(sleep=sleep) as api:
docs = api.find_documents_in_docket(docket) d = store.docket_get(docket)
typer.echo(f" docket {docket}: {len(docs)} FR documents indexed") if d is None:
total = 0 d = discover_docket(api, docket, rule_cms_id=cms_id)
for fr_doc in docs: store.docket_upsert(d)
attrs = fr_doc.get("attributes") or {} elif cms_id and not d.rule_cms_id:
fr_id = fr_doc["id"] from dataclasses import replace
object_id = attrs.get("objectId")
if not object_id or not attrs.get("commentEndDate"): d = replace(d, rule_cms_id=cms_id)
continue store.docket_upsert(d)
typer.echo(f" ↓ comments on {fr_id} (objectId={object_id})") if d.sealed and not force:
for c in api.iter_comments(object_id): typer.echo(
if limit and total >= limit: f" {docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipping — use --force to re-walk"
return )
key = upsert_comment(store, c, cms_id=cms_id) return
if attachments and c.attachment_count: if not d.fr_object_id:
for att in api.attachments_for(c.id): typer.echo(f" {docket}: no commentable FR document found")
path = api.download_attachment(att.url, scratch / c.id) return
if path: _walk_and_report(
store.attach_file(key, path, title=att.filename) store,
total += 1 api,
if total % 50 == 0: d,
typer.echo(f" processed {total} comments") attachments=attachments,
store._con().commit() # noqa: SLF001 limit=limit,
typer.echo(f" total: {total} comments upserted") force=force,
echo=typer.echo,
)
@app.command(name="fetch-pfs-comments") @app.command(name="fetch-pfs-comments")
@@ -192,46 +214,59 @@ def fetch_pfs_comments(
until: str = typer.Option("", "--until"), until: str = typer.Option("", "--until"),
attachments: bool = typer.Option(False, "--attachments"), attachments: bool = typer.Option(False, "--attachments"),
per_docket_limit: int = typer.Option( per_docket_limit: int = typer.Option(
0, 0, "--per-docket-limit", help="Cap comments fetched per docket. 0 = unlimited."
"--per-docket-limit",
help="Cap comments fetched per docket. 0 = unlimited.",
), ),
sleep: float = typer.Option(1.3, "--sleep"), sleep: float = typer.Option(1.3, "--sleep"),
docket: str = typer.Option( docket: str = typer.Option(
"", "",
"--docket", "--docket",
help="Only pull this reg.gov docket (e.g. CMS-2026-2377); " help="Only pull this reg.gov docket (e.g. CMS-2026-2377); other dockets are skipped before any API call once known.",
"the other PFS dockets are skipped without any API calls.", ),
force: bool = typer.Option(
False,
"--force",
help="Walk every docket from page 1, sealed or not. Never unseals.",
), ),
) -> None: ) -> None:
"""One-shot: discover every PFS rule since *since* and pull every """Discover every PFS proposed rule since *since* and pull new comments
comment on each of their dockets. on each docket from its stored watermark.
Heavy run. A single PFS proposed rule can hold 5K–25K comments; Sealed dockets (comment period closed + quiet period + an empty pull)
times ~20 rules and with attachments this runs for many hours cost nothing: no rule-metadata fetch, no resolve call, no walk.
under the reg.gov rate budget. Use ``--per-docket-limit`` to
smoke-test first.
""" """
from pathlib import Path
from bib import connect from bib import connect
from bib.federalregister import pfs_rules, split_docket_ids from bib.federalregister import pfs_rules, split_docket_ids
from bib.regulations_gov import Client, upsert_comment from bib.regulations_gov import Client, discover_docket
from bib.tag import Tag from bib.tag import Tag
from bib.translate import federal_register from bib.translate import federal_register
store = connect() store = connect()
rules = pfs_rules(since=since, until=until or None) rules = pfs_rules(since=since, until=until or None)
typer.echo(f"==> {len(rules)} PFS rules since {since}") typer.echo(f"==> {len(rules)} PFS rules since {since}")
# Only proposed rules have public comment periods — skip finals +
# corrections to keep the work focused.
proposed = [r for r in rules if r.type == "Proposed Rule"] proposed = [r for r in rules if r.type == "Proposed Rule"]
typer.echo(f" {len(proposed)} proposed rules (the ones with comments)") typer.echo(f" {len(proposed)} proposed rules (the ones with comments)")
with Client(sleep=sleep) as api: with Client(sleep=sleep) as api:
for doc in proposed: for doc in proposed:
cms_ids = split_docket_ids(doc.dockets) cms_ids = split_docket_ids(doc.dockets)
known = {cid: store.docket_for_rule(cid) for cid in cms_ids}
# Decide what this rule needs BEFORE touching the network.
def _wanted(cid: str) -> bool:
d = known[cid]
if docket and d is not None and d.id != docket:
return False
if d is not None and d.sealed and not force:
typer.echo(
f" {d.id}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipping"
)
return False
return True
todo = [cid for cid in cms_ids if _wanted(cid)]
if not todo:
continue
try: try:
rule = federal_register( rule = federal_register(
doc.html_url doc.html_url
@@ -244,58 +279,33 @@ def fetch_pfs_comments(
rule.add_tag("module:pfs") rule.add_tag("module:pfs")
for cid in cms_ids: for cid in cms_ids:
rule.add_tag(f"cms-rule:{cid}") rule.add_tag(f"cms-rule:{cid}")
store.upsert(rule)
# Resolve each CMS-XXXX-P to its reg.gov docket id. for cms_id in todo:
for cms_id in cms_ids: d = known[cms_id]
reg_docket = api.resolve_docket(cms_id) if d is None:
if not reg_docket: reg_docket = api.resolve_docket(cms_id)
typer.echo(f" skip {cms_id}: no reg.gov docket found") if not reg_docket:
continue typer.echo(f" skip {cms_id}: no reg.gov docket found")
if docket and reg_docket != docket: continue
continue if docket and reg_docket != docket:
typer.echo(f" {cms_id} → {reg_docket} ({doc.publication_date})") continue
rule.add_tag(f"reg-docket:{reg_docket}") d = discover_docket(api, reg_docket, rule_cms_id=cms_id)
store.docket_upsert(d)
typer.echo(f" {cms_id} → {d.id} ({doc.publication_date})")
rule.add_tag(f"reg-docket:{d.id}")
store.upsert(rule) store.upsert(rule)
if not d.fr_object_id:
fr_docs = api.find_documents_in_docket(reg_docket) typer.echo(f" {d.id}: no commentable FR document; skipping")
per_docket_count = 0 continue
scratch = Path(f".state/comments/{reg_docket}") _walk_and_report(
for fr_doc in fr_docs: store,
attrs = fr_doc.get("attributes") or {} api,
object_id = attrs.get("objectId") d,
# Skip docs with no objectId or no real comment attachments=attachments,
# window — final rules and internal display versions limit=per_docket_limit,
# don't carry meaningful comment traffic. force=force,
if not object_id: echo=typer.echo,
continue )
if not attrs.get("commentEndDate"):
continue
for c in api.iter_comments(object_id):
if per_docket_limit and per_docket_count >= per_docket_limit:
break
key = upsert_comment(
store,
c,
cms_id=cms_id,
extra_tags=[f"reg-docket:{reg_docket}"],
)
if attachments and c.attachment_count:
for att in api.attachments_for(c.id):
path = api.download_attachment(
att.url,
scratch / c.id,
)
if path:
store.attach_file(key, path, title=att.filename)
per_docket_count += 1
if per_docket_count % 50 == 0:
typer.echo(f" {per_docket_count} comments")
store._con().commit() # noqa: SLF001
if per_docket_limit and per_docket_count >= per_docket_limit:
break
store._con().commit() # noqa: SLF001
typer.echo(f" docket total: {per_docket_count}")
@app.command(name="ingest-mail") @app.command(name="ingest-mail")
@@ -378,6 +388,14 @@ def backfill_comments(
"reg.gov API — bulk speed, no rate cap, also ingests comments " "reg.gov API — bulk speed, no rate cap, also ingests comments "
"the original farm never listed. Requires --docket.", "the original farm never listed. Requires --docket.",
), ),
force: bool = typer.Option(
False,
"--force",
help="Run even if the docket is sealed. Bypasses the docket seal "
"only: items already tagged enriched:ok (or enriched:gone) are "
"still skipped, so this resumes an interrupted enrichment — it "
"never re-enriches an item.",
),
) -> None: ) -> None:
"""Enrich every reg-gov comment stub with body + attachments. """Enrich every reg-gov comment stub with body + attachments.
@@ -403,6 +421,7 @@ def backfill_comments(
store, store,
Mirror(), Mirror(),
docket=docket, docket=docket,
force=force,
limit=limit or None, limit=limit or None,
log_path=log_path, log_path=log_path,
) )
@@ -414,6 +433,7 @@ def backfill_comments(
store, store,
api, api,
docket=docket, docket=docket,
force=force,
limit=limit or None, limit=limit or None,
log_path=log_path, log_path=log_path,
) )

View File

@@ -7,6 +7,9 @@ Phase 1 (this file):
extract — write combined.md per comment (PDF/DOCX → text) extract — write combined.md per comment (PDF/DOCX → text)
extract-ocr — phase-2 stub (raises until tesseract integration lands) extract-ocr — phase-2 stub (raises until tesseract integration lands)
stats — count comments by extraction status stats — count comments by extraction status
dockets — list known dockets (optionally discover new ones)
seal — mark a docket complete so later stages skip it
unseal — reopen a sealed docket
""" """
from __future__ import annotations from __future__ import annotations
@@ -28,54 +31,63 @@ NOTE_TITLE = "Comment text"
def _build_bib_helpers(use_bib: bool, attach: bool): def _build_bib_helpers(use_bib: bool, attach: bool):
"""Return ``(body_lookup, attach_callback)``. """Return ``(body_lookup, attach_callback, store_or_None)``.
``body_lookup(comment_id) -> str`` is called from worker threads and ``body_lookup`` queries bib per comment id (``items.url`` is indexed)
returns the inline comment body from bib.sqlite (or "" if missing or under a lock — worker threads share one connection. Nothing is
--no-bib). Backed by a comment_id → (item_key, body) cache built loaded up front, so a run that extracts nothing reads nothing.
once at startup — bib has 164k+ rows and the URL → comment_id parse
can't use any index, so we materialize the whole map up front.
``attach_callback(comment_id, comment_dir) -> None`` runs on the main ``attach_callback(comment_id, comment_dir) -> None`` runs on the main
thread and registers a single ``"Comment text"`` note on the bib item thread and registers a single ``"Comment text"`` note on the bib item
containing combined.md rendered to HTML. Skips items that already have containing combined.md rendered to HTML. Skips items that already have
one. Returns None when --no-bib or --no-attach. one. ``attach_callback`` is ``None`` for --no-bib or --no-attach; the
store is ``None`` only for --no-bib, so ``extract`` can still read
seals with --no-attach.
""" """
if not use_bib: if not use_bib:
return lambda _cid: "", None return (lambda _cid: ""), None, None
import sqlite3 import sqlite3
import threading
from conf import path
db = str(path("db.bib"))
con = sqlite3.connect(db, check_same_thread=False)
try:
cache: dict[str, tuple[str, str]] = {}
for row in con.execute(
"SELECT key, url, abstract FROM items "
"WHERE url LIKE 'https://www.regulations.gov/comment/%'"
):
cid = (row[1] or "").rsplit("/", 1)[-1]
if cid:
cache[cid] = (row[0], row[2] or "")
finally:
con.close()
def body_lookup(comment_id: str) -> str:
return cache.get(comment_id, ("", ""))[1]
if not attach:
return body_lookup, None
from bib import connect from bib import connect
from rex.comments.render import render_combined_md
store = connect() store = connect()
own_con = False
if store._db_path == ":memory:": # noqa: SLF001
con = store._con() # noqa: SLF001 — tests only: a fresh connection would be an empty DB
else:
con = sqlite3.connect(store._db_path, check_same_thread=False) # noqa: SLF001
con.row_factory = sqlite3.Row
own_con = True
lock = threading.Lock()
def _row(comment_id: str):
with lock:
return con.execute(
"SELECT key, abstract FROM items WHERE url = ?",
(f"https://www.regulations.gov/comment/{comment_id}",),
).fetchone()
def body_lookup(comment_id: str) -> str:
row = _row(comment_id)
return (row["abstract"] or "") if row else ""
# extract() closes this on our behalf when it opened a private
# read connection (the file-backed branch above) — the :memory:
# connection is store's own and must outlive this call.
if own_con:
body_lookup.close = con.close
if not attach:
return body_lookup, None, store
from rex.comments.render import render_combined_md
# Items that already have a NOTE_TITLE note — keeps re-runs idempotent. # Items that already have a NOTE_TITLE note — keeps re-runs idempotent.
items_with_note: set[str] = { items_with_note: set[str] = {
row[0] row[0]
for row in store._con().execute( for row in con.execute(
"SELECT i.key FROM notes n " "SELECT i.key FROM notes n "
"JOIN items i ON n.item_id = i.id " "JOIN items i ON n.item_id = i.id "
"WHERE n.title = ?", "WHERE n.title = ?",
@@ -84,10 +96,10 @@ def _build_bib_helpers(use_bib: bool, attach: bool):
} }
def attach_callback(comment_id: str, comment_dir: Path) -> None: def attach_callback(comment_id: str, comment_dir: Path) -> None:
info = cache.get(comment_id) row = _row(comment_id)
if not info: if not row:
return # comment not in bib return # comment not in bib
item_key = info[0] item_key = row["key"]
if item_key in items_with_note: if item_key in items_with_note:
return return
combined = comment_dir / "combined.md" combined = comment_dir / "combined.md"
@@ -95,17 +107,18 @@ def _build_bib_helpers(use_bib: bool, attach: bool):
return # extraction must have failed; nothing to attach return # extraction must have failed; nothing to attach
try: try:
html = render_combined_md(combined.read_text(encoding="utf-8")) html = render_combined_md(combined.read_text(encoding="utf-8"))
store.attach_note(item_key, html, title=NOTE_TITLE) with lock:
store.attach_note(item_key, html, title=NOTE_TITLE)
items_with_note.add(item_key) items_with_note.add(item_key)
except Exception as e: # noqa: BLE001 except Exception as e: # noqa: BLE001
log.warning("attach_note failed for %s: %s", item_key, e) log.warning("attach_note failed for %s: %s", item_key, e)
return body_lookup, attach_callback return body_lookup, attach_callback, store
# Back-compat alias — older code/tests may still import this name. # Back-compat alias — older code/tests may still import this name.
def _bib_lookup_factory(use_bib: bool): def _bib_lookup_factory(use_bib: bool):
body, _ = _build_bib_helpers(use_bib, attach=False) body, _, _ = _build_bib_helpers(use_bib, attach=False)
return body return body
@@ -133,6 +146,12 @@ def extract(
help="Attach combined.md back to its bib item (so it flows to " help="Attach combined.md back to its bib item (so it flows to "
"Zotero via `bib sync`). Implies --bib; no-op when --no-bib.", "Zotero via `bib sync`). Implies --bib; no-op when --no-bib.",
), ),
reattach: bool = typer.Option(
False,
"--reattach",
help="Also attach combined.md notes for dirs that were already "
"current (repair).",
),
) -> None: ) -> None:
"""Walk comment dirs and write combined.md (PDF/DOCX → text).""" """Walk comment dirs and write combined.md (PDF/DOCX → text)."""
from rex.comments.walker import walk_and_extract from rex.comments.walker import walk_and_extract
@@ -141,16 +160,28 @@ def extract(
typer.echo(f"root not found: {root}", err=True) typer.echo(f"root not found: {root}", err=True)
raise typer.Exit(1) raise typer.Exit(1)
lookup, attach_cb = _build_bib_helpers(use_bib, attach=use_bib and attach) lookup, attach_cb, store = _build_bib_helpers(use_bib, attach=use_bib and attach)
stats = walk_and_extract( try:
root, skip = (
inline_body_lookup=lookup, set(store.sealed_dockets()) if (store is not None and not force) else set()
docket=docket, )
limit=limit or None, stats = walk_and_extract(
workers=workers, root,
force=force, inline_body_lookup=lookup,
on_extracted=attach_cb, docket=docket,
) limit=limit or None,
workers=workers,
force=force,
on_extracted=attach_cb,
skip_dockets=skip,
reattach=reattach,
)
finally:
close = getattr(lookup, "close", None)
if close is not None:
close()
if docket and docket in skip:
typer.echo(f" {docket}: sealed; skipped — use --force")
for k, v in stats.items(): for k, v in stats.items():
typer.echo(f" {k}: {v}") typer.echo(f" {k}: {v}")
@@ -254,3 +285,86 @@ def stats(
cols = ["comments", "attachments", "ok", "ocr_needed", "failed"] cols = ["comments", "attachments", "ok", "ocr_needed", "failed"]
for k, v in zip(cols, rows, strict=True): for k, v in zip(cols, rows, strict=True):
typer.echo(f" {k}: {v or 0}") typer.echo(f" {k}: {v or 0}")
@app.command()
def dockets(
discover: bool = typer.Option(
False,
"--discover",
help="Populate rows for every reg-docket: tag in bib that has no "
"dockets row (one reg.gov listing call each).",
),
) -> None:
"""List dockets: id, rule, close date, watermark, counts, seal."""
from bib import connect
from bib.regulations_gov import docket_counts
store = connect()
if discover:
from bib.regulations_gov import Client, _rule_tag_for_docket, discover_docket
tagged = [
r[0].split(":", 1)[1]
for r in store._con().execute( # noqa: SLF001
"SELECT name FROM tags WHERE name LIKE 'reg-docket:%' ORDER BY name"
)
]
missing = [d for d in tagged if store.docket_get(d) is None]
with Client(sleep=1.3) as api:
for d in missing:
row = discover_docket(
api, d, rule_cms_id=_rule_tag_for_docket(store, d)
)
store.docket_upsert(row)
typer.echo(
f" discovered {d}: closes {row.comment_end_date or '?'} "
f"rule {row.rule_cms_id or '?'}"
)
rows = store.dockets()
if not rows:
typer.echo("no dockets recorded — run `stack comments dockets --discover`")
return
typer.echo(
f"{'docket':<16}{'rule':<12}{'closes':<12}{'comments':>9}{'enriched':>9} "
f"{'watermark':<20} state"
)
for d in rows:
c = docket_counts(store, d.id)
state = f"sealed {d.sealed_at[:10]} ({d.seal_reason})" if d.sealed else "open"
typer.echo(
f"{d.id:<16}{d.rule_cms_id:<12}{d.comment_end_date:<12}"
f"{c['comments']:>9}{c['enriched']:>9} {d.pull_watermark[:19]:<20} {state}"
)
@app.command()
def seal(
docket: str = typer.Argument(..., help="reg.gov docket id"),
reason: str = typer.Option("manual", "--reason"),
) -> None:
"""Mark a docket complete: every stage skips it until unsealed or --force."""
from bib import connect
from bib.regulations_gov import docket_counts
store = connect()
if store.docket_get(docket) is None:
typer.echo(
f"unknown docket {docket} — run `stack comments dockets --discover` first"
)
raise typer.Exit(1)
store.docket_seal(docket, reason=reason, counts=docket_counts(store, docket))
typer.echo(f"sealed {docket} ({reason})")
@app.command()
def unseal(docket: str = typer.Argument(..., help="reg.gov docket id")) -> None:
"""Reopen a sealed docket."""
from bib import connect
store = connect()
if store.docket_get(docket) is None:
typer.echo(f"unknown docket {docket}")
raise typer.Exit(1)
store.docket_unseal(docket)
typer.echo(f"unsealed {docket}")

View File

@@ -7,24 +7,26 @@ app = typer.Typer(no_args_is_help=True)
_COLLECTIONS = ("comments", "rules", "corpus") _COLLECTIONS = ("comments", "rules", "corpus")
def _docs_for(collection: str, store, docket: str, keys: tuple[str, ...]): def _refs_for(
collection: str, store, docket: str, keys: tuple[str, ...], skip_dockets: set[str]
):
from llm.source import ( from llm.source import (
ZoteroPdfIndex, ZoteroPdfIndex,
iter_comment_docs, iter_comment_refs,
iter_corpus_docs, iter_corpus_refs,
iter_rule_docs, iter_rule_refs,
) )
if collection == "comments": if collection == "comments":
return iter_comment_docs(store, docket=docket) return iter_comment_refs(store, docket=docket, skip_dockets=skip_dockets)
if collection == "rules": if collection == "rules":
return iter_rule_docs(store, keys=keys) return iter_rule_refs(store, keys=keys)
from conf import ROOT, path from conf import ROOT, path
zotero = ZoteroPdfIndex.snapshot( zotero = ZoteroPdfIndex.lazy(
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm" path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
) )
return iter_corpus_docs(store, zotero=zotero) return iter_corpus_refs(store, zotero=zotero)
@app.command() @app.command()
@@ -47,7 +49,8 @@ def index(
from conf.connect import bib from conf.connect import bib
from llm import config as llm_config from llm import config as llm_config
from llm.index import index_docs from llm.index import _engine, docket_complete, index_refs
from llm.migrate import migrate
from llm.pool import HostPool from llm.pool import HostPool
targets = _COLLECTIONS if collection == "all" else (collection,) targets = _COLLECTIONS if collection == "all" else (collection,)
@@ -55,20 +58,43 @@ def index(
raise typer.BadParameter("collection must be comments, rules, corpus or all") raise typer.BadParameter("collection must be comments, rules, corpus or all")
cfg = llm_config.load() cfg = llm_config.load()
store = bib() store = bib()
sealed_all = {} if force else store.sealed_dockets()
# One engine for the whole run, migrated *before* anything reads
# index_docket_state — that table does not exist until migrate() has
# run, and the first sealed run on an un-migrated DB gets here first.
engine = None
if sealed_all and "comments" in targets:
engine = _engine(cfg)
migrate(engine)
for target in targets: for target in targets:
docs = _docs_for(target, store, docket, tuple(key)) complete = (
docket_complete(engine, target)
if (engine is not None and target == "comments")
else {}
)
skip = {d for d, s in sealed_all.items() if complete.get(d) == s}
pending_seals = {d: s for d, s in sealed_all.items() if d not in skip}
refs = _refs_for(target, store, docket, tuple(key), skip)
if limit: if limit:
docs = itertools.islice(docs, limit) refs = itertools.islice(refs, limit)
stats = index_docs( stats = index_refs(
docs, refs,
collection=target, collection=target,
cfg=cfg, cfg=cfg,
pool=HostPool.from_config(cfg), pool=HostPool.from_config(cfg),
force=force, force=force,
sealed=pending_seals if target == "comments" else None,
mark_complete=not limit,
engine=engine,
) )
if skip:
typer.echo(
f"{target}: {len(skip)} sealed docket(s) already complete — not listed"
)
typer.echo( typer.echo(
f"{target}: indexed={stats['indexed']} skipped={stats['skipped']} " f"{target}: indexed={stats['indexed']} skipped={stats['skipped']} chunks={stats['chunks']} "
f"chunks={stats['chunks']}" f"fp_skipped={stats['fingerprint_skipped']} hash_skipped={stats['hash_skipped']} "
f"docket_complete={stats['docket_complete']}"
) )

View File

@@ -31,6 +31,7 @@ from llm.config import LlmConfig, pg_url
from llm.migrate import ensure_hnsw, migrate from llm.migrate import ensure_hnsw, migrate
from llm.pages import enrich_pdf_pages from llm.pages import enrich_pdf_pages
from llm.pool import HostPool, PoolEmbeddings, embed_texts from llm.pool import HostPool, PoolEmbeddings, embed_texts
from llm.source import DocRef
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
@@ -51,15 +52,28 @@ def vectorstore(collection: str, cfg: LlmConfig, pool: HostPool):
) )
def _state(engine: Engine, collection: str) -> dict[str, str]: def _state(engine: Engine, collection: str) -> dict[str, tuple[str, str]]:
with engine.begin() as conn: with engine.begin() as conn:
rows = conn.execute( rows = conn.execute(
text( text(
"SELECT item_key, content_hash FROM index_state WHERE collection = :c" "SELECT item_key, content_hash, fingerprint FROM index_state "
"WHERE collection = :c"
), ),
{"c": collection}, {"c": collection},
).fetchall() ).fetchall()
return dict(rows) return {r[0]: (r[1], r[2] or "") for r in rows}
def docket_complete(engine: Engine, collection: str) -> dict[str, str]:
"""docket → sealed_at for dockets fully indexed under that seal."""
with engine.begin() as conn:
rows = conn.execute(
text(
"SELECT docket, sealed_at FROM index_docket_state WHERE collection = :c"
),
{"c": collection},
).fetchall()
return {r[0]: r[1] for r in rows}
def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None: def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None:
@@ -87,6 +101,154 @@ def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None:
log.debug("langchain_pg_embedding not present yet; skipping delete") log.debug("langchain_pg_embedding not present yet; skipping delete")
def _record_state(
engine: Engine, key: str, collection: str, h: str, n: int, fingerprint: str
) -> None:
"""Upsert this item's row in ``index_state``.
``h == "" and n == 0`` is the *stamped-final* form: an item with no
text at all. It is a finished state, not a failure — no chunks exist
to delete or embed — and the empty hash can never collide with a real
``content_hash`` (a sha256 hexdigest), so the next run recognises it
by fingerprint alone and never loads it again. If text does appear
later the fingerprint changes and the item is embedded normally.
"""
with engine.begin() as conn:
conn.execute(
text(
"INSERT INTO index_state "
"(item_key, collection, content_hash, chunk_count, fingerprint) "
"VALUES (:k, :c, :h, :n, :f) "
"ON CONFLICT (item_key, collection) DO UPDATE SET "
"content_hash = :h, chunk_count = :n, fingerprint = :f, "
"indexed_at = now()"
),
{"k": key, "c": collection, "h": h, "n": n, "f": fingerprint},
)
def index_refs(
refs: Iterable[DocRef],
*,
collection: str,
cfg: LlmConfig,
pool: HostPool,
force: bool = False,
sealed: dict[str, str] | None = None,
mark_complete: bool = True,
engine: Engine | None = None,
) -> dict:
"""Fingerprint-first incremental indexing.
Per ref: (1) fingerprint equal to the stored one → skip without
loading; (2) load, hash the text; hash equal → record the new
fingerprint, skip embedding; (3) chunk, locate PDF pages, embed,
write. PDFs are opened only on path (3). A ref that loads blank is
*stamped final* (see :func:`_record_state`) rather than treated as a
failure — plenty of real comments carry no text at all.
After the loop, a docket in *sealed* is recorded in
``index_docket_state`` — so the next run does not list it at all —
only when every one of its refs was accounted for. A ref that loaded
*with* text but chunked to nothing is *unindexable* and blocks
completion, because a seal must never claim coverage this run did not
achieve (an exception on the embed/write path aborts the run and so
blocks it too). *mark_complete* False (a ``--limit`` run is never
complete) blocks it as well. Pass *engine* to reuse a caller's engine
instead of building a second one.
"""
engine = engine if engine is not None else _engine(cfg)
migrate(engine)
pool.check(cfg.embed_model)
store = vectorstore(collection, cfg, pool)
seen = _state(engine, collection)
stats = {
"indexed": 0,
"skipped": 0,
"chunks": 0,
"fingerprint_skipped": 0,
"hash_skipped": 0,
"docket_complete": 0,
}
dockets_seen: set[str] = set()
unindexable: dict[str, int] = {}
def _unindexable(ref: DocRef) -> None:
stats["skipped"] += 1
if ref.docket:
unindexable[ref.docket] = unindexable.get(ref.docket, 0) + 1
for ref in refs:
if ref.docket:
dockets_seen.add(ref.docket)
prev_hash, prev_fp = seen.get(ref.key, ("", ""))
if not force and ref.fingerprint and ref.fingerprint == prev_fp:
stats["fingerprint_skipped"] += 1
continue
doc = ref.load()
if doc is None or not doc.text.strip():
_record_state(engine, ref.key, collection, "", 0, ref.fingerprint)
stats["skipped"] += 1
continue
h = content_hash(doc.text)
if not force and prev_hash == h:
with engine.begin() as conn:
conn.execute(
text(
"UPDATE index_state SET fingerprint = :f "
"WHERE item_key = :k AND collection = :c"
),
{"f": ref.fingerprint, "k": ref.key, "c": collection},
)
stats["hash_skipped"] += 1
continue
chunks = enrich_pdf_pages(doc, chunk_doc(doc))
if not chunks:
_unindexable(ref)
continue
vectors = embed_texts(pool, cfg.embed_model, [c.text for c in chunks])
_delete_old_chunks(engine, ref.key, collection)
store.add_embeddings(
texts=[c.text for c in chunks],
embeddings=vectors,
metadatas=[c.metadata for c in chunks],
ids=[c.id for c in chunks],
)
_record_state(engine, ref.key, collection, h, len(chunks), ref.fingerprint)
stats["indexed"] += 1
stats["chunks"] += len(chunks)
# needs 100+ docs in one run to hit this line
if stats["indexed"] % 100 == 0:
log.info("indexed %(indexed)s (+%(chunks)s)", stats) # pragma: no cover
if mark_complete and sealed:
for d in sorted(dockets_seen & set(sealed)):
if unindexable.get(d):
log.info(
"docket %s not marked complete: %s unindexable item(s)",
d,
unindexable[d],
)
continue
with engine.begin() as conn:
conn.execute(
text(
"INSERT INTO index_docket_state (collection, docket, sealed_at) "
"VALUES (:c, :d, :s) "
"ON CONFLICT (collection, docket) DO UPDATE SET "
"sealed_at = :s, indexed_at = now()"
),
{"c": collection, "d": d, "s": sealed[d]},
)
stats["docket_complete"] += 1
if cfg.build_ann_index:
ensure_hnsw(engine, cfg.embed_dim)
else:
log.info("ANN index skipped (build_ann_index=false); using exact search")
return stats
def index_docs( def index_docs(
docs: Iterable[Doc], docs: Iterable[Doc],
*, *,
@@ -95,49 +257,22 @@ def index_docs(
pool: HostPool, pool: HostPool,
force: bool = False, force: bool = False,
) -> dict: ) -> dict:
engine = _engine(cfg) """Back-compat wrapper: Docs without fingerprints (never fingerprint-skipped)."""
migrate(engine) refs = (
pool.check(cfg.embed_model) DocRef(
store = vectorstore(collection, cfg, pool) key=d.key,
seen = _state(engine, collection) collection=collection,
stats = {"indexed": 0, "skipped": 0, "chunks": 0} docket=d.metadata.get("docket") or None,
fingerprint="",
for doc in docs: load=(lambda d=d: d),
chunks = enrich_pdf_pages(doc, chunk_doc(doc))
if not chunks:
stats["skipped"] += 1
continue
h = content_hash(doc.text)
if not force and seen.get(doc.key) == h:
stats["skipped"] += 1
continue
vectors = embed_texts(pool, cfg.embed_model, [c.text for c in chunks])
_delete_old_chunks(engine, doc.key, collection)
store.add_embeddings(
texts=[c.text for c in chunks],
embeddings=vectors,
metadatas=[c.metadata for c in chunks],
ids=[c.id for c in chunks],
) )
with engine.begin() as conn: for d in docs
conn.execute( )
text( stats = index_refs(
"INSERT INTO index_state " refs, collection=collection, cfg=cfg, pool=pool, force=force, sealed=None
"(item_key, collection, content_hash, chunk_count) " )
"VALUES (:k, :c, :h, :n) " # Fold the fingerprint/hash skip counters into "skipped" so the 3-key
"ON CONFLICT (item_key, collection) DO UPDATE SET " # contract still means "not embedded" — index_refs itself keeps them
"content_hash = :h, chunk_count = :n, indexed_at = now()" # separate; this is only a local projection.
), skipped = stats["skipped"] + stats["fingerprint_skipped"] + stats["hash_skipped"]
{"k": doc.key, "c": collection, "h": h, "n": len(chunks)}, return {"indexed": stats["indexed"], "skipped": skipped, "chunks": stats["chunks"]}
)
stats["indexed"] += 1
stats["chunks"] += len(chunks)
# needs 100+ docs in one run to hit this line
if stats["indexed"] % 100 == 0:
log.info("indexed %(indexed)s (+%(chunks)s)", stats) # pragma: no cover
if cfg.build_ann_index:
ensure_hnsw(engine, cfg.embed_dim)
else:
log.info("ANN index skipped (build_ann_index=false); using exact search")
return stats

View File

@@ -21,10 +21,27 @@ CREATE TABLE IF NOT EXISTS index_state (
content_hash TEXT NOT NULL, content_hash TEXT NOT NULL,
chunk_count INTEGER NOT NULL DEFAULT 0, chunk_count INTEGER NOT NULL DEFAULT 0,
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(), indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
fingerprint TEXT NOT NULL DEFAULT '',
PRIMARY KEY (item_key, collection) PRIMARY KEY (item_key, collection)
) )
""" """
# Existing databases predate the column.
INDEX_STATE_ALTER = "ALTER TABLE index_state ADD COLUMN IF NOT EXISTS fingerprint TEXT NOT NULL DEFAULT ''"
# A sealed docket that was fully indexed under a given seal: the iterator
# does not even list its items until the seal changes (unseal/reseal
# writes a new sealed_at in bib, which no longer matches).
INDEX_DOCKET_STATE_DDL = """
CREATE TABLE IF NOT EXISTS index_docket_state (
collection TEXT NOT NULL,
docket TEXT NOT NULL,
sealed_at TEXT NOT NULL,
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
PRIMARY KEY (collection, docket)
)
"""
# langchain-postgres creates langchain_pg_embedding with an untyped vector # langchain-postgres creates langchain_pg_embedding with an untyped vector
# column; HNSW needs a typed one. ALTER is a no-op when already typed. # column; HNSW needs a typed one. ALTER is a no-op when already typed.
_HNSW_DDL = [ _HNSW_DDL = [
@@ -35,9 +52,11 @@ _HNSW_DDL = [
def migrate(engine: Engine) -> None: def migrate(engine: Engine) -> None:
"""Create llm-owned tables. Safe to run on every start.""" """Create llm-owned tables/columns. Safe to run on every start."""
with engine.begin() as conn: with engine.begin() as conn:
conn.execute(text(INDEX_STATE_DDL)) conn.execute(text(INDEX_STATE_DDL))
conn.execute(text(INDEX_STATE_ALTER))
conn.execute(text(INDEX_DOCKET_STATE_DDL))
def ensure_hnsw(engine: Engine, dim: int) -> None: def ensure_hnsw(engine: Engine, dim: int) -> None:

View File

@@ -1,5 +1,12 @@
"""Doc iterators over the bib store + comment extraction tree. """Doc iterators over the bib store + comment extraction tree.
Every source is listed as a :class:`DocRef` first — key, collection,
docket and a cheap *fingerprint* (file stats + ``updated_at``, or the FR
anchor sha) — so the indexer can decide what changed before any file is
read. ``ref.load()`` then does the expensive part (extraction, per-item
SQL) and returns ``None`` for a text-less item. ``iter_*_docs`` are thin
wrappers that load every ref.
Comments: text prefers the #253 extraction output Comments: text prefers the #253 extraction output
(``.state/comments/<docket>/<comment_id>/combined.md`` body via (``.state/comments/<docket>/<comment_id>/combined.md`` body via
``rex.comments.combine.parse_combined``); falls back to the bib ``rex.comments.combine.parse_combined``); falls back to the bib
@@ -16,11 +23,14 @@ so parsing the comment id is the reliable path to a comment's docket.
from __future__ import annotations from __future__ import annotations
import logging import logging
import os
import shutil import shutil
import sqlite3 import sqlite3
from dataclasses import dataclass
from pathlib import Path from pathlib import Path
from typing import Iterator from typing import Callable, Iterator
from bib.dockets import fingerprint_files
from bib.store import Store from bib.store import Store
from llm.chunk import Doc, Paragraph from llm.chunk import Doc, Paragraph
@@ -30,6 +40,19 @@ _COMMENT_URL_PREFIX = "https://www.regulations.gov/comment/"
_ATTACHMENT_EXT = (".pdf", ".docx", ".doc", ".txt") _ATTACHMENT_EXT = (".pdf", ".docx", ".doc", ".txt")
@dataclass(frozen=True)
class DocRef:
"""A document the indexer *may* need: enough to decide (key +
fingerprint) without building it. ``load()`` does the expensive part
and returns ``None`` when the item has no text."""
key: str
collection: str
docket: str | None
fingerprint: str
load: Callable[[], "Doc | None"]
def _default_root() -> Path: def _default_root() -> Path:
from conf import ROOT from conf import ROOT
@@ -51,12 +74,26 @@ def _year_of(store: Store, item_key: str) -> str:
return row[0].split(":", 1)[1] if row else "" return row[0].split(":", 1)[1] if row else ""
def _comment_rows( def _attachment_paths(store: Store, item_key: str) -> list[Path]:
store: Store, docket: str = "" rows = (
) -> list[tuple[str, str, str, str, str]]: store._con()
"""(docket, comment_id, item key, date_published, title) for every .execute(
comment matching *docket*, newest posted first so incremental index "SELECT a.storage_path FROM attachments a "
runs surface the latest comments before older backlog. "JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(item_key,),
)
.fetchall()
)
return [Path(r[0]) for r in rows]
# ── comments ──
def _comment_listing(store: Store, docket: str = "") -> list[sqlite3.Row]:
"""One query: key, comment url, date, title, updated_at, year — newest
posted first so incremental index runs surface the latest comments
before older backlog.
*docket* filters via the URL (``.../comment/<docket>-%``); an empty *docket* filters via the URL (``.../comment/<docket>-%``); an empty
docket matches every comment and the docket is recovered from each docket matches every comment and the docket is recovered from each
@@ -65,29 +102,26 @@ def _comment_rows(
pattern = ( pattern = (
f"{_COMMENT_URL_PREFIX}{docket}-%" if docket else f"{_COMMENT_URL_PREFIX}%" f"{_COMMENT_URL_PREFIX}{docket}-%" if docket else f"{_COMMENT_URL_PREFIX}%"
) )
rows = ( return (
store._con() store._con()
.execute( .execute(
"SELECT i.key, i.url, COALESCE(i.date_published, ''), " "SELECT i.key, i.url, COALESCE(i.date_published,'') AS date, "
"COALESCE(i.title, '') FROM items i WHERE i.url LIKE ? " "COALESCE(i.title,'') AS title, COALESCE(i.updated_at,'') AS updated_at, "
"COALESCE((SELECT t.name FROM tags t JOIN item_tags it ON it.tag_id = t.id "
" WHERE it.item_id = i.id AND t.name LIKE 'year:%' LIMIT 1), '') AS year "
"FROM items i WHERE i.url LIKE ? "
"ORDER BY i.date_published DESC, i.key", "ORDER BY i.date_published DESC, i.key",
(pattern,), (pattern,),
) )
.fetchall() .fetchall()
) )
out = []
for key, url, date, title in rows:
comment_id = url.rsplit("/", 1)[-1]
dk = docket or comment_id.rsplit("-", 1)[0]
out.append((dk, comment_id, key, date, title))
return out
def comment_key_map(store: Store, docket: str) -> dict[str, tuple[str, str]]: def comment_key_map(store: Store, docket: str) -> dict[str, tuple[str, str]]:
"""comment_id -> (bib item key, year) for one docket.""" """comment_id -> (bib item key, year) for one docket."""
return { return {
comment_id: (key, _year_of(store, key)) row["url"].rsplit("/", 1)[-1]: (row["key"], row["year"].split(":", 1)[-1])
for _, comment_id, key, _, _ in _comment_rows(store, docket) for row in _comment_listing(store, docket)
} }
@@ -101,52 +135,84 @@ def _comment_files(comment_dir: Path) -> tuple[tuple[str, str], ...]:
) )
def iter_comment_docs( def iter_comment_refs(
store: Store, *, docket: str = "", root: Path | None = None store: Store,
) -> Iterator[Doc]: *,
"""One Doc per comment, newest first: extraction body, else abstract.""" docket: str = "",
from rex.comments.combine import parse_combined root: Path | None = None,
skip_dockets: set[str] | frozenset[str] = frozenset(),
) -> Iterator[DocRef]:
"""One DocRef per comment, newest first. Fingerprint = file stats of
combined.md + attachments + the item's updated_at; no file is read
and the extraction machinery is not even imported."""
root = root if root is not None else _default_root() root = root if root is not None else _default_root()
for dk, comment_id, key, date, title in _comment_rows(store, docket): for row in _comment_listing(store, docket):
comment_id = row["url"].rsplit("/", 1)[-1]
dk = docket or comment_id.rsplit("-", 1)[0]
if dk in skip_dockets:
continue
key, date, title, updated_at = (
row["key"],
row["date"],
row["title"],
row["updated_at"],
)
year = row["year"].split(":", 1)[-1]
comment_dir = root / dk / comment_id
combined = comment_dir / "combined.md"
files = _comment_files(comment_dir)
fp = (
fingerprint_files([combined, *(Path(p) for _, p in files)])
+ "|"
+ updated_at
)
meta = { meta = {
"docket": dk, "docket": dk,
"comment_id": comment_id, "comment_id": comment_id,
"doctype": "comment", "doctype": "comment",
"kind": "comment", "kind": "comment",
"year": _year_of(store, key), "year": year,
"date": date[:10], "date": date[:10],
"title": title, "title": title,
} }
comment_dir = root / dk / comment_id
combined = comment_dir / "combined.md" def _load(key=key, meta=meta, combined=combined, files=files) -> Doc | None:
if combined.exists(): from rex.comments.combine import parse_combined
_, body = parse_combined(combined.read_text())
if body.strip(): if combined.exists():
yield Doc( _, body = parse_combined(combined.read_text())
key=key, text=body, metadata=meta, files=_comment_files(comment_dir) if body.strip():
) return Doc(key=key, text=body, metadata=meta, files=files)
continue row = (
item = store.get(key) store._con()
if item.abstract.strip(): .execute("SELECT abstract FROM items WHERE key = ?", (key,))
yield Doc(key=key, text=item.abstract, metadata=meta) .fetchone()
)
abstract = (row["abstract"] if row else "") or ""
return (
Doc(key=key, text=abstract, metadata=meta) if abstract.strip() else None
)
yield DocRef(
key=key, collection="comments", docket=dk, fingerprint=fp, load=_load
)
def iter_comment_docs(
store: Store, *, docket: str = "", root: Path | None = None
) -> Iterator[Doc]:
"""One Doc per comment, newest first: extraction body, else abstract."""
for ref in iter_comment_refs(store, docket=docket, root=root):
doc = ref.load()
if doc is not None:
yield doc
def _attachment_text(store: Store, item_key: str) -> str: def _attachment_text(store: Store, item_key: str) -> str:
from rex.comments.combine import extract_attachment from rex.comments.combine import extract_attachment
rows = (
store._con()
.execute(
"SELECT a.storage_path FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(item_key,),
)
.fetchall()
)
parts = [] parts = []
for (storage_path,) in rows: for path in _attachment_paths(store, item_key):
path = Path(storage_path)
if path.exists(): if path.exists():
result = extract_attachment(path) result = extract_attachment(path)
if result.status == "ok": if result.status == "ok":
@@ -166,17 +232,7 @@ def _rule_text(store: Store, item_key: str) -> str:
""" """
from rex.frtext import clean_fr_text from rex.frtext import clean_fr_text
rows = ( for path in _attachment_paths(store, item_key):
store._con()
.execute(
"SELECT a.storage_path FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(item_key,),
)
.fetchall()
)
for (storage_path,) in rows:
path = Path(storage_path)
if path.suffix.lower() == ".txt" and path.exists(): if path.suffix.lower() == ".txt" and path.exists():
return clean_fr_text(path.read_text(encoding="utf-8", errors="replace")) return clean_fr_text(path.read_text(encoding="utf-8", errors="replace"))
return _attachment_text(store, item_key) return _attachment_text(store, item_key)
@@ -209,42 +265,122 @@ def _anchor_doc(store: Store, item_key: str) -> tuple[str, str]:
return (row[0], str(row[1])) if row else ("", "") return (row[0], str(row[1])) if row else ("", "")
# ── rules ──
def iter_rule_refs(
store: Store, *, keys: tuple[str, ...] = (), tag: str = ""
) -> Iterator[DocRef]:
"""One DocRef per FR rule item. Fingerprint = the grabbed anchor
document's sha256 when present, else the attachment file stats +
updated_at."""
rows = (
store._con()
.execute(
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at, "
"(SELECT d.sha256 FROM fr_anchor_docs d WHERE d.item_key = i.key) AS anchor_sha "
"FROM items i WHERE i.item_type = 'rule'"
+ (
" AND i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
"(SELECT id FROM tags WHERE name = ?))"
if tag
else ""
)
+ " ORDER BY i.id",
(tag,) if tag else (),
)
.fetchall()
)
for row in rows:
key = row["key"]
if keys and key not in keys:
continue
if row["anchor_sha"]:
# updated_at too: the anchor sha covers the FR body, not the
# item's own metadata (title, cms-rule: tag, date).
fp = f"anchors:{row['anchor_sha']}|{row['updated_at']}"
else:
fp = (
fingerprint_files(_attachment_paths(store, key))
+ "|"
+ row["updated_at"]
)
def _load(key=key) -> Doc | None:
return _build_rule_doc(store, store.get(key))
yield DocRef(
key=key, collection="rules", docket=None, fingerprint=fp, load=_load
)
def _build_rule_doc(store: Store, item) -> Doc | None:
"""Anchor paragraphs when grabbed (exact ``#p-N`` provenance per
chunk), else TXT attachment, else PDF-extract; None when empty."""
paragraphs = rule_paragraphs(store, item.key)
html_url, volume = _anchor_doc(store, item.key)
if paragraphs:
text = "\n\n".join(p.text for p in paragraphs if p.text.strip())
else:
text = _rule_text(store, item.key)
if not text.strip():
return None
cms_rule = next(
(t.split(":", 1)[1] for t in item.tags if t.startswith("cms-rule:")), ""
)
return Doc(
key=item.key,
text=text,
metadata={
"doctype": "rule",
"kind": "rule",
"cms_rule_id": cms_rule,
"fr_document_number": item.document_number or "",
"year": _year_of(store, item.key),
"date": (item.date_published or "")[:10],
"title": item.title,
"item_key": item.key,
"html_url": html_url,
"fr_volume": volume,
},
paragraphs=paragraphs,
)
def iter_rule_docs( def iter_rule_docs(
store: Store, *, keys: tuple[str, ...] = (), tag: str = "" store: Store, *, keys: tuple[str, ...] = (), tag: str = ""
) -> Iterator[Doc]: ) -> Iterator[Doc]:
"""One Doc per FR rule item: anchor paragraphs when grabbed (exact """One Doc per FR rule item: anchor paragraphs when grabbed (exact
``#p-N`` provenance per chunk), else TXT attachment, else PDF-extract.""" ``#p-N`` provenance per chunk), else TXT attachment, else PDF-extract."""
for item in store.list_items(item_type="rule", tag=tag): for ref in iter_rule_refs(store, keys=keys, tag=tag):
if keys and item.key not in keys: doc = ref.load()
continue if doc is not None:
paragraphs = rule_paragraphs(store, item.key) yield doc
html_url, volume = _anchor_doc(store, item.key)
if paragraphs:
text = "\n\n".join(p.text for p in paragraphs if p.text.strip()) def _source_mtime_ns(src: Path) -> int:
else: """Newest mtime across zotero.sqlite and its ``-wal`` sidecar.
text = _rule_text(store, item.key)
if not text.strip(): Zotero commits into the WAL and only touches the main database at a
continue checkpoint, so ``src.stat()`` alone reports a database that has not
cms_rule = next( changed for weeks while the library is being edited daily.
(t.split(":", 1)[1] for t in item.tags if t.startswith("cms-rule:")), "" """
) newest = src.stat().st_mtime_ns if src.exists() else 0
yield Doc( wal = src.with_name(src.name + "-wal")
key=item.key, if wal.exists():
text=text, newest = max(newest, wal.stat().st_mtime_ns)
metadata={ return newest
"doctype": "rule",
"kind": "rule",
"cms_rule_id": cms_rule, def _copy_snapshot(src: Path, snap: Path) -> bool:
"fr_document_number": item.document_number or "", """Copy the Zotero database to *snap*. False (logged) on failure —
"year": _year_of(store, item.key), the caller must not treat whatever is at *snap* as fresh."""
"date": (item.date_published or "")[:10], try:
"title": item.title, shutil.copy2(src, snap)
"item_key": item.key, except OSError as e:
"html_url": html_url, log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e)
"fr_volume": volume, return False
}, return True
paragraphs=paragraphs,
)
class ZoteroPdfIndex: class ZoteroPdfIndex:
@@ -254,6 +390,44 @@ class ZoteroPdfIndex:
def __init__(self, by_key: dict[str, list[Path]]) -> None: def __init__(self, by_key: dict[str, list[Path]]) -> None:
self._by_key = by_key self._by_key = by_key
self._loader: Callable[[], dict[str, list[Path]]] | None = None
@classmethod
def lazy(
cls, sqlite_path: Path, storage_dir: Path, tmp_dir: Path
) -> "ZoteroPdfIndex":
"""Snapshot on first ``pdfs_for``; skip the copy when the existing
snapshot is already as new as the source (the 1.95 GB copy is pure
waste on a run that never reaches the Zotero fallback)."""
inst = cls({})
def _load() -> dict[str, list[Path]]:
snap = Path(tmp_dir) / "zotero.sqlite"
src = Path(sqlite_path)
# Zotero writes to zotero.sqlite-wal and only touches the main
# file at a checkpoint, so the main mtime alone would leave a
# stale snapshot in place indefinitely — take the newer of the two.
newest = _source_mtime_ns(src)
if snap.exists() and src.exists() and snap.stat().st_mtime_ns >= newest:
return cls._read(snap, Path(storage_dir))
if not src.exists():
log.warning("zotero db %s missing — no Zotero PDF fallback", src)
return {}
Path(tmp_dir).mkdir(parents=True, exist_ok=True)
if not _copy_snapshot(src, snap):
# Only a *successful* copy may be stamped: stamping the
# stale snapshot left behind by a failed one would pass
# it off as current on every future run.
return {}
if newest:
# copy2 carries the *main* file's mtime, which is older
# than the WAL; stamp what we actually captured so the
# next run doesn't re-copy 1.95 GB for nothing.
os.utime(snap, ns=(newest, newest))
return cls._read(snap, Path(storage_dir))
inst._loader = _load
return inst
@classmethod @classmethod
def snapshot( def snapshot(
@@ -264,8 +438,14 @@ class ZoteroPdfIndex:
return cls({}) return cls({})
tmp_dir.mkdir(parents=True, exist_ok=True) tmp_dir.mkdir(parents=True, exist_ok=True)
snap = tmp_dir / "zotero.sqlite" snap = tmp_dir / "zotero.sqlite"
if not _copy_snapshot(Path(sqlite_path), snap):
return cls({})
return cls(cls._read(snap, Path(storage_dir)))
@classmethod
def _read(cls, snap: Path, storage_dir: Path) -> dict[str, list[Path]]:
"""parent item key → existing storage PDFs, from a snapshot copy."""
try: try:
shutil.copy2(sqlite_path, snap)
con = sqlite3.connect(f"file:{snap}?mode=ro", uri=True) con = sqlite3.connect(f"file:{snap}?mode=ro", uri=True)
rows = con.execute( rows = con.execute(
"SELECT p.key, a.key, ia.path FROM itemAttachments ia " "SELECT p.key, a.key, ia.path FROM itemAttachments ia "
@@ -277,15 +457,18 @@ class ZoteroPdfIndex:
con.close() con.close()
except (OSError, sqlite3.Error) as e: except (OSError, sqlite3.Error) as e:
log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e) log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e)
return cls({}) return {}
by_key: dict[str, list[Path]] = {} by_key: dict[str, list[Path]] = {}
for parent_key, att_key, path in rows: for parent_key, att_key, path in rows:
pdf = Path(storage_dir) / att_key / path[len("storage:") :] pdf = Path(storage_dir) / att_key / path[len("storage:") :]
if pdf.exists(): if pdf.exists():
by_key.setdefault(parent_key, []).append(pdf) by_key.setdefault(parent_key, []).append(pdf)
return cls(by_key) return by_key
def pdfs_for(self, key: str) -> list[Path]: def pdfs_for(self, key: str) -> list[Path]:
if self._loader is not None:
self._by_key = self._loader()
self._loader = None
return list(self._by_key.get(key, [])) return list(self._by_key.get(key, []))
@@ -295,19 +478,9 @@ def _attachment_sections(
"""(markdown sections, files) for an item's bib attachments.""" """(markdown sections, files) for an item's bib attachments."""
from rex.comments.combine import extract_attachment from rex.comments.combine import extract_attachment
rows = (
store._con()
.execute(
"SELECT a.storage_path FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(item_key,),
)
.fetchall()
)
sections: list[str] = [] sections: list[str] = []
files: list[tuple[str, str]] = [] files: list[tuple[str, str]] = []
for (storage_path,) in rows: for path in _attachment_paths(store, item_key):
path = Path(storage_path)
if path.exists(): if path.exists():
result = extract_attachment(path) result = extract_attachment(path)
if result.status == "ok" and result.text.strip(): if result.status == "ok" and result.text.strip():
@@ -316,42 +489,83 @@ def _attachment_sections(
return sections, files return sections, files
# ── corpus ──
def iter_corpus_refs(
store: Store, *, tag: str = "", zotero: "ZoteroPdfIndex | None" = None
) -> Iterator[DocRef]:
"""One DocRef per non-comment item. Fingerprint = updated_at + the
bib attachment file stats; the Zotero fallback is only consulted by
``load()``."""
sql = (
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at FROM items i "
"WHERE i.id NOT IN (SELECT item_id FROM item_tags WHERE tag_id IN "
"(SELECT id FROM tags WHERE name = 'doctype:comment'))"
+ (
" AND i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
"(SELECT id FROM tags WHERE name = ?))"
if tag
else ""
)
+ " ORDER BY i.id"
)
for row in store._con().execute(sql, (tag,) if tag else ()).fetchall():
key = row["key"]
fp = row["updated_at"] + "|" + fingerprint_files(_attachment_paths(store, key))
def _load(key=key) -> Doc | None:
return _build_corpus_doc(store, store.get(key), zotero)
yield DocRef(
key=key, collection="corpus", docket=None, fingerprint=fp, load=_load
)
def _build_corpus_doc(
store: Store, item, zotero: "ZoteroPdfIndex | None"
) -> Doc | None:
"""Attachment sections (bib, else Zotero-only storage PDFs) +
abstract; None when empty."""
from rex.comments.combine import extract_attachment
sections, files = _attachment_sections(store, item.key)
if not sections and zotero is not None:
for pdf in zotero.pdfs_for(item.key):
result = extract_attachment(pdf)
if result.status == "ok" and result.text.strip():
sections.append(f"## {pdf.name}\n\n{result.text.strip()}")
files.append((pdf.name, str(pdf)))
abstract = item.abstract.strip()
parts = sections + ([abstract] if abstract else [])
text = "\n\n".join(parts)
if not text.strip():
return None
project = next(
(t.split(":", 1)[1] for t in item.tags if t.startswith("project:")), ""
)
return Doc(
key=item.key,
text=text,
metadata={
"doctype": item.item_type,
"kind": "corpus",
"year": _year_of(store, item.key),
"date": (item.date_published or "")[:10],
"title": item.title,
"url": item.url or "",
"project": project,
},
files=tuple(files),
)
def iter_corpus_docs( def iter_corpus_docs(
store: Store, *, tag: str = "", zotero: ZoteroPdfIndex | None = None store: Store, *, tag: str = "", zotero: ZoteroPdfIndex | None = None
) -> Iterator[Doc]: ) -> Iterator[Doc]:
"""Every non-comment item: attachment sections (bib, else Zotero-only """Every non-comment item: attachment sections (bib, else Zotero-only
storage PDFs) + abstract.""" storage PDFs) + abstract."""
from rex.comments.combine import extract_attachment for ref in iter_corpus_refs(store, tag=tag, zotero=zotero):
doc = ref.load()
for item in store.list_items(tag=tag): if doc is not None:
if "doctype:comment" in item.tags: yield doc
continue
sections, files = _attachment_sections(store, item.key)
if not sections and zotero is not None:
for pdf in zotero.pdfs_for(item.key):
result = extract_attachment(pdf)
if result.status == "ok" and result.text.strip():
sections.append(f"## {pdf.name}\n\n{result.text.strip()}")
files.append((pdf.name, str(pdf)))
abstract = item.abstract.strip()
parts = sections + ([abstract] if abstract else [])
text = "\n\n".join(parts)
if not text.strip():
continue
project = next(
(t.split(":", 1)[1] for t in item.tags if t.startswith("project:")), ""
)
yield Doc(
key=item.key,
text=text,
metadata={
"doctype": item.item_type,
"kind": "corpus",
"year": _year_of(store, item.key),
"date": (item.date_published or "")[:10],
"title": item.title,
"url": item.url or "",
"project": project,
},
files=tuple(files),
)

View File

@@ -120,6 +120,38 @@ def derive_siblings_from_combined(comment_dir: Path) -> list[Path]:
return written return written
def source_paths(comment_dir: Path) -> list[Path]:
"""Files that feed extraction: attachments and bodies, never our own
outputs (``combined.md``, ``*.md`` siblings, ``*.tmp``) or dotfiles."""
out: list[Path] = []
for p in comment_dir.iterdir():
n = p.name
if not p.is_file() or n.startswith(".") or n == _FILENAME:
continue
if n.endswith(".md") or n.endswith(".tmp"):
continue
out.append(p)
return sorted(out)
def is_current(comment_dir: Path) -> bool:
"""True when ``combined.md`` exists and is at least as new as every
source file — a stat-only check, no reads."""
out = comment_dir / _FILENAME
try:
out_m = out.stat().st_mtime_ns
except FileNotFoundError:
return False
for p in source_paths(comment_dir):
try:
newer = p.stat().st_mtime_ns > out_m
except FileNotFoundError:
continue # vanished between iterdir() and stat() — not our business
if newer:
return False
return True
def parse_combined(text: str) -> tuple[dict[str, Any], str]: def parse_combined(text: str) -> tuple[dict[str, Any], str]:
"""Split a combined.md into (frontmatter dict, body markdown).""" """Split a combined.md into (frontmatter dict, body markdown)."""
if not text.startswith("---\n"): if not text.startswith("---\n"):
@@ -136,11 +168,7 @@ def parse_combined(text: str) -> tuple[dict[str, Any], str]:
def _attachment_paths(comment_dir: Path) -> list[Path]: def _attachment_paths(comment_dir: Path) -> list[Path]:
return [ return source_paths(comment_dir)
p
for p in comment_dir.iterdir()
if p.is_file() and p.name != _FILENAME and not p.name.startswith(".")
]
def _render_section(filename: str, result: ExtractResult) -> str: def _render_section(filename: str, result: ExtractResult) -> str:

View File

@@ -12,12 +12,14 @@ from collections.abc import Callable
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path from pathlib import Path
from rex.comments.combine import derive_siblings_from_combined, extract_comment from rex.comments.combine import (
derive_siblings_from_combined,
extract_comment,
is_current,
)
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
_COMBINED = "combined.md"
def walk_and_extract( def walk_and_extract(
root: Path, root: Path,
@@ -28,6 +30,8 @@ def walk_and_extract(
workers: int = 8, workers: int = 8,
force: bool = False, force: bool = False,
on_extracted: Callable[[str, Path], None] | None = None, on_extracted: Callable[[str, Path], None] | None = None,
skip_dockets: set[str] | frozenset[str] = frozenset(),
reattach: bool = False,
) -> dict[str, int]: ) -> dict[str, int]:
"""Process every comment dir under *root*. """Process every comment dir under *root*.
@@ -36,21 +40,22 @@ def walk_and_extract(
``comment_id`` and returns the inline body text from bib.sqlite (or ``comment_id`` and returns the inline body text from bib.sqlite (or
"" if missing). "" if missing).
*on_extracted*, if given, is called on the main thread with A dir is *current* when its combined.md is newer than every source
``(comment_id, comment_dir)`` for every dir that ends up with a file (``combine.is_current``); current dirs are skipped without being
combined.md — both newly-written ones and pre-existing skips. The read. *on_extracted* fires for newly written dirs; with *reattach*
callback is responsible for discovering whatever markdown files it it also fires for skipped dirs (after deriving any missing sibling
wants to consume in the dir (combined.md plus per-attachment MDs), which is the repair path for bib notes. *skip_dockets* names
siblings). Skipped dirs that pre-date the sibling-write feature get docket dirs to ignore entirely (sealed dockets), unless *force*.
siblings derived from combined.md before the callback fires, so the
callback always sees a complete set.
Returns counts: ``{"written": N, "skipped": N, "failed": N}``. Returns counts: ``{"written": N, "skipped": N, "failed": N}``, plus
``"skipped_sealed": N`` when any docket was skipped via *skip_dockets*.
""" """
candidate_dirs, skipped_dirs = _collect_dirs( candidate_dirs, skipped_dirs, sealed = _collect_dirs(
root, docket=docket, limit=limit, force=force root, docket=docket, limit=limit, force=force, skip_dockets=skip_dockets
) )
stats = {"written": 0, "skipped": len(skipped_dirs), "failed": 0} stats = {"written": 0, "skipped": len(skipped_dirs), "failed": 0}
if sealed:
stats["skipped_sealed"] = sealed
if candidate_dirs: if candidate_dirs:
with ThreadPoolExecutor(max_workers=workers) as pool: with ThreadPoolExecutor(max_workers=workers) as pool:
@@ -65,7 +70,7 @@ def walk_and_extract(
cdir = futures[fut] cdir = futures[fut]
on_extracted(cdir.name, cdir) on_extracted(cdir.name, cdir)
if on_extracted: if on_extracted and reattach:
for cdir in skipped_dirs: for cdir in skipped_dirs:
try: try:
derive_siblings_from_combined(cdir) derive_siblings_from_combined(cdir)
@@ -82,31 +87,39 @@ def _collect_dirs(
docket: str | None, docket: str | None,
limit: int | None, limit: int | None,
force: bool, force: bool,
) -> tuple[list[Path], list[Path]]: skip_dockets: set[str] | frozenset[str] = frozenset(),
"""Return ``(dirs_to_process, dirs_skipped)``. ) -> tuple[list[Path], list[Path], int]:
"""Return ``(dirs_to_process, dirs_skipped, sealed_docket_count)``.
``dirs_skipped`` are dirs that already have combined.md and would not ``dirs_skipped`` are dirs that are already current (``is_current``)
be re-processed. We return them as a list (not a count) so callers and would not be re-processed. We return them as a list (not a count)
can still drive per-dir post-processing — e.g. attaching the existing so callers can still drive per-dir post-processing — e.g. attaching
combined.md to bib on a re-run. the existing combined.md to bib on a re-run (``reattach``).
``sealed_docket_count`` counts docket dirs skipped entirely because
their name is in *skip_dockets* (and *force* is False).
""" """
todo: list[Path] = [] todo: list[Path] = []
skipped: list[Path] = [] skipped: list[Path] = []
sealed = 0
for docket_dir in sorted(root.iterdir()): for docket_dir in sorted(root.iterdir()):
if not docket_dir.is_dir(): if not docket_dir.is_dir():
continue continue
if docket and docket_dir.name != docket: if docket and docket_dir.name != docket:
continue continue
if docket_dir.name in skip_dockets and not force:
sealed += 1
continue
for cdir in sorted(docket_dir.iterdir()): for cdir in sorted(docket_dir.iterdir()):
if not cdir.is_dir(): if not cdir.is_dir():
continue continue
if (cdir / _COMBINED).is_file() and not force: if not force and is_current(cdir):
skipped.append(cdir) skipped.append(cdir)
continue continue
todo.append(cdir) todo.append(cdir)
if limit and len(todo) >= limit: if limit and len(todo) >= limit:
return todo, skipped return todo, skipped, sealed
return todo, skipped return todo, skipped, sealed
def _one( def _one(
@@ -115,7 +128,7 @@ def _one(
force: bool, force: bool,
) -> str: ) -> str:
"""Process one comment dir. Returns "written" / "skipped" / "failed".""" """Process one comment dir. Returns "written" / "skipped" / "failed"."""
if (comment_dir / _COMBINED).is_file() and not force: if not force and is_current(comment_dir):
return "skipped" return "skipped"
try: try:
body = inline_body_lookup(comment_dir.name) body = inline_body_lookup(comment_dir.name)
@@ -123,7 +136,7 @@ def _one(
log.warning("inline body lookup failed for %s: %s", comment_dir.name, e) log.warning("inline body lookup failed for %s: %s", comment_dir.name, e)
body = "" body = ""
try: try:
extract_comment(comment_dir, inline_body=body, force=force) extract_comment(comment_dir, inline_body=body, force=True)
except Exception as e: # noqa: BLE001 except Exception as e: # noqa: BLE001
log.warning("extract_comment failed for %s: %s", comment_dir, e) log.warning("extract_comment failed for %s: %s", comment_dir, e)
return "failed" return "failed"

View File

@@ -79,6 +79,9 @@ aco = "data/aco.duckdb"
bib = "data/bib.sqlite" bib = "data/bib.sqlite"
zotero = "data/zotero/data/zotero.sqlite" zotero = "data/zotero/data/zotero.sqlite"
[comments]
seal_quiet_days = 30 # days after a docket's comment close date before an empty pull seals it
[storage] [storage]
bib = "data/bib/storage" bib = "data/bib/storage"
zotero = "data/zotero/data/storage" zotero = "data/zotero/data/storage"

View File

@@ -0,0 +1,44 @@
from __future__ import annotations
from unittest.mock import MagicMock
from bib.dockets import Docket
from bib.regulations_gov import backfill_details, backfill_from_mirror
from bib.store import Store
D = "CMS-2019-0111"
def _sealed_store() -> Store:
s = Store(":memory:", storage_dir="/tmp/nope")
s.docket_upsert(
Docket(id=D, sealed_at="2020-06-01T00:00:00Z", seal_reason="manual")
)
return s
def test_api_backfill_skips_sealed_without_calls():
s = _sealed_store()
api = MagicMock()
stats = backfill_details(s, api, docket=D)
assert stats["skipped_sealed"] == 1 and stats["seen"] == 0
api.get_comment_detail.assert_not_called()
def test_mirror_backfill_skips_sealed_without_listing():
s = _sealed_store()
m = MagicMock()
stats = backfill_from_mirror(s, m, docket=D)
assert stats["skipped_sealed"] == 1 and stats["seen"] == 0
m.comment_ids.assert_not_called()
m.attachment_keys.assert_not_called()
def test_force_runs_sealed_mirror_backfill():
s = _sealed_store()
m = MagicMock()
m.comment_ids.return_value = []
m.attachment_keys.return_value = {}
stats = backfill_from_mirror(s, m, docket=D, force=True)
assert "skipped_sealed" not in stats
m.comment_ids.assert_called_once_with(D)

115
tests/bib/test_dockets.py Normal file
View File

@@ -0,0 +1,115 @@
"""bib.dockets — pure docket state helpers."""
from __future__ import annotations
import os
from datetime import date
from pathlib import Path
from bib.dockets import Docket, fingerprint_files, quiet_days, should_seal
def _d(**kw) -> Docket:
base = dict(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_document_id="CMS-2026-2377-0001",
fr_object_id="0900006482921ba1",
comment_end_date="2026-09-14",
pull_watermark="",
last_pull_at="",
last_pull_new=None,
sealed_at="",
seal_reason="",
counts_json="{}",
)
base.update(kw)
return Docket(**base)
class TestShouldSeal:
def test_seals_after_quiet_period_with_no_new(self):
d = _d(last_pull_at="2026-10-15T03:00:00Z", last_pull_new=0)
assert should_seal(d, date(2026, 10, 15), 30) is True
def test_not_before_quiet_period(self):
d = _d(last_pull_at="2026-10-13T03:00:00Z", last_pull_new=0)
assert should_seal(d, date(2026, 10, 13), 30) is False
def test_not_when_last_pull_found_new(self):
d = _d(last_pull_at="2026-10-20T03:00:00Z", last_pull_new=3)
assert should_seal(d, date(2026, 10, 20), 30) is False
def test_not_when_last_pull_predates_quiet_period(self):
# Pull happened before end+quiet even though today is well past it.
d = _d(last_pull_at="2026-09-20T03:00:00Z", last_pull_new=0)
assert should_seal(d, date(2026, 12, 1), 30) is False
def test_not_without_close_date(self):
d = _d(
comment_end_date="", last_pull_at="2027-01-01T00:00:00Z", last_pull_new=0
)
assert should_seal(d, date(2027, 1, 1), 30) is False
def test_not_when_already_sealed(self):
d = _d(
sealed_at="2026-10-15T00:00:00Z",
last_pull_at="2026-10-15T03:00:00Z",
last_pull_new=0,
)
assert should_seal(d, date(2026, 10, 15), 30) is False
def test_not_when_never_pulled(self):
d = _d(last_pull_at="", last_pull_new=None)
assert should_seal(d, date(2027, 1, 1), 30) is False
class TestFingerprintFiles:
def test_order_independent(self, tmp_path: Path):
a = tmp_path / "a.pdf"
b = tmp_path / "b.pdf"
a.write_bytes(b"aa")
b.write_bytes(b"bbb")
assert fingerprint_files([a, b]) == fingerprint_files([b, a])
def test_changes_when_size_changes(self, tmp_path: Path):
a = tmp_path / "a.pdf"
a.write_bytes(b"aa")
f1 = fingerprint_files([a])
a.write_bytes(b"aaaa")
assert fingerprint_files([a]) != f1
def test_changes_when_mtime_changes(self, tmp_path: Path):
a = tmp_path / "a.pdf"
a.write_bytes(b"aa")
f1 = fingerprint_files([a])
os.utime(a, ns=(1_000_000_000_000_000_000, 1_000_000_000_000_000_000))
assert fingerprint_files([a]) != f1
def test_missing_file_is_recorded_not_fatal(self, tmp_path: Path):
assert fingerprint_files([tmp_path / "nope.pdf"]) == fingerprint_files(
[tmp_path / "nope.pdf"]
)
assert fingerprint_files([tmp_path / "nope.pdf"]) != fingerprint_files([])
def test_empty_is_stable(self):
assert fingerprint_files([]) == fingerprint_files([])
assert len(fingerprint_files([])) == 64
class TestQuietDays:
def test_default_when_section_missing(self, monkeypatch):
import bib.dockets as mod
from conf import _Cfg
monkeypatch.setattr(mod, "_cfg", lambda: _Cfg({}))
assert quiet_days() == 30
def test_reads_section(self, monkeypatch):
import bib.dockets as mod
from conf import _Cfg
monkeypatch.setattr(
mod, "_cfg", lambda: _Cfg({"comments": {"seal_quiet_days": 7}})
)
assert quiet_days() == 7

View File

@@ -156,7 +156,7 @@ class TestParseComment:
class TestUpsertComment: class TestUpsertComment:
def test_creates_item(self): def test_creates_item(self):
store = MagicMock() store = MagicMock()
store.upsert.return_value = "KEY1" store.upsert_status.return_value = ("KEY1", "created")
c = Comment( c = Comment(
id="CMS-2017-0092-0002", id="CMS-2017-0092-0002",
title="Test", title="Test",
@@ -168,8 +168,8 @@ class TestUpsertComment:
) )
key = upsert_comment(store, c, cms_id="CMS-1676-P") key = upsert_comment(store, c, cms_id="CMS-1676-P")
assert key == "KEY1" assert key == "KEY1"
store.upsert.assert_called_once() store.upsert_status.assert_called_once()
item = store.upsert.call_args[0][0] item = store.upsert_status.call_args[0][0]
assert "AMA" in item.title assert "AMA" in item.title
assert item.url == "https://www.regulations.gov/comment/CMS-2017-0092-0002" assert item.url == "https://www.regulations.gov/comment/CMS-2017-0092-0002"

View File

@@ -114,7 +114,7 @@ class TestByline:
class TestUpsertComment: class TestUpsertComment:
def test_full(self): def test_full(self):
store = MagicMock() store = MagicMock()
store.upsert.return_value = "KEY1" store.upsert_status.return_value = ("KEY1", "created")
c = Comment( c = Comment(
id="CMS-2023-0001-0001", id="CMS-2023-0001-0001",
title="My Comment", title="My Comment",
@@ -130,7 +130,7 @@ class TestUpsertComment:
def test_no_title_uses_comment_on_id(self): def test_no_title_uses_comment_on_id(self):
store = MagicMock() store = MagicMock()
store.upsert.return_value = "KEY2" store.upsert_status.return_value = ("KEY2", "created")
c = Comment( c = Comment(
id="C2", id="C2",
title="", title="",

View File

@@ -0,0 +1,65 @@
from __future__ import annotations
import httpx
from bib.regulations_gov import Client
def _row(cid: str, lm: str) -> dict:
return {
"id": cid,
"attributes": {
"lastModifiedDate": lm,
"postedDate": lm,
"docketId": "D",
"commentOnId": "x",
},
}
def _client(handler) -> Client:
http = httpx.Client(
transport=httpx.MockTransport(handler), headers={"X-Api-Key": "k"}
)
return Client(api_key="k", sleep=0, client=http)
def test_since_seeds_the_date_filter():
seen: list[dict] = []
def handler(req: httpx.Request) -> httpx.Response:
seen.append(dict(req.url.params))
return httpx.Response(
200,
json={
"data": [_row("D-1", "2026-09-08T12:00:00Z")],
"meta": {"totalPages": 1},
},
)
api = _client(handler)
out = list(api.iter_comments("obj", since="2026-09-01T00:00:00Z"))
assert [c.id for c in out] == ["D-1"]
assert seen[0]["filter[lastModifiedDate][ge]"] == "2026-09-01 00:00:00"
def test_no_since_means_no_date_filter():
seen: list[dict] = []
def handler(req: httpx.Request) -> httpx.Response:
seen.append(dict(req.url.params))
return httpx.Response(200, json={"data": [], "meta": {"totalPages": 1}})
list(_client(handler).iter_comments("obj"))
assert "filter[lastModifiedDate][ge]" not in seen[0]
def test_on_error_called_when_page_fails():
errors: list[Exception] = []
def handler(req: httpx.Request) -> httpx.Response:
return httpx.Response(500, json={})
out = list(_client(handler).iter_comments("obj", on_error=errors.append))
assert out == []
assert len(errors) == 1 and isinstance(errors[0], httpx.HTTPStatusError)

View File

@@ -0,0 +1,30 @@
from __future__ import annotations
from pathlib import Path
from bib.item import Source
from bib.store import Store
def test_second_attach_of_same_filename_returns_existing_key(tmp_path: Path):
s = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
key = s.create(Source(title="T", url="https://x/1"))
f = tmp_path / "attachment_1.pdf"
f.write_bytes(b"%PDF")
k1 = s.attach_file(key, f, title="attachment_1.pdf")
k2 = s.attach_file(key, f, title="attachment_1.pdf")
assert k1 == k2
n = s._con().execute("SELECT count(*) FROM attachments").fetchone()[0]
assert n == 1
# exactly one storage copy
assert len(list((tmp_path / "storage").iterdir())) == 1
def test_different_filename_creates_second_row(tmp_path: Path):
s = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
key = s.create(Source(title="T", url="https://x/1"))
f1 = tmp_path / "a.pdf"
f1.write_bytes(b"a")
f2 = tmp_path / "b.pdf"
f2.write_bytes(b"b")
assert s.attach_file(key, f1) != s.attach_file(key, f2)

View File

@@ -0,0 +1,76 @@
"""Store.docket_* — persistence for bib.dockets.Docket."""
from __future__ import annotations
from bib.dockets import Docket
from bib.store import Store
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def test_table_exists_after_init():
s = _store()
names = {
r[0]
for r in s._con().execute("SELECT name FROM sqlite_master WHERE type='table'")
}
assert "dockets" in names
idx = {
r[0]
for r in s._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
}
assert "idx_items_url" in idx
def test_upsert_then_get_roundtrip():
s = _store()
d = Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="obj",
comment_end_date="2026-09-14",
)
s.docket_upsert(d)
got = s.docket_get("CMS-2026-2377")
assert got == d
assert s.docket_get("CMS-0000-0000") is None
def test_upsert_replaces_fields():
s = _store()
s.docket_upsert(Docket(id="D1", pull_watermark="a"))
s.docket_upsert(Docket(id="D1", pull_watermark="b", last_pull_new=4))
got = s.docket_get("D1")
assert got.pull_watermark == "b"
assert got.last_pull_new == 4
def test_docket_for_rule():
s = _store()
s.docket_upsert(Docket(id="D1", rule_cms_id="CMS-1848-P"))
assert s.docket_for_rule("CMS-1848-P").id == "D1"
assert s.docket_for_rule("CMS-9999-P") is None
def test_seal_unseal_and_sealed_dockets():
s = _store()
s.docket_upsert(Docket(id="D1"))
s.docket_upsert(Docket(id="D2"))
s.docket_seal("D1", reason="manual", counts={"comments": 3})
d1 = s.docket_get("D1")
assert d1.sealed and d1.seal_reason == "manual"
assert '"comments": 3' in d1.counts_json
assert set(s.sealed_dockets()) == {"D1"}
assert s.sealed_dockets()["D1"] == d1.sealed_at
s.docket_unseal("D1")
assert not s.docket_get("D1").sealed
assert s.sealed_dockets() == {}
def test_dockets_lists_sorted_by_id():
s = _store()
s.docket_upsert(Docket(id="CMS-2019-0111"))
s.docket_upsert(Docket(id="CMS-2017-0092"))
assert [d.id for d in s.dockets()] == ["CMS-2017-0092", "CMS-2019-0111"]

View File

@@ -0,0 +1,98 @@
"""Tag/collection changes must stamp ``items.updated_at``.
The llm indexer fingerprints an item as ``updated_at`` + its file stats,
so a tag-only edit (``year:``, ``project:``, ``cms-rule:`` — all of which
land in chunk metadata) is invisible to a re-index unless the row is
stamped. An *unchanged* upsert must still write nothing.
"""
from __future__ import annotations
from bib.item import Source
from bib.store import Store
OLD = "2000-01-01T00:00:00Z"
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def _item(**kw) -> Source:
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
it.abstract = kw.get("abstract", "body")
for t in kw.get("tags", ["a:1", "b:2"]):
it.add_tag(t)
return it
def _backdate(s: Store, key: str) -> None:
"""updated_at has 1s granularity — backdate so a bump is observable."""
s._con().execute("UPDATE items SET updated_at = ? WHERE key = ?", (OLD, key))
s._con().commit()
def _updated_at(s: Store, key: str) -> str:
return (
s._con()
.execute("SELECT updated_at FROM items WHERE key = ?", (key,))
.fetchone()["updated_at"]
)
def test_add_tag_bumps_updated_at():
s = _store()
key = s.create(_item())
_backdate(s, key)
s.add_tag(key, "year:2026")
assert _updated_at(s, key) != OLD
def test_re_adding_the_same_tag_does_not_bump():
s = _store()
key = s.create(_item())
s.add_tag(key, "year:2026")
_backdate(s, key)
s.add_tag(key, "year:2026") # already there — nothing changed
assert _updated_at(s, key) == OLD
def test_remove_tag_bumps_updated_at():
s = _store()
key = s.create(_item(tags=["year:2026"]))
_backdate(s, key)
s.remove_tag(key, "year:2026")
assert _updated_at(s, key) != OLD
def test_removing_an_absent_tag_does_not_bump():
s = _store()
key = s.create(_item())
_backdate(s, key)
s.remove_tag(key, "nope:1")
assert _updated_at(s, key) == OLD
def test_update_with_only_tags_bumps_updated_at():
s = _store()
key = s.create(_item())
_backdate(s, key)
s.update(key, tags=["year:2026", "project:pfs"])
assert _updated_at(s, key) != OLD
def test_update_with_only_collections_bumps_updated_at():
s = _store()
key = s.create(_item())
_backdate(s, key)
s.update(key, collections=[])
assert _updated_at(s, key) != OLD
def test_unchanged_upsert_still_does_not_bump():
s = _store()
key, _ = s.upsert_status(_item())
_backdate(s, key)
key2, status = s.upsert_status(_item())
assert (key2, status) == (key, "unchanged")
assert _updated_at(s, key) == OLD

View File

@@ -0,0 +1,104 @@
"""Store.upsert must not rewrite rows/tags when nothing changed."""
from __future__ import annotations
from bib.item import Source
from bib.store import Store
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def _item(**kw) -> Source:
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
it.abstract = kw.get("abstract", "body")
for t in kw.get("tags", ["a:1", "b:2"]):
it.add_tag(t)
return it
def _snapshot(s: Store, key: str) -> tuple:
con = s._con()
row = con.execute(
"SELECT access_date, updated_at FROM items WHERE key=?", (key,)
).fetchone()
tags = con.execute(
"SELECT it.rowid FROM item_tags it JOIN items i ON i.id=it.item_id WHERE i.key=? ORDER BY 1",
(key,),
).fetchall()
return (row["access_date"], row["updated_at"], [t[0] for t in tags])
def test_first_upsert_is_created():
s = _store()
key, status = s.upsert_status(_item())
assert status == "created" and key
def test_identical_upsert_is_unchanged_and_writes_nothing():
s = _store()
key, _ = s.upsert_status(_item())
before = _snapshot(s, key)
# different Python object, same content
key2, status = s.upsert_status(_item())
assert key2 == key and status == "unchanged"
assert _snapshot(s, key) == before # no access/updated stamp, no tag row churn
def test_changed_abstract_is_updated():
s = _store()
key, _ = s.upsert_status(_item(abstract="v1"))
_, status = s.upsert_status(_item(abstract="v2"))
assert status == "updated"
assert s.get(key).abstract == "v2"
def test_new_tag_is_updated_and_merged():
s = _store()
key, _ = s.upsert_status(_item(tags=["a:1"]))
_, status = s.upsert_status(_item(tags=["c:3"]))
assert status == "updated"
assert set(s.get(key).tags) >= {"a:1", "c:3"}
def test_subset_of_existing_tags_is_unchanged():
"""Upsert never removes tags (#624); a subset therefore changes nothing."""
s = _store()
key, _ = s.upsert_status(_item(tags=["a:1", "b:2"]))
_, status = s.upsert_status(_item(tags=["a:1"]))
assert status == "unchanged"
def test_upsert_keeps_returning_key():
s = _store()
key = s.upsert(_item())
assert s.upsert(_item()) == key
def test_empty_incoming_abstract_keeps_stored_body():
"""A list-walk row (no body) must not blank an enriched comment."""
s = _store()
key, _ = s.upsert_status(_item(abstract="enriched body"))
_, status = s.upsert_status(_item(abstract=""))
assert status == "unchanged"
assert s.get(key).abstract == "enriched body"
def test_empty_incoming_title_keeps_stored_title():
s = _store()
it = _item()
it.title = "Org: CMS-2026-2377-1"
key, _ = s.upsert_status(it)
it2 = _item()
it2.title = ""
_, status = s.upsert_status(it2)
assert status == "unchanged"
assert s.get(key).title == "Org: CMS-2026-2377-1"
def test_non_empty_incoming_still_updates():
s = _store()
key, _ = s.upsert_status(_item(abstract="v1"))
_, status = s.upsert_status(_item(abstract="v2"))
assert status == "updated" and s.get(key).abstract == "v2"

View File

@@ -0,0 +1,212 @@
from __future__ import annotations
import json
from datetime import date
from unittest.mock import MagicMock
import httpx
from bib.dockets import Docket
from bib.regulations_gov import (
Attachment,
Comment,
WalkResult,
discover_docket,
walk_docket,
)
from bib.store import Store
D = "CMS-2026-2377"
def _c(n: int, lm: str) -> Comment:
return Comment(
id=f"{D}-{n}",
title="",
posted_date=lm[:10],
received_date=lm[:10],
docket_id=D,
comment_on_id="x",
raw={"attributes": {"lastModifiedDate": lm}},
)
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def _docket(**kw) -> Docket:
base = dict(
id=D,
rule_cms_id="CMS-1848-P",
fr_object_id="obj",
comment_end_date="2026-09-14",
)
base.update(kw)
return Docket(**base)
def test_discover_docket_picks_commentable_doc():
api = MagicMock()
api.find_documents_in_docket.return_value = [
{"id": "X-1", "attributes": {"objectId": "o1"}}, # no comment window
{
"id": "X-2",
"attributes": {"objectId": "o2", "commentEndDate": "2026-09-14T03:59:59Z"},
},
]
d = discover_docket(api, D, rule_cms_id="CMS-1848-P")
assert d == Docket(
id=D,
rule_cms_id="CMS-1848-P",
fr_document_id="X-2",
fr_object_id="o2",
comment_end_date="2026-09-14",
)
api.find_documents_in_docket.assert_called_once_with(D)
def test_walk_creates_counts_and_advances_watermark():
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = [
_c(1, "2026-09-01T00:00:00Z"),
_c(2, "2026-09-02T00:00:00Z"),
]
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
assert r == WalkResult(
created=2,
updated=0,
unchanged=0,
watermark="2026-09-02T00:00:00Z",
clean=True,
sealed=False,
)
d = s.docket_get(D)
assert d.pull_watermark == "2026-09-02T00:00:00Z"
assert d.last_pull_new == 2 and d.last_pull_at
api.iter_comments.assert_called_once()
assert api.iter_comments.call_args.kwargs["since"] == ""
def test_walk_passes_watermark_and_counts_unchanged():
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = [_c(1, "2026-09-01T00:00:00Z")]
walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
api.iter_comments.return_value = [
_c(1, "2026-09-01T00:00:00Z")
] # boundary re-yield
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
assert api.iter_comments.call_args.kwargs["since"] == "2026-09-01T00:00:00Z"
assert (r.created, r.unchanged) == (0, 1)
assert s.docket_get(D).last_pull_new == 0
def test_unclean_walk_keeps_old_watermark():
s = _store()
s.docket_upsert(_docket(pull_watermark="2026-08-01T00:00:00Z"))
api = MagicMock()
def _iter(_obj, *, since="", on_error=None):
yield _c(9, "2026-09-05T00:00:00Z")
on_error(
httpx.HTTPStatusError("boom", request=MagicMock(), response=MagicMock())
)
api.iter_comments.side_effect = _iter
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
assert r.clean is False and r.created == 1
assert s.docket_get(D).pull_watermark == "2026-08-01T00:00:00Z"
def test_force_walks_from_scratch_and_never_seals():
s = _store()
s.docket_upsert(_docket(pull_watermark="2026-08-01T00:00:00Z"))
api = MagicMock()
api.iter_comments.return_value = []
r = walk_docket(s, api, s.docket_get(D), force=True, today=date(2027, 1, 1))
assert api.iter_comments.call_args.kwargs["since"] == ""
assert r.sealed is False and not s.docket_get(D).sealed
def test_auto_seal_after_quiet_empty_pull():
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = [_c(1, "2026-09-01T00:00:00Z")]
walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
api.iter_comments.return_value = []
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 10, 20), quiet_days=30)
assert r.sealed is True
d = s.docket_get(D)
assert d.sealed and d.seal_reason == "auto"
assert json.loads(d.counts_json)["comments"] == 1
def test_no_auto_seal_when_the_docket_has_no_stored_comments():
"""An empty walk over a docket bib never ingested is a failed farm,
not a finished one — sealing it would freeze zero comments forever."""
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = []
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 10, 20), quiet_days=30)
assert r.sealed is False
assert not s.docket_get(D).sealed
def test_attachments_fetched_for_new_comments_only_unless_forced(tmp_path):
s = _store()
s.docket_upsert(_docket())
att = Attachment(url="https://cdn/att.pdf", filename="att.pdf")
blob = tmp_path / "att.pdf"
blob.write_bytes(b"%PDF-1.4")
api = MagicMock()
api.attachments_for.return_value = [att]
api.download_attachment.return_value = blob
c = _c(1, "2026-09-01T00:00:00Z")
c.attachment_count = 1
api.iter_comments.return_value = [c]
walk_docket(s, api, s.docket_get(D), attachments=True, today=date(2026, 9, 8))
assert api.download_attachment.call_count == 1 # created → fetched
api.iter_comments.return_value = [c]
r = walk_docket(s, api, s.docket_get(D), attachments=True, today=date(2026, 9, 8))
assert r.unchanged == 1
assert api.download_attachment.call_count == 1 # unchanged → not re-fetched
api.iter_comments.return_value = [c]
walk_docket(
s, api, s.docket_get(D), attachments=True, force=True, today=date(2026, 9, 8)
)
assert api.download_attachment.call_count == 2 # --force re-fetches
def test_progress_echo_and_commit_every_50():
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = [_c(i, "2026-09-01T00:00:00Z") for i in range(101)]
echoed: list[str] = []
r = walk_docket(s, api, s.docket_get(D), echo=echoed.append, today=date(2026, 9, 8))
assert r.created == 101
assert " 50 comments" in echoed
assert " 100 comments" in echoed
def test_limit_stops_early_and_leaves_watermark_unadvanced():
s = _store()
s.docket_upsert(_docket())
api = MagicMock()
api.iter_comments.return_value = [
_c(1, "2026-09-01T00:00:00Z"),
_c(2, "2026-09-02T00:00:00Z"),
]
r = walk_docket(s, api, s.docket_get(D), limit=1, today=date(2026, 9, 8))
assert r.created == 1
assert r.clean is False
assert s.docket_get(D).pull_watermark == ""

View File

@@ -106,10 +106,14 @@ class TestDiscoverPfsRulesExercise:
class TestFetchDocketComments: class TestFetchDocketComments:
@patch("bib.connect") @patch("bib.connect")
@patch("bib.regulations_gov.Client") @patch("bib.regulations_gov.Client")
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1") @patch("bib.regulations_gov.discover_docket")
def test_basic(self, mc_upsert, mc_client_cls, mc_connect): @patch("bib.regulations_gov.walk_docket")
store = MagicMock() def test_basic(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
store._con.return_value = MagicMock() from bib.dockets import Docket
from bib.regulations_gov import WalkResult
from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
api = MagicMock() api = MagicMock()
@@ -117,26 +121,26 @@ class TestFetchDocketComments:
api.__exit__ = MagicMock(return_value=False) api.__exit__ = MagicMock(return_value=False)
mc_client_cls.return_value = api mc_client_cls.return_value = api
fr_doc = { mc_disc.return_value = Docket(
"id": "DOC1", id="CMS-1676-P", fr_object_id="09000001", comment_end_date="2023-12-31"
"attributes": {"objectId": "09000001", "commentEndDate": "2023-12-31"}, )
} mc_walk.return_value = WalkResult(created=1)
api.find_documents_in_docket.return_value = [fr_doc]
comment = MagicMock()
comment.id = "C1"
comment.attachment_count = 0
api.iter_comments.return_value = [comment]
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P", "-n", "5"]) result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P", "-n", "5"])
assert result.exit_code == 0 assert result.exit_code == 0
mc_disc.assert_called_once()
mc_walk.assert_called_once()
@patch("bib.connect") @patch("bib.connect")
@patch("bib.regulations_gov.Client") @patch("bib.regulations_gov.Client")
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1") @patch("bib.regulations_gov.discover_docket")
def test_with_attachments(self, mc_upsert, mc_client_cls, mc_connect): @patch("bib.regulations_gov.walk_docket")
store = MagicMock() def test_with_attachments(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
store._con.return_value = MagicMock() from bib.dockets import Docket
from bib.regulations_gov import WalkResult
from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
api = MagicMock() api = MagicMock()
@@ -144,33 +148,27 @@ class TestFetchDocketComments:
api.__exit__ = MagicMock(return_value=False) api.__exit__ = MagicMock(return_value=False)
mc_client_cls.return_value = api mc_client_cls.return_value = api
fr_doc = { mc_disc.return_value = Docket(
"id": "DOC1", id="CMS-1676-P", fr_object_id="09000001", comment_end_date="2023-12-31"
"attributes": {"objectId": "09000001", "commentEndDate": "2023-12-31"}, )
} mc_walk.return_value = WalkResult(created=1)
api.find_documents_in_docket.return_value = [fr_doc]
comment = MagicMock()
comment.id = "C1"
comment.attachment_count = 1
api.iter_comments.return_value = [comment]
att = MagicMock()
att.url = "https://example.com/att.pdf"
att.filename = "att.pdf"
api.attachments_for.return_value = [att]
api.download_attachment.return_value = Path("/tmp/att.pdf")
result = runner.invoke( result = runner.invoke(
app, app,
["fetch-docket-comments", "CMS-1676-P", "--attachments", "-n", "5"], ["fetch-docket-comments", "CMS-1676-P", "--attachments", "-n", "5"],
) )
assert result.exit_code == 0 assert result.exit_code == 0
assert mc_walk.call_args.kwargs["attachments"] is True
@patch("bib.connect") @patch("bib.connect")
@patch("bib.regulations_gov.Client") @patch("bib.regulations_gov.Client")
def test_skip_no_object_id(self, mc_client_cls, mc_connect): @patch("bib.regulations_gov.discover_docket")
store = MagicMock() @patch("bib.regulations_gov.walk_docket")
def test_skip_no_object_id(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
from bib.dockets import Docket
from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
api = MagicMock() api = MagicMock()
@@ -178,11 +176,12 @@ class TestFetchDocketComments:
api.__exit__ = MagicMock(return_value=False) api.__exit__ = MagicMock(return_value=False)
mc_client_cls.return_value = api mc_client_cls.return_value = api
fr_doc = {"id": "DOC1", "attributes": {"objectId": None}} mc_disc.return_value = Docket(id="CMS-1676-P")
api.find_documents_in_docket.return_value = [fr_doc]
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P"]) result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P"])
assert result.exit_code == 0 assert result.exit_code == 0
assert "no commentable FR document" in result.output
mc_walk.assert_not_called()
class TestFetchPfsComments: class TestFetchPfsComments:
@@ -191,12 +190,24 @@ class TestFetchPfsComments:
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1676-P"]) @patch("bib.federalregister.split_docket_ids", return_value=["CMS-1676-P"])
@patch("bib.translate.federal_register") @patch("bib.translate.federal_register")
@patch("bib.regulations_gov.Client") @patch("bib.regulations_gov.Client")
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1") @patch("bib.regulations_gov.discover_docket")
@patch("bib.regulations_gov.walk_docket")
def test_full_loop( def test_full_loop(
self, mc_upsert, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect self,
mc_walk,
mc_disc,
mc_client_cls,
mc_translate,
mc_split,
mc_pfs,
mc_connect,
): ):
store = MagicMock() from bib.dockets import Docket
store._con.return_value = MagicMock() from bib.item import Rule
from bib.regulations_gov import WalkResult
from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
doc = MagicMock() doc = MagicMock()
@@ -207,7 +218,7 @@ class TestFetchPfsComments:
doc.document_number = "2023-12345" doc.document_number = "2023-12345"
mc_pfs.return_value = [doc] mc_pfs.return_value = [doc]
rule = MagicMock() rule = Rule(title="CY2024 PFS NPRM", url="https://example.com")
mc_translate.return_value = rule mc_translate.return_value = rule
api = MagicMock() api = MagicMock()
@@ -216,22 +227,18 @@ class TestFetchPfsComments:
mc_client_cls.return_value = api mc_client_cls.return_value = api
api.resolve_docket.return_value = "CMS-2023-0001" api.resolve_docket.return_value = "CMS-2023-0001"
fr_doc = { mc_disc.return_value = Docket(
"id": "DOC1", id="CMS-2023-0001",
"attributes": { rule_cms_id="CMS-1676-P",
"objectId": "09000001", fr_object_id="09000001",
"commentEndDate": "2023-12-31", comment_end_date="2023-12-31",
}, )
} mc_walk.return_value = WalkResult(created=1)
api.find_documents_in_docket.return_value = [fr_doc]
comment = MagicMock()
comment.id = "C1"
comment.attachment_count = 0
api.iter_comments.return_value = [comment]
result = runner.invoke(app, ["fetch-pfs-comments", "--per-docket-limit", "1"]) result = runner.invoke(app, ["fetch-pfs-comments", "--per-docket-limit", "1"])
assert result.exit_code == 0 assert result.exit_code == 0
mc_disc.assert_called_once()
mc_walk.assert_called_once()
@patch("bib.connect") @patch("bib.connect")
@patch("bib.federalregister.pfs_rules", return_value=[]) @patch("bib.federalregister.pfs_rules", return_value=[])
@@ -252,7 +259,9 @@ class TestFetchPfsComments:
def test_skip_bad_rule( def test_skip_bad_rule(
self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect
): ):
store = MagicMock() from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
doc = MagicMock() doc = MagicMock()
@@ -269,6 +278,8 @@ class TestFetchPfsComments:
result = runner.invoke(app, ["fetch-pfs-comments"]) result = runner.invoke(app, ["fetch-pfs-comments"])
assert result.exit_code == 0 assert result.exit_code == 0
mc_translate.assert_called_once()
assert "skipped rule meta" in result.output
@patch("bib.connect") @patch("bib.connect")
@patch("bib.federalregister.pfs_rules") @patch("bib.federalregister.pfs_rules")
@@ -276,7 +287,9 @@ class TestFetchPfsComments:
@patch("bib.translate.federal_register") @patch("bib.translate.federal_register")
@patch("bib.regulations_gov.Client") @patch("bib.regulations_gov.Client")
def test_no_docket(self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect): def test_no_docket(self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect):
store = MagicMock() from bib.store import Store
store = Store(":memory:", storage_dir="/tmp/nope")
mc_connect.return_value = store mc_connect.return_value = store
doc = MagicMock() doc = MagicMock()

View File

@@ -0,0 +1,244 @@
"""fetch-pfs-comments / fetch-docket-comments must not call reg.gov for
sealed dockets and must reuse stored document ids for known ones."""
from __future__ import annotations
from unittest.mock import MagicMock, patch
from typer.testing import CliRunner
from bib.dockets import Docket
from bib.item import Rule
from bib.regulations_gov import WalkResult
from bib.store import Store
from cli.bib import app
runner = CliRunner()
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def _rule_doc():
doc = MagicMock()
doc.type = "Proposed Rule"
doc.publication_date = "2026-07-16"
doc.dockets = ["CMS-1848-P"]
doc.html_url = "https://example.com"
doc.document_number = "2026-1"
return doc
def _api():
api = MagicMock()
api.__enter__ = MagicMock(return_value=api)
api.__exit__ = MagicMock(return_value=False)
return api
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
@patch("bib.regulations_gov.Client")
@patch("bib.translate.federal_register")
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
@patch("bib.federalregister.pfs_rules")
@patch("bib.connect")
def test_sealed_docket_skipped_before_any_call(
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
):
s = _store()
s.docket_upsert(
Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="o",
sealed_at="2026-10-20T00:00:00Z",
seal_reason="auto",
)
)
mc_connect.return_value = s
mc_pfs.return_value = [_rule_doc()]
api = _api()
mc_client.return_value = api
result = runner.invoke(app, ["fetch-pfs-comments"])
assert result.exit_code == 0, result.output
assert "sealed" in result.output
mc_fr.assert_not_called() # no rule-metadata fetch
api.resolve_docket.assert_not_called()
api.find_documents_in_docket.assert_not_called()
mc_walk.assert_not_called()
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult(created=2))
@patch("bib.regulations_gov.Client")
@patch("bib.translate.federal_register")
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
@patch("bib.federalregister.pfs_rules")
@patch("bib.connect")
def test_known_open_docket_walks_without_resolve(
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
):
s = _store()
s.docket_upsert(
Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="o",
comment_end_date="2026-09-14",
)
)
mc_connect.return_value = s
mc_pfs.return_value = [_rule_doc()]
mc_fr.return_value = Rule(
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
)
api = _api()
mc_client.return_value = api
result = runner.invoke(app, ["fetch-pfs-comments"])
assert result.exit_code == 0, result.output
api.resolve_docket.assert_not_called()
api.find_documents_in_docket.assert_not_called()
mc_walk.assert_called_once()
assert mc_walk.call_args.args[2].id == "CMS-2026-2377"
assert "created=2" in result.output
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
@patch("bib.regulations_gov.discover_docket")
@patch("bib.regulations_gov.Client")
@patch("bib.translate.federal_register")
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
@patch("bib.federalregister.pfs_rules")
@patch("bib.connect")
def test_unknown_docket_is_resolved_once_and_stored(
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_disc, mc_walk
):
s = _store()
mc_connect.return_value = s
mc_pfs.return_value = [_rule_doc()]
mc_fr.return_value = Rule(
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
)
api = _api()
api.resolve_docket.return_value = "CMS-2026-2377"
mc_client.return_value = api
mc_disc.return_value = Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="o",
comment_end_date="2026-09-14",
)
result = runner.invoke(app, ["fetch-pfs-comments"])
assert result.exit_code == 0, result.output
api.resolve_docket.assert_called_once_with("CMS-1848-P")
mc_disc.assert_called_once()
assert s.docket_get("CMS-2026-2377").fr_object_id == "o"
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
@patch("bib.regulations_gov.Client")
@patch("bib.translate.federal_register")
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
@patch("bib.federalregister.pfs_rules")
@patch("bib.connect")
def test_docket_filter_skips_other_known_dockets_without_calls(
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
):
s = _store()
s.docket_upsert(
Docket(id="CMS-2026-2377", rule_cms_id="CMS-1848-P", fr_object_id="o")
)
mc_connect.return_value = s
mc_pfs.return_value = [_rule_doc()]
api = _api()
mc_client.return_value = api
result = runner.invoke(app, ["fetch-pfs-comments", "--docket", "CMS-2019-0111"])
assert result.exit_code == 0, result.output
mc_fr.assert_not_called()
api.resolve_docket.assert_not_called()
mc_walk.assert_not_called()
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
@patch("bib.regulations_gov.Client")
@patch("bib.translate.federal_register")
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
@patch("bib.federalregister.pfs_rules")
@patch("bib.connect")
def test_force_walks_sealed_docket(
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
):
s = _store()
s.docket_upsert(
Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="o",
sealed_at="2026-10-20T00:00:00Z",
)
)
mc_connect.return_value = s
mc_pfs.return_value = [_rule_doc()]
mc_fr.return_value = Rule(
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
)
mc_client.return_value = _api()
result = runner.invoke(app, ["fetch-pfs-comments", "--force"])
assert result.exit_code == 0, result.output
mc_walk.assert_called_once()
assert mc_walk.call_args.kwargs["force"] is True
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult(created=1))
@patch("bib.regulations_gov.discover_docket")
@patch("bib.regulations_gov.Client")
@patch("bib.connect")
def test_fetch_docket_comments_discovers_then_walks(
mc_connect, mc_client, mc_disc, mc_walk
):
s = _store()
mc_connect.return_value = s
mc_client.return_value = _api()
mc_disc.return_value = Docket(
id="CMS-2026-2377",
rule_cms_id="CMS-1848-P",
fr_object_id="o",
comment_end_date="2026-09-14",
)
result = runner.invoke(
app, ["fetch-docket-comments", "CMS-2026-2377", "--cms-id", "CMS-1848-P"]
)
assert result.exit_code == 0, result.output
mc_disc.assert_called_once()
mc_walk.assert_called_once()
assert s.docket_get("CMS-2026-2377") is not None
@patch("bib.regulations_gov.walk_docket")
@patch("bib.regulations_gov.Client")
@patch("bib.connect")
def test_fetch_docket_comments_sealed_notice(mc_connect, mc_client, mc_walk):
s = _store()
s.docket_upsert(
Docket(id="CMS-2019-0111", fr_object_id="o", sealed_at="2020-01-01T00:00:00Z")
)
mc_connect.return_value = s
mc_client.return_value = _api()
result = runner.invoke(app, ["fetch-docket-comments", "CMS-2019-0111"])
assert result.exit_code == 0
assert "sealed" in result.output
mc_walk.assert_not_called()

View File

@@ -0,0 +1,151 @@
from __future__ import annotations
from pathlib import Path
from unittest.mock import MagicMock, patch
import fitz
import pytest
from typer.testing import CliRunner
from bib.dockets import Docket
from bib.store import Store
from cli.comments import app
runner = CliRunner()
@pytest.fixture
def store() -> Store:
s = Store(":memory:", storage_dir="/tmp/nope")
try:
yield s
finally:
s.close()
def _seed_dir(root: Path, docket: str) -> Path:
cdir = root / docket / f"{docket}-0001"
cdir.mkdir(parents=True)
doc = fitz.open()
doc.new_page().insert_text((50, 72), "Long body content " * 10)
doc.save(str(cdir / "attachment_1.pdf"))
doc.close()
return cdir
@patch("bib.connect")
def test_dockets_table_lists_rows(mc_connect, store):
s = store
s.docket_upsert(
Docket(
id="CMS-2019-0111",
rule_cms_id="CMS-1693-P",
comment_end_date="2019-09-27",
sealed_at="2020-01-01T00:00:00Z",
seal_reason="manual",
)
)
s.docket_upsert(
Docket(
id="CMS-2026-2377", rule_cms_id="CMS-1848-P", comment_end_date="2026-09-14"
)
)
mc_connect.return_value = s
result = runner.invoke(app, ["dockets"])
assert result.exit_code == 0, result.output
assert "CMS-2019-0111" in result.output and "sealed" in result.output
assert "CMS-2026-2377" in result.output and "open" in result.output
@patch("bib.connect")
def test_seal_and_unseal(mc_connect, store):
s = store
s.docket_upsert(Docket(id="CMS-2019-0111"))
mc_connect.return_value = s
assert (
runner.invoke(
app, ["seal", "CMS-2019-0111", "--reason", "historical"]
).exit_code
== 0
)
assert s.docket_get("CMS-2019-0111").seal_reason == "historical"
assert runner.invoke(app, ["unseal", "CMS-2019-0111"]).exit_code == 0
assert not s.docket_get("CMS-2019-0111").sealed
@patch("bib.connect")
def test_seal_unknown_docket_fails(mc_connect, store):
mc_connect.return_value = store
result = runner.invoke(app, ["seal", "CMS-0000-0000"])
assert result.exit_code == 1
assert "unknown docket" in result.output
@patch("bib.regulations_gov.discover_docket")
@patch("bib.regulations_gov.Client")
@patch("bib.connect")
def test_dockets_discover_populates_from_tags(mc_connect, mc_client, mc_disc, store):
from bib.item import Source
s = store
it = Source(title="c", url="https://www.regulations.gov/comment/CMS-2019-0111-1")
it.add_tag("reg-docket:CMS-2019-0111")
it.add_tag("rule:CMS-1693-P")
s.create(it)
mc_connect.return_value = s
api = MagicMock()
api.__enter__ = MagicMock(return_value=api)
api.__exit__ = MagicMock(return_value=False)
mc_client.return_value = api
mc_disc.return_value = Docket(
id="CMS-2019-0111",
rule_cms_id="CMS-1693-P",
fr_object_id="o",
comment_end_date="2019-09-27",
)
result = runner.invoke(app, ["dockets", "--discover"])
assert result.exit_code == 0, result.output
mc_disc.assert_called_once_with(api, "CMS-2019-0111", rule_cms_id="CMS-1693-P")
assert s.docket_get("CMS-2019-0111").comment_end_date == "2019-09-27"
@patch("bib.connect")
def test_extract_skips_sealed_docket(mc_connect, tmp_path: Path, store):
s = store
s.docket_upsert(
Docket(
id="CMS-2024-0001", sealed_at="2025-01-01T00:00:00Z", seal_reason="manual"
)
)
mc_connect.return_value = s
cdir = _seed_dir(tmp_path, "CMS-2024-0001")
result = runner.invoke(
app, ["extract", "--root", str(tmp_path), "--workers", "1", "--no-attach"]
)
assert result.exit_code == 0, result.output
assert "skipped_sealed: 1" in result.output
assert not (cdir / "combined.md").exists()
@patch("bib.connect")
def test_extract_force_runs_sealed_docket(mc_connect, tmp_path: Path, store):
s = store
s.docket_upsert(Docket(id="CMS-2024-0001", sealed_at="2025-01-01T00:00:00Z"))
mc_connect.return_value = s
cdir = _seed_dir(tmp_path, "CMS-2024-0001")
result = runner.invoke(
app,
[
"extract",
"--root",
str(tmp_path),
"--workers",
"1",
"--no-attach",
"--force",
],
)
assert result.exit_code == 0, result.output
assert (cdir / "combined.md").is_file()

View File

@@ -10,34 +10,51 @@ from cli.llm import app
runner = CliRunner() runner = CliRunner()
_STATS = {"indexed": 3, "skipped": 1, "chunks": 7} _STATS = {
"indexed": 3,
"skipped": 1,
"chunks": 7,
"fingerprint_skipped": 0,
"hash_skipped": 0,
"docket_complete": 0,
}
class TestIndexComments: class TestIndexComments:
@patch("llm.source.iter_comment_docs") @patch("llm.index.docket_complete", return_value={})
@patch("llm.index._engine")
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config") @patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_docs") @patch("llm.index.index_refs")
@patch("conf.connect.bib") @patch("conf.connect.bib")
@patch("llm.config.load") @patch("llm.config.load")
def test_default_collection_is_comments( def test_default_collection_is_comments(
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter self,
mock_load,
mock_bib,
mock_index_refs,
mock_from_config,
mock_iter,
_engine,
_complete,
): ):
cfg = MagicMock() cfg = MagicMock()
mock_load.return_value = cfg mock_load.return_value = cfg
store = MagicMock() store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store mock_bib.return_value = store
docs = iter(["doc1", "doc2"]) docs = iter(["doc1", "doc2"])
mock_iter.return_value = docs mock_iter.return_value = docs
pool = MagicMock() pool = MagicMock()
mock_from_config.return_value = pool mock_from_config.return_value = pool
mock_index_docs.return_value = _STATS mock_index_refs.return_value = _STATS
result = runner.invoke(app, ["index"]) result = runner.invoke(app, ["index"])
assert result.exit_code == 0 assert result.exit_code == 0, result.output
mock_iter.assert_called_once_with(store, docket="") mock_iter.assert_called_once_with(store, docket="", skip_dockets=set())
mock_index_docs.assert_called_once() mock_index_refs.assert_called_once()
kwargs = mock_index_docs.call_args.kwargs kwargs = mock_index_refs.call_args.kwargs
assert kwargs["collection"] == "comments" assert kwargs["collection"] == "comments"
assert kwargs["cfg"] is cfg assert kwargs["cfg"] is cfg
assert kwargs["pool"] is pool assert kwargs["pool"] is pool
@@ -46,71 +63,82 @@ class TestIndexComments:
class TestIndexCorpus: class TestIndexCorpus:
@patch("llm.source.ZoteroPdfIndex.snapshot") @patch("llm.index.docket_complete", return_value={})
@patch("llm.source.iter_corpus_docs") @patch("llm.index._engine")
@patch("llm.source.ZoteroPdfIndex.lazy")
@patch("llm.source.iter_corpus_refs")
@patch("llm.pool.HostPool.from_config") @patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_docs") @patch("llm.index.index_refs")
@patch("conf.connect.bib") @patch("conf.connect.bib")
@patch("llm.config.load") @patch("llm.config.load")
def test_corpus_collection( def test_corpus_collection(
self, self,
mock_load, mock_load,
mock_bib, mock_bib,
mock_index_docs, mock_index_refs,
mock_from_config, mock_from_config,
mock_iter, mock_iter,
mock_snapshot, mock_lazy,
_engine,
_complete,
): ):
cfg = MagicMock() cfg = MagicMock()
mock_load.return_value = cfg mock_load.return_value = cfg
store = MagicMock() store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store mock_bib.return_value = store
mock_iter.return_value = iter(["doc1"]) mock_iter.return_value = iter(["doc1"])
mock_from_config.return_value = MagicMock() mock_from_config.return_value = MagicMock()
mock_index_docs.return_value = _STATS mock_index_refs.return_value = _STATS
zot = MagicMock() zot = MagicMock()
mock_snapshot.return_value = zot mock_lazy.return_value = zot
result = runner.invoke(app, ["index", "--collection", "corpus"]) result = runner.invoke(app, ["index", "--collection", "corpus"])
assert result.exit_code == 0 assert result.exit_code == 0, result.output
mock_iter.assert_called_once_with(store, zotero=zot) mock_iter.assert_called_once_with(store, zotero=zot)
kwargs = mock_index_docs.call_args.kwargs kwargs = mock_index_refs.call_args.kwargs
assert kwargs["collection"] == "corpus" assert kwargs["collection"] == "corpus"
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
class TestIndexAll: class TestIndexAll:
@patch("llm.source.ZoteroPdfIndex.snapshot") @patch("llm.index.docket_complete", return_value={})
@patch("llm.source.iter_corpus_docs") @patch("llm.index._engine")
@patch("llm.source.iter_rule_docs") @patch("llm.source.ZoteroPdfIndex.lazy")
@patch("llm.source.iter_comment_docs") @patch("llm.source.iter_corpus_refs")
@patch("llm.source.iter_rule_refs")
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config") @patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_docs") @patch("llm.index.index_refs")
@patch("conf.connect.bib") @patch("conf.connect.bib")
@patch("llm.config.load") @patch("llm.config.load")
def test_all_runs_comments_rules_corpus_in_order( def test_all_runs_comments_rules_corpus_in_order(
self, self,
mock_load, mock_load,
mock_bib, mock_bib,
mock_index_docs, mock_index_refs,
mock_from_config, mock_from_config,
mock_comments, mock_comments,
mock_rules, mock_rules,
mock_corpus, mock_corpus,
mock_snapshot, mock_lazy,
_engine,
_complete,
): ):
mock_load.return_value = MagicMock() mock_load.return_value = MagicMock()
mock_bib.return_value = MagicMock() store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store
mock_from_config.return_value = MagicMock() mock_from_config.return_value = MagicMock()
mock_index_docs.return_value = _STATS mock_index_refs.return_value = _STATS
for m in (mock_comments, mock_rules, mock_corpus): for m in (mock_comments, mock_rules, mock_corpus):
m.return_value = iter(["d"]) m.return_value = iter(["d"])
result = runner.invoke(app, ["index", "--collection", "all"]) result = runner.invoke(app, ["index", "--collection", "all"])
assert result.exit_code == 0 assert result.exit_code == 0, result.output
assert [c.kwargs["collection"] for c in mock_index_docs.call_args_list] == [ assert [c.kwargs["collection"] for c in mock_index_refs.call_args_list] == [
"comments", "comments",
"rules", "rules",
"corpus", "corpus",
@@ -119,30 +147,40 @@ class TestIndexAll:
class TestIndexRules: class TestIndexRules:
@patch("llm.source.iter_rule_docs") @patch("llm.index.docket_complete", return_value={})
@patch("llm.index._engine")
@patch("llm.source.iter_rule_refs")
@patch("llm.pool.HostPool.from_config") @patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_docs") @patch("llm.index.index_refs")
@patch("conf.connect.bib") @patch("conf.connect.bib")
@patch("llm.config.load") @patch("llm.config.load")
def test_rules_collection( def test_rules_collection(
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter self,
mock_load,
mock_bib,
mock_index_refs,
mock_from_config,
mock_iter,
_engine,
_complete,
): ):
cfg = MagicMock() cfg = MagicMock()
mock_load.return_value = cfg mock_load.return_value = cfg
store = MagicMock() store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store mock_bib.return_value = store
mock_iter.return_value = iter(["doc1"]) mock_iter.return_value = iter(["doc1"])
mock_from_config.return_value = MagicMock() mock_from_config.return_value = MagicMock()
mock_index_docs.return_value = _STATS mock_index_refs.return_value = _STATS
result = runner.invoke( result = runner.invoke(
app, app,
["index", "--collection", "rules", "--key", "abc123", "--key", "def456"], ["index", "--collection", "rules", "--key", "abc123", "--key", "def456"],
) )
assert result.exit_code == 0 assert result.exit_code == 0, result.output
mock_iter.assert_called_once_with(store, keys=("abc123", "def456")) mock_iter.assert_called_once_with(store, keys=("abc123", "def456"))
kwargs = mock_index_docs.call_args.kwargs kwargs = mock_index_refs.call_args.kwargs
assert kwargs["collection"] == "rules" assert kwargs["collection"] == "rules"
assert "indexed=3 skipped=1 chunks=7" in result.output assert "indexed=3 skipped=1 chunks=7" in result.output
@@ -161,25 +199,37 @@ class TestIndexBadCollection:
class TestIndexLimit: class TestIndexLimit:
@patch("llm.source.iter_comment_docs") @patch("llm.index.docket_complete", return_value={})
@patch("llm.index._engine")
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config") @patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_docs") @patch("llm.index.index_refs")
@patch("conf.connect.bib") @patch("conf.connect.bib")
@patch("llm.config.load") @patch("llm.config.load")
def test_limit_truncates_docs( def test_limit_truncates_docs(
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter self,
mock_load,
mock_bib,
mock_index_refs,
mock_from_config,
mock_iter,
_engine,
_complete,
): ):
mock_load.return_value = MagicMock() mock_load.return_value = MagicMock()
mock_bib.return_value = MagicMock() store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store
mock_iter.return_value = iter([f"doc{i}" for i in range(5)]) mock_iter.return_value = iter([f"doc{i}" for i in range(5)])
mock_from_config.return_value = MagicMock() mock_from_config.return_value = MagicMock()
mock_index_docs.return_value = _STATS mock_index_refs.return_value = _STATS
result = runner.invoke(app, ["index", "--limit", "2"]) result = runner.invoke(app, ["index", "--limit", "2"])
assert result.exit_code == 0 assert result.exit_code == 0, result.output
docs_arg = mock_index_docs.call_args.args[0] refs_arg = mock_index_refs.call_args.args[0]
assert list(docs_arg) == ["doc0", "doc1"] assert list(refs_arg) == ["doc0", "doc1"]
assert mock_index_refs.call_args.kwargs["mark_complete"] is False
class TestServe: class TestServe:

View File

@@ -0,0 +1,151 @@
"""Exercise cli/llm.py's `index` seal-skip wiring: sealed/complete dockets
are excluded from the comment listing, and only pending seals are passed
through to be re-marked."""
from __future__ import annotations
from unittest.mock import MagicMock, patch
from typer.testing import CliRunner
from cli.llm import app
runner = CliRunner()
_STATS = {
"indexed": 0,
"skipped": 0,
"chunks": 0,
"fingerprint_skipped": 5,
"hash_skipped": 0,
"docket_complete": 0,
}
@patch("llm.index._engine")
@patch(
"llm.index.docket_complete",
return_value={"CMS-2019-0111": "s1", "CMS-2020-0088": "old"},
)
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_refs")
@patch("conf.connect.bib")
@patch("llm.config.load")
def test_complete_sealed_dockets_are_not_listed(
mock_load, mock_bib, mock_index, mock_pool, mock_iter, mock_complete, _engine
):
mock_load.return_value = MagicMock()
store = MagicMock()
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1", "CMS-2020-0088": "s2"}
mock_bib.return_value = store
mock_iter.return_value = iter([])
mock_index.return_value = _STATS
result = runner.invoke(app, ["index"])
assert result.exit_code == 0, result.output
kwargs = mock_iter.call_args.kwargs
assert kwargs["skip_dockets"] == {"CMS-2019-0111"} # seal matches → skipped
ikw = mock_index.call_args.kwargs
assert ikw["sealed"] == {
"CMS-2020-0088": "s2"
} # re-sealed docket will be re-marked
assert ikw["mark_complete"] is True
assert "fp_skipped=5" in result.output
@patch("llm.index._engine")
@patch("llm.index.docket_complete", return_value={})
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_refs")
@patch("conf.connect.bib")
@patch("llm.config.load")
def test_force_and_limit_disable_skips_and_completion(
mock_load, mock_bib, mock_index, mock_pool, mock_iter, mock_complete, _engine
):
mock_load.return_value = MagicMock()
store = MagicMock()
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
mock_bib.return_value = store
mock_iter.return_value = iter([])
mock_index.return_value = _STATS
result = runner.invoke(app, ["index", "--force", "--limit", "5"])
assert result.exit_code == 0, result.output
assert mock_iter.call_args.kwargs["skip_dockets"] == set()
assert mock_index.call_args.kwargs["mark_complete"] is False
@patch("llm.index._engine")
@patch("llm.migrate.migrate")
@patch("llm.index.docket_complete", return_value={})
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_refs")
@patch("conf.connect.bib")
@patch("llm.config.load")
def test_migrate_runs_before_docket_complete_and_engine_is_reused(
mock_load,
mock_bib,
mock_index,
mock_pool,
mock_iter,
mock_complete,
mock_migrate,
mock_engine,
):
"""index_docket_state may not exist yet on the first sealed run."""
order: list[str] = []
mock_migrate.side_effect = lambda e: order.append("migrate")
mock_complete.side_effect = lambda e, c: (order.append("complete"), {})[1]
mock_load.return_value = MagicMock()
store = MagicMock()
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
mock_bib.return_value = store
mock_iter.return_value = iter([])
mock_index.return_value = _STATS
result = runner.invoke(app, ["index"])
assert result.exit_code == 0, result.output
assert order == ["migrate", "complete"]
mock_engine.assert_called_once() # one engine, shared
assert mock_migrate.call_args.args[0] is mock_engine.return_value
assert mock_complete.call_args.args[0] is mock_engine.return_value
assert mock_index.call_args.kwargs["engine"] is mock_engine.return_value
@patch("llm.index._engine")
@patch("llm.migrate.migrate")
@patch("llm.index.docket_complete", return_value={})
@patch("llm.source.iter_comment_refs")
@patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_refs")
@patch("conf.connect.bib")
@patch("llm.config.load")
def test_force_touches_neither_engine_nor_migrate(
mock_load,
mock_bib,
mock_index,
mock_pool,
mock_iter,
mock_complete,
mock_migrate,
mock_engine,
):
mock_load.return_value = MagicMock()
store = MagicMock()
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
mock_bib.return_value = store
mock_iter.return_value = iter([])
mock_index.return_value = _STATS
result = runner.invoke(app, ["index", "--force"])
assert result.exit_code == 0, result.output
mock_engine.assert_not_called()
mock_migrate.assert_not_called()
mock_complete.assert_not_called()
assert mock_index.call_args.kwargs["engine"] is None

View File

@@ -0,0 +1,158 @@
from __future__ import annotations
import importlib.util
import sqlite3
import sys
from importlib import resources
from pathlib import Path
from bib.item import Source
from bib.store import Store
_SCRIPT = (
Path(__file__).resolve().parents[2] / "dev" / "scripts" / "dedupe_attachments.py"
)
spec = importlib.util.spec_from_file_location("dedupe_attachments", _SCRIPT)
mod = importlib.util.module_from_spec(spec)
sys.modules["dedupe_attachments"] = mod
spec.loader.exec_module(mod)
def _seed(tmp_path: Path) -> tuple[Store, str]:
# Seed the item + duplicate attachment rows via a raw connection,
# *before* any Store ever opens this file. Store's schema guard adds
# the (item_id, filename) unique index the moment it sees zero
# duplicate groups — true the instant the attachments table exists —
# so seeding duplicates through a Store-opened connection can never
# succeed. This mirrors the real migration scenario: the duplicates
# were written by pre-Task-4 code that predates this guard.
db_path = tmp_path / "bib.sqlite"
storage = tmp_path / "storage"
ddl = resources.files("bib").joinpath("schema.sql").read_text()
raw = sqlite3.connect(str(db_path))
raw.executescript(ddl)
item = Source(title="T", url="https://x/1")
item.stamp_access()
row = item.to_row()
key = row["key"] or "SEEDKEY1"
row["key"] = key
raw.execute(
"""INSERT INTO items
(key, item_type, title, url, date_published,
access_date, abstract, institution, extra, extra_json)
VALUES (:key, :item_type, :title, :url,
:date_published, :access_date, :abstract,
:institution, :extra, :extra_json)""",
row,
)
item_id = raw.execute("SELECT id FROM items WHERE key=?", (key,)).fetchone()[0]
# Simulate the old non-idempotent attach: three rows, three copies.
for k in ("AAAAAAAA", "BBBBBBBB", "CCCCCCCC"):
d = storage / k
d.mkdir(parents=True)
(d / "attachment_1.pdf").write_bytes(b"%PDF-dup")
raw.execute(
"INSERT INTO attachments (item_id, key, filename, content_type, storage_path) VALUES (?,?,?,?,?)",
(
item_id,
k,
"attachment_1.pdf",
"application/pdf",
str(d / "attachment_1.pdf"),
),
)
raw.commit()
raw.close()
s = Store(str(db_path), storage_dir=storage)
return s, key
def test_plan_finds_group_and_keeps_oldest(tmp_path: Path):
s, _ = _seed(tmp_path)
groups = mod.plan(s._con())
assert len(groups) == 1
g = groups[0]
assert g.keep == "AAAAAAAA"
assert sorted(g.remove) == ["BBBBBBBB", "CCCCCCCC"]
def test_apply_removes_rows_and_files(tmp_path: Path):
s, _ = _seed(tmp_path)
rep = mod.apply(s._con(), mod.plan(s._con()))
assert rep.rows_removed == 2
assert rep.files_removed == 2
assert s._con().execute("SELECT count(*) FROM attachments").fetchone()[0] == 1
assert (tmp_path / "storage" / "AAAAAAAA" / "attachment_1.pdf").is_file()
assert not (tmp_path / "storage" / "BBBBBBBB").exists()
assert mod.plan(s._con()) == []
def test_refuses_group_with_differing_sizes(tmp_path: Path):
s, _ = _seed(tmp_path)
(tmp_path / "storage" / "CCCCCCCC" / "attachment_1.pdf").write_bytes(
b"different-longer"
)
groups = mod.plan(s._con())
assert groups[0].conflict is True
rep = mod.apply(s._con(), groups)
assert rep.rows_removed == 0 and rep.skipped_conflicts == 1
def test_unique_index_created_only_when_clean(tmp_path: Path):
s, _ = _seed(tmp_path)
s.close()
s2 = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
idx = {
r[0]
for r in s2._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
}
assert "idx_attachments_item_filename" not in idx
mod.apply(s2._con(), mod.plan(s2._con()))
s2.close()
s3 = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
idx = {
r[0]
for r in s3._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
}
assert "idx_attachments_item_filename" in idx
class _FailAfterFirstDelete:
"""Connection proxy that dies partway through the transaction."""
def __init__(self, con: sqlite3.Connection) -> None:
self._con = con
self.deletes = 0
def execute(self, sql: str, *args):
if sql.lstrip().upper().startswith("DELETE"):
self.deletes += 1
if self.deletes == 2:
raise sqlite3.OperationalError("disk I/O error")
return self._con.execute(sql, *args)
def test_apply_removes_no_file_when_the_transaction_fails(tmp_path: Path):
"""A rollback restores the rows — so the files they point at must
still be there. Unlinking inside the transaction loses data."""
import pytest
s, _ = _seed(tmp_path)
groups = mod.plan(s._con())
flaky = _FailAfterFirstDelete(s._con())
with pytest.raises(sqlite3.OperationalError):
mod.apply(flaky, groups)
assert s._con().execute("SELECT count(*) FROM attachments").fetchone()[0] == 3
for k in ("AAAAAAAA", "BBBBBBBB", "CCCCCCCC"):
assert (tmp_path / "storage" / k / "attachment_1.pdf").is_file()
def test_main_apply_reports_removed_keys(tmp_path: Path, capsys):
s, _ = _seed(tmp_path)
s.close()
rc = mod.main(["--db", str(tmp_path / "bib.sqlite"), "--apply"])
out = capsys.readouterr().out
assert rc == 0
assert "removed rows=2" in out
assert "removed keys (2)" in out
assert "BBBBBBBB" in out and "CCCCCCCC" in out

View File

@@ -57,12 +57,14 @@ class TestIndexDocs:
def test_unchanged_doc_skipped(self): def test_unchanged_doc_skipped(self):
h = content_hash(DOC.text) h = content_hash(DOC.text)
stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h)]) stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h, "")])
assert stats["skipped"] == 1 assert stats["skipped"] == 1
store.add_embeddings.assert_not_called() store.add_embeddings.assert_not_called()
def test_changed_doc_deletes_old_chunks_first(self): def test_changed_doc_deletes_old_chunks_first(self):
stats, store, conn, _ensure_hnsw = _run([DOC], state_rows=[("K1", "stalehash")]) stats, store, conn, _ensure_hnsw = _run(
[DOC], state_rows=[("K1", "stalehash", "")]
)
assert stats["indexed"] == 1 assert stats["indexed"] == 1
deletes = [ deletes = [
c c
@@ -81,7 +83,9 @@ class TestIndexDocs:
def test_force_reembeds_unchanged(self): def test_force_reembeds_unchanged(self):
h = content_hash(DOC.text) h = content_hash(DOC.text)
stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h)], force=True) stats, store, _, _ensure_hnsw = _run(
[DOC], state_rows=[("K1", h, "")], force=True
)
assert stats["indexed"] == 1 assert stats["indexed"] == 1
def test_empty_doc_counts_skipped(self): def test_empty_doc_counts_skipped(self):
@@ -91,19 +95,22 @@ class TestIndexDocs:
class TestPageEnrichmentHook: class TestPageEnrichmentHook:
def test_enriches_chunks_before_add(self, monkeypatch): def test_enriches_only_docs_that_embed(self, monkeypatch):
"""index_docs runs llm.pages.enrich_pdf_pages on every doc's chunks.""" """index_docs runs llm.pages.enrich_pdf_pages only on docs that reach
the embed path — a hash match must not open the doc's PDFs."""
from llm import index as index_mod from llm import index as index_mod
seen = [] seen = []
monkeypatch.setattr(
def fake_enrich(doc, chunks): index_mod,
seen.append(doc.key) "enrich_pdf_pages",
return chunks lambda doc, chunks: seen.append(doc.key) or chunks,
)
monkeypatch.setattr(index_mod, "enrich_pdf_pages", fake_enrich)
_run([DOC], []) _run([DOC], [])
assert seen == ["K1"] assert seen == ["K1"]
seen.clear()
_run([DOC], [("K1", content_hash(DOC.text), "")])
assert seen == []
class TestDeleteOldChunksFallback: class TestDeleteOldChunksFallback:

View File

@@ -0,0 +1,218 @@
from __future__ import annotations
from unittest.mock import MagicMock, patch
from llm.chunk import Doc, content_hash
from llm.config import LlmConfig
from llm.index import index_refs
from llm.source import DocRef
CFG = LlmConfig(
ollama_hosts=("http://h1:11434",),
embed_model="m",
instruct_model="g",
embed_dim=768,
build_ann_index=False,
pg_host="x",
pg_port=5432,
pg_db="llm",
pg_user="llm",
)
DOC = Doc(key="K1", text="Some body text.", metadata={"docket": "D"})
def _ref(fp="fp1", docket="D", loads=None, doc=DOC):
def _load():
if loads is not None:
loads.append(doc.key)
return doc
return DocRef(
key=doc.key, collection="comments", docket=docket, fingerprint=fp, load=_load
)
def _maybe_chunk_doc(fn):
"""Patch llm.index.chunk_doc only when a test asks for it."""
from contextlib import nullcontext
return nullcontext() if fn is None else patch("llm.index.chunk_doc", side_effect=fn)
def _run(
refs,
state_rows,
*,
force=False,
sealed=None,
mark_complete=True,
complete_rows=(),
chunk_doc=None,
):
store = MagicMock()
engine = MagicMock()
conn = engine.begin.return_value.__enter__.return_value
def fake_execute(clause, *a, **k):
sql = str(clause)
r = MagicMock()
if "FROM index_docket_state" in sql:
r.fetchall.return_value = list(complete_rows)
elif "FROM index_state" in sql:
r.fetchall.return_value = state_rows
else:
r.fetchall.return_value = []
return r
conn.execute.side_effect = fake_execute
with (
patch("llm.index._engine", return_value=engine),
patch("llm.index.vectorstore", return_value=store),
patch("llm.index.embed_texts", return_value=[[0.0] * 3]),
patch("llm.index.ensure_hnsw"),
patch("llm.index.enrich_pdf_pages", side_effect=lambda d, c: c) as enrich,
patch("llm.index.HostPool") as MockPool,
_maybe_chunk_doc(chunk_doc),
):
MockPool.return_value.check.return_value = ["http://h1:11434"]
stats = index_refs(
refs,
collection="comments",
cfg=CFG,
pool=MockPool.return_value,
force=force,
sealed=sealed,
mark_complete=mark_complete,
)
return stats, store, conn, enrich
def test_fingerprint_match_skips_without_load():
loads = []
stats, store, _, enrich = _run([_ref("fp1", loads=loads)], [("K1", "h-old", "fp1")])
assert stats["fingerprint_skipped"] == 1 and stats["indexed"] == 0
assert loads == []
store.add_embeddings.assert_not_called()
enrich.assert_not_called()
def test_hash_match_updates_fingerprint_without_embedding():
h = content_hash(DOC.text)
stats, store, conn, enrich = _run([_ref("fp2")], [("K1", h, "fp1")])
assert stats["hash_skipped"] == 1 and stats["indexed"] == 0
store.add_embeddings.assert_not_called()
enrich.assert_not_called()
upd = [
c
for c in conn.execute.call_args_list
if "UPDATE index_state SET fingerprint" in str(c.args[0])
]
assert len(upd) == 1 and upd[0].args[1]["f"] == "fp2"
def test_changed_doc_embeds_and_records_fingerprint():
stats, store, conn, enrich = _run([_ref("fp2")], [("K1", "stale", "fp1")])
assert stats["indexed"] == 1 and stats["chunks"] == 1
store.add_embeddings.assert_called_once()
enrich.assert_called_once()
ins = [
c
for c in conn.execute.call_args_list
if "INSERT INTO index_state" in str(c.args[0])
]
assert ins[0].args[1]["f"] == "fp2"
def test_empty_fingerprint_never_matches():
stats, store, _, _ = _run([_ref("")], [("K1", "stale", "")])
assert stats["indexed"] == 1
def test_force_ignores_fingerprint_and_hash():
h = content_hash(DOC.text)
stats, store, _, _ = _run([_ref("fp1")], [("K1", h, "fp1")], force=True)
assert stats["indexed"] == 1
def test_load_none_counts_skipped():
ref = DocRef(
key="K9", collection="comments", docket="D", fingerprint="x", load=lambda: None
)
stats, *_ = _run([ref], [])
assert stats["skipped"] == 1 and stats["indexed"] == 0
def test_sealed_docket_marked_complete_after_clean_run():
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={"D": "2026-10-20T00:00:00Z"})
assert stats["docket_complete"] == 1
ins = [
c
for c in conn.execute.call_args_list
if "INSERT INTO index_docket_state" in str(c.args[0])
]
assert ins[0].args[1] == {"c": "comments", "d": "D", "s": "2026-10-20T00:00:00Z"}
def test_not_marked_when_mark_complete_false_or_unsealed():
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={"D": "s"}, mark_complete=False)
assert stats["docket_complete"] == 0
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={})
assert stats["docket_complete"] == 0
def _state_inserts(conn):
return [
c
for c in conn.execute.call_args_list
if "INSERT INTO index_state" in str(c.args[0])
]
def _docket_inserts(conn):
return [
c
for c in conn.execute.call_args_list
if "INSERT INTO index_docket_state" in str(c.args[0])
]
def test_empty_text_ref_is_stamped_final_and_counts_toward_completion():
"""No text is a finished state, not a failure: stamp it so the next
run fingerprint-skips it, and let the docket still seal complete."""
blank = DocRef(
key="K9", collection="comments", docket="D", fingerprint="x", load=lambda: None
)
stats, store, conn, _ = _run([blank], [], sealed={"D": "s"})
assert stats["skipped"] == 1 and stats["indexed"] == 0
store.add_embeddings.assert_not_called()
ins = _state_inserts(conn)
assert len(ins) == 1
assert ins[0].args[1] == {"k": "K9", "c": "comments", "h": "", "n": 0, "f": "x"}
assert stats["docket_complete"] == 1
assert _docket_inserts(conn)
def test_stamped_empty_row_is_fingerprint_skipped_on_the_next_run():
loads = []
blank = DocRef(
key="K9",
collection="comments",
docket="D",
fingerprint="x",
load=lambda: loads.append("K9"),
)
stats, _, conn, _ = _run([blank], [("K9", "", "x")])
assert stats["fingerprint_skipped"] == 1 and stats["skipped"] == 0
assert loads == [] # stamped-empty rows are never re-loaded
assert _state_inserts(conn) == []
def test_sealed_docket_not_marked_when_a_ref_yields_no_chunks():
"""Text that chunks to nothing is a real failure — it blocks the seal
and is not stamped."""
stats, _, conn, _ = _run(
[_ref("fp1")], [], sealed={"D": "s"}, chunk_doc=lambda d: []
)
assert stats["skipped"] == 1 and stats["docket_complete"] == 0
assert _state_inserts(conn) == []
assert _docket_inserts(conn) == []

View File

@@ -24,3 +24,20 @@ class TestDdl:
executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list) executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list)
assert "vector(768)" in executed assert "vector(768)" in executed
assert "USING hnsw" in executed assert "USING hnsw" in executed
def test_fingerprint_column_and_docket_state(self):
assert "fingerprint" in migrate.INDEX_STATE_DDL
assert "ADD COLUMN IF NOT EXISTS fingerprint" in migrate.INDEX_STATE_ALTER
assert (
"CREATE TABLE IF NOT EXISTS index_docket_state"
in migrate.INDEX_DOCKET_STATE_DDL
)
assert "PRIMARY KEY (collection, docket)" in migrate.INDEX_DOCKET_STATE_DDL
def test_migrate_runs_alter_and_docket_state(self):
engine = MagicMock()
conn = engine.begin.return_value.__enter__.return_value
migrate.migrate(engine)
executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list)
assert "ADD COLUMN IF NOT EXISTS fingerprint" in executed
assert "index_docket_state" in executed

View File

@@ -0,0 +1,276 @@
"""llm.source — lazy DocRefs: fingerprints without loads, docket skips,
lazy Zotero snapshot."""
from __future__ import annotations
import os
from pathlib import Path
import pytest
from bib.item import Item, Rule
from bib.store import Store
from llm.source import (
DocRef,
ZoteroPdfIndex,
iter_comment_refs,
iter_corpus_refs,
iter_rule_refs,
)
DOCKET = "CMS-2019-0111"
CID = f"{DOCKET}-0042"
COMBINED = f"---\ncomment_id: {CID}\ndocket_id: {DOCKET}\n---\n\nWe object.\n"
@pytest.fixture
def store(tmp_path):
s = Store(":memory:", storage_dir=tmp_path / "storage")
key = s.create(
Item(
item_type="report",
title="A comment",
url=f"https://www.regulations.gov/comment/{CID}",
abstract="Inline.",
date_published="2019-09-27",
)
)
for tag in ("doctype:comment", "year:2019", f"reg-docket:{DOCKET}"):
s.add_tag(key, tag)
s._comment_key = key
return s
@pytest.fixture
def root(tmp_path):
d = tmp_path / DOCKET / CID
d.mkdir(parents=True)
(d / "combined.md").write_text(COMBINED)
return tmp_path
class TestCommentRefs:
def test_ref_has_fingerprint_and_lazy_load(self, store, root, monkeypatch):
reads = []
real = Path.read_text
monkeypatch.setattr(
Path,
"read_text",
lambda self, *a, **k: reads.append(self) or real(self, *a, **k),
)
refs = list(iter_comment_refs(store, docket=DOCKET, root=root))
assert len(refs) == 1
r = refs[0]
assert isinstance(r, DocRef)
assert (r.key, r.collection, r.docket) == (
store._comment_key,
"comments",
DOCKET,
)
assert len(r.fingerprint) > 64 and "|" in r.fingerprint
assert reads == [] # nothing read yet
doc = r.load()
assert "We object" in doc.text
assert doc.metadata["year"] == "2019" and doc.metadata["docket"] == DOCKET
assert reads # load() read combined.md
def test_fingerprint_changes_with_new_attachment(self, store, root):
f1 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
(root / DOCKET / CID / "attachment_1.pdf").write_bytes(b"%PDF")
f2 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
assert f1 != f2
def test_fingerprint_changes_with_updated_at(self, store, root):
f1 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
store._con().execute(
"UPDATE items SET updated_at='2030-01-01T00:00:00Z' WHERE key=?",
(store._comment_key,),
)
f2 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
assert f1 != f2
def test_skip_dockets_excludes_rows(self, store, root):
assert list(iter_comment_refs(store, root=root, skip_dockets={DOCKET})) == []
def test_unextracted_load_falls_back_to_abstract(self, store, tmp_path):
r = next(iter_comment_refs(store, docket=DOCKET, root=tmp_path))
assert r.load().text == "Inline."
def test_empty_comment_load_returns_none(self, store, tmp_path):
store._con().execute(
"UPDATE items SET abstract='' WHERE key=?", (store._comment_key,)
)
r = next(iter_comment_refs(store, docket=DOCKET, root=tmp_path))
assert r.load() is None
def test_single_query_for_year(self, store, root, monkeypatch):
"""No per-comment year lookups: exactly one SELECT on items for the listing."""
calls: list[str] = []
store._con().set_trace_callback(calls.append)
list(iter_comment_refs(store, docket=DOCKET, root=root))
store._con().set_trace_callback(None)
selects = [c for c in calls if c.lstrip().upper().startswith("SELECT")]
assert len(selects) == 1
def _anchored_rule(store) -> str:
"""A rule with one grabbed FR anchor paragraph (sha256 "abc")."""
key = store.create(Rule(title="R", url="https://fr/1", document_number="2019-1"))
store._con().execute(
"INSERT INTO fr_anchor_docs (item_key, document_number, html_url, start_page, end_page, fr_volume, sha256) VALUES (?,?,?,?,?,?,?)",
(key, "2019-1", "https://fr/1", 1, 2, 84, "abc"),
)
store._con().execute(
"INSERT INTO fr_anchors (item_key, p_id, page, ordinal, text) VALUES (?,?,?,?,?)",
(key, 1, 1, 1, "Para one."),
)
return key
class TestRuleRefs:
def test_fingerprint_from_anchor_sha(self, store):
key = _anchored_rule(store)
refs = list(iter_rule_refs(store))
assert [r.key for r in refs] == [key]
assert refs[0].fingerprint.startswith("anchors:abc|")
assert refs[0].collection == "rules" and refs[0].docket is None
assert refs[0].load().text == "Para one."
def test_anchor_fingerprint_changes_with_updated_at(self, store):
"""The sha covers the FR body, not the item's own metadata."""
key = _anchored_rule(store)
f1 = next(iter_rule_refs(store)).fingerprint
store._con().execute(
"UPDATE items SET updated_at='2030-01-01T00:00:00Z' WHERE key=?", (key,)
)
assert next(iter_rule_refs(store)).fingerprint != f1
def test_tag_filter_selects_only_tagged_rules(self, store):
tagged = store.create(Rule(title="Tagged", document_number="2019-2"))
store.add_tag(tagged, "project:pfs")
store.create(Rule(title="Untagged", document_number="2019-3"))
assert [r.key for r in iter_rule_refs(store, tag="project:pfs")] == [tagged]
class TestCorpusRefs:
def test_excludes_comments_and_is_lazy(self, store):
key = store.create(
Item(
item_type="report", title="Report", url="https://x/r", abstract="Body."
)
)
refs = list(iter_corpus_refs(store))
assert [r.key for r in refs] == [key]
assert refs[0].collection == "corpus"
assert refs[0].load().metadata["kind"] == "corpus"
def test_zotero_not_consulted_until_load(self, store, tmp_path):
store.create(
Item(
item_type="report", title="Report", url="https://x/r", abstract="Body."
)
)
calls = []
class Z(ZoteroPdfIndex):
def pdfs_for(self, key):
calls.append(key)
return []
refs = list(iter_corpus_refs(store, zotero=Z({})))
assert calls == []
refs[0].load()
assert calls
def test_tag_filter_selects_only_tagged_items(self, store):
tagged = store.create(Item(item_type="report", title="T", abstract="Body."))
store.add_tag(tagged, "project:pfs")
store.create(Item(item_type="report", title="U", abstract="Body."))
assert [r.key for r in iter_corpus_refs(store, tag="project:pfs")] == [tagged]
class TestLazyZotero:
def test_no_copy_until_first_lookup(self, tmp_path):
src = tmp_path / "zotero.sqlite"
import sqlite3
con = sqlite3.connect(src)
con.executescript(
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
)
con.close()
snap_dir = tmp_path / "snap"
z = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
assert not (snap_dir / "zotero.sqlite").exists()
assert z.pdfs_for("ABCD1234") == []
assert (snap_dir / "zotero.sqlite").exists()
m1 = (snap_dir / "zotero.sqlite").stat().st_mtime_ns
z2 = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
z2.pdfs_for("ABCD1234")
assert (
snap_dir / "zotero.sqlite"
).stat().st_mtime_ns == m1 # source unchanged → no recopy
bump = src.stat().st_mtime_ns + 10**9
os.utime(src, ns=(bump, bump)) # source now newer than the snapshot
z3 = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
z3.pdfs_for("ABCD1234")
assert (snap_dir / "zotero.sqlite").stat().st_mtime_ns != m1
def test_newer_wal_forces_a_recopy(self, tmp_path):
"""Zotero commits into zotero.sqlite-wal and only touches the main
file at a checkpoint — the WAL's mtime has to count too."""
src = tmp_path / "zotero.sqlite"
import sqlite3
con = sqlite3.connect(src)
con.executescript(
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
)
con.close()
snap_dir = tmp_path / "snap"
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
snap = snap_dir / "zotero.sqlite"
m1 = snap.stat().st_mtime_ns
wal = tmp_path / "zotero.sqlite-wal"
wal.write_bytes(b"wal")
bump = m1 + 10**9
os.utime(wal, ns=(bump, bump)) # main file untouched, WAL is newer
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
m2 = snap.stat().st_mtime_ns
assert m2 != m1 # re-copied
# and the snapshot now records what it captured, so the next run
# doesn't re-copy 1.95 GB for the same unchanged WAL
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
assert snap.stat().st_mtime_ns == m2
def test_failed_copy_does_not_stamp_the_stale_snapshot(self, tmp_path, monkeypatch):
"""Stamping after a failed copy would pass a stale snapshot off as
current forever."""
import shutil
import sqlite3
src = tmp_path / "zotero.sqlite"
con = sqlite3.connect(src)
con.executescript(
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
)
con.close()
snap_dir = tmp_path / "snap"
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
snap = snap_dir / "zotero.sqlite"
m1 = snap.stat().st_mtime_ns
bump = m1 + 10**9
os.utime(src, ns=(bump, bump)) # source now newer → a copy is due
monkeypatch.setattr(
shutil, "copy2", lambda *a, **k: (_ for _ in ()).throw(OSError("no space"))
)
assert (
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for(
"ABCD1234"
)
== []
)
assert snap.stat().st_mtime_ns == m1 # untouched, so the next run retries

View File

@@ -0,0 +1,126 @@
from __future__ import annotations
import os
from pathlib import Path
from rex.comments.combine import is_current, source_paths
from rex.comments.walker import walk_and_extract
def _dir(tmp_path: Path, docket="CMS-2024-0001", cid="CMS-2024-0001-0001") -> Path:
d = tmp_path / docket / cid
d.mkdir(parents=True)
return d
def test_source_paths_excludes_outputs(tmp_path: Path):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"x")
(d / "attachment_1.pdf.md").write_text("sibling")
(d / "combined.md").write_text("c")
(d / "combined.md.tmp").write_text("t")
(d / ".hidden").write_text("h")
assert [p.name for p in source_paths(d)] == ["attachment_1.pdf"]
def test_is_current_false_without_combined(tmp_path: Path):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"x")
assert is_current(d) is False
def test_is_current_true_when_combined_newer(tmp_path: Path):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"x")
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
(d / "combined.md").write_text("c")
assert is_current(d) is True
def test_is_current_false_when_source_newer(tmp_path: Path):
d = _dir(tmp_path)
(d / "combined.md").write_text("c")
os.utime(d / "combined.md", ns=(1_000, 1_000))
(d / "attachment_2.pdf").write_bytes(b"new")
assert is_current(d) is False
def test_walker_reextracts_stale_dir(tmp_path: Path, monkeypatch):
d = _dir(tmp_path)
(d / "combined.md").write_text("---\ncomment_id: x\n---\n\nold\n")
os.utime(d / "combined.md", ns=(1_000, 1_000))
(d / "attachment_1.pdf").write_bytes(b"%PDF") # newer than combined.md
called = []
monkeypatch.setattr(
"rex.comments.walker.extract_comment",
lambda cdir, **kw: called.append(cdir) or (cdir / "combined.md"),
)
stats = walk_and_extract(tmp_path, inline_body_lookup=lambda _c: "", workers=1)
assert stats["written"] == 1 and called == [d]
def test_walker_skips_current_dir_without_reading(tmp_path: Path, monkeypatch):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"%PDF")
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
(d / "combined.md").write_text("---\ncomment_id: x\n---\n\nbody\n")
reads = []
real_read_text = Path.read_text
def spy(self, *a, **kw):
if self.name == "combined.md":
reads.append(self)
return real_read_text(self, *a, **kw)
monkeypatch.setattr(Path, "read_text", spy)
seen = []
stats = walk_and_extract(
tmp_path,
inline_body_lookup=lambda _c: "",
workers=1,
on_extracted=lambda cid, _d: seen.append(cid),
)
assert stats == {"written": 0, "skipped": 1, "failed": 0}
assert reads == [] # skipped dirs are not read
assert seen == [] # and the callback does not fire without --reattach
def test_walker_reattach_fires_callback_for_skipped(tmp_path: Path):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"%PDF")
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
(d / "combined.md").write_text(
"---\ncomment_id: x\ndocket_id: y\n---\n\n## attachment_1.pdf\n\nbody\n"
)
seen = []
walk_and_extract(
tmp_path,
inline_body_lookup=lambda _c: "",
workers=1,
reattach=True,
on_extracted=lambda cid, _d: seen.append(cid),
)
assert seen == [d.name]
assert (d / "attachment_1.pdf.md").is_file() # siblings derived on reattach
def test_walker_skip_dockets(tmp_path: Path):
d = _dir(tmp_path)
(d / "attachment_1.pdf").write_bytes(b"%PDF")
stats = walk_and_extract(
tmp_path,
inline_body_lookup=lambda _c: "",
workers=1,
skip_dockets={"CMS-2024-0001"},
)
assert stats == {"written": 0, "skipped": 0, "failed": 0, "skipped_sealed": 1}
assert not (d / "combined.md").exists()
def test_is_current_ignores_a_source_that_vanished(tmp_path: Path, monkeypatch):
"""A file removed between iterdir() and stat() must not kill the run."""
d = _dir(tmp_path)
(d / "combined.md").write_text("c")
ghost = d / "attachment_9.pdf"
monkeypatch.setattr("rex.comments.combine.source_paths", lambda _d: [ghost])
assert is_current(d) is True

View File

@@ -2,6 +2,7 @@
from __future__ import annotations from __future__ import annotations
import os
from pathlib import Path from pathlib import Path
import fitz import fitz
@@ -44,6 +45,7 @@ def test_walk_and_extract_writes_all_combined(tmp_path: Path):
def test_walk_and_extract_skips_existing(tmp_path: Path): def test_walk_and_extract_skips_existing(tmp_path: Path):
a, _b = _setup_two_comments(tmp_path) a, _b = _setup_two_comments(tmp_path)
(a / "combined.md").write_text("---\ncomment_id: x\n---\n\npre-existing\n") (a / "combined.md").write_text("---\ncomment_id: x\n---\n\npre-existing\n")
os.utime(a / "attachment_1.pdf", ns=(1_000, 1_000))
stats = walk_and_extract(tmp_path, inline_body_lookup=_stub_inline_body, workers=1) stats = walk_and_extract(tmp_path, inline_body_lookup=_stub_inline_body, workers=1)
@@ -96,10 +98,10 @@ def test_on_extracted_called_for_written_dirs(tmp_path: Path):
def test_on_extracted_called_for_skipped_dirs(tmp_path: Path): def test_on_extracted_called_for_skipped_dirs(tmp_path: Path):
"""Pre-existing combined.md still gets the callback, and any missing """With reattach=True, pre-existing combined.md still gets the callback,
sibling MDs are derived from it before the callback fires — so and any missing sibling MDs are derived from it before the callback
previously-extracted dirs end up with the same set of MDs as fires — so previously-extracted dirs end up with the same set of MDs
freshly-written ones.""" as freshly-written ones."""
a, b = _setup_two_comments(tmp_path) a, b = _setup_two_comments(tmp_path)
# Pre-write a stale combined.md with one section so derive_siblings # Pre-write a stale combined.md with one section so derive_siblings
# has something to split. # has something to split.
@@ -112,6 +114,7 @@ def test_on_extracted_called_for_skipped_dirs(tmp_path: Path):
tmp_path, tmp_path,
inline_body_lookup=_stub_inline_body, inline_body_lookup=_stub_inline_body,
workers=1, workers=1,
reattach=True,
on_extracted=lambda cid, _d: seen.append(cid), on_extracted=lambda cid, _d: seen.append(cid),
) )

View File

@@ -13,9 +13,11 @@ from unittest.mock import MagicMock, patch
import pytest import pytest
# ═══════════════════════════════════════════════════════════════════ # ═══════════════════════════════════════════════════════════════════
# Priority 1: cli/bib.py (14 lines) # Priority 1: cli/bib.py
# Lines: 113, 175, 184, 185, 263, 265, 268, 276, 277, 281, 282, # Lines: 113, 175, 184, 185, 325
# 285, 286, 325 # (the fetch-pfs-comments block moved to tests/bib/test_walk_docket.py
# and tests/cli/test_bib_fetch_sealed.py when that loop moved into
# walk_docket — MagicMock-store coverage of deleted lines proved nothing)
# ═══════════════════════════════════════════════════════════════════ # ═══════════════════════════════════════════════════════════════════
@@ -53,244 +55,6 @@ class TestCliBibDiscoverPfsRulesTranslateError:
assert "skipped" in result.output # line 113 hit via exception branch assert "skipped" in result.output # line 113 hit via exception branch
class TestCliBibFetchDocketComments:
"""Lines 175, 184, 185: early return on limit, progress echo, commit."""
def test_fetch_docket_comments_with_limit(self):
from typer.testing import CliRunner
from cli.bib import app
runner = CliRunner()
fake_comment = MagicMock()
fake_comment.id = "C-001"
fake_comment.attachment_count = 0
fake_fr_doc = {
"id": "FR-001",
"attributes": {
"objectId": "obj1",
"commentEndDate": "2024-12-31",
},
}
mock_api = MagicMock()
mock_api.__enter__ = lambda s: s
mock_api.__exit__ = lambda s, *a: None
mock_api.find_documents_in_docket.return_value = [fake_fr_doc]
# Return 2 comments to hit limit=1 → line 175 (return)
mock_api.iter_comments.return_value = iter([fake_comment, fake_comment])
mock_store = MagicMock()
with (
patch("bib.connect", return_value=mock_store),
patch("bib.regulations_gov.Client", return_value=mock_api),
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
):
result = runner.invoke(
app,
["fetch-docket-comments", "CMS-1234-P", "--limit", "1"],
)
assert result.exit_code == 0 # line 175 hit
def test_fetch_docket_comments_progress_and_commit(self):
"""Lines 184, 185: progress echo at 50-comment intervals and commit."""
from typer.testing import CliRunner
from cli.bib import app
runner = CliRunner()
fake_comment = MagicMock()
fake_comment.id = "C-001"
fake_comment.attachment_count = 0
fake_fr_doc = {
"id": "FR-001",
"attributes": {
"objectId": "obj1",
"commentEndDate": "2024-12-31",
},
}
mock_api = MagicMock()
mock_api.__enter__ = lambda s: s
mock_api.__exit__ = lambda s, *a: None
mock_api.find_documents_in_docket.return_value = [fake_fr_doc]
# Return exactly 50 comments to trigger progress print
mock_api.iter_comments.return_value = iter([fake_comment] * 50)
mock_store = MagicMock()
mock_con = MagicMock()
mock_store._con.return_value = mock_con
with (
patch("bib.connect", return_value=mock_store),
patch("bib.regulations_gov.Client", return_value=mock_api),
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
):
result = runner.invoke(
app,
["fetch-docket-comments", "CMS-1234-P"],
)
assert result.exit_code == 0
assert "processed 50" in result.output # line 184
mock_con.commit.assert_called() # line 185
class TestCliBibFetchPfsComments:
"""Lines 263, 265, 268, 276, 277, 281, 282, 285, 286."""
def test_fetch_pfs_comments_full_flow(self):
from typer.testing import CliRunner
from cli.bib import app
runner = CliRunner()
fake_doc = MagicMock()
fake_doc.type = "Proposed Rule"
fake_doc.publication_date = "2024-01-01"
fake_doc.document_number = "2024-00001"
fake_doc.dockets = ["CMS-1234-P"]
fake_doc.html_url = "https://example.com/doc"
fake_rule = MagicMock()
fake_comment = MagicMock()
fake_comment.id = "C-001"
fake_comment.attachment_count = 1
fake_att = MagicMock()
fake_att.url = "https://example.com/att.pdf"
fake_att.filename = "att.pdf"
# fr_doc with no objectId → line 263 (continue)
fr_doc_no_obj = {
"id": "FR-A",
"attributes": {"commentEndDate": "2024-12-31"},
}
# fr_doc with no commentEndDate → line 265 (continue)
fr_doc_no_end = {
"id": "FR-B",
"attributes": {"objectId": "obj2"},
}
# fr_doc that works normally
fr_doc_ok = {
"id": "FR-C",
"attributes": {
"objectId": "obj3",
"commentEndDate": "2024-12-31",
},
}
mock_api = MagicMock()
mock_api.__enter__ = lambda s: s
mock_api.__exit__ = lambda s, *a: None
mock_api.resolve_docket.return_value = "REG-DOCKET-1"
mock_api.find_documents_in_docket.return_value = [
fr_doc_no_obj,
fr_doc_no_end,
fr_doc_ok,
]
# Return 50 comments to hit lines 285-286 (progress + commit)
mock_api.iter_comments.return_value = iter([fake_comment] * 50)
mock_api.attachments_for.return_value = [fake_att]
mock_api.download_attachment.return_value = Path("/fake/att.pdf")
mock_store = MagicMock()
mock_con = MagicMock()
mock_store._con.return_value = mock_con
with (
patch("bib.connect", return_value=mock_store),
patch("bib.federalregister.pfs_rules", return_value=[fake_doc]),
patch("bib.federalregister.split_docket_ids", return_value=["CMS-1234-P"]),
patch("bib.regulations_gov.Client", return_value=mock_api),
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
patch("bib.translate.federal_register", return_value=fake_rule),
):
result = runner.invoke(
app,
[
"fetch-pfs-comments",
"--attachments",
"--per-docket-limit",
"50",
],
)
assert result.exit_code == 0
# line 263: fr_doc_no_obj skipped
# line 265: fr_doc_no_end skipped
# line 268: per_docket_limit break
# lines 276-277: attachment download
# lines 281-282: attach_file called
# lines 285-286: progress + commit at 50
def test_fetch_pfs_comments_attachment_path_none(self):
"""Line 281: download_attachment returns None → skip attach_file."""
from typer.testing import CliRunner
from cli.bib import app
runner = CliRunner()
fake_doc = MagicMock()
fake_doc.type = "Proposed Rule"
fake_doc.publication_date = "2024-01-01"
fake_doc.document_number = "2024-00001"
fake_doc.dockets = ["CMS-5678-P"]
fake_doc.html_url = "https://example.com/doc"
fake_rule = MagicMock()
fake_comment = MagicMock()
fake_comment.id = "C-002"
fake_comment.attachment_count = 1
fake_att = MagicMock()
fake_att.url = "https://example.com/att.pdf"
fake_att.filename = "att.pdf"
fr_doc_ok = {
"id": "FR-D",
"attributes": {
"objectId": "obj4",
"commentEndDate": "2024-12-31",
},
}
mock_api = MagicMock()
mock_api.__enter__ = lambda s: s
mock_api.__exit__ = lambda s, *a: None
mock_api.resolve_docket.return_value = "REG-DOCKET-2"
mock_api.find_documents_in_docket.return_value = [fr_doc_ok]
mock_api.iter_comments.return_value = iter([fake_comment])
mock_api.attachments_for.return_value = [fake_att]
mock_api.download_attachment.return_value = None # failed download
mock_store = MagicMock()
mock_con = MagicMock()
mock_store._con.return_value = mock_con
with (
patch("bib.connect", return_value=mock_store),
patch("bib.federalregister.pfs_rules", return_value=[fake_doc]),
patch("bib.federalregister.split_docket_ids", return_value=["CMS-5678-P"]),
patch("bib.regulations_gov.Client", return_value=mock_api),
patch("bib.regulations_gov.upsert_comment", return_value="key2"),
patch("bib.translate.federal_register", return_value=fake_rule),
):
result = runner.invoke(
app,
["fetch-pfs-comments", "--attachments", "--per-docket-limit", "1"],
)
assert result.exit_code == 0
mock_store.attach_file.assert_not_called() # line 281 path=None
class TestCliBibIngestMailMissingPassword: class TestCliBibIngestMailMissingPassword:
"""Line 325: no cached password for user.""" """Line 325: no cached password for user."""