merge: P47 comment pipeline docket seals + fingerprint-first skips — never redo known-complete work (refs #615 #680)
This commit is contained in:
183
dev/scripts/dedupe_attachments.py
Executable file
183
dev/scripts/dedupe_attachments.py
Executable file
@@ -0,0 +1,183 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""One-time cleanup of duplicate bib attachments (refs #615).
|
||||||
|
|
||||||
|
``Store.attach_file`` used to mint a new row + storage copy on every
|
||||||
|
call, so re-farms produced thousands of ``(item_id, filename)``
|
||||||
|
duplicates (33,379 groups / 66,917 of 78,017 rows on 2026-09-08). This
|
||||||
|
keeps the oldest row of each group, deletes the others' rows and their
|
||||||
|
storage copies, and prints a report. Dry-run by default.
|
||||||
|
|
||||||
|
Zotero is not touched: ``bib.sync`` already dedupes child attachments by
|
||||||
|
filename, so a duplicate that was synced once is a single Zotero child
|
||||||
|
attachment and stays valid.
|
||||||
|
|
||||||
|
Usage::
|
||||||
|
|
||||||
|
uv run python dev/scripts/dedupe_attachments.py # report only
|
||||||
|
uv run python dev/scripts/dedupe_attachments.py --apply # delete
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import os
|
||||||
|
import sqlite3
|
||||||
|
import sys
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Group:
|
||||||
|
item_id: int
|
||||||
|
filename: str
|
||||||
|
keep: str
|
||||||
|
remove: list[str] = field(default_factory=list)
|
||||||
|
remove_paths: list[str] = field(default_factory=list)
|
||||||
|
conflict: bool = False # sizes differ → do not touch
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Report:
|
||||||
|
groups: int = 0
|
||||||
|
rows_removed: int = 0
|
||||||
|
files_removed: int = 0
|
||||||
|
bytes_freed: int = 0
|
||||||
|
skipped_conflicts: int = 0
|
||||||
|
removed_keys: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
def plan(con: sqlite3.Connection) -> list[Group]:
|
||||||
|
con.row_factory = sqlite3.Row
|
||||||
|
rows = con.execute(
|
||||||
|
"""
|
||||||
|
SELECT a.item_id, a.filename, a.key, a.storage_path, a.rowid AS rid
|
||||||
|
FROM attachments a
|
||||||
|
WHERE (a.item_id, a.filename) IN (
|
||||||
|
SELECT item_id, filename FROM attachments
|
||||||
|
GROUP BY item_id, filename HAVING count(*) > 1
|
||||||
|
)
|
||||||
|
ORDER BY a.item_id, a.filename, a.rowid
|
||||||
|
"""
|
||||||
|
).fetchall()
|
||||||
|
groups: dict[tuple[int, str], Group] = {}
|
||||||
|
for r in rows:
|
||||||
|
gkey = (r["item_id"], r["filename"])
|
||||||
|
g = groups.get(gkey)
|
||||||
|
if g is None:
|
||||||
|
groups[gkey] = Group(
|
||||||
|
item_id=r["item_id"], filename=r["filename"], keep=r["key"]
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
g.remove.append(r["key"])
|
||||||
|
g.remove_paths.append(r["storage_path"])
|
||||||
|
# conflict check: every copy must have the same size as the kept one
|
||||||
|
for g in groups.values():
|
||||||
|
keep_path = con.execute(
|
||||||
|
"SELECT storage_path FROM attachments WHERE key = ?", (g.keep,)
|
||||||
|
).fetchone()["storage_path"]
|
||||||
|
keep_size = _size(keep_path)
|
||||||
|
for p in g.remove_paths:
|
||||||
|
if _size(p) not in (keep_size, -1):
|
||||||
|
g.conflict = True
|
||||||
|
break
|
||||||
|
return list(groups.values())
|
||||||
|
|
||||||
|
|
||||||
|
def _size(path: str) -> int:
|
||||||
|
try:
|
||||||
|
return os.stat(path).st_size
|
||||||
|
except OSError:
|
||||||
|
return -1
|
||||||
|
|
||||||
|
|
||||||
|
def apply(con: sqlite3.Connection, groups: list[Group]) -> Report:
|
||||||
|
"""Delete the duplicate rows, then their storage copies.
|
||||||
|
|
||||||
|
Rows first, files second, with the COMMIT in between: a failure
|
||||||
|
mid-transaction rolls the rows back, and rows that still exist must
|
||||||
|
still have their files. Unlinking inside the transaction would leave
|
||||||
|
surviving rows pointing at nothing — the one outcome this cleanup
|
||||||
|
must never produce.
|
||||||
|
"""
|
||||||
|
rep = Report(groups=len(groups))
|
||||||
|
doomed: list[Path] = []
|
||||||
|
con.execute("BEGIN")
|
||||||
|
try:
|
||||||
|
for g in groups:
|
||||||
|
if g.conflict:
|
||||||
|
rep.skipped_conflicts += 1
|
||||||
|
continue
|
||||||
|
for key, path in zip(g.remove, g.remove_paths):
|
||||||
|
still_referenced = con.execute(
|
||||||
|
"SELECT count(*) FROM attachments WHERE storage_path = ? AND key <> ?",
|
||||||
|
(path, key),
|
||||||
|
).fetchone()[0]
|
||||||
|
con.execute("DELETE FROM attachments WHERE key = ?", (key,))
|
||||||
|
rep.rows_removed += 1
|
||||||
|
rep.removed_keys.append(key)
|
||||||
|
if not still_referenced:
|
||||||
|
doomed.append(Path(path))
|
||||||
|
con.execute("COMMIT")
|
||||||
|
except Exception:
|
||||||
|
con.execute("ROLLBACK")
|
||||||
|
raise
|
||||||
|
for p in doomed:
|
||||||
|
if p.is_file():
|
||||||
|
rep.bytes_freed += p.stat().st_size
|
||||||
|
p.unlink()
|
||||||
|
rep.files_removed += 1
|
||||||
|
try:
|
||||||
|
p.parent.rmdir() # the per-key dir, if now empty
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return rep
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
ap = argparse.ArgumentParser(
|
||||||
|
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
||||||
|
)
|
||||||
|
ap.add_argument(
|
||||||
|
"--db", default="", help="bib.sqlite path (default: stack.toml db.bib)"
|
||||||
|
)
|
||||||
|
ap.add_argument(
|
||||||
|
"--apply", action="store_true", help="delete duplicates (default: report only)"
|
||||||
|
)
|
||||||
|
args = ap.parse_args(argv)
|
||||||
|
if args.db:
|
||||||
|
db = args.db
|
||||||
|
else:
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "src"))
|
||||||
|
from conf import path
|
||||||
|
|
||||||
|
db = str(path("db.bib"))
|
||||||
|
con = sqlite3.connect(db, isolation_level=None)
|
||||||
|
try:
|
||||||
|
groups = plan(con)
|
||||||
|
conflicts = sum(1 for g in groups if g.conflict)
|
||||||
|
rows = sum(len(g.remove) for g in groups)
|
||||||
|
print(
|
||||||
|
f"{db}: {len(groups)} duplicate groups, {rows} rows to remove, "
|
||||||
|
f"{conflicts} conflicts (size mismatch, skipped)"
|
||||||
|
)
|
||||||
|
if not args.apply:
|
||||||
|
print("dry run — pass --apply to delete")
|
||||||
|
return 0
|
||||||
|
rep = apply(con, groups)
|
||||||
|
finally:
|
||||||
|
con.close()
|
||||||
|
print(
|
||||||
|
f"removed rows={rep.rows_removed} files={rep.files_removed} "
|
||||||
|
f"freed={rep.bytes_freed / 1e6:.1f} MB skipped_conflicts={rep.skipped_conflicts}"
|
||||||
|
)
|
||||||
|
if rep.removed_keys:
|
||||||
|
head = ", ".join(rep.removed_keys[:20])
|
||||||
|
more = " …" if len(rep.removed_keys) > 20 else ""
|
||||||
|
print(f"removed keys ({len(rep.removed_keys)}): {head}{more}")
|
||||||
|
print("reopen the store once (any `stack bib` command) to create the unique index")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -22,24 +22,15 @@ step() { # step <label> <cmd...>: run, count failures, keep going
|
|||||||
"$@" || { echo "WARN: $label rc=$?"; FAILURES=$((FAILURES+1)); }
|
"$@" || { echo "WARN: $label rc=$?"; FAILURES=$((FAILURES+1)); }
|
||||||
}
|
}
|
||||||
|
|
||||||
# Fast path first: the Mirrulations S3 mirror (#662) is a superset of the
|
# Every step is incremental: sealed dockets are skipped outright, open
|
||||||
# live docket and has no rate cap (~1,500 comments/min vs the API's ~970/hr).
|
# dockets pull from the stored watermark, extract/index compare cheap
|
||||||
# It creates comments the farm never listed, enriches, attaches and tags
|
# fingerprints before doing any work. Mirror first (bulk, no cap), then
|
||||||
# them identically (reg-docket:/rule:/enriched:ok), so the index is usable
|
# the API for anything the mirror lags on.
|
||||||
# minutes in rather than the next day.
|
|
||||||
step "mirror backfill" uv run stack bib backfill-comments --docket CMS-2026-2377 --mirror
|
step "mirror backfill" uv run stack bib backfill-comments --docket CMS-2026-2377 --mirror
|
||||||
step "extract" uv run stack comments extract --docket CMS-2026-2377
|
|
||||||
# extract-ocr is a phase-2 stub (always exits 2, no --docket) — see #253.
|
|
||||||
# ocr_needed attachments stay flagged in combined.md frontmatter for now.
|
|
||||||
step "index" uv run stack llm index --collection comments --docket CMS-2026-2377
|
|
||||||
|
|
||||||
# Gap-fill from the reg.gov API for anything the mirror lags on. All
|
|
||||||
# resumable: the list walk upserts, the backfill skips enriched:ok, and
|
|
||||||
# extract/index only touch new or changed items.
|
|
||||||
step "api fetch" uv run stack bib fetch-pfs-comments --docket CMS-2026-2377
|
step "api fetch" uv run stack bib fetch-pfs-comments --docket CMS-2026-2377
|
||||||
step "api backfill" uv run stack bib backfill-comments --docket CMS-2026-2377
|
step "api backfill" uv run stack bib backfill-comments --docket CMS-2026-2377
|
||||||
step "extract (gap)" uv run stack comments extract --docket CMS-2026-2377
|
step "extract" uv run stack comments extract --docket CMS-2026-2377
|
||||||
step "index (gap)" uv run stack llm index --collection comments --docket CMS-2026-2377
|
step "index" uv run stack llm index --collection comments --docket CMS-2026-2377
|
||||||
|
|
||||||
COUNTS=$(uv run python - <<'EOF'
|
COUNTS=$(uv run python - <<'EOF'
|
||||||
from conf import connect
|
from conf import connect
|
||||||
|
|||||||
@@ -668,6 +668,82 @@ git commit -m "feat(bib): Store.upsert_status — identical re-upsert writes not
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
### Task 3b: Data-preserving upsert — never blank a stored value
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `src/bib/store.py` (`upsert_status`)
|
||||||
|
- Test: `tests/bib/test_store_upsert_noop.py` (extend)
|
||||||
|
|
||||||
|
**Why (incident 2026-09-08):** the fetch list walk builds a `Source` whose `abstract` is empty (the list endpoint has no body) and re-upserts every comment it sees. With the old upsert that overwrote 17,607 enriched bodies in CMS-2026-2377 and 1,115 in CMS-2017-0092. Task 3's comparison alone would still call that "updated" and write the empty abstract. Upsert must merge, not replace: an empty incoming value carries no information.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: `Store.upsert_status` from Task 3.
|
||||||
|
- Produces: same signature; for every column in `_COMPARE_COLS` (except `extra_json`) an empty incoming string keeps the stored value. `extra_json` is compared/written as today (it is always a full serialization).
|
||||||
|
|
||||||
|
- [ ] **Step 1: Write the failing tests** (append to `tests/bib/test_store_upsert_noop.py`)
|
||||||
|
|
||||||
|
```python
|
||||||
|
def test_empty_incoming_abstract_keeps_stored_body():
|
||||||
|
"""A list-walk row (no body) must not blank an enriched comment."""
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(abstract="enriched body"))
|
||||||
|
_, status = s.upsert_status(_item(abstract=""))
|
||||||
|
assert status == "unchanged"
|
||||||
|
assert s.get(key).abstract == "enriched body"
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_incoming_title_keeps_stored_title():
|
||||||
|
s = _store()
|
||||||
|
it = _item(); it.title = "Org: CMS-2026-2377-1"
|
||||||
|
key, _ = s.upsert_status(it)
|
||||||
|
it2 = _item(); it2.title = ""
|
||||||
|
_, status = s.upsert_status(it2)
|
||||||
|
assert status == "unchanged"
|
||||||
|
assert s.get(key).title == "Org: CMS-2026-2377-1"
|
||||||
|
|
||||||
|
|
||||||
|
def test_non_empty_incoming_still_updates():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(abstract="v1"))
|
||||||
|
_, status = s.upsert_status(_item(abstract="v2"))
|
||||||
|
assert status == "updated" and s.get(key).abstract == "v2"
|
||||||
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2: Run to verify failure**
|
||||||
|
|
||||||
|
Run: `uv run --no-sync pytest tests/bib/test_store_upsert_noop.py -q -p no:cacheprovider`
|
||||||
|
Expected: the first two new tests FAIL (`status == "updated"`, abstract/title blanked).
|
||||||
|
|
||||||
|
- [ ] **Step 3: Implement**
|
||||||
|
|
||||||
|
In `upsert_status`, after `cur_row = current.to_row()` and before the `same_cols` check:
|
||||||
|
|
||||||
|
```python
|
||||||
|
# Merge, don't replace: an empty incoming value carries no
|
||||||
|
# information (a list-walk row has no body), so the stored value
|
||||||
|
# wins. extra_json is always a full serialization and is exempt.
|
||||||
|
for c in self._COMPARE_COLS:
|
||||||
|
if c != "extra_json" and not new_row.get(c) and cur_row.get(c):
|
||||||
|
new_row[c] = cur_row[c]
|
||||||
|
setattr(item, c, cur_row[c])
|
||||||
|
```
|
||||||
|
|
||||||
|
`setattr(item, c, …)` keeps the later `item.to_row()` (used to build the written row) consistent with the merged values. `Item` is a pydantic model; plain attribute assignment is allowed on these fields.
|
||||||
|
|
||||||
|
- [ ] **Step 4: Run to verify pass**
|
||||||
|
|
||||||
|
Run: `uv run --no-sync pytest tests/bib -q -p no:cacheprovider`
|
||||||
|
Expected: pass.
|
||||||
|
|
||||||
|
- [ ] **Step 5: Commit**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git add src/bib/store.py tests/bib/test_store_upsert_noop.py
|
||||||
|
git commit -m "fix(bib): upsert never blanks a stored value — empty incoming columns keep the stored ones (refs #615)"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
### Task 4: Idempotent `attach_file`
|
### Task 4: Idempotent `attach_file`
|
||||||
|
|
||||||
**Files:**
|
**Files:**
|
||||||
@@ -3409,6 +3485,19 @@ uv run stack comments dockets # reopens the store
|
|||||||
sqlite3 data/bib.sqlite "SELECT name FROM sqlite_master WHERE name='idx_attachments_item_filename'"
|
sqlite3 data/bib.sqlite "SELECT name FROM sqlite_master WHERE name='idx_attachments_item_filename'"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
- [ ] **Step 2b: Recover the abstracts blanked on 2026-09-08**
|
||||||
|
|
||||||
|
The old upsert overwrote 17,607 enriched bodies in CMS-2026-2377 and 1,115 in CMS-2017-0092 with '' (incident in the SDD ledger). They still carry `enriched:ok`, so backfill skips them. Clear the tag on those rows and re-enrich from the mirror (idempotent attach + data-preserving upsert are in place by now):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
for d in CMS-2026-2377 CMS-2017-0092; do
|
||||||
|
sqlite3 data/bib.sqlite "DELETE FROM item_tags WHERE tag_id=(SELECT id FROM tags WHERE name='enriched:ok') AND item_id IN (SELECT id FROM items WHERE url LIKE 'https://www.regulations.gov/comment/$d-%' AND coalesce(abstract,'')='')"
|
||||||
|
uv run stack bib backfill-comments --docket $d --mirror
|
||||||
|
done
|
||||||
|
sqlite3 data/bib.sqlite "SELECT substr(url,37,13), count(*), sum(coalesce(abstract,'')<>'') FROM items WHERE url LIKE 'https://www.regulations.gov/comment/CMS-2026-2377-%' OR url LIKE 'https://www.regulations.gov/comment/CMS-2017-0092-%' GROUP BY 1"
|
||||||
|
```
|
||||||
|
Expected: non-empty abstracts ≈ total for both dockets (a few genuinely empty bodies are fine). Then re-run `stack comments extract` and `stack llm index --collection comments` so any comment whose text was lost from combined.md/index is restored (the index still holds pre-wipe text; fingerprints change because `updated_at` moved, hash-skip absorbs unchanged text).
|
||||||
|
|
||||||
- [ ] **Step 3: Seal the historical dockets**
|
- [ ] **Step 3: Seal the historical dockets**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
86
src/bib/dockets.py
Normal file
86
src/bib/dockets.py
Normal file
@@ -0,0 +1,86 @@
|
|||||||
|
"""Docket-level state for the regulations.gov comment pipeline.
|
||||||
|
|
||||||
|
A *docket* row (``dockets`` table in bib.sqlite) remembers what every
|
||||||
|
stage would otherwise re-derive from the network: the reg.gov document
|
||||||
|
that carries the comments, the comment close date, the pull watermark,
|
||||||
|
and — once the docket is known complete — a *seal*. Sealed dockets are
|
||||||
|
skipped by fetch, backfill, extract and index unless ``--force``.
|
||||||
|
|
||||||
|
Pure helpers only; the store owns persistence (``Store.docket_*``).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from datetime import date, timedelta
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Iterable
|
||||||
|
|
||||||
|
DEFAULT_QUIET_DAYS = 30
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class Docket:
|
||||||
|
id: str
|
||||||
|
rule_cms_id: str = ""
|
||||||
|
fr_document_id: str = ""
|
||||||
|
fr_object_id: str = ""
|
||||||
|
comment_end_date: str = "" # YYYY-MM-DD
|
||||||
|
pull_watermark: str = "" # max lastModifiedDate seen on a clean walk
|
||||||
|
last_pull_at: str = "" # ISO-8601 UTC
|
||||||
|
last_pull_new: int | None = None
|
||||||
|
sealed_at: str = ""
|
||||||
|
seal_reason: str = ""
|
||||||
|
counts_json: str = "{}"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def sealed(self) -> bool:
|
||||||
|
return bool(self.sealed_at)
|
||||||
|
|
||||||
|
|
||||||
|
def should_seal(docket: Docket, today: date, quiet_days: int) -> bool:
|
||||||
|
"""True when the comment period closed ≥ *quiet_days* ago and the
|
||||||
|
most recent completed pull, itself after that quiet boundary, found
|
||||||
|
nothing new."""
|
||||||
|
if docket.sealed or not docket.comment_end_date or not docket.last_pull_at:
|
||||||
|
return False
|
||||||
|
if docket.last_pull_new is None or docket.last_pull_new > 0:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
end = date.fromisoformat(docket.comment_end_date[:10])
|
||||||
|
pulled = date.fromisoformat(docket.last_pull_at[:10])
|
||||||
|
except ValueError:
|
||||||
|
return False
|
||||||
|
boundary = end + timedelta(days=quiet_days)
|
||||||
|
return today >= boundary and pulled >= boundary
|
||||||
|
|
||||||
|
|
||||||
|
def fingerprint_files(paths: Iterable[Path]) -> str:
|
||||||
|
"""sha256 over sorted ``(name, size, mtime_ns)`` — cheap change
|
||||||
|
detection without reading contents. Missing files contribute their
|
||||||
|
name with size -1 so a deletion changes the fingerprint too."""
|
||||||
|
rows: list[str] = []
|
||||||
|
for p in paths:
|
||||||
|
p = Path(p)
|
||||||
|
try:
|
||||||
|
st = p.stat()
|
||||||
|
rows.append(f"{p.name}\x00{st.st_size}\x00{st.st_mtime_ns}")
|
||||||
|
except FileNotFoundError:
|
||||||
|
rows.append(f"{p.name}\x00-1\x000")
|
||||||
|
rows.sort()
|
||||||
|
return hashlib.sha256("\n".join(rows).encode()).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def _cfg():
|
||||||
|
from conf import cfg
|
||||||
|
|
||||||
|
return cfg
|
||||||
|
|
||||||
|
|
||||||
|
def quiet_days() -> int:
|
||||||
|
"""``[comments] seal_quiet_days`` from stack.toml, default 30."""
|
||||||
|
c = _cfg()
|
||||||
|
if "comments" in c and "seal_quiet_days" in c.comments:
|
||||||
|
return int(c.comments.seal_quiet_days)
|
||||||
|
return DEFAULT_QUIET_DAYS
|
||||||
@@ -29,12 +29,14 @@ import os
|
|||||||
import re
|
import re
|
||||||
import time
|
import time
|
||||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field, replace
|
||||||
|
from datetime import date
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING, Iterator
|
from typing import TYPE_CHECKING, Callable, Iterator
|
||||||
|
|
||||||
import httpx
|
import httpx
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
from bib.item import Source
|
from bib.item import Source
|
||||||
from bib.tag import Tag
|
from bib.tag import Tag
|
||||||
|
|
||||||
@@ -195,7 +197,13 @@ class Client:
|
|||||||
|
|
||||||
# ── Comment iteration ──────────────────────────────────────
|
# ── Comment iteration ──────────────────────────────────────
|
||||||
|
|
||||||
def iter_comments(self, object_id: str) -> Iterator[Comment]:
|
def iter_comments(
|
||||||
|
self,
|
||||||
|
object_id: str,
|
||||||
|
*,
|
||||||
|
since: str = "",
|
||||||
|
on_error: "Callable[[Exception], None] | None" = None,
|
||||||
|
) -> Iterator[Comment]:
|
||||||
"""Yield every comment against a single FR document.
|
"""Yield every comment against a single FR document.
|
||||||
|
|
||||||
The ``commentOnId`` filter on ``/comments`` takes a document's
|
The ``commentOnId`` filter on ``/comments`` takes a document's
|
||||||
@@ -211,8 +219,14 @@ class Client:
|
|||||||
2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` —
|
2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` —
|
||||||
returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space-
|
returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space-
|
||||||
separated, no timezone suffix). Normalize before sending.
|
separated, no timezone suffix). Normalize before sending.
|
||||||
|
|
||||||
|
*since* seeds the ``lastModifiedDate`` cursor so an incremental
|
||||||
|
walk starts where the last clean one ended (the boundary row is
|
||||||
|
re-yielded; the store's no-op upsert absorbs it). *on_error* is
|
||||||
|
invoked with the ``HTTPStatusError`` before the walk stops, so a
|
||||||
|
caller can tell "walked to the end" from "gave up".
|
||||||
"""
|
"""
|
||||||
cursor: str | None = None
|
cursor: str | None = _reg_date(since) if since else None
|
||||||
page = 1
|
page = 1
|
||||||
while True:
|
while True:
|
||||||
params: dict[str, str] = {
|
params: dict[str, str] = {
|
||||||
@@ -233,6 +247,8 @@ class Client:
|
|||||||
page,
|
page,
|
||||||
cursor,
|
cursor,
|
||||||
)
|
)
|
||||||
|
if on_error is not None:
|
||||||
|
on_error(e)
|
||||||
break
|
break
|
||||||
rows = data.get("data", [])
|
rows = data.get("data", [])
|
||||||
if not rows:
|
if not rows:
|
||||||
@@ -380,6 +396,7 @@ def backfill_details(
|
|||||||
client: Client,
|
client: Client,
|
||||||
*,
|
*,
|
||||||
docket: str = "",
|
docket: str = "",
|
||||||
|
force: bool = False,
|
||||||
limit: int | None = None,
|
limit: int | None = None,
|
||||||
log_path: Path | None = None,
|
log_path: Path | None = None,
|
||||||
commit_every: int = 25,
|
commit_every: int = 25,
|
||||||
@@ -401,6 +418,27 @@ def backfill_details(
|
|||||||
so we stop hammering it. Commits land every ``commit_every`` items
|
so we stop hammering it. Commits land every ``commit_every`` items
|
||||||
so a crash loses at most that many items of work.
|
so a crash loses at most that many items of work.
|
||||||
"""
|
"""
|
||||||
|
if docket and not force:
|
||||||
|
d = store.docket_get(docket)
|
||||||
|
if d is not None and d.sealed:
|
||||||
|
log.info(
|
||||||
|
"%s sealed %s (%s); backfill skipped",
|
||||||
|
docket,
|
||||||
|
d.sealed_at[:10],
|
||||||
|
d.seal_reason,
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
f"{docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipped — use --force",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"skipped_sealed": 1,
|
||||||
|
"seen": 0,
|
||||||
|
"enriched": 0,
|
||||||
|
"created": 0,
|
||||||
|
"attached": 0,
|
||||||
|
"errors": 0,
|
||||||
|
}
|
||||||
con = store._con() # noqa: SLF001
|
con = store._con() # noqa: SLF001
|
||||||
# Resume rule: treat ``enriched:ok`` as the completion marker.
|
# Resume rule: treat ``enriched:ok`` as the completion marker.
|
||||||
# Using a tag (not just the abstract) lets us handle "see attached"
|
# Using a tag (not just the abstract) lets us handle "see attached"
|
||||||
@@ -659,6 +697,7 @@ def backfill_from_mirror(
|
|||||||
mirror: Mirror,
|
mirror: Mirror,
|
||||||
*,
|
*,
|
||||||
docket: str,
|
docket: str,
|
||||||
|
force: bool = False,
|
||||||
limit: int | None = None,
|
limit: int | None = None,
|
||||||
log_path: Path | None = None,
|
log_path: Path | None = None,
|
||||||
commit_every: int = 200,
|
commit_every: int = 200,
|
||||||
@@ -672,6 +711,27 @@ def backfill_from_mirror(
|
|||||||
S3 fetches fan out over a thread pool; every Store/sqlite call stays
|
S3 fetches fan out over a thread pool; every Store/sqlite call stays
|
||||||
on this thread. Idempotent: ``enriched:ok`` items are skipped.
|
on this thread. Idempotent: ``enriched:ok`` items are skipped.
|
||||||
"""
|
"""
|
||||||
|
if docket and not force:
|
||||||
|
d = store.docket_get(docket)
|
||||||
|
if d is not None and d.sealed:
|
||||||
|
log.info(
|
||||||
|
"%s sealed %s (%s); backfill skipped",
|
||||||
|
docket,
|
||||||
|
d.sealed_at[:10],
|
||||||
|
d.seal_reason,
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
f"{docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipped — use --force",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"skipped_sealed": 1,
|
||||||
|
"seen": 0,
|
||||||
|
"enriched": 0,
|
||||||
|
"created": 0,
|
||||||
|
"attached": 0,
|
||||||
|
"errors": 0,
|
||||||
|
}
|
||||||
con = store._con() # noqa: SLF001
|
con = store._con() # noqa: SLF001
|
||||||
existing: dict[str, tuple[str, bool]] = {} # cid -> (key, enriched)
|
existing: dict[str, tuple[str, bool]] = {} # cid -> (key, enriched)
|
||||||
for key, url, enriched in con.execute(
|
for key, url, enriched in con.execute(
|
||||||
@@ -784,6 +844,19 @@ def upsert_comment(
|
|||||||
extra_tags: list[str] | None = None,
|
extra_tags: list[str] | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Upsert the comment as a Source item, return bib key."""
|
"""Upsert the comment as a Source item, return bib key."""
|
||||||
|
return upsert_comment_status(store, comment, cms_id=cms_id, extra_tags=extra_tags)[
|
||||||
|
0
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def upsert_comment_status(
|
||||||
|
store: Store,
|
||||||
|
comment: Comment,
|
||||||
|
*,
|
||||||
|
cms_id: str = "",
|
||||||
|
extra_tags: list[str] | None = None,
|
||||||
|
) -> tuple[str, str]:
|
||||||
|
"""Like :func:`upsert_comment`, also returning created/updated/unchanged."""
|
||||||
# Title = the comment's own CMS-DOCKET-XXXX-NNNN id, optionally
|
# Title = the comment's own CMS-DOCKET-XXXX-NNNN id, optionally
|
||||||
# prefixed with the byline (org or first/last name) when the
|
# prefixed with the byline (org or first/last name) when the
|
||||||
# commenter filled those fields in. The previous "Comment on
|
# commenter filled those fields in. The previous "Comment on
|
||||||
@@ -816,7 +889,146 @@ def upsert_comment(
|
|||||||
for t in tags:
|
for t in tags:
|
||||||
item.add_tag(t)
|
item.add_tag(t)
|
||||||
|
|
||||||
return store.upsert(item)
|
return store.upsert_status(item)
|
||||||
|
|
||||||
|
|
||||||
|
# ── Docket walks ───────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def discover_docket(client: Client, docket_id: str, *, rule_cms_id: str = "") -> Docket:
|
||||||
|
"""One ``/documents`` listing → the docket's commentable document.
|
||||||
|
|
||||||
|
Picks the document that has an ``objectId`` and a ``commentEndDate``
|
||||||
|
(latest close date wins when several qualify). Called once per
|
||||||
|
docket; the result is persisted so later runs make no API call.
|
||||||
|
"""
|
||||||
|
best: dict | None = None
|
||||||
|
for fr_doc in client.find_documents_in_docket(docket_id):
|
||||||
|
attrs = fr_doc.get("attributes") or {}
|
||||||
|
if not attrs.get("objectId") or not attrs.get("commentEndDate"):
|
||||||
|
continue
|
||||||
|
if (
|
||||||
|
best is None
|
||||||
|
or attrs["commentEndDate"]
|
||||||
|
> (best.get("attributes") or {})["commentEndDate"]
|
||||||
|
):
|
||||||
|
best = fr_doc
|
||||||
|
attrs = (best or {}).get("attributes") or {}
|
||||||
|
return Docket(
|
||||||
|
id=docket_id,
|
||||||
|
rule_cms_id=rule_cms_id,
|
||||||
|
fr_document_id=(best or {}).get("id", ""),
|
||||||
|
fr_object_id=attrs.get("objectId", ""),
|
||||||
|
comment_end_date=(attrs.get("commentEndDate") or "")[:10],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class WalkResult:
|
||||||
|
created: int = 0
|
||||||
|
updated: int = 0
|
||||||
|
unchanged: int = 0
|
||||||
|
watermark: str = ""
|
||||||
|
clean: bool = True
|
||||||
|
sealed: bool = False
|
||||||
|
|
||||||
|
|
||||||
|
def walk_docket(
|
||||||
|
store: Store,
|
||||||
|
client: Client,
|
||||||
|
docket: Docket,
|
||||||
|
*,
|
||||||
|
attachments: bool = False,
|
||||||
|
limit: int = 0,
|
||||||
|
force: bool = False,
|
||||||
|
quiet_days: int = 30,
|
||||||
|
today: date | None = None,
|
||||||
|
scratch_root: Path = Path(".state/comments"),
|
||||||
|
echo: Callable[[str], None] | None = None,
|
||||||
|
) -> WalkResult:
|
||||||
|
"""Pull *docket*'s comments from the stored watermark, upsert them,
|
||||||
|
advance the watermark on a clean walk, and auto-seal when
|
||||||
|
:func:`should_seal` says so. ``force`` walks from page 1 and never
|
||||||
|
seals. Returns per-status counts.
|
||||||
|
|
||||||
|
Attachments are fetched for created/updated comments, and for every
|
||||||
|
comment under ``force`` (an unchanged row can still be missing its
|
||||||
|
files). A docket is only auto-sealed when bib actually holds at
|
||||||
|
least one of its comments.
|
||||||
|
"""
|
||||||
|
from bib.dockets import should_seal
|
||||||
|
|
||||||
|
res = WalkResult()
|
||||||
|
since = "" if force else docket.pull_watermark
|
||||||
|
max_lm = docket.pull_watermark if not force else ""
|
||||||
|
n = 0
|
||||||
|
|
||||||
|
def _err(_e: Exception) -> None:
|
||||||
|
res.clean = False
|
||||||
|
|
||||||
|
scratch = scratch_root / docket.id
|
||||||
|
for c in client.iter_comments(docket.fr_object_id, since=since, on_error=_err):
|
||||||
|
if limit and n >= limit:
|
||||||
|
res.clean = False # a capped walk is not a complete one
|
||||||
|
break
|
||||||
|
key, status = upsert_comment_status(
|
||||||
|
store, c, cms_id=docket.rule_cms_id, extra_tags=[f"reg-docket:{docket.id}"]
|
||||||
|
)
|
||||||
|
setattr(res, status, getattr(res, status) + 1)
|
||||||
|
if attachments and c.attachment_count and (force or status != "unchanged"):
|
||||||
|
for att in client.attachments_for(c.id):
|
||||||
|
path = client.download_attachment(att.url, scratch / c.id)
|
||||||
|
if path:
|
||||||
|
store.attach_file(key, path, title=att.filename)
|
||||||
|
lm = (c.raw.get("attributes") or {}).get("lastModifiedDate") or ""
|
||||||
|
if lm > max_lm:
|
||||||
|
max_lm = lm
|
||||||
|
n += 1
|
||||||
|
if n % 50 == 0:
|
||||||
|
store._con().commit() # noqa: SLF001
|
||||||
|
if echo:
|
||||||
|
echo(f" {n} comments")
|
||||||
|
store._con().commit() # noqa: SLF001
|
||||||
|
res.watermark = max_lm
|
||||||
|
|
||||||
|
if res.clean and not force:
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
# ``today`` doubles as a deterministic override of "now" for
|
||||||
|
# tests exercising the auto-seal quiet-window boundary; real
|
||||||
|
# callers leave it unset and get the actual wall-clock time.
|
||||||
|
if today is not None:
|
||||||
|
now = f"{today.isoformat()}T00:00:00Z"
|
||||||
|
else:
|
||||||
|
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||||
|
updated = replace(
|
||||||
|
docket, pull_watermark=max_lm, last_pull_at=now, last_pull_new=res.created
|
||||||
|
)
|
||||||
|
store.docket_upsert(updated)
|
||||||
|
counts = docket_counts(store, docket.id)
|
||||||
|
# A docket with nothing in bib is a farm that never landed, not a
|
||||||
|
# finished one: an empty walk over it must not seal it shut.
|
||||||
|
if counts["comments"] and should_seal(
|
||||||
|
updated, today or date.today(), quiet_days
|
||||||
|
):
|
||||||
|
store.docket_seal(docket.id, reason="auto", counts=counts)
|
||||||
|
res.sealed = True
|
||||||
|
return res
|
||||||
|
|
||||||
|
|
||||||
|
def docket_counts(store: Store, docket_id: str) -> dict[str, int]:
|
||||||
|
"""``{"comments": n, "enriched": n}`` from SQL only (no filesystem)."""
|
||||||
|
con = store._con() # noqa: SLF001
|
||||||
|
like = f"https://www.regulations.gov/comment/{docket_id}-%"
|
||||||
|
comments = con.execute(
|
||||||
|
"SELECT count(*) FROM items WHERE url LIKE ?", (like,)
|
||||||
|
).fetchone()[0]
|
||||||
|
enriched = con.execute(
|
||||||
|
"""SELECT count(*) FROM items i WHERE i.url LIKE ? AND i.id IN (
|
||||||
|
SELECT item_id FROM item_tags WHERE tag_id IN (SELECT id FROM tags WHERE name='enriched:ok'))""",
|
||||||
|
(like,),
|
||||||
|
).fetchone()[0]
|
||||||
|
return {"comments": comments, "enriched": enriched}
|
||||||
|
|
||||||
|
|
||||||
# ── Internals ──────────────────────────────────────────────────
|
# ── Internals ──────────────────────────────────────────────────
|
||||||
|
|||||||
@@ -137,3 +137,25 @@ CREATE TABLE IF NOT EXISTS fr_links (
|
|||||||
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')),
|
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')),
|
||||||
UNIQUE (item_key, url)
|
UNIQUE (item_key, url)
|
||||||
);
|
);
|
||||||
|
|
||||||
|
-- ── Dockets (P47) ────────────────────────────────────────────────
|
||||||
|
-- One row per reg.gov docket: what every stage would otherwise re-fetch
|
||||||
|
-- (document/object ids, close date), the pull watermark, and the seal
|
||||||
|
-- that marks the docket complete. See bib/dockets.py.
|
||||||
|
CREATE TABLE IF NOT EXISTS dockets (
|
||||||
|
id TEXT PRIMARY KEY,
|
||||||
|
rule_cms_id TEXT NOT NULL DEFAULT '',
|
||||||
|
fr_document_id TEXT NOT NULL DEFAULT '',
|
||||||
|
fr_object_id TEXT NOT NULL DEFAULT '',
|
||||||
|
comment_end_date TEXT NOT NULL DEFAULT '',
|
||||||
|
pull_watermark TEXT NOT NULL DEFAULT '',
|
||||||
|
last_pull_at TEXT NOT NULL DEFAULT '',
|
||||||
|
last_pull_new INTEGER,
|
||||||
|
sealed_at TEXT NOT NULL DEFAULT '',
|
||||||
|
seal_reason TEXT NOT NULL DEFAULT '',
|
||||||
|
counts_json TEXT NOT NULL DEFAULT '{}'
|
||||||
|
);
|
||||||
|
|
||||||
|
-- upsert() dedupes by URL on every comment; without this every upsert
|
||||||
|
-- was a full scan of items.
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_items_url ON items(url);
|
||||||
|
|||||||
258
src/bib/store.py
258
src/bib/store.py
@@ -17,6 +17,7 @@ Usage::
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
import mimetypes
|
import mimetypes
|
||||||
import random
|
import random
|
||||||
import shutil
|
import shutil
|
||||||
@@ -25,6 +26,7 @@ from importlib import resources
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
from bib.item import Item
|
from bib.item import Item
|
||||||
|
|
||||||
|
|
||||||
@@ -81,6 +83,19 @@ class Store:
|
|||||||
def _init_schema(self) -> None:
|
def _init_schema(self) -> None:
|
||||||
ddl = resources.files("bib").joinpath("schema.sql").read_text()
|
ddl = resources.files("bib").joinpath("schema.sql").read_text()
|
||||||
self._con().executescript(ddl)
|
self._con().executescript(ddl)
|
||||||
|
# The (item_id, filename) unique index can only exist once the
|
||||||
|
# dedupe migration has run (dev/scripts/dedupe_attachments.py);
|
||||||
|
# until then we leave it off rather than fail to open the store.
|
||||||
|
con = self._con()
|
||||||
|
dup = con.execute(
|
||||||
|
"SELECT 1 FROM attachments GROUP BY item_id, filename HAVING count(*) > 1 LIMIT 1"
|
||||||
|
).fetchone()
|
||||||
|
if dup is None:
|
||||||
|
con.execute(
|
||||||
|
"CREATE UNIQUE INDEX IF NOT EXISTS idx_attachments_item_filename "
|
||||||
|
"ON attachments(item_id, filename)"
|
||||||
|
)
|
||||||
|
con.commit()
|
||||||
|
|
||||||
def close(self) -> None:
|
def close(self) -> None:
|
||||||
if self._connection is not None:
|
if self._connection is not None:
|
||||||
@@ -195,6 +210,11 @@ class Store:
|
|||||||
self._sync_tags(item_id, new_tags)
|
self._sync_tags(item_id, new_tags)
|
||||||
if new_collections is not None:
|
if new_collections is not None:
|
||||||
self._sync_collections(item_id, new_collections)
|
self._sync_collections(item_id, new_collections)
|
||||||
|
if not fields:
|
||||||
|
# A tags/collections-only rewrite still changes the item
|
||||||
|
# as the indexer sees it (year:/project:/cms-rule: land in
|
||||||
|
# chunk metadata), and nothing else stamped the row.
|
||||||
|
self._stamp(key)
|
||||||
|
|
||||||
con.commit()
|
con.commit()
|
||||||
|
|
||||||
@@ -204,6 +224,17 @@ class Store:
|
|||||||
con.execute("DELETE FROM items WHERE key = ?", (key,))
|
con.execute("DELETE FROM items WHERE key = ?", (key,))
|
||||||
con.commit()
|
con.commit()
|
||||||
|
|
||||||
|
_COMPARE_COLS = (
|
||||||
|
"item_type",
|
||||||
|
"title",
|
||||||
|
"url",
|
||||||
|
"date_published",
|
||||||
|
"abstract",
|
||||||
|
"institution",
|
||||||
|
"extra",
|
||||||
|
"extra_json",
|
||||||
|
)
|
||||||
|
|
||||||
def upsert(
|
def upsert(
|
||||||
self,
|
self,
|
||||||
item: Item,
|
item: Item,
|
||||||
@@ -212,37 +243,86 @@ class Store:
|
|||||||
collection: str = "",
|
collection: str = "",
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Create or update an item, deduplicating by URL."""
|
"""Create or update an item, deduplicating by URL."""
|
||||||
if item.url:
|
return self.upsert_status(item, tags=tags, collection=collection)[0]
|
||||||
con = self._con()
|
|
||||||
existing = con.execute(
|
|
||||||
"SELECT key FROM items WHERE url = ?", (item.url,)
|
|
||||||
).fetchone()
|
|
||||||
if existing:
|
|
||||||
ekey = existing["key"]
|
|
||||||
if tags:
|
|
||||||
for tag in tags:
|
|
||||||
label = tag.label if hasattr(tag, "label") else str(tag)
|
|
||||||
item.add_tag(label)
|
|
||||||
if collection and collection not in item.collections:
|
|
||||||
item.collections.append(collection)
|
|
||||||
item.stamp_access()
|
|
||||||
# Merge with what's already stored — a re-ingest must
|
|
||||||
# never clobber tags/collections curated on the row
|
|
||||||
# since the last ingest (#624: the CY2027 NPRM lost its
|
|
||||||
# sup: registration tag to this and silently vanished
|
|
||||||
# from the tag-scoped Zotero sync). Deliberate removal
|
|
||||||
# goes through remove_tag, not upsert.
|
|
||||||
current = self.get(ekey)
|
|
||||||
row = item.to_row()
|
|
||||||
row.pop("key", None)
|
|
||||||
row["tags"] = list(dict.fromkeys([*current.tags, *item.tags]))
|
|
||||||
row["collections"] = list(
|
|
||||||
dict.fromkeys([*current.collections, *item.collections])
|
|
||||||
)
|
|
||||||
self.update(ekey, **row)
|
|
||||||
return ekey
|
|
||||||
|
|
||||||
return self.create(item, tags=tags, collection=collection)
|
def upsert_status(
|
||||||
|
self,
|
||||||
|
item: Item,
|
||||||
|
*,
|
||||||
|
tags: list[Any] | None = None,
|
||||||
|
collection: str = "",
|
||||||
|
) -> tuple[str, str]:
|
||||||
|
"""Like :meth:`upsert` but also report what happened:
|
||||||
|
``"created"``, ``"updated"`` or ``"unchanged"``.
|
||||||
|
|
||||||
|
``unchanged`` means every compared column, the merged tag set and
|
||||||
|
the merged collection set already match the stored row — nothing
|
||||||
|
is written, no ``access_date``/``updated_at`` stamp, no tag row
|
||||||
|
churn. A re-farm over a complete docket is therefore a read-only
|
||||||
|
pass over ``items``.
|
||||||
|
|
||||||
|
Merge rules, all of which make ``unchanged`` a comparison against
|
||||||
|
the *merged* row rather than the incoming one:
|
||||||
|
|
||||||
|
* columns: an empty incoming value keeps the stored one (a
|
||||||
|
list-walk row carries no body, and must not blank an enriched
|
||||||
|
abstract). ``extra_json`` is exempt — it is always a full
|
||||||
|
serialization, so an incoming ``{}`` is a real value and
|
||||||
|
replaces what is stored.
|
||||||
|
* tags and collections: union-merged with what is stored, never
|
||||||
|
replaced. Removal goes through :meth:`remove_tag`.
|
||||||
|
"""
|
||||||
|
if tags:
|
||||||
|
for tag in tags:
|
||||||
|
label = tag.label if hasattr(tag, "label") else str(tag)
|
||||||
|
item.add_tag(label)
|
||||||
|
if collection and collection not in item.collections:
|
||||||
|
item.collections.append(collection)
|
||||||
|
|
||||||
|
if not item.url:
|
||||||
|
return self.create(item), "created"
|
||||||
|
|
||||||
|
con = self._con()
|
||||||
|
existing = con.execute(
|
||||||
|
"SELECT key FROM items WHERE url = ?", (item.url,)
|
||||||
|
).fetchone()
|
||||||
|
if not existing:
|
||||||
|
return self.create(item), "created"
|
||||||
|
|
||||||
|
ekey = existing["key"]
|
||||||
|
# Merge with what's already stored — a re-ingest must never
|
||||||
|
# clobber tags/collections curated on the row since the last
|
||||||
|
# ingest (#624). Deliberate removal goes through remove_tag.
|
||||||
|
current = self.get(ekey)
|
||||||
|
merged_tags = list(dict.fromkeys([*current.tags, *item.tags]))
|
||||||
|
merged_cols = list(dict.fromkeys([*current.collections, *item.collections]))
|
||||||
|
|
||||||
|
new_row = item.to_row()
|
||||||
|
cur_row = current.to_row()
|
||||||
|
# Merge, don't replace: an empty incoming value carries no
|
||||||
|
# information (a list-walk row has no body), so the stored value
|
||||||
|
# wins. extra_json is always a full serialization and is exempt.
|
||||||
|
for c in self._COMPARE_COLS:
|
||||||
|
if c != "extra_json" and not new_row.get(c) and cur_row.get(c):
|
||||||
|
new_row[c] = cur_row[c]
|
||||||
|
setattr(item, c, cur_row[c])
|
||||||
|
same_cols = all(
|
||||||
|
new_row.get(c, "") == cur_row.get(c, "") for c in self._COMPARE_COLS
|
||||||
|
)
|
||||||
|
if (
|
||||||
|
same_cols
|
||||||
|
and set(merged_tags) == set(current.tags)
|
||||||
|
and set(merged_cols) == set(current.collections)
|
||||||
|
):
|
||||||
|
return ekey, "unchanged"
|
||||||
|
|
||||||
|
item.stamp_access()
|
||||||
|
row = item.to_row()
|
||||||
|
row.pop("key", None)
|
||||||
|
row["tags"] = merged_tags
|
||||||
|
row["collections"] = merged_cols
|
||||||
|
self.update(ekey, **row)
|
||||||
|
return ekey, "updated"
|
||||||
|
|
||||||
# ── Query ────────────────────────────────────────────────────
|
# ── Query ────────────────────────────────────────────────────
|
||||||
|
|
||||||
@@ -452,6 +532,19 @@ class Store:
|
|||||||
(row["id"], item_id),
|
(row["id"], item_id),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def _stamp(self, item_key: str) -> None:
|
||||||
|
"""Mark the item row as changed *now*.
|
||||||
|
|
||||||
|
Tags carry indexed metadata (``year:``, ``project:``,
|
||||||
|
``cms-rule:``), so a tag-only edit has to move ``updated_at`` or
|
||||||
|
the llm indexer's fingerprint can never see it (#615).
|
||||||
|
"""
|
||||||
|
self._con().execute(
|
||||||
|
"UPDATE items SET updated_at = strftime('%Y-%m-%dT%H:%M:%SZ','now') "
|
||||||
|
"WHERE key = ?",
|
||||||
|
(item_key,),
|
||||||
|
)
|
||||||
|
|
||||||
def add_tag(self, item_key: str, tag: str) -> None:
|
def add_tag(self, item_key: str, tag: str) -> None:
|
||||||
"""Add a single tag to an item."""
|
"""Add a single tag to an item."""
|
||||||
con = self._con()
|
con = self._con()
|
||||||
@@ -459,21 +552,25 @@ class Store:
|
|||||||
if row is None:
|
if row is None:
|
||||||
raise KeyError(f"Item not found: {item_key}")
|
raise KeyError(f"Item not found: {item_key}")
|
||||||
tag_id = self._ensure_tag(tag)
|
tag_id = self._ensure_tag(tag)
|
||||||
con.execute(
|
cur = con.execute(
|
||||||
"INSERT OR IGNORE INTO item_tags (item_id, tag_id) VALUES (?, ?)",
|
"INSERT OR IGNORE INTO item_tags (item_id, tag_id) VALUES (?, ?)",
|
||||||
(row["id"], tag_id),
|
(row["id"], tag_id),
|
||||||
)
|
)
|
||||||
|
if cur.rowcount: # re-tagging an already-tagged item changes nothing
|
||||||
|
self._stamp(item_key)
|
||||||
con.commit()
|
con.commit()
|
||||||
|
|
||||||
def remove_tag(self, item_key: str, tag: str) -> None:
|
def remove_tag(self, item_key: str, tag: str) -> None:
|
||||||
"""Remove a tag from an item."""
|
"""Remove a tag from an item."""
|
||||||
con = self._con()
|
con = self._con()
|
||||||
con.execute(
|
cur = con.execute(
|
||||||
"""DELETE FROM item_tags
|
"""DELETE FROM item_tags
|
||||||
WHERE item_id = (SELECT id FROM items WHERE key = ?)
|
WHERE item_id = (SELECT id FROM items WHERE key = ?)
|
||||||
AND tag_id = (SELECT id FROM tags WHERE name = ?)""",
|
AND tag_id = (SELECT id FROM tags WHERE name = ?)""",
|
||||||
(item_key, tag),
|
(item_key, tag),
|
||||||
)
|
)
|
||||||
|
if cur.rowcount: # the tag wasn't there — nothing changed
|
||||||
|
self._stamp(item_key)
|
||||||
con.commit()
|
con.commit()
|
||||||
|
|
||||||
def list_tags(self, *, namespace: str = "") -> list[dict[str, int]]:
|
def list_tags(self, *, namespace: str = "") -> list[dict[str, int]]:
|
||||||
@@ -497,6 +594,93 @@ class Store:
|
|||||||
).fetchall()
|
).fetchall()
|
||||||
return [{"name": r["name"], "count": r["cnt"]} for r in rows]
|
return [{"name": r["name"], "count": r["cnt"]} for r in rows]
|
||||||
|
|
||||||
|
# ── Dockets ──────────────────────────────────────────────────
|
||||||
|
|
||||||
|
_DOCKET_COLS = (
|
||||||
|
"id",
|
||||||
|
"rule_cms_id",
|
||||||
|
"fr_document_id",
|
||||||
|
"fr_object_id",
|
||||||
|
"comment_end_date",
|
||||||
|
"pull_watermark",
|
||||||
|
"last_pull_at",
|
||||||
|
"last_pull_new",
|
||||||
|
"sealed_at",
|
||||||
|
"seal_reason",
|
||||||
|
"counts_json",
|
||||||
|
)
|
||||||
|
|
||||||
|
def _docket_from_row(self, row: sqlite3.Row | None) -> Docket | None:
|
||||||
|
if row is None:
|
||||||
|
return None
|
||||||
|
d = {c: row[c] for c in self._DOCKET_COLS}
|
||||||
|
return Docket(**d)
|
||||||
|
|
||||||
|
def docket_get(self, docket_id: str) -> Docket | None:
|
||||||
|
row = (
|
||||||
|
self._con()
|
||||||
|
.execute("SELECT * FROM dockets WHERE id = ?", (docket_id,))
|
||||||
|
.fetchone()
|
||||||
|
)
|
||||||
|
return self._docket_from_row(row)
|
||||||
|
|
||||||
|
def docket_for_rule(self, cms_rule_id: str) -> Docket | None:
|
||||||
|
row = (
|
||||||
|
self._con()
|
||||||
|
.execute(
|
||||||
|
"SELECT * FROM dockets WHERE rule_cms_id = ? ORDER BY id LIMIT 1",
|
||||||
|
(cms_rule_id,),
|
||||||
|
)
|
||||||
|
.fetchone()
|
||||||
|
)
|
||||||
|
return self._docket_from_row(row)
|
||||||
|
|
||||||
|
def docket_upsert(self, docket: Docket) -> None:
|
||||||
|
cols = ", ".join(self._DOCKET_COLS)
|
||||||
|
marks = ", ".join(f":{c}" for c in self._DOCKET_COLS)
|
||||||
|
sets = ", ".join(f"{c} = excluded.{c}" for c in self._DOCKET_COLS if c != "id")
|
||||||
|
con = self._con()
|
||||||
|
con.execute(
|
||||||
|
f"INSERT INTO dockets ({cols}) VALUES ({marks}) " # noqa: S608
|
||||||
|
f"ON CONFLICT(id) DO UPDATE SET {sets}",
|
||||||
|
{c: getattr(docket, c) for c in self._DOCKET_COLS},
|
||||||
|
)
|
||||||
|
con.commit()
|
||||||
|
|
||||||
|
def dockets(self) -> list[Docket]:
|
||||||
|
rows = self._con().execute("SELECT * FROM dockets ORDER BY id").fetchall()
|
||||||
|
return [self._docket_from_row(r) for r in rows]
|
||||||
|
|
||||||
|
def sealed_dockets(self) -> dict[str, str]:
|
||||||
|
"""docket id → sealed_at for every sealed docket."""
|
||||||
|
rows = (
|
||||||
|
self._con()
|
||||||
|
.execute("SELECT id, sealed_at FROM dockets WHERE sealed_at <> ''")
|
||||||
|
.fetchall()
|
||||||
|
)
|
||||||
|
return {r["id"]: r["sealed_at"] for r in rows}
|
||||||
|
|
||||||
|
def docket_seal(
|
||||||
|
self, docket_id: str, *, reason: str, counts: dict[str, int]
|
||||||
|
) -> None:
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||||
|
con = self._con()
|
||||||
|
con.execute(
|
||||||
|
"UPDATE dockets SET sealed_at = ?, seal_reason = ?, counts_json = ? WHERE id = ?",
|
||||||
|
(now, reason, json.dumps(counts, sort_keys=True), docket_id),
|
||||||
|
)
|
||||||
|
con.commit()
|
||||||
|
|
||||||
|
def docket_unseal(self, docket_id: str) -> None:
|
||||||
|
con = self._con()
|
||||||
|
con.execute(
|
||||||
|
"UPDATE dockets SET sealed_at = '', seal_reason = '' WHERE id = ?",
|
||||||
|
(docket_id,),
|
||||||
|
)
|
||||||
|
con.commit()
|
||||||
|
|
||||||
# ── Attachments & Notes ──────────────────────────────────────
|
# ── Attachments & Notes ──────────────────────────────────────
|
||||||
|
|
||||||
def attach_file(self, item_key: str, path: Path, *, title: str = "") -> str:
|
def attach_file(self, item_key: str, path: Path, *, title: str = "") -> str:
|
||||||
@@ -506,6 +690,14 @@ class Store:
|
|||||||
if row is None:
|
if row is None:
|
||||||
raise KeyError(f"Item not found: {item_key}")
|
raise KeyError(f"Item not found: {item_key}")
|
||||||
|
|
||||||
|
filename = title or path.name
|
||||||
|
dup = con.execute(
|
||||||
|
"SELECT key FROM attachments WHERE item_id = ? AND filename = ?",
|
||||||
|
(row["id"], filename),
|
||||||
|
).fetchone()
|
||||||
|
if dup:
|
||||||
|
return dup["key"]
|
||||||
|
|
||||||
att_key = _generate_key()
|
att_key = _generate_key()
|
||||||
dest_dir = self._storage / att_key
|
dest_dir = self._storage / att_key
|
||||||
dest_dir.mkdir(parents=True, exist_ok=True)
|
dest_dir.mkdir(parents=True, exist_ok=True)
|
||||||
@@ -521,7 +713,7 @@ class Store:
|
|||||||
(
|
(
|
||||||
row["id"],
|
row["id"],
|
||||||
att_key,
|
att_key,
|
||||||
title or path.name,
|
filename,
|
||||||
content_type,
|
content_type,
|
||||||
# resolve(): through a symlinked storage dir (a git
|
# resolve(): through a symlinked storage dir (a git
|
||||||
# worktree's data/ pointing at the real data dir) the
|
# worktree's data/ pointing at the real data dir) the
|
||||||
|
|||||||
226
src/cli/bib.py
226
src/cli/bib.py
@@ -126,9 +126,29 @@ def discover_pfs_rules(
|
|||||||
store.upsert(rule)
|
store.upsert(rule)
|
||||||
|
|
||||||
|
|
||||||
|
def _walk_and_report(store, api, docket, *, attachments, limit, force, echo) -> None:
|
||||||
|
from bib.dockets import quiet_days
|
||||||
|
from bib.regulations_gov import walk_docket
|
||||||
|
|
||||||
|
r = walk_docket(
|
||||||
|
store,
|
||||||
|
api,
|
||||||
|
docket,
|
||||||
|
attachments=attachments,
|
||||||
|
limit=limit,
|
||||||
|
force=force,
|
||||||
|
quiet_days=quiet_days(),
|
||||||
|
echo=echo,
|
||||||
|
)
|
||||||
|
echo(
|
||||||
|
f" {docket.id}: created={r.created} updated={r.updated} "
|
||||||
|
f"unchanged={r.unchanged} clean={r.clean}" + (" — sealed" if r.sealed else "")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@app.command(name="fetch-docket-comments")
|
@app.command(name="fetch-docket-comments")
|
||||||
def fetch_docket_comments(
|
def fetch_docket_comments(
|
||||||
docket: str = typer.Argument(..., help="e.g. CMS-1676-P"),
|
docket: str = typer.Argument(..., help="reg.gov docket id, e.g. CMS-2026-2377"),
|
||||||
cms_id: str = typer.Option(
|
cms_id: str = typer.Option(
|
||||||
"", "--cms-id", help="Associate comments with a specific CMS rule tag."
|
"", "--cms-id", help="Associate comments with a specific CMS rule tag."
|
||||||
),
|
),
|
||||||
@@ -138,52 +158,54 @@ def fetch_docket_comments(
|
|||||||
attachments: bool = typer.Option(
|
attachments: bool = typer.Option(
|
||||||
False,
|
False,
|
||||||
"--attachments",
|
"--attachments",
|
||||||
help="Also download each comment's PDF/DOCX attachments "
|
help="Also download each comment's PDF/DOCX attachments.",
|
||||||
"(expensive — skim first without).",
|
|
||||||
),
|
),
|
||||||
sleep: float = typer.Option(
|
sleep: float = typer.Option(
|
||||||
1.3, "--sleep", help="Seconds between API calls (rate budget)."
|
1.3, "--sleep", help="Seconds between API calls (rate budget)."
|
||||||
),
|
),
|
||||||
|
force: bool = typer.Option(
|
||||||
|
False,
|
||||||
|
"--force",
|
||||||
|
help="Walk from page 1 even if sealed / watermarked. Never unseals.",
|
||||||
|
),
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Walk every comment on one docket; upsert as Source items.
|
"""Walk one docket's comments from its stored watermark; upsert as Source items.
|
||||||
|
|
||||||
Policy: we iterate every FR document indexed under the docket and
|
The first run discovers the docket's commentable FR document (one
|
||||||
pull their comments. For one PFS rule that usually means one parent
|
listing call) and stores it; later runs make no discovery calls.
|
||||||
document with a few thousand to tens of thousands of comments.
|
Sealed dockets are skipped unless --force.
|
||||||
"""
|
"""
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from bib import connect
|
from bib import connect
|
||||||
from bib.regulations_gov import Client, upsert_comment
|
from bib.regulations_gov import Client, discover_docket
|
||||||
|
|
||||||
store = connect()
|
store = connect()
|
||||||
scratch = Path(f".state/comments/{docket}")
|
|
||||||
|
|
||||||
with Client(sleep=sleep) as api:
|
with Client(sleep=sleep) as api:
|
||||||
docs = api.find_documents_in_docket(docket)
|
d = store.docket_get(docket)
|
||||||
typer.echo(f" docket {docket}: {len(docs)} FR documents indexed")
|
if d is None:
|
||||||
total = 0
|
d = discover_docket(api, docket, rule_cms_id=cms_id)
|
||||||
for fr_doc in docs:
|
store.docket_upsert(d)
|
||||||
attrs = fr_doc.get("attributes") or {}
|
elif cms_id and not d.rule_cms_id:
|
||||||
fr_id = fr_doc["id"]
|
from dataclasses import replace
|
||||||
object_id = attrs.get("objectId")
|
|
||||||
if not object_id or not attrs.get("commentEndDate"):
|
d = replace(d, rule_cms_id=cms_id)
|
||||||
continue
|
store.docket_upsert(d)
|
||||||
typer.echo(f" ↓ comments on {fr_id} (objectId={object_id})")
|
if d.sealed and not force:
|
||||||
for c in api.iter_comments(object_id):
|
typer.echo(
|
||||||
if limit and total >= limit:
|
f" {docket}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipping — use --force to re-walk"
|
||||||
return
|
)
|
||||||
key = upsert_comment(store, c, cms_id=cms_id)
|
return
|
||||||
if attachments and c.attachment_count:
|
if not d.fr_object_id:
|
||||||
for att in api.attachments_for(c.id):
|
typer.echo(f" {docket}: no commentable FR document found")
|
||||||
path = api.download_attachment(att.url, scratch / c.id)
|
return
|
||||||
if path:
|
_walk_and_report(
|
||||||
store.attach_file(key, path, title=att.filename)
|
store,
|
||||||
total += 1
|
api,
|
||||||
if total % 50 == 0:
|
d,
|
||||||
typer.echo(f" processed {total} comments")
|
attachments=attachments,
|
||||||
store._con().commit() # noqa: SLF001
|
limit=limit,
|
||||||
typer.echo(f" total: {total} comments upserted")
|
force=force,
|
||||||
|
echo=typer.echo,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@app.command(name="fetch-pfs-comments")
|
@app.command(name="fetch-pfs-comments")
|
||||||
@@ -192,46 +214,59 @@ def fetch_pfs_comments(
|
|||||||
until: str = typer.Option("", "--until"),
|
until: str = typer.Option("", "--until"),
|
||||||
attachments: bool = typer.Option(False, "--attachments"),
|
attachments: bool = typer.Option(False, "--attachments"),
|
||||||
per_docket_limit: int = typer.Option(
|
per_docket_limit: int = typer.Option(
|
||||||
0,
|
0, "--per-docket-limit", help="Cap comments fetched per docket. 0 = unlimited."
|
||||||
"--per-docket-limit",
|
|
||||||
help="Cap comments fetched per docket. 0 = unlimited.",
|
|
||||||
),
|
),
|
||||||
sleep: float = typer.Option(1.3, "--sleep"),
|
sleep: float = typer.Option(1.3, "--sleep"),
|
||||||
docket: str = typer.Option(
|
docket: str = typer.Option(
|
||||||
"",
|
"",
|
||||||
"--docket",
|
"--docket",
|
||||||
help="Only pull this reg.gov docket (e.g. CMS-2026-2377); "
|
help="Only pull this reg.gov docket (e.g. CMS-2026-2377); other dockets are skipped before any API call once known.",
|
||||||
"the other PFS dockets are skipped without any API calls.",
|
),
|
||||||
|
force: bool = typer.Option(
|
||||||
|
False,
|
||||||
|
"--force",
|
||||||
|
help="Walk every docket from page 1, sealed or not. Never unseals.",
|
||||||
),
|
),
|
||||||
) -> None:
|
) -> None:
|
||||||
"""One-shot: discover every PFS rule since *since* and pull every
|
"""Discover every PFS proposed rule since *since* and pull new comments
|
||||||
comment on each of their dockets.
|
on each docket from its stored watermark.
|
||||||
|
|
||||||
Heavy run. A single PFS proposed rule can hold 5K–25K comments;
|
Sealed dockets (comment period closed + quiet period + an empty pull)
|
||||||
times ~20 rules and with attachments this runs for many hours
|
cost nothing: no rule-metadata fetch, no resolve call, no walk.
|
||||||
under the reg.gov rate budget. Use ``--per-docket-limit`` to
|
|
||||||
smoke-test first.
|
|
||||||
"""
|
"""
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from bib import connect
|
from bib import connect
|
||||||
from bib.federalregister import pfs_rules, split_docket_ids
|
from bib.federalregister import pfs_rules, split_docket_ids
|
||||||
from bib.regulations_gov import Client, upsert_comment
|
from bib.regulations_gov import Client, discover_docket
|
||||||
from bib.tag import Tag
|
from bib.tag import Tag
|
||||||
from bib.translate import federal_register
|
from bib.translate import federal_register
|
||||||
|
|
||||||
store = connect()
|
store = connect()
|
||||||
rules = pfs_rules(since=since, until=until or None)
|
rules = pfs_rules(since=since, until=until or None)
|
||||||
typer.echo(f"==> {len(rules)} PFS rules since {since}")
|
typer.echo(f"==> {len(rules)} PFS rules since {since}")
|
||||||
|
|
||||||
# Only proposed rules have public comment periods — skip finals +
|
|
||||||
# corrections to keep the work focused.
|
|
||||||
proposed = [r for r in rules if r.type == "Proposed Rule"]
|
proposed = [r for r in rules if r.type == "Proposed Rule"]
|
||||||
typer.echo(f" {len(proposed)} proposed rules (the ones with comments)")
|
typer.echo(f" {len(proposed)} proposed rules (the ones with comments)")
|
||||||
|
|
||||||
with Client(sleep=sleep) as api:
|
with Client(sleep=sleep) as api:
|
||||||
for doc in proposed:
|
for doc in proposed:
|
||||||
cms_ids = split_docket_ids(doc.dockets)
|
cms_ids = split_docket_ids(doc.dockets)
|
||||||
|
known = {cid: store.docket_for_rule(cid) for cid in cms_ids}
|
||||||
|
|
||||||
|
# Decide what this rule needs BEFORE touching the network.
|
||||||
|
def _wanted(cid: str) -> bool:
|
||||||
|
d = known[cid]
|
||||||
|
if docket and d is not None and d.id != docket:
|
||||||
|
return False
|
||||||
|
if d is not None and d.sealed and not force:
|
||||||
|
typer.echo(
|
||||||
|
f" {d.id}: sealed {d.sealed_at[:10]} ({d.seal_reason}); skipping"
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
todo = [cid for cid in cms_ids if _wanted(cid)]
|
||||||
|
if not todo:
|
||||||
|
continue
|
||||||
|
|
||||||
try:
|
try:
|
||||||
rule = federal_register(
|
rule = federal_register(
|
||||||
doc.html_url
|
doc.html_url
|
||||||
@@ -244,58 +279,33 @@ def fetch_pfs_comments(
|
|||||||
rule.add_tag("module:pfs")
|
rule.add_tag("module:pfs")
|
||||||
for cid in cms_ids:
|
for cid in cms_ids:
|
||||||
rule.add_tag(f"cms-rule:{cid}")
|
rule.add_tag(f"cms-rule:{cid}")
|
||||||
store.upsert(rule)
|
|
||||||
|
|
||||||
# Resolve each CMS-XXXX-P to its reg.gov docket id.
|
for cms_id in todo:
|
||||||
for cms_id in cms_ids:
|
d = known[cms_id]
|
||||||
reg_docket = api.resolve_docket(cms_id)
|
if d is None:
|
||||||
if not reg_docket:
|
reg_docket = api.resolve_docket(cms_id)
|
||||||
typer.echo(f" skip {cms_id}: no reg.gov docket found")
|
if not reg_docket:
|
||||||
continue
|
typer.echo(f" skip {cms_id}: no reg.gov docket found")
|
||||||
if docket and reg_docket != docket:
|
continue
|
||||||
continue
|
if docket and reg_docket != docket:
|
||||||
typer.echo(f" {cms_id} → {reg_docket} ({doc.publication_date})")
|
continue
|
||||||
rule.add_tag(f"reg-docket:{reg_docket}")
|
d = discover_docket(api, reg_docket, rule_cms_id=cms_id)
|
||||||
|
store.docket_upsert(d)
|
||||||
|
typer.echo(f" {cms_id} → {d.id} ({doc.publication_date})")
|
||||||
|
rule.add_tag(f"reg-docket:{d.id}")
|
||||||
store.upsert(rule)
|
store.upsert(rule)
|
||||||
|
if not d.fr_object_id:
|
||||||
fr_docs = api.find_documents_in_docket(reg_docket)
|
typer.echo(f" {d.id}: no commentable FR document; skipping")
|
||||||
per_docket_count = 0
|
continue
|
||||||
scratch = Path(f".state/comments/{reg_docket}")
|
_walk_and_report(
|
||||||
for fr_doc in fr_docs:
|
store,
|
||||||
attrs = fr_doc.get("attributes") or {}
|
api,
|
||||||
object_id = attrs.get("objectId")
|
d,
|
||||||
# Skip docs with no objectId or no real comment
|
attachments=attachments,
|
||||||
# window — final rules and internal display versions
|
limit=per_docket_limit,
|
||||||
# don't carry meaningful comment traffic.
|
force=force,
|
||||||
if not object_id:
|
echo=typer.echo,
|
||||||
continue
|
)
|
||||||
if not attrs.get("commentEndDate"):
|
|
||||||
continue
|
|
||||||
for c in api.iter_comments(object_id):
|
|
||||||
if per_docket_limit and per_docket_count >= per_docket_limit:
|
|
||||||
break
|
|
||||||
key = upsert_comment(
|
|
||||||
store,
|
|
||||||
c,
|
|
||||||
cms_id=cms_id,
|
|
||||||
extra_tags=[f"reg-docket:{reg_docket}"],
|
|
||||||
)
|
|
||||||
if attachments and c.attachment_count:
|
|
||||||
for att in api.attachments_for(c.id):
|
|
||||||
path = api.download_attachment(
|
|
||||||
att.url,
|
|
||||||
scratch / c.id,
|
|
||||||
)
|
|
||||||
if path:
|
|
||||||
store.attach_file(key, path, title=att.filename)
|
|
||||||
per_docket_count += 1
|
|
||||||
if per_docket_count % 50 == 0:
|
|
||||||
typer.echo(f" {per_docket_count} comments")
|
|
||||||
store._con().commit() # noqa: SLF001
|
|
||||||
if per_docket_limit and per_docket_count >= per_docket_limit:
|
|
||||||
break
|
|
||||||
store._con().commit() # noqa: SLF001
|
|
||||||
typer.echo(f" docket total: {per_docket_count}")
|
|
||||||
|
|
||||||
|
|
||||||
@app.command(name="ingest-mail")
|
@app.command(name="ingest-mail")
|
||||||
@@ -378,6 +388,14 @@ def backfill_comments(
|
|||||||
"reg.gov API — bulk speed, no rate cap, also ingests comments "
|
"reg.gov API — bulk speed, no rate cap, also ingests comments "
|
||||||
"the original farm never listed. Requires --docket.",
|
"the original farm never listed. Requires --docket.",
|
||||||
),
|
),
|
||||||
|
force: bool = typer.Option(
|
||||||
|
False,
|
||||||
|
"--force",
|
||||||
|
help="Run even if the docket is sealed. Bypasses the docket seal "
|
||||||
|
"only: items already tagged enriched:ok (or enriched:gone) are "
|
||||||
|
"still skipped, so this resumes an interrupted enrichment — it "
|
||||||
|
"never re-enriches an item.",
|
||||||
|
),
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Enrich every reg-gov comment stub with body + attachments.
|
"""Enrich every reg-gov comment stub with body + attachments.
|
||||||
|
|
||||||
@@ -403,6 +421,7 @@ def backfill_comments(
|
|||||||
store,
|
store,
|
||||||
Mirror(),
|
Mirror(),
|
||||||
docket=docket,
|
docket=docket,
|
||||||
|
force=force,
|
||||||
limit=limit or None,
|
limit=limit or None,
|
||||||
log_path=log_path,
|
log_path=log_path,
|
||||||
)
|
)
|
||||||
@@ -414,6 +433,7 @@ def backfill_comments(
|
|||||||
store,
|
store,
|
||||||
api,
|
api,
|
||||||
docket=docket,
|
docket=docket,
|
||||||
|
force=force,
|
||||||
limit=limit or None,
|
limit=limit or None,
|
||||||
log_path=log_path,
|
log_path=log_path,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -7,6 +7,9 @@ Phase 1 (this file):
|
|||||||
extract — write combined.md per comment (PDF/DOCX → text)
|
extract — write combined.md per comment (PDF/DOCX → text)
|
||||||
extract-ocr — phase-2 stub (raises until tesseract integration lands)
|
extract-ocr — phase-2 stub (raises until tesseract integration lands)
|
||||||
stats — count comments by extraction status
|
stats — count comments by extraction status
|
||||||
|
dockets — list known dockets (optionally discover new ones)
|
||||||
|
seal — mark a docket complete so later stages skip it
|
||||||
|
unseal — reopen a sealed docket
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -28,54 +31,63 @@ NOTE_TITLE = "Comment text"
|
|||||||
|
|
||||||
|
|
||||||
def _build_bib_helpers(use_bib: bool, attach: bool):
|
def _build_bib_helpers(use_bib: bool, attach: bool):
|
||||||
"""Return ``(body_lookup, attach_callback)``.
|
"""Return ``(body_lookup, attach_callback, store_or_None)``.
|
||||||
|
|
||||||
``body_lookup(comment_id) -> str`` is called from worker threads and
|
``body_lookup`` queries bib per comment id (``items.url`` is indexed)
|
||||||
returns the inline comment body from bib.sqlite (or "" if missing or
|
under a lock — worker threads share one connection. Nothing is
|
||||||
--no-bib). Backed by a comment_id → (item_key, body) cache built
|
loaded up front, so a run that extracts nothing reads nothing.
|
||||||
once at startup — bib has 164k+ rows and the URL → comment_id parse
|
|
||||||
can't use any index, so we materialize the whole map up front.
|
|
||||||
|
|
||||||
``attach_callback(comment_id, comment_dir) -> None`` runs on the main
|
``attach_callback(comment_id, comment_dir) -> None`` runs on the main
|
||||||
thread and registers a single ``"Comment text"`` note on the bib item
|
thread and registers a single ``"Comment text"`` note on the bib item
|
||||||
containing combined.md rendered to HTML. Skips items that already have
|
containing combined.md rendered to HTML. Skips items that already have
|
||||||
one. Returns None when --no-bib or --no-attach.
|
one. ``attach_callback`` is ``None`` for --no-bib or --no-attach; the
|
||||||
|
store is ``None`` only for --no-bib, so ``extract`` can still read
|
||||||
|
seals with --no-attach.
|
||||||
"""
|
"""
|
||||||
if not use_bib:
|
if not use_bib:
|
||||||
return lambda _cid: "", None
|
return (lambda _cid: ""), None, None
|
||||||
|
|
||||||
import sqlite3
|
import sqlite3
|
||||||
|
import threading
|
||||||
from conf import path
|
|
||||||
|
|
||||||
db = str(path("db.bib"))
|
|
||||||
con = sqlite3.connect(db, check_same_thread=False)
|
|
||||||
try:
|
|
||||||
cache: dict[str, tuple[str, str]] = {}
|
|
||||||
for row in con.execute(
|
|
||||||
"SELECT key, url, abstract FROM items "
|
|
||||||
"WHERE url LIKE 'https://www.regulations.gov/comment/%'"
|
|
||||||
):
|
|
||||||
cid = (row[1] or "").rsplit("/", 1)[-1]
|
|
||||||
if cid:
|
|
||||||
cache[cid] = (row[0], row[2] or "")
|
|
||||||
finally:
|
|
||||||
con.close()
|
|
||||||
|
|
||||||
def body_lookup(comment_id: str) -> str:
|
|
||||||
return cache.get(comment_id, ("", ""))[1]
|
|
||||||
|
|
||||||
if not attach:
|
|
||||||
return body_lookup, None
|
|
||||||
|
|
||||||
from bib import connect
|
from bib import connect
|
||||||
from rex.comments.render import render_combined_md
|
|
||||||
|
|
||||||
store = connect()
|
store = connect()
|
||||||
|
own_con = False
|
||||||
|
if store._db_path == ":memory:": # noqa: SLF001
|
||||||
|
con = store._con() # noqa: SLF001 — tests only: a fresh connection would be an empty DB
|
||||||
|
else:
|
||||||
|
con = sqlite3.connect(store._db_path, check_same_thread=False) # noqa: SLF001
|
||||||
|
con.row_factory = sqlite3.Row
|
||||||
|
own_con = True
|
||||||
|
lock = threading.Lock()
|
||||||
|
|
||||||
|
def _row(comment_id: str):
|
||||||
|
with lock:
|
||||||
|
return con.execute(
|
||||||
|
"SELECT key, abstract FROM items WHERE url = ?",
|
||||||
|
(f"https://www.regulations.gov/comment/{comment_id}",),
|
||||||
|
).fetchone()
|
||||||
|
|
||||||
|
def body_lookup(comment_id: str) -> str:
|
||||||
|
row = _row(comment_id)
|
||||||
|
return (row["abstract"] or "") if row else ""
|
||||||
|
|
||||||
|
# extract() closes this on our behalf when it opened a private
|
||||||
|
# read connection (the file-backed branch above) — the :memory:
|
||||||
|
# connection is store's own and must outlive this call.
|
||||||
|
if own_con:
|
||||||
|
body_lookup.close = con.close
|
||||||
|
|
||||||
|
if not attach:
|
||||||
|
return body_lookup, None, store
|
||||||
|
|
||||||
|
from rex.comments.render import render_combined_md
|
||||||
|
|
||||||
# Items that already have a NOTE_TITLE note — keeps re-runs idempotent.
|
# Items that already have a NOTE_TITLE note — keeps re-runs idempotent.
|
||||||
items_with_note: set[str] = {
|
items_with_note: set[str] = {
|
||||||
row[0]
|
row[0]
|
||||||
for row in store._con().execute(
|
for row in con.execute(
|
||||||
"SELECT i.key FROM notes n "
|
"SELECT i.key FROM notes n "
|
||||||
"JOIN items i ON n.item_id = i.id "
|
"JOIN items i ON n.item_id = i.id "
|
||||||
"WHERE n.title = ?",
|
"WHERE n.title = ?",
|
||||||
@@ -84,10 +96,10 @@ def _build_bib_helpers(use_bib: bool, attach: bool):
|
|||||||
}
|
}
|
||||||
|
|
||||||
def attach_callback(comment_id: str, comment_dir: Path) -> None:
|
def attach_callback(comment_id: str, comment_dir: Path) -> None:
|
||||||
info = cache.get(comment_id)
|
row = _row(comment_id)
|
||||||
if not info:
|
if not row:
|
||||||
return # comment not in bib
|
return # comment not in bib
|
||||||
item_key = info[0]
|
item_key = row["key"]
|
||||||
if item_key in items_with_note:
|
if item_key in items_with_note:
|
||||||
return
|
return
|
||||||
combined = comment_dir / "combined.md"
|
combined = comment_dir / "combined.md"
|
||||||
@@ -95,17 +107,18 @@ def _build_bib_helpers(use_bib: bool, attach: bool):
|
|||||||
return # extraction must have failed; nothing to attach
|
return # extraction must have failed; nothing to attach
|
||||||
try:
|
try:
|
||||||
html = render_combined_md(combined.read_text(encoding="utf-8"))
|
html = render_combined_md(combined.read_text(encoding="utf-8"))
|
||||||
store.attach_note(item_key, html, title=NOTE_TITLE)
|
with lock:
|
||||||
|
store.attach_note(item_key, html, title=NOTE_TITLE)
|
||||||
items_with_note.add(item_key)
|
items_with_note.add(item_key)
|
||||||
except Exception as e: # noqa: BLE001
|
except Exception as e: # noqa: BLE001
|
||||||
log.warning("attach_note failed for %s: %s", item_key, e)
|
log.warning("attach_note failed for %s: %s", item_key, e)
|
||||||
|
|
||||||
return body_lookup, attach_callback
|
return body_lookup, attach_callback, store
|
||||||
|
|
||||||
|
|
||||||
# Back-compat alias — older code/tests may still import this name.
|
# Back-compat alias — older code/tests may still import this name.
|
||||||
def _bib_lookup_factory(use_bib: bool):
|
def _bib_lookup_factory(use_bib: bool):
|
||||||
body, _ = _build_bib_helpers(use_bib, attach=False)
|
body, _, _ = _build_bib_helpers(use_bib, attach=False)
|
||||||
return body
|
return body
|
||||||
|
|
||||||
|
|
||||||
@@ -133,6 +146,12 @@ def extract(
|
|||||||
help="Attach combined.md back to its bib item (so it flows to "
|
help="Attach combined.md back to its bib item (so it flows to "
|
||||||
"Zotero via `bib sync`). Implies --bib; no-op when --no-bib.",
|
"Zotero via `bib sync`). Implies --bib; no-op when --no-bib.",
|
||||||
),
|
),
|
||||||
|
reattach: bool = typer.Option(
|
||||||
|
False,
|
||||||
|
"--reattach",
|
||||||
|
help="Also attach combined.md notes for dirs that were already "
|
||||||
|
"current (repair).",
|
||||||
|
),
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Walk comment dirs and write combined.md (PDF/DOCX → text)."""
|
"""Walk comment dirs and write combined.md (PDF/DOCX → text)."""
|
||||||
from rex.comments.walker import walk_and_extract
|
from rex.comments.walker import walk_and_extract
|
||||||
@@ -141,16 +160,28 @@ def extract(
|
|||||||
typer.echo(f"root not found: {root}", err=True)
|
typer.echo(f"root not found: {root}", err=True)
|
||||||
raise typer.Exit(1)
|
raise typer.Exit(1)
|
||||||
|
|
||||||
lookup, attach_cb = _build_bib_helpers(use_bib, attach=use_bib and attach)
|
lookup, attach_cb, store = _build_bib_helpers(use_bib, attach=use_bib and attach)
|
||||||
stats = walk_and_extract(
|
try:
|
||||||
root,
|
skip = (
|
||||||
inline_body_lookup=lookup,
|
set(store.sealed_dockets()) if (store is not None and not force) else set()
|
||||||
docket=docket,
|
)
|
||||||
limit=limit or None,
|
stats = walk_and_extract(
|
||||||
workers=workers,
|
root,
|
||||||
force=force,
|
inline_body_lookup=lookup,
|
||||||
on_extracted=attach_cb,
|
docket=docket,
|
||||||
)
|
limit=limit or None,
|
||||||
|
workers=workers,
|
||||||
|
force=force,
|
||||||
|
on_extracted=attach_cb,
|
||||||
|
skip_dockets=skip,
|
||||||
|
reattach=reattach,
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
close = getattr(lookup, "close", None)
|
||||||
|
if close is not None:
|
||||||
|
close()
|
||||||
|
if docket and docket in skip:
|
||||||
|
typer.echo(f" {docket}: sealed; skipped — use --force")
|
||||||
for k, v in stats.items():
|
for k, v in stats.items():
|
||||||
typer.echo(f" {k}: {v}")
|
typer.echo(f" {k}: {v}")
|
||||||
|
|
||||||
@@ -254,3 +285,86 @@ def stats(
|
|||||||
cols = ["comments", "attachments", "ok", "ocr_needed", "failed"]
|
cols = ["comments", "attachments", "ok", "ocr_needed", "failed"]
|
||||||
for k, v in zip(cols, rows, strict=True):
|
for k, v in zip(cols, rows, strict=True):
|
||||||
typer.echo(f" {k}: {v or 0}")
|
typer.echo(f" {k}: {v or 0}")
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def dockets(
|
||||||
|
discover: bool = typer.Option(
|
||||||
|
False,
|
||||||
|
"--discover",
|
||||||
|
help="Populate rows for every reg-docket: tag in bib that has no "
|
||||||
|
"dockets row (one reg.gov listing call each).",
|
||||||
|
),
|
||||||
|
) -> None:
|
||||||
|
"""List dockets: id, rule, close date, watermark, counts, seal."""
|
||||||
|
from bib import connect
|
||||||
|
from bib.regulations_gov import docket_counts
|
||||||
|
|
||||||
|
store = connect()
|
||||||
|
if discover:
|
||||||
|
from bib.regulations_gov import Client, _rule_tag_for_docket, discover_docket
|
||||||
|
|
||||||
|
tagged = [
|
||||||
|
r[0].split(":", 1)[1]
|
||||||
|
for r in store._con().execute( # noqa: SLF001
|
||||||
|
"SELECT name FROM tags WHERE name LIKE 'reg-docket:%' ORDER BY name"
|
||||||
|
)
|
||||||
|
]
|
||||||
|
missing = [d for d in tagged if store.docket_get(d) is None]
|
||||||
|
with Client(sleep=1.3) as api:
|
||||||
|
for d in missing:
|
||||||
|
row = discover_docket(
|
||||||
|
api, d, rule_cms_id=_rule_tag_for_docket(store, d)
|
||||||
|
)
|
||||||
|
store.docket_upsert(row)
|
||||||
|
typer.echo(
|
||||||
|
f" discovered {d}: closes {row.comment_end_date or '?'} "
|
||||||
|
f"rule {row.rule_cms_id or '?'}"
|
||||||
|
)
|
||||||
|
rows = store.dockets()
|
||||||
|
if not rows:
|
||||||
|
typer.echo("no dockets recorded — run `stack comments dockets --discover`")
|
||||||
|
return
|
||||||
|
typer.echo(
|
||||||
|
f"{'docket':<16}{'rule':<12}{'closes':<12}{'comments':>9}{'enriched':>9} "
|
||||||
|
f"{'watermark':<20} state"
|
||||||
|
)
|
||||||
|
for d in rows:
|
||||||
|
c = docket_counts(store, d.id)
|
||||||
|
state = f"sealed {d.sealed_at[:10]} ({d.seal_reason})" if d.sealed else "open"
|
||||||
|
typer.echo(
|
||||||
|
f"{d.id:<16}{d.rule_cms_id:<12}{d.comment_end_date:<12}"
|
||||||
|
f"{c['comments']:>9}{c['enriched']:>9} {d.pull_watermark[:19]:<20} {state}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def seal(
|
||||||
|
docket: str = typer.Argument(..., help="reg.gov docket id"),
|
||||||
|
reason: str = typer.Option("manual", "--reason"),
|
||||||
|
) -> None:
|
||||||
|
"""Mark a docket complete: every stage skips it until unsealed or --force."""
|
||||||
|
from bib import connect
|
||||||
|
from bib.regulations_gov import docket_counts
|
||||||
|
|
||||||
|
store = connect()
|
||||||
|
if store.docket_get(docket) is None:
|
||||||
|
typer.echo(
|
||||||
|
f"unknown docket {docket} — run `stack comments dockets --discover` first"
|
||||||
|
)
|
||||||
|
raise typer.Exit(1)
|
||||||
|
store.docket_seal(docket, reason=reason, counts=docket_counts(store, docket))
|
||||||
|
typer.echo(f"sealed {docket} ({reason})")
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def unseal(docket: str = typer.Argument(..., help="reg.gov docket id")) -> None:
|
||||||
|
"""Reopen a sealed docket."""
|
||||||
|
from bib import connect
|
||||||
|
|
||||||
|
store = connect()
|
||||||
|
if store.docket_get(docket) is None:
|
||||||
|
typer.echo(f"unknown docket {docket}")
|
||||||
|
raise typer.Exit(1)
|
||||||
|
store.docket_unseal(docket)
|
||||||
|
typer.echo(f"unsealed {docket}")
|
||||||
|
|||||||
@@ -7,24 +7,26 @@ app = typer.Typer(no_args_is_help=True)
|
|||||||
_COLLECTIONS = ("comments", "rules", "corpus")
|
_COLLECTIONS = ("comments", "rules", "corpus")
|
||||||
|
|
||||||
|
|
||||||
def _docs_for(collection: str, store, docket: str, keys: tuple[str, ...]):
|
def _refs_for(
|
||||||
|
collection: str, store, docket: str, keys: tuple[str, ...], skip_dockets: set[str]
|
||||||
|
):
|
||||||
from llm.source import (
|
from llm.source import (
|
||||||
ZoteroPdfIndex,
|
ZoteroPdfIndex,
|
||||||
iter_comment_docs,
|
iter_comment_refs,
|
||||||
iter_corpus_docs,
|
iter_corpus_refs,
|
||||||
iter_rule_docs,
|
iter_rule_refs,
|
||||||
)
|
)
|
||||||
|
|
||||||
if collection == "comments":
|
if collection == "comments":
|
||||||
return iter_comment_docs(store, docket=docket)
|
return iter_comment_refs(store, docket=docket, skip_dockets=skip_dockets)
|
||||||
if collection == "rules":
|
if collection == "rules":
|
||||||
return iter_rule_docs(store, keys=keys)
|
return iter_rule_refs(store, keys=keys)
|
||||||
from conf import ROOT, path
|
from conf import ROOT, path
|
||||||
|
|
||||||
zotero = ZoteroPdfIndex.snapshot(
|
zotero = ZoteroPdfIndex.lazy(
|
||||||
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
|
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
|
||||||
)
|
)
|
||||||
return iter_corpus_docs(store, zotero=zotero)
|
return iter_corpus_refs(store, zotero=zotero)
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
@app.command()
|
||||||
@@ -47,7 +49,8 @@ def index(
|
|||||||
|
|
||||||
from conf.connect import bib
|
from conf.connect import bib
|
||||||
from llm import config as llm_config
|
from llm import config as llm_config
|
||||||
from llm.index import index_docs
|
from llm.index import _engine, docket_complete, index_refs
|
||||||
|
from llm.migrate import migrate
|
||||||
from llm.pool import HostPool
|
from llm.pool import HostPool
|
||||||
|
|
||||||
targets = _COLLECTIONS if collection == "all" else (collection,)
|
targets = _COLLECTIONS if collection == "all" else (collection,)
|
||||||
@@ -55,20 +58,43 @@ def index(
|
|||||||
raise typer.BadParameter("collection must be comments, rules, corpus or all")
|
raise typer.BadParameter("collection must be comments, rules, corpus or all")
|
||||||
cfg = llm_config.load()
|
cfg = llm_config.load()
|
||||||
store = bib()
|
store = bib()
|
||||||
|
sealed_all = {} if force else store.sealed_dockets()
|
||||||
|
# One engine for the whole run, migrated *before* anything reads
|
||||||
|
# index_docket_state — that table does not exist until migrate() has
|
||||||
|
# run, and the first sealed run on an un-migrated DB gets here first.
|
||||||
|
engine = None
|
||||||
|
if sealed_all and "comments" in targets:
|
||||||
|
engine = _engine(cfg)
|
||||||
|
migrate(engine)
|
||||||
for target in targets:
|
for target in targets:
|
||||||
docs = _docs_for(target, store, docket, tuple(key))
|
complete = (
|
||||||
|
docket_complete(engine, target)
|
||||||
|
if (engine is not None and target == "comments")
|
||||||
|
else {}
|
||||||
|
)
|
||||||
|
skip = {d for d, s in sealed_all.items() if complete.get(d) == s}
|
||||||
|
pending_seals = {d: s for d, s in sealed_all.items() if d not in skip}
|
||||||
|
refs = _refs_for(target, store, docket, tuple(key), skip)
|
||||||
if limit:
|
if limit:
|
||||||
docs = itertools.islice(docs, limit)
|
refs = itertools.islice(refs, limit)
|
||||||
stats = index_docs(
|
stats = index_refs(
|
||||||
docs,
|
refs,
|
||||||
collection=target,
|
collection=target,
|
||||||
cfg=cfg,
|
cfg=cfg,
|
||||||
pool=HostPool.from_config(cfg),
|
pool=HostPool.from_config(cfg),
|
||||||
force=force,
|
force=force,
|
||||||
|
sealed=pending_seals if target == "comments" else None,
|
||||||
|
mark_complete=not limit,
|
||||||
|
engine=engine,
|
||||||
)
|
)
|
||||||
|
if skip:
|
||||||
|
typer.echo(
|
||||||
|
f"{target}: {len(skip)} sealed docket(s) already complete — not listed"
|
||||||
|
)
|
||||||
typer.echo(
|
typer.echo(
|
||||||
f"{target}: indexed={stats['indexed']} skipped={stats['skipped']} "
|
f"{target}: indexed={stats['indexed']} skipped={stats['skipped']} chunks={stats['chunks']} "
|
||||||
f"chunks={stats['chunks']}"
|
f"fp_skipped={stats['fingerprint_skipped']} hash_skipped={stats['hash_skipped']} "
|
||||||
|
f"docket_complete={stats['docket_complete']}"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
231
src/llm/index.py
231
src/llm/index.py
@@ -31,6 +31,7 @@ from llm.config import LlmConfig, pg_url
|
|||||||
from llm.migrate import ensure_hnsw, migrate
|
from llm.migrate import ensure_hnsw, migrate
|
||||||
from llm.pages import enrich_pdf_pages
|
from llm.pages import enrich_pdf_pages
|
||||||
from llm.pool import HostPool, PoolEmbeddings, embed_texts
|
from llm.pool import HostPool, PoolEmbeddings, embed_texts
|
||||||
|
from llm.source import DocRef
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -51,15 +52,28 @@ def vectorstore(collection: str, cfg: LlmConfig, pool: HostPool):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _state(engine: Engine, collection: str) -> dict[str, str]:
|
def _state(engine: Engine, collection: str) -> dict[str, tuple[str, str]]:
|
||||||
with engine.begin() as conn:
|
with engine.begin() as conn:
|
||||||
rows = conn.execute(
|
rows = conn.execute(
|
||||||
text(
|
text(
|
||||||
"SELECT item_key, content_hash FROM index_state WHERE collection = :c"
|
"SELECT item_key, content_hash, fingerprint FROM index_state "
|
||||||
|
"WHERE collection = :c"
|
||||||
),
|
),
|
||||||
{"c": collection},
|
{"c": collection},
|
||||||
).fetchall()
|
).fetchall()
|
||||||
return dict(rows)
|
return {r[0]: (r[1], r[2] or "") for r in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def docket_complete(engine: Engine, collection: str) -> dict[str, str]:
|
||||||
|
"""docket → sealed_at for dockets fully indexed under that seal."""
|
||||||
|
with engine.begin() as conn:
|
||||||
|
rows = conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT docket, sealed_at FROM index_docket_state WHERE collection = :c"
|
||||||
|
),
|
||||||
|
{"c": collection},
|
||||||
|
).fetchall()
|
||||||
|
return {r[0]: r[1] for r in rows}
|
||||||
|
|
||||||
|
|
||||||
def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None:
|
def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None:
|
||||||
@@ -87,6 +101,154 @@ def _delete_old_chunks(engine: Engine, item_key: str, collection: str) -> None:
|
|||||||
log.debug("langchain_pg_embedding not present yet; skipping delete")
|
log.debug("langchain_pg_embedding not present yet; skipping delete")
|
||||||
|
|
||||||
|
|
||||||
|
def _record_state(
|
||||||
|
engine: Engine, key: str, collection: str, h: str, n: int, fingerprint: str
|
||||||
|
) -> None:
|
||||||
|
"""Upsert this item's row in ``index_state``.
|
||||||
|
|
||||||
|
``h == "" and n == 0`` is the *stamped-final* form: an item with no
|
||||||
|
text at all. It is a finished state, not a failure — no chunks exist
|
||||||
|
to delete or embed — and the empty hash can never collide with a real
|
||||||
|
``content_hash`` (a sha256 hexdigest), so the next run recognises it
|
||||||
|
by fingerprint alone and never loads it again. If text does appear
|
||||||
|
later the fingerprint changes and the item is embedded normally.
|
||||||
|
"""
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"INSERT INTO index_state "
|
||||||
|
"(item_key, collection, content_hash, chunk_count, fingerprint) "
|
||||||
|
"VALUES (:k, :c, :h, :n, :f) "
|
||||||
|
"ON CONFLICT (item_key, collection) DO UPDATE SET "
|
||||||
|
"content_hash = :h, chunk_count = :n, fingerprint = :f, "
|
||||||
|
"indexed_at = now()"
|
||||||
|
),
|
||||||
|
{"k": key, "c": collection, "h": h, "n": n, "f": fingerprint},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def index_refs(
|
||||||
|
refs: Iterable[DocRef],
|
||||||
|
*,
|
||||||
|
collection: str,
|
||||||
|
cfg: LlmConfig,
|
||||||
|
pool: HostPool,
|
||||||
|
force: bool = False,
|
||||||
|
sealed: dict[str, str] | None = None,
|
||||||
|
mark_complete: bool = True,
|
||||||
|
engine: Engine | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""Fingerprint-first incremental indexing.
|
||||||
|
|
||||||
|
Per ref: (1) fingerprint equal to the stored one → skip without
|
||||||
|
loading; (2) load, hash the text; hash equal → record the new
|
||||||
|
fingerprint, skip embedding; (3) chunk, locate PDF pages, embed,
|
||||||
|
write. PDFs are opened only on path (3). A ref that loads blank is
|
||||||
|
*stamped final* (see :func:`_record_state`) rather than treated as a
|
||||||
|
failure — plenty of real comments carry no text at all.
|
||||||
|
|
||||||
|
After the loop, a docket in *sealed* is recorded in
|
||||||
|
``index_docket_state`` — so the next run does not list it at all —
|
||||||
|
only when every one of its refs was accounted for. A ref that loaded
|
||||||
|
*with* text but chunked to nothing is *unindexable* and blocks
|
||||||
|
completion, because a seal must never claim coverage this run did not
|
||||||
|
achieve (an exception on the embed/write path aborts the run and so
|
||||||
|
blocks it too). *mark_complete* False (a ``--limit`` run is never
|
||||||
|
complete) blocks it as well. Pass *engine* to reuse a caller's engine
|
||||||
|
instead of building a second one.
|
||||||
|
"""
|
||||||
|
engine = engine if engine is not None else _engine(cfg)
|
||||||
|
migrate(engine)
|
||||||
|
pool.check(cfg.embed_model)
|
||||||
|
store = vectorstore(collection, cfg, pool)
|
||||||
|
seen = _state(engine, collection)
|
||||||
|
stats = {
|
||||||
|
"indexed": 0,
|
||||||
|
"skipped": 0,
|
||||||
|
"chunks": 0,
|
||||||
|
"fingerprint_skipped": 0,
|
||||||
|
"hash_skipped": 0,
|
||||||
|
"docket_complete": 0,
|
||||||
|
}
|
||||||
|
dockets_seen: set[str] = set()
|
||||||
|
unindexable: dict[str, int] = {}
|
||||||
|
|
||||||
|
def _unindexable(ref: DocRef) -> None:
|
||||||
|
stats["skipped"] += 1
|
||||||
|
if ref.docket:
|
||||||
|
unindexable[ref.docket] = unindexable.get(ref.docket, 0) + 1
|
||||||
|
|
||||||
|
for ref in refs:
|
||||||
|
if ref.docket:
|
||||||
|
dockets_seen.add(ref.docket)
|
||||||
|
prev_hash, prev_fp = seen.get(ref.key, ("", ""))
|
||||||
|
if not force and ref.fingerprint and ref.fingerprint == prev_fp:
|
||||||
|
stats["fingerprint_skipped"] += 1
|
||||||
|
continue
|
||||||
|
doc = ref.load()
|
||||||
|
if doc is None or not doc.text.strip():
|
||||||
|
_record_state(engine, ref.key, collection, "", 0, ref.fingerprint)
|
||||||
|
stats["skipped"] += 1
|
||||||
|
continue
|
||||||
|
h = content_hash(doc.text)
|
||||||
|
if not force and prev_hash == h:
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"UPDATE index_state SET fingerprint = :f "
|
||||||
|
"WHERE item_key = :k AND collection = :c"
|
||||||
|
),
|
||||||
|
{"f": ref.fingerprint, "k": ref.key, "c": collection},
|
||||||
|
)
|
||||||
|
stats["hash_skipped"] += 1
|
||||||
|
continue
|
||||||
|
chunks = enrich_pdf_pages(doc, chunk_doc(doc))
|
||||||
|
if not chunks:
|
||||||
|
_unindexable(ref)
|
||||||
|
continue
|
||||||
|
vectors = embed_texts(pool, cfg.embed_model, [c.text for c in chunks])
|
||||||
|
_delete_old_chunks(engine, ref.key, collection)
|
||||||
|
store.add_embeddings(
|
||||||
|
texts=[c.text for c in chunks],
|
||||||
|
embeddings=vectors,
|
||||||
|
metadatas=[c.metadata for c in chunks],
|
||||||
|
ids=[c.id for c in chunks],
|
||||||
|
)
|
||||||
|
_record_state(engine, ref.key, collection, h, len(chunks), ref.fingerprint)
|
||||||
|
stats["indexed"] += 1
|
||||||
|
stats["chunks"] += len(chunks)
|
||||||
|
# needs 100+ docs in one run to hit this line
|
||||||
|
if stats["indexed"] % 100 == 0:
|
||||||
|
log.info("indexed %(indexed)s (+%(chunks)s)", stats) # pragma: no cover
|
||||||
|
|
||||||
|
if mark_complete and sealed:
|
||||||
|
for d in sorted(dockets_seen & set(sealed)):
|
||||||
|
if unindexable.get(d):
|
||||||
|
log.info(
|
||||||
|
"docket %s not marked complete: %s unindexable item(s)",
|
||||||
|
d,
|
||||||
|
unindexable[d],
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"INSERT INTO index_docket_state (collection, docket, sealed_at) "
|
||||||
|
"VALUES (:c, :d, :s) "
|
||||||
|
"ON CONFLICT (collection, docket) DO UPDATE SET "
|
||||||
|
"sealed_at = :s, indexed_at = now()"
|
||||||
|
),
|
||||||
|
{"c": collection, "d": d, "s": sealed[d]},
|
||||||
|
)
|
||||||
|
stats["docket_complete"] += 1
|
||||||
|
|
||||||
|
if cfg.build_ann_index:
|
||||||
|
ensure_hnsw(engine, cfg.embed_dim)
|
||||||
|
else:
|
||||||
|
log.info("ANN index skipped (build_ann_index=false); using exact search")
|
||||||
|
return stats
|
||||||
|
|
||||||
|
|
||||||
def index_docs(
|
def index_docs(
|
||||||
docs: Iterable[Doc],
|
docs: Iterable[Doc],
|
||||||
*,
|
*,
|
||||||
@@ -95,49 +257,22 @@ def index_docs(
|
|||||||
pool: HostPool,
|
pool: HostPool,
|
||||||
force: bool = False,
|
force: bool = False,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
engine = _engine(cfg)
|
"""Back-compat wrapper: Docs without fingerprints (never fingerprint-skipped)."""
|
||||||
migrate(engine)
|
refs = (
|
||||||
pool.check(cfg.embed_model)
|
DocRef(
|
||||||
store = vectorstore(collection, cfg, pool)
|
key=d.key,
|
||||||
seen = _state(engine, collection)
|
collection=collection,
|
||||||
stats = {"indexed": 0, "skipped": 0, "chunks": 0}
|
docket=d.metadata.get("docket") or None,
|
||||||
|
fingerprint="",
|
||||||
for doc in docs:
|
load=(lambda d=d: d),
|
||||||
chunks = enrich_pdf_pages(doc, chunk_doc(doc))
|
|
||||||
if not chunks:
|
|
||||||
stats["skipped"] += 1
|
|
||||||
continue
|
|
||||||
h = content_hash(doc.text)
|
|
||||||
if not force and seen.get(doc.key) == h:
|
|
||||||
stats["skipped"] += 1
|
|
||||||
continue
|
|
||||||
vectors = embed_texts(pool, cfg.embed_model, [c.text for c in chunks])
|
|
||||||
_delete_old_chunks(engine, doc.key, collection)
|
|
||||||
store.add_embeddings(
|
|
||||||
texts=[c.text for c in chunks],
|
|
||||||
embeddings=vectors,
|
|
||||||
metadatas=[c.metadata for c in chunks],
|
|
||||||
ids=[c.id for c in chunks],
|
|
||||||
)
|
)
|
||||||
with engine.begin() as conn:
|
for d in docs
|
||||||
conn.execute(
|
)
|
||||||
text(
|
stats = index_refs(
|
||||||
"INSERT INTO index_state "
|
refs, collection=collection, cfg=cfg, pool=pool, force=force, sealed=None
|
||||||
"(item_key, collection, content_hash, chunk_count) "
|
)
|
||||||
"VALUES (:k, :c, :h, :n) "
|
# Fold the fingerprint/hash skip counters into "skipped" so the 3-key
|
||||||
"ON CONFLICT (item_key, collection) DO UPDATE SET "
|
# contract still means "not embedded" — index_refs itself keeps them
|
||||||
"content_hash = :h, chunk_count = :n, indexed_at = now()"
|
# separate; this is only a local projection.
|
||||||
),
|
skipped = stats["skipped"] + stats["fingerprint_skipped"] + stats["hash_skipped"]
|
||||||
{"k": doc.key, "c": collection, "h": h, "n": len(chunks)},
|
return {"indexed": stats["indexed"], "skipped": skipped, "chunks": stats["chunks"]}
|
||||||
)
|
|
||||||
stats["indexed"] += 1
|
|
||||||
stats["chunks"] += len(chunks)
|
|
||||||
# needs 100+ docs in one run to hit this line
|
|
||||||
if stats["indexed"] % 100 == 0:
|
|
||||||
log.info("indexed %(indexed)s (+%(chunks)s)", stats) # pragma: no cover
|
|
||||||
|
|
||||||
if cfg.build_ann_index:
|
|
||||||
ensure_hnsw(engine, cfg.embed_dim)
|
|
||||||
else:
|
|
||||||
log.info("ANN index skipped (build_ann_index=false); using exact search")
|
|
||||||
return stats
|
|
||||||
|
|||||||
@@ -21,10 +21,27 @@ CREATE TABLE IF NOT EXISTS index_state (
|
|||||||
content_hash TEXT NOT NULL,
|
content_hash TEXT NOT NULL,
|
||||||
chunk_count INTEGER NOT NULL DEFAULT 0,
|
chunk_count INTEGER NOT NULL DEFAULT 0,
|
||||||
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||||
|
fingerprint TEXT NOT NULL DEFAULT '',
|
||||||
PRIMARY KEY (item_key, collection)
|
PRIMARY KEY (item_key, collection)
|
||||||
)
|
)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
# Existing databases predate the column.
|
||||||
|
INDEX_STATE_ALTER = "ALTER TABLE index_state ADD COLUMN IF NOT EXISTS fingerprint TEXT NOT NULL DEFAULT ''"
|
||||||
|
|
||||||
|
# A sealed docket that was fully indexed under a given seal: the iterator
|
||||||
|
# does not even list its items until the seal changes (unseal/reseal
|
||||||
|
# writes a new sealed_at in bib, which no longer matches).
|
||||||
|
INDEX_DOCKET_STATE_DDL = """
|
||||||
|
CREATE TABLE IF NOT EXISTS index_docket_state (
|
||||||
|
collection TEXT NOT NULL,
|
||||||
|
docket TEXT NOT NULL,
|
||||||
|
sealed_at TEXT NOT NULL,
|
||||||
|
indexed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||||
|
PRIMARY KEY (collection, docket)
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
|
||||||
# langchain-postgres creates langchain_pg_embedding with an untyped vector
|
# langchain-postgres creates langchain_pg_embedding with an untyped vector
|
||||||
# column; HNSW needs a typed one. ALTER is a no-op when already typed.
|
# column; HNSW needs a typed one. ALTER is a no-op when already typed.
|
||||||
_HNSW_DDL = [
|
_HNSW_DDL = [
|
||||||
@@ -35,9 +52,11 @@ _HNSW_DDL = [
|
|||||||
|
|
||||||
|
|
||||||
def migrate(engine: Engine) -> None:
|
def migrate(engine: Engine) -> None:
|
||||||
"""Create llm-owned tables. Safe to run on every start."""
|
"""Create llm-owned tables/columns. Safe to run on every start."""
|
||||||
with engine.begin() as conn:
|
with engine.begin() as conn:
|
||||||
conn.execute(text(INDEX_STATE_DDL))
|
conn.execute(text(INDEX_STATE_DDL))
|
||||||
|
conn.execute(text(INDEX_STATE_ALTER))
|
||||||
|
conn.execute(text(INDEX_DOCKET_STATE_DDL))
|
||||||
|
|
||||||
|
|
||||||
def ensure_hnsw(engine: Engine, dim: int) -> None:
|
def ensure_hnsw(engine: Engine, dim: int) -> None:
|
||||||
|
|||||||
@@ -1,5 +1,12 @@
|
|||||||
"""Doc iterators over the bib store + comment extraction tree.
|
"""Doc iterators over the bib store + comment extraction tree.
|
||||||
|
|
||||||
|
Every source is listed as a :class:`DocRef` first — key, collection,
|
||||||
|
docket and a cheap *fingerprint* (file stats + ``updated_at``, or the FR
|
||||||
|
anchor sha) — so the indexer can decide what changed before any file is
|
||||||
|
read. ``ref.load()`` then does the expensive part (extraction, per-item
|
||||||
|
SQL) and returns ``None`` for a text-less item. ``iter_*_docs`` are thin
|
||||||
|
wrappers that load every ref.
|
||||||
|
|
||||||
Comments: text prefers the #253 extraction output
|
Comments: text prefers the #253 extraction output
|
||||||
(``.state/comments/<docket>/<comment_id>/combined.md`` body via
|
(``.state/comments/<docket>/<comment_id>/combined.md`` body via
|
||||||
``rex.comments.combine.parse_combined``); falls back to the bib
|
``rex.comments.combine.parse_combined``); falls back to the bib
|
||||||
@@ -16,11 +23,14 @@ so parsing the comment id is the reliable path to a comment's docket.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import sqlite3
|
import sqlite3
|
||||||
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterator
|
from typing import Callable, Iterator
|
||||||
|
|
||||||
|
from bib.dockets import fingerprint_files
|
||||||
from bib.store import Store
|
from bib.store import Store
|
||||||
from llm.chunk import Doc, Paragraph
|
from llm.chunk import Doc, Paragraph
|
||||||
|
|
||||||
@@ -30,6 +40,19 @@ _COMMENT_URL_PREFIX = "https://www.regulations.gov/comment/"
|
|||||||
_ATTACHMENT_EXT = (".pdf", ".docx", ".doc", ".txt")
|
_ATTACHMENT_EXT = (".pdf", ".docx", ".doc", ".txt")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class DocRef:
|
||||||
|
"""A document the indexer *may* need: enough to decide (key +
|
||||||
|
fingerprint) without building it. ``load()`` does the expensive part
|
||||||
|
and returns ``None`` when the item has no text."""
|
||||||
|
|
||||||
|
key: str
|
||||||
|
collection: str
|
||||||
|
docket: str | None
|
||||||
|
fingerprint: str
|
||||||
|
load: Callable[[], "Doc | None"]
|
||||||
|
|
||||||
|
|
||||||
def _default_root() -> Path:
|
def _default_root() -> Path:
|
||||||
from conf import ROOT
|
from conf import ROOT
|
||||||
|
|
||||||
@@ -51,12 +74,26 @@ def _year_of(store: Store, item_key: str) -> str:
|
|||||||
return row[0].split(":", 1)[1] if row else ""
|
return row[0].split(":", 1)[1] if row else ""
|
||||||
|
|
||||||
|
|
||||||
def _comment_rows(
|
def _attachment_paths(store: Store, item_key: str) -> list[Path]:
|
||||||
store: Store, docket: str = ""
|
rows = (
|
||||||
) -> list[tuple[str, str, str, str, str]]:
|
store._con()
|
||||||
"""(docket, comment_id, item key, date_published, title) for every
|
.execute(
|
||||||
comment matching *docket*, newest posted first so incremental index
|
"SELECT a.storage_path FROM attachments a "
|
||||||
runs surface the latest comments before older backlog.
|
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
|
||||||
|
(item_key,),
|
||||||
|
)
|
||||||
|
.fetchall()
|
||||||
|
)
|
||||||
|
return [Path(r[0]) for r in rows]
|
||||||
|
|
||||||
|
|
||||||
|
# ── comments ──
|
||||||
|
|
||||||
|
|
||||||
|
def _comment_listing(store: Store, docket: str = "") -> list[sqlite3.Row]:
|
||||||
|
"""One query: key, comment url, date, title, updated_at, year — newest
|
||||||
|
posted first so incremental index runs surface the latest comments
|
||||||
|
before older backlog.
|
||||||
|
|
||||||
*docket* filters via the URL (``.../comment/<docket>-%``); an empty
|
*docket* filters via the URL (``.../comment/<docket>-%``); an empty
|
||||||
docket matches every comment and the docket is recovered from each
|
docket matches every comment and the docket is recovered from each
|
||||||
@@ -65,29 +102,26 @@ def _comment_rows(
|
|||||||
pattern = (
|
pattern = (
|
||||||
f"{_COMMENT_URL_PREFIX}{docket}-%" if docket else f"{_COMMENT_URL_PREFIX}%"
|
f"{_COMMENT_URL_PREFIX}{docket}-%" if docket else f"{_COMMENT_URL_PREFIX}%"
|
||||||
)
|
)
|
||||||
rows = (
|
return (
|
||||||
store._con()
|
store._con()
|
||||||
.execute(
|
.execute(
|
||||||
"SELECT i.key, i.url, COALESCE(i.date_published, ''), "
|
"SELECT i.key, i.url, COALESCE(i.date_published,'') AS date, "
|
||||||
"COALESCE(i.title, '') FROM items i WHERE i.url LIKE ? "
|
"COALESCE(i.title,'') AS title, COALESCE(i.updated_at,'') AS updated_at, "
|
||||||
|
"COALESCE((SELECT t.name FROM tags t JOIN item_tags it ON it.tag_id = t.id "
|
||||||
|
" WHERE it.item_id = i.id AND t.name LIKE 'year:%' LIMIT 1), '') AS year "
|
||||||
|
"FROM items i WHERE i.url LIKE ? "
|
||||||
"ORDER BY i.date_published DESC, i.key",
|
"ORDER BY i.date_published DESC, i.key",
|
||||||
(pattern,),
|
(pattern,),
|
||||||
)
|
)
|
||||||
.fetchall()
|
.fetchall()
|
||||||
)
|
)
|
||||||
out = []
|
|
||||||
for key, url, date, title in rows:
|
|
||||||
comment_id = url.rsplit("/", 1)[-1]
|
|
||||||
dk = docket or comment_id.rsplit("-", 1)[0]
|
|
||||||
out.append((dk, comment_id, key, date, title))
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def comment_key_map(store: Store, docket: str) -> dict[str, tuple[str, str]]:
|
def comment_key_map(store: Store, docket: str) -> dict[str, tuple[str, str]]:
|
||||||
"""comment_id -> (bib item key, year) for one docket."""
|
"""comment_id -> (bib item key, year) for one docket."""
|
||||||
return {
|
return {
|
||||||
comment_id: (key, _year_of(store, key))
|
row["url"].rsplit("/", 1)[-1]: (row["key"], row["year"].split(":", 1)[-1])
|
||||||
for _, comment_id, key, _, _ in _comment_rows(store, docket)
|
for row in _comment_listing(store, docket)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -101,52 +135,84 @@ def _comment_files(comment_dir: Path) -> tuple[tuple[str, str], ...]:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def iter_comment_docs(
|
def iter_comment_refs(
|
||||||
store: Store, *, docket: str = "", root: Path | None = None
|
store: Store,
|
||||||
) -> Iterator[Doc]:
|
*,
|
||||||
"""One Doc per comment, newest first: extraction body, else abstract."""
|
docket: str = "",
|
||||||
from rex.comments.combine import parse_combined
|
root: Path | None = None,
|
||||||
|
skip_dockets: set[str] | frozenset[str] = frozenset(),
|
||||||
|
) -> Iterator[DocRef]:
|
||||||
|
"""One DocRef per comment, newest first. Fingerprint = file stats of
|
||||||
|
combined.md + attachments + the item's updated_at; no file is read
|
||||||
|
and the extraction machinery is not even imported."""
|
||||||
root = root if root is not None else _default_root()
|
root = root if root is not None else _default_root()
|
||||||
for dk, comment_id, key, date, title in _comment_rows(store, docket):
|
for row in _comment_listing(store, docket):
|
||||||
|
comment_id = row["url"].rsplit("/", 1)[-1]
|
||||||
|
dk = docket or comment_id.rsplit("-", 1)[0]
|
||||||
|
if dk in skip_dockets:
|
||||||
|
continue
|
||||||
|
key, date, title, updated_at = (
|
||||||
|
row["key"],
|
||||||
|
row["date"],
|
||||||
|
row["title"],
|
||||||
|
row["updated_at"],
|
||||||
|
)
|
||||||
|
year = row["year"].split(":", 1)[-1]
|
||||||
|
comment_dir = root / dk / comment_id
|
||||||
|
combined = comment_dir / "combined.md"
|
||||||
|
files = _comment_files(comment_dir)
|
||||||
|
fp = (
|
||||||
|
fingerprint_files([combined, *(Path(p) for _, p in files)])
|
||||||
|
+ "|"
|
||||||
|
+ updated_at
|
||||||
|
)
|
||||||
meta = {
|
meta = {
|
||||||
"docket": dk,
|
"docket": dk,
|
||||||
"comment_id": comment_id,
|
"comment_id": comment_id,
|
||||||
"doctype": "comment",
|
"doctype": "comment",
|
||||||
"kind": "comment",
|
"kind": "comment",
|
||||||
"year": _year_of(store, key),
|
"year": year,
|
||||||
"date": date[:10],
|
"date": date[:10],
|
||||||
"title": title,
|
"title": title,
|
||||||
}
|
}
|
||||||
comment_dir = root / dk / comment_id
|
|
||||||
combined = comment_dir / "combined.md"
|
def _load(key=key, meta=meta, combined=combined, files=files) -> Doc | None:
|
||||||
if combined.exists():
|
from rex.comments.combine import parse_combined
|
||||||
_, body = parse_combined(combined.read_text())
|
|
||||||
if body.strip():
|
if combined.exists():
|
||||||
yield Doc(
|
_, body = parse_combined(combined.read_text())
|
||||||
key=key, text=body, metadata=meta, files=_comment_files(comment_dir)
|
if body.strip():
|
||||||
)
|
return Doc(key=key, text=body, metadata=meta, files=files)
|
||||||
continue
|
row = (
|
||||||
item = store.get(key)
|
store._con()
|
||||||
if item.abstract.strip():
|
.execute("SELECT abstract FROM items WHERE key = ?", (key,))
|
||||||
yield Doc(key=key, text=item.abstract, metadata=meta)
|
.fetchone()
|
||||||
|
)
|
||||||
|
abstract = (row["abstract"] if row else "") or ""
|
||||||
|
return (
|
||||||
|
Doc(key=key, text=abstract, metadata=meta) if abstract.strip() else None
|
||||||
|
)
|
||||||
|
|
||||||
|
yield DocRef(
|
||||||
|
key=key, collection="comments", docket=dk, fingerprint=fp, load=_load
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def iter_comment_docs(
|
||||||
|
store: Store, *, docket: str = "", root: Path | None = None
|
||||||
|
) -> Iterator[Doc]:
|
||||||
|
"""One Doc per comment, newest first: extraction body, else abstract."""
|
||||||
|
for ref in iter_comment_refs(store, docket=docket, root=root):
|
||||||
|
doc = ref.load()
|
||||||
|
if doc is not None:
|
||||||
|
yield doc
|
||||||
|
|
||||||
|
|
||||||
def _attachment_text(store: Store, item_key: str) -> str:
|
def _attachment_text(store: Store, item_key: str) -> str:
|
||||||
from rex.comments.combine import extract_attachment
|
from rex.comments.combine import extract_attachment
|
||||||
|
|
||||||
rows = (
|
|
||||||
store._con()
|
|
||||||
.execute(
|
|
||||||
"SELECT a.storage_path FROM attachments a "
|
|
||||||
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
|
|
||||||
(item_key,),
|
|
||||||
)
|
|
||||||
.fetchall()
|
|
||||||
)
|
|
||||||
parts = []
|
parts = []
|
||||||
for (storage_path,) in rows:
|
for path in _attachment_paths(store, item_key):
|
||||||
path = Path(storage_path)
|
|
||||||
if path.exists():
|
if path.exists():
|
||||||
result = extract_attachment(path)
|
result = extract_attachment(path)
|
||||||
if result.status == "ok":
|
if result.status == "ok":
|
||||||
@@ -166,17 +232,7 @@ def _rule_text(store: Store, item_key: str) -> str:
|
|||||||
"""
|
"""
|
||||||
from rex.frtext import clean_fr_text
|
from rex.frtext import clean_fr_text
|
||||||
|
|
||||||
rows = (
|
for path in _attachment_paths(store, item_key):
|
||||||
store._con()
|
|
||||||
.execute(
|
|
||||||
"SELECT a.storage_path FROM attachments a "
|
|
||||||
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
|
|
||||||
(item_key,),
|
|
||||||
)
|
|
||||||
.fetchall()
|
|
||||||
)
|
|
||||||
for (storage_path,) in rows:
|
|
||||||
path = Path(storage_path)
|
|
||||||
if path.suffix.lower() == ".txt" and path.exists():
|
if path.suffix.lower() == ".txt" and path.exists():
|
||||||
return clean_fr_text(path.read_text(encoding="utf-8", errors="replace"))
|
return clean_fr_text(path.read_text(encoding="utf-8", errors="replace"))
|
||||||
return _attachment_text(store, item_key)
|
return _attachment_text(store, item_key)
|
||||||
@@ -209,42 +265,122 @@ def _anchor_doc(store: Store, item_key: str) -> tuple[str, str]:
|
|||||||
return (row[0], str(row[1])) if row else ("", "")
|
return (row[0], str(row[1])) if row else ("", "")
|
||||||
|
|
||||||
|
|
||||||
|
# ── rules ──
|
||||||
|
|
||||||
|
|
||||||
|
def iter_rule_refs(
|
||||||
|
store: Store, *, keys: tuple[str, ...] = (), tag: str = ""
|
||||||
|
) -> Iterator[DocRef]:
|
||||||
|
"""One DocRef per FR rule item. Fingerprint = the grabbed anchor
|
||||||
|
document's sha256 when present, else the attachment file stats +
|
||||||
|
updated_at."""
|
||||||
|
rows = (
|
||||||
|
store._con()
|
||||||
|
.execute(
|
||||||
|
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at, "
|
||||||
|
"(SELECT d.sha256 FROM fr_anchor_docs d WHERE d.item_key = i.key) AS anchor_sha "
|
||||||
|
"FROM items i WHERE i.item_type = 'rule'"
|
||||||
|
+ (
|
||||||
|
" AND i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
||||||
|
"(SELECT id FROM tags WHERE name = ?))"
|
||||||
|
if tag
|
||||||
|
else ""
|
||||||
|
)
|
||||||
|
+ " ORDER BY i.id",
|
||||||
|
(tag,) if tag else (),
|
||||||
|
)
|
||||||
|
.fetchall()
|
||||||
|
)
|
||||||
|
for row in rows:
|
||||||
|
key = row["key"]
|
||||||
|
if keys and key not in keys:
|
||||||
|
continue
|
||||||
|
if row["anchor_sha"]:
|
||||||
|
# updated_at too: the anchor sha covers the FR body, not the
|
||||||
|
# item's own metadata (title, cms-rule: tag, date).
|
||||||
|
fp = f"anchors:{row['anchor_sha']}|{row['updated_at']}"
|
||||||
|
else:
|
||||||
|
fp = (
|
||||||
|
fingerprint_files(_attachment_paths(store, key))
|
||||||
|
+ "|"
|
||||||
|
+ row["updated_at"]
|
||||||
|
)
|
||||||
|
|
||||||
|
def _load(key=key) -> Doc | None:
|
||||||
|
return _build_rule_doc(store, store.get(key))
|
||||||
|
|
||||||
|
yield DocRef(
|
||||||
|
key=key, collection="rules", docket=None, fingerprint=fp, load=_load
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _build_rule_doc(store: Store, item) -> Doc | None:
|
||||||
|
"""Anchor paragraphs when grabbed (exact ``#p-N`` provenance per
|
||||||
|
chunk), else TXT attachment, else PDF-extract; None when empty."""
|
||||||
|
paragraphs = rule_paragraphs(store, item.key)
|
||||||
|
html_url, volume = _anchor_doc(store, item.key)
|
||||||
|
if paragraphs:
|
||||||
|
text = "\n\n".join(p.text for p in paragraphs if p.text.strip())
|
||||||
|
else:
|
||||||
|
text = _rule_text(store, item.key)
|
||||||
|
if not text.strip():
|
||||||
|
return None
|
||||||
|
cms_rule = next(
|
||||||
|
(t.split(":", 1)[1] for t in item.tags if t.startswith("cms-rule:")), ""
|
||||||
|
)
|
||||||
|
return Doc(
|
||||||
|
key=item.key,
|
||||||
|
text=text,
|
||||||
|
metadata={
|
||||||
|
"doctype": "rule",
|
||||||
|
"kind": "rule",
|
||||||
|
"cms_rule_id": cms_rule,
|
||||||
|
"fr_document_number": item.document_number or "",
|
||||||
|
"year": _year_of(store, item.key),
|
||||||
|
"date": (item.date_published or "")[:10],
|
||||||
|
"title": item.title,
|
||||||
|
"item_key": item.key,
|
||||||
|
"html_url": html_url,
|
||||||
|
"fr_volume": volume,
|
||||||
|
},
|
||||||
|
paragraphs=paragraphs,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def iter_rule_docs(
|
def iter_rule_docs(
|
||||||
store: Store, *, keys: tuple[str, ...] = (), tag: str = ""
|
store: Store, *, keys: tuple[str, ...] = (), tag: str = ""
|
||||||
) -> Iterator[Doc]:
|
) -> Iterator[Doc]:
|
||||||
"""One Doc per FR rule item: anchor paragraphs when grabbed (exact
|
"""One Doc per FR rule item: anchor paragraphs when grabbed (exact
|
||||||
``#p-N`` provenance per chunk), else TXT attachment, else PDF-extract."""
|
``#p-N`` provenance per chunk), else TXT attachment, else PDF-extract."""
|
||||||
for item in store.list_items(item_type="rule", tag=tag):
|
for ref in iter_rule_refs(store, keys=keys, tag=tag):
|
||||||
if keys and item.key not in keys:
|
doc = ref.load()
|
||||||
continue
|
if doc is not None:
|
||||||
paragraphs = rule_paragraphs(store, item.key)
|
yield doc
|
||||||
html_url, volume = _anchor_doc(store, item.key)
|
|
||||||
if paragraphs:
|
|
||||||
text = "\n\n".join(p.text for p in paragraphs if p.text.strip())
|
def _source_mtime_ns(src: Path) -> int:
|
||||||
else:
|
"""Newest mtime across zotero.sqlite and its ``-wal`` sidecar.
|
||||||
text = _rule_text(store, item.key)
|
|
||||||
if not text.strip():
|
Zotero commits into the WAL and only touches the main database at a
|
||||||
continue
|
checkpoint, so ``src.stat()`` alone reports a database that has not
|
||||||
cms_rule = next(
|
changed for weeks while the library is being edited daily.
|
||||||
(t.split(":", 1)[1] for t in item.tags if t.startswith("cms-rule:")), ""
|
"""
|
||||||
)
|
newest = src.stat().st_mtime_ns if src.exists() else 0
|
||||||
yield Doc(
|
wal = src.with_name(src.name + "-wal")
|
||||||
key=item.key,
|
if wal.exists():
|
||||||
text=text,
|
newest = max(newest, wal.stat().st_mtime_ns)
|
||||||
metadata={
|
return newest
|
||||||
"doctype": "rule",
|
|
||||||
"kind": "rule",
|
|
||||||
"cms_rule_id": cms_rule,
|
def _copy_snapshot(src: Path, snap: Path) -> bool:
|
||||||
"fr_document_number": item.document_number or "",
|
"""Copy the Zotero database to *snap*. False (logged) on failure —
|
||||||
"year": _year_of(store, item.key),
|
the caller must not treat whatever is at *snap* as fresh."""
|
||||||
"date": (item.date_published or "")[:10],
|
try:
|
||||||
"title": item.title,
|
shutil.copy2(src, snap)
|
||||||
"item_key": item.key,
|
except OSError as e:
|
||||||
"html_url": html_url,
|
log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e)
|
||||||
"fr_volume": volume,
|
return False
|
||||||
},
|
return True
|
||||||
paragraphs=paragraphs,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class ZoteroPdfIndex:
|
class ZoteroPdfIndex:
|
||||||
@@ -254,6 +390,44 @@ class ZoteroPdfIndex:
|
|||||||
|
|
||||||
def __init__(self, by_key: dict[str, list[Path]]) -> None:
|
def __init__(self, by_key: dict[str, list[Path]]) -> None:
|
||||||
self._by_key = by_key
|
self._by_key = by_key
|
||||||
|
self._loader: Callable[[], dict[str, list[Path]]] | None = None
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def lazy(
|
||||||
|
cls, sqlite_path: Path, storage_dir: Path, tmp_dir: Path
|
||||||
|
) -> "ZoteroPdfIndex":
|
||||||
|
"""Snapshot on first ``pdfs_for``; skip the copy when the existing
|
||||||
|
snapshot is already as new as the source (the 1.95 GB copy is pure
|
||||||
|
waste on a run that never reaches the Zotero fallback)."""
|
||||||
|
inst = cls({})
|
||||||
|
|
||||||
|
def _load() -> dict[str, list[Path]]:
|
||||||
|
snap = Path(tmp_dir) / "zotero.sqlite"
|
||||||
|
src = Path(sqlite_path)
|
||||||
|
# Zotero writes to zotero.sqlite-wal and only touches the main
|
||||||
|
# file at a checkpoint, so the main mtime alone would leave a
|
||||||
|
# stale snapshot in place indefinitely — take the newer of the two.
|
||||||
|
newest = _source_mtime_ns(src)
|
||||||
|
if snap.exists() and src.exists() and snap.stat().st_mtime_ns >= newest:
|
||||||
|
return cls._read(snap, Path(storage_dir))
|
||||||
|
if not src.exists():
|
||||||
|
log.warning("zotero db %s missing — no Zotero PDF fallback", src)
|
||||||
|
return {}
|
||||||
|
Path(tmp_dir).mkdir(parents=True, exist_ok=True)
|
||||||
|
if not _copy_snapshot(src, snap):
|
||||||
|
# Only a *successful* copy may be stamped: stamping the
|
||||||
|
# stale snapshot left behind by a failed one would pass
|
||||||
|
# it off as current on every future run.
|
||||||
|
return {}
|
||||||
|
if newest:
|
||||||
|
# copy2 carries the *main* file's mtime, which is older
|
||||||
|
# than the WAL; stamp what we actually captured so the
|
||||||
|
# next run doesn't re-copy 1.95 GB for nothing.
|
||||||
|
os.utime(snap, ns=(newest, newest))
|
||||||
|
return cls._read(snap, Path(storage_dir))
|
||||||
|
|
||||||
|
inst._loader = _load
|
||||||
|
return inst
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def snapshot(
|
def snapshot(
|
||||||
@@ -264,8 +438,14 @@ class ZoteroPdfIndex:
|
|||||||
return cls({})
|
return cls({})
|
||||||
tmp_dir.mkdir(parents=True, exist_ok=True)
|
tmp_dir.mkdir(parents=True, exist_ok=True)
|
||||||
snap = tmp_dir / "zotero.sqlite"
|
snap = tmp_dir / "zotero.sqlite"
|
||||||
|
if not _copy_snapshot(Path(sqlite_path), snap):
|
||||||
|
return cls({})
|
||||||
|
return cls(cls._read(snap, Path(storage_dir)))
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _read(cls, snap: Path, storage_dir: Path) -> dict[str, list[Path]]:
|
||||||
|
"""parent item key → existing storage PDFs, from a snapshot copy."""
|
||||||
try:
|
try:
|
||||||
shutil.copy2(sqlite_path, snap)
|
|
||||||
con = sqlite3.connect(f"file:{snap}?mode=ro", uri=True)
|
con = sqlite3.connect(f"file:{snap}?mode=ro", uri=True)
|
||||||
rows = con.execute(
|
rows = con.execute(
|
||||||
"SELECT p.key, a.key, ia.path FROM itemAttachments ia "
|
"SELECT p.key, a.key, ia.path FROM itemAttachments ia "
|
||||||
@@ -277,15 +457,18 @@ class ZoteroPdfIndex:
|
|||||||
con.close()
|
con.close()
|
||||||
except (OSError, sqlite3.Error) as e:
|
except (OSError, sqlite3.Error) as e:
|
||||||
log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e)
|
log.warning("zotero snapshot failed (%s) — no Zotero PDF fallback", e)
|
||||||
return cls({})
|
return {}
|
||||||
by_key: dict[str, list[Path]] = {}
|
by_key: dict[str, list[Path]] = {}
|
||||||
for parent_key, att_key, path in rows:
|
for parent_key, att_key, path in rows:
|
||||||
pdf = Path(storage_dir) / att_key / path[len("storage:") :]
|
pdf = Path(storage_dir) / att_key / path[len("storage:") :]
|
||||||
if pdf.exists():
|
if pdf.exists():
|
||||||
by_key.setdefault(parent_key, []).append(pdf)
|
by_key.setdefault(parent_key, []).append(pdf)
|
||||||
return cls(by_key)
|
return by_key
|
||||||
|
|
||||||
def pdfs_for(self, key: str) -> list[Path]:
|
def pdfs_for(self, key: str) -> list[Path]:
|
||||||
|
if self._loader is not None:
|
||||||
|
self._by_key = self._loader()
|
||||||
|
self._loader = None
|
||||||
return list(self._by_key.get(key, []))
|
return list(self._by_key.get(key, []))
|
||||||
|
|
||||||
|
|
||||||
@@ -295,19 +478,9 @@ def _attachment_sections(
|
|||||||
"""(markdown sections, files) for an item's bib attachments."""
|
"""(markdown sections, files) for an item's bib attachments."""
|
||||||
from rex.comments.combine import extract_attachment
|
from rex.comments.combine import extract_attachment
|
||||||
|
|
||||||
rows = (
|
|
||||||
store._con()
|
|
||||||
.execute(
|
|
||||||
"SELECT a.storage_path FROM attachments a "
|
|
||||||
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
|
|
||||||
(item_key,),
|
|
||||||
)
|
|
||||||
.fetchall()
|
|
||||||
)
|
|
||||||
sections: list[str] = []
|
sections: list[str] = []
|
||||||
files: list[tuple[str, str]] = []
|
files: list[tuple[str, str]] = []
|
||||||
for (storage_path,) in rows:
|
for path in _attachment_paths(store, item_key):
|
||||||
path = Path(storage_path)
|
|
||||||
if path.exists():
|
if path.exists():
|
||||||
result = extract_attachment(path)
|
result = extract_attachment(path)
|
||||||
if result.status == "ok" and result.text.strip():
|
if result.status == "ok" and result.text.strip():
|
||||||
@@ -316,42 +489,83 @@ def _attachment_sections(
|
|||||||
return sections, files
|
return sections, files
|
||||||
|
|
||||||
|
|
||||||
|
# ── corpus ──
|
||||||
|
|
||||||
|
|
||||||
|
def iter_corpus_refs(
|
||||||
|
store: Store, *, tag: str = "", zotero: "ZoteroPdfIndex | None" = None
|
||||||
|
) -> Iterator[DocRef]:
|
||||||
|
"""One DocRef per non-comment item. Fingerprint = updated_at + the
|
||||||
|
bib attachment file stats; the Zotero fallback is only consulted by
|
||||||
|
``load()``."""
|
||||||
|
sql = (
|
||||||
|
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at FROM items i "
|
||||||
|
"WHERE i.id NOT IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
||||||
|
"(SELECT id FROM tags WHERE name = 'doctype:comment'))"
|
||||||
|
+ (
|
||||||
|
" AND i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
||||||
|
"(SELECT id FROM tags WHERE name = ?))"
|
||||||
|
if tag
|
||||||
|
else ""
|
||||||
|
)
|
||||||
|
+ " ORDER BY i.id"
|
||||||
|
)
|
||||||
|
for row in store._con().execute(sql, (tag,) if tag else ()).fetchall():
|
||||||
|
key = row["key"]
|
||||||
|
fp = row["updated_at"] + "|" + fingerprint_files(_attachment_paths(store, key))
|
||||||
|
|
||||||
|
def _load(key=key) -> Doc | None:
|
||||||
|
return _build_corpus_doc(store, store.get(key), zotero)
|
||||||
|
|
||||||
|
yield DocRef(
|
||||||
|
key=key, collection="corpus", docket=None, fingerprint=fp, load=_load
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _build_corpus_doc(
|
||||||
|
store: Store, item, zotero: "ZoteroPdfIndex | None"
|
||||||
|
) -> Doc | None:
|
||||||
|
"""Attachment sections (bib, else Zotero-only storage PDFs) +
|
||||||
|
abstract; None when empty."""
|
||||||
|
from rex.comments.combine import extract_attachment
|
||||||
|
|
||||||
|
sections, files = _attachment_sections(store, item.key)
|
||||||
|
if not sections and zotero is not None:
|
||||||
|
for pdf in zotero.pdfs_for(item.key):
|
||||||
|
result = extract_attachment(pdf)
|
||||||
|
if result.status == "ok" and result.text.strip():
|
||||||
|
sections.append(f"## {pdf.name}\n\n{result.text.strip()}")
|
||||||
|
files.append((pdf.name, str(pdf)))
|
||||||
|
abstract = item.abstract.strip()
|
||||||
|
parts = sections + ([abstract] if abstract else [])
|
||||||
|
text = "\n\n".join(parts)
|
||||||
|
if not text.strip():
|
||||||
|
return None
|
||||||
|
project = next(
|
||||||
|
(t.split(":", 1)[1] for t in item.tags if t.startswith("project:")), ""
|
||||||
|
)
|
||||||
|
return Doc(
|
||||||
|
key=item.key,
|
||||||
|
text=text,
|
||||||
|
metadata={
|
||||||
|
"doctype": item.item_type,
|
||||||
|
"kind": "corpus",
|
||||||
|
"year": _year_of(store, item.key),
|
||||||
|
"date": (item.date_published or "")[:10],
|
||||||
|
"title": item.title,
|
||||||
|
"url": item.url or "",
|
||||||
|
"project": project,
|
||||||
|
},
|
||||||
|
files=tuple(files),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def iter_corpus_docs(
|
def iter_corpus_docs(
|
||||||
store: Store, *, tag: str = "", zotero: ZoteroPdfIndex | None = None
|
store: Store, *, tag: str = "", zotero: ZoteroPdfIndex | None = None
|
||||||
) -> Iterator[Doc]:
|
) -> Iterator[Doc]:
|
||||||
"""Every non-comment item: attachment sections (bib, else Zotero-only
|
"""Every non-comment item: attachment sections (bib, else Zotero-only
|
||||||
storage PDFs) + abstract."""
|
storage PDFs) + abstract."""
|
||||||
from rex.comments.combine import extract_attachment
|
for ref in iter_corpus_refs(store, tag=tag, zotero=zotero):
|
||||||
|
doc = ref.load()
|
||||||
for item in store.list_items(tag=tag):
|
if doc is not None:
|
||||||
if "doctype:comment" in item.tags:
|
yield doc
|
||||||
continue
|
|
||||||
sections, files = _attachment_sections(store, item.key)
|
|
||||||
if not sections and zotero is not None:
|
|
||||||
for pdf in zotero.pdfs_for(item.key):
|
|
||||||
result = extract_attachment(pdf)
|
|
||||||
if result.status == "ok" and result.text.strip():
|
|
||||||
sections.append(f"## {pdf.name}\n\n{result.text.strip()}")
|
|
||||||
files.append((pdf.name, str(pdf)))
|
|
||||||
abstract = item.abstract.strip()
|
|
||||||
parts = sections + ([abstract] if abstract else [])
|
|
||||||
text = "\n\n".join(parts)
|
|
||||||
if not text.strip():
|
|
||||||
continue
|
|
||||||
project = next(
|
|
||||||
(t.split(":", 1)[1] for t in item.tags if t.startswith("project:")), ""
|
|
||||||
)
|
|
||||||
yield Doc(
|
|
||||||
key=item.key,
|
|
||||||
text=text,
|
|
||||||
metadata={
|
|
||||||
"doctype": item.item_type,
|
|
||||||
"kind": "corpus",
|
|
||||||
"year": _year_of(store, item.key),
|
|
||||||
"date": (item.date_published or "")[:10],
|
|
||||||
"title": item.title,
|
|
||||||
"url": item.url or "",
|
|
||||||
"project": project,
|
|
||||||
},
|
|
||||||
files=tuple(files),
|
|
||||||
)
|
|
||||||
|
|||||||
@@ -120,6 +120,38 @@ def derive_siblings_from_combined(comment_dir: Path) -> list[Path]:
|
|||||||
return written
|
return written
|
||||||
|
|
||||||
|
|
||||||
|
def source_paths(comment_dir: Path) -> list[Path]:
|
||||||
|
"""Files that feed extraction: attachments and bodies, never our own
|
||||||
|
outputs (``combined.md``, ``*.md`` siblings, ``*.tmp``) or dotfiles."""
|
||||||
|
out: list[Path] = []
|
||||||
|
for p in comment_dir.iterdir():
|
||||||
|
n = p.name
|
||||||
|
if not p.is_file() or n.startswith(".") or n == _FILENAME:
|
||||||
|
continue
|
||||||
|
if n.endswith(".md") or n.endswith(".tmp"):
|
||||||
|
continue
|
||||||
|
out.append(p)
|
||||||
|
return sorted(out)
|
||||||
|
|
||||||
|
|
||||||
|
def is_current(comment_dir: Path) -> bool:
|
||||||
|
"""True when ``combined.md`` exists and is at least as new as every
|
||||||
|
source file — a stat-only check, no reads."""
|
||||||
|
out = comment_dir / _FILENAME
|
||||||
|
try:
|
||||||
|
out_m = out.stat().st_mtime_ns
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False
|
||||||
|
for p in source_paths(comment_dir):
|
||||||
|
try:
|
||||||
|
newer = p.stat().st_mtime_ns > out_m
|
||||||
|
except FileNotFoundError:
|
||||||
|
continue # vanished between iterdir() and stat() — not our business
|
||||||
|
if newer:
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
def parse_combined(text: str) -> tuple[dict[str, Any], str]:
|
def parse_combined(text: str) -> tuple[dict[str, Any], str]:
|
||||||
"""Split a combined.md into (frontmatter dict, body markdown)."""
|
"""Split a combined.md into (frontmatter dict, body markdown)."""
|
||||||
if not text.startswith("---\n"):
|
if not text.startswith("---\n"):
|
||||||
@@ -136,11 +168,7 @@ def parse_combined(text: str) -> tuple[dict[str, Any], str]:
|
|||||||
|
|
||||||
|
|
||||||
def _attachment_paths(comment_dir: Path) -> list[Path]:
|
def _attachment_paths(comment_dir: Path) -> list[Path]:
|
||||||
return [
|
return source_paths(comment_dir)
|
||||||
p
|
|
||||||
for p in comment_dir.iterdir()
|
|
||||||
if p.is_file() and p.name != _FILENAME and not p.name.startswith(".")
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def _render_section(filename: str, result: ExtractResult) -> str:
|
def _render_section(filename: str, result: ExtractResult) -> str:
|
||||||
|
|||||||
@@ -12,12 +12,14 @@ from collections.abc import Callable
|
|||||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from rex.comments.combine import derive_siblings_from_combined, extract_comment
|
from rex.comments.combine import (
|
||||||
|
derive_siblings_from_combined,
|
||||||
|
extract_comment,
|
||||||
|
is_current,
|
||||||
|
)
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
_COMBINED = "combined.md"
|
|
||||||
|
|
||||||
|
|
||||||
def walk_and_extract(
|
def walk_and_extract(
|
||||||
root: Path,
|
root: Path,
|
||||||
@@ -28,6 +30,8 @@ def walk_and_extract(
|
|||||||
workers: int = 8,
|
workers: int = 8,
|
||||||
force: bool = False,
|
force: bool = False,
|
||||||
on_extracted: Callable[[str, Path], None] | None = None,
|
on_extracted: Callable[[str, Path], None] | None = None,
|
||||||
|
skip_dockets: set[str] | frozenset[str] = frozenset(),
|
||||||
|
reattach: bool = False,
|
||||||
) -> dict[str, int]:
|
) -> dict[str, int]:
|
||||||
"""Process every comment dir under *root*.
|
"""Process every comment dir under *root*.
|
||||||
|
|
||||||
@@ -36,21 +40,22 @@ def walk_and_extract(
|
|||||||
``comment_id`` and returns the inline body text from bib.sqlite (or
|
``comment_id`` and returns the inline body text from bib.sqlite (or
|
||||||
"" if missing).
|
"" if missing).
|
||||||
|
|
||||||
*on_extracted*, if given, is called on the main thread with
|
A dir is *current* when its combined.md is newer than every source
|
||||||
``(comment_id, comment_dir)`` for every dir that ends up with a
|
file (``combine.is_current``); current dirs are skipped without being
|
||||||
combined.md — both newly-written ones and pre-existing skips. The
|
read. *on_extracted* fires for newly written dirs; with *reattach*
|
||||||
callback is responsible for discovering whatever markdown files it
|
it also fires for skipped dirs (after deriving any missing sibling
|
||||||
wants to consume in the dir (combined.md plus per-attachment
|
MDs), which is the repair path for bib notes. *skip_dockets* names
|
||||||
siblings). Skipped dirs that pre-date the sibling-write feature get
|
docket dirs to ignore entirely (sealed dockets), unless *force*.
|
||||||
siblings derived from combined.md before the callback fires, so the
|
|
||||||
callback always sees a complete set.
|
|
||||||
|
|
||||||
Returns counts: ``{"written": N, "skipped": N, "failed": N}``.
|
Returns counts: ``{"written": N, "skipped": N, "failed": N}``, plus
|
||||||
|
``"skipped_sealed": N`` when any docket was skipped via *skip_dockets*.
|
||||||
"""
|
"""
|
||||||
candidate_dirs, skipped_dirs = _collect_dirs(
|
candidate_dirs, skipped_dirs, sealed = _collect_dirs(
|
||||||
root, docket=docket, limit=limit, force=force
|
root, docket=docket, limit=limit, force=force, skip_dockets=skip_dockets
|
||||||
)
|
)
|
||||||
stats = {"written": 0, "skipped": len(skipped_dirs), "failed": 0}
|
stats = {"written": 0, "skipped": len(skipped_dirs), "failed": 0}
|
||||||
|
if sealed:
|
||||||
|
stats["skipped_sealed"] = sealed
|
||||||
|
|
||||||
if candidate_dirs:
|
if candidate_dirs:
|
||||||
with ThreadPoolExecutor(max_workers=workers) as pool:
|
with ThreadPoolExecutor(max_workers=workers) as pool:
|
||||||
@@ -65,7 +70,7 @@ def walk_and_extract(
|
|||||||
cdir = futures[fut]
|
cdir = futures[fut]
|
||||||
on_extracted(cdir.name, cdir)
|
on_extracted(cdir.name, cdir)
|
||||||
|
|
||||||
if on_extracted:
|
if on_extracted and reattach:
|
||||||
for cdir in skipped_dirs:
|
for cdir in skipped_dirs:
|
||||||
try:
|
try:
|
||||||
derive_siblings_from_combined(cdir)
|
derive_siblings_from_combined(cdir)
|
||||||
@@ -82,31 +87,39 @@ def _collect_dirs(
|
|||||||
docket: str | None,
|
docket: str | None,
|
||||||
limit: int | None,
|
limit: int | None,
|
||||||
force: bool,
|
force: bool,
|
||||||
) -> tuple[list[Path], list[Path]]:
|
skip_dockets: set[str] | frozenset[str] = frozenset(),
|
||||||
"""Return ``(dirs_to_process, dirs_skipped)``.
|
) -> tuple[list[Path], list[Path], int]:
|
||||||
|
"""Return ``(dirs_to_process, dirs_skipped, sealed_docket_count)``.
|
||||||
|
|
||||||
``dirs_skipped`` are dirs that already have combined.md and would not
|
``dirs_skipped`` are dirs that are already current (``is_current``)
|
||||||
be re-processed. We return them as a list (not a count) so callers
|
and would not be re-processed. We return them as a list (not a count)
|
||||||
can still drive per-dir post-processing — e.g. attaching the existing
|
so callers can still drive per-dir post-processing — e.g. attaching
|
||||||
combined.md to bib on a re-run.
|
the existing combined.md to bib on a re-run (``reattach``).
|
||||||
|
|
||||||
|
``sealed_docket_count`` counts docket dirs skipped entirely because
|
||||||
|
their name is in *skip_dockets* (and *force* is False).
|
||||||
"""
|
"""
|
||||||
todo: list[Path] = []
|
todo: list[Path] = []
|
||||||
skipped: list[Path] = []
|
skipped: list[Path] = []
|
||||||
|
sealed = 0
|
||||||
for docket_dir in sorted(root.iterdir()):
|
for docket_dir in sorted(root.iterdir()):
|
||||||
if not docket_dir.is_dir():
|
if not docket_dir.is_dir():
|
||||||
continue
|
continue
|
||||||
if docket and docket_dir.name != docket:
|
if docket and docket_dir.name != docket:
|
||||||
continue
|
continue
|
||||||
|
if docket_dir.name in skip_dockets and not force:
|
||||||
|
sealed += 1
|
||||||
|
continue
|
||||||
for cdir in sorted(docket_dir.iterdir()):
|
for cdir in sorted(docket_dir.iterdir()):
|
||||||
if not cdir.is_dir():
|
if not cdir.is_dir():
|
||||||
continue
|
continue
|
||||||
if (cdir / _COMBINED).is_file() and not force:
|
if not force and is_current(cdir):
|
||||||
skipped.append(cdir)
|
skipped.append(cdir)
|
||||||
continue
|
continue
|
||||||
todo.append(cdir)
|
todo.append(cdir)
|
||||||
if limit and len(todo) >= limit:
|
if limit and len(todo) >= limit:
|
||||||
return todo, skipped
|
return todo, skipped, sealed
|
||||||
return todo, skipped
|
return todo, skipped, sealed
|
||||||
|
|
||||||
|
|
||||||
def _one(
|
def _one(
|
||||||
@@ -115,7 +128,7 @@ def _one(
|
|||||||
force: bool,
|
force: bool,
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Process one comment dir. Returns "written" / "skipped" / "failed"."""
|
"""Process one comment dir. Returns "written" / "skipped" / "failed"."""
|
||||||
if (comment_dir / _COMBINED).is_file() and not force:
|
if not force and is_current(comment_dir):
|
||||||
return "skipped"
|
return "skipped"
|
||||||
try:
|
try:
|
||||||
body = inline_body_lookup(comment_dir.name)
|
body = inline_body_lookup(comment_dir.name)
|
||||||
@@ -123,7 +136,7 @@ def _one(
|
|||||||
log.warning("inline body lookup failed for %s: %s", comment_dir.name, e)
|
log.warning("inline body lookup failed for %s: %s", comment_dir.name, e)
|
||||||
body = ""
|
body = ""
|
||||||
try:
|
try:
|
||||||
extract_comment(comment_dir, inline_body=body, force=force)
|
extract_comment(comment_dir, inline_body=body, force=True)
|
||||||
except Exception as e: # noqa: BLE001
|
except Exception as e: # noqa: BLE001
|
||||||
log.warning("extract_comment failed for %s: %s", comment_dir, e)
|
log.warning("extract_comment failed for %s: %s", comment_dir, e)
|
||||||
return "failed"
|
return "failed"
|
||||||
|
|||||||
@@ -79,6 +79,9 @@ aco = "data/aco.duckdb"
|
|||||||
bib = "data/bib.sqlite"
|
bib = "data/bib.sqlite"
|
||||||
zotero = "data/zotero/data/zotero.sqlite"
|
zotero = "data/zotero/data/zotero.sqlite"
|
||||||
|
|
||||||
|
[comments]
|
||||||
|
seal_quiet_days = 30 # days after a docket's comment close date before an empty pull seals it
|
||||||
|
|
||||||
[storage]
|
[storage]
|
||||||
bib = "data/bib/storage"
|
bib = "data/bib/storage"
|
||||||
zotero = "data/zotero/data/storage"
|
zotero = "data/zotero/data/storage"
|
||||||
|
|||||||
44
tests/bib/test_backfill_sealed.py
Normal file
44
tests/bib/test_backfill_sealed.py
Normal file
@@ -0,0 +1,44 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.regulations_gov import backfill_details, backfill_from_mirror
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
D = "CMS-2019-0111"
|
||||||
|
|
||||||
|
|
||||||
|
def _sealed_store() -> Store:
|
||||||
|
s = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(id=D, sealed_at="2020-06-01T00:00:00Z", seal_reason="manual")
|
||||||
|
)
|
||||||
|
return s
|
||||||
|
|
||||||
|
|
||||||
|
def test_api_backfill_skips_sealed_without_calls():
|
||||||
|
s = _sealed_store()
|
||||||
|
api = MagicMock()
|
||||||
|
stats = backfill_details(s, api, docket=D)
|
||||||
|
assert stats["skipped_sealed"] == 1 and stats["seen"] == 0
|
||||||
|
api.get_comment_detail.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
def test_mirror_backfill_skips_sealed_without_listing():
|
||||||
|
s = _sealed_store()
|
||||||
|
m = MagicMock()
|
||||||
|
stats = backfill_from_mirror(s, m, docket=D)
|
||||||
|
assert stats["skipped_sealed"] == 1 and stats["seen"] == 0
|
||||||
|
m.comment_ids.assert_not_called()
|
||||||
|
m.attachment_keys.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
def test_force_runs_sealed_mirror_backfill():
|
||||||
|
s = _sealed_store()
|
||||||
|
m = MagicMock()
|
||||||
|
m.comment_ids.return_value = []
|
||||||
|
m.attachment_keys.return_value = {}
|
||||||
|
stats = backfill_from_mirror(s, m, docket=D, force=True)
|
||||||
|
assert "skipped_sealed" not in stats
|
||||||
|
m.comment_ids.assert_called_once_with(D)
|
||||||
115
tests/bib/test_dockets.py
Normal file
115
tests/bib/test_dockets.py
Normal file
@@ -0,0 +1,115 @@
|
|||||||
|
"""bib.dockets — pure docket state helpers."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from datetime import date
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from bib.dockets import Docket, fingerprint_files, quiet_days, should_seal
|
||||||
|
|
||||||
|
|
||||||
|
def _d(**kw) -> Docket:
|
||||||
|
base = dict(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_document_id="CMS-2026-2377-0001",
|
||||||
|
fr_object_id="0900006482921ba1",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
pull_watermark="",
|
||||||
|
last_pull_at="",
|
||||||
|
last_pull_new=None,
|
||||||
|
sealed_at="",
|
||||||
|
seal_reason="",
|
||||||
|
counts_json="{}",
|
||||||
|
)
|
||||||
|
base.update(kw)
|
||||||
|
return Docket(**base)
|
||||||
|
|
||||||
|
|
||||||
|
class TestShouldSeal:
|
||||||
|
def test_seals_after_quiet_period_with_no_new(self):
|
||||||
|
d = _d(last_pull_at="2026-10-15T03:00:00Z", last_pull_new=0)
|
||||||
|
assert should_seal(d, date(2026, 10, 15), 30) is True
|
||||||
|
|
||||||
|
def test_not_before_quiet_period(self):
|
||||||
|
d = _d(last_pull_at="2026-10-13T03:00:00Z", last_pull_new=0)
|
||||||
|
assert should_seal(d, date(2026, 10, 13), 30) is False
|
||||||
|
|
||||||
|
def test_not_when_last_pull_found_new(self):
|
||||||
|
d = _d(last_pull_at="2026-10-20T03:00:00Z", last_pull_new=3)
|
||||||
|
assert should_seal(d, date(2026, 10, 20), 30) is False
|
||||||
|
|
||||||
|
def test_not_when_last_pull_predates_quiet_period(self):
|
||||||
|
# Pull happened before end+quiet even though today is well past it.
|
||||||
|
d = _d(last_pull_at="2026-09-20T03:00:00Z", last_pull_new=0)
|
||||||
|
assert should_seal(d, date(2026, 12, 1), 30) is False
|
||||||
|
|
||||||
|
def test_not_without_close_date(self):
|
||||||
|
d = _d(
|
||||||
|
comment_end_date="", last_pull_at="2027-01-01T00:00:00Z", last_pull_new=0
|
||||||
|
)
|
||||||
|
assert should_seal(d, date(2027, 1, 1), 30) is False
|
||||||
|
|
||||||
|
def test_not_when_already_sealed(self):
|
||||||
|
d = _d(
|
||||||
|
sealed_at="2026-10-15T00:00:00Z",
|
||||||
|
last_pull_at="2026-10-15T03:00:00Z",
|
||||||
|
last_pull_new=0,
|
||||||
|
)
|
||||||
|
assert should_seal(d, date(2026, 10, 15), 30) is False
|
||||||
|
|
||||||
|
def test_not_when_never_pulled(self):
|
||||||
|
d = _d(last_pull_at="", last_pull_new=None)
|
||||||
|
assert should_seal(d, date(2027, 1, 1), 30) is False
|
||||||
|
|
||||||
|
|
||||||
|
class TestFingerprintFiles:
|
||||||
|
def test_order_independent(self, tmp_path: Path):
|
||||||
|
a = tmp_path / "a.pdf"
|
||||||
|
b = tmp_path / "b.pdf"
|
||||||
|
a.write_bytes(b"aa")
|
||||||
|
b.write_bytes(b"bbb")
|
||||||
|
assert fingerprint_files([a, b]) == fingerprint_files([b, a])
|
||||||
|
|
||||||
|
def test_changes_when_size_changes(self, tmp_path: Path):
|
||||||
|
a = tmp_path / "a.pdf"
|
||||||
|
a.write_bytes(b"aa")
|
||||||
|
f1 = fingerprint_files([a])
|
||||||
|
a.write_bytes(b"aaaa")
|
||||||
|
assert fingerprint_files([a]) != f1
|
||||||
|
|
||||||
|
def test_changes_when_mtime_changes(self, tmp_path: Path):
|
||||||
|
a = tmp_path / "a.pdf"
|
||||||
|
a.write_bytes(b"aa")
|
||||||
|
f1 = fingerprint_files([a])
|
||||||
|
os.utime(a, ns=(1_000_000_000_000_000_000, 1_000_000_000_000_000_000))
|
||||||
|
assert fingerprint_files([a]) != f1
|
||||||
|
|
||||||
|
def test_missing_file_is_recorded_not_fatal(self, tmp_path: Path):
|
||||||
|
assert fingerprint_files([tmp_path / "nope.pdf"]) == fingerprint_files(
|
||||||
|
[tmp_path / "nope.pdf"]
|
||||||
|
)
|
||||||
|
assert fingerprint_files([tmp_path / "nope.pdf"]) != fingerprint_files([])
|
||||||
|
|
||||||
|
def test_empty_is_stable(self):
|
||||||
|
assert fingerprint_files([]) == fingerprint_files([])
|
||||||
|
assert len(fingerprint_files([])) == 64
|
||||||
|
|
||||||
|
|
||||||
|
class TestQuietDays:
|
||||||
|
def test_default_when_section_missing(self, monkeypatch):
|
||||||
|
import bib.dockets as mod
|
||||||
|
from conf import _Cfg
|
||||||
|
|
||||||
|
monkeypatch.setattr(mod, "_cfg", lambda: _Cfg({}))
|
||||||
|
assert quiet_days() == 30
|
||||||
|
|
||||||
|
def test_reads_section(self, monkeypatch):
|
||||||
|
import bib.dockets as mod
|
||||||
|
from conf import _Cfg
|
||||||
|
|
||||||
|
monkeypatch.setattr(
|
||||||
|
mod, "_cfg", lambda: _Cfg({"comments": {"seal_quiet_days": 7}})
|
||||||
|
)
|
||||||
|
assert quiet_days() == 7
|
||||||
@@ -156,7 +156,7 @@ class TestParseComment:
|
|||||||
class TestUpsertComment:
|
class TestUpsertComment:
|
||||||
def test_creates_item(self):
|
def test_creates_item(self):
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
store.upsert.return_value = "KEY1"
|
store.upsert_status.return_value = ("KEY1", "created")
|
||||||
c = Comment(
|
c = Comment(
|
||||||
id="CMS-2017-0092-0002",
|
id="CMS-2017-0092-0002",
|
||||||
title="Test",
|
title="Test",
|
||||||
@@ -168,8 +168,8 @@ class TestUpsertComment:
|
|||||||
)
|
)
|
||||||
key = upsert_comment(store, c, cms_id="CMS-1676-P")
|
key = upsert_comment(store, c, cms_id="CMS-1676-P")
|
||||||
assert key == "KEY1"
|
assert key == "KEY1"
|
||||||
store.upsert.assert_called_once()
|
store.upsert_status.assert_called_once()
|
||||||
item = store.upsert.call_args[0][0]
|
item = store.upsert_status.call_args[0][0]
|
||||||
assert "AMA" in item.title
|
assert "AMA" in item.title
|
||||||
assert item.url == "https://www.regulations.gov/comment/CMS-2017-0092-0002"
|
assert item.url == "https://www.regulations.gov/comment/CMS-2017-0092-0002"
|
||||||
|
|
||||||
|
|||||||
@@ -114,7 +114,7 @@ class TestByline:
|
|||||||
class TestUpsertComment:
|
class TestUpsertComment:
|
||||||
def test_full(self):
|
def test_full(self):
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
store.upsert.return_value = "KEY1"
|
store.upsert_status.return_value = ("KEY1", "created")
|
||||||
c = Comment(
|
c = Comment(
|
||||||
id="CMS-2023-0001-0001",
|
id="CMS-2023-0001-0001",
|
||||||
title="My Comment",
|
title="My Comment",
|
||||||
@@ -130,7 +130,7 @@ class TestUpsertComment:
|
|||||||
|
|
||||||
def test_no_title_uses_comment_on_id(self):
|
def test_no_title_uses_comment_on_id(self):
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
store.upsert.return_value = "KEY2"
|
store.upsert_status.return_value = ("KEY2", "created")
|
||||||
c = Comment(
|
c = Comment(
|
||||||
id="C2",
|
id="C2",
|
||||||
title="",
|
title="",
|
||||||
|
|||||||
65
tests/bib/test_regulations_gov_since.py
Normal file
65
tests/bib/test_regulations_gov_since.py
Normal file
@@ -0,0 +1,65 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from bib.regulations_gov import Client
|
||||||
|
|
||||||
|
|
||||||
|
def _row(cid: str, lm: str) -> dict:
|
||||||
|
return {
|
||||||
|
"id": cid,
|
||||||
|
"attributes": {
|
||||||
|
"lastModifiedDate": lm,
|
||||||
|
"postedDate": lm,
|
||||||
|
"docketId": "D",
|
||||||
|
"commentOnId": "x",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _client(handler) -> Client:
|
||||||
|
http = httpx.Client(
|
||||||
|
transport=httpx.MockTransport(handler), headers={"X-Api-Key": "k"}
|
||||||
|
)
|
||||||
|
return Client(api_key="k", sleep=0, client=http)
|
||||||
|
|
||||||
|
|
||||||
|
def test_since_seeds_the_date_filter():
|
||||||
|
seen: list[dict] = []
|
||||||
|
|
||||||
|
def handler(req: httpx.Request) -> httpx.Response:
|
||||||
|
seen.append(dict(req.url.params))
|
||||||
|
return httpx.Response(
|
||||||
|
200,
|
||||||
|
json={
|
||||||
|
"data": [_row("D-1", "2026-09-08T12:00:00Z")],
|
||||||
|
"meta": {"totalPages": 1},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
api = _client(handler)
|
||||||
|
out = list(api.iter_comments("obj", since="2026-09-01T00:00:00Z"))
|
||||||
|
assert [c.id for c in out] == ["D-1"]
|
||||||
|
assert seen[0]["filter[lastModifiedDate][ge]"] == "2026-09-01 00:00:00"
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_since_means_no_date_filter():
|
||||||
|
seen: list[dict] = []
|
||||||
|
|
||||||
|
def handler(req: httpx.Request) -> httpx.Response:
|
||||||
|
seen.append(dict(req.url.params))
|
||||||
|
return httpx.Response(200, json={"data": [], "meta": {"totalPages": 1}})
|
||||||
|
|
||||||
|
list(_client(handler).iter_comments("obj"))
|
||||||
|
assert "filter[lastModifiedDate][ge]" not in seen[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_on_error_called_when_page_fails():
|
||||||
|
errors: list[Exception] = []
|
||||||
|
|
||||||
|
def handler(req: httpx.Request) -> httpx.Response:
|
||||||
|
return httpx.Response(500, json={})
|
||||||
|
|
||||||
|
out = list(_client(handler).iter_comments("obj", on_error=errors.append))
|
||||||
|
assert out == []
|
||||||
|
assert len(errors) == 1 and isinstance(errors[0], httpx.HTTPStatusError)
|
||||||
30
tests/bib/test_store_attach_idempotent.py
Normal file
30
tests/bib/test_store_attach_idempotent.py
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from bib.item import Source
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
|
||||||
|
def test_second_attach_of_same_filename_returns_existing_key(tmp_path: Path):
|
||||||
|
s = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
|
||||||
|
key = s.create(Source(title="T", url="https://x/1"))
|
||||||
|
f = tmp_path / "attachment_1.pdf"
|
||||||
|
f.write_bytes(b"%PDF")
|
||||||
|
k1 = s.attach_file(key, f, title="attachment_1.pdf")
|
||||||
|
k2 = s.attach_file(key, f, title="attachment_1.pdf")
|
||||||
|
assert k1 == k2
|
||||||
|
n = s._con().execute("SELECT count(*) FROM attachments").fetchone()[0]
|
||||||
|
assert n == 1
|
||||||
|
# exactly one storage copy
|
||||||
|
assert len(list((tmp_path / "storage").iterdir())) == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_different_filename_creates_second_row(tmp_path: Path):
|
||||||
|
s = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
|
||||||
|
key = s.create(Source(title="T", url="https://x/1"))
|
||||||
|
f1 = tmp_path / "a.pdf"
|
||||||
|
f1.write_bytes(b"a")
|
||||||
|
f2 = tmp_path / "b.pdf"
|
||||||
|
f2.write_bytes(b"b")
|
||||||
|
assert s.attach_file(key, f1) != s.attach_file(key, f2)
|
||||||
76
tests/bib/test_store_dockets.py
Normal file
76
tests/bib/test_store_dockets.py
Normal file
@@ -0,0 +1,76 @@
|
|||||||
|
"""Store.docket_* — persistence for bib.dockets.Docket."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
|
||||||
|
def _store() -> Store:
|
||||||
|
return Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
|
||||||
|
|
||||||
|
def test_table_exists_after_init():
|
||||||
|
s = _store()
|
||||||
|
names = {
|
||||||
|
r[0]
|
||||||
|
for r in s._con().execute("SELECT name FROM sqlite_master WHERE type='table'")
|
||||||
|
}
|
||||||
|
assert "dockets" in names
|
||||||
|
idx = {
|
||||||
|
r[0]
|
||||||
|
for r in s._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
|
||||||
|
}
|
||||||
|
assert "idx_items_url" in idx
|
||||||
|
|
||||||
|
|
||||||
|
def test_upsert_then_get_roundtrip():
|
||||||
|
s = _store()
|
||||||
|
d = Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="obj",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
s.docket_upsert(d)
|
||||||
|
got = s.docket_get("CMS-2026-2377")
|
||||||
|
assert got == d
|
||||||
|
assert s.docket_get("CMS-0000-0000") is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_upsert_replaces_fields():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(Docket(id="D1", pull_watermark="a"))
|
||||||
|
s.docket_upsert(Docket(id="D1", pull_watermark="b", last_pull_new=4))
|
||||||
|
got = s.docket_get("D1")
|
||||||
|
assert got.pull_watermark == "b"
|
||||||
|
assert got.last_pull_new == 4
|
||||||
|
|
||||||
|
|
||||||
|
def test_docket_for_rule():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(Docket(id="D1", rule_cms_id="CMS-1848-P"))
|
||||||
|
assert s.docket_for_rule("CMS-1848-P").id == "D1"
|
||||||
|
assert s.docket_for_rule("CMS-9999-P") is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_seal_unseal_and_sealed_dockets():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(Docket(id="D1"))
|
||||||
|
s.docket_upsert(Docket(id="D2"))
|
||||||
|
s.docket_seal("D1", reason="manual", counts={"comments": 3})
|
||||||
|
d1 = s.docket_get("D1")
|
||||||
|
assert d1.sealed and d1.seal_reason == "manual"
|
||||||
|
assert '"comments": 3' in d1.counts_json
|
||||||
|
assert set(s.sealed_dockets()) == {"D1"}
|
||||||
|
assert s.sealed_dockets()["D1"] == d1.sealed_at
|
||||||
|
s.docket_unseal("D1")
|
||||||
|
assert not s.docket_get("D1").sealed
|
||||||
|
assert s.sealed_dockets() == {}
|
||||||
|
|
||||||
|
|
||||||
|
def test_dockets_lists_sorted_by_id():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(Docket(id="CMS-2019-0111"))
|
||||||
|
s.docket_upsert(Docket(id="CMS-2017-0092"))
|
||||||
|
assert [d.id for d in s.dockets()] == ["CMS-2017-0092", "CMS-2019-0111"]
|
||||||
98
tests/bib/test_store_updated_at.py
Normal file
98
tests/bib/test_store_updated_at.py
Normal file
@@ -0,0 +1,98 @@
|
|||||||
|
"""Tag/collection changes must stamp ``items.updated_at``.
|
||||||
|
|
||||||
|
The llm indexer fingerprints an item as ``updated_at`` + its file stats,
|
||||||
|
so a tag-only edit (``year:``, ``project:``, ``cms-rule:`` — all of which
|
||||||
|
land in chunk metadata) is invisible to a re-index unless the row is
|
||||||
|
stamped. An *unchanged* upsert must still write nothing.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from bib.item import Source
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
OLD = "2000-01-01T00:00:00Z"
|
||||||
|
|
||||||
|
|
||||||
|
def _store() -> Store:
|
||||||
|
return Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
|
||||||
|
|
||||||
|
def _item(**kw) -> Source:
|
||||||
|
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
|
||||||
|
it.abstract = kw.get("abstract", "body")
|
||||||
|
for t in kw.get("tags", ["a:1", "b:2"]):
|
||||||
|
it.add_tag(t)
|
||||||
|
return it
|
||||||
|
|
||||||
|
|
||||||
|
def _backdate(s: Store, key: str) -> None:
|
||||||
|
"""updated_at has 1s granularity — backdate so a bump is observable."""
|
||||||
|
s._con().execute("UPDATE items SET updated_at = ? WHERE key = ?", (OLD, key))
|
||||||
|
s._con().commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _updated_at(s: Store, key: str) -> str:
|
||||||
|
return (
|
||||||
|
s._con()
|
||||||
|
.execute("SELECT updated_at FROM items WHERE key = ?", (key,))
|
||||||
|
.fetchone()["updated_at"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_add_tag_bumps_updated_at():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item())
|
||||||
|
_backdate(s, key)
|
||||||
|
s.add_tag(key, "year:2026")
|
||||||
|
assert _updated_at(s, key) != OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_re_adding_the_same_tag_does_not_bump():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item())
|
||||||
|
s.add_tag(key, "year:2026")
|
||||||
|
_backdate(s, key)
|
||||||
|
s.add_tag(key, "year:2026") # already there — nothing changed
|
||||||
|
assert _updated_at(s, key) == OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_remove_tag_bumps_updated_at():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item(tags=["year:2026"]))
|
||||||
|
_backdate(s, key)
|
||||||
|
s.remove_tag(key, "year:2026")
|
||||||
|
assert _updated_at(s, key) != OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_removing_an_absent_tag_does_not_bump():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item())
|
||||||
|
_backdate(s, key)
|
||||||
|
s.remove_tag(key, "nope:1")
|
||||||
|
assert _updated_at(s, key) == OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_update_with_only_tags_bumps_updated_at():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item())
|
||||||
|
_backdate(s, key)
|
||||||
|
s.update(key, tags=["year:2026", "project:pfs"])
|
||||||
|
assert _updated_at(s, key) != OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_update_with_only_collections_bumps_updated_at():
|
||||||
|
s = _store()
|
||||||
|
key = s.create(_item())
|
||||||
|
_backdate(s, key)
|
||||||
|
s.update(key, collections=[])
|
||||||
|
assert _updated_at(s, key) != OLD
|
||||||
|
|
||||||
|
|
||||||
|
def test_unchanged_upsert_still_does_not_bump():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item())
|
||||||
|
_backdate(s, key)
|
||||||
|
key2, status = s.upsert_status(_item())
|
||||||
|
assert (key2, status) == (key, "unchanged")
|
||||||
|
assert _updated_at(s, key) == OLD
|
||||||
104
tests/bib/test_store_upsert_noop.py
Normal file
104
tests/bib/test_store_upsert_noop.py
Normal file
@@ -0,0 +1,104 @@
|
|||||||
|
"""Store.upsert must not rewrite rows/tags when nothing changed."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from bib.item import Source
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
|
||||||
|
def _store() -> Store:
|
||||||
|
return Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
|
||||||
|
|
||||||
|
def _item(**kw) -> Source:
|
||||||
|
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
|
||||||
|
it.abstract = kw.get("abstract", "body")
|
||||||
|
for t in kw.get("tags", ["a:1", "b:2"]):
|
||||||
|
it.add_tag(t)
|
||||||
|
return it
|
||||||
|
|
||||||
|
|
||||||
|
def _snapshot(s: Store, key: str) -> tuple:
|
||||||
|
con = s._con()
|
||||||
|
row = con.execute(
|
||||||
|
"SELECT access_date, updated_at FROM items WHERE key=?", (key,)
|
||||||
|
).fetchone()
|
||||||
|
tags = con.execute(
|
||||||
|
"SELECT it.rowid FROM item_tags it JOIN items i ON i.id=it.item_id WHERE i.key=? ORDER BY 1",
|
||||||
|
(key,),
|
||||||
|
).fetchall()
|
||||||
|
return (row["access_date"], row["updated_at"], [t[0] for t in tags])
|
||||||
|
|
||||||
|
|
||||||
|
def test_first_upsert_is_created():
|
||||||
|
s = _store()
|
||||||
|
key, status = s.upsert_status(_item())
|
||||||
|
assert status == "created" and key
|
||||||
|
|
||||||
|
|
||||||
|
def test_identical_upsert_is_unchanged_and_writes_nothing():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item())
|
||||||
|
before = _snapshot(s, key)
|
||||||
|
# different Python object, same content
|
||||||
|
key2, status = s.upsert_status(_item())
|
||||||
|
assert key2 == key and status == "unchanged"
|
||||||
|
assert _snapshot(s, key) == before # no access/updated stamp, no tag row churn
|
||||||
|
|
||||||
|
|
||||||
|
def test_changed_abstract_is_updated():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(abstract="v1"))
|
||||||
|
_, status = s.upsert_status(_item(abstract="v2"))
|
||||||
|
assert status == "updated"
|
||||||
|
assert s.get(key).abstract == "v2"
|
||||||
|
|
||||||
|
|
||||||
|
def test_new_tag_is_updated_and_merged():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(tags=["a:1"]))
|
||||||
|
_, status = s.upsert_status(_item(tags=["c:3"]))
|
||||||
|
assert status == "updated"
|
||||||
|
assert set(s.get(key).tags) >= {"a:1", "c:3"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_subset_of_existing_tags_is_unchanged():
|
||||||
|
"""Upsert never removes tags (#624); a subset therefore changes nothing."""
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(tags=["a:1", "b:2"]))
|
||||||
|
_, status = s.upsert_status(_item(tags=["a:1"]))
|
||||||
|
assert status == "unchanged"
|
||||||
|
|
||||||
|
|
||||||
|
def test_upsert_keeps_returning_key():
|
||||||
|
s = _store()
|
||||||
|
key = s.upsert(_item())
|
||||||
|
assert s.upsert(_item()) == key
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_incoming_abstract_keeps_stored_body():
|
||||||
|
"""A list-walk row (no body) must not blank an enriched comment."""
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(abstract="enriched body"))
|
||||||
|
_, status = s.upsert_status(_item(abstract=""))
|
||||||
|
assert status == "unchanged"
|
||||||
|
assert s.get(key).abstract == "enriched body"
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_incoming_title_keeps_stored_title():
|
||||||
|
s = _store()
|
||||||
|
it = _item()
|
||||||
|
it.title = "Org: CMS-2026-2377-1"
|
||||||
|
key, _ = s.upsert_status(it)
|
||||||
|
it2 = _item()
|
||||||
|
it2.title = ""
|
||||||
|
_, status = s.upsert_status(it2)
|
||||||
|
assert status == "unchanged"
|
||||||
|
assert s.get(key).title == "Org: CMS-2026-2377-1"
|
||||||
|
|
||||||
|
|
||||||
|
def test_non_empty_incoming_still_updates():
|
||||||
|
s = _store()
|
||||||
|
key, _ = s.upsert_status(_item(abstract="v1"))
|
||||||
|
_, status = s.upsert_status(_item(abstract="v2"))
|
||||||
|
assert status == "updated" and s.get(key).abstract == "v2"
|
||||||
212
tests/bib/test_walk_docket.py
Normal file
212
tests/bib/test_walk_docket.py
Normal file
@@ -0,0 +1,212 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from datetime import date
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.regulations_gov import (
|
||||||
|
Attachment,
|
||||||
|
Comment,
|
||||||
|
WalkResult,
|
||||||
|
discover_docket,
|
||||||
|
walk_docket,
|
||||||
|
)
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
D = "CMS-2026-2377"
|
||||||
|
|
||||||
|
|
||||||
|
def _c(n: int, lm: str) -> Comment:
|
||||||
|
return Comment(
|
||||||
|
id=f"{D}-{n}",
|
||||||
|
title="",
|
||||||
|
posted_date=lm[:10],
|
||||||
|
received_date=lm[:10],
|
||||||
|
docket_id=D,
|
||||||
|
comment_on_id="x",
|
||||||
|
raw={"attributes": {"lastModifiedDate": lm}},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _store() -> Store:
|
||||||
|
return Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
|
||||||
|
|
||||||
|
def _docket(**kw) -> Docket:
|
||||||
|
base = dict(
|
||||||
|
id=D,
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="obj",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
base.update(kw)
|
||||||
|
return Docket(**base)
|
||||||
|
|
||||||
|
|
||||||
|
def test_discover_docket_picks_commentable_doc():
|
||||||
|
api = MagicMock()
|
||||||
|
api.find_documents_in_docket.return_value = [
|
||||||
|
{"id": "X-1", "attributes": {"objectId": "o1"}}, # no comment window
|
||||||
|
{
|
||||||
|
"id": "X-2",
|
||||||
|
"attributes": {"objectId": "o2", "commentEndDate": "2026-09-14T03:59:59Z"},
|
||||||
|
},
|
||||||
|
]
|
||||||
|
d = discover_docket(api, D, rule_cms_id="CMS-1848-P")
|
||||||
|
assert d == Docket(
|
||||||
|
id=D,
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_document_id="X-2",
|
||||||
|
fr_object_id="o2",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
api.find_documents_in_docket.assert_called_once_with(D)
|
||||||
|
|
||||||
|
|
||||||
|
def test_walk_creates_counts_and_advances_watermark():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = [
|
||||||
|
_c(1, "2026-09-01T00:00:00Z"),
|
||||||
|
_c(2, "2026-09-02T00:00:00Z"),
|
||||||
|
]
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
|
||||||
|
assert r == WalkResult(
|
||||||
|
created=2,
|
||||||
|
updated=0,
|
||||||
|
unchanged=0,
|
||||||
|
watermark="2026-09-02T00:00:00Z",
|
||||||
|
clean=True,
|
||||||
|
sealed=False,
|
||||||
|
)
|
||||||
|
d = s.docket_get(D)
|
||||||
|
assert d.pull_watermark == "2026-09-02T00:00:00Z"
|
||||||
|
assert d.last_pull_new == 2 and d.last_pull_at
|
||||||
|
api.iter_comments.assert_called_once()
|
||||||
|
assert api.iter_comments.call_args.kwargs["since"] == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_walk_passes_watermark_and_counts_unchanged():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = [_c(1, "2026-09-01T00:00:00Z")]
|
||||||
|
walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
|
||||||
|
api.iter_comments.return_value = [
|
||||||
|
_c(1, "2026-09-01T00:00:00Z")
|
||||||
|
] # boundary re-yield
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
|
||||||
|
assert api.iter_comments.call_args.kwargs["since"] == "2026-09-01T00:00:00Z"
|
||||||
|
assert (r.created, r.unchanged) == (0, 1)
|
||||||
|
assert s.docket_get(D).last_pull_new == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_unclean_walk_keeps_old_watermark():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket(pull_watermark="2026-08-01T00:00:00Z"))
|
||||||
|
api = MagicMock()
|
||||||
|
|
||||||
|
def _iter(_obj, *, since="", on_error=None):
|
||||||
|
yield _c(9, "2026-09-05T00:00:00Z")
|
||||||
|
on_error(
|
||||||
|
httpx.HTTPStatusError("boom", request=MagicMock(), response=MagicMock())
|
||||||
|
)
|
||||||
|
|
||||||
|
api.iter_comments.side_effect = _iter
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
|
||||||
|
assert r.clean is False and r.created == 1
|
||||||
|
assert s.docket_get(D).pull_watermark == "2026-08-01T00:00:00Z"
|
||||||
|
|
||||||
|
|
||||||
|
def test_force_walks_from_scratch_and_never_seals():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket(pull_watermark="2026-08-01T00:00:00Z"))
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = []
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), force=True, today=date(2027, 1, 1))
|
||||||
|
assert api.iter_comments.call_args.kwargs["since"] == ""
|
||||||
|
assert r.sealed is False and not s.docket_get(D).sealed
|
||||||
|
|
||||||
|
|
||||||
|
def test_auto_seal_after_quiet_empty_pull():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = [_c(1, "2026-09-01T00:00:00Z")]
|
||||||
|
walk_docket(s, api, s.docket_get(D), today=date(2026, 9, 8))
|
||||||
|
api.iter_comments.return_value = []
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 10, 20), quiet_days=30)
|
||||||
|
assert r.sealed is True
|
||||||
|
d = s.docket_get(D)
|
||||||
|
assert d.sealed and d.seal_reason == "auto"
|
||||||
|
assert json.loads(d.counts_json)["comments"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_auto_seal_when_the_docket_has_no_stored_comments():
|
||||||
|
"""An empty walk over a docket bib never ingested is a failed farm,
|
||||||
|
not a finished one — sealing it would freeze zero comments forever."""
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = []
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), today=date(2026, 10, 20), quiet_days=30)
|
||||||
|
assert r.sealed is False
|
||||||
|
assert not s.docket_get(D).sealed
|
||||||
|
|
||||||
|
|
||||||
|
def test_attachments_fetched_for_new_comments_only_unless_forced(tmp_path):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
att = Attachment(url="https://cdn/att.pdf", filename="att.pdf")
|
||||||
|
blob = tmp_path / "att.pdf"
|
||||||
|
blob.write_bytes(b"%PDF-1.4")
|
||||||
|
api = MagicMock()
|
||||||
|
api.attachments_for.return_value = [att]
|
||||||
|
api.download_attachment.return_value = blob
|
||||||
|
c = _c(1, "2026-09-01T00:00:00Z")
|
||||||
|
c.attachment_count = 1
|
||||||
|
|
||||||
|
api.iter_comments.return_value = [c]
|
||||||
|
walk_docket(s, api, s.docket_get(D), attachments=True, today=date(2026, 9, 8))
|
||||||
|
assert api.download_attachment.call_count == 1 # created → fetched
|
||||||
|
|
||||||
|
api.iter_comments.return_value = [c]
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), attachments=True, today=date(2026, 9, 8))
|
||||||
|
assert r.unchanged == 1
|
||||||
|
assert api.download_attachment.call_count == 1 # unchanged → not re-fetched
|
||||||
|
|
||||||
|
api.iter_comments.return_value = [c]
|
||||||
|
walk_docket(
|
||||||
|
s, api, s.docket_get(D), attachments=True, force=True, today=date(2026, 9, 8)
|
||||||
|
)
|
||||||
|
assert api.download_attachment.call_count == 2 # --force re-fetches
|
||||||
|
|
||||||
|
|
||||||
|
def test_progress_echo_and_commit_every_50():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = [_c(i, "2026-09-01T00:00:00Z") for i in range(101)]
|
||||||
|
echoed: list[str] = []
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), echo=echoed.append, today=date(2026, 9, 8))
|
||||||
|
assert r.created == 101
|
||||||
|
assert " 50 comments" in echoed
|
||||||
|
assert " 100 comments" in echoed
|
||||||
|
|
||||||
|
|
||||||
|
def test_limit_stops_early_and_leaves_watermark_unadvanced():
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(_docket())
|
||||||
|
api = MagicMock()
|
||||||
|
api.iter_comments.return_value = [
|
||||||
|
_c(1, "2026-09-01T00:00:00Z"),
|
||||||
|
_c(2, "2026-09-02T00:00:00Z"),
|
||||||
|
]
|
||||||
|
r = walk_docket(s, api, s.docket_get(D), limit=1, today=date(2026, 9, 8))
|
||||||
|
assert r.created == 1
|
||||||
|
assert r.clean is False
|
||||||
|
assert s.docket_get(D).pull_watermark == ""
|
||||||
@@ -106,10 +106,14 @@ class TestDiscoverPfsRulesExercise:
|
|||||||
class TestFetchDocketComments:
|
class TestFetchDocketComments:
|
||||||
@patch("bib.connect")
|
@patch("bib.connect")
|
||||||
@patch("bib.regulations_gov.Client")
|
@patch("bib.regulations_gov.Client")
|
||||||
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1")
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
def test_basic(self, mc_upsert, mc_client_cls, mc_connect):
|
@patch("bib.regulations_gov.walk_docket")
|
||||||
store = MagicMock()
|
def test_basic(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
|
||||||
store._con.return_value = MagicMock()
|
from bib.dockets import Docket
|
||||||
|
from bib.regulations_gov import WalkResult
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
api = MagicMock()
|
api = MagicMock()
|
||||||
@@ -117,26 +121,26 @@ class TestFetchDocketComments:
|
|||||||
api.__exit__ = MagicMock(return_value=False)
|
api.__exit__ = MagicMock(return_value=False)
|
||||||
mc_client_cls.return_value = api
|
mc_client_cls.return_value = api
|
||||||
|
|
||||||
fr_doc = {
|
mc_disc.return_value = Docket(
|
||||||
"id": "DOC1",
|
id="CMS-1676-P", fr_object_id="09000001", comment_end_date="2023-12-31"
|
||||||
"attributes": {"objectId": "09000001", "commentEndDate": "2023-12-31"},
|
)
|
||||||
}
|
mc_walk.return_value = WalkResult(created=1)
|
||||||
api.find_documents_in_docket.return_value = [fr_doc]
|
|
||||||
|
|
||||||
comment = MagicMock()
|
|
||||||
comment.id = "C1"
|
|
||||||
comment.attachment_count = 0
|
|
||||||
api.iter_comments.return_value = [comment]
|
|
||||||
|
|
||||||
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P", "-n", "5"])
|
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P", "-n", "5"])
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0
|
||||||
|
mc_disc.assert_called_once()
|
||||||
|
mc_walk.assert_called_once()
|
||||||
|
|
||||||
@patch("bib.connect")
|
@patch("bib.connect")
|
||||||
@patch("bib.regulations_gov.Client")
|
@patch("bib.regulations_gov.Client")
|
||||||
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1")
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
def test_with_attachments(self, mc_upsert, mc_client_cls, mc_connect):
|
@patch("bib.regulations_gov.walk_docket")
|
||||||
store = MagicMock()
|
def test_with_attachments(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
|
||||||
store._con.return_value = MagicMock()
|
from bib.dockets import Docket
|
||||||
|
from bib.regulations_gov import WalkResult
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
api = MagicMock()
|
api = MagicMock()
|
||||||
@@ -144,33 +148,27 @@ class TestFetchDocketComments:
|
|||||||
api.__exit__ = MagicMock(return_value=False)
|
api.__exit__ = MagicMock(return_value=False)
|
||||||
mc_client_cls.return_value = api
|
mc_client_cls.return_value = api
|
||||||
|
|
||||||
fr_doc = {
|
mc_disc.return_value = Docket(
|
||||||
"id": "DOC1",
|
id="CMS-1676-P", fr_object_id="09000001", comment_end_date="2023-12-31"
|
||||||
"attributes": {"objectId": "09000001", "commentEndDate": "2023-12-31"},
|
)
|
||||||
}
|
mc_walk.return_value = WalkResult(created=1)
|
||||||
api.find_documents_in_docket.return_value = [fr_doc]
|
|
||||||
|
|
||||||
comment = MagicMock()
|
|
||||||
comment.id = "C1"
|
|
||||||
comment.attachment_count = 1
|
|
||||||
api.iter_comments.return_value = [comment]
|
|
||||||
|
|
||||||
att = MagicMock()
|
|
||||||
att.url = "https://example.com/att.pdf"
|
|
||||||
att.filename = "att.pdf"
|
|
||||||
api.attachments_for.return_value = [att]
|
|
||||||
api.download_attachment.return_value = Path("/tmp/att.pdf")
|
|
||||||
|
|
||||||
result = runner.invoke(
|
result = runner.invoke(
|
||||||
app,
|
app,
|
||||||
["fetch-docket-comments", "CMS-1676-P", "--attachments", "-n", "5"],
|
["fetch-docket-comments", "CMS-1676-P", "--attachments", "-n", "5"],
|
||||||
)
|
)
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0
|
||||||
|
assert mc_walk.call_args.kwargs["attachments"] is True
|
||||||
|
|
||||||
@patch("bib.connect")
|
@patch("bib.connect")
|
||||||
@patch("bib.regulations_gov.Client")
|
@patch("bib.regulations_gov.Client")
|
||||||
def test_skip_no_object_id(self, mc_client_cls, mc_connect):
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
store = MagicMock()
|
@patch("bib.regulations_gov.walk_docket")
|
||||||
|
def test_skip_no_object_id(self, mc_walk, mc_disc, mc_client_cls, mc_connect):
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
api = MagicMock()
|
api = MagicMock()
|
||||||
@@ -178,11 +176,12 @@ class TestFetchDocketComments:
|
|||||||
api.__exit__ = MagicMock(return_value=False)
|
api.__exit__ = MagicMock(return_value=False)
|
||||||
mc_client_cls.return_value = api
|
mc_client_cls.return_value = api
|
||||||
|
|
||||||
fr_doc = {"id": "DOC1", "attributes": {"objectId": None}}
|
mc_disc.return_value = Docket(id="CMS-1676-P")
|
||||||
api.find_documents_in_docket.return_value = [fr_doc]
|
|
||||||
|
|
||||||
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P"])
|
result = runner.invoke(app, ["fetch-docket-comments", "CMS-1676-P"])
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0
|
||||||
|
assert "no commentable FR document" in result.output
|
||||||
|
mc_walk.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
class TestFetchPfsComments:
|
class TestFetchPfsComments:
|
||||||
@@ -191,12 +190,24 @@ class TestFetchPfsComments:
|
|||||||
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1676-P"])
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1676-P"])
|
||||||
@patch("bib.translate.federal_register")
|
@patch("bib.translate.federal_register")
|
||||||
@patch("bib.regulations_gov.Client")
|
@patch("bib.regulations_gov.Client")
|
||||||
@patch("bib.regulations_gov.upsert_comment", return_value="KEY1")
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
|
@patch("bib.regulations_gov.walk_docket")
|
||||||
def test_full_loop(
|
def test_full_loop(
|
||||||
self, mc_upsert, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect
|
self,
|
||||||
|
mc_walk,
|
||||||
|
mc_disc,
|
||||||
|
mc_client_cls,
|
||||||
|
mc_translate,
|
||||||
|
mc_split,
|
||||||
|
mc_pfs,
|
||||||
|
mc_connect,
|
||||||
):
|
):
|
||||||
store = MagicMock()
|
from bib.dockets import Docket
|
||||||
store._con.return_value = MagicMock()
|
from bib.item import Rule
|
||||||
|
from bib.regulations_gov import WalkResult
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
doc = MagicMock()
|
doc = MagicMock()
|
||||||
@@ -207,7 +218,7 @@ class TestFetchPfsComments:
|
|||||||
doc.document_number = "2023-12345"
|
doc.document_number = "2023-12345"
|
||||||
mc_pfs.return_value = [doc]
|
mc_pfs.return_value = [doc]
|
||||||
|
|
||||||
rule = MagicMock()
|
rule = Rule(title="CY2024 PFS NPRM", url="https://example.com")
|
||||||
mc_translate.return_value = rule
|
mc_translate.return_value = rule
|
||||||
|
|
||||||
api = MagicMock()
|
api = MagicMock()
|
||||||
@@ -216,22 +227,18 @@ class TestFetchPfsComments:
|
|||||||
mc_client_cls.return_value = api
|
mc_client_cls.return_value = api
|
||||||
api.resolve_docket.return_value = "CMS-2023-0001"
|
api.resolve_docket.return_value = "CMS-2023-0001"
|
||||||
|
|
||||||
fr_doc = {
|
mc_disc.return_value = Docket(
|
||||||
"id": "DOC1",
|
id="CMS-2023-0001",
|
||||||
"attributes": {
|
rule_cms_id="CMS-1676-P",
|
||||||
"objectId": "09000001",
|
fr_object_id="09000001",
|
||||||
"commentEndDate": "2023-12-31",
|
comment_end_date="2023-12-31",
|
||||||
},
|
)
|
||||||
}
|
mc_walk.return_value = WalkResult(created=1)
|
||||||
api.find_documents_in_docket.return_value = [fr_doc]
|
|
||||||
|
|
||||||
comment = MagicMock()
|
|
||||||
comment.id = "C1"
|
|
||||||
comment.attachment_count = 0
|
|
||||||
api.iter_comments.return_value = [comment]
|
|
||||||
|
|
||||||
result = runner.invoke(app, ["fetch-pfs-comments", "--per-docket-limit", "1"])
|
result = runner.invoke(app, ["fetch-pfs-comments", "--per-docket-limit", "1"])
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0
|
||||||
|
mc_disc.assert_called_once()
|
||||||
|
mc_walk.assert_called_once()
|
||||||
|
|
||||||
@patch("bib.connect")
|
@patch("bib.connect")
|
||||||
@patch("bib.federalregister.pfs_rules", return_value=[])
|
@patch("bib.federalregister.pfs_rules", return_value=[])
|
||||||
@@ -252,7 +259,9 @@ class TestFetchPfsComments:
|
|||||||
def test_skip_bad_rule(
|
def test_skip_bad_rule(
|
||||||
self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect
|
self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect
|
||||||
):
|
):
|
||||||
store = MagicMock()
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
doc = MagicMock()
|
doc = MagicMock()
|
||||||
@@ -269,6 +278,8 @@ class TestFetchPfsComments:
|
|||||||
|
|
||||||
result = runner.invoke(app, ["fetch-pfs-comments"])
|
result = runner.invoke(app, ["fetch-pfs-comments"])
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0
|
||||||
|
mc_translate.assert_called_once()
|
||||||
|
assert "skipped rule meta" in result.output
|
||||||
|
|
||||||
@patch("bib.connect")
|
@patch("bib.connect")
|
||||||
@patch("bib.federalregister.pfs_rules")
|
@patch("bib.federalregister.pfs_rules")
|
||||||
@@ -276,7 +287,9 @@ class TestFetchPfsComments:
|
|||||||
@patch("bib.translate.federal_register")
|
@patch("bib.translate.federal_register")
|
||||||
@patch("bib.regulations_gov.Client")
|
@patch("bib.regulations_gov.Client")
|
||||||
def test_no_docket(self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect):
|
def test_no_docket(self, mc_client_cls, mc_translate, mc_split, mc_pfs, mc_connect):
|
||||||
store = MagicMock()
|
from bib.store import Store
|
||||||
|
|
||||||
|
store = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
mc_connect.return_value = store
|
mc_connect.return_value = store
|
||||||
|
|
||||||
doc = MagicMock()
|
doc = MagicMock()
|
||||||
|
|||||||
244
tests/cli/test_bib_fetch_sealed.py
Normal file
244
tests/cli/test_bib_fetch_sealed.py
Normal file
@@ -0,0 +1,244 @@
|
|||||||
|
"""fetch-pfs-comments / fetch-docket-comments must not call reg.gov for
|
||||||
|
sealed dockets and must reuse stored document ids for known ones."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
from typer.testing import CliRunner
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.item import Rule
|
||||||
|
from bib.regulations_gov import WalkResult
|
||||||
|
from bib.store import Store
|
||||||
|
from cli.bib import app
|
||||||
|
|
||||||
|
runner = CliRunner()
|
||||||
|
|
||||||
|
|
||||||
|
def _store() -> Store:
|
||||||
|
return Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
|
||||||
|
|
||||||
|
def _rule_doc():
|
||||||
|
doc = MagicMock()
|
||||||
|
doc.type = "Proposed Rule"
|
||||||
|
doc.publication_date = "2026-07-16"
|
||||||
|
doc.dockets = ["CMS-1848-P"]
|
||||||
|
doc.html_url = "https://example.com"
|
||||||
|
doc.document_number = "2026-1"
|
||||||
|
return doc
|
||||||
|
|
||||||
|
|
||||||
|
def _api():
|
||||||
|
api = MagicMock()
|
||||||
|
api.__enter__ = MagicMock(return_value=api)
|
||||||
|
api.__exit__ = MagicMock(return_value=False)
|
||||||
|
return api
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.translate.federal_register")
|
||||||
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
|
||||||
|
@patch("bib.federalregister.pfs_rules")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_sealed_docket_skipped_before_any_call(
|
||||||
|
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
sealed_at="2026-10-20T00:00:00Z",
|
||||||
|
seal_reason="auto",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_pfs.return_value = [_rule_doc()]
|
||||||
|
api = _api()
|
||||||
|
mc_client.return_value = api
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-pfs-comments"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert "sealed" in result.output
|
||||||
|
mc_fr.assert_not_called() # no rule-metadata fetch
|
||||||
|
api.resolve_docket.assert_not_called()
|
||||||
|
api.find_documents_in_docket.assert_not_called()
|
||||||
|
mc_walk.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult(created=2))
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.translate.federal_register")
|
||||||
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
|
||||||
|
@patch("bib.federalregister.pfs_rules")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_known_open_docket_walks_without_resolve(
|
||||||
|
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_pfs.return_value = [_rule_doc()]
|
||||||
|
mc_fr.return_value = Rule(
|
||||||
|
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
|
||||||
|
)
|
||||||
|
api = _api()
|
||||||
|
mc_client.return_value = api
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-pfs-comments"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
api.resolve_docket.assert_not_called()
|
||||||
|
api.find_documents_in_docket.assert_not_called()
|
||||||
|
mc_walk.assert_called_once()
|
||||||
|
assert mc_walk.call_args.args[2].id == "CMS-2026-2377"
|
||||||
|
assert "created=2" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
|
||||||
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.translate.federal_register")
|
||||||
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
|
||||||
|
@patch("bib.federalregister.pfs_rules")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_unknown_docket_is_resolved_once_and_stored(
|
||||||
|
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_disc, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_pfs.return_value = [_rule_doc()]
|
||||||
|
mc_fr.return_value = Rule(
|
||||||
|
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
|
||||||
|
)
|
||||||
|
api = _api()
|
||||||
|
api.resolve_docket.return_value = "CMS-2026-2377"
|
||||||
|
mc_client.return_value = api
|
||||||
|
mc_disc.return_value = Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-pfs-comments"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
api.resolve_docket.assert_called_once_with("CMS-1848-P")
|
||||||
|
mc_disc.assert_called_once()
|
||||||
|
assert s.docket_get("CMS-2026-2377").fr_object_id == "o"
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.translate.federal_register")
|
||||||
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
|
||||||
|
@patch("bib.federalregister.pfs_rules")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_docket_filter_skips_other_known_dockets_without_calls(
|
||||||
|
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(id="CMS-2026-2377", rule_cms_id="CMS-1848-P", fr_object_id="o")
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_pfs.return_value = [_rule_doc()]
|
||||||
|
api = _api()
|
||||||
|
mc_client.return_value = api
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-pfs-comments", "--docket", "CMS-2019-0111"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mc_fr.assert_not_called()
|
||||||
|
api.resolve_docket.assert_not_called()
|
||||||
|
mc_walk.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult())
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.translate.federal_register")
|
||||||
|
@patch("bib.federalregister.split_docket_ids", return_value=["CMS-1848-P"])
|
||||||
|
@patch("bib.federalregister.pfs_rules")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_force_walks_sealed_docket(
|
||||||
|
mc_connect, mc_pfs, _split, mc_fr, mc_client, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
sealed_at="2026-10-20T00:00:00Z",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_pfs.return_value = [_rule_doc()]
|
||||||
|
mc_fr.return_value = Rule(
|
||||||
|
title="CY2027 PFS NPRM", url="https://www.federalregister.gov/d/2026-1"
|
||||||
|
)
|
||||||
|
mc_client.return_value = _api()
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-pfs-comments", "--force"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mc_walk.assert_called_once()
|
||||||
|
assert mc_walk.call_args.kwargs["force"] is True
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket", return_value=WalkResult(created=1))
|
||||||
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_fetch_docket_comments_discovers_then_walks(
|
||||||
|
mc_connect, mc_client, mc_disc, mc_walk
|
||||||
|
):
|
||||||
|
s = _store()
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_client.return_value = _api()
|
||||||
|
mc_disc.return_value = Docket(
|
||||||
|
id="CMS-2026-2377",
|
||||||
|
rule_cms_id="CMS-1848-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
comment_end_date="2026-09-14",
|
||||||
|
)
|
||||||
|
|
||||||
|
result = runner.invoke(
|
||||||
|
app, ["fetch-docket-comments", "CMS-2026-2377", "--cms-id", "CMS-1848-P"]
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mc_disc.assert_called_once()
|
||||||
|
mc_walk.assert_called_once()
|
||||||
|
assert s.docket_get("CMS-2026-2377") is not None
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.walk_docket")
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_fetch_docket_comments_sealed_notice(mc_connect, mc_client, mc_walk):
|
||||||
|
s = _store()
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(id="CMS-2019-0111", fr_object_id="o", sealed_at="2020-01-01T00:00:00Z")
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
mc_client.return_value = _api()
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["fetch-docket-comments", "CMS-2019-0111"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0
|
||||||
|
assert "sealed" in result.output
|
||||||
|
mc_walk.assert_not_called()
|
||||||
151
tests/cli/test_comments_dockets.py
Normal file
151
tests/cli/test_comments_dockets.py
Normal file
@@ -0,0 +1,151 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
import fitz
|
||||||
|
import pytest
|
||||||
|
from typer.testing import CliRunner
|
||||||
|
|
||||||
|
from bib.dockets import Docket
|
||||||
|
from bib.store import Store
|
||||||
|
from cli.comments import app
|
||||||
|
|
||||||
|
runner = CliRunner()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def store() -> Store:
|
||||||
|
s = Store(":memory:", storage_dir="/tmp/nope")
|
||||||
|
try:
|
||||||
|
yield s
|
||||||
|
finally:
|
||||||
|
s.close()
|
||||||
|
|
||||||
|
|
||||||
|
def _seed_dir(root: Path, docket: str) -> Path:
|
||||||
|
cdir = root / docket / f"{docket}-0001"
|
||||||
|
cdir.mkdir(parents=True)
|
||||||
|
doc = fitz.open()
|
||||||
|
doc.new_page().insert_text((50, 72), "Long body content " * 10)
|
||||||
|
doc.save(str(cdir / "attachment_1.pdf"))
|
||||||
|
doc.close()
|
||||||
|
return cdir
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_dockets_table_lists_rows(mc_connect, store):
|
||||||
|
s = store
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2019-0111",
|
||||||
|
rule_cms_id="CMS-1693-P",
|
||||||
|
comment_end_date="2019-09-27",
|
||||||
|
sealed_at="2020-01-01T00:00:00Z",
|
||||||
|
seal_reason="manual",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2026-2377", rule_cms_id="CMS-1848-P", comment_end_date="2026-09-14"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
result = runner.invoke(app, ["dockets"])
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert "CMS-2019-0111" in result.output and "sealed" in result.output
|
||||||
|
assert "CMS-2026-2377" in result.output and "open" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_seal_and_unseal(mc_connect, store):
|
||||||
|
s = store
|
||||||
|
s.docket_upsert(Docket(id="CMS-2019-0111"))
|
||||||
|
mc_connect.return_value = s
|
||||||
|
assert (
|
||||||
|
runner.invoke(
|
||||||
|
app, ["seal", "CMS-2019-0111", "--reason", "historical"]
|
||||||
|
).exit_code
|
||||||
|
== 0
|
||||||
|
)
|
||||||
|
assert s.docket_get("CMS-2019-0111").seal_reason == "historical"
|
||||||
|
assert runner.invoke(app, ["unseal", "CMS-2019-0111"]).exit_code == 0
|
||||||
|
assert not s.docket_get("CMS-2019-0111").sealed
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_seal_unknown_docket_fails(mc_connect, store):
|
||||||
|
mc_connect.return_value = store
|
||||||
|
result = runner.invoke(app, ["seal", "CMS-0000-0000"])
|
||||||
|
assert result.exit_code == 1
|
||||||
|
assert "unknown docket" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.regulations_gov.discover_docket")
|
||||||
|
@patch("bib.regulations_gov.Client")
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_dockets_discover_populates_from_tags(mc_connect, mc_client, mc_disc, store):
|
||||||
|
from bib.item import Source
|
||||||
|
|
||||||
|
s = store
|
||||||
|
it = Source(title="c", url="https://www.regulations.gov/comment/CMS-2019-0111-1")
|
||||||
|
it.add_tag("reg-docket:CMS-2019-0111")
|
||||||
|
it.add_tag("rule:CMS-1693-P")
|
||||||
|
s.create(it)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
api = MagicMock()
|
||||||
|
api.__enter__ = MagicMock(return_value=api)
|
||||||
|
api.__exit__ = MagicMock(return_value=False)
|
||||||
|
mc_client.return_value = api
|
||||||
|
mc_disc.return_value = Docket(
|
||||||
|
id="CMS-2019-0111",
|
||||||
|
rule_cms_id="CMS-1693-P",
|
||||||
|
fr_object_id="o",
|
||||||
|
comment_end_date="2019-09-27",
|
||||||
|
)
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["dockets", "--discover"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mc_disc.assert_called_once_with(api, "CMS-2019-0111", rule_cms_id="CMS-1693-P")
|
||||||
|
assert s.docket_get("CMS-2019-0111").comment_end_date == "2019-09-27"
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_extract_skips_sealed_docket(mc_connect, tmp_path: Path, store):
|
||||||
|
s = store
|
||||||
|
s.docket_upsert(
|
||||||
|
Docket(
|
||||||
|
id="CMS-2024-0001", sealed_at="2025-01-01T00:00:00Z", seal_reason="manual"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
mc_connect.return_value = s
|
||||||
|
cdir = _seed_dir(tmp_path, "CMS-2024-0001")
|
||||||
|
result = runner.invoke(
|
||||||
|
app, ["extract", "--root", str(tmp_path), "--workers", "1", "--no-attach"]
|
||||||
|
)
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert "skipped_sealed: 1" in result.output
|
||||||
|
assert not (cdir / "combined.md").exists()
|
||||||
|
|
||||||
|
|
||||||
|
@patch("bib.connect")
|
||||||
|
def test_extract_force_runs_sealed_docket(mc_connect, tmp_path: Path, store):
|
||||||
|
s = store
|
||||||
|
s.docket_upsert(Docket(id="CMS-2024-0001", sealed_at="2025-01-01T00:00:00Z"))
|
||||||
|
mc_connect.return_value = s
|
||||||
|
cdir = _seed_dir(tmp_path, "CMS-2024-0001")
|
||||||
|
result = runner.invoke(
|
||||||
|
app,
|
||||||
|
[
|
||||||
|
"extract",
|
||||||
|
"--root",
|
||||||
|
str(tmp_path),
|
||||||
|
"--workers",
|
||||||
|
"1",
|
||||||
|
"--no-attach",
|
||||||
|
"--force",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert (cdir / "combined.md").is_file()
|
||||||
@@ -10,34 +10,51 @@ from cli.llm import app
|
|||||||
|
|
||||||
runner = CliRunner()
|
runner = CliRunner()
|
||||||
|
|
||||||
_STATS = {"indexed": 3, "skipped": 1, "chunks": 7}
|
_STATS = {
|
||||||
|
"indexed": 3,
|
||||||
|
"skipped": 1,
|
||||||
|
"chunks": 7,
|
||||||
|
"fingerprint_skipped": 0,
|
||||||
|
"hash_skipped": 0,
|
||||||
|
"docket_complete": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class TestIndexComments:
|
class TestIndexComments:
|
||||||
@patch("llm.source.iter_comment_docs")
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
@patch("llm.pool.HostPool.from_config")
|
@patch("llm.pool.HostPool.from_config")
|
||||||
@patch("llm.index.index_docs")
|
@patch("llm.index.index_refs")
|
||||||
@patch("conf.connect.bib")
|
@patch("conf.connect.bib")
|
||||||
@patch("llm.config.load")
|
@patch("llm.config.load")
|
||||||
def test_default_collection_is_comments(
|
def test_default_collection_is_comments(
|
||||||
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter
|
self,
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index_refs,
|
||||||
|
mock_from_config,
|
||||||
|
mock_iter,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
):
|
):
|
||||||
cfg = MagicMock()
|
cfg = MagicMock()
|
||||||
mock_load.return_value = cfg
|
mock_load.return_value = cfg
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
mock_bib.return_value = store
|
mock_bib.return_value = store
|
||||||
docs = iter(["doc1", "doc2"])
|
docs = iter(["doc1", "doc2"])
|
||||||
mock_iter.return_value = docs
|
mock_iter.return_value = docs
|
||||||
pool = MagicMock()
|
pool = MagicMock()
|
||||||
mock_from_config.return_value = pool
|
mock_from_config.return_value = pool
|
||||||
mock_index_docs.return_value = _STATS
|
mock_index_refs.return_value = _STATS
|
||||||
|
|
||||||
result = runner.invoke(app, ["index"])
|
result = runner.invoke(app, ["index"])
|
||||||
|
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0, result.output
|
||||||
mock_iter.assert_called_once_with(store, docket="")
|
mock_iter.assert_called_once_with(store, docket="", skip_dockets=set())
|
||||||
mock_index_docs.assert_called_once()
|
mock_index_refs.assert_called_once()
|
||||||
kwargs = mock_index_docs.call_args.kwargs
|
kwargs = mock_index_refs.call_args.kwargs
|
||||||
assert kwargs["collection"] == "comments"
|
assert kwargs["collection"] == "comments"
|
||||||
assert kwargs["cfg"] is cfg
|
assert kwargs["cfg"] is cfg
|
||||||
assert kwargs["pool"] is pool
|
assert kwargs["pool"] is pool
|
||||||
@@ -46,71 +63,82 @@ class TestIndexComments:
|
|||||||
|
|
||||||
|
|
||||||
class TestIndexCorpus:
|
class TestIndexCorpus:
|
||||||
@patch("llm.source.ZoteroPdfIndex.snapshot")
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
@patch("llm.source.iter_corpus_docs")
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.source.ZoteroPdfIndex.lazy")
|
||||||
|
@patch("llm.source.iter_corpus_refs")
|
||||||
@patch("llm.pool.HostPool.from_config")
|
@patch("llm.pool.HostPool.from_config")
|
||||||
@patch("llm.index.index_docs")
|
@patch("llm.index.index_refs")
|
||||||
@patch("conf.connect.bib")
|
@patch("conf.connect.bib")
|
||||||
@patch("llm.config.load")
|
@patch("llm.config.load")
|
||||||
def test_corpus_collection(
|
def test_corpus_collection(
|
||||||
self,
|
self,
|
||||||
mock_load,
|
mock_load,
|
||||||
mock_bib,
|
mock_bib,
|
||||||
mock_index_docs,
|
mock_index_refs,
|
||||||
mock_from_config,
|
mock_from_config,
|
||||||
mock_iter,
|
mock_iter,
|
||||||
mock_snapshot,
|
mock_lazy,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
):
|
):
|
||||||
cfg = MagicMock()
|
cfg = MagicMock()
|
||||||
mock_load.return_value = cfg
|
mock_load.return_value = cfg
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
mock_bib.return_value = store
|
mock_bib.return_value = store
|
||||||
mock_iter.return_value = iter(["doc1"])
|
mock_iter.return_value = iter(["doc1"])
|
||||||
mock_from_config.return_value = MagicMock()
|
mock_from_config.return_value = MagicMock()
|
||||||
mock_index_docs.return_value = _STATS
|
mock_index_refs.return_value = _STATS
|
||||||
zot = MagicMock()
|
zot = MagicMock()
|
||||||
mock_snapshot.return_value = zot
|
mock_lazy.return_value = zot
|
||||||
|
|
||||||
result = runner.invoke(app, ["index", "--collection", "corpus"])
|
result = runner.invoke(app, ["index", "--collection", "corpus"])
|
||||||
|
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0, result.output
|
||||||
mock_iter.assert_called_once_with(store, zotero=zot)
|
mock_iter.assert_called_once_with(store, zotero=zot)
|
||||||
kwargs = mock_index_docs.call_args.kwargs
|
kwargs = mock_index_refs.call_args.kwargs
|
||||||
assert kwargs["collection"] == "corpus"
|
assert kwargs["collection"] == "corpus"
|
||||||
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
|
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
|
||||||
|
|
||||||
|
|
||||||
class TestIndexAll:
|
class TestIndexAll:
|
||||||
@patch("llm.source.ZoteroPdfIndex.snapshot")
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
@patch("llm.source.iter_corpus_docs")
|
@patch("llm.index._engine")
|
||||||
@patch("llm.source.iter_rule_docs")
|
@patch("llm.source.ZoteroPdfIndex.lazy")
|
||||||
@patch("llm.source.iter_comment_docs")
|
@patch("llm.source.iter_corpus_refs")
|
||||||
|
@patch("llm.source.iter_rule_refs")
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
@patch("llm.pool.HostPool.from_config")
|
@patch("llm.pool.HostPool.from_config")
|
||||||
@patch("llm.index.index_docs")
|
@patch("llm.index.index_refs")
|
||||||
@patch("conf.connect.bib")
|
@patch("conf.connect.bib")
|
||||||
@patch("llm.config.load")
|
@patch("llm.config.load")
|
||||||
def test_all_runs_comments_rules_corpus_in_order(
|
def test_all_runs_comments_rules_corpus_in_order(
|
||||||
self,
|
self,
|
||||||
mock_load,
|
mock_load,
|
||||||
mock_bib,
|
mock_bib,
|
||||||
mock_index_docs,
|
mock_index_refs,
|
||||||
mock_from_config,
|
mock_from_config,
|
||||||
mock_comments,
|
mock_comments,
|
||||||
mock_rules,
|
mock_rules,
|
||||||
mock_corpus,
|
mock_corpus,
|
||||||
mock_snapshot,
|
mock_lazy,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
):
|
):
|
||||||
mock_load.return_value = MagicMock()
|
mock_load.return_value = MagicMock()
|
||||||
mock_bib.return_value = MagicMock()
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
|
mock_bib.return_value = store
|
||||||
mock_from_config.return_value = MagicMock()
|
mock_from_config.return_value = MagicMock()
|
||||||
mock_index_docs.return_value = _STATS
|
mock_index_refs.return_value = _STATS
|
||||||
for m in (mock_comments, mock_rules, mock_corpus):
|
for m in (mock_comments, mock_rules, mock_corpus):
|
||||||
m.return_value = iter(["d"])
|
m.return_value = iter(["d"])
|
||||||
|
|
||||||
result = runner.invoke(app, ["index", "--collection", "all"])
|
result = runner.invoke(app, ["index", "--collection", "all"])
|
||||||
|
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0, result.output
|
||||||
assert [c.kwargs["collection"] for c in mock_index_docs.call_args_list] == [
|
assert [c.kwargs["collection"] for c in mock_index_refs.call_args_list] == [
|
||||||
"comments",
|
"comments",
|
||||||
"rules",
|
"rules",
|
||||||
"corpus",
|
"corpus",
|
||||||
@@ -119,30 +147,40 @@ class TestIndexAll:
|
|||||||
|
|
||||||
|
|
||||||
class TestIndexRules:
|
class TestIndexRules:
|
||||||
@patch("llm.source.iter_rule_docs")
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.source.iter_rule_refs")
|
||||||
@patch("llm.pool.HostPool.from_config")
|
@patch("llm.pool.HostPool.from_config")
|
||||||
@patch("llm.index.index_docs")
|
@patch("llm.index.index_refs")
|
||||||
@patch("conf.connect.bib")
|
@patch("conf.connect.bib")
|
||||||
@patch("llm.config.load")
|
@patch("llm.config.load")
|
||||||
def test_rules_collection(
|
def test_rules_collection(
|
||||||
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter
|
self,
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index_refs,
|
||||||
|
mock_from_config,
|
||||||
|
mock_iter,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
):
|
):
|
||||||
cfg = MagicMock()
|
cfg = MagicMock()
|
||||||
mock_load.return_value = cfg
|
mock_load.return_value = cfg
|
||||||
store = MagicMock()
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
mock_bib.return_value = store
|
mock_bib.return_value = store
|
||||||
mock_iter.return_value = iter(["doc1"])
|
mock_iter.return_value = iter(["doc1"])
|
||||||
mock_from_config.return_value = MagicMock()
|
mock_from_config.return_value = MagicMock()
|
||||||
mock_index_docs.return_value = _STATS
|
mock_index_refs.return_value = _STATS
|
||||||
|
|
||||||
result = runner.invoke(
|
result = runner.invoke(
|
||||||
app,
|
app,
|
||||||
["index", "--collection", "rules", "--key", "abc123", "--key", "def456"],
|
["index", "--collection", "rules", "--key", "abc123", "--key", "def456"],
|
||||||
)
|
)
|
||||||
|
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0, result.output
|
||||||
mock_iter.assert_called_once_with(store, keys=("abc123", "def456"))
|
mock_iter.assert_called_once_with(store, keys=("abc123", "def456"))
|
||||||
kwargs = mock_index_docs.call_args.kwargs
|
kwargs = mock_index_refs.call_args.kwargs
|
||||||
assert kwargs["collection"] == "rules"
|
assert kwargs["collection"] == "rules"
|
||||||
assert "indexed=3 skipped=1 chunks=7" in result.output
|
assert "indexed=3 skipped=1 chunks=7" in result.output
|
||||||
|
|
||||||
@@ -161,25 +199,37 @@ class TestIndexBadCollection:
|
|||||||
|
|
||||||
|
|
||||||
class TestIndexLimit:
|
class TestIndexLimit:
|
||||||
@patch("llm.source.iter_comment_docs")
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
@patch("llm.pool.HostPool.from_config")
|
@patch("llm.pool.HostPool.from_config")
|
||||||
@patch("llm.index.index_docs")
|
@patch("llm.index.index_refs")
|
||||||
@patch("conf.connect.bib")
|
@patch("conf.connect.bib")
|
||||||
@patch("llm.config.load")
|
@patch("llm.config.load")
|
||||||
def test_limit_truncates_docs(
|
def test_limit_truncates_docs(
|
||||||
self, mock_load, mock_bib, mock_index_docs, mock_from_config, mock_iter
|
self,
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index_refs,
|
||||||
|
mock_from_config,
|
||||||
|
mock_iter,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
):
|
):
|
||||||
mock_load.return_value = MagicMock()
|
mock_load.return_value = MagicMock()
|
||||||
mock_bib.return_value = MagicMock()
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
|
mock_bib.return_value = store
|
||||||
mock_iter.return_value = iter([f"doc{i}" for i in range(5)])
|
mock_iter.return_value = iter([f"doc{i}" for i in range(5)])
|
||||||
mock_from_config.return_value = MagicMock()
|
mock_from_config.return_value = MagicMock()
|
||||||
mock_index_docs.return_value = _STATS
|
mock_index_refs.return_value = _STATS
|
||||||
|
|
||||||
result = runner.invoke(app, ["index", "--limit", "2"])
|
result = runner.invoke(app, ["index", "--limit", "2"])
|
||||||
|
|
||||||
assert result.exit_code == 0
|
assert result.exit_code == 0, result.output
|
||||||
docs_arg = mock_index_docs.call_args.args[0]
|
refs_arg = mock_index_refs.call_args.args[0]
|
||||||
assert list(docs_arg) == ["doc0", "doc1"]
|
assert list(refs_arg) == ["doc0", "doc1"]
|
||||||
|
assert mock_index_refs.call_args.kwargs["mark_complete"] is False
|
||||||
|
|
||||||
|
|
||||||
class TestServe:
|
class TestServe:
|
||||||
|
|||||||
151
tests/cli/test_llm_index_sealed.py
Normal file
151
tests/cli/test_llm_index_sealed.py
Normal file
@@ -0,0 +1,151 @@
|
|||||||
|
"""Exercise cli/llm.py's `index` seal-skip wiring: sealed/complete dockets
|
||||||
|
are excluded from the comment listing, and only pending seals are passed
|
||||||
|
through to be re-marked."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
from typer.testing import CliRunner
|
||||||
|
|
||||||
|
from cli.llm import app
|
||||||
|
|
||||||
|
runner = CliRunner()
|
||||||
|
_STATS = {
|
||||||
|
"indexed": 0,
|
||||||
|
"skipped": 0,
|
||||||
|
"chunks": 0,
|
||||||
|
"fingerprint_skipped": 5,
|
||||||
|
"hash_skipped": 0,
|
||||||
|
"docket_complete": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch(
|
||||||
|
"llm.index.docket_complete",
|
||||||
|
return_value={"CMS-2019-0111": "s1", "CMS-2020-0088": "old"},
|
||||||
|
)
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
|
@patch("llm.pool.HostPool.from_config")
|
||||||
|
@patch("llm.index.index_refs")
|
||||||
|
@patch("conf.connect.bib")
|
||||||
|
@patch("llm.config.load")
|
||||||
|
def test_complete_sealed_dockets_are_not_listed(
|
||||||
|
mock_load, mock_bib, mock_index, mock_pool, mock_iter, mock_complete, _engine
|
||||||
|
):
|
||||||
|
mock_load.return_value = MagicMock()
|
||||||
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1", "CMS-2020-0088": "s2"}
|
||||||
|
mock_bib.return_value = store
|
||||||
|
mock_iter.return_value = iter([])
|
||||||
|
mock_index.return_value = _STATS
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["index"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
kwargs = mock_iter.call_args.kwargs
|
||||||
|
assert kwargs["skip_dockets"] == {"CMS-2019-0111"} # seal matches → skipped
|
||||||
|
ikw = mock_index.call_args.kwargs
|
||||||
|
assert ikw["sealed"] == {
|
||||||
|
"CMS-2020-0088": "s2"
|
||||||
|
} # re-sealed docket will be re-marked
|
||||||
|
assert ikw["mark_complete"] is True
|
||||||
|
assert "fp_skipped=5" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
|
@patch("llm.pool.HostPool.from_config")
|
||||||
|
@patch("llm.index.index_refs")
|
||||||
|
@patch("conf.connect.bib")
|
||||||
|
@patch("llm.config.load")
|
||||||
|
def test_force_and_limit_disable_skips_and_completion(
|
||||||
|
mock_load, mock_bib, mock_index, mock_pool, mock_iter, mock_complete, _engine
|
||||||
|
):
|
||||||
|
mock_load.return_value = MagicMock()
|
||||||
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
|
||||||
|
mock_bib.return_value = store
|
||||||
|
mock_iter.return_value = iter([])
|
||||||
|
mock_index.return_value = _STATS
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["index", "--force", "--limit", "5"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert mock_iter.call_args.kwargs["skip_dockets"] == set()
|
||||||
|
assert mock_index.call_args.kwargs["mark_complete"] is False
|
||||||
|
|
||||||
|
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.migrate.migrate")
|
||||||
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
|
@patch("llm.pool.HostPool.from_config")
|
||||||
|
@patch("llm.index.index_refs")
|
||||||
|
@patch("conf.connect.bib")
|
||||||
|
@patch("llm.config.load")
|
||||||
|
def test_migrate_runs_before_docket_complete_and_engine_is_reused(
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index,
|
||||||
|
mock_pool,
|
||||||
|
mock_iter,
|
||||||
|
mock_complete,
|
||||||
|
mock_migrate,
|
||||||
|
mock_engine,
|
||||||
|
):
|
||||||
|
"""index_docket_state may not exist yet on the first sealed run."""
|
||||||
|
order: list[str] = []
|
||||||
|
mock_migrate.side_effect = lambda e: order.append("migrate")
|
||||||
|
mock_complete.side_effect = lambda e, c: (order.append("complete"), {})[1]
|
||||||
|
mock_load.return_value = MagicMock()
|
||||||
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
|
||||||
|
mock_bib.return_value = store
|
||||||
|
mock_iter.return_value = iter([])
|
||||||
|
mock_index.return_value = _STATS
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["index"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert order == ["migrate", "complete"]
|
||||||
|
mock_engine.assert_called_once() # one engine, shared
|
||||||
|
assert mock_migrate.call_args.args[0] is mock_engine.return_value
|
||||||
|
assert mock_complete.call_args.args[0] is mock_engine.return_value
|
||||||
|
assert mock_index.call_args.kwargs["engine"] is mock_engine.return_value
|
||||||
|
|
||||||
|
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.migrate.migrate")
|
||||||
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.source.iter_comment_refs")
|
||||||
|
@patch("llm.pool.HostPool.from_config")
|
||||||
|
@patch("llm.index.index_refs")
|
||||||
|
@patch("conf.connect.bib")
|
||||||
|
@patch("llm.config.load")
|
||||||
|
def test_force_touches_neither_engine_nor_migrate(
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index,
|
||||||
|
mock_pool,
|
||||||
|
mock_iter,
|
||||||
|
mock_complete,
|
||||||
|
mock_migrate,
|
||||||
|
mock_engine,
|
||||||
|
):
|
||||||
|
mock_load.return_value = MagicMock()
|
||||||
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {"CMS-2019-0111": "s1"}
|
||||||
|
mock_bib.return_value = store
|
||||||
|
mock_iter.return_value = iter([])
|
||||||
|
mock_index.return_value = _STATS
|
||||||
|
|
||||||
|
result = runner.invoke(app, ["index", "--force"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mock_engine.assert_not_called()
|
||||||
|
mock_migrate.assert_not_called()
|
||||||
|
mock_complete.assert_not_called()
|
||||||
|
assert mock_index.call_args.kwargs["engine"] is None
|
||||||
158
tests/dev/test_dedupe_attachments.py
Normal file
158
tests/dev/test_dedupe_attachments.py
Normal file
@@ -0,0 +1,158 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import importlib.util
|
||||||
|
import sqlite3
|
||||||
|
import sys
|
||||||
|
from importlib import resources
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from bib.item import Source
|
||||||
|
from bib.store import Store
|
||||||
|
|
||||||
|
_SCRIPT = (
|
||||||
|
Path(__file__).resolve().parents[2] / "dev" / "scripts" / "dedupe_attachments.py"
|
||||||
|
)
|
||||||
|
spec = importlib.util.spec_from_file_location("dedupe_attachments", _SCRIPT)
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
sys.modules["dedupe_attachments"] = mod
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
|
||||||
|
|
||||||
|
def _seed(tmp_path: Path) -> tuple[Store, str]:
|
||||||
|
# Seed the item + duplicate attachment rows via a raw connection,
|
||||||
|
# *before* any Store ever opens this file. Store's schema guard adds
|
||||||
|
# the (item_id, filename) unique index the moment it sees zero
|
||||||
|
# duplicate groups — true the instant the attachments table exists —
|
||||||
|
# so seeding duplicates through a Store-opened connection can never
|
||||||
|
# succeed. This mirrors the real migration scenario: the duplicates
|
||||||
|
# were written by pre-Task-4 code that predates this guard.
|
||||||
|
db_path = tmp_path / "bib.sqlite"
|
||||||
|
storage = tmp_path / "storage"
|
||||||
|
ddl = resources.files("bib").joinpath("schema.sql").read_text()
|
||||||
|
raw = sqlite3.connect(str(db_path))
|
||||||
|
raw.executescript(ddl)
|
||||||
|
item = Source(title="T", url="https://x/1")
|
||||||
|
item.stamp_access()
|
||||||
|
row = item.to_row()
|
||||||
|
key = row["key"] or "SEEDKEY1"
|
||||||
|
row["key"] = key
|
||||||
|
raw.execute(
|
||||||
|
"""INSERT INTO items
|
||||||
|
(key, item_type, title, url, date_published,
|
||||||
|
access_date, abstract, institution, extra, extra_json)
|
||||||
|
VALUES (:key, :item_type, :title, :url,
|
||||||
|
:date_published, :access_date, :abstract,
|
||||||
|
:institution, :extra, :extra_json)""",
|
||||||
|
row,
|
||||||
|
)
|
||||||
|
item_id = raw.execute("SELECT id FROM items WHERE key=?", (key,)).fetchone()[0]
|
||||||
|
# Simulate the old non-idempotent attach: three rows, three copies.
|
||||||
|
for k in ("AAAAAAAA", "BBBBBBBB", "CCCCCCCC"):
|
||||||
|
d = storage / k
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF-dup")
|
||||||
|
raw.execute(
|
||||||
|
"INSERT INTO attachments (item_id, key, filename, content_type, storage_path) VALUES (?,?,?,?,?)",
|
||||||
|
(
|
||||||
|
item_id,
|
||||||
|
k,
|
||||||
|
"attachment_1.pdf",
|
||||||
|
"application/pdf",
|
||||||
|
str(d / "attachment_1.pdf"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
raw.commit()
|
||||||
|
raw.close()
|
||||||
|
s = Store(str(db_path), storage_dir=storage)
|
||||||
|
return s, key
|
||||||
|
|
||||||
|
|
||||||
|
def test_plan_finds_group_and_keeps_oldest(tmp_path: Path):
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
groups = mod.plan(s._con())
|
||||||
|
assert len(groups) == 1
|
||||||
|
g = groups[0]
|
||||||
|
assert g.keep == "AAAAAAAA"
|
||||||
|
assert sorted(g.remove) == ["BBBBBBBB", "CCCCCCCC"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_apply_removes_rows_and_files(tmp_path: Path):
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
rep = mod.apply(s._con(), mod.plan(s._con()))
|
||||||
|
assert rep.rows_removed == 2
|
||||||
|
assert rep.files_removed == 2
|
||||||
|
assert s._con().execute("SELECT count(*) FROM attachments").fetchone()[0] == 1
|
||||||
|
assert (tmp_path / "storage" / "AAAAAAAA" / "attachment_1.pdf").is_file()
|
||||||
|
assert not (tmp_path / "storage" / "BBBBBBBB").exists()
|
||||||
|
assert mod.plan(s._con()) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_refuses_group_with_differing_sizes(tmp_path: Path):
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
(tmp_path / "storage" / "CCCCCCCC" / "attachment_1.pdf").write_bytes(
|
||||||
|
b"different-longer"
|
||||||
|
)
|
||||||
|
groups = mod.plan(s._con())
|
||||||
|
assert groups[0].conflict is True
|
||||||
|
rep = mod.apply(s._con(), groups)
|
||||||
|
assert rep.rows_removed == 0 and rep.skipped_conflicts == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_unique_index_created_only_when_clean(tmp_path: Path):
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
s.close()
|
||||||
|
s2 = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
|
||||||
|
idx = {
|
||||||
|
r[0]
|
||||||
|
for r in s2._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
|
||||||
|
}
|
||||||
|
assert "idx_attachments_item_filename" not in idx
|
||||||
|
mod.apply(s2._con(), mod.plan(s2._con()))
|
||||||
|
s2.close()
|
||||||
|
s3 = Store(str(tmp_path / "bib.sqlite"), storage_dir=tmp_path / "storage")
|
||||||
|
idx = {
|
||||||
|
r[0]
|
||||||
|
for r in s3._con().execute("SELECT name FROM sqlite_master WHERE type='index'")
|
||||||
|
}
|
||||||
|
assert "idx_attachments_item_filename" in idx
|
||||||
|
|
||||||
|
|
||||||
|
class _FailAfterFirstDelete:
|
||||||
|
"""Connection proxy that dies partway through the transaction."""
|
||||||
|
|
||||||
|
def __init__(self, con: sqlite3.Connection) -> None:
|
||||||
|
self._con = con
|
||||||
|
self.deletes = 0
|
||||||
|
|
||||||
|
def execute(self, sql: str, *args):
|
||||||
|
if sql.lstrip().upper().startswith("DELETE"):
|
||||||
|
self.deletes += 1
|
||||||
|
if self.deletes == 2:
|
||||||
|
raise sqlite3.OperationalError("disk I/O error")
|
||||||
|
return self._con.execute(sql, *args)
|
||||||
|
|
||||||
|
|
||||||
|
def test_apply_removes_no_file_when_the_transaction_fails(tmp_path: Path):
|
||||||
|
"""A rollback restores the rows — so the files they point at must
|
||||||
|
still be there. Unlinking inside the transaction loses data."""
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
groups = mod.plan(s._con())
|
||||||
|
flaky = _FailAfterFirstDelete(s._con())
|
||||||
|
with pytest.raises(sqlite3.OperationalError):
|
||||||
|
mod.apply(flaky, groups)
|
||||||
|
assert s._con().execute("SELECT count(*) FROM attachments").fetchone()[0] == 3
|
||||||
|
for k in ("AAAAAAAA", "BBBBBBBB", "CCCCCCCC"):
|
||||||
|
assert (tmp_path / "storage" / k / "attachment_1.pdf").is_file()
|
||||||
|
|
||||||
|
|
||||||
|
def test_main_apply_reports_removed_keys(tmp_path: Path, capsys):
|
||||||
|
s, _ = _seed(tmp_path)
|
||||||
|
s.close()
|
||||||
|
rc = mod.main(["--db", str(tmp_path / "bib.sqlite"), "--apply"])
|
||||||
|
out = capsys.readouterr().out
|
||||||
|
assert rc == 0
|
||||||
|
assert "removed rows=2" in out
|
||||||
|
assert "removed keys (2)" in out
|
||||||
|
assert "BBBBBBBB" in out and "CCCCCCCC" in out
|
||||||
@@ -57,12 +57,14 @@ class TestIndexDocs:
|
|||||||
|
|
||||||
def test_unchanged_doc_skipped(self):
|
def test_unchanged_doc_skipped(self):
|
||||||
h = content_hash(DOC.text)
|
h = content_hash(DOC.text)
|
||||||
stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h)])
|
stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h, "")])
|
||||||
assert stats["skipped"] == 1
|
assert stats["skipped"] == 1
|
||||||
store.add_embeddings.assert_not_called()
|
store.add_embeddings.assert_not_called()
|
||||||
|
|
||||||
def test_changed_doc_deletes_old_chunks_first(self):
|
def test_changed_doc_deletes_old_chunks_first(self):
|
||||||
stats, store, conn, _ensure_hnsw = _run([DOC], state_rows=[("K1", "stalehash")])
|
stats, store, conn, _ensure_hnsw = _run(
|
||||||
|
[DOC], state_rows=[("K1", "stalehash", "")]
|
||||||
|
)
|
||||||
assert stats["indexed"] == 1
|
assert stats["indexed"] == 1
|
||||||
deletes = [
|
deletes = [
|
||||||
c
|
c
|
||||||
@@ -81,7 +83,9 @@ class TestIndexDocs:
|
|||||||
|
|
||||||
def test_force_reembeds_unchanged(self):
|
def test_force_reembeds_unchanged(self):
|
||||||
h = content_hash(DOC.text)
|
h = content_hash(DOC.text)
|
||||||
stats, store, _, _ensure_hnsw = _run([DOC], state_rows=[("K1", h)], force=True)
|
stats, store, _, _ensure_hnsw = _run(
|
||||||
|
[DOC], state_rows=[("K1", h, "")], force=True
|
||||||
|
)
|
||||||
assert stats["indexed"] == 1
|
assert stats["indexed"] == 1
|
||||||
|
|
||||||
def test_empty_doc_counts_skipped(self):
|
def test_empty_doc_counts_skipped(self):
|
||||||
@@ -91,19 +95,22 @@ class TestIndexDocs:
|
|||||||
|
|
||||||
|
|
||||||
class TestPageEnrichmentHook:
|
class TestPageEnrichmentHook:
|
||||||
def test_enriches_chunks_before_add(self, monkeypatch):
|
def test_enriches_only_docs_that_embed(self, monkeypatch):
|
||||||
"""index_docs runs llm.pages.enrich_pdf_pages on every doc's chunks."""
|
"""index_docs runs llm.pages.enrich_pdf_pages only on docs that reach
|
||||||
|
the embed path — a hash match must not open the doc's PDFs."""
|
||||||
from llm import index as index_mod
|
from llm import index as index_mod
|
||||||
|
|
||||||
seen = []
|
seen = []
|
||||||
|
monkeypatch.setattr(
|
||||||
def fake_enrich(doc, chunks):
|
index_mod,
|
||||||
seen.append(doc.key)
|
"enrich_pdf_pages",
|
||||||
return chunks
|
lambda doc, chunks: seen.append(doc.key) or chunks,
|
||||||
|
)
|
||||||
monkeypatch.setattr(index_mod, "enrich_pdf_pages", fake_enrich)
|
|
||||||
_run([DOC], [])
|
_run([DOC], [])
|
||||||
assert seen == ["K1"]
|
assert seen == ["K1"]
|
||||||
|
seen.clear()
|
||||||
|
_run([DOC], [("K1", content_hash(DOC.text), "")])
|
||||||
|
assert seen == []
|
||||||
|
|
||||||
|
|
||||||
class TestDeleteOldChunksFallback:
|
class TestDeleteOldChunksFallback:
|
||||||
|
|||||||
218
tests/llm/test_index_refs.py
Normal file
218
tests/llm/test_index_refs.py
Normal file
@@ -0,0 +1,218 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
from llm.chunk import Doc, content_hash
|
||||||
|
from llm.config import LlmConfig
|
||||||
|
from llm.index import index_refs
|
||||||
|
from llm.source import DocRef
|
||||||
|
|
||||||
|
CFG = LlmConfig(
|
||||||
|
ollama_hosts=("http://h1:11434",),
|
||||||
|
embed_model="m",
|
||||||
|
instruct_model="g",
|
||||||
|
embed_dim=768,
|
||||||
|
build_ann_index=False,
|
||||||
|
pg_host="x",
|
||||||
|
pg_port=5432,
|
||||||
|
pg_db="llm",
|
||||||
|
pg_user="llm",
|
||||||
|
)
|
||||||
|
DOC = Doc(key="K1", text="Some body text.", metadata={"docket": "D"})
|
||||||
|
|
||||||
|
|
||||||
|
def _ref(fp="fp1", docket="D", loads=None, doc=DOC):
|
||||||
|
def _load():
|
||||||
|
if loads is not None:
|
||||||
|
loads.append(doc.key)
|
||||||
|
return doc
|
||||||
|
|
||||||
|
return DocRef(
|
||||||
|
key=doc.key, collection="comments", docket=docket, fingerprint=fp, load=_load
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _maybe_chunk_doc(fn):
|
||||||
|
"""Patch llm.index.chunk_doc only when a test asks for it."""
|
||||||
|
from contextlib import nullcontext
|
||||||
|
|
||||||
|
return nullcontext() if fn is None else patch("llm.index.chunk_doc", side_effect=fn)
|
||||||
|
|
||||||
|
|
||||||
|
def _run(
|
||||||
|
refs,
|
||||||
|
state_rows,
|
||||||
|
*,
|
||||||
|
force=False,
|
||||||
|
sealed=None,
|
||||||
|
mark_complete=True,
|
||||||
|
complete_rows=(),
|
||||||
|
chunk_doc=None,
|
||||||
|
):
|
||||||
|
store = MagicMock()
|
||||||
|
engine = MagicMock()
|
||||||
|
conn = engine.begin.return_value.__enter__.return_value
|
||||||
|
|
||||||
|
def fake_execute(clause, *a, **k):
|
||||||
|
sql = str(clause)
|
||||||
|
r = MagicMock()
|
||||||
|
if "FROM index_docket_state" in sql:
|
||||||
|
r.fetchall.return_value = list(complete_rows)
|
||||||
|
elif "FROM index_state" in sql:
|
||||||
|
r.fetchall.return_value = state_rows
|
||||||
|
else:
|
||||||
|
r.fetchall.return_value = []
|
||||||
|
return r
|
||||||
|
|
||||||
|
conn.execute.side_effect = fake_execute
|
||||||
|
with (
|
||||||
|
patch("llm.index._engine", return_value=engine),
|
||||||
|
patch("llm.index.vectorstore", return_value=store),
|
||||||
|
patch("llm.index.embed_texts", return_value=[[0.0] * 3]),
|
||||||
|
patch("llm.index.ensure_hnsw"),
|
||||||
|
patch("llm.index.enrich_pdf_pages", side_effect=lambda d, c: c) as enrich,
|
||||||
|
patch("llm.index.HostPool") as MockPool,
|
||||||
|
_maybe_chunk_doc(chunk_doc),
|
||||||
|
):
|
||||||
|
MockPool.return_value.check.return_value = ["http://h1:11434"]
|
||||||
|
stats = index_refs(
|
||||||
|
refs,
|
||||||
|
collection="comments",
|
||||||
|
cfg=CFG,
|
||||||
|
pool=MockPool.return_value,
|
||||||
|
force=force,
|
||||||
|
sealed=sealed,
|
||||||
|
mark_complete=mark_complete,
|
||||||
|
)
|
||||||
|
return stats, store, conn, enrich
|
||||||
|
|
||||||
|
|
||||||
|
def test_fingerprint_match_skips_without_load():
|
||||||
|
loads = []
|
||||||
|
stats, store, _, enrich = _run([_ref("fp1", loads=loads)], [("K1", "h-old", "fp1")])
|
||||||
|
assert stats["fingerprint_skipped"] == 1 and stats["indexed"] == 0
|
||||||
|
assert loads == []
|
||||||
|
store.add_embeddings.assert_not_called()
|
||||||
|
enrich.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
def test_hash_match_updates_fingerprint_without_embedding():
|
||||||
|
h = content_hash(DOC.text)
|
||||||
|
stats, store, conn, enrich = _run([_ref("fp2")], [("K1", h, "fp1")])
|
||||||
|
assert stats["hash_skipped"] == 1 and stats["indexed"] == 0
|
||||||
|
store.add_embeddings.assert_not_called()
|
||||||
|
enrich.assert_not_called()
|
||||||
|
upd = [
|
||||||
|
c
|
||||||
|
for c in conn.execute.call_args_list
|
||||||
|
if "UPDATE index_state SET fingerprint" in str(c.args[0])
|
||||||
|
]
|
||||||
|
assert len(upd) == 1 and upd[0].args[1]["f"] == "fp2"
|
||||||
|
|
||||||
|
|
||||||
|
def test_changed_doc_embeds_and_records_fingerprint():
|
||||||
|
stats, store, conn, enrich = _run([_ref("fp2")], [("K1", "stale", "fp1")])
|
||||||
|
assert stats["indexed"] == 1 and stats["chunks"] == 1
|
||||||
|
store.add_embeddings.assert_called_once()
|
||||||
|
enrich.assert_called_once()
|
||||||
|
ins = [
|
||||||
|
c
|
||||||
|
for c in conn.execute.call_args_list
|
||||||
|
if "INSERT INTO index_state" in str(c.args[0])
|
||||||
|
]
|
||||||
|
assert ins[0].args[1]["f"] == "fp2"
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_fingerprint_never_matches():
|
||||||
|
stats, store, _, _ = _run([_ref("")], [("K1", "stale", "")])
|
||||||
|
assert stats["indexed"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_force_ignores_fingerprint_and_hash():
|
||||||
|
h = content_hash(DOC.text)
|
||||||
|
stats, store, _, _ = _run([_ref("fp1")], [("K1", h, "fp1")], force=True)
|
||||||
|
assert stats["indexed"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_none_counts_skipped():
|
||||||
|
ref = DocRef(
|
||||||
|
key="K9", collection="comments", docket="D", fingerprint="x", load=lambda: None
|
||||||
|
)
|
||||||
|
stats, *_ = _run([ref], [])
|
||||||
|
assert stats["skipped"] == 1 and stats["indexed"] == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_sealed_docket_marked_complete_after_clean_run():
|
||||||
|
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={"D": "2026-10-20T00:00:00Z"})
|
||||||
|
assert stats["docket_complete"] == 1
|
||||||
|
ins = [
|
||||||
|
c
|
||||||
|
for c in conn.execute.call_args_list
|
||||||
|
if "INSERT INTO index_docket_state" in str(c.args[0])
|
||||||
|
]
|
||||||
|
assert ins[0].args[1] == {"c": "comments", "d": "D", "s": "2026-10-20T00:00:00Z"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_not_marked_when_mark_complete_false_or_unsealed():
|
||||||
|
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={"D": "s"}, mark_complete=False)
|
||||||
|
assert stats["docket_complete"] == 0
|
||||||
|
stats, _, conn, _ = _run([_ref("fp1")], [], sealed={})
|
||||||
|
assert stats["docket_complete"] == 0
|
||||||
|
|
||||||
|
|
||||||
|
def _state_inserts(conn):
|
||||||
|
return [
|
||||||
|
c
|
||||||
|
for c in conn.execute.call_args_list
|
||||||
|
if "INSERT INTO index_state" in str(c.args[0])
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _docket_inserts(conn):
|
||||||
|
return [
|
||||||
|
c
|
||||||
|
for c in conn.execute.call_args_list
|
||||||
|
if "INSERT INTO index_docket_state" in str(c.args[0])
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_text_ref_is_stamped_final_and_counts_toward_completion():
|
||||||
|
"""No text is a finished state, not a failure: stamp it so the next
|
||||||
|
run fingerprint-skips it, and let the docket still seal complete."""
|
||||||
|
blank = DocRef(
|
||||||
|
key="K9", collection="comments", docket="D", fingerprint="x", load=lambda: None
|
||||||
|
)
|
||||||
|
stats, store, conn, _ = _run([blank], [], sealed={"D": "s"})
|
||||||
|
assert stats["skipped"] == 1 and stats["indexed"] == 0
|
||||||
|
store.add_embeddings.assert_not_called()
|
||||||
|
ins = _state_inserts(conn)
|
||||||
|
assert len(ins) == 1
|
||||||
|
assert ins[0].args[1] == {"k": "K9", "c": "comments", "h": "", "n": 0, "f": "x"}
|
||||||
|
assert stats["docket_complete"] == 1
|
||||||
|
assert _docket_inserts(conn)
|
||||||
|
|
||||||
|
|
||||||
|
def test_stamped_empty_row_is_fingerprint_skipped_on_the_next_run():
|
||||||
|
loads = []
|
||||||
|
blank = DocRef(
|
||||||
|
key="K9",
|
||||||
|
collection="comments",
|
||||||
|
docket="D",
|
||||||
|
fingerprint="x",
|
||||||
|
load=lambda: loads.append("K9"),
|
||||||
|
)
|
||||||
|
stats, _, conn, _ = _run([blank], [("K9", "", "x")])
|
||||||
|
assert stats["fingerprint_skipped"] == 1 and stats["skipped"] == 0
|
||||||
|
assert loads == [] # stamped-empty rows are never re-loaded
|
||||||
|
assert _state_inserts(conn) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_sealed_docket_not_marked_when_a_ref_yields_no_chunks():
|
||||||
|
"""Text that chunks to nothing is a real failure — it blocks the seal
|
||||||
|
and is not stamped."""
|
||||||
|
stats, _, conn, _ = _run(
|
||||||
|
[_ref("fp1")], [], sealed={"D": "s"}, chunk_doc=lambda d: []
|
||||||
|
)
|
||||||
|
assert stats["skipped"] == 1 and stats["docket_complete"] == 0
|
||||||
|
assert _state_inserts(conn) == []
|
||||||
|
assert _docket_inserts(conn) == []
|
||||||
@@ -24,3 +24,20 @@ class TestDdl:
|
|||||||
executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list)
|
executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list)
|
||||||
assert "vector(768)" in executed
|
assert "vector(768)" in executed
|
||||||
assert "USING hnsw" in executed
|
assert "USING hnsw" in executed
|
||||||
|
|
||||||
|
def test_fingerprint_column_and_docket_state(self):
|
||||||
|
assert "fingerprint" in migrate.INDEX_STATE_DDL
|
||||||
|
assert "ADD COLUMN IF NOT EXISTS fingerprint" in migrate.INDEX_STATE_ALTER
|
||||||
|
assert (
|
||||||
|
"CREATE TABLE IF NOT EXISTS index_docket_state"
|
||||||
|
in migrate.INDEX_DOCKET_STATE_DDL
|
||||||
|
)
|
||||||
|
assert "PRIMARY KEY (collection, docket)" in migrate.INDEX_DOCKET_STATE_DDL
|
||||||
|
|
||||||
|
def test_migrate_runs_alter_and_docket_state(self):
|
||||||
|
engine = MagicMock()
|
||||||
|
conn = engine.begin.return_value.__enter__.return_value
|
||||||
|
migrate.migrate(engine)
|
||||||
|
executed = " ".join(str(call.args[0]) for call in conn.execute.call_args_list)
|
||||||
|
assert "ADD COLUMN IF NOT EXISTS fingerprint" in executed
|
||||||
|
assert "index_docket_state" in executed
|
||||||
|
|||||||
276
tests/llm/test_source_refs.py
Normal file
276
tests/llm/test_source_refs.py
Normal file
@@ -0,0 +1,276 @@
|
|||||||
|
"""llm.source — lazy DocRefs: fingerprints without loads, docket skips,
|
||||||
|
lazy Zotero snapshot."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from bib.item import Item, Rule
|
||||||
|
from bib.store import Store
|
||||||
|
from llm.source import (
|
||||||
|
DocRef,
|
||||||
|
ZoteroPdfIndex,
|
||||||
|
iter_comment_refs,
|
||||||
|
iter_corpus_refs,
|
||||||
|
iter_rule_refs,
|
||||||
|
)
|
||||||
|
|
||||||
|
DOCKET = "CMS-2019-0111"
|
||||||
|
CID = f"{DOCKET}-0042"
|
||||||
|
COMBINED = f"---\ncomment_id: {CID}\ndocket_id: {DOCKET}\n---\n\nWe object.\n"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def store(tmp_path):
|
||||||
|
s = Store(":memory:", storage_dir=tmp_path / "storage")
|
||||||
|
key = s.create(
|
||||||
|
Item(
|
||||||
|
item_type="report",
|
||||||
|
title="A comment",
|
||||||
|
url=f"https://www.regulations.gov/comment/{CID}",
|
||||||
|
abstract="Inline.",
|
||||||
|
date_published="2019-09-27",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
for tag in ("doctype:comment", "year:2019", f"reg-docket:{DOCKET}"):
|
||||||
|
s.add_tag(key, tag)
|
||||||
|
s._comment_key = key
|
||||||
|
return s
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def root(tmp_path):
|
||||||
|
d = tmp_path / DOCKET / CID
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
(d / "combined.md").write_text(COMBINED)
|
||||||
|
return tmp_path
|
||||||
|
|
||||||
|
|
||||||
|
class TestCommentRefs:
|
||||||
|
def test_ref_has_fingerprint_and_lazy_load(self, store, root, monkeypatch):
|
||||||
|
reads = []
|
||||||
|
real = Path.read_text
|
||||||
|
monkeypatch.setattr(
|
||||||
|
Path,
|
||||||
|
"read_text",
|
||||||
|
lambda self, *a, **k: reads.append(self) or real(self, *a, **k),
|
||||||
|
)
|
||||||
|
refs = list(iter_comment_refs(store, docket=DOCKET, root=root))
|
||||||
|
assert len(refs) == 1
|
||||||
|
r = refs[0]
|
||||||
|
assert isinstance(r, DocRef)
|
||||||
|
assert (r.key, r.collection, r.docket) == (
|
||||||
|
store._comment_key,
|
||||||
|
"comments",
|
||||||
|
DOCKET,
|
||||||
|
)
|
||||||
|
assert len(r.fingerprint) > 64 and "|" in r.fingerprint
|
||||||
|
assert reads == [] # nothing read yet
|
||||||
|
doc = r.load()
|
||||||
|
assert "We object" in doc.text
|
||||||
|
assert doc.metadata["year"] == "2019" and doc.metadata["docket"] == DOCKET
|
||||||
|
assert reads # load() read combined.md
|
||||||
|
|
||||||
|
def test_fingerprint_changes_with_new_attachment(self, store, root):
|
||||||
|
f1 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
|
||||||
|
(root / DOCKET / CID / "attachment_1.pdf").write_bytes(b"%PDF")
|
||||||
|
f2 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
|
||||||
|
assert f1 != f2
|
||||||
|
|
||||||
|
def test_fingerprint_changes_with_updated_at(self, store, root):
|
||||||
|
f1 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
|
||||||
|
store._con().execute(
|
||||||
|
"UPDATE items SET updated_at='2030-01-01T00:00:00Z' WHERE key=?",
|
||||||
|
(store._comment_key,),
|
||||||
|
)
|
||||||
|
f2 = next(iter_comment_refs(store, docket=DOCKET, root=root)).fingerprint
|
||||||
|
assert f1 != f2
|
||||||
|
|
||||||
|
def test_skip_dockets_excludes_rows(self, store, root):
|
||||||
|
assert list(iter_comment_refs(store, root=root, skip_dockets={DOCKET})) == []
|
||||||
|
|
||||||
|
def test_unextracted_load_falls_back_to_abstract(self, store, tmp_path):
|
||||||
|
r = next(iter_comment_refs(store, docket=DOCKET, root=tmp_path))
|
||||||
|
assert r.load().text == "Inline."
|
||||||
|
|
||||||
|
def test_empty_comment_load_returns_none(self, store, tmp_path):
|
||||||
|
store._con().execute(
|
||||||
|
"UPDATE items SET abstract='' WHERE key=?", (store._comment_key,)
|
||||||
|
)
|
||||||
|
r = next(iter_comment_refs(store, docket=DOCKET, root=tmp_path))
|
||||||
|
assert r.load() is None
|
||||||
|
|
||||||
|
def test_single_query_for_year(self, store, root, monkeypatch):
|
||||||
|
"""No per-comment year lookups: exactly one SELECT on items for the listing."""
|
||||||
|
calls: list[str] = []
|
||||||
|
store._con().set_trace_callback(calls.append)
|
||||||
|
list(iter_comment_refs(store, docket=DOCKET, root=root))
|
||||||
|
store._con().set_trace_callback(None)
|
||||||
|
selects = [c for c in calls if c.lstrip().upper().startswith("SELECT")]
|
||||||
|
assert len(selects) == 1
|
||||||
|
|
||||||
|
|
||||||
|
def _anchored_rule(store) -> str:
|
||||||
|
"""A rule with one grabbed FR anchor paragraph (sha256 "abc")."""
|
||||||
|
key = store.create(Rule(title="R", url="https://fr/1", document_number="2019-1"))
|
||||||
|
store._con().execute(
|
||||||
|
"INSERT INTO fr_anchor_docs (item_key, document_number, html_url, start_page, end_page, fr_volume, sha256) VALUES (?,?,?,?,?,?,?)",
|
||||||
|
(key, "2019-1", "https://fr/1", 1, 2, 84, "abc"),
|
||||||
|
)
|
||||||
|
store._con().execute(
|
||||||
|
"INSERT INTO fr_anchors (item_key, p_id, page, ordinal, text) VALUES (?,?,?,?,?)",
|
||||||
|
(key, 1, 1, 1, "Para one."),
|
||||||
|
)
|
||||||
|
return key
|
||||||
|
|
||||||
|
|
||||||
|
class TestRuleRefs:
|
||||||
|
def test_fingerprint_from_anchor_sha(self, store):
|
||||||
|
key = _anchored_rule(store)
|
||||||
|
refs = list(iter_rule_refs(store))
|
||||||
|
assert [r.key for r in refs] == [key]
|
||||||
|
assert refs[0].fingerprint.startswith("anchors:abc|")
|
||||||
|
assert refs[0].collection == "rules" and refs[0].docket is None
|
||||||
|
assert refs[0].load().text == "Para one."
|
||||||
|
|
||||||
|
def test_anchor_fingerprint_changes_with_updated_at(self, store):
|
||||||
|
"""The sha covers the FR body, not the item's own metadata."""
|
||||||
|
key = _anchored_rule(store)
|
||||||
|
f1 = next(iter_rule_refs(store)).fingerprint
|
||||||
|
store._con().execute(
|
||||||
|
"UPDATE items SET updated_at='2030-01-01T00:00:00Z' WHERE key=?", (key,)
|
||||||
|
)
|
||||||
|
assert next(iter_rule_refs(store)).fingerprint != f1
|
||||||
|
|
||||||
|
def test_tag_filter_selects_only_tagged_rules(self, store):
|
||||||
|
tagged = store.create(Rule(title="Tagged", document_number="2019-2"))
|
||||||
|
store.add_tag(tagged, "project:pfs")
|
||||||
|
store.create(Rule(title="Untagged", document_number="2019-3"))
|
||||||
|
assert [r.key for r in iter_rule_refs(store, tag="project:pfs")] == [tagged]
|
||||||
|
|
||||||
|
|
||||||
|
class TestCorpusRefs:
|
||||||
|
def test_excludes_comments_and_is_lazy(self, store):
|
||||||
|
key = store.create(
|
||||||
|
Item(
|
||||||
|
item_type="report", title="Report", url="https://x/r", abstract="Body."
|
||||||
|
)
|
||||||
|
)
|
||||||
|
refs = list(iter_corpus_refs(store))
|
||||||
|
assert [r.key for r in refs] == [key]
|
||||||
|
assert refs[0].collection == "corpus"
|
||||||
|
assert refs[0].load().metadata["kind"] == "corpus"
|
||||||
|
|
||||||
|
def test_zotero_not_consulted_until_load(self, store, tmp_path):
|
||||||
|
store.create(
|
||||||
|
Item(
|
||||||
|
item_type="report", title="Report", url="https://x/r", abstract="Body."
|
||||||
|
)
|
||||||
|
)
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
class Z(ZoteroPdfIndex):
|
||||||
|
def pdfs_for(self, key):
|
||||||
|
calls.append(key)
|
||||||
|
return []
|
||||||
|
|
||||||
|
refs = list(iter_corpus_refs(store, zotero=Z({})))
|
||||||
|
assert calls == []
|
||||||
|
refs[0].load()
|
||||||
|
assert calls
|
||||||
|
|
||||||
|
def test_tag_filter_selects_only_tagged_items(self, store):
|
||||||
|
tagged = store.create(Item(item_type="report", title="T", abstract="Body."))
|
||||||
|
store.add_tag(tagged, "project:pfs")
|
||||||
|
store.create(Item(item_type="report", title="U", abstract="Body."))
|
||||||
|
assert [r.key for r in iter_corpus_refs(store, tag="project:pfs")] == [tagged]
|
||||||
|
|
||||||
|
|
||||||
|
class TestLazyZotero:
|
||||||
|
def test_no_copy_until_first_lookup(self, tmp_path):
|
||||||
|
src = tmp_path / "zotero.sqlite"
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
con = sqlite3.connect(src)
|
||||||
|
con.executescript(
|
||||||
|
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
|
||||||
|
)
|
||||||
|
con.close()
|
||||||
|
snap_dir = tmp_path / "snap"
|
||||||
|
z = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
|
||||||
|
assert not (snap_dir / "zotero.sqlite").exists()
|
||||||
|
assert z.pdfs_for("ABCD1234") == []
|
||||||
|
assert (snap_dir / "zotero.sqlite").exists()
|
||||||
|
m1 = (snap_dir / "zotero.sqlite").stat().st_mtime_ns
|
||||||
|
z2 = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
|
||||||
|
z2.pdfs_for("ABCD1234")
|
||||||
|
assert (
|
||||||
|
snap_dir / "zotero.sqlite"
|
||||||
|
).stat().st_mtime_ns == m1 # source unchanged → no recopy
|
||||||
|
bump = src.stat().st_mtime_ns + 10**9
|
||||||
|
os.utime(src, ns=(bump, bump)) # source now newer than the snapshot
|
||||||
|
z3 = ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir)
|
||||||
|
z3.pdfs_for("ABCD1234")
|
||||||
|
assert (snap_dir / "zotero.sqlite").stat().st_mtime_ns != m1
|
||||||
|
|
||||||
|
def test_newer_wal_forces_a_recopy(self, tmp_path):
|
||||||
|
"""Zotero commits into zotero.sqlite-wal and only touches the main
|
||||||
|
file at a checkpoint — the WAL's mtime has to count too."""
|
||||||
|
src = tmp_path / "zotero.sqlite"
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
con = sqlite3.connect(src)
|
||||||
|
con.executescript(
|
||||||
|
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
|
||||||
|
)
|
||||||
|
con.close()
|
||||||
|
snap_dir = tmp_path / "snap"
|
||||||
|
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
|
||||||
|
snap = snap_dir / "zotero.sqlite"
|
||||||
|
m1 = snap.stat().st_mtime_ns
|
||||||
|
|
||||||
|
wal = tmp_path / "zotero.sqlite-wal"
|
||||||
|
wal.write_bytes(b"wal")
|
||||||
|
bump = m1 + 10**9
|
||||||
|
os.utime(wal, ns=(bump, bump)) # main file untouched, WAL is newer
|
||||||
|
|
||||||
|
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
|
||||||
|
m2 = snap.stat().st_mtime_ns
|
||||||
|
assert m2 != m1 # re-copied
|
||||||
|
# and the snapshot now records what it captured, so the next run
|
||||||
|
# doesn't re-copy 1.95 GB for the same unchanged WAL
|
||||||
|
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
|
||||||
|
assert snap.stat().st_mtime_ns == m2
|
||||||
|
|
||||||
|
def test_failed_copy_does_not_stamp_the_stale_snapshot(self, tmp_path, monkeypatch):
|
||||||
|
"""Stamping after a failed copy would pass a stale snapshot off as
|
||||||
|
current forever."""
|
||||||
|
import shutil
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
src = tmp_path / "zotero.sqlite"
|
||||||
|
con = sqlite3.connect(src)
|
||||||
|
con.executescript(
|
||||||
|
"CREATE TABLE items(itemID INTEGER, key TEXT); CREATE TABLE itemAttachments(itemID INTEGER, parentItemID INTEGER, path TEXT); CREATE TABLE deletedItems(itemID INTEGER);"
|
||||||
|
)
|
||||||
|
con.close()
|
||||||
|
snap_dir = tmp_path / "snap"
|
||||||
|
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for("ABCD1234")
|
||||||
|
snap = snap_dir / "zotero.sqlite"
|
||||||
|
m1 = snap.stat().st_mtime_ns
|
||||||
|
|
||||||
|
bump = m1 + 10**9
|
||||||
|
os.utime(src, ns=(bump, bump)) # source now newer → a copy is due
|
||||||
|
monkeypatch.setattr(
|
||||||
|
shutil, "copy2", lambda *a, **k: (_ for _ in ()).throw(OSError("no space"))
|
||||||
|
)
|
||||||
|
assert (
|
||||||
|
ZoteroPdfIndex.lazy(src, tmp_path / "storage", snap_dir).pdfs_for(
|
||||||
|
"ABCD1234"
|
||||||
|
)
|
||||||
|
== []
|
||||||
|
)
|
||||||
|
assert snap.stat().st_mtime_ns == m1 # untouched, so the next run retries
|
||||||
126
tests/rex/comments/test_currency.py
Normal file
126
tests/rex/comments/test_currency.py
Normal file
@@ -0,0 +1,126 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from rex.comments.combine import is_current, source_paths
|
||||||
|
from rex.comments.walker import walk_and_extract
|
||||||
|
|
||||||
|
|
||||||
|
def _dir(tmp_path: Path, docket="CMS-2024-0001", cid="CMS-2024-0001-0001") -> Path:
|
||||||
|
d = tmp_path / docket / cid
|
||||||
|
d.mkdir(parents=True)
|
||||||
|
return d
|
||||||
|
|
||||||
|
|
||||||
|
def test_source_paths_excludes_outputs(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"x")
|
||||||
|
(d / "attachment_1.pdf.md").write_text("sibling")
|
||||||
|
(d / "combined.md").write_text("c")
|
||||||
|
(d / "combined.md.tmp").write_text("t")
|
||||||
|
(d / ".hidden").write_text("h")
|
||||||
|
assert [p.name for p in source_paths(d)] == ["attachment_1.pdf"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_is_current_false_without_combined(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"x")
|
||||||
|
assert is_current(d) is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_is_current_true_when_combined_newer(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"x")
|
||||||
|
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
|
||||||
|
(d / "combined.md").write_text("c")
|
||||||
|
assert is_current(d) is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_is_current_false_when_source_newer(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "combined.md").write_text("c")
|
||||||
|
os.utime(d / "combined.md", ns=(1_000, 1_000))
|
||||||
|
(d / "attachment_2.pdf").write_bytes(b"new")
|
||||||
|
assert is_current(d) is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_walker_reextracts_stale_dir(tmp_path: Path, monkeypatch):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "combined.md").write_text("---\ncomment_id: x\n---\n\nold\n")
|
||||||
|
os.utime(d / "combined.md", ns=(1_000, 1_000))
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF") # newer than combined.md
|
||||||
|
called = []
|
||||||
|
monkeypatch.setattr(
|
||||||
|
"rex.comments.walker.extract_comment",
|
||||||
|
lambda cdir, **kw: called.append(cdir) or (cdir / "combined.md"),
|
||||||
|
)
|
||||||
|
stats = walk_and_extract(tmp_path, inline_body_lookup=lambda _c: "", workers=1)
|
||||||
|
assert stats["written"] == 1 and called == [d]
|
||||||
|
|
||||||
|
|
||||||
|
def test_walker_skips_current_dir_without_reading(tmp_path: Path, monkeypatch):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF")
|
||||||
|
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
|
||||||
|
(d / "combined.md").write_text("---\ncomment_id: x\n---\n\nbody\n")
|
||||||
|
reads = []
|
||||||
|
real_read_text = Path.read_text
|
||||||
|
|
||||||
|
def spy(self, *a, **kw):
|
||||||
|
if self.name == "combined.md":
|
||||||
|
reads.append(self)
|
||||||
|
return real_read_text(self, *a, **kw)
|
||||||
|
|
||||||
|
monkeypatch.setattr(Path, "read_text", spy)
|
||||||
|
seen = []
|
||||||
|
stats = walk_and_extract(
|
||||||
|
tmp_path,
|
||||||
|
inline_body_lookup=lambda _c: "",
|
||||||
|
workers=1,
|
||||||
|
on_extracted=lambda cid, _d: seen.append(cid),
|
||||||
|
)
|
||||||
|
assert stats == {"written": 0, "skipped": 1, "failed": 0}
|
||||||
|
assert reads == [] # skipped dirs are not read
|
||||||
|
assert seen == [] # and the callback does not fire without --reattach
|
||||||
|
|
||||||
|
|
||||||
|
def test_walker_reattach_fires_callback_for_skipped(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF")
|
||||||
|
os.utime(d / "attachment_1.pdf", ns=(1_000, 1_000))
|
||||||
|
(d / "combined.md").write_text(
|
||||||
|
"---\ncomment_id: x\ndocket_id: y\n---\n\n## attachment_1.pdf\n\nbody\n"
|
||||||
|
)
|
||||||
|
seen = []
|
||||||
|
walk_and_extract(
|
||||||
|
tmp_path,
|
||||||
|
inline_body_lookup=lambda _c: "",
|
||||||
|
workers=1,
|
||||||
|
reattach=True,
|
||||||
|
on_extracted=lambda cid, _d: seen.append(cid),
|
||||||
|
)
|
||||||
|
assert seen == [d.name]
|
||||||
|
assert (d / "attachment_1.pdf.md").is_file() # siblings derived on reattach
|
||||||
|
|
||||||
|
|
||||||
|
def test_walker_skip_dockets(tmp_path: Path):
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "attachment_1.pdf").write_bytes(b"%PDF")
|
||||||
|
stats = walk_and_extract(
|
||||||
|
tmp_path,
|
||||||
|
inline_body_lookup=lambda _c: "",
|
||||||
|
workers=1,
|
||||||
|
skip_dockets={"CMS-2024-0001"},
|
||||||
|
)
|
||||||
|
assert stats == {"written": 0, "skipped": 0, "failed": 0, "skipped_sealed": 1}
|
||||||
|
assert not (d / "combined.md").exists()
|
||||||
|
|
||||||
|
|
||||||
|
def test_is_current_ignores_a_source_that_vanished(tmp_path: Path, monkeypatch):
|
||||||
|
"""A file removed between iterdir() and stat() must not kill the run."""
|
||||||
|
d = _dir(tmp_path)
|
||||||
|
(d / "combined.md").write_text("c")
|
||||||
|
ghost = d / "attachment_9.pdf"
|
||||||
|
monkeypatch.setattr("rex.comments.combine.source_paths", lambda _d: [ghost])
|
||||||
|
assert is_current(d) is True
|
||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import fitz
|
import fitz
|
||||||
@@ -44,6 +45,7 @@ def test_walk_and_extract_writes_all_combined(tmp_path: Path):
|
|||||||
def test_walk_and_extract_skips_existing(tmp_path: Path):
|
def test_walk_and_extract_skips_existing(tmp_path: Path):
|
||||||
a, _b = _setup_two_comments(tmp_path)
|
a, _b = _setup_two_comments(tmp_path)
|
||||||
(a / "combined.md").write_text("---\ncomment_id: x\n---\n\npre-existing\n")
|
(a / "combined.md").write_text("---\ncomment_id: x\n---\n\npre-existing\n")
|
||||||
|
os.utime(a / "attachment_1.pdf", ns=(1_000, 1_000))
|
||||||
|
|
||||||
stats = walk_and_extract(tmp_path, inline_body_lookup=_stub_inline_body, workers=1)
|
stats = walk_and_extract(tmp_path, inline_body_lookup=_stub_inline_body, workers=1)
|
||||||
|
|
||||||
@@ -96,10 +98,10 @@ def test_on_extracted_called_for_written_dirs(tmp_path: Path):
|
|||||||
|
|
||||||
|
|
||||||
def test_on_extracted_called_for_skipped_dirs(tmp_path: Path):
|
def test_on_extracted_called_for_skipped_dirs(tmp_path: Path):
|
||||||
"""Pre-existing combined.md still gets the callback, and any missing
|
"""With reattach=True, pre-existing combined.md still gets the callback,
|
||||||
sibling MDs are derived from it before the callback fires — so
|
and any missing sibling MDs are derived from it before the callback
|
||||||
previously-extracted dirs end up with the same set of MDs as
|
fires — so previously-extracted dirs end up with the same set of MDs
|
||||||
freshly-written ones."""
|
as freshly-written ones."""
|
||||||
a, b = _setup_two_comments(tmp_path)
|
a, b = _setup_two_comments(tmp_path)
|
||||||
# Pre-write a stale combined.md with one section so derive_siblings
|
# Pre-write a stale combined.md with one section so derive_siblings
|
||||||
# has something to split.
|
# has something to split.
|
||||||
@@ -112,6 +114,7 @@ def test_on_extracted_called_for_skipped_dirs(tmp_path: Path):
|
|||||||
tmp_path,
|
tmp_path,
|
||||||
inline_body_lookup=_stub_inline_body,
|
inline_body_lookup=_stub_inline_body,
|
||||||
workers=1,
|
workers=1,
|
||||||
|
reattach=True,
|
||||||
on_extracted=lambda cid, _d: seen.append(cid),
|
on_extracted=lambda cid, _d: seen.append(cid),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -13,9 +13,11 @@ from unittest.mock import MagicMock, patch
|
|||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
# ═══════════════════════════════════════════════════════════════════
|
# ═══════════════════════════════════════════════════════════════════
|
||||||
# Priority 1: cli/bib.py (14 lines)
|
# Priority 1: cli/bib.py
|
||||||
# Lines: 113, 175, 184, 185, 263, 265, 268, 276, 277, 281, 282,
|
# Lines: 113, 175, 184, 185, 325
|
||||||
# 285, 286, 325
|
# (the fetch-pfs-comments block moved to tests/bib/test_walk_docket.py
|
||||||
|
# and tests/cli/test_bib_fetch_sealed.py when that loop moved into
|
||||||
|
# walk_docket — MagicMock-store coverage of deleted lines proved nothing)
|
||||||
# ═══════════════════════════════════════════════════════════════════
|
# ═══════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
|
||||||
@@ -53,244 +55,6 @@ class TestCliBibDiscoverPfsRulesTranslateError:
|
|||||||
assert "skipped" in result.output # line 113 hit via exception branch
|
assert "skipped" in result.output # line 113 hit via exception branch
|
||||||
|
|
||||||
|
|
||||||
class TestCliBibFetchDocketComments:
|
|
||||||
"""Lines 175, 184, 185: early return on limit, progress echo, commit."""
|
|
||||||
|
|
||||||
def test_fetch_docket_comments_with_limit(self):
|
|
||||||
from typer.testing import CliRunner
|
|
||||||
|
|
||||||
from cli.bib import app
|
|
||||||
|
|
||||||
runner = CliRunner()
|
|
||||||
|
|
||||||
fake_comment = MagicMock()
|
|
||||||
fake_comment.id = "C-001"
|
|
||||||
fake_comment.attachment_count = 0
|
|
||||||
|
|
||||||
fake_fr_doc = {
|
|
||||||
"id": "FR-001",
|
|
||||||
"attributes": {
|
|
||||||
"objectId": "obj1",
|
|
||||||
"commentEndDate": "2024-12-31",
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
mock_api = MagicMock()
|
|
||||||
mock_api.__enter__ = lambda s: s
|
|
||||||
mock_api.__exit__ = lambda s, *a: None
|
|
||||||
mock_api.find_documents_in_docket.return_value = [fake_fr_doc]
|
|
||||||
# Return 2 comments to hit limit=1 → line 175 (return)
|
|
||||||
mock_api.iter_comments.return_value = iter([fake_comment, fake_comment])
|
|
||||||
|
|
||||||
mock_store = MagicMock()
|
|
||||||
|
|
||||||
with (
|
|
||||||
patch("bib.connect", return_value=mock_store),
|
|
||||||
patch("bib.regulations_gov.Client", return_value=mock_api),
|
|
||||||
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
|
|
||||||
):
|
|
||||||
result = runner.invoke(
|
|
||||||
app,
|
|
||||||
["fetch-docket-comments", "CMS-1234-P", "--limit", "1"],
|
|
||||||
)
|
|
||||||
assert result.exit_code == 0 # line 175 hit
|
|
||||||
|
|
||||||
def test_fetch_docket_comments_progress_and_commit(self):
|
|
||||||
"""Lines 184, 185: progress echo at 50-comment intervals and commit."""
|
|
||||||
from typer.testing import CliRunner
|
|
||||||
|
|
||||||
from cli.bib import app
|
|
||||||
|
|
||||||
runner = CliRunner()
|
|
||||||
|
|
||||||
fake_comment = MagicMock()
|
|
||||||
fake_comment.id = "C-001"
|
|
||||||
fake_comment.attachment_count = 0
|
|
||||||
|
|
||||||
fake_fr_doc = {
|
|
||||||
"id": "FR-001",
|
|
||||||
"attributes": {
|
|
||||||
"objectId": "obj1",
|
|
||||||
"commentEndDate": "2024-12-31",
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
mock_api = MagicMock()
|
|
||||||
mock_api.__enter__ = lambda s: s
|
|
||||||
mock_api.__exit__ = lambda s, *a: None
|
|
||||||
mock_api.find_documents_in_docket.return_value = [fake_fr_doc]
|
|
||||||
# Return exactly 50 comments to trigger progress print
|
|
||||||
mock_api.iter_comments.return_value = iter([fake_comment] * 50)
|
|
||||||
|
|
||||||
mock_store = MagicMock()
|
|
||||||
mock_con = MagicMock()
|
|
||||||
mock_store._con.return_value = mock_con
|
|
||||||
|
|
||||||
with (
|
|
||||||
patch("bib.connect", return_value=mock_store),
|
|
||||||
patch("bib.regulations_gov.Client", return_value=mock_api),
|
|
||||||
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
|
|
||||||
):
|
|
||||||
result = runner.invoke(
|
|
||||||
app,
|
|
||||||
["fetch-docket-comments", "CMS-1234-P"],
|
|
||||||
)
|
|
||||||
assert result.exit_code == 0
|
|
||||||
assert "processed 50" in result.output # line 184
|
|
||||||
mock_con.commit.assert_called() # line 185
|
|
||||||
|
|
||||||
|
|
||||||
class TestCliBibFetchPfsComments:
|
|
||||||
"""Lines 263, 265, 268, 276, 277, 281, 282, 285, 286."""
|
|
||||||
|
|
||||||
def test_fetch_pfs_comments_full_flow(self):
|
|
||||||
from typer.testing import CliRunner
|
|
||||||
|
|
||||||
from cli.bib import app
|
|
||||||
|
|
||||||
runner = CliRunner()
|
|
||||||
|
|
||||||
fake_doc = MagicMock()
|
|
||||||
fake_doc.type = "Proposed Rule"
|
|
||||||
fake_doc.publication_date = "2024-01-01"
|
|
||||||
fake_doc.document_number = "2024-00001"
|
|
||||||
fake_doc.dockets = ["CMS-1234-P"]
|
|
||||||
fake_doc.html_url = "https://example.com/doc"
|
|
||||||
|
|
||||||
fake_rule = MagicMock()
|
|
||||||
|
|
||||||
fake_comment = MagicMock()
|
|
||||||
fake_comment.id = "C-001"
|
|
||||||
fake_comment.attachment_count = 1
|
|
||||||
|
|
||||||
fake_att = MagicMock()
|
|
||||||
fake_att.url = "https://example.com/att.pdf"
|
|
||||||
fake_att.filename = "att.pdf"
|
|
||||||
|
|
||||||
# fr_doc with no objectId → line 263 (continue)
|
|
||||||
fr_doc_no_obj = {
|
|
||||||
"id": "FR-A",
|
|
||||||
"attributes": {"commentEndDate": "2024-12-31"},
|
|
||||||
}
|
|
||||||
# fr_doc with no commentEndDate → line 265 (continue)
|
|
||||||
fr_doc_no_end = {
|
|
||||||
"id": "FR-B",
|
|
||||||
"attributes": {"objectId": "obj2"},
|
|
||||||
}
|
|
||||||
# fr_doc that works normally
|
|
||||||
fr_doc_ok = {
|
|
||||||
"id": "FR-C",
|
|
||||||
"attributes": {
|
|
||||||
"objectId": "obj3",
|
|
||||||
"commentEndDate": "2024-12-31",
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
mock_api = MagicMock()
|
|
||||||
mock_api.__enter__ = lambda s: s
|
|
||||||
mock_api.__exit__ = lambda s, *a: None
|
|
||||||
mock_api.resolve_docket.return_value = "REG-DOCKET-1"
|
|
||||||
mock_api.find_documents_in_docket.return_value = [
|
|
||||||
fr_doc_no_obj,
|
|
||||||
fr_doc_no_end,
|
|
||||||
fr_doc_ok,
|
|
||||||
]
|
|
||||||
# Return 50 comments to hit lines 285-286 (progress + commit)
|
|
||||||
mock_api.iter_comments.return_value = iter([fake_comment] * 50)
|
|
||||||
mock_api.attachments_for.return_value = [fake_att]
|
|
||||||
mock_api.download_attachment.return_value = Path("/fake/att.pdf")
|
|
||||||
|
|
||||||
mock_store = MagicMock()
|
|
||||||
mock_con = MagicMock()
|
|
||||||
mock_store._con.return_value = mock_con
|
|
||||||
|
|
||||||
with (
|
|
||||||
patch("bib.connect", return_value=mock_store),
|
|
||||||
patch("bib.federalregister.pfs_rules", return_value=[fake_doc]),
|
|
||||||
patch("bib.federalregister.split_docket_ids", return_value=["CMS-1234-P"]),
|
|
||||||
patch("bib.regulations_gov.Client", return_value=mock_api),
|
|
||||||
patch("bib.regulations_gov.upsert_comment", return_value="key1"),
|
|
||||||
patch("bib.translate.federal_register", return_value=fake_rule),
|
|
||||||
):
|
|
||||||
result = runner.invoke(
|
|
||||||
app,
|
|
||||||
[
|
|
||||||
"fetch-pfs-comments",
|
|
||||||
"--attachments",
|
|
||||||
"--per-docket-limit",
|
|
||||||
"50",
|
|
||||||
],
|
|
||||||
)
|
|
||||||
assert result.exit_code == 0
|
|
||||||
# line 263: fr_doc_no_obj skipped
|
|
||||||
# line 265: fr_doc_no_end skipped
|
|
||||||
# line 268: per_docket_limit break
|
|
||||||
# lines 276-277: attachment download
|
|
||||||
# lines 281-282: attach_file called
|
|
||||||
# lines 285-286: progress + commit at 50
|
|
||||||
|
|
||||||
def test_fetch_pfs_comments_attachment_path_none(self):
|
|
||||||
"""Line 281: download_attachment returns None → skip attach_file."""
|
|
||||||
from typer.testing import CliRunner
|
|
||||||
|
|
||||||
from cli.bib import app
|
|
||||||
|
|
||||||
runner = CliRunner()
|
|
||||||
|
|
||||||
fake_doc = MagicMock()
|
|
||||||
fake_doc.type = "Proposed Rule"
|
|
||||||
fake_doc.publication_date = "2024-01-01"
|
|
||||||
fake_doc.document_number = "2024-00001"
|
|
||||||
fake_doc.dockets = ["CMS-5678-P"]
|
|
||||||
fake_doc.html_url = "https://example.com/doc"
|
|
||||||
|
|
||||||
fake_rule = MagicMock()
|
|
||||||
|
|
||||||
fake_comment = MagicMock()
|
|
||||||
fake_comment.id = "C-002"
|
|
||||||
fake_comment.attachment_count = 1
|
|
||||||
|
|
||||||
fake_att = MagicMock()
|
|
||||||
fake_att.url = "https://example.com/att.pdf"
|
|
||||||
fake_att.filename = "att.pdf"
|
|
||||||
|
|
||||||
fr_doc_ok = {
|
|
||||||
"id": "FR-D",
|
|
||||||
"attributes": {
|
|
||||||
"objectId": "obj4",
|
|
||||||
"commentEndDate": "2024-12-31",
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
mock_api = MagicMock()
|
|
||||||
mock_api.__enter__ = lambda s: s
|
|
||||||
mock_api.__exit__ = lambda s, *a: None
|
|
||||||
mock_api.resolve_docket.return_value = "REG-DOCKET-2"
|
|
||||||
mock_api.find_documents_in_docket.return_value = [fr_doc_ok]
|
|
||||||
mock_api.iter_comments.return_value = iter([fake_comment])
|
|
||||||
mock_api.attachments_for.return_value = [fake_att]
|
|
||||||
mock_api.download_attachment.return_value = None # failed download
|
|
||||||
|
|
||||||
mock_store = MagicMock()
|
|
||||||
mock_con = MagicMock()
|
|
||||||
mock_store._con.return_value = mock_con
|
|
||||||
|
|
||||||
with (
|
|
||||||
patch("bib.connect", return_value=mock_store),
|
|
||||||
patch("bib.federalregister.pfs_rules", return_value=[fake_doc]),
|
|
||||||
patch("bib.federalregister.split_docket_ids", return_value=["CMS-5678-P"]),
|
|
||||||
patch("bib.regulations_gov.Client", return_value=mock_api),
|
|
||||||
patch("bib.regulations_gov.upsert_comment", return_value="key2"),
|
|
||||||
patch("bib.translate.federal_register", return_value=fake_rule),
|
|
||||||
):
|
|
||||||
result = runner.invoke(
|
|
||||||
app,
|
|
||||||
["fetch-pfs-comments", "--attachments", "--per-docket-limit", "1"],
|
|
||||||
)
|
|
||||||
assert result.exit_code == 0
|
|
||||||
mock_store.attach_file.assert_not_called() # line 281 path=None
|
|
||||||
|
|
||||||
|
|
||||||
class TestCliBibIngestMailMissingPassword:
|
class TestCliBibIngestMailMissingPassword:
|
||||||
"""Line 325: no cached password for user."""
|
"""Line 325: no cached password for user."""
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user