feat(bib): iter_comments since= watermark + on_error hook (refs #615)

This commit is contained in:
kert
2026-09-08 12:11:22 -04:00
parent 7be6ca8157
commit 272a5f3bfd
2 changed files with 82 additions and 3 deletions

View File

@@ -31,7 +31,7 @@ import time
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import dataclass, field from dataclasses import dataclass, field
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING, Iterator from typing import TYPE_CHECKING, Callable, Iterator
import httpx import httpx
@@ -195,7 +195,13 @@ class Client:
# ── Comment iteration ────────────────────────────────────── # ── Comment iteration ──────────────────────────────────────
def iter_comments(self, object_id: str) -> Iterator[Comment]: def iter_comments(
self,
object_id: str,
*,
since: str = "",
on_error: "Callable[[Exception], None] | None" = None,
) -> Iterator[Comment]:
"""Yield every comment against a single FR document. """Yield every comment against a single FR document.
The ``commentOnId`` filter on ``/comments`` takes a document's The ``commentOnId`` filter on ``/comments`` takes a document's
@@ -211,8 +217,14 @@ class Client:
2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` — 2. That filter rejects ISO-8601 timestamps with ``T``/``Z`` —
returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space- returns 400. It wants ``YYYY-MM-DD HH:MM:SS`` (space-
separated, no timezone suffix). Normalize before sending. separated, no timezone suffix). Normalize before sending.
*since* seeds the ``lastModifiedDate`` cursor so an incremental
walk starts where the last clean one ended (the boundary row is
re-yielded; the store's no-op upsert absorbs it). *on_error* is
invoked with the ``HTTPStatusError`` before the walk stops, so a
caller can tell "walked to the end" from "gave up".
""" """
cursor: str | None = None cursor: str | None = _reg_date(since) if since else None
page = 1 page = 1
while True: while True:
params: dict[str, str] = { params: dict[str, str] = {
@@ -233,6 +245,8 @@ class Client:
page, page,
cursor, cursor,
) )
if on_error is not None:
on_error(e)
break break
rows = data.get("data", []) rows = data.get("data", [])
if not rows: if not rows:

View File

@@ -0,0 +1,65 @@
from __future__ import annotations
import httpx
from bib.regulations_gov import Client
def _row(cid: str, lm: str) -> dict:
return {
"id": cid,
"attributes": {
"lastModifiedDate": lm,
"postedDate": lm,
"docketId": "D",
"commentOnId": "x",
},
}
def _client(handler) -> Client:
http = httpx.Client(
transport=httpx.MockTransport(handler), headers={"X-Api-Key": "k"}
)
return Client(api_key="k", sleep=0, client=http)
def test_since_seeds_the_date_filter():
seen: list[dict] = []
def handler(req: httpx.Request) -> httpx.Response:
seen.append(dict(req.url.params))
return httpx.Response(
200,
json={
"data": [_row("D-1", "2026-09-08T12:00:00Z")],
"meta": {"totalPages": 1},
},
)
api = _client(handler)
out = list(api.iter_comments("obj", since="2026-09-01T00:00:00Z"))
assert [c.id for c in out] == ["D-1"]
assert seen[0]["filter[lastModifiedDate][ge]"] == "2026-09-01 00:00:00"
def test_no_since_means_no_date_filter():
seen: list[dict] = []
def handler(req: httpx.Request) -> httpx.Response:
seen.append(dict(req.url.params))
return httpx.Response(200, json={"data": [], "meta": {"totalPages": 1}})
list(_client(handler).iter_comments("obj"))
assert "filter[lastModifiedDate][ge]" not in seen[0]
def test_on_error_called_when_page_fails():
errors: list[Exception] = []
def handler(req: httpx.Request) -> httpx.Response:
return httpx.Response(500, json={})
out = list(_client(handler).iter_comments("obj", on_error=errors.append))
assert out == []
assert len(errors) == 1 and isinstance(errors[0], httpx.HTTPStatusError)