feat(bib): fr_anchors schema + FR body-HTML grabber + fr-grab CLI (closes #634)

This commit is contained in:
kert
2026-08-17 14:24:03 -04:00
parent 7ac4024c76
commit 5ea78cffd7
4 changed files with 431 additions and 0 deletions

212
src/bib/frlink.py Normal file
View File

@@ -0,0 +1,212 @@
"""FR jump links — grab and place federalregister.gov deep links (P40).
The web version of every Federal Register document exposes two kinds of
jump anchors: ``#page-NNNNN`` for each printed FR page, and ``#p-N``
for each paragraph (the hover link icon). This module **grabs** a
rule's full anchor map from the FR API's ``body_html_url`` — one fetch
per rule; each paragraph element carries an explicit ``data-page``
attribute, so page association is exact — persists it in ``bib.sqlite``
(``fr_anchors`` / ``fr_anchor_docs``), and **places** deep links for
any FR citation:
- ``"91 FR 44218"`` → page link (``#page-44218``)
- ``"91 FR 44218 ¶3"`` → 3rd paragraph on that page (``#p-N``)
- ``"p-3600"`` → validated raw paragraph anchor
- a quoted passage → the paragraph containing it
Transmutation runs both directions (:func:`page_of`,
:func:`paragraphs_of`), :func:`md_link` renders markdown for
notebooks, and :func:`place` records a link in ``fr_links`` — the
curated subset that ``bib.sync`` carries into Zotero as child link
attachments. Resolution is pure over stored data; only
:func:`grab`/:func:`backfill` touch the network.
"""
from __future__ import annotations
import hashlib
import html as _html
import re
from dataclasses import dataclass
from typing import TYPE_CHECKING, Any, Callable
if TYPE_CHECKING:
from bib.store import Store
_API_DOC = "https://www.federalregister.gov/api/v1/documents/{doc}.json"
# Paragraph anchors in the body HTML: <p> or <li> elements carrying
# id="p-N" ... data-page="NNNNN" (attribute order is stable in FR's
# rendering; verified on doc 2026-14327 — 4,194 <p> + 662 <li>).
_P_ANCHOR_RE = re.compile(r'<(\w+)[^>]*\bid="p-(\d+)"[^>]*\bdata-page="(\d+)"[^>]*>')
_TAG_RE = re.compile(r"<[^>]+>")
# Ref grammar: raw anchor, or "VV FR PPPPP [¶k | para k | p.k]".
_RAW_ANCHOR_RE = re.compile(r"^p-(\d+)$")
_FR_CITE_RE = re.compile(
r"^(\d+)\s+FR\s+(\d+)"
r"(?:[,\s]+(?:¶|para\.?\s*|p\.\s*)(\d+))?$",
re.IGNORECASE,
)
_MIN_QUOTE_LEN = 15
@dataclass(frozen=True)
class Anchor:
"""One paragraph anchor: ``#p-{p_id}`` on printed page ``page``,
the ``ordinal``-th paragraph on that page (1-based)."""
p_id: int
page: int
ordinal: int
text: str
@dataclass(frozen=True)
class JumpLink:
"""A resolved deep link into the FR web version of a rule."""
url: str
item_key: str
page: int
p_id: int | None = None
ordinal: int | None = None
snippet: str = ""
def _norm(text: str) -> str:
"""Entity-decode, strip tags, collapse whitespace — the canonical
stored/compared text form."""
return " ".join(_TAG_RE.sub("", _html.unescape(text)).split())
def parse_anchors(html: str) -> list[Anchor]:
"""Extract every paragraph anchor from an FR ``body_html`` page.
Pure function. Ordinals are assigned 1-based per printed page in
document order; text is normalized via :func:`_norm` (the element's
content up to its closing ``</p>``).
"""
anchors: list[Anchor] = []
per_page: dict[int, int] = {}
for m in _P_ANCHOR_RE.finditer(html):
tag, p_id, page = m.group(1), int(m.group(2)), int(m.group(3))
per_page[page] = per_page.get(page, 0) + 1
end = html.find(f"</{tag}>", m.end())
text = _norm(html[m.end() : end]) if end >= 0 else ""
anchors.append(Anchor(p_id=p_id, page=page, ordinal=per_page[page], text=text))
return anchors
# ── Grab ────────────────────────────────────────────────────────────
def _doc_number(store: Store, item_key: str) -> str:
"""The item's FR document number, from extra_json or its URL."""
item = store.get(item_key)
doc = getattr(item, "document_number", "") or ""
if doc:
return doc
# https://www.federalregister.gov/documents/YYYY/MM/DD/<doc>/<slug>
m = re.search(r"/documents/\d{4}/\d{2}/\d{2}/([^/]+)/", item.url or "")
if m:
return m.group(1)
raise ValueError(f"{item_key}: no FR document number on item or URL")
def _default_fetch(doc: str) -> tuple[dict, str]:
"""Fetch document metadata + body HTML from federalregister.gov."""
import httpx
fields = "fields[]=html_url&fields[]=body_html_url&fields[]=start_page&fields[]=end_page&fields[]=volume"
with httpx.Client(timeout=120, follow_redirects=True) as client:
meta = (
client.get(f"{_API_DOC.format(doc=doc)}?{fields}").raise_for_status().json()
)
body = client.get(meta["body_html_url"]).raise_for_status().text
return meta, body
def grab(
store: Store,
item_key: str,
*,
force: bool = False,
fetch: Callable[[str], tuple[dict, str]] | None = None,
) -> dict[str, Any]:
"""Grab and persist ``item_key``'s FR anchor map.
One network round-trip pair (doc API + body HTML) via ``fetch``
(injectable for tests). Replaces the item's ``fr_anchors``
partition and ``fr_anchor_docs`` row. Skips (returns
``{"skipped": True}``) when already grabbed unless ``force``.
"""
con = store._con() # noqa: SLF001 — same pattern as bib/oig.py
existing = con.execute(
"SELECT 1 FROM fr_anchor_docs WHERE item_key = ?", (item_key,)
).fetchone()
if existing and not force:
return {"item_key": item_key, "skipped": True}
doc = _doc_number(store, item_key)
meta, body = (fetch or _default_fetch)(doc)
anchors = parse_anchors(body)
pages = {a.page for a in anchors}
con.execute("DELETE FROM fr_anchors WHERE item_key = ?", (item_key,))
con.executemany(
"INSERT INTO fr_anchors (item_key, p_id, page, ordinal, text) "
"VALUES (?, ?, ?, ?, ?)",
[(item_key, a.p_id, a.page, a.ordinal, a.text) for a in anchors],
)
con.execute(
"""INSERT OR REPLACE INTO fr_anchor_docs
(item_key, document_number, html_url, body_html_url,
start_page, end_page, fr_volume, sha256, n_paragraphs, n_pages)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
(
item_key,
doc,
meta["html_url"],
meta.get("body_html_url", ""),
int(meta["start_page"]),
int(meta["end_page"]),
int(meta["volume"]),
hashlib.sha256(body.encode()).hexdigest(),
len(anchors),
len(pages),
),
)
con.commit()
return {
"item_key": item_key,
"document_number": doc,
"n_paragraphs": len(anchors),
"n_pages": len(pages),
"skipped": False,
}
def backfill(
store: Store, *, force: bool = False, sleep: float = 1.0
) -> list[dict[str, Any]]:
"""Grab anchor maps for every rule item with an FR URL.
Sequential with a polite ``sleep`` between live fetches; failures
are recorded per item and never abort the sweep.
"""
import time
results: list[dict[str, Any]] = []
for item in store.list_items(item_type="rule"):
if "federalregister.gov" not in (item.url or ""):
continue
try:
out = grab(store, item.key, force=force)
except Exception as e: # noqa: BLE001 — sweep must survive odd rules
out = {"item_key": item.key, "error": str(e)}
results.append(out)
if not out.get("skipped") and "error" not in out:
time.sleep(sleep)
return results

View File

@@ -93,3 +93,47 @@ CREATE UNIQUE INDEX IF NOT EXISTS idx_pincites_unique
ON pincites(fn_path, item_key, locator); ON pincites(fn_path, item_key, locator);
CREATE INDEX IF NOT EXISTS idx_pincites_fn ON pincites(fn_path); CREATE INDEX IF NOT EXISTS idx_pincites_fn ON pincites(fn_path);
CREATE INDEX IF NOT EXISTS idx_pincites_item ON pincites(item_key); CREATE INDEX IF NOT EXISTS idx_pincites_item ON pincites(item_key);
-- ── FR jump links (P40, #634) ────────────────────────────────────
-- Anchor maps grabbed from the federalregister.gov web version of a
-- rule: one row per paragraph anchor (id="p-N"), with the printed FR
-- page (explicit data-page attribute) and 1-based ordinal-on-page.
-- Text is entity-decoded/tag-stripped for quote search.
CREATE TABLE IF NOT EXISTS fr_anchors (
id INTEGER PRIMARY KEY AUTOINCREMENT,
item_key TEXT NOT NULL REFERENCES items(key) ON DELETE CASCADE,
p_id INTEGER NOT NULL,
page INTEGER NOT NULL,
ordinal INTEGER NOT NULL,
text TEXT NOT NULL DEFAULT '',
UNIQUE (item_key, p_id)
);
CREATE INDEX IF NOT EXISTS idx_fr_anchors_page ON fr_anchors(item_key, page);
CREATE TABLE IF NOT EXISTS fr_anchor_docs (
item_key TEXT PRIMARY KEY REFERENCES items(key) ON DELETE CASCADE,
document_number TEXT NOT NULL,
html_url TEXT NOT NULL,
body_html_url TEXT NOT NULL DEFAULT '',
start_page INTEGER NOT NULL,
end_page INTEGER NOT NULL,
fr_volume INTEGER NOT NULL,
fetched_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')),
sha256 TEXT NOT NULL DEFAULT '',
n_paragraphs INTEGER NOT NULL DEFAULT 0,
n_pages INTEGER NOT NULL DEFAULT 0
);
-- Placed links: the curated subset of jump links actually cited —
-- these (and only these) sync to Zotero as child link attachments.
CREATE TABLE IF NOT EXISTS fr_links (
id INTEGER PRIMARY KEY AUTOINCREMENT,
item_key TEXT NOT NULL REFERENCES items(key) ON DELETE CASCADE,
p_id INTEGER,
page INTEGER NOT NULL,
label TEXT NOT NULL DEFAULT '',
url TEXT NOT NULL,
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')),
UNIQUE (item_key, url)
);

View File

@@ -661,3 +661,39 @@ def refresh_iom(
f" created={stats['created']} skipped={stats['skipped']} " f" created={stats['created']} skipped={stats['skipped']} "
f"attachments={stats['attachments']}" f"attachments={stats['attachments']}"
) )
@app.command(name="fr-grab")
def fr_grab(
key: str = typer.Option("", "--key", help="Grab one rule item by bib key."),
all_rules: bool = typer.Option(
False, "--all", help="Backfill every rule item with an FR URL."
),
force: bool = typer.Option(
False, "--force", help="Re-grab even if an anchor map already exists."
),
) -> None:
"""Grab federalregister.gov paragraph/page anchor maps into bib (#634).
One body-HTML fetch per rule; anchors persist in ``fr_anchors`` /
``fr_anchor_docs`` for offline jump-link resolution.
"""
from bib import connect, frlink
store = connect()
if bool(key) == all_rules:
raise typer.BadParameter("pass exactly one of --key or --all")
if key:
results = [frlink.grab(store, key, force=force)]
else:
results = frlink.backfill(store, force=force)
for r in results:
if r.get("skipped"):
typer.echo(f" {r['item_key']}: skipped (already grabbed)")
elif "error" in r:
typer.echo(f" {r['item_key']}: ERROR {r['error']}")
else:
typer.echo(
f" {r['item_key']}: {r['document_number']} — "
f"{r['n_paragraphs']} paragraphs / {r['n_pages']} pages"
)

139
tests/bib/test_frlink.py Normal file
View File

@@ -0,0 +1,139 @@
"""Tests for bib.frlink — FR web anchor maps + jump-link resolution (P40)."""
from __future__ import annotations
from bib import frlink
from bib.item import Rule
from bib.store import Store
# Mirrors the real body-HTML shape: paragraph elements carry id="p-N"
# AND an explicit data-page attribute (verified on doc 2026-14327);
# page divs carry id="page-NNNNN"; text has entities + inline markup.
FIXTURE_HTML = """
<html><body>
<div id="page-100"></div>
<p id="p-1" data-page="100">First para on 100 with &sect;&thinsp;414.1425 text.</p>
<p id="p-2" data-page="100">Second para <em>with markup</em> on 100.</p>
<div id="page-101"></div>
<p id="p-3" data-page="101">Only para on 101.</p>
<ul><li id="p-4" data-page="101">A list item anchor on 101.</li></ul>
</body></html>
"""
FIXTURE_HTML_SMALLER = """
<html><body>
<div id="page-100"></div>
<p id="p-1" data-page="100">First para on 100.</p>
<p id="p-2" data-page="100">Second para on 100.</p>
</body></html>
"""
DOC_META = {
"html_url": "https://example.test/doc",
"body_html_url": "https://example.test/full_text/doc.html",
"start_page": 100,
"end_page": 101,
"volume": 91,
}
def _store_with_rule() -> tuple[Store, str]:
s = Store(":memory:")
key = s.create(
Rule(
title="Test Rule",
url="https://www.federalregister.gov/documents/2026/07/16/2026-14327/test",
document_number="2026-14327",
fr_volume="91",
fr_page="100",
)
)
return s, key
def _grabbed_store(html: str = FIXTURE_HTML) -> tuple[Store, str]:
s, key = _store_with_rule()
frlink.grab(s, key, fetch=lambda _doc: (DOC_META, html))
return s, key
# ── parse_anchors ───────────────────────────────────────────────────
class TestParseAnchors:
def test_pages_ordinals_text(self) -> None:
a = frlink.parse_anchors(FIXTURE_HTML)
assert [(x.p_id, x.page, x.ordinal) for x in a] == [
(1, 100, 1),
(2, 100, 2),
(3, 101, 1),
(4, 101, 2),
]
# entities decoded (&sect; &thinsp;), whitespace collapsed
assert "414.1425" in a[0].text
assert "§" in a[0].text
assert "&sect;" not in a[0].text
# inline markup stripped
assert a[1].text == "Second para with markup on 100."
# <li> anchors close with </li>, not </p> — text must not bleed
assert a[3].text == "A list item anchor on 101."
def test_empty_html_yields_nothing(self) -> None:
assert frlink.parse_anchors("<html><body></body></html>") == []
# ── grab ────────────────────────────────────────────────────────────
class TestGrab:
def test_grab_persists_anchors_and_doc_row(self) -> None:
s, key = _grabbed_store()
con = s._con() # noqa: SLF001
rows = con.execute(
"SELECT p_id, page, ordinal FROM fr_anchors WHERE item_key = ? ORDER BY p_id",
(key,),
).fetchall()
assert [tuple(r) for r in rows] == [
(1, 100, 1),
(2, 100, 2),
(3, 101, 1),
(4, 101, 2),
]
doc = con.execute(
"SELECT document_number, html_url, start_page, end_page, fr_volume, "
"n_paragraphs, n_pages FROM fr_anchor_docs WHERE item_key = ?",
(key,),
).fetchone()
assert tuple(doc) == (
"2026-14327",
"https://example.test/doc",
100,
101,
91,
4,
2,
)
s.close()
def test_regrab_replaces_partition(self) -> None:
s, key = _grabbed_store()
frlink.grab(
s, key, force=True, fetch=lambda _doc: (DOC_META, FIXTURE_HTML_SMALLER)
)
con = s._con() # noqa: SLF001
n = con.execute(
"SELECT count(*) FROM fr_anchors WHERE item_key = ?", (key,)
).fetchone()[0]
assert n == 2
s.close()
def test_grab_skips_when_already_grabbed(self) -> None:
s, key = _grabbed_store()
out = frlink.grab(s, key, fetch=lambda _doc: (DOC_META, FIXTURE_HTML_SMALLER))
assert out["skipped"] is True
con = s._con() # noqa: SLF001
n = con.execute(
"SELECT count(*) FROM fr_anchors WHERE item_key = ?", (key,)
).fetchone()[0]
assert n == 4 # untouched
s.close()