Files
stack/tests/bib/test_frlink.py

363 lines
13 KiB
Python

"""Tests for bib.frlink — FR web anchor maps + jump-link resolution (P40)."""
from __future__ import annotations
import pytest
from bib import frlink
from bib.item import Rule
from bib.store import Store
# Mirrors the real body-HTML shape: paragraph elements carry id="p-N"
# AND an explicit data-page attribute (verified on doc 2026-14327);
# page divs carry id="page-NNNNN"; text has entities + inline markup.
FIXTURE_HTML = """
<html><body>
<div id="page-100"></div>
<p id="p-1" data-page="100">First para on 100 with &sect;&thinsp;414.1425 text.</p>
<p id="p-2" data-page="100">Second para <em>with markup</em> on 100.</p>
<div id="page-101"></div>
<p id="p-3" data-page="101">Only para on 101.</p>
<ul><li id="p-4" data-page="101">A list item anchor on 101.</li></ul>
</body></html>
"""
FIXTURE_HTML_SMALLER = """
<html><body>
<div id="page-100"></div>
<p id="p-1" data-page="100">First para on 100.</p>
<p id="p-2" data-page="100">Second para on 100.</p>
</body></html>
"""
DOC_META = {
"html_url": "https://example.test/doc",
"body_html_url": "https://example.test/full_text/doc.html",
"start_page": 100,
"end_page": 101,
"volume": 91,
}
def _store_with_rule() -> tuple[Store, str]:
s = Store(":memory:")
key = s.create(
Rule(
title="Test Rule",
url="https://www.federalregister.gov/documents/2026/07/16/2026-14327/test",
document_number="2026-14327",
fr_volume="91",
fr_page="100",
)
)
return s, key
def _grabbed_store(html: str = FIXTURE_HTML) -> tuple[Store, str]:
s, key = _store_with_rule()
frlink.grab(s, key, fetch=lambda _doc: (DOC_META, html))
return s, key
# ── parse_anchors ───────────────────────────────────────────────────
class TestParseAnchors:
def test_pages_ordinals_text(self) -> None:
a = frlink.parse_anchors(FIXTURE_HTML)
assert [(x.p_id, x.page, x.ordinal) for x in a] == [
(1, 100, 1),
(2, 100, 2),
(3, 101, 1),
(4, 101, 2),
]
# entities decoded (&sect; &thinsp;), whitespace collapsed
assert "414.1425" in a[0].text
assert "§" in a[0].text
assert "&sect;" not in a[0].text
# inline markup stripped
assert a[1].text == "Second para with markup on 100."
# <li> anchors close with </li>, not </p> — text must not bleed
assert a[3].text == "A list item anchor on 101."
def test_empty_html_yields_nothing(self) -> None:
assert frlink.parse_anchors("<html><body></body></html>") == []
# ── grab ────────────────────────────────────────────────────────────
class TestGrab:
def test_grab_persists_anchors_and_doc_row(self) -> None:
s, key = _grabbed_store()
con = s._con() # noqa: SLF001
rows = con.execute(
"SELECT p_id, page, ordinal FROM fr_anchors WHERE item_key = ? ORDER BY p_id",
(key,),
).fetchall()
assert [tuple(r) for r in rows] == [
(1, 100, 1),
(2, 100, 2),
(3, 101, 1),
(4, 101, 2),
]
doc = con.execute(
"SELECT document_number, html_url, start_page, end_page, fr_volume, "
"n_paragraphs, n_pages FROM fr_anchor_docs WHERE item_key = ?",
(key,),
).fetchone()
assert tuple(doc) == (
"2026-14327",
"https://example.test/doc",
100,
101,
91,
4,
2,
)
s.close()
def test_regrab_replaces_partition(self) -> None:
s, key = _grabbed_store()
frlink.grab(
s, key, force=True, fetch=lambda _doc: (DOC_META, FIXTURE_HTML_SMALLER)
)
con = s._con() # noqa: SLF001
n = con.execute(
"SELECT count(*) FROM fr_anchors WHERE item_key = ?", (key,)
).fetchone()[0]
assert n == 2
s.close()
def test_grab_skips_when_already_grabbed(self) -> None:
s, key = _grabbed_store()
out = frlink.grab(s, key, fetch=lambda _doc: (DOC_META, FIXTURE_HTML_SMALLER))
assert out["skipped"] is True
con = s._con() # noqa: SLF001
n = con.execute(
"SELECT count(*) FROM fr_anchors WHERE item_key = ?", (key,)
).fetchone()[0]
assert n == 4 # untouched
s.close()
# ── resolve / transmutations / md_link (#635) ───────────────────────
class TestResolve:
def test_page_ref(self) -> None:
s, key = _grabbed_store()
link = frlink.resolve("91 FR 100", store=s)
assert link.url == "https://example.test/doc#page-100"
assert (link.item_key, link.page, link.p_id) == (key, 100, None)
s.close()
def test_page_plus_ordinal(self) -> None:
s, _key = _grabbed_store()
link = frlink.resolve("91 FR 100 ¶2", store=s)
assert link.url == "https://example.test/doc#p-2"
assert (link.page, link.p_id, link.ordinal) == (100, 2, 2)
assert "markup" in link.snippet
# alternate spellings
assert frlink.resolve("91 FR 100, para 2", store=s).p_id == 2
assert frlink.resolve("91 FR 101 p.2", store=s).p_id == 4
s.close()
def test_raw_anchor(self) -> None:
s, key = _grabbed_store()
link = frlink.resolve("p-3", store=s, item_key=key)
assert link.url == "https://example.test/doc#p-3"
assert link.page == 101
with pytest.raises(ValueError, match="p-99"):
frlink.resolve("p-99", store=s, item_key=key)
s.close()
def test_raw_anchor_without_key_single_doc_context(self) -> None:
s, _key = _grabbed_store()
# only one grabbed rule → unambiguous without item_key
assert frlink.resolve("p-1", store=s).page == 100
s.close()
def test_quote(self) -> None:
s, _key = _grabbed_store()
link = frlink.resolve("Only para on 101", store=s)
assert link.p_id == 3
assert link.page == 101
s.close()
def test_quote_multi_hit_prefers_shortest_start_match(self) -> None:
dup_html = FIXTURE_HTML.replace(
"</body>",
'<p id="p-9" data-page="101">Only para on 101, repeated.</p></body>',
)
s, _key = _grabbed_store(dup_html)
# both start with the quote → the tighter paragraph wins
assert frlink.resolve("Only para on 101", store=s).p_id == 3
s.close()
def test_quote_no_hit_raises(self) -> None:
s, _key = _grabbed_store()
with pytest.raises(ValueError, match="[Nn]o paragraph"):
frlink.resolve("nothing like this appears anywhere", store=s)
s.close()
def test_unknown_page_raises(self) -> None:
s, _key = _grabbed_store()
with pytest.raises(ValueError, match="99 FR 100"):
frlink.resolve("99 FR 100", store=s)
s.close()
def test_ordinal_out_of_range_raises(self) -> None:
s, _key = _grabbed_store()
with pytest.raises(ValueError, match="2 paragraph"):
frlink.resolve("91 FR 100 ¶7", store=s)
s.close()
class TestTransmutations:
def test_page_of(self) -> None:
s, key = _grabbed_store()
assert frlink.page_of(s, key, 3) == 101
s.close()
def test_paragraphs_of(self) -> None:
s, _key = _grabbed_store()
anchors = frlink.paragraphs_of("91 FR 100", store=s)
assert [(a.p_id, a.ordinal) for a in anchors] == [(1, 1), (2, 2)]
s.close()
class TestMdLink:
def test_default_text_is_ref(self) -> None:
s, _key = _grabbed_store()
md = frlink.md_link("91 FR 100 ¶2", store=s, highlight=False)
assert md == "[91 FR 100 ¶2](https://example.test/doc#p-2)"
s.close()
def test_custom_text(self) -> None:
s, _key = _grabbed_store()
md = frlink.md_link(
"91 FR 100", store=s, text="the CF discussion", highlight=False
)
assert md == "[the CF discussion](https://example.test/doc#page-100)"
s.close()
# ── place() (#637) ──────────────────────────────────────────────────
class TestPlace:
def test_place_writes_fr_links_row(self) -> None:
s, key = _grabbed_store()
link = frlink.place(s, "91 FR 100 ¶2", label="the markup para")
assert link.p_id == 2
con = s._con() # noqa: SLF001
rows = con.execute(
"SELECT item_key, p_id, page, label, url FROM fr_links"
).fetchall()
assert [tuple(r) for r in rows] == [
(key, 2, 100, "the markup para", "https://example.test/doc#p-2")
]
s.close()
def test_place_default_label_and_idempotency(self) -> None:
s, _key = _grabbed_store()
frlink.place(s, "91 FR 100")
frlink.place(s, "91 FR 100") # same URL → no duplicate
con = s._con() # noqa: SLF001
rows = con.execute("SELECT label, url, p_id FROM fr_links").fetchall()
assert len(rows) == 1
assert rows[0]["label"] == "91 FR 100"
assert rows[0]["p_id"] is None
s.close()
# ── text fragments + highlight ──────────────────────────────────────
class TestTextFragment:
def test_first_sentence_encoded(self) -> None:
frag = frlink.text_fragment(
"Under this proposal, the new G-codes apply. Second sentence."
)
assert frag == "Under%20this%20proposal%2C%20the%20new%20G%2Dcodes%20apply."
def test_long_sentence_cut_at_word_boundary(self) -> None:
text = "word " * 40
frag = frlink.text_fragment(text.strip(), max_chars=22)
assert frag == "word%20word%20word%20word"
def test_empty(self) -> None:
assert frlink.text_fragment(" ") == ""
class TestHighlight:
def test_paragraph_ref_gets_fragment(self) -> None:
s, _key = _grabbed_store()
link = frlink.resolve("91 FR 100 ¶2", store=s, highlight=True)
assert link.url == (
"https://example.test/doc#p-2:~:text=Second%20para%20with%20markup%20on%20100."
)
assert link.p_id == 2
s.close()
def test_page_cite_upgrades_to_first_paragraph_on_page(self) -> None:
s, _key = _grabbed_store()
link = frlink.resolve("91 FR 101", store=s, highlight=True)
assert link.p_id == 3
assert link.url.startswith("https://example.test/doc#p-3:~:text=Only%20para")
s.close()
def test_page_cite_without_highlight_unchanged(self) -> None:
s, _key = _grabbed_store()
link = frlink.resolve("91 FR 101", store=s)
assert link.url == "https://example.test/doc#page-101"
assert link.p_id is None
s.close()
def test_page_without_anchors_stays_page_link_even_when_highlighting(
self,
) -> None:
s, _key = _grabbed_store()
s._con().execute("DELETE FROM fr_anchors WHERE page = 101")
link = frlink.resolve("91 FR 101", store=s, highlight=True)
assert link.url == "https://example.test/doc#page-101"
s.close()
def test_md_link_highlights_by_default(self) -> None:
s, _key = _grabbed_store()
md = frlink.md_link("91 FR 100 ¶1", store=s)
assert ":~:text=" in md
assert md.startswith("[91 FR 100 ¶1](")
s.close()
def test_place_never_records_fragment(self) -> None:
s, _key = _grabbed_store()
link = frlink.place(s, "91 FR 100 ¶1", label="x")
assert ":~:text=" not in link.url
(row,) = s._con().execute("SELECT url FROM fr_links").fetchall()
assert ":~:text=" not in row["url"]
s.close()
class TestQuoteTieBreak:
def test_prefers_paragraph_starting_with_quote(self) -> None:
html = """
<p id="p-1" data-page="100">Shared quote text appears here first.</p>
<p id="p-2" data-page="100">Before it, shared quote text appears here again.</p>
"""
s, _key = _grabbed_store(html)
link = frlink.resolve("shared quote text appears here", store=s)
assert link.p_id == 1
s.close()
def test_still_raises_when_truly_ambiguous(self) -> None:
html = """
<p id="p-1" data-page="100">Shared quote text appears here first.</p>
<p id="p-2" data-page="100">Shared quote text appears here again.</p>
"""
s, _key = _grabbed_store(html)
with pytest.raises(ValueError, match="matches 2 paragraphs"):
frlink.resolve("shared quote text appears here", store=s)
s.close()