Files
stack/tests/zot/test_extract.py
kert cc418cefde fix(tests): close leaked sqlite connections; gate ResourceWarning as error (refs #618)
zot Db.__init__ leak on schema-parity raise; test fixtures not closing
raw Db/create_db connections; cli comments _build_bib_helpers cache
connection never closed. Gate: error::ResourceWarning (coverage's own
sqlitedb warning ignored). Root causes tracemalloc-verified; most
per-test attributions in #618 were GC-timing misattribution across
xdist workers.
2026-08-14 09:38:15 -04:00

373 lines
13 KiB
Python

"""Tests for zot.extract — tag/annotation extraction for docstrings."""
from __future__ import annotations
from pathlib import Path
import pytest
from zot.db import TYPE_MAP, Db
from zot.extract import (
Extractor,
Quote,
_extract_quotes_from_note,
_page_from_link,
_parse_note_html,
)
HOST_DB = Path("data/zotero/data/zotero.sqlite")
# ── HTML parsing ─────────────────────────────────────────────────
class TestParseNoteHtml:
def test_simple_paragraph(self):
segs = _parse_note_html("<p>Hello world</p>")
assert len(segs) == 1
assert segs[0]["text"] == "Hello world"
def test_multiple_paragraphs(self):
segs = _parse_note_html("<p>First</p><p>Second</p>")
assert len(segs) == 2
def test_link_extraction(self):
html = '<p><a href="zotero://open-pdf/0_ABC12345/14">text</a></p>'
segs = _parse_note_html(html)
assert segs[0]["link"] == "zotero://open-pdf/0_ABC12345/14"
def test_nested_tags(self):
html = "<p><strong>Bold</strong> and <em>italic</em></p>"
segs = _parse_note_html(html)
assert "Bold" in segs[0]["text"]
def test_zotero_note_format(self):
html = (
'<div class="zotero-note znv1">'
"<p><strong>Extracted Annotations</strong></p>"
'<p>"Some quoted text" (<a href="zotero://open-pdf/0_KEY/5">Author :5</a>)</p>'
"</div>"
)
segs = _parse_note_html(html)
assert len(segs) >= 2
class TestPageFromLink:
def test_extracts_page(self):
assert _page_from_link("zotero://open-pdf/0_CUNC7XC6/14") == 14
def test_page_one(self):
assert _page_from_link("zotero://open-pdf/0_KEY/1") == 1
def test_no_match(self):
assert _page_from_link("https://example.com") is None
def test_empty(self):
assert _page_from_link("") is None
# ── Quote extraction ─────────────────────────────────────────────
class TestExtractQuotes:
def test_simple_quote(self):
html = '<p>"This is a quoted annotation."</p>'
quotes = _extract_quotes_from_note(html, "TESTKEY2")
assert len(quotes) == 1
assert quotes[0].text == "This is a quoted annotation."
assert quotes[0].item_key == "TESTKEY2"
def test_quote_with_page_link(self):
html = (
'<p>"Important finding about outcomes."'
' (<a href="zotero://open-pdf/0_ABC/14">Smith :14</a>)</p>'
)
quotes = _extract_quotes_from_note(html, "ABCD2345")
assert len(quotes) == 1
assert quotes[0].page == 14
def test_skips_header(self):
html = (
"<p><strong>Extracted Annotations (1/1/2025)</strong></p>"
'<p>"Actual quote."</p>'
)
quotes = _extract_quotes_from_note(html, "KEY12345")
assert len(quotes) == 1
assert quotes[0].text == "Actual quote."
def test_multiple_quotes(self):
html = '<p>"First quote."</p><p>"Second quote."</p>'
quotes = _extract_quotes_from_note(html, "KEY12345")
assert len(quotes) == 2
def test_no_quotes(self):
html = "<p>This is just a regular note.</p>"
quotes = _extract_quotes_from_note(html, "KEY12345")
assert len(quotes) == 0
# ── Quote model ──────────────────────────────────────────────────
class TestQuote:
def test_pincite_directive_basic(self):
q = Quote(text="Important finding.", item_key="ABCD2345")
assert q.pincite_directive() == ':pincite:`ABCD2345` — "Important finding."'
def test_pincite_directive_with_page(self):
q = Quote(text="Finding.", page=14, item_key="ABCD2345")
assert "p.14" in q.pincite_directive()
def test_pincite_directive_with_section(self):
q = Quote(text="Finding.", section="§2.2.1", page=8, item_key="ABCD2345")
d = q.pincite_directive()
assert "§2.2.1" in d
assert "p.8" in d
def test_pincite_truncates_long_text(self):
q = Quote(text="x" * 200, item_key="ABCD2345")
d = q.pincite_directive()
assert "..." in d
assert len(d) < 300
def test_as_tag(self):
q = Quote(text="t", page=14, item_key="ABCD2345")
assert q.as_tag() == "pin:ABCD2345/p.14"
def test_as_tag_with_section(self):
q = Quote(text="t", section="§2.2.1", item_key="ABCD2345")
assert q.as_tag() == "pin:ABCD2345/§2.2.1"
# ── Extractor with DB ────────────────────────────────────────────
@pytest.fixture()
def extractor(zotero_db) -> Extractor:
path = zotero_db
db = Db(path)
# Create items with notes
id1 = db.create_item(TYPE_MAP["statute"], key="STATABCD")
db.set_fields(
id1, {"nameOfAct": "PFS 2026 Final Rule", "url": "https://ex.com/pfs"}
)
db.sync_tags(id1, ["module:pfs", "year:2026"])
db.add_note(
id1,
"<p><strong>Extracted Annotations</strong></p>"
'<p>"The physician fee schedule determines payment rates."'
' (<a href="zotero://open-pdf/0_XYZ/3">CMS :3</a>)</p>'
'<p>"Geographic adjustments apply to all services."</p>',
)
id2 = db.create_item(TYPE_MAP["journalArticle"], key="JRNKEFGH")
db.set_fields(
id2,
{
"title": "Skin Substitute Review",
"DOI": "10.1234/test",
"publicationTitle": "JAMA",
},
)
db.sync_tags(id2, ["module:skin-subs"])
db.add_creators(id2, [("Jane", "Smith")])
# Item with no notes
id3 = db.create_item(TYPE_MAP["webpage"], key="WEBIJKKM")
db.set_fields(id3, {"title": "CMS Data Portal", "url": "https://cms.gov"})
db.sync_tags(id3, ["module:pfs"])
db.commit()
ex = Extractor(db)
yield ex
ex.close()
db.close()
class TestExtractorQuotes:
def test_quotes_for_item(self, extractor: Extractor):
quotes = extractor.quotes_for_item("STATABCD")
assert len(quotes) == 2
assert quotes[0].text == "The physician fee schedule determines payment rates."
assert quotes[0].page == 3
def test_quotes_for_missing_item(self, extractor: Extractor):
assert extractor.quotes_for_item("ZZZZZZZZ") == []
def test_quotes_for_item_no_notes(self, extractor: Extractor):
assert extractor.quotes_for_item("WEBIJKKM") == []
def test_quotes_for_tag(self, extractor: Extractor):
result = extractor.quotes_for_tag("module:pfs")
assert "STATABCD" in result
assert len(result["STATABCD"]) == 2
class TestExtractorTags:
def test_tags_for_item(self, extractor: Extractor):
tags = extractor.tags_for_item("STATABCD")
assert "module:pfs" in tags
assert "year:2026" in tags
def test_tags_for_missing(self, extractor: Extractor):
assert extractor.tags_for_item("ZZZZZZZZ") == []
def test_items_by_tag_namespace(self, extractor: Extractor):
result = extractor.items_by_tag_namespace("module")
assert "pfs" in result
assert "STATABCD" in result["pfs"]
assert "skin-subs" in result
assert "JRNKEFGH" in result["skin-subs"]
class TestExtractorDocstring:
def test_docstring_block_with_quotes(self, extractor: Extractor):
block = extractor.docstring_block("STATABCD")
assert "References" in block
assert ":pincite:" in block
assert "physician fee schedule" in block
def test_docstring_block_no_notes(self, extractor: Extractor):
block = extractor.docstring_block("WEBIJKKM")
assert "References" in block
assert "CMS Data Portal" in block
def test_pincite_directives(self, extractor: Extractor):
directives = extractor.pincite_directives("STATABCD")
assert len(directives) == 2
assert all(d.startswith(":pincite:") for d in directives)
class TestExtractorExport:
def test_export_provenance(self, extractor: Extractor):
items = extractor.export_provenance()
assert len(items) == 4
keys = {i["key"] for i in items}
assert "STATABCD" in keys
stat = next(i for i in items if i["key"] == "STATABCD")
assert stat["item_type"] == "statute"
assert "module:pfs" in stat["tags"]
assert len(stat["quotes"]) == 2
assert (
stat["quotes"][0]["text"]
== "The physician fee schedule determines payment rates."
)
def test_export_provenance_filtered(self, extractor: Extractor):
items = extractor.export_provenance(tag="module:skin-subs")
assert len(items) == 1
assert items[0]["key"] == "JRNKEFGH"
# ── Integration with real DB ─────────────────────────────────────
@pytest.mark.skipif(
not HOST_DB.exists(),
reason="zotero.sqlite not available — only runs against live Zotero",
)
class TestExtractorRealDb:
def test_extract_real_notes(self, host_db):
with Extractor(str(host_db)) as ex:
# Find items with notes
rows = ex.db.con.execute(
"SELECT DISTINCT i.key FROM items i "
"JOIN itemNotes n ON i.itemID = n.parentItemID "
"LIMIT 5"
).fetchall()
for row in rows:
quotes = ex.quotes_for_item(row[0])
# Just verify it doesn't crash
assert isinstance(quotes, list)
def test_export_provenance_real(self, host_db):
with Extractor(str(host_db)) as ex:
# Small export
items = (
ex.db.search_by_tag("module:skin-subs")
if ex.db.search_by_tag("module:skin-subs")
else []
)
if items:
result = ex.export_provenance(tag="module:skin-subs")
assert isinstance(result, list)
# ── Gap coverage — missed lines ────────────────────────────────
class TestExtractQuotesPageFromSourceRef:
"""Lines 167-169: page extracted from source_ref like ':5' when no PDF link."""
def test_page_from_source_ref(self):
html = '<p>"Quote with ref" (Author :5)</p>'
quotes = _extract_quotes_from_note(html, "REFKEY23")
assert len(quotes) == 1
assert quotes[0].page == 5
assert quotes[0].source_ref == "Author :5"
def test_page_from_source_ref_no_page(self):
html = '<p>"Quote no page" (Author)</p>'
quotes = _extract_quotes_from_note(html, "REFKEY24")
assert len(quotes) == 1
assert quotes[0].page is None
class TestFieldsForItemMissing:
"""Line 282: fields_for_item returns {} for missing item."""
def test_fields_for_missing_item(self, extractor: Extractor):
assert extractor.fields_for_item("ZZZZZZZZ") == {}
class TestDocstringBlockSections:
"""Lines 306-311, 314-315, 334-336: docstring_block with sections filter."""
def test_docstring_block_with_sections(self, extractor: Extractor):
block = extractor.docstring_block("STATABCD", sections=["§2.2.1"])
assert "References" in block
assert ":pincite:" in block
# Section should be assigned to quotes that lack one
assert "§2.2.1" in block
def test_docstring_block_sections_no_match(self, extractor: Extractor):
"""Sections filter keeps all quotes via the else branch (line 314)."""
block = extractor.docstring_block("STATABCD", sections=["§99.99"])
assert "References" in block
assert ":pincite:" in block
class TestExportProvenanceSkipsNone:
"""Line 373: export_provenance skips items where get_item returns None."""
def test_export_provenance_skips_none(self, extractor: Extractor):
from unittest.mock import patch
# Mock get_item to return None for one item to exercise line 373
orig_get_item = extractor._db.get_item
def _patched_get_item(item_id):
# Return None for the first call to trigger the continue
if item_id == 99999:
return None
return orig_get_item(item_id)
# Mock search_by_tag to include a nonexistent item
orig_search = extractor._db.search_by_tag
def _patched_search(tag_name):
result = orig_search(tag_name)
return [99999] + result
with (
patch.object(extractor._db, "get_item", side_effect=_patched_get_item),
patch.object(extractor._db, "search_by_tag", side_effect=_patched_search),
):
items = extractor.export_provenance(tag="module:pfs")
# Should not crash; the 99999 item is skipped
assert isinstance(items, list)
# Only the real items should be present
assert all(i["key"] != "" for i in items)