Files
stack/tests/bib/test_zotero_import.py

255 lines
9.1 KiB
Python

"""bib.zotero_import — pull a Zotero collection's book items (with
files) into bib."""
from __future__ import annotations
import sqlite3
import pytest
from bib.store import Store
from bib.zotero_import import ZoteroBook, import_books, list_collection
def _zotero_db(tmp_path):
"""A tiny sqlite db with the Zotero schema subset list_collection needs:
two book items in "AMA Coding Publications" (one with a storage:
attachment, one without), plus a distractor item in another
collection and a deleted item that must never surface."""
db = tmp_path / "zotero.sqlite"
con = sqlite3.connect(db)
con.executescript(
"""
CREATE TABLE items (itemID INTEGER PRIMARY KEY, itemTypeID INTEGER, key TEXT);
CREATE TABLE itemTypes (itemTypeID INTEGER PRIMARY KEY, typeName TEXT);
CREATE TABLE fields (fieldID INTEGER PRIMARY KEY, fieldName TEXT);
CREATE TABLE itemDataValues (valueID INTEGER PRIMARY KEY, value TEXT);
CREATE TABLE itemData (itemID INTEGER, fieldID INTEGER, valueID INTEGER);
CREATE TABLE collections (collectionID INTEGER PRIMARY KEY, collectionName TEXT);
CREATE TABLE collectionItems (collectionID INTEGER, itemID INTEGER);
CREATE TABLE itemAttachments (itemID INTEGER, parentItemID INTEGER, path TEXT, contentType TEXT);
CREATE TABLE deletedItems (itemID INTEGER);
INSERT INTO itemTypes VALUES (1, 'book'), (2, 'note');
-- item 1: CPT 2018, has a pdf attachment (item 10)
INSERT INTO items VALUES (1, 1, 'CPT2018K');
-- item 2: CPT Changes 2020, no attachment
INSERT INTO items VALUES (2, 1, 'CPTCHG20');
-- item 3: a note item in the same collection — never a "book"
INSERT INTO items VALUES (3, 2, 'NOTEKEY1');
-- item 4: deleted book — must never surface
INSERT INTO items VALUES (4, 1, 'DELETED1');
-- item 10: the attachment for item 1
INSERT INTO items VALUES (10, 3, 'ATTKEY01');
INSERT INTO fields VALUES (1, 'title'), (2, 'url'), (3, 'date'), (4, 'publisher');
INSERT INTO itemDataValues VALUES
(100, 'CPT 2018'), (101, 'https://ama.example/cpt2018'),
(102, '2017-11-01'), (103, 'American Medical Association'),
(200, 'CPT Changes 2020');
INSERT INTO itemData VALUES
(1, 1, 100), (1, 2, 101), (1, 3, 102), (1, 4, 103),
(2, 1, 200);
INSERT INTO collections VALUES (1, 'AMA Coding Publications'), (2, 'Other');
INSERT INTO collectionItems VALUES (1, 1), (1, 2), (1, 3), (1, 4), (2, 1);
INSERT INTO itemAttachments VALUES (10, 1, 'storage:CPT 2018.pdf', 'application/pdf');
INSERT INTO deletedItems VALUES (4);
"""
)
con.commit()
con.close()
return db
@pytest.fixture
def zcon(tmp_path):
con = sqlite3.connect(_zotero_db(tmp_path))
yield con
con.close()
@pytest.fixture
def storage_dir(tmp_path):
d = tmp_path / "storage" / "ATTKEY01"
d.mkdir(parents=True)
(d / "CPT 2018.pdf").write_bytes(b"%PDF-1.4 fake cpt 2018")
return tmp_path / "storage"
@pytest.fixture
def store(tmp_path):
s = Store(":memory:", storage_dir=tmp_path / "bib-storage")
yield s
s.close()
class TestListCollection:
def test_two_books_one_with_attachment(self, zcon, storage_dir):
books = list_collection(
zcon, "AMA Coding Publications", with_files=False, storage_dir=storage_dir
)
assert {b.key for b in books} == {"CPT2018K", "CPTCHG20"}
cpt = next(b for b in books if b.key == "CPT2018K")
assert cpt.title == "CPT 2018"
assert cpt.url == "https://ama.example/cpt2018"
assert cpt.year == "2017"
assert cpt.publisher == "American Medical Association"
assert cpt.attachments == (storage_dir / "ATTKEY01" / "CPT 2018.pdf",)
changes = next(b for b in books if b.key == "CPTCHG20")
assert changes.attachments == ()
assert changes.url == ""
def test_excludes_notes_and_deleted_items(self, zcon, storage_dir):
books = list_collection(
zcon, "AMA Coding Publications", with_files=False, storage_dir=storage_dir
)
keys = {b.key for b in books}
assert "NOTEKEY1" not in keys
assert "DELETED1" not in keys
def test_with_files_filters_to_items_with_attachments(self, zcon, storage_dir):
books = list_collection(
zcon, "AMA Coding Publications", with_files=True, storage_dir=storage_dir
)
assert [b.key for b in books] == ["CPT2018K"]
def test_unknown_collection_returns_empty(self, zcon, storage_dir):
assert (
list_collection(zcon, "Nope", with_files=False, storage_dir=storage_dir)
== []
)
class TestImportBooks:
def _cpt_book(self, storage_dir):
return ZoteroBook(
key="CPT2018K",
title="CPT 2018",
url="https://ama.example/cpt2018",
year="2017",
publisher="American Medical Association",
attachments=(storage_dir / "ATTKEY01" / "CPT 2018.pdf",),
)
def test_creates_source_item_with_tags_year_and_attachment(
self, store, storage_dir
):
book = self._cpt_book(storage_dir)
stats = import_books(
store, [book], tags=["source:ama", "module:coding", "llm:skip"]
)
assert stats["created"] == 1
assert stats["updated"] == 0
assert stats["attached"] == 1
(rec,) = stats["items"]
assert rec["zotero_key"] == "CPT2018K"
item = store.get(rec["key"])
assert item.item_type == "source"
assert item.title == "CPT 2018"
assert item.url == "https://ama.example/cpt2018"
assert set(item.tags) == {
"source:ama",
"module:coding",
"llm:skip",
"year:2017",
}
attachments = (
store._con()
.execute(
"SELECT filename FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(rec["key"],),
)
.fetchall()
)
assert [a[0] for a in attachments] == ["CPT 2018.pdf"]
def test_fallback_url_used_when_book_has_none(self, store):
book = ZoteroBook(
key="NOURLKEY",
title="No URL Book",
url="",
year="2020",
publisher="",
attachments=(),
)
import_books(store, [book], tags=["source:ama"])
item = store.get(store.list_items(query="No URL Book")[0].key)
assert item.url == "zotero://select/library/items/NOURLKEY"
def test_rerun_is_unchanged_with_no_duplicate_attachment(self, store, storage_dir):
book = self._cpt_book(storage_dir)
tags = ["source:ama", "module:coding", "llm:skip"]
first = import_books(store, [book], tags=tags)
second = import_books(store, [book], tags=tags)
assert second["created"] == 0
assert second["unchanged"] == 1
assert second["attached"] == 0
bib_key = first["items"][0]["key"]
rows = (
store._con()
.execute(
"SELECT filename FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(bib_key,),
)
.fetchall()
)
assert len(rows) == 1 # no duplicate attachment
def test_dry_run_makes_no_writes(self, store, storage_dir):
book = self._cpt_book(storage_dir)
stats = import_books(store, [book], tags=["source:ama"], dry_run=True)
assert stats["created"] == 1
assert stats["attached"] == 1
assert store.list_items() == [] # nothing written
def test_dry_run_after_real_import_reports_unchanged(self, store, storage_dir):
book = self._cpt_book(storage_dir)
tags = ["source:ama"]
import_books(store, [book], tags=tags)
stats = import_books(store, [book], tags=tags, dry_run=True)
assert stats["unchanged"] == 1
assert stats["attached"] == 0
assert stats["items"][0]["key"] # bib key resolved from the existing url
def test_missing_attachment_file_counts_and_skips(self, store, storage_dir):
ghost_path = storage_dir / "GHOSTKEY" / "missing.pdf"
book = ZoteroBook(
key="GHOSTBK1",
title="Ghost Book",
url="https://ama.example/ghost",
year="2020",
publisher="American Medical Association",
attachments=(ghost_path,),
)
stats = import_books(store, [book], tags=["source:ama"])
assert stats["missing"] == 1
assert stats["attached"] == 0
(rec,) = stats["items"]
assert rec["missing"] == 1
assert rec["missing_paths"] == [str(ghost_path)]
rows = (
store._con()
.execute(
"SELECT filename FROM attachments a "
"JOIN items i ON i.id = a.item_id WHERE i.key = ?",
(rec["key"],),
)
.fetchall()
)
assert rows == [] # no attachment row for the missing file