feat(zotero): mirror AMA coding/CPT publications into a collection
All checks were successful
CI / lint (pull_request) Successful in 29s
CI / notebooks-smoke (pull_request) Successful in 1m27s
Infra CI / notebooks (pull_request) Successful in 53s
Infra CI / zotero (pull_request) Successful in 16s
Infra CI / docs (pull_request) Successful in 13s
Infra CI / api (pull_request) Successful in 23s
Infra CI / llm (pull_request) Successful in 12s
Infra CI / mc (pull_request) Successful in 13s
CI / lint (push) Successful in 32s
CI / test (pull_request) Successful in 14m33s
CI / notebooks-smoke (push) Successful in 1m25s
CI / test (push) Successful in 12m40s
All checks were successful
CI / lint (pull_request) Successful in 29s
CI / notebooks-smoke (pull_request) Successful in 1m27s
Infra CI / notebooks (pull_request) Successful in 53s
Infra CI / zotero (pull_request) Successful in 16s
Infra CI / docs (pull_request) Successful in 13s
Infra CI / api (pull_request) Successful in 23s
Infra CI / llm (pull_request) Successful in 12s
Infra CI / mc (pull_request) Successful in 13s
CI / lint (push) Successful in 32s
CI / test (pull_request) Successful in 14m33s
CI / notebooks-smoke (push) Successful in 1m25s
CI / test (push) Successful in 12m40s
One book item per OpenLibrary work from data/ama/ama_coding_titles.json under 'AMA Coding Publications', with ISBNs, year tags, and the OL URL as the dedupe key. Attaches ebooks LazyLibrarian has landed under the books library; re-runs only add what is new.
This commit is contained in:
174
dev/scripts/add_ama_coding_to_zotero.py
Normal file
174
dev/scripts/add_ama_coding_to_zotero.py
Normal file
@@ -0,0 +1,174 @@
|
|||||||
|
"""Mirror the AMA coding/CPT publications into Zotero.
|
||||||
|
|
||||||
|
Reads the OpenLibrary-derived title list (``data/ama/ama_coding_titles.json``,
|
||||||
|
one entry per OpenLibrary work: title, first publish year, ISBNs, OL key)
|
||||||
|
and creates one ``book`` item per work under a top-level collection, then
|
||||||
|
attaches whatever ebook LazyLibrarian has already landed under the books
|
||||||
|
library for that title. Re-running is safe: items are matched on their
|
||||||
|
OpenLibrary URL, attachments on file name.
|
||||||
|
|
||||||
|
Tags:
|
||||||
|
- source:ama
|
||||||
|
- module:coding
|
||||||
|
- year:YYYY
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
uv run python dev/scripts/add_ama_coding_to_zotero.py [--titles PATH]
|
||||||
|
[--books DIR] [--collection NAME] [--dry-run]
|
||||||
|
|
||||||
|
Restart Zotero afterwards so the desktop picks up the new rows.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from conf import path as _conf_path
|
||||||
|
from zot.db import TYPE_MAP, Db, generate_key, now_iso
|
||||||
|
|
||||||
|
ZOTERO_DB = str(_conf_path("db.zotero"))
|
||||||
|
ZOTERO_STORAGE = Path(str(_conf_path("storage.zotero")))
|
||||||
|
|
||||||
|
DEFAULT_TITLES = Path("data/ama/ama_coding_titles.json")
|
||||||
|
DEFAULT_BOOKS = Path("/srv/pool/media/books/American Medical Association")
|
||||||
|
DEFAULT_COLLECTION = "AMA Coding Publications"
|
||||||
|
PUBLISHER = "American Medical Association"
|
||||||
|
CONTENT_TYPES = {
|
||||||
|
".epub": "application/epub+zip",
|
||||||
|
".pdf": "application/pdf",
|
||||||
|
".mobi": "application/x-mobipocket-ebook",
|
||||||
|
".azw3": "application/vnd.amazon.mobi8-ebook",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _norm(s: str) -> str:
|
||||||
|
return re.sub(r"[^a-z0-9]+", " ", s.lower()).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _ol_url(key: str) -> str:
|
||||||
|
olid = key.rsplit("/", 1)[-1]
|
||||||
|
kind = "works" if olid.endswith("W") else "books"
|
||||||
|
return f"https://openlibrary.org/{kind}/{olid}"
|
||||||
|
|
||||||
|
|
||||||
|
def _index_books(books_dir: Path) -> dict[str, list[Path]]:
|
||||||
|
"""Map normalised folder title -> ebook files LazyLibrarian has landed."""
|
||||||
|
index: dict[str, list[Path]] = {}
|
||||||
|
if not books_dir.is_dir():
|
||||||
|
return index
|
||||||
|
for folder in books_dir.iterdir():
|
||||||
|
if not folder.is_dir():
|
||||||
|
continue
|
||||||
|
files = [p for p in folder.iterdir() if p.suffix.lower() in CONTENT_TYPES]
|
||||||
|
if files:
|
||||||
|
index[_norm(folder.name)] = sorted(files)
|
||||||
|
return index
|
||||||
|
|
||||||
|
|
||||||
|
def _existing_attachment_names(db: Db, parent_id: int) -> set[str]:
|
||||||
|
rows = db.con.execute(
|
||||||
|
"SELECT path FROM itemAttachments WHERE parentItemID = ?", (parent_id,)
|
||||||
|
).fetchall()
|
||||||
|
return {str(r[0]).removeprefix("storage:") for r in rows if r[0]}
|
||||||
|
|
||||||
|
|
||||||
|
def _attach(db: Db, parent_id: int, src: Path, *, dry_run: bool) -> None:
|
||||||
|
att_key = generate_key()
|
||||||
|
storage_dir = ZOTERO_STORAGE / att_key
|
||||||
|
if dry_run:
|
||||||
|
print(f" would attach {src.name}")
|
||||||
|
return
|
||||||
|
subprocess.run(["sudo", "mkdir", "-p", str(storage_dir)], check=True)
|
||||||
|
subprocess.run(["sudo", "cp", str(src), str(storage_dir / src.name)], check=True)
|
||||||
|
subprocess.run(["sudo", "chown", "-R", "100999:100999", str(storage_dir)], check=True)
|
||||||
|
att_id = db.add_attachment(
|
||||||
|
parent_id,
|
||||||
|
key=att_key,
|
||||||
|
content_type=CONTENT_TYPES[src.suffix.lower()],
|
||||||
|
path=f"storage:{src.name}",
|
||||||
|
)
|
||||||
|
db.set_field(att_id, "title", src.name)
|
||||||
|
print(f" attached {src.name} ({att_key})")
|
||||||
|
|
||||||
|
|
||||||
|
def _ensure_item(db: Db, entry: dict, collection_key: str, *, dry_run: bool) -> int | None:
|
||||||
|
url = _ol_url(entry["key"])
|
||||||
|
item_id = db.find_item_by_url(url)
|
||||||
|
if item_id is not None:
|
||||||
|
db.add_to_collection(item_id, collection_key=collection_key)
|
||||||
|
return item_id
|
||||||
|
if dry_run:
|
||||||
|
print(f" would create: {entry['title']}")
|
||||||
|
return None
|
||||||
|
now = now_iso()
|
||||||
|
item_id = db.create_item(TYPE_MAP["book"], now=now)
|
||||||
|
fields = {
|
||||||
|
"title": entry["title"],
|
||||||
|
"publisher": PUBLISHER,
|
||||||
|
"url": url,
|
||||||
|
"accessDate": now,
|
||||||
|
"language": "en",
|
||||||
|
"libraryCatalog": "Open Library",
|
||||||
|
"extra": f"OpenLibrary: {entry['key'].rsplit('/', 1)[-1]}",
|
||||||
|
}
|
||||||
|
if entry.get("year"):
|
||||||
|
fields["date"] = str(entry["year"])
|
||||||
|
if entry.get("isbns"):
|
||||||
|
fields["ISBN"] = " ".join(entry["isbns"])
|
||||||
|
db.set_fields(item_id, fields)
|
||||||
|
db.add_creators(item_id, [("", PUBLISHER)], creator_type="author")
|
||||||
|
tags = ["source:ama", "module:coding"]
|
||||||
|
if entry.get("year"):
|
||||||
|
tags.append(f"year:{entry['year']}")
|
||||||
|
db.sync_tags(item_id, tags)
|
||||||
|
db.add_to_collection(item_id, collection_key=collection_key)
|
||||||
|
print(f" created: {entry['title']}")
|
||||||
|
return item_id
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__ or "")
|
||||||
|
ap.add_argument("--titles", type=Path, default=DEFAULT_TITLES)
|
||||||
|
ap.add_argument("--books", type=Path, default=DEFAULT_BOOKS)
|
||||||
|
ap.add_argument("--collection", default=DEFAULT_COLLECTION)
|
||||||
|
ap.add_argument("--dry-run", action="store_true")
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
entries = json.loads(args.titles.read_text())
|
||||||
|
books = _index_books(args.books)
|
||||||
|
print(f"{len(entries)} titles, {len(books)} ebook folders on disk")
|
||||||
|
|
||||||
|
created = attached = existing = 0
|
||||||
|
with Db(ZOTERO_DB) as db:
|
||||||
|
collection_key = db.ensure_collection(args.collection)
|
||||||
|
for entry in entries:
|
||||||
|
before = db.find_item_by_url(_ol_url(entry["key"]))
|
||||||
|
item_id = _ensure_item(db, entry, collection_key, dry_run=args.dry_run)
|
||||||
|
if item_id is None:
|
||||||
|
continue
|
||||||
|
if before is None:
|
||||||
|
created += 1
|
||||||
|
else:
|
||||||
|
existing += 1
|
||||||
|
have = _existing_attachment_names(db, item_id)
|
||||||
|
for f in books.get(_norm(entry["title"]), []):
|
||||||
|
if f.name in have:
|
||||||
|
continue
|
||||||
|
_attach(db, item_id, f, dry_run=args.dry_run)
|
||||||
|
attached += 1
|
||||||
|
if not args.dry_run:
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
print()
|
||||||
|
print(f"collection '{args.collection}' ({collection_key}): "
|
||||||
|
f"{created} created, {existing} already present, {attached} files attached")
|
||||||
|
if not args.dry_run:
|
||||||
|
print("Restart Zotero so the desktop reloads the database.")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user