Omitting clientDateModified lets SQLite's DEFAULT CURRENT_TIMESTAMP fill it with 'YYYY-MM-DD HH:MM:SS' (no T, no Z), which the Zotero sync engine rejects. All INSERT INTO items statements now include clientDateModified with the ISO 8601 now value.
198 lines
6.3 KiB
Python
198 lines
6.3 KiB
Python
"""Add 2010 and 2017 PFS carrier files to Zotero library.
|
|
|
|
Downloads from CMS, extracts, and creates Zotero items with proper
|
|
tags and storage-linked attachments.
|
|
|
|
Usage:
|
|
uv run python dev/scripts/add_carrier_to_zotero.py
|
|
"""
|
|
|
|
import os
|
|
import random
|
|
import shutil
|
|
import sqlite3
|
|
import string
|
|
import subprocess
|
|
import zipfile
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
|
|
from conf import path as _conf_path
|
|
|
|
ZOTERO_DB = str(_conf_path("db.zotero"))
|
|
ZOTERO_STORAGE = str(_conf_path("storage.zotero"))
|
|
|
|
# Files already downloaded to /tmp
|
|
CARRIER_FILES = {
|
|
2010: {
|
|
"zip": "/tmp/pfs_carrier_dl/cy2010_carrier.zip",
|
|
"extracted": "/tmp/pfs_carrier_dl/2010",
|
|
"url": "https://www.cms.gov/medicare/medicare-fee-for-service-payment/physicianfeesched/downloads/cy2010qtr1carrierfiles3.zip",
|
|
"title": "CY 2010 PFS Carrier — Cy2010 Carrier Files",
|
|
},
|
|
2017: {
|
|
"zip": "/tmp/pfs_carrier_dl/cy2017_carrier.zip",
|
|
"extracted": "/tmp/pfs_carrier_dl/2017",
|
|
"url": "https://www.cms.gov/medicare/medicare-fee-for-service-payment/physicianfeesched/downloads/cy2017-carrierfiles.zip",
|
|
"title": "CY 2017 PFS Carrier — Cy2017 Carrier Files",
|
|
},
|
|
}
|
|
|
|
# Zotero schema constants
|
|
ITEM_TYPE_WEBPAGE = 40
|
|
ITEM_TYPE_ATTACHMENT = 3
|
|
FIELD_TITLE = 1
|
|
FIELD_DATE = 6
|
|
FIELD_URL = 10
|
|
FIELD_ACCESS_DATE = 11
|
|
FIELD_WEBSITE_TYPE = 42
|
|
FIELD_WEBSITE_TITLE = 123
|
|
TAG_MODULE_PFS = 8
|
|
TAG_YEARS = {2010: 36, 2017: 20}
|
|
|
|
|
|
def _zotero_key() -> str:
|
|
"""Generate a random 8-char Zotero key using the allowed character set."""
|
|
chars = "23456789ABCDEFGHIJKLMNPQRSTUVWXYZ"
|
|
return "".join(random.choices(chars, k=8))
|
|
|
|
|
|
def _get_or_create_value(db: sqlite3.Connection, value: str) -> int:
|
|
"""Get or create an itemDataValues row, return valueID."""
|
|
row = db.execute(
|
|
"SELECT valueID FROM itemDataValues WHERE value = ?", (value,)
|
|
).fetchone()
|
|
if row:
|
|
return row[0]
|
|
cur = db.execute("INSERT INTO itemDataValues (value) VALUES (?)", (value,))
|
|
return cur.lastrowid
|
|
|
|
|
|
def _next_item_id(db: sqlite3.Connection) -> int:
|
|
return db.execute("SELECT max(itemID) + 1 FROM items").fetchone()[0]
|
|
|
|
|
|
def add_carrier_year(db: sqlite3.Connection, year: int, info: dict) -> None:
|
|
"""Add one carrier year to Zotero: parent item + file attachments."""
|
|
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
|
|
# --- Create parent item (webpage) ---
|
|
parent_id = _next_item_id(db)
|
|
parent_key = _zotero_key()
|
|
db.execute(
|
|
"""INSERT INTO items (itemID, itemTypeID, dateAdded, dateModified,
|
|
clientDateModified, key, version, synced, libraryID)
|
|
VALUES (?, ?, ?, ?, ?, ?, 0, 0, 1)""",
|
|
(parent_id, ITEM_TYPE_WEBPAGE, now, now, now, parent_key),
|
|
)
|
|
|
|
# Set fields: title, date, url, accessDate, websiteType, websiteTitle
|
|
fields = {
|
|
FIELD_TITLE: info["title"],
|
|
FIELD_DATE: f"{year}-01-01",
|
|
FIELD_URL: info["url"],
|
|
FIELD_ACCESS_DATE: now,
|
|
FIELD_WEBSITE_TYPE: "Government Data Portal",
|
|
FIELD_WEBSITE_TITLE: "Centers for Medicare & Medicaid Services",
|
|
}
|
|
for field_id, value in fields.items():
|
|
value_id = _get_or_create_value(db, value)
|
|
db.execute(
|
|
"INSERT INTO itemData (itemID, fieldID, valueID) VALUES (?, ?, ?)",
|
|
(parent_id, field_id, value_id),
|
|
)
|
|
|
|
# Tags: module:pfs + year:YYYY
|
|
db.execute(
|
|
"INSERT INTO itemTags (itemID, tagID, type) VALUES (?, ?, 0)",
|
|
(parent_id, TAG_MODULE_PFS),
|
|
)
|
|
db.execute(
|
|
"INSERT INTO itemTags (itemID, tagID, type) VALUES (?, ?, 0)",
|
|
(parent_id, TAG_YEARS[year]),
|
|
)
|
|
|
|
print(f"Created parent item: {parent_key} ({info['title']})")
|
|
|
|
# --- Create attachments for each .TXT and .pdf file ---
|
|
extracted_dir = Path(info["extracted"])
|
|
files = sorted(extracted_dir.iterdir())
|
|
att_count = 0
|
|
|
|
for filepath in files:
|
|
if filepath.suffix.upper() not in (".TXT", ".PDF"):
|
|
continue
|
|
|
|
att_id = _next_item_id(db)
|
|
att_key = _zotero_key()
|
|
|
|
# Copy file to Zotero storage (owned by container uid 100999)
|
|
storage_dir = Path(ZOTERO_STORAGE) / att_key
|
|
subprocess.run(["sudo", "mkdir", "-p", str(storage_dir)], check=True)
|
|
dest = storage_dir / filepath.name
|
|
subprocess.run(["sudo", "cp", str(filepath), str(dest)], check=True)
|
|
subprocess.run(
|
|
["sudo", "chown", "-R", "100999:100999", str(storage_dir)],
|
|
check=True,
|
|
)
|
|
|
|
# Determine content type
|
|
ext = filepath.suffix.upper()
|
|
content_type = "text/plain" if ext == ".TXT" else "application/pdf"
|
|
|
|
# Create item record
|
|
db.execute(
|
|
"""INSERT INTO items (itemID, itemTypeID, dateAdded, dateModified,
|
|
key, version, synced, libraryID)
|
|
VALUES (?, ?, ?, ?, ?, 0, 0, 1)""",
|
|
(att_id, ITEM_TYPE_ATTACHMENT, now, now, now, att_key),
|
|
)
|
|
|
|
# Create attachment record (linkMode=0 = imported file)
|
|
db.execute(
|
|
"""INSERT INTO itemAttachments
|
|
(itemID, parentItemID, linkMode, contentType, path)
|
|
VALUES (?, ?, 0, ?, ?)""",
|
|
(att_id, parent_id, content_type, f"storage:{filepath.name}"),
|
|
)
|
|
|
|
# Set title field on attachment
|
|
value_id = _get_or_create_value(db, filepath.name)
|
|
db.execute(
|
|
"INSERT INTO itemData (itemID, fieldID, valueID) VALUES (?, ?, ?)",
|
|
(att_id, FIELD_TITLE, value_id),
|
|
)
|
|
|
|
att_count += 1
|
|
|
|
print(f" Added {att_count} attachments for year {year}")
|
|
|
|
|
|
def main() -> None:
|
|
# Verify extracted files exist
|
|
for year, info in CARRIER_FILES.items():
|
|
extracted = Path(info["extracted"])
|
|
if not extracted.exists():
|
|
print(f"Extracting {info['zip']}...")
|
|
with zipfile.ZipFile(info["zip"]) as zf:
|
|
zf.extractall(extracted)
|
|
|
|
txt_count = len(list(extracted.glob("*.TXT")))
|
|
print(f"Year {year}: {txt_count} .TXT files ready")
|
|
|
|
db = sqlite3.connect(ZOTERO_DB)
|
|
try:
|
|
for year, info in sorted(CARRIER_FILES.items()):
|
|
add_carrier_year(db, year, info)
|
|
db.commit()
|
|
print("\nDone. Committed to Zotero database.")
|
|
except Exception:
|
|
db.rollback()
|
|
raise
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|