Files
stack/dev/scripts/add_carrier_to_zotero.py
kert 2cfd61a57a fix dev scripts: explicitly set clientDateModified in Zotero INSERTs
Omitting clientDateModified lets SQLite's DEFAULT CURRENT_TIMESTAMP
fill it with 'YYYY-MM-DD HH:MM:SS' (no T, no Z), which the Zotero
sync engine rejects. All INSERT INTO items statements now include
clientDateModified with the ISO 8601 now value.
2026-03-22 00:09:24 -04:00

198 lines
6.3 KiB
Python

"""Add 2010 and 2017 PFS carrier files to Zotero library.
Downloads from CMS, extracts, and creates Zotero items with proper
tags and storage-linked attachments.
Usage:
uv run python dev/scripts/add_carrier_to_zotero.py
"""
import os
import random
import shutil
import sqlite3
import string
import subprocess
import zipfile
from datetime import datetime, timezone
from pathlib import Path
from conf import path as _conf_path
ZOTERO_DB = str(_conf_path("db.zotero"))
ZOTERO_STORAGE = str(_conf_path("storage.zotero"))
# Files already downloaded to /tmp
CARRIER_FILES = {
2010: {
"zip": "/tmp/pfs_carrier_dl/cy2010_carrier.zip",
"extracted": "/tmp/pfs_carrier_dl/2010",
"url": "https://www.cms.gov/medicare/medicare-fee-for-service-payment/physicianfeesched/downloads/cy2010qtr1carrierfiles3.zip",
"title": "CY 2010 PFS Carrier — Cy2010 Carrier Files",
},
2017: {
"zip": "/tmp/pfs_carrier_dl/cy2017_carrier.zip",
"extracted": "/tmp/pfs_carrier_dl/2017",
"url": "https://www.cms.gov/medicare/medicare-fee-for-service-payment/physicianfeesched/downloads/cy2017-carrierfiles.zip",
"title": "CY 2017 PFS Carrier — Cy2017 Carrier Files",
},
}
# Zotero schema constants
ITEM_TYPE_WEBPAGE = 40
ITEM_TYPE_ATTACHMENT = 3
FIELD_TITLE = 1
FIELD_DATE = 6
FIELD_URL = 10
FIELD_ACCESS_DATE = 11
FIELD_WEBSITE_TYPE = 42
FIELD_WEBSITE_TITLE = 123
TAG_MODULE_PFS = 8
TAG_YEARS = {2010: 36, 2017: 20}
def _zotero_key() -> str:
"""Generate a random 8-char Zotero key using the allowed character set."""
chars = "23456789ABCDEFGHIJKLMNPQRSTUVWXYZ"
return "".join(random.choices(chars, k=8))
def _get_or_create_value(db: sqlite3.Connection, value: str) -> int:
"""Get or create an itemDataValues row, return valueID."""
row = db.execute(
"SELECT valueID FROM itemDataValues WHERE value = ?", (value,)
).fetchone()
if row:
return row[0]
cur = db.execute("INSERT INTO itemDataValues (value) VALUES (?)", (value,))
return cur.lastrowid
def _next_item_id(db: sqlite3.Connection) -> int:
return db.execute("SELECT max(itemID) + 1 FROM items").fetchone()[0]
def add_carrier_year(db: sqlite3.Connection, year: int, info: dict) -> None:
"""Add one carrier year to Zotero: parent item + file attachments."""
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
# --- Create parent item (webpage) ---
parent_id = _next_item_id(db)
parent_key = _zotero_key()
db.execute(
"""INSERT INTO items (itemID, itemTypeID, dateAdded, dateModified,
clientDateModified, key, version, synced, libraryID)
VALUES (?, ?, ?, ?, ?, ?, 0, 0, 1)""",
(parent_id, ITEM_TYPE_WEBPAGE, now, now, now, parent_key),
)
# Set fields: title, date, url, accessDate, websiteType, websiteTitle
fields = {
FIELD_TITLE: info["title"],
FIELD_DATE: f"{year}-01-01",
FIELD_URL: info["url"],
FIELD_ACCESS_DATE: now,
FIELD_WEBSITE_TYPE: "Government Data Portal",
FIELD_WEBSITE_TITLE: "Centers for Medicare & Medicaid Services",
}
for field_id, value in fields.items():
value_id = _get_or_create_value(db, value)
db.execute(
"INSERT INTO itemData (itemID, fieldID, valueID) VALUES (?, ?, ?)",
(parent_id, field_id, value_id),
)
# Tags: module:pfs + year:YYYY
db.execute(
"INSERT INTO itemTags (itemID, tagID, type) VALUES (?, ?, 0)",
(parent_id, TAG_MODULE_PFS),
)
db.execute(
"INSERT INTO itemTags (itemID, tagID, type) VALUES (?, ?, 0)",
(parent_id, TAG_YEARS[year]),
)
print(f"Created parent item: {parent_key} ({info['title']})")
# --- Create attachments for each .TXT and .pdf file ---
extracted_dir = Path(info["extracted"])
files = sorted(extracted_dir.iterdir())
att_count = 0
for filepath in files:
if filepath.suffix.upper() not in (".TXT", ".PDF"):
continue
att_id = _next_item_id(db)
att_key = _zotero_key()
# Copy file to Zotero storage (owned by container uid 100999)
storage_dir = Path(ZOTERO_STORAGE) / att_key
subprocess.run(["sudo", "mkdir", "-p", str(storage_dir)], check=True)
dest = storage_dir / filepath.name
subprocess.run(["sudo", "cp", str(filepath), str(dest)], check=True)
subprocess.run(
["sudo", "chown", "-R", "100999:100999", str(storage_dir)],
check=True,
)
# Determine content type
ext = filepath.suffix.upper()
content_type = "text/plain" if ext == ".TXT" else "application/pdf"
# Create item record
db.execute(
"""INSERT INTO items (itemID, itemTypeID, dateAdded, dateModified,
key, version, synced, libraryID)
VALUES (?, ?, ?, ?, ?, 0, 0, 1)""",
(att_id, ITEM_TYPE_ATTACHMENT, now, now, now, att_key),
)
# Create attachment record (linkMode=0 = imported file)
db.execute(
"""INSERT INTO itemAttachments
(itemID, parentItemID, linkMode, contentType, path)
VALUES (?, ?, 0, ?, ?)""",
(att_id, parent_id, content_type, f"storage:{filepath.name}"),
)
# Set title field on attachment
value_id = _get_or_create_value(db, filepath.name)
db.execute(
"INSERT INTO itemData (itemID, fieldID, valueID) VALUES (?, ?, ?)",
(att_id, FIELD_TITLE, value_id),
)
att_count += 1
print(f" Added {att_count} attachments for year {year}")
def main() -> None:
# Verify extracted files exist
for year, info in CARRIER_FILES.items():
extracted = Path(info["extracted"])
if not extracted.exists():
print(f"Extracting {info['zip']}...")
with zipfile.ZipFile(info["zip"]) as zf:
zf.extractall(extracted)
txt_count = len(list(extracted.glob("*.TXT")))
print(f"Year {year}: {txt_count} .TXT files ready")
db = sqlite3.connect(ZOTERO_DB)
try:
for year, info in sorted(CARRIER_FILES.items()):
add_carrier_year(db, year, info)
db.commit()
print("\nDone. Committed to Zotero database.")
except Exception:
db.rollback()
raise
finally:
db.close()
if __name__ == "__main__":
main()