chore(scripts): remove one-shot comment-md→notes migration

Run successfully against the corpus: bib + Zotero now hold 23,610
'Comment text' notes (rendered HTML, wrapped in zotero-note
envelope). All .md file attachments are gone from both stores.
Restoring this script for re-runs is one git revert away.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
kert
2026-05-07 09:05:22 -04:00
parent 5aa043daea
commit eaead0c348

View File

@@ -1,260 +0,0 @@
#!/usr/bin/env python
"""One-shot: comment .md attachments → Zotero notes.
Phases:
A (default): for every .state/comments/<docket>/<id>/combined.md,
attach a single 'Comment text' note (rendered HTML)
to the bib item. Idempotent; safe to retry.
C (--cleanup): destructively delete every .md row from bib.attachments,
matching itemAttachments + items rows in Zotero, and the
per-attachment storage dirs. Prompts before destruction.
Phase B (push notes to Zotero) lives outside this script — operator runs
'stack bib sync' between A and C.
After a successful migration, this file should be removed in a follow-up
commit.
"""
from __future__ import annotations
import argparse
import logging
import shutil
import sqlite3
import sys
from pathlib import Path
from cli.comments import NOTE_TITLE
logging.basicConfig(format="%(asctime)s %(levelname)s %(message)s", level=logging.INFO)
log = logging.getLogger("migrate-comments-md-to-notes")
COMMENTS_ROOT = Path(".state/comments")
def _safety_preflight() -> None:
"""Refuse to run unless _sync_notes is on disk — sanity check that
re-syncing won't recreate the same problem we're cleaning up."""
sync_py = Path("src/bib/sync.py")
if not sync_py.is_file():
sys.exit("FATAL: src/bib/sync.py missing — run from repo root.")
if "_sync_notes" not in sync_py.read_text(encoding="utf-8"):
sys.exit(
"FATAL: src/bib/sync.py has no _sync_notes — apply Task 3 first."
)
def phase_a_backfill(root: Path) -> dict[str, int]:
"""Walk comment dirs, attach combined.md as a 'Comment text' note."""
from bib import connect
from rex.comments.render import render_combined_md
store = connect()
con = store._con() # noqa: SLF001
# comment_id → bib item_key
keys: dict[str, str] = {
(row[1] or "").rsplit("/", 1)[-1]: row[0]
for row in con.execute(
"SELECT key, url FROM items "
"WHERE url LIKE 'https://www.regulations.gov/comment/%'"
)
}
# item_keys that already have a 'Comment text' note → skip
have_note: set[str] = {
row[0]
for row in con.execute(
"SELECT i.key FROM notes n "
"JOIN items i ON n.item_id = i.id "
"WHERE n.title = ?",
(NOTE_TITLE,),
)
}
stats = {"scanned": 0, "attached": 0, "skipped_have_note": 0,
"skipped_unknown_item": 0, "errors": 0}
for combined in root.rglob("combined.md"):
stats["scanned"] += 1
comment_id = combined.parent.name
item_key = keys.get(comment_id)
if not item_key:
stats["skipped_unknown_item"] += 1
continue
if item_key in have_note:
stats["skipped_have_note"] += 1
continue
try:
html = render_combined_md(combined.read_text(encoding="utf-8"))
store.attach_note(item_key, html, title=NOTE_TITLE)
have_note.add(item_key)
stats["attached"] += 1
except Exception as e: # noqa: BLE001
log.warning("attach_note failed for %s: %s", item_key, e)
stats["errors"] += 1
if stats["scanned"] % 1000 == 0:
log.info("phase A progress: %s", stats)
return stats
def _confirm(msg: str) -> bool:
"""Prompt y/N. Defaults to N."""
reply = input(f"{msg} [y/N] ").strip().lower()
return reply == "y"
def _phase_c_preview() -> dict[str, int]:
from conf import path
bib_db = sqlite3.connect(str(path("db.bib")))
n_bib = bib_db.execute(
"SELECT COUNT(*) FROM attachments WHERE filename LIKE '%.md'"
).fetchone()[0]
bib_db.close()
zot_db = sqlite3.connect(str(path("db.zotero")))
n_zot = zot_db.execute(
"SELECT COUNT(*) FROM itemAttachments WHERE path LIKE 'storage:%.md'"
).fetchone()[0]
zot_db.close()
return {"bib_md_attachments": n_bib, "zotero_md_attachments": n_zot}
def phase_c_cleanup() -> dict[str, int]:
"""Delete every .md attachment row + storage dir on bib and Zotero."""
from conf import path
stats = {"bib_rows_deleted": 0, "bib_dirs_removed": 0,
"bib_dirs_skipped_nonempty": 0, "bib_dirs_rmtree_failed": 0,
"zot_rows_deleted": 0, "zot_dirs_removed": 0,
"zot_dirs_skipped_nonempty": 0, "zot_dirs_rmtree_failed": 0}
# ── bib ──────────────────────────────────────────
# bib's storage layout is one attachment per dir (att_key/<filename>),
# so the parent dir should contain only the .md we're deleting. Guard
# against unexpected siblings the same way the Zotero branch does.
bib_con = sqlite3.connect(str(path("db.bib")))
bib_rows = bib_con.execute(
"SELECT id, storage_path FROM attachments WHERE filename LIKE '%.md'"
).fetchall()
for _att_id, storage_path in bib_rows:
if not storage_path:
continue
d = Path(storage_path).parent
if not d.is_dir():
continue
non_md = [p for p in d.iterdir() if not p.name.endswith(".md")]
if non_md:
log.warning("skip bib storage dir %s (contains non-md: %s)",
d, [p.name for p in non_md])
stats["bib_dirs_skipped_nonempty"] += 1
continue
try:
shutil.rmtree(d)
except OSError as e:
log.warning("rmtree failed for %s: %s", d, e)
stats["bib_dirs_rmtree_failed"] += 1
continue
stats["bib_dirs_removed"] += 1
bib_con.execute("DELETE FROM attachments WHERE filename LIKE '%.md'")
stats["bib_rows_deleted"] = bib_con.total_changes
bib_con.commit()
bib_con.close()
# ── Zotero ───────────────────────────────────────
zot_db_path = str(path("db.zotero"))
zot_storage = Path(zot_db_path).parent / "storage"
zot_con = sqlite3.connect(zot_db_path)
zot_rows = zot_con.execute(
"SELECT i.itemID, i.key FROM items i "
"JOIN itemAttachments ia ON ia.itemID = i.itemID "
"WHERE ia.path LIKE 'storage:%.md'"
).fetchall()
item_ids_to_delete: list[int] = []
for item_id, item_key in zot_rows:
d = zot_storage / item_key
if d.is_dir():
non_md = [p for p in d.iterdir() if not p.name.endswith(".md")]
if non_md:
log.warning("skip zot storage dir %s (contains non-md: %s)",
d, [p.name for p in non_md])
stats["zot_dirs_skipped_nonempty"] += 1
# We still drop the rows — Zotero will show a missing
# attachment, easier to clean than a phantom row.
else:
try:
shutil.rmtree(d)
stats["zot_dirs_removed"] += 1
except OSError as e:
log.warning("rmtree failed for %s: %s", d, e)
stats["zot_dirs_rmtree_failed"] += 1
item_ids_to_delete.append(item_id)
if item_ids_to_delete:
# SQLite caps host parameters per statement (default 32766 on 3.32+,
# 999 on older builds). 33k+ IDs at once trips it — chunk to be safe.
BATCH = 500
for i in range(0, len(item_ids_to_delete), BATCH):
chunk = item_ids_to_delete[i : i + BATCH]
placeholders = ",".join("?" * len(chunk))
zot_con.execute(
f"DELETE FROM itemAttachments WHERE itemID IN ({placeholders})",
chunk,
)
zot_con.execute(
f"DELETE FROM items WHERE itemID IN ({placeholders})",
chunk,
)
stats["zot_rows_deleted"] = len(item_ids_to_delete)
zot_con.commit()
zot_con.close()
return stats
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument(
"--root", type=Path, default=COMMENTS_ROOT,
help="Comments root (default: .state/comments)",
)
parser.add_argument(
"--cleanup", action="store_true",
help="After Phase A, also run Phase C (destructive: deletes all "
"*.md rows from bib + Zotero and removes their storage dirs).",
)
args = parser.parse_args()
_safety_preflight()
if not args.root.is_dir():
sys.exit(f"FATAL: comments root not found: {args.root}")
log.info("Phase A: backfilling notes from %s", args.root)
stats_a = phase_a_backfill(args.root)
log.info("Phase A complete: %s", stats_a)
if not args.cleanup:
log.info(
"Done. Run 'stack bib sync' to push the notes to Zotero, "
"then re-run with --cleanup to delete the old .md attachments."
)
return
log.info("Phase C: destructive cleanup of .md file attachments")
counts = _phase_c_preview()
log.info("Will delete: %s", counts)
if not _confirm("Proceed with destructive cleanup?"):
log.info("Aborted.")
return
stats_c = phase_c_cleanup()
log.info("Phase C complete: %s", stats_c)
if __name__ == "__main__":
main()