- _init_schema skips the duplicate-attachments GROUP BY scan once the unique index already exists (checks sqlite_master first). - upsert_status merges extra_json per key instead of replacing the whole column: a stored key now survives unless the incoming payload sets that specific key to a non-empty value. - docket_counts is one grouped query (a single items scan, left-joined against the enriched-tag set) instead of two separate url LIKE scans. - fingerprint_files widens `except FileNotFoundError` to `except OSError` so NotADirectoryError/PermissionError are handled the same way as a plain missing file. - backfill_details and backfill_from_mirror shared an identical sealed-guard block; factored into one _sealed_backfill_skip helper. Adds coverage: extra_json per-key merge (both directions), docket_counts, the widened OSError catch, attach_file's explicit-title dedupe path (two different source files, same explicit title), a dedupe-migration conflict test for a missing kept file, and a test confirming _init_schema's new short-circuit actually skips the scan.
146 lines
4.6 KiB
Python
146 lines
4.6 KiB
Python
"""Store.upsert must not rewrite rows/tags when nothing changed."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from bib.item import Source
|
|
from bib.store import Store
|
|
|
|
|
|
def _store() -> Store:
|
|
return Store(":memory:", storage_dir="/tmp/nope")
|
|
|
|
|
|
def _item(**kw) -> Source:
|
|
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
|
|
it.abstract = kw.get("abstract", "body")
|
|
for t in kw.get("tags", ["a:1", "b:2"]):
|
|
it.add_tag(t)
|
|
return it
|
|
|
|
|
|
def _snapshot(s: Store, key: str) -> tuple:
|
|
con = s._con()
|
|
row = con.execute(
|
|
"SELECT access_date, updated_at FROM items WHERE key=?", (key,)
|
|
).fetchone()
|
|
tags = con.execute(
|
|
"SELECT it.rowid FROM item_tags it JOIN items i ON i.id=it.item_id WHERE i.key=? ORDER BY 1",
|
|
(key,),
|
|
).fetchall()
|
|
return (row["access_date"], row["updated_at"], [t[0] for t in tags])
|
|
|
|
|
|
def test_first_upsert_is_created():
|
|
s = _store()
|
|
key, status = s.upsert_status(_item())
|
|
assert status == "created" and key
|
|
|
|
|
|
def test_identical_upsert_is_unchanged_and_writes_nothing():
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item())
|
|
before = _snapshot(s, key)
|
|
# different Python object, same content
|
|
key2, status = s.upsert_status(_item())
|
|
assert key2 == key and status == "unchanged"
|
|
assert _snapshot(s, key) == before # no access/updated stamp, no tag row churn
|
|
|
|
|
|
def test_changed_abstract_is_updated():
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item(abstract="v1"))
|
|
_, status = s.upsert_status(_item(abstract="v2"))
|
|
assert status == "updated"
|
|
assert s.get(key).abstract == "v2"
|
|
|
|
|
|
def test_new_tag_is_updated_and_merged():
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item(tags=["a:1"]))
|
|
_, status = s.upsert_status(_item(tags=["c:3"]))
|
|
assert status == "updated"
|
|
assert set(s.get(key).tags) >= {"a:1", "c:3"}
|
|
|
|
|
|
def test_subset_of_existing_tags_is_unchanged():
|
|
"""Upsert never removes tags (#624); a subset therefore changes nothing."""
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item(tags=["a:1", "b:2"]))
|
|
_, status = s.upsert_status(_item(tags=["a:1"]))
|
|
assert status == "unchanged"
|
|
|
|
|
|
def test_upsert_keeps_returning_key():
|
|
s = _store()
|
|
key = s.upsert(_item())
|
|
assert s.upsert(_item()) == key
|
|
|
|
|
|
def test_empty_incoming_abstract_keeps_stored_body():
|
|
"""A list-walk row (no body) must not blank an enriched comment."""
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item(abstract="enriched body"))
|
|
_, status = s.upsert_status(_item(abstract=""))
|
|
assert status == "unchanged"
|
|
assert s.get(key).abstract == "enriched body"
|
|
|
|
|
|
def test_empty_incoming_title_keeps_stored_title():
|
|
s = _store()
|
|
it = _item()
|
|
it.title = "Org: CMS-2026-2377-1"
|
|
key, _ = s.upsert_status(it)
|
|
it2 = _item()
|
|
it2.title = ""
|
|
_, status = s.upsert_status(it2)
|
|
assert status == "unchanged"
|
|
assert s.get(key).title == "Org: CMS-2026-2377-1"
|
|
|
|
|
|
def test_non_empty_incoming_still_updates():
|
|
s = _store()
|
|
key, _ = s.upsert_status(_item(abstract="v1"))
|
|
_, status = s.upsert_status(_item(abstract="v2"))
|
|
assert status == "updated" and s.get(key).abstract == "v2"
|
|
|
|
|
|
# ── extra_json merges per key (refs #680) ────────────────────────────
|
|
|
|
|
|
def _rule(**kw):
|
|
from bib.item import Rule
|
|
|
|
it = Rule(title="T", url="https://example.com/rule-1")
|
|
it.cms_id = kw.get("cms_id", "")
|
|
it.effective_date = kw.get("effective_date", "")
|
|
it.fr_page = kw.get("fr_page", "")
|
|
return it
|
|
|
|
|
|
def test_empty_incoming_extra_json_key_keeps_stored_key():
|
|
"""A list-walk Rule with a blank effective_date must not blank an
|
|
already-stored one — extra_json is merged per key, not replaced
|
|
wholesale."""
|
|
s = _store()
|
|
key, _ = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date="2026-01-01"))
|
|
_, status = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date=""))
|
|
assert status == "unchanged"
|
|
stored = s.get(key)
|
|
assert stored.effective_date == "2026-01-01"
|
|
assert stored.cms_id == "CMS-1848-P"
|
|
|
|
|
|
def test_non_empty_incoming_extra_json_key_updates_that_key_only():
|
|
s = _store()
|
|
key, _ = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date="2026-01-01"))
|
|
_, status = s.upsert_status(
|
|
_rule(cms_id="CMS-1848-P", effective_date="", fr_page="12345")
|
|
)
|
|
assert status == "updated"
|
|
stored = s.get(key)
|
|
# New key applied...
|
|
assert stored.fr_page == "12345"
|
|
# ...but the blank incoming effective_date did not blank the stored one.
|
|
assert stored.effective_date == "2026-01-01"
|
|
assert stored.cms_id == "CMS-1848-P"
|