Files
stack/tests/bib/test_store_upsert_noop.py
kert f61eae8656 fix(bib): stop over-scanning on Store open, upsert_status, and sealed backfills (refs #680)
- _init_schema skips the duplicate-attachments GROUP BY scan once the
  unique index already exists (checks sqlite_master first).
- upsert_status merges extra_json per key instead of replacing the
  whole column: a stored key now survives unless the incoming payload
  sets that specific key to a non-empty value.
- docket_counts is one grouped query (a single items scan, left-joined
  against the enriched-tag set) instead of two separate url LIKE scans.
- fingerprint_files widens `except FileNotFoundError` to `except
  OSError` so NotADirectoryError/PermissionError are handled the same
  way as a plain missing file.
- backfill_details and backfill_from_mirror shared an identical
  sealed-guard block; factored into one _sealed_backfill_skip helper.

Adds coverage: extra_json per-key merge (both directions),
docket_counts, the widened OSError catch, attach_file's explicit-title
dedupe path (two different source files, same explicit title), a
dedupe-migration conflict test for a missing kept file, and a test
confirming _init_schema's new short-circuit actually skips the scan.
2026-09-11 17:23:07 -04:00

146 lines
4.6 KiB
Python

"""Store.upsert must not rewrite rows/tags when nothing changed."""
from __future__ import annotations
from bib.item import Source
from bib.store import Store
def _store() -> Store:
return Store(":memory:", storage_dir="/tmp/nope")
def _item(**kw) -> Source:
it = Source(title="T", url="https://www.regulations.gov/comment/CMS-2026-2377-1")
it.abstract = kw.get("abstract", "body")
for t in kw.get("tags", ["a:1", "b:2"]):
it.add_tag(t)
return it
def _snapshot(s: Store, key: str) -> tuple:
con = s._con()
row = con.execute(
"SELECT access_date, updated_at FROM items WHERE key=?", (key,)
).fetchone()
tags = con.execute(
"SELECT it.rowid FROM item_tags it JOIN items i ON i.id=it.item_id WHERE i.key=? ORDER BY 1",
(key,),
).fetchall()
return (row["access_date"], row["updated_at"], [t[0] for t in tags])
def test_first_upsert_is_created():
s = _store()
key, status = s.upsert_status(_item())
assert status == "created" and key
def test_identical_upsert_is_unchanged_and_writes_nothing():
s = _store()
key, _ = s.upsert_status(_item())
before = _snapshot(s, key)
# different Python object, same content
key2, status = s.upsert_status(_item())
assert key2 == key and status == "unchanged"
assert _snapshot(s, key) == before # no access/updated stamp, no tag row churn
def test_changed_abstract_is_updated():
s = _store()
key, _ = s.upsert_status(_item(abstract="v1"))
_, status = s.upsert_status(_item(abstract="v2"))
assert status == "updated"
assert s.get(key).abstract == "v2"
def test_new_tag_is_updated_and_merged():
s = _store()
key, _ = s.upsert_status(_item(tags=["a:1"]))
_, status = s.upsert_status(_item(tags=["c:3"]))
assert status == "updated"
assert set(s.get(key).tags) >= {"a:1", "c:3"}
def test_subset_of_existing_tags_is_unchanged():
"""Upsert never removes tags (#624); a subset therefore changes nothing."""
s = _store()
key, _ = s.upsert_status(_item(tags=["a:1", "b:2"]))
_, status = s.upsert_status(_item(tags=["a:1"]))
assert status == "unchanged"
def test_upsert_keeps_returning_key():
s = _store()
key = s.upsert(_item())
assert s.upsert(_item()) == key
def test_empty_incoming_abstract_keeps_stored_body():
"""A list-walk row (no body) must not blank an enriched comment."""
s = _store()
key, _ = s.upsert_status(_item(abstract="enriched body"))
_, status = s.upsert_status(_item(abstract=""))
assert status == "unchanged"
assert s.get(key).abstract == "enriched body"
def test_empty_incoming_title_keeps_stored_title():
s = _store()
it = _item()
it.title = "Org: CMS-2026-2377-1"
key, _ = s.upsert_status(it)
it2 = _item()
it2.title = ""
_, status = s.upsert_status(it2)
assert status == "unchanged"
assert s.get(key).title == "Org: CMS-2026-2377-1"
def test_non_empty_incoming_still_updates():
s = _store()
key, _ = s.upsert_status(_item(abstract="v1"))
_, status = s.upsert_status(_item(abstract="v2"))
assert status == "updated" and s.get(key).abstract == "v2"
# ── extra_json merges per key (refs #680) ────────────────────────────
def _rule(**kw):
from bib.item import Rule
it = Rule(title="T", url="https://example.com/rule-1")
it.cms_id = kw.get("cms_id", "")
it.effective_date = kw.get("effective_date", "")
it.fr_page = kw.get("fr_page", "")
return it
def test_empty_incoming_extra_json_key_keeps_stored_key():
"""A list-walk Rule with a blank effective_date must not blank an
already-stored one — extra_json is merged per key, not replaced
wholesale."""
s = _store()
key, _ = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date="2026-01-01"))
_, status = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date=""))
assert status == "unchanged"
stored = s.get(key)
assert stored.effective_date == "2026-01-01"
assert stored.cms_id == "CMS-1848-P"
def test_non_empty_incoming_extra_json_key_updates_that_key_only():
s = _store()
key, _ = s.upsert_status(_rule(cms_id="CMS-1848-P", effective_date="2026-01-01"))
_, status = s.upsert_status(
_rule(cms_id="CMS-1848-P", effective_date="", fr_page="12345")
)
assert status == "updated"
stored = s.get(key)
# New key applied...
assert stored.fr_page == "12345"
# ...but the blank incoming effective_date did not blank the stored one.
assert stored.effective_date == "2026-01-01"
assert stored.cms_id == "CMS-1848-P"