feat(llm): stack llm index --key also limits the corpus collection

F2's live check needs to index just the 6 CPT bib items, not the whole
corpus — --key was wired only into the rules collection
(iter_rule_refs); the corpus branch (_refs_for -> iter_corpus_refs)
ignored it entirely, so "--collection corpus --key ..." silently
indexed everything. iter_corpus_refs takes an optional keys tuple
(same Python-side post-filter iter_rule_refs already uses) and
_refs_for now passes --key through on the corpus path too.
This commit is contained in:
kert
2026-09-09 21:28:04 -04:00
parent 5628b50548
commit 272d653ef9
4 changed files with 76 additions and 5 deletions

View File

@@ -26,7 +26,7 @@ def _refs_for(
zotero = ZoteroPdfIndex.lazy(
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
)
return iter_corpus_refs(store, zotero=zotero)
return iter_corpus_refs(store, keys=keys, zotero=zotero)
@app.command()
@@ -35,7 +35,9 @@ def index(
"comments", help="Which collection: comments | rules | corpus | all."
),
docket: str = typer.Option("", help="Limit comments to one docket id."),
key: list[str] = typer.Option([], "--key", help="Limit rules to these item keys."),
key: list[str] = typer.Option(
[], "--key", help="Limit rules/corpus to these item keys."
),
force: bool = typer.Option(False, help="Re-embed even when unchanged."),
limit: int = typer.Option(0, help="Stop after N docs per collection (0 = all)."),
) -> None:

View File

@@ -493,7 +493,11 @@ def _attachment_sections(
def iter_corpus_refs(
store: Store, *, tag: str = "", zotero: "ZoteroPdfIndex | None" = None
store: Store,
*,
tag: str = "",
keys: tuple[str, ...] = (),
zotero: "ZoteroPdfIndex | None" = None,
) -> Iterator[DocRef]:
"""One DocRef per non-comment, non-skipped item. Fingerprint =
updated_at + the bib attachment file stats; the Zotero fallback is
@@ -502,7 +506,12 @@ def iter_corpus_refs(
Items tagged ``llm:skip`` are excluded — a generic opt-out for
material that should not be embedded (e.g. a scratch export, or a
duplicate scan of something already in the corpus). It's never
applied automatically; a caller opts an item in by tagging it."""
applied automatically; a caller opts an item in by tagging it.
*keys*, when given, limits the yield to those item keys (same
Python-side post-filter ``iter_rule_refs`` uses for its own
``keys`` — the corpus scan is cheap enough that this doesn't need
to be pushed into the SQL)."""
sql = (
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at FROM items i "
"WHERE i.id NOT IN (SELECT item_id FROM item_tags WHERE tag_id IN "
@@ -517,6 +526,8 @@ def iter_corpus_refs(
)
for row in store._con().execute(sql, (tag,) if tag else ()).fetchall():
key = row["key"]
if keys and key not in keys:
continue
fp = row["updated_at"] + "|" + fingerprint_files(_attachment_paths(store, key))
def _load(key=key) -> Doc | None:

View File

@@ -96,11 +96,58 @@ class TestIndexCorpus:
result = runner.invoke(app, ["index", "--collection", "corpus"])
assert result.exit_code == 0, result.output
mock_iter.assert_called_once_with(store, zotero=zot)
mock_iter.assert_called_once_with(store, keys=(), zotero=zot)
kwargs = mock_index_refs.call_args.kwargs
assert kwargs["collection"] == "corpus"
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
@patch("llm.index.docket_complete", return_value={})
@patch("llm.index._engine")
@patch("llm.source.ZoteroPdfIndex.lazy")
@patch("llm.source.iter_corpus_refs")
@patch("llm.pool.HostPool.from_config")
@patch("llm.index.index_refs")
@patch("conf.connect.bib")
@patch("llm.config.load")
def test_key_option_limits_corpus_to_those_items(
self,
mock_load,
mock_bib,
mock_index_refs,
mock_from_config,
mock_iter,
mock_lazy,
_engine,
_complete,
):
mock_load.return_value = MagicMock()
store = MagicMock()
store.sealed_dockets.return_value = {}
mock_bib.return_value = store
mock_iter.return_value = iter(["doc1"])
mock_from_config.return_value = MagicMock()
mock_index_refs.return_value = _STATS
zot = MagicMock()
mock_lazy.return_value = zot
result = runner.invoke(
app,
[
"index",
"--collection",
"corpus",
"--key",
"GQGTPGYV",
"--key",
"T6H4ZPEQ",
],
)
assert result.exit_code == 0, result.output
mock_iter.assert_called_once_with(
store, keys=("GQGTPGYV", "T6H4ZPEQ"), zotero=zot
)
class TestIndexAll:
@patch("llm.index.docket_complete", return_value={})

View File

@@ -188,6 +188,17 @@ class TestCorpusRefs:
store.create(Item(item_type="report", title="U", abstract="Body."))
assert [r.key for r in iter_corpus_refs(store, tag="project:pfs")] == [tagged]
def test_keys_filter_selects_only_named_items(self, store):
wanted = store.create(Item(item_type="source", title="CPT 2024", abstract="."))
store.create(Item(item_type="source", title="Other book", abstract="."))
refs = list(iter_corpus_refs(store, keys=(wanted,)))
assert [r.key for r in refs] == [wanted]
def test_keys_empty_tuple_yields_every_item(self, store):
a = store.create(Item(item_type="source", title="A", abstract="."))
b = store.create(Item(item_type="source", title="B", abstract="."))
assert {r.key for r in iter_corpus_refs(store, keys=())} == {a, b}
class TestLazyZotero:
def test_no_copy_until_first_lookup(self, tmp_path):