feat(llm): stack llm index --key also limits the corpus collection
F2's live check needs to index just the 6 CPT bib items, not the whole corpus — --key was wired only into the rules collection (iter_rule_refs); the corpus branch (_refs_for -> iter_corpus_refs) ignored it entirely, so "--collection corpus --key ..." silently indexed everything. iter_corpus_refs takes an optional keys tuple (same Python-side post-filter iter_rule_refs already uses) and _refs_for now passes --key through on the corpus path too.
This commit is contained in:
@@ -26,7 +26,7 @@ def _refs_for(
|
|||||||
zotero = ZoteroPdfIndex.lazy(
|
zotero = ZoteroPdfIndex.lazy(
|
||||||
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
|
path("db.zotero"), path("storage.zotero"), ROOT / ".state" / "llm"
|
||||||
)
|
)
|
||||||
return iter_corpus_refs(store, zotero=zotero)
|
return iter_corpus_refs(store, keys=keys, zotero=zotero)
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
@app.command()
|
||||||
@@ -35,7 +35,9 @@ def index(
|
|||||||
"comments", help="Which collection: comments | rules | corpus | all."
|
"comments", help="Which collection: comments | rules | corpus | all."
|
||||||
),
|
),
|
||||||
docket: str = typer.Option("", help="Limit comments to one docket id."),
|
docket: str = typer.Option("", help="Limit comments to one docket id."),
|
||||||
key: list[str] = typer.Option([], "--key", help="Limit rules to these item keys."),
|
key: list[str] = typer.Option(
|
||||||
|
[], "--key", help="Limit rules/corpus to these item keys."
|
||||||
|
),
|
||||||
force: bool = typer.Option(False, help="Re-embed even when unchanged."),
|
force: bool = typer.Option(False, help="Re-embed even when unchanged."),
|
||||||
limit: int = typer.Option(0, help="Stop after N docs per collection (0 = all)."),
|
limit: int = typer.Option(0, help="Stop after N docs per collection (0 = all)."),
|
||||||
) -> None:
|
) -> None:
|
||||||
|
|||||||
@@ -493,7 +493,11 @@ def _attachment_sections(
|
|||||||
|
|
||||||
|
|
||||||
def iter_corpus_refs(
|
def iter_corpus_refs(
|
||||||
store: Store, *, tag: str = "", zotero: "ZoteroPdfIndex | None" = None
|
store: Store,
|
||||||
|
*,
|
||||||
|
tag: str = "",
|
||||||
|
keys: tuple[str, ...] = (),
|
||||||
|
zotero: "ZoteroPdfIndex | None" = None,
|
||||||
) -> Iterator[DocRef]:
|
) -> Iterator[DocRef]:
|
||||||
"""One DocRef per non-comment, non-skipped item. Fingerprint =
|
"""One DocRef per non-comment, non-skipped item. Fingerprint =
|
||||||
updated_at + the bib attachment file stats; the Zotero fallback is
|
updated_at + the bib attachment file stats; the Zotero fallback is
|
||||||
@@ -502,7 +506,12 @@ def iter_corpus_refs(
|
|||||||
Items tagged ``llm:skip`` are excluded — a generic opt-out for
|
Items tagged ``llm:skip`` are excluded — a generic opt-out for
|
||||||
material that should not be embedded (e.g. a scratch export, or a
|
material that should not be embedded (e.g. a scratch export, or a
|
||||||
duplicate scan of something already in the corpus). It's never
|
duplicate scan of something already in the corpus). It's never
|
||||||
applied automatically; a caller opts an item in by tagging it."""
|
applied automatically; a caller opts an item in by tagging it.
|
||||||
|
|
||||||
|
*keys*, when given, limits the yield to those item keys (same
|
||||||
|
Python-side post-filter ``iter_rule_refs`` uses for its own
|
||||||
|
``keys`` — the corpus scan is cheap enough that this doesn't need
|
||||||
|
to be pushed into the SQL)."""
|
||||||
sql = (
|
sql = (
|
||||||
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at FROM items i "
|
"SELECT i.key, COALESCE(i.updated_at,'') AS updated_at FROM items i "
|
||||||
"WHERE i.id NOT IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
"WHERE i.id NOT IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
||||||
@@ -517,6 +526,8 @@ def iter_corpus_refs(
|
|||||||
)
|
)
|
||||||
for row in store._con().execute(sql, (tag,) if tag else ()).fetchall():
|
for row in store._con().execute(sql, (tag,) if tag else ()).fetchall():
|
||||||
key = row["key"]
|
key = row["key"]
|
||||||
|
if keys and key not in keys:
|
||||||
|
continue
|
||||||
fp = row["updated_at"] + "|" + fingerprint_files(_attachment_paths(store, key))
|
fp = row["updated_at"] + "|" + fingerprint_files(_attachment_paths(store, key))
|
||||||
|
|
||||||
def _load(key=key) -> Doc | None:
|
def _load(key=key) -> Doc | None:
|
||||||
|
|||||||
@@ -96,11 +96,58 @@ class TestIndexCorpus:
|
|||||||
result = runner.invoke(app, ["index", "--collection", "corpus"])
|
result = runner.invoke(app, ["index", "--collection", "corpus"])
|
||||||
|
|
||||||
assert result.exit_code == 0, result.output
|
assert result.exit_code == 0, result.output
|
||||||
mock_iter.assert_called_once_with(store, zotero=zot)
|
mock_iter.assert_called_once_with(store, keys=(), zotero=zot)
|
||||||
kwargs = mock_index_refs.call_args.kwargs
|
kwargs = mock_index_refs.call_args.kwargs
|
||||||
assert kwargs["collection"] == "corpus"
|
assert kwargs["collection"] == "corpus"
|
||||||
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
|
assert "corpus: indexed=3 skipped=1 chunks=7" in result.output
|
||||||
|
|
||||||
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
@patch("llm.index._engine")
|
||||||
|
@patch("llm.source.ZoteroPdfIndex.lazy")
|
||||||
|
@patch("llm.source.iter_corpus_refs")
|
||||||
|
@patch("llm.pool.HostPool.from_config")
|
||||||
|
@patch("llm.index.index_refs")
|
||||||
|
@patch("conf.connect.bib")
|
||||||
|
@patch("llm.config.load")
|
||||||
|
def test_key_option_limits_corpus_to_those_items(
|
||||||
|
self,
|
||||||
|
mock_load,
|
||||||
|
mock_bib,
|
||||||
|
mock_index_refs,
|
||||||
|
mock_from_config,
|
||||||
|
mock_iter,
|
||||||
|
mock_lazy,
|
||||||
|
_engine,
|
||||||
|
_complete,
|
||||||
|
):
|
||||||
|
mock_load.return_value = MagicMock()
|
||||||
|
store = MagicMock()
|
||||||
|
store.sealed_dockets.return_value = {}
|
||||||
|
mock_bib.return_value = store
|
||||||
|
mock_iter.return_value = iter(["doc1"])
|
||||||
|
mock_from_config.return_value = MagicMock()
|
||||||
|
mock_index_refs.return_value = _STATS
|
||||||
|
zot = MagicMock()
|
||||||
|
mock_lazy.return_value = zot
|
||||||
|
|
||||||
|
result = runner.invoke(
|
||||||
|
app,
|
||||||
|
[
|
||||||
|
"index",
|
||||||
|
"--collection",
|
||||||
|
"corpus",
|
||||||
|
"--key",
|
||||||
|
"GQGTPGYV",
|
||||||
|
"--key",
|
||||||
|
"T6H4ZPEQ",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
mock_iter.assert_called_once_with(
|
||||||
|
store, keys=("GQGTPGYV", "T6H4ZPEQ"), zotero=zot
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class TestIndexAll:
|
class TestIndexAll:
|
||||||
@patch("llm.index.docket_complete", return_value={})
|
@patch("llm.index.docket_complete", return_value={})
|
||||||
|
|||||||
@@ -188,6 +188,17 @@ class TestCorpusRefs:
|
|||||||
store.create(Item(item_type="report", title="U", abstract="Body."))
|
store.create(Item(item_type="report", title="U", abstract="Body."))
|
||||||
assert [r.key for r in iter_corpus_refs(store, tag="project:pfs")] == [tagged]
|
assert [r.key for r in iter_corpus_refs(store, tag="project:pfs")] == [tagged]
|
||||||
|
|
||||||
|
def test_keys_filter_selects_only_named_items(self, store):
|
||||||
|
wanted = store.create(Item(item_type="source", title="CPT 2024", abstract="."))
|
||||||
|
store.create(Item(item_type="source", title="Other book", abstract="."))
|
||||||
|
refs = list(iter_corpus_refs(store, keys=(wanted,)))
|
||||||
|
assert [r.key for r in refs] == [wanted]
|
||||||
|
|
||||||
|
def test_keys_empty_tuple_yields_every_item(self, store):
|
||||||
|
a = store.create(Item(item_type="source", title="A", abstract="."))
|
||||||
|
b = store.create(Item(item_type="source", title="B", abstract="."))
|
||||||
|
assert {r.key for r in iter_corpus_refs(store, keys=())} == {a, b}
|
||||||
|
|
||||||
|
|
||||||
class TestLazyZotero:
|
class TestLazyZotero:
|
||||||
def test_no_copy_until_first_lookup(self, tmp_path):
|
def test_no_copy_until_first_lookup(self, tmp_path):
|
||||||
|
|||||||
Reference in New Issue
Block a user