From 7693e97d36bfb4dc70f38184875ddbe80201a5d6 Mon Sep 17 00:00:00 2001 From: kert Date: Tue, 22 Sep 2026 16:48:51 -0400 Subject: [PATCH 1/2] =?UTF-8?q?feat(llm):=20closed=20theme=20vocabulary=20?= =?UTF-8?q?for=20comment=20tagging=20=E2=80=94=20themes.yaml=20v1,=20llm.v?= =?UTF-8?q?ocab=20loader,=20stack=20llm=20vocab,=20P35=20design=20spec=20(?= =?UTF-8?q?refs=20#574)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docs/superpowers/specs/2026-09-22-llm-tagging-design.md lays out P35: a versioned hand-curated vocabulary, embedding shortlist + per-theme yes/no judgement with an evidence chunk on the largest live host, resumable docket-scoped runs with state in bib extra_json, write-back that replaces only llm: tags, and a golden-set gate before fan-out. Slice 1: src/llm/vocab/themes.yaml — 53 themes (conversion factor, practice expense, telehealth, care management, behavioral health, Part B drugs, MSSP, QPP, …), each with a label, one-sentence definition, letter-language synonyms and the FR section stems it was seeded from (dev/scripts/llm_vocab_candidates.py mines heading paragraphs from the CY2018+ rules' fr_anchors). llm.vocab.load() validates unique kebab-case slugs, non-empty definitions, integer version ≥ 1 and that retired slugs are not still active; Theme.card is the text embedded per theme; split_tags() separates llm: tags from hand tags for the write-back. stack llm vocab [--path] [--slug]. --- dev/scripts/llm_vocab_candidates.py | 81 +++++ docs/docs/cli/api-serve.md | 2 +- docs/docs/cli/api.md | 2 +- docs/docs/cli/db-comment.md | 2 +- docs/docs/cli/db-inspect.md | 2 +- docs/docs/cli/db.md | 2 +- docs/docs/cli/docs-build.md | 2 +- docs/docs/cli/docs-generate.md | 2 +- docs/docs/cli/docs-serve.md | 2 +- docs/docs/cli/docs.md | 2 +- docs/docs/cli/llm-vocab.md | 21 ++ docs/docs/cli/llm.md | 4 + docs/docs/cli/mail-attach-smarthost.md | 2 +- docs/docs/cli/mail-dkim-export.md | 2 +- docs/docs/cli/mail-dns.md | 2 +- docs/docs/cli/mail-down.md | 2 +- docs/docs/cli/mail-provision.md | 2 +- docs/docs/cli/mail-rotate-creds.md | 2 +- docs/docs/cli/mail-seed-mailboxes.md | 2 +- docs/docs/cli/mail-status.md | 2 +- docs/docs/cli/mail-up.md | 2 +- docs/docs/cli/mail-wire-git.md | 2 +- docs/docs/cli/mail.md | 2 +- docs/docs/cli/perf-show.md | 2 +- docs/docs/cli/perf.md | 2 +- docs/docs/cli/pfs-cpt-ingest.md | 2 +- docs/docs/cli/pfs-elements.md | 2 +- docs/docs/cli/pfs-exposure.md | 2 +- docs/docs/cli/pfs-families.md | 2 +- docs/docs/cli/pfs-guidance.md | 2 +- docs/docs/cli/pfs-lineage.md | 2 +- docs/docs/cli/pfs-reaction.md | 2 +- docs/docs/cli/pfs-review.md | 2 +- docs/docs/cli/pfs-utilization.md | 2 +- docs/docs/cli/pfs.md | 2 +- docs/docs/cli/prisma-eligible.md | 2 +- docs/docs/cli/prisma-export.md | 2 +- docs/docs/cli/prisma-extract.md | 2 +- docs/docs/cli/prisma-fetch.md | 2 +- docs/docs/cli/prisma-flow.md | 2 +- docs/docs/cli/prisma-init.md | 2 +- docs/docs/cli/prisma-ping-llm.md | 2 +- docs/docs/cli/prisma-run.md | 2 +- docs/docs/cli/prisma-screen.md | 2 +- docs/docs/cli/prisma-vpn.md | 2 +- docs/docs/cli/prisma.md | 2 +- docs/docs/cli/rec-list.md | 2 +- docs/docs/cli/rec-opps.md | 2 +- docs/docs/cli/rec-pfs.md | 2 +- docs/docs/cli/rec.md | 2 +- docs/docs/cli/zot-dump-schema.md | 2 +- docs/docs/cli/zot-fix-dates.md | 2 +- docs/docs/cli/zot-fix-fields.md | 2 +- docs/docs/cli/zot-fix-keys.md | 2 +- docs/docs/cli/zot-verify-parity.md | 2 +- docs/docs/cli/zot.md | 2 +- .../specs/2026-09-22-llm-tagging-design.md | 46 +++ src/cli/llm.py | 33 +++ src/llm/vocab/__init__.py | 151 ++++++++++ src/llm/vocab/themes.yaml | 276 ++++++++++++++++++ tests/llm/test_vocab.py | 114 ++++++++ 61 files changed, 779 insertions(+), 53 deletions(-) create mode 100644 dev/scripts/llm_vocab_candidates.py create mode 100644 docs/docs/cli/llm-vocab.md create mode 100644 docs/superpowers/specs/2026-09-22-llm-tagging-design.md create mode 100644 src/llm/vocab/__init__.py create mode 100644 src/llm/vocab/themes.yaml create mode 100644 tests/llm/test_vocab.py diff --git a/dev/scripts/llm_vocab_candidates.py b/dev/scripts/llm_vocab_candidates.py new file mode 100644 index 0000000..0174fc3 --- /dev/null +++ b/dev/scripts/llm_vocab_candidates.py @@ -0,0 +1,81 @@ +"""Seed a theme-vocabulary revision from Federal Register section headings (#574). + +Walks ``fr_anchors`` for heading-like paragraphs ("B. Determination of +Practice Expense…") in every PFS rule whose rule year is at or after +``--since`` (default 2018 — the regulations.gov comment dockets start in +2017), stems the titles, and prints the stems ordered by how many rule +years carry them, with an example title each. The output is raw material +for a human editing ``src/llm/vocab/themes.yaml`` — it is not the vocab. + +Usage: + uv run python dev/scripts/llm_vocab_candidates.py [--since 2018] [--top 200] +""" + +from __future__ import annotations + +import argparse +import re +from collections import defaultdict + +_HEADING = re.compile(r"^(?:[IVX]{1,4}|[A-Z]|\d{1,2}|[a-z])\. +([A-Z][^.]{6,120})$") +_STOP = re.compile( + r"\b(cy|20\d\d|proposed|final|rule|for|the|of|and|to|in|a|an|under|pfs|physician fee schedule)\b" +) +_SKIP = ("authority citation", "on page", "column", "paragraph") + + +def stem(title: str) -> str: + norm = re.sub(r"[^a-z0-9 ]", " ", title.lower()) + norm = _STOP.sub(" ", norm) + return re.sub(r"\s+", " ", norm).strip() + + +def candidates(store, *, since: int) -> list[tuple[int, str, str]]: + """``(n_rule_years, stem, example title)`` sorted by n desc.""" + from pfs.descriptors import rule_year_of + + con = store._con() # noqa: SLF001 + years: dict[str, int] = {} + for key, title, published in con.execute( + "SELECT key, title, date_published FROM items WHERE item_type = 'rule'" + ).fetchall(): + # rule_year_of reads the payment year from the title, else falls + # back to publication year + 1 (PFS rules publish July–December). + years[key] = int(rule_year_of(title or "", published or "")) + recent = {k for k, y in years.items() if y and y >= since} + by: dict[str, set[int]] = defaultdict(set) + example: dict[str, str] = {} + for key, text in con.execute( + "SELECT item_key, text FROM fr_anchors WHERE length(text) < 140" + ).fetchall(): + if key not in recent: + continue + m = _HEADING.match((text or "").strip()) + if not m: + continue + title = m.group(1).strip() + s = stem(title) + if len(s) < 5 or any(s.startswith(x) or x in s for x in _SKIP): + continue + by[s].add(years[key]) + example.setdefault(s, title) + return sorted( + ((len(v), k, example[k]) for k, v in by.items()), key=lambda r: (-r[0], r[1]) + ) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("--since", type=int, default=2018) + ap.add_argument("--top", type=int, default=200) + args = ap.parse_args() + from conf import connect + + store = connect.bib() + for n, s, ex in candidates(store, since=args.since)[: args.top]: + print(f"{n:3d} {ex}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/docs/cli/api-serve.md b/docs/docs/cli/api-serve.md index 3a6d624..78fccfa 100644 --- a/docs/docs/cli/api-serve.md +++ b/docs/docs/cli/api-serve.md @@ -1,6 +1,6 @@ --- title: stack api serve -sidebar_position: 61 +sidebar_position: 62 --- # `stack api serve` diff --git a/docs/docs/cli/api.md b/docs/docs/cli/api.md index 680af44..acf8249 100644 --- a/docs/docs/cli/api.md +++ b/docs/docs/cli/api.md @@ -1,6 +1,6 @@ --- title: stack api -sidebar_position: 60 +sidebar_position: 61 --- # `stack api` diff --git a/docs/docs/cli/db-comment.md b/docs/docs/cli/db-comment.md index 3e82e9b..c0f7f9c 100644 --- a/docs/docs/cli/db-comment.md +++ b/docs/docs/cli/db-comment.md @@ -1,6 +1,6 @@ --- title: stack db comment -sidebar_position: 54 +sidebar_position: 55 --- # `stack db comment` diff --git a/docs/docs/cli/db-inspect.md b/docs/docs/cli/db-inspect.md index 653c2cc..29ae60d 100644 --- a/docs/docs/cli/db-inspect.md +++ b/docs/docs/cli/db-inspect.md @@ -1,6 +1,6 @@ --- title: stack db inspect -sidebar_position: 55 +sidebar_position: 56 --- # `stack db inspect` diff --git a/docs/docs/cli/db.md b/docs/docs/cli/db.md index 2e50b2f..8c73d5b 100644 --- a/docs/docs/cli/db.md +++ b/docs/docs/cli/db.md @@ -1,6 +1,6 @@ --- title: stack db -sidebar_position: 53 +sidebar_position: 54 --- # `stack db` diff --git a/docs/docs/cli/docs-build.md b/docs/docs/cli/docs-build.md index 56b53ab..e243395 100644 --- a/docs/docs/cli/docs-build.md +++ b/docs/docs/cli/docs-build.md @@ -1,6 +1,6 @@ --- title: stack docs build -sidebar_position: 57 +sidebar_position: 58 --- # `stack docs build` diff --git a/docs/docs/cli/docs-generate.md b/docs/docs/cli/docs-generate.md index 9fcb075..6b0b71f 100644 --- a/docs/docs/cli/docs-generate.md +++ b/docs/docs/cli/docs-generate.md @@ -1,6 +1,6 @@ --- title: stack docs generate -sidebar_position: 59 +sidebar_position: 60 --- # `stack docs generate` diff --git a/docs/docs/cli/docs-serve.md b/docs/docs/cli/docs-serve.md index d908da8..b0649b3 100644 --- a/docs/docs/cli/docs-serve.md +++ b/docs/docs/cli/docs-serve.md @@ -1,6 +1,6 @@ --- title: stack docs serve -sidebar_position: 58 +sidebar_position: 59 --- # `stack docs serve` diff --git a/docs/docs/cli/docs.md b/docs/docs/cli/docs.md index 6c6fca4..2067ef6 100644 --- a/docs/docs/cli/docs.md +++ b/docs/docs/cli/docs.md @@ -1,6 +1,6 @@ --- title: stack docs -sidebar_position: 56 +sidebar_position: 57 --- # `stack docs` diff --git a/docs/docs/cli/llm-vocab.md b/docs/docs/cli/llm-vocab.md new file mode 100644 index 0000000..a337cf7 --- /dev/null +++ b/docs/docs/cli/llm-vocab.md @@ -0,0 +1,21 @@ +--- +title: stack llm vocab +sidebar_position: 53 +--- + +# `stack llm vocab` + +``` +Usage: stack llm vocab [OPTIONS] + + The closed theme vocabulary for comment tagging (#574): validate it and list + its slugs, or show one theme's definition, synonyms and the FR section stems + it was seeded from. + +╭─ Options ────────────────────────────────────────────────────────────────────╮ +│ --path TEXT A vocabulary file to validate instead of the packaged │ +│ one. │ +│ --slug TEXT Show one theme in full. │ +│ --help Show this message and exit. │ +╰──────────────────────────────────────────────────────────────────────────────╯ +``` diff --git a/docs/docs/cli/llm.md b/docs/docs/cli/llm.md index 8eb5241..4128367 100644 --- a/docs/docs/cli/llm.md +++ b/docs/docs/cli/llm.md @@ -19,5 +19,9 @@ Usage: stack llm [OPTIONS] COMMAND [ARGS]... │ hosts Show the Ollama fleet: declared VRAM, liveness, models, and which │ │ host + model would answer a chat right now. │ │ serve Serve the SSO-guarded chat UI (llm.api:app). │ +│ vocab The closed theme vocabulary for comment tagging (#574): validate it │ +│ and list its slugs, or show one theme's definition, synonyms and │ +│ the FR │ +│ section stems it was seeded from. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/docs/docs/cli/mail-attach-smarthost.md b/docs/docs/cli/mail-attach-smarthost.md index e5ff99f..f2b38f0 100644 --- a/docs/docs/cli/mail-attach-smarthost.md +++ b/docs/docs/cli/mail-attach-smarthost.md @@ -1,6 +1,6 @@ --- title: stack mail attach-smarthost -sidebar_position: 102 +sidebar_position: 103 --- # `stack mail attach-smarthost` diff --git a/docs/docs/cli/mail-dkim-export.md b/docs/docs/cli/mail-dkim-export.md index 7063e19..a97e00b 100644 --- a/docs/docs/cli/mail-dkim-export.md +++ b/docs/docs/cli/mail-dkim-export.md @@ -1,6 +1,6 @@ --- title: stack mail dkim-export -sidebar_position: 101 +sidebar_position: 102 --- # `stack mail dkim-export` diff --git a/docs/docs/cli/mail-dns.md b/docs/docs/cli/mail-dns.md index 7164ae1..595e177 100644 --- a/docs/docs/cli/mail-dns.md +++ b/docs/docs/cli/mail-dns.md @@ -1,6 +1,6 @@ --- title: stack mail dns -sidebar_position: 100 +sidebar_position: 101 --- # `stack mail dns` diff --git a/docs/docs/cli/mail-down.md b/docs/docs/cli/mail-down.md index 486143b..e23a231 100644 --- a/docs/docs/cli/mail-down.md +++ b/docs/docs/cli/mail-down.md @@ -1,6 +1,6 @@ --- title: stack mail down -sidebar_position: 98 +sidebar_position: 99 --- # `stack mail down` diff --git a/docs/docs/cli/mail-provision.md b/docs/docs/cli/mail-provision.md index 85810f8..cceec5d 100644 --- a/docs/docs/cli/mail-provision.md +++ b/docs/docs/cli/mail-provision.md @@ -1,6 +1,6 @@ --- title: stack mail provision -sidebar_position: 96 +sidebar_position: 97 --- # `stack mail provision` diff --git a/docs/docs/cli/mail-rotate-creds.md b/docs/docs/cli/mail-rotate-creds.md index 08c8603..0d98cd4 100644 --- a/docs/docs/cli/mail-rotate-creds.md +++ b/docs/docs/cli/mail-rotate-creds.md @@ -1,6 +1,6 @@ --- title: stack mail rotate-creds -sidebar_position: 103 +sidebar_position: 104 --- # `stack mail rotate-creds` diff --git a/docs/docs/cli/mail-seed-mailboxes.md b/docs/docs/cli/mail-seed-mailboxes.md index 2526f30..ffe2741 100644 --- a/docs/docs/cli/mail-seed-mailboxes.md +++ b/docs/docs/cli/mail-seed-mailboxes.md @@ -1,6 +1,6 @@ --- title: stack mail seed-mailboxes -sidebar_position: 104 +sidebar_position: 105 --- # `stack mail seed-mailboxes` diff --git a/docs/docs/cli/mail-status.md b/docs/docs/cli/mail-status.md index 0f0b988..b0a565a 100644 --- a/docs/docs/cli/mail-status.md +++ b/docs/docs/cli/mail-status.md @@ -1,6 +1,6 @@ --- title: stack mail status -sidebar_position: 99 +sidebar_position: 100 --- # `stack mail status` diff --git a/docs/docs/cli/mail-up.md b/docs/docs/cli/mail-up.md index 8fa0e70..1bda771 100644 --- a/docs/docs/cli/mail-up.md +++ b/docs/docs/cli/mail-up.md @@ -1,6 +1,6 @@ --- title: stack mail up -sidebar_position: 97 +sidebar_position: 98 --- # `stack mail up` diff --git a/docs/docs/cli/mail-wire-git.md b/docs/docs/cli/mail-wire-git.md index 60cdfde..21a087e 100644 --- a/docs/docs/cli/mail-wire-git.md +++ b/docs/docs/cli/mail-wire-git.md @@ -1,6 +1,6 @@ --- title: stack mail wire-git -sidebar_position: 105 +sidebar_position: 106 --- # `stack mail wire-git` diff --git a/docs/docs/cli/mail.md b/docs/docs/cli/mail.md index 699bb87..99f002e 100644 --- a/docs/docs/cli/mail.md +++ b/docs/docs/cli/mail.md @@ -1,6 +1,6 @@ --- title: stack mail -sidebar_position: 95 +sidebar_position: 96 --- # `stack mail` diff --git a/docs/docs/cli/perf-show.md b/docs/docs/cli/perf-show.md index a620821..7f0bec3 100644 --- a/docs/docs/cli/perf-show.md +++ b/docs/docs/cli/perf-show.md @@ -1,6 +1,6 @@ --- title: stack perf show -sidebar_position: 63 +sidebar_position: 64 --- # `stack perf show` diff --git a/docs/docs/cli/perf.md b/docs/docs/cli/perf.md index 59dfef9..a07728d 100644 --- a/docs/docs/cli/perf.md +++ b/docs/docs/cli/perf.md @@ -1,6 +1,6 @@ --- title: stack perf -sidebar_position: 62 +sidebar_position: 63 --- # `stack perf` diff --git a/docs/docs/cli/pfs-cpt-ingest.md b/docs/docs/cli/pfs-cpt-ingest.md index 921c977..b4d6a02 100644 --- a/docs/docs/cli/pfs-cpt-ingest.md +++ b/docs/docs/cli/pfs-cpt-ingest.md @@ -1,6 +1,6 @@ --- title: stack pfs cpt-ingest -sidebar_position: 77 +sidebar_position: 78 --- # `stack pfs cpt-ingest` diff --git a/docs/docs/cli/pfs-elements.md b/docs/docs/cli/pfs-elements.md index d2ad7ff..cffc9c9 100644 --- a/docs/docs/cli/pfs-elements.md +++ b/docs/docs/cli/pfs-elements.md @@ -1,6 +1,6 @@ --- title: stack pfs elements -sidebar_position: 69 +sidebar_position: 70 --- # `stack pfs elements` diff --git a/docs/docs/cli/pfs-exposure.md b/docs/docs/cli/pfs-exposure.md index 046ff2b..3b443d3 100644 --- a/docs/docs/cli/pfs-exposure.md +++ b/docs/docs/cli/pfs-exposure.md @@ -1,6 +1,6 @@ --- title: stack pfs exposure -sidebar_position: 74 +sidebar_position: 75 --- # `stack pfs exposure` diff --git a/docs/docs/cli/pfs-families.md b/docs/docs/cli/pfs-families.md index 8073374..87c34cd 100644 --- a/docs/docs/cli/pfs-families.md +++ b/docs/docs/cli/pfs-families.md @@ -1,6 +1,6 @@ --- title: stack pfs families -sidebar_position: 71 +sidebar_position: 72 --- # `stack pfs families` diff --git a/docs/docs/cli/pfs-guidance.md b/docs/docs/cli/pfs-guidance.md index 669098c..e5837e5 100644 --- a/docs/docs/cli/pfs-guidance.md +++ b/docs/docs/cli/pfs-guidance.md @@ -1,6 +1,6 @@ --- title: stack pfs guidance -sidebar_position: 72 +sidebar_position: 73 --- # `stack pfs guidance` diff --git a/docs/docs/cli/pfs-lineage.md b/docs/docs/cli/pfs-lineage.md index b743fe8..a12df77 100644 --- a/docs/docs/cli/pfs-lineage.md +++ b/docs/docs/cli/pfs-lineage.md @@ -1,6 +1,6 @@ --- title: stack pfs lineage -sidebar_position: 70 +sidebar_position: 71 --- # `stack pfs lineage` diff --git a/docs/docs/cli/pfs-reaction.md b/docs/docs/cli/pfs-reaction.md index fe2f247..ebd436a 100644 --- a/docs/docs/cli/pfs-reaction.md +++ b/docs/docs/cli/pfs-reaction.md @@ -1,6 +1,6 @@ --- title: stack pfs reaction -sidebar_position: 73 +sidebar_position: 74 --- # `stack pfs reaction` diff --git a/docs/docs/cli/pfs-review.md b/docs/docs/cli/pfs-review.md index 7285581..e8836a8 100644 --- a/docs/docs/cli/pfs-review.md +++ b/docs/docs/cli/pfs-review.md @@ -1,6 +1,6 @@ --- title: stack pfs review -sidebar_position: 76 +sidebar_position: 77 --- # `stack pfs review` diff --git a/docs/docs/cli/pfs-utilization.md b/docs/docs/cli/pfs-utilization.md index 9aacf18..9c2a8d9 100644 --- a/docs/docs/cli/pfs-utilization.md +++ b/docs/docs/cli/pfs-utilization.md @@ -1,6 +1,6 @@ --- title: stack pfs utilization -sidebar_position: 75 +sidebar_position: 76 --- # `stack pfs utilization` diff --git a/docs/docs/cli/pfs.md b/docs/docs/cli/pfs.md index b07448c..7578636 100644 --- a/docs/docs/cli/pfs.md +++ b/docs/docs/cli/pfs.md @@ -1,6 +1,6 @@ --- title: stack pfs -sidebar_position: 68 +sidebar_position: 69 --- # `stack pfs` diff --git a/docs/docs/cli/prisma-eligible.md b/docs/docs/cli/prisma-eligible.md index 12166ce..08610fb 100644 --- a/docs/docs/cli/prisma-eligible.md +++ b/docs/docs/cli/prisma-eligible.md @@ -1,6 +1,6 @@ --- title: stack prisma eligible -sidebar_position: 89 +sidebar_position: 90 --- # `stack prisma eligible` diff --git a/docs/docs/cli/prisma-export.md b/docs/docs/cli/prisma-export.md index 51ef71b..26a5bcc 100644 --- a/docs/docs/cli/prisma-export.md +++ b/docs/docs/cli/prisma-export.md @@ -1,6 +1,6 @@ --- title: stack prisma export -sidebar_position: 86 +sidebar_position: 87 --- # `stack prisma export` diff --git a/docs/docs/cli/prisma-extract.md b/docs/docs/cli/prisma-extract.md index e084280..08cbf09 100644 --- a/docs/docs/cli/prisma-extract.md +++ b/docs/docs/cli/prisma-extract.md @@ -1,6 +1,6 @@ --- title: stack prisma extract -sidebar_position: 90 +sidebar_position: 91 --- # `stack prisma extract` diff --git a/docs/docs/cli/prisma-fetch.md b/docs/docs/cli/prisma-fetch.md index 15e7a37..33bf0eb 100644 --- a/docs/docs/cli/prisma-fetch.md +++ b/docs/docs/cli/prisma-fetch.md @@ -1,6 +1,6 @@ --- title: stack prisma fetch -sidebar_position: 92 +sidebar_position: 93 --- # `stack prisma fetch` diff --git a/docs/docs/cli/prisma-flow.md b/docs/docs/cli/prisma-flow.md index 929cc33..44a0781 100644 --- a/docs/docs/cli/prisma-flow.md +++ b/docs/docs/cli/prisma-flow.md @@ -1,6 +1,6 @@ --- title: stack prisma flow -sidebar_position: 91 +sidebar_position: 92 --- # `stack prisma flow` diff --git a/docs/docs/cli/prisma-init.md b/docs/docs/cli/prisma-init.md index 59e8313..f93b2fb 100644 --- a/docs/docs/cli/prisma-init.md +++ b/docs/docs/cli/prisma-init.md @@ -1,6 +1,6 @@ --- title: stack prisma init -sidebar_position: 85 +sidebar_position: 86 --- # `stack prisma init` diff --git a/docs/docs/cli/prisma-ping-llm.md b/docs/docs/cli/prisma-ping-llm.md index 0dd46c1..9d161f3 100644 --- a/docs/docs/cli/prisma-ping-llm.md +++ b/docs/docs/cli/prisma-ping-llm.md @@ -1,6 +1,6 @@ --- title: stack prisma ping-llm -sidebar_position: 87 +sidebar_position: 88 --- # `stack prisma ping-llm` diff --git a/docs/docs/cli/prisma-run.md b/docs/docs/cli/prisma-run.md index 8da04e3..404bd45 100644 --- a/docs/docs/cli/prisma-run.md +++ b/docs/docs/cli/prisma-run.md @@ -1,6 +1,6 @@ --- title: stack prisma run -sidebar_position: 93 +sidebar_position: 94 --- # `stack prisma run` diff --git a/docs/docs/cli/prisma-screen.md b/docs/docs/cli/prisma-screen.md index 60d0d33..152cc1d 100644 --- a/docs/docs/cli/prisma-screen.md +++ b/docs/docs/cli/prisma-screen.md @@ -1,6 +1,6 @@ --- title: stack prisma screen -sidebar_position: 88 +sidebar_position: 89 --- # `stack prisma screen` diff --git a/docs/docs/cli/prisma-vpn.md b/docs/docs/cli/prisma-vpn.md index b687d45..6b63da7 100644 --- a/docs/docs/cli/prisma-vpn.md +++ b/docs/docs/cli/prisma-vpn.md @@ -1,6 +1,6 @@ --- title: stack prisma vpn -sidebar_position: 94 +sidebar_position: 95 --- # `stack prisma vpn` diff --git a/docs/docs/cli/prisma.md b/docs/docs/cli/prisma.md index 79377b1..42d5a5e 100644 --- a/docs/docs/cli/prisma.md +++ b/docs/docs/cli/prisma.md @@ -1,6 +1,6 @@ --- title: stack prisma -sidebar_position: 84 +sidebar_position: 85 --- # `stack prisma` diff --git a/docs/docs/cli/rec-list.md b/docs/docs/cli/rec-list.md index d701b7c..455d9f4 100644 --- a/docs/docs/cli/rec-list.md +++ b/docs/docs/cli/rec-list.md @@ -1,6 +1,6 @@ --- title: stack rec list -sidebar_position: 65 +sidebar_position: 66 --- # `stack rec list` diff --git a/docs/docs/cli/rec-opps.md b/docs/docs/cli/rec-opps.md index 49951f7..073871c 100644 --- a/docs/docs/cli/rec-opps.md +++ b/docs/docs/cli/rec-opps.md @@ -1,6 +1,6 @@ --- title: stack rec opps -sidebar_position: 67 +sidebar_position: 68 --- # `stack rec opps` diff --git a/docs/docs/cli/rec-pfs.md b/docs/docs/cli/rec-pfs.md index 31ce428..f2d5ec9 100644 --- a/docs/docs/cli/rec-pfs.md +++ b/docs/docs/cli/rec-pfs.md @@ -1,6 +1,6 @@ --- title: stack rec pfs -sidebar_position: 66 +sidebar_position: 67 --- # `stack rec pfs` diff --git a/docs/docs/cli/rec.md b/docs/docs/cli/rec.md index 8ab513b..9efb9db 100644 --- a/docs/docs/cli/rec.md +++ b/docs/docs/cli/rec.md @@ -1,6 +1,6 @@ --- title: stack rec -sidebar_position: 64 +sidebar_position: 65 --- # `stack rec` diff --git a/docs/docs/cli/zot-dump-schema.md b/docs/docs/cli/zot-dump-schema.md index 66934a4..cc95b01 100644 --- a/docs/docs/cli/zot-dump-schema.md +++ b/docs/docs/cli/zot-dump-schema.md @@ -1,6 +1,6 @@ --- title: stack zot dump-schema -sidebar_position: 79 +sidebar_position: 80 --- # `stack zot dump-schema` diff --git a/docs/docs/cli/zot-fix-dates.md b/docs/docs/cli/zot-fix-dates.md index 6a499a6..545efe9 100644 --- a/docs/docs/cli/zot-fix-dates.md +++ b/docs/docs/cli/zot-fix-dates.md @@ -1,6 +1,6 @@ --- title: stack zot fix-dates -sidebar_position: 80 +sidebar_position: 81 --- # `stack zot fix-dates` diff --git a/docs/docs/cli/zot-fix-fields.md b/docs/docs/cli/zot-fix-fields.md index a7bb541..aba4715 100644 --- a/docs/docs/cli/zot-fix-fields.md +++ b/docs/docs/cli/zot-fix-fields.md @@ -1,6 +1,6 @@ --- title: stack zot fix-fields -sidebar_position: 82 +sidebar_position: 83 --- # `stack zot fix-fields` diff --git a/docs/docs/cli/zot-fix-keys.md b/docs/docs/cli/zot-fix-keys.md index cd1fa1b..67ae1af 100644 --- a/docs/docs/cli/zot-fix-keys.md +++ b/docs/docs/cli/zot-fix-keys.md @@ -1,6 +1,6 @@ --- title: stack zot fix-keys -sidebar_position: 81 +sidebar_position: 82 --- # `stack zot fix-keys` diff --git a/docs/docs/cli/zot-verify-parity.md b/docs/docs/cli/zot-verify-parity.md index 36cad32..4fd4338 100644 --- a/docs/docs/cli/zot-verify-parity.md +++ b/docs/docs/cli/zot-verify-parity.md @@ -1,6 +1,6 @@ --- title: stack zot verify-parity -sidebar_position: 83 +sidebar_position: 84 --- # `stack zot verify-parity` diff --git a/docs/docs/cli/zot.md b/docs/docs/cli/zot.md index 809cf09..190db74 100644 --- a/docs/docs/cli/zot.md +++ b/docs/docs/cli/zot.md @@ -1,6 +1,6 @@ --- title: stack zot -sidebar_position: 78 +sidebar_position: 79 --- # `stack zot` diff --git a/docs/superpowers/specs/2026-09-22-llm-tagging-design.md b/docs/superpowers/specs/2026-09-22-llm-tagging-design.md new file mode 100644 index 0000000..f3ce782 --- /dev/null +++ b/docs/superpowers/specs/2026-09-22-llm-tagging-design.md @@ -0,0 +1,46 @@ +# P35 — closed-vocabulary RAG tagging of rulemaking comments into bib/Zotero + +Tracker: milestone P35 (#574 vocabulary, #575 chain + batch runner, #576 write-back + Zotero sync, #577 golden-set gate). Parent spec: `2026-07-16-llm-module-design.md` §P35. Builds on the comment pipeline (P47 dockets/seals, `.state/comments///combined.md` bodies), the pgvector index (`comments` collection, chunk metadata `docket`/`comment_id`/`item_key`/`date`), `llm.classify.closed_vocab_classifier` (one slug or `none` from the largest live host), `llm.pool` (dead-host drop, #796), and the bib tag conventions (`bib.tag` namespaces, `Store.add_tags`, `remove_tag`, nightly `zotero-sync --tag`). + +## Goal + +Every regulations.gov comment on a PFS docket carries a small set of **theme tags** from a closed vocabulary — which parts of the rule the letter is about — written as `llm:` tags on its bib item so they reach Zotero with the nightly sync, are filterable in `stack bib query` and the search UI, and give #254–#256 (position classification, stakeholder segmentation, provision mapping) a stable thematic axis. Self-hosted inference only (the llm module never calls a cloud API). + +## Decisions + +1. **The vocabulary is a versioned YAML file in the repo**, `src/llm/vocab/themes.yaml`, hand-curated. It is *seeded* from the Federal Register section headings of the CY2018–CY2027 PFS rules (mined from `fr_anchors`, `dev/scripts/llm_vocab_candidates.py`) and from the hand code families, but the file is the artifact; a run records the `version` it was tagged with. Slugs are kebab-case, ~60 themes, each with `label`, `definition` (one sentence the model sees), `synonyms` (phrases that appear in letters), and `sections` (the FR heading stems it maps to). Mechanical bib namespaces (`docket:`, `year:`, `rule:`, `reg-docket:`, `enriched:`, `code:`, `family:`) are excluded by construction — they are provenance, not themes. +2. **Multi-label, evidence-first.** A comment gets zero to `max_tags` (default 4) themes. The chain does not ask the model to pick from 60 slugs cold: it first shortlists candidates by embedding similarity between the comment's chunks and each theme's definition+synonyms (the theme "cards" are embedded once per vocab version), then asks the model a yes/no per candidate with the comment's most relevant chunk as evidence, on the largest live host, `temperature 0`. Anything outside `{yes, no}` counts as `no`. This is the closed-vocab discipline of `llm.classify` applied per theme, and the evidence chunk is stored so a human can audit any tag. +3. **Resumable, docket-scoped, idempotent.** State lives in bib: `extra_json["llm_tags"] = {"version": v, "model": m, "tags": {slug: {"confidence": c, "evidence_chunk": seq}}, "tagged_at": iso}`. A comment is skipped when its stored `version`+`model`+`content_hash` match; `--force` redoes it. `stack llm tag --docket D [--limit N] [--force]` walks a docket newest-first through the pool (`HostPool` fan-out; a dead host is dropped, #796). +4. **Write-back replaces only `llm:` tags.** `apply_tags(store, key, slugs)` removes every existing `llm:*` tag on the item and adds the new set — hand tags (`topic:`, `stance:`, …) are never touched. Confidence below `min_confidence` (default 0.6) is recorded in `extra_json` but not written as a tag. The nightly `zotero-sync` already carries tags; `--tag llm:` scopes a manual push. +5. **Golden set gates fan-out.** `tests/llm/golden_themes.yaml` holds ~200 hand-labelled comments across ≥3 dockets (comment id, expected slugs). `stack llm eval-tags` computes per-slug precision/recall and the abstain rate; CI runs it mocked (the harness only), the nightly `llm-golden` job runs it live. Corpus-wide tagging beyond the pilot docket is blocked until micro-F1 ≥ 0.70 and no theme with support ≥ 5 has precision < 0.5. +6. **Observability** reuses `llm.metrics` (`dispatch`, a `tagged` counter) — the P34 Grafana panel gains a tagging row later, not in this milestone. + +## Data flow + +``` +themes.yaml ──embed cards──▶ pgvector "themes" collection (version-stamped) +comment chunks (pgvector) ──similarity vs cards──▶ shortlist (top 8 themes) +shortlist × evidence chunk ──yes/no on largest host──▶ tags + confidence + └──▶ bib extra_json.llm_tags (state) + └──▶ bib tags llm: (write-back) + └──▶ nightly zotero-sync +golden_themes.yaml ──▶ stack llm eval-tags ──▶ precision/recall report (gate) +``` + +## Modules + +- `src/llm/vocab.py` — `Theme` dataclass, `load()` (validates unique slugs, kebab-case, non-empty definitions), `version()` (the file's `version` field), `cards()` (text per theme for embedding). +- `src/llm/tagger.py` — `shortlist()`, `judge()`, `tag_comment()`, `run()` (batch), `apply_tags()`; pure pieces take injected callables so tests never touch Ollama or pgvector. +- `src/cli/llm.py` — `vocab` (list/validate), `tag` (batch), `eval-tags` (gate). +- `dev/scripts/llm_vocab_candidates.py` — the heading miner that seeds a vocab revision (not run in CI). + +## Slices + +1. **#574** vocabulary YAML + `llm.vocab` + `stack llm vocab` (this slice). +2. **#575** theme cards in pgvector, shortlist + judge chain, batch runner with state. +3. **#576** write-back + sync verification on a pilot docket (CMS-2026-2377). +4. **#577** golden set + `eval-tags` + the gate; then corpus-wide run. + +## Out of scope + +Position/stance classification (#254's other half — reuse `stance:` from P43 tooling later), stakeholder segmentation (#255), cloud LLMs, tagging rules or the Zotero corpus (comments only). diff --git a/src/cli/llm.py b/src/cli/llm.py index 3125428..14fa0c9 100644 --- a/src/cli/llm.py +++ b/src/cli/llm.py @@ -203,3 +203,36 @@ def serve( import uvicorn uvicorn.run("llm.api:app", host=host, port=port, log_level="info") + + +@app.command() +def vocab( + path: str = typer.Option( + "", "--path", help="A vocabulary file to validate instead of the packaged one." + ), + slug: str = typer.Option("", "--slug", help="Show one theme in full."), +) -> None: + """The closed theme vocabulary for comment tagging (#574): validate it + and list its slugs, or show one theme's definition, synonyms and the FR + section stems it was seeded from.""" + from llm.vocab import VocabError, load + + try: + v = load(path or None) + except VocabError as exc: + typer.echo(f"invalid vocabulary: {exc}") + raise typer.Exit(1) from exc + if slug: + if slug not in v: + raise typer.BadParameter(f"unknown slug {slug!r}") + t = v.get(slug) + typer.echo(f"{t.tag} {t.label}") + typer.echo(f" {t.definition}") + typer.echo(f" synonyms: {', '.join(t.synonyms) or '—'}") + typer.echo(f" sections: {' | '.join(t.sections) or '—'}") + return + for t in v.themes: + typer.echo(f"{t.slug:<34} {t.label}") + typer.echo( + f"vocabulary v{v.version}: {len(v)} themes, {len(v.retired)} retired ({v.source})" + ) diff --git a/src/llm/vocab/__init__.py b/src/llm/vocab/__init__.py new file mode 100644 index 0000000..0a75a91 --- /dev/null +++ b/src/llm/vocab/__init__.py @@ -0,0 +1,151 @@ +"""The closed theme vocabulary for comment tagging (P35, #574). + +``themes.yaml`` next to this file is the artifact: hand-curated, versioned +(``version:``), seeded from the CY2018+ PFS rules' section headings +(``dev/scripts/llm_vocab_candidates.py``). Every tagging run records the +version it used, so a vocabulary change never silently re-labels old +work — bump ``version`` when a slug, definition or synonym changes and +retire slugs into ``retired:`` instead of deleting them. + + load() → Vocab (validated: unique kebab-case slugs, definitions) + Vocab.cards → {slug: text} — what gets embedded per theme (#575) + Vocab.tag(slug) → "llm:", the bib tag label +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from importlib import resources +from pathlib import Path +from typing import Any, Sequence + +import yaml + +TAG_NAMESPACE = "llm" +_SLUG = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$") + + +@dataclass(frozen=True) +class Theme: + slug: str + label: str + definition: str + synonyms: tuple[str, ...] = () + sections: tuple[str, ...] = () + + @property + def tag(self) -> str: + return f"{TAG_NAMESPACE}:{self.slug}" + + @property + def card(self) -> str: + """The text embedded for shortlisting: label, definition and the + phrases letters actually use.""" + syn = ", ".join(self.synonyms) + return f"{self.label}. {self.definition}" + (f" Also: {syn}." if syn else "") + + +@dataclass(frozen=True) +class Vocab: + version: int + themes: tuple[Theme, ...] + retired: tuple[str, ...] = () + source: str = "" + _by_slug: dict[str, Theme] = field(default_factory=dict, repr=False, compare=False) + + def __post_init__(self) -> None: + object.__setattr__(self, "_by_slug", {t.slug: t for t in self.themes}) + + @property + def slugs(self) -> tuple[str, ...]: + return tuple(t.slug for t in self.themes) + + def get(self, slug: str) -> Theme: + return self._by_slug[slug] + + def __contains__(self, slug: object) -> bool: + return slug in self._by_slug + + def __len__(self) -> int: + return len(self.themes) + + @property + def cards(self) -> dict[str, str]: + return {t.slug: t.card for t in self.themes} + + def tag(self, slug: str) -> str: + return self.get(slug).tag + + +class VocabError(ValueError): + """The YAML is not a valid vocabulary — say exactly what is wrong.""" + + +def _theme(raw: dict[str, Any]) -> Theme: + slug = str(raw.get("slug") or "").strip() + if not _SLUG.match(slug): + raise VocabError(f"slug {slug!r} is not kebab-case") + definition = str(raw.get("definition") or "").strip() + if not definition: + raise VocabError(f"{slug}: definition is empty") + label = str(raw.get("label") or "").strip() or slug + return Theme( + slug=slug, + label=label, + definition=definition, + synonyms=tuple( + str(s).strip() for s in (raw.get("synonyms") or []) if str(s).strip() + ), + sections=tuple( + str(s).strip() for s in (raw.get("sections") or []) if str(s).strip() + ), + ) + + +def parse(text: str, *, source: str = "") -> Vocab: + data = yaml.safe_load(text) or {} + if not isinstance(data, dict) or "themes" not in data: + raise VocabError( + f"{source or 'vocabulary'}: expected a mapping with a `themes` list" + ) + try: + version = int(data.get("version", 0)) + except (TypeError, ValueError) as exc: + raise VocabError(f"{source}: version must be an integer") from exc + if version < 1: + raise VocabError(f"{source}: version must be >= 1") + themes = tuple(_theme(t) for t in data.get("themes") or []) + if not themes: + raise VocabError(f"{source}: no themes") + seen: set[str] = set() + for t in themes: + if t.slug in seen: + raise VocabError(f"duplicate slug {t.slug!r}") + seen.add(t.slug) + retired = tuple( + str(s).strip() for s in (data.get("retired") or []) if str(s).strip() + ) + clash = seen & set(retired) + if clash: + raise VocabError(f"retired slugs still active: {sorted(clash)}") + return Vocab(version=version, themes=themes, retired=retired, source=source) + + +def load(path: Path | str | None = None) -> Vocab: + """The packaged ``themes.yaml`` (default) or a file on disk.""" + if path is None: + res = resources.files(__name__).joinpath("themes.yaml") + return parse(res.read_text(encoding="utf-8"), source=str(res)) + p = Path(path) + return parse(p.read_text(encoding="utf-8"), source=str(p)) + + +def is_llm_tag(label: str) -> bool: + return label.startswith(f"{TAG_NAMESPACE}:") + + +def split_tags(labels: Sequence[str]) -> tuple[list[str], list[str]]: + """``(llm tags, everything else)`` — the write-back only ever touches the first.""" + ours = [t for t in labels if is_llm_tag(t)] + return ours, [t for t in labels if not is_llm_tag(t)] diff --git a/src/llm/vocab/themes.yaml b/src/llm/vocab/themes.yaml new file mode 100644 index 0000000..759c7b3 --- /dev/null +++ b/src/llm/vocab/themes.yaml @@ -0,0 +1,276 @@ +# Closed theme vocabulary for tagging PFS rulemaking comments (P35, #574). +# +# Seeded from the section headings of the CY2018–CY2027 PFS proposed and +# final rules (dev/scripts/llm_vocab_candidates.py over fr_anchors) and the +# hand code families; curated by hand. Slugs are stable once published — +# retire a theme by moving it to `retired:` rather than deleting it, and +# bump `version` whenever a slug, definition or synonym list changes, since +# every tag run records the version it used. +version: 1 +themes: + - slug: conversion-factor + label: Conversion factor and payment update + definition: The annual PFS conversion factor, statutory update percentage, budget-neutrality adjustment, or the overall payment cut or increase. + synonyms: [conversion factor, CF, payment update, budget neutrality, statutory update, payment cut, physician payment cut] + sections: [Conversion Factor, Physician Fee Schedule Update, Budget Neutrality] + - slug: practice-expense + label: Practice expense RVUs + definition: Practice expense methodology, direct and indirect PE inputs, clinical labor pricing, supply and equipment pricing, or the PE/HR survey data. + synonyms: [practice expense, PE RVU, clinical labor, supply pricing, equipment pricing, indirect practice expense, PE/HR, AMA PPIS survey] + sections: [Determination of Practice Expense (PE) Relative Value Units (RVUs), Practice Expense Methodology] + - slug: work-rvu + label: Work RVUs and code valuation + definition: Physician work RVUs, RUC recommendations, time assumptions, or the valuation of specific CPT/HCPCS codes. + synonyms: [work RVU, RUC recommendation, valuation of specific codes, intraservice time, survey data, code valuation] + sections: [Valuation of Specific Codes, Work RVUs] + - slug: malpractice-rvu + label: Malpractice RVUs + definition: Malpractice (MP) RVU methodology, premium data, or specialty risk factors. + synonyms: [malpractice RVU, MP RVU, professional liability insurance premium, risk factor] + sections: [Determination of Malpractice Relative Value Units (RVUs)] + - slug: misvalued-codes + label: Potentially misvalued codes + definition: Identification, review or revaluation of potentially misvalued services, including the statutory misvalued-code target. + synonyms: [misvalued, potentially misvalued codes, misvalued code target, revaluation] + sections: [Potentially Misvalued Services Under the PFS] + - slug: gpci-localities + label: GPCIs and payment localities + definition: Geographic practice cost indices, locality definitions, or geographic adjustment of payment. + synonyms: [GPCI, geographic practice cost index, payment locality, locality, geographic adjustment, work GPCI floor] + sections: [Geographic Practice Cost Indices (GPCIs), Payment Localities] + - slug: telehealth + label: Medicare telehealth services + definition: The Medicare telehealth services list, originating-site and geographic restrictions, audio-only, virtual direct supervision, or telehealth flexibilities after the public health emergency. + synonyms: [telehealth, telemedicine, audio-only, originating site, distant site, virtual supervision, telehealth list, home as originating site] + sections: [Medicare Telehealth Services, Payment for Medicare Telehealth Services Under Section 1834(m) of the Act] + - slug: communication-technology-services + label: Communication technology-based services + definition: Virtual check-ins, e-visits, remote evaluation of images, interprofessional consultations and other communication technology-based services that are not telehealth. + synonyms: [virtual check-in, e-visit, online digital E/M, interprofessional consultation, remote evaluation, CTBS, G2012, 99421] + sections: [Communication Technology-Based Services (CTBS)] + - slug: remote-monitoring + label: Remote physiologic and therapeutic monitoring + definition: Remote physiologic monitoring (RPM) or remote therapeutic monitoring (RTM) codes, device supply days, or monitoring time requirements. + synonyms: [remote physiologic monitoring, RPM, remote therapeutic monitoring, RTM, 99454, 99457, 98977, 16 days] + sections: [Remote Physiologic Monitoring] + - slug: evaluation-management + label: Evaluation and management visits + definition: Office/outpatient, hospital, nursing facility or home E/M visit coding, documentation guidelines, level selection, prolonged services, or split/shared visits. + synonyms: [E/M, evaluation and management, office visit, 99213, 99214, prolonged services, split/shared, level selection, documentation guidelines] + sections: [Evaluation & Management (E/M) Visits, Evaluation & Management (E/M) Guidelines and Care Management Services] + - slug: visit-complexity-add-on + label: Visit complexity add-on (G2211) + definition: The office/outpatient visit complexity add-on code G2211, its use with modifier 25, or its budget impact. + synonyms: [G2211, visit complexity, complexity add-on, longitudinal relationship, modifier 25] + sections: [Visit Complexity] + - slug: care-management + label: Care management services + definition: Chronic care management, principal care management, transitional care management, advanced primary care management, or other monthly care-management codes and their consent, staffing and time requirements. + synonyms: [chronic care management, CCM, principal care management, PCM, transitional care management, TCM, advanced primary care management, APCM, 99490, 99439, G0556, care management] + sections: [Care Management Services, Advanced Primary Care Management] + - slug: behavioral-health + label: Behavioral health integration and services + definition: Behavioral health integration, collaborative care, psychotherapy, marriage and family therapists and mental health counselors, or crisis and safety-planning services. + synonyms: [behavioral health, mental health, collaborative care, BHI, psychotherapy, MFT, MHC, safety planning, crisis psychotherapy, community health integration] + sections: [Behavioral Health Services] + - slug: opioid-treatment + label: Opioid treatment programs and substance use disorder + definition: Opioid treatment program bundled payments, medications for opioid use disorder, or substance use disorder treatment via telehealth. + synonyms: [opioid treatment program, OTP, medication for opioid use disorder, MOUD, buprenorphine, methadone, substance use disorder, SUD] + sections: [Requirements for Opioid Treatment Programs (OTP)] + - slug: caregiver-training + label: Caregiver training services + definition: Caregiver training services furnished to a patient's caregivers, with or without the patient present. + synonyms: [caregiver training, CTS, 97550, 96202] + sections: [Caregiver Training Services] + - slug: community-health-social-needs + label: Community health integration and social needs + definition: Community health integration, principal illness navigation, social determinants of health risk assessment, or community health worker services. + synonyms: [community health integration, CHI, principal illness navigation, PIN, social determinants of health, SDOH risk assessment, community health worker, G0019, G0136] + sections: [Community Health Integration, Principal Illness Navigation] + - slug: palliative-care + label: Palliative and serious-illness care + definition: Community-based palliative care, advance care planning, hospice interaction, or serious-illness care management. + synonyms: [palliative care, advance care planning, ACP, serious illness, hospice, end-of-life] + sections: [Request for Information on Community-Based Palliative Care] + - slug: supervision + label: Supervision requirements + definition: Direct, general or virtual supervision of auxiliary personnel, incident-to services, or supervision of diagnostic tests and residents. + synonyms: [direct supervision, general supervision, incident to, incident-to, supervision of residents, virtual presence, teaching physician] + sections: [Direct Supervision by Interactive Telecommunications Technology, Teaching Physician Documentation Requirements] + - slug: non-physician-practitioners + label: Non-physician practitioners and scope + definition: Nurse practitioners, physician assistants, clinical nurse specialists, therapists or other non-physician practitioners' billing, supervision or scope-of-practice policies. + synonyms: [nurse practitioner, physician assistant, PA, NP, advanced practice, scope of practice, CRNA, non-physician practitioner, NPP] + sections: [Physician Assistants, Nurse Practitioners] + - slug: therapy-services + label: Physical, occupational and speech therapy + definition: Therapy services, therapy assistants and the CQ/CO modifiers, the KX modifier threshold, plan-of-care certification, or therapy supervision. + synonyms: [physical therapy, occupational therapy, speech-language pathology, therapy assistant, CQ modifier, CO modifier, KX modifier, therapy threshold, plan of care] + sections: [Therapy Services, Therapy Caps] + - slug: drugs-biologicals + label: Part B drugs and biologicals + definition: Average sales price methodology, drug add-on percentages, biosimilars, discarded-drug refunds, or drug payment under Part B. + synonyms: [average sales price, ASP, WAC, biosimilar, discarded drug, JW modifier, JZ modifier, Part B drug, drug payment, 340B] + sections: [Part B Drug Payment, Payment for Biosimilar Biological Products] + - slug: vaccines-preventive + label: Vaccines and preventive services + definition: Vaccine administration payment, preventive services coverage, screening services, or the Medicare Diabetes Prevention Program. + synonyms: [vaccine administration, preventive services, screening, colorectal cancer screening, Medicare Diabetes Prevention Program, MDPP, hepatitis B vaccine, HIV PrEP] + sections: [Medicare Diabetes Prevention Program, Preventive Services] + - slug: skin-substitutes + label: Skin substitutes and wound care + definition: Payment or coding for skin substitute products, cellular and tissue-based products, or wound care services. + synonyms: [skin substitute, cellular and tissue-based product, CTP, wound care, Q4 code, amniotic, dermal substitute] + sections: [Skin Substitutes] + - slug: dental-services + label: Dental services + definition: Medicare payment for dental services inextricably linked to covered medical services. + synonyms: [dental, oral health, dental services, inextricably linked] + sections: [Dental Services] + - slug: global-surgery + label: Global surgical packages + definition: Global surgery periods, post-operative visit reporting, or the transfer-of-care modifier for global packages. + synonyms: [global surgery, global period, 010-day, 090-day, post-operative visit, modifier 54, modifier 55, transfer of care, G0559] + sections: [Global Surgery] + - slug: anesthesia + label: Anesthesia services + definition: The anesthesia conversion factor, anesthesia base units, or anesthesia billing and supervision. + synonyms: [anesthesia, anesthesia conversion factor, base units, CRNA, medical direction] + sections: [Anesthesia Services, Anesthesia Conversion Factor] + - slug: imaging-radiology + label: Imaging and radiology + definition: Diagnostic imaging payment, appropriate use criteria for advanced imaging, or radiology assistants and supervision. + synonyms: [imaging, radiology, appropriate use criteria, AUC, clinical decision support, CT, MRI, radiologist assistant] + sections: [Appropriate Use Criteria for Advanced Diagnostic Imaging Services] + - slug: laboratory + label: Clinical laboratory fee schedule + definition: The clinical laboratory fee schedule, private payer rate reporting, specimen collection, or laboratory test payment. + synonyms: [clinical laboratory fee schedule, CLFS, private payor rate, PAMA, laboratory test, specimen collection] + sections: [Clinical Laboratory Fee Schedule] + - slug: ambulance + label: Ambulance services + definition: The ambulance fee schedule, ground or air ambulance payment, or ambulance cost data collection. + synonyms: [ambulance, ambulance fee schedule, ground ambulance, air ambulance, ambulance cost collection] + sections: [Ambulance Fee Schedule] + - slug: dme-supplies + label: Durable medical equipment and supplies + definition: DME infusion drugs, DMEPOS competitive bidding, or supplies furnished with physician services. + synonyms: [durable medical equipment, DME, DMEPOS, infusion drugs, competitive bidding] + sections: [Payment for DME Infusion Drugs] + - slug: esrd-dialysis + label: ESRD and dialysis services + definition: Monthly capitation payment for ESRD-related services, home dialysis, or kidney disease education. + synonyms: [ESRD, dialysis, monthly capitation payment, MCP, home dialysis, kidney disease education] + sections: [Monthly Capitation Payment (MCP) for ESRD-Related Services] + - slug: rhc-fqhc + label: Rural health clinics and FQHCs + definition: Payment or care-coordination policies for rural health clinics and federally qualified health centers. + synonyms: [rural health clinic, RHC, federally qualified health center, FQHC, G0511, all-inclusive rate] + sections: [Rural Health Clinics (RHCs) and Federally Qualified Health Centers (FQHCs)] + - slug: rural-access + label: Rural and underserved access + definition: Access to care in rural or underserved areas, workforce shortages, or health equity concerns raised about a policy's effect on access. + synonyms: [rural, underserved, health equity, access to care, workforce shortage, health professional shortage area, HPSA] + sections: [Health Equity] + - slug: shared-savings-program + label: Medicare Shared Savings Program + definition: Shared Savings Program ACO participation, benchmarking, risk tracks, quality reporting, beneficiary assignment, or advance investment payments. + synonyms: [Shared Savings Program, MSSP, accountable care organization, ACO, benchmark, ENHANCED track, BASIC track, beneficiary assignment, advance investment payment] + sections: [Medicare Shared Savings Program] + - slug: quality-payment-program + label: Quality Payment Program and MIPS + definition: MIPS categories, MIPS Value Pathways, scoring, performance thresholds, or the QPP reporting requirements. + synonyms: [Quality Payment Program, QPP, MIPS, MIPS Value Pathway, MVP, performance threshold, promoting interoperability, cost category, improvement activities] + sections: [Updates to the Quality Payment Program] + - slug: advanced-apm + label: Advanced APMs and APM incentives + definition: Advanced alternative payment model participation, qualifying participant thresholds, the APM incentive payment, or the APM conversion factor. + synonyms: [advanced APM, alternative payment model, qualifying participant, QP threshold, APM incentive, APM conversion factor] + sections: [Advanced Alternative Payment Models] + - slug: quality-measures + label: Quality measures and reporting + definition: Specific quality measures, measure specifications, electronic clinical quality measures, or measure reporting burden. + synonyms: [quality measure, eCQM, measure specification, reporting burden, measure set, benchmarking of measures] + sections: [Quality Measures] + - slug: innovation-models + label: Innovation Center models + definition: CMS Innovation Center models and demonstrations, including primary care, kidney, oncology or radiation models. + synonyms: [Innovation Center, CMMI, model, demonstration, Primary Care First, ACO REACH, oncology care model, radiation oncology model] + sections: [Innovation Center Models] + - slug: self-referral + label: Physician self-referral law + definition: Stark law exceptions, the annual code list update, or self-referral compliance. + synonyms: [self-referral, Stark, physician self-referral law, designated health services] + sections: [Physician Self-Referral Law] + - slug: enrollment-program-integrity + label: Enrollment and program integrity + definition: Provider enrollment requirements, revocations, prior authorization, fraud, waste and abuse controls, or medical review. + synonyms: [enrollment, revocation, program integrity, fraud waste and abuse, prior authorization, medical review, overpayment, audit] + sections: [Medicare Provider Enrollment, Program Integrity] + - slug: coverage-policy + label: Coverage and benefit categories + definition: Whether a service is a covered Medicare benefit, benefit-category determinations, or national coverage policy raised in the rule. + synonyms: [coverage, benefit category, covered service, national coverage determination, NCD, reasonable and necessary] + sections: [Coverage] + - slug: beneficiary-cost-sharing + label: Beneficiary cost sharing + definition: Coinsurance, deductibles, or beneficiary out-of-pocket effects of a payment policy. + synonyms: [cost sharing, coinsurance, deductible, out-of-pocket, beneficiary liability, copay] + sections: [Telehealth Modalities and Cost-sharing] + - slug: administrative-burden + label: Administrative burden and documentation + definition: Documentation, paperwork, reporting or compliance burden imposed on practices, including collection-of-information estimates. + synonyms: [administrative burden, documentation burden, paperwork, compliance burden, collection of information, patients over paperwork, prior authorization burden] + sections: [Collection of Information Requirements] + - slug: regulatory-impact + label: Regulatory impact and specialty impacts + definition: The regulatory impact analysis, specialty-level payment impact tables, or estimated effects on practices and beneficiaries. + synonyms: [regulatory impact analysis, impact table, specialty impact, Table 118, estimated impact, small entities] + sections: [Regulatory Impact Analysis] + - slug: medicare-economic-index + label: Medicare Economic Index and cost weights + definition: The Medicare Economic Index, its rebasing or revised cost-share weights, and their effect on RVU allocation. + synonyms: [Medicare Economic Index, MEI, cost share weights, rebasing] + sections: [The Percentage Change in the Medicare Economic Index (MEI)] + - slug: efficiency-adjustment + label: Efficiency adjustment to work RVUs + definition: An across-the-board efficiency adjustment to work RVUs or intraservice time for non-time-based services. + synonyms: [efficiency adjustment, efficiency gains, across-the-board reduction, non-time-based services] + sections: [Efficiency Adjustment] + - slug: site-of-service + label: Site of service and facility payment + definition: Facility versus non-facility payment, site-neutral policies, hospital outpatient department comparisons, or indirect PE by setting. + synonyms: [site of service, site neutral, facility rate, non-facility rate, hospital outpatient department, HOPD, office-based] + sections: [Site of Service] + - slug: hospice-home-health + label: Hospice and home health interaction + definition: Hospice face-to-face encounters, homebound status, home health certification, or physician services interacting with hospice and home health benefits. + synonyms: [hospice, home health, face-to-face encounter, homebound, certification of home health] + sections: [Clarification of Homebound Status Under the Medicare Home Health Benefit, Telehealth and the Medicare Hospice Face-to-Face Encounter Requirement] + - slug: interoperability-health-it + label: Health IT and interoperability + definition: Certified EHR technology, promoting interoperability requirements, data exchange, or the EHR incentive programs. + synonyms: [electronic health record, EHR, certified EHR technology, CEHRT, interoperability, health information exchange, promoting interoperability] + sections: [Medicare EHR Incentive Program, Medicaid Promoting Interoperability Program Requirements] + - slug: price-transparency + label: Price transparency + definition: Public disclosure of prices, charges or payment rates to beneficiaries. + synonyms: [price transparency, charge transparency, disclosure of prices] + sections: [Request for Information on Price Transparency] + - slug: covid-phe + label: COVID-19 public health emergency flexibilities + definition: Policies tied to the COVID-19 public health emergency and their extension or expiration. + synonyms: [public health emergency, PHE, COVID-19, pandemic, flexibilities, waiver] + sections: [Counting of Resident Time During the PHE for the COVID-19 Pandemic] + - slug: data-requests-rfi + label: Requests for information + definition: Responses to a request for information or comment solicitation rather than a specific proposal. + synonyms: [request for information, RFI, comment solicitation, seeking comment, solicit feedback] + sections: [Requests for Information] + - slug: coding-descriptors + label: Coding and code descriptors + definition: Creation, deletion or crosswalk of CPT/HCPCS codes, code descriptors, or bundling of services into other codes. + synonyms: [new code, deleted code, HCPCS code, G code, crosswalk, bundled, code descriptor, CPT Editorial Panel, unbundle] + sections: [Proposed Valuation of Specific Codes] +retired: [] diff --git a/tests/llm/test_vocab.py b/tests/llm/test_vocab.py new file mode 100644 index 0000000..a043524 --- /dev/null +++ b/tests/llm/test_vocab.py @@ -0,0 +1,114 @@ +"""llm.vocab — the closed theme vocabulary (#574).""" + +from __future__ import annotations + +import pytest + +from llm.vocab import Theme, Vocab, VocabError, is_llm_tag, load, parse, split_tags + +GOOD = """ +version: 3 +themes: + - slug: telehealth + label: Telehealth + definition: Medicare telehealth services. + synonyms: [telehealth, audio-only] + sections: [Medicare Telehealth Services] + - slug: care-management + definition: Monthly care management codes. +retired: [old-theme] +""" + + +class TestParse: + def test_reads_themes_and_version(self): + v = parse(GOOD, source="t") + assert isinstance(v, Vocab) and v.version == 3 and len(v) == 2 + assert v.slugs == ("telehealth", "care-management") + t = v.get("telehealth") + assert isinstance(t, Theme) and t.synonyms == ("telehealth", "audio-only") + assert ( + v.get("care-management").label == "care-management" + ) # label defaults to slug + assert v.retired == ("old-theme",) and "telehealth" in v and "nope" not in v + + def test_cards_and_tags(self): + v = parse(GOOD) + assert ( + v.cards["telehealth"] + == "Telehealth. Medicare telehealth services. Also: telehealth, audio-only." + ) + assert ( + v.cards["care-management"] + == "care-management. Monthly care management codes." + ) + assert ( + v.tag("telehealth") == "llm:telehealth" + and v.get("telehealth").tag == "llm:telehealth" + ) + + @pytest.mark.parametrize( + "text, msg", + [ + ( + "version: 1\nthemes:\n - slug: Bad Slug\n definition: x\n", + "kebab-case", + ), + ( + "version: 1\nthemes:\n - slug: a\n definition: ''\n", + "definition is empty", + ), + ( + "version: 1\nthemes:\n - slug: a\n definition: x\n - slug: a\n definition: y\n", + "duplicate slug", + ), + ( + "version: 0\nthemes:\n - slug: a\n definition: x\n", + "version must be >= 1", + ), + ("version: x\nthemes: []\n", "version must be an integer"), + ("version: 1\nthemes: []\n", "no themes"), + ("- a\n- b\n", "expected a mapping"), + ( + "version: 1\nthemes:\n - slug: a\n definition: x\nretired: [a]\n", + "retired slugs still active", + ), + ], + ) + def test_rejects_bad_files(self, text, msg): + with pytest.raises(VocabError, match=msg): + parse(text, source="t") + + +class TestPackagedVocab: + def test_loads_and_is_sane(self): + v = load() + assert v.version >= 1 and 40 <= len(v) <= 150 + assert all(t.definition.endswith(".") for t in v.themes) + assert all(len(t.synonyms) >= 1 for t in v.themes) + for must in ( + "telehealth", + "care-management", + "conversion-factor", + "practice-expense", + "palliative-care", + "skin-substitutes", + ): + assert must in v + + def test_load_from_path(self, tmp_path): + p = tmp_path / "v.yaml" + p.write_text(GOOD) + assert load(p).source == str(p) + + +class TestTagHelpers: + def test_split(self): + ours, rest = split_tags( + ["llm:telehealth", "topic:palliative", "llm:care-management", "docket:x"] + ) + assert ours == ["llm:telehealth", "llm:care-management"] and rest == [ + "topic:palliative", + "docket:x", + ] + assert is_llm_tag("llm:x") and not is_llm_tag("llmx:y") From 65f8d785a1243f7d2b3fd97e1c2ed296c13f2199 Mon Sep 17 00:00:00 2001 From: kert Date: Tue, 22 Sep 2026 16:53:23 -0400 Subject: [PATCH 2/2] =?UTF-8?q?feat(llm):=20theme=20tagging=20chain=20+=20?= =?UTF-8?q?docket=20runner=20+=20write-back=20=E2=80=94=20llm.tagger,=20St?= =?UTF-8?q?ore.merge=5Fextra,=20stack=20llm=20tag=20(refs=20#575,=20#576)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit llm.tagger: cosine shortlist of the vocabulary's theme cards against a comment's already-indexed chunks (no re-embedding — text and vector come from langchain_pg_embedding), one yes/no judgement per candidate on the largest live host at temperature 0 with the scoring chunk as evidence, up to max_tags accepted with the shortlist similarity as confidence. State on the bib item (extra_json.llm_tags: vocab version, model, content hash, tags with evidence chunk, the full shortlist with verdicts) makes runs resumable; apply_tags replaces exactly the item's llm:* tags and never touches hand tags. Store.merge_extra merges top-level keys into extra_json. Every collaborator is injected so the pure pieces are tested without Ollama or pgvector. stack llm tag --docket D [--limit N] [--force] [--dry-run] [--top] [--max-tags] [--min-confidence] [--verbose]; _tagging_runtime wires the store, pgvector loader, card vectors, judge and model. --- docs/docs/cli/api-serve.md | 2 +- docs/docs/cli/api.md | 2 +- docs/docs/cli/db-comment.md | 2 +- docs/docs/cli/db-inspect.md | 2 +- docs/docs/cli/db.md | 2 +- docs/docs/cli/docs-build.md | 2 +- docs/docs/cli/docs-generate.md | 2 +- docs/docs/cli/docs-serve.md | 2 +- docs/docs/cli/docs.md | 2 +- docs/docs/cli/llm-tag.md | 38 +++ docs/docs/cli/llm.md | 5 + docs/docs/cli/mail-attach-smarthost.md | 2 +- docs/docs/cli/mail-dkim-export.md | 2 +- docs/docs/cli/mail-dns.md | 2 +- docs/docs/cli/mail-down.md | 2 +- docs/docs/cli/mail-provision.md | 2 +- docs/docs/cli/mail-rotate-creds.md | 2 +- docs/docs/cli/mail-seed-mailboxes.md | 2 +- docs/docs/cli/mail-status.md | 2 +- docs/docs/cli/mail-up.md | 2 +- docs/docs/cli/mail-wire-git.md | 2 +- docs/docs/cli/mail.md | 2 +- docs/docs/cli/perf-show.md | 2 +- docs/docs/cli/perf.md | 2 +- docs/docs/cli/pfs-cpt-ingest.md | 2 +- docs/docs/cli/pfs-elements.md | 2 +- docs/docs/cli/pfs-exposure.md | 2 +- docs/docs/cli/pfs-families.md | 2 +- docs/docs/cli/pfs-guidance.md | 2 +- docs/docs/cli/pfs-lineage.md | 2 +- docs/docs/cli/pfs-reaction.md | 2 +- docs/docs/cli/pfs-review.md | 2 +- docs/docs/cli/pfs-utilization.md | 2 +- docs/docs/cli/pfs.md | 2 +- docs/docs/cli/prisma-eligible.md | 2 +- docs/docs/cli/prisma-export.md | 2 +- docs/docs/cli/prisma-extract.md | 2 +- docs/docs/cli/prisma-fetch.md | 2 +- docs/docs/cli/prisma-flow.md | 2 +- docs/docs/cli/prisma-init.md | 2 +- docs/docs/cli/prisma-ping-llm.md | 2 +- docs/docs/cli/prisma-run.md | 2 +- docs/docs/cli/prisma-screen.md | 2 +- docs/docs/cli/prisma-vpn.md | 2 +- docs/docs/cli/prisma.md | 2 +- docs/docs/cli/rec-list.md | 2 +- docs/docs/cli/rec-opps.md | 2 +- docs/docs/cli/rec-pfs.md | 2 +- docs/docs/cli/rec.md | 2 +- docs/docs/cli/zot-dump-schema.md | 2 +- docs/docs/cli/zot-fix-dates.md | 2 +- docs/docs/cli/zot-fix-fields.md | 2 +- docs/docs/cli/zot-fix-keys.md | 2 +- docs/docs/cli/zot-verify-parity.md | 2 +- docs/docs/cli/zot.md | 2 +- src/bib/store.py | 18 ++ src/cli/llm.py | 102 +++++++ src/llm/tagger.py | 369 +++++++++++++++++++++++ tests/cli/test_llm_tag_cli.py | 103 +++++++ tests/llm/test_tagger.py | 390 +++++++++++++++++++++++++ 60 files changed, 1078 insertions(+), 53 deletions(-) create mode 100644 docs/docs/cli/llm-tag.md create mode 100644 src/llm/tagger.py create mode 100644 tests/cli/test_llm_tag_cli.py create mode 100644 tests/llm/test_tagger.py diff --git a/docs/docs/cli/api-serve.md b/docs/docs/cli/api-serve.md index 78fccfa..6bc5db2 100644 --- a/docs/docs/cli/api-serve.md +++ b/docs/docs/cli/api-serve.md @@ -1,6 +1,6 @@ --- title: stack api serve -sidebar_position: 62 +sidebar_position: 63 --- # `stack api serve` diff --git a/docs/docs/cli/api.md b/docs/docs/cli/api.md index acf8249..3c8cbbe 100644 --- a/docs/docs/cli/api.md +++ b/docs/docs/cli/api.md @@ -1,6 +1,6 @@ --- title: stack api -sidebar_position: 61 +sidebar_position: 62 --- # `stack api` diff --git a/docs/docs/cli/db-comment.md b/docs/docs/cli/db-comment.md index c0f7f9c..b3364a1 100644 --- a/docs/docs/cli/db-comment.md +++ b/docs/docs/cli/db-comment.md @@ -1,6 +1,6 @@ --- title: stack db comment -sidebar_position: 55 +sidebar_position: 56 --- # `stack db comment` diff --git a/docs/docs/cli/db-inspect.md b/docs/docs/cli/db-inspect.md index 29ae60d..78db95e 100644 --- a/docs/docs/cli/db-inspect.md +++ b/docs/docs/cli/db-inspect.md @@ -1,6 +1,6 @@ --- title: stack db inspect -sidebar_position: 56 +sidebar_position: 57 --- # `stack db inspect` diff --git a/docs/docs/cli/db.md b/docs/docs/cli/db.md index 8c73d5b..529cd59 100644 --- a/docs/docs/cli/db.md +++ b/docs/docs/cli/db.md @@ -1,6 +1,6 @@ --- title: stack db -sidebar_position: 54 +sidebar_position: 55 --- # `stack db` diff --git a/docs/docs/cli/docs-build.md b/docs/docs/cli/docs-build.md index e243395..d37c434 100644 --- a/docs/docs/cli/docs-build.md +++ b/docs/docs/cli/docs-build.md @@ -1,6 +1,6 @@ --- title: stack docs build -sidebar_position: 58 +sidebar_position: 59 --- # `stack docs build` diff --git a/docs/docs/cli/docs-generate.md b/docs/docs/cli/docs-generate.md index 6b0b71f..e37d9dd 100644 --- a/docs/docs/cli/docs-generate.md +++ b/docs/docs/cli/docs-generate.md @@ -1,6 +1,6 @@ --- title: stack docs generate -sidebar_position: 60 +sidebar_position: 61 --- # `stack docs generate` diff --git a/docs/docs/cli/docs-serve.md b/docs/docs/cli/docs-serve.md index b0649b3..76a6fd1 100644 --- a/docs/docs/cli/docs-serve.md +++ b/docs/docs/cli/docs-serve.md @@ -1,6 +1,6 @@ --- title: stack docs serve -sidebar_position: 59 +sidebar_position: 60 --- # `stack docs serve` diff --git a/docs/docs/cli/docs.md b/docs/docs/cli/docs.md index 2067ef6..09edd1b 100644 --- a/docs/docs/cli/docs.md +++ b/docs/docs/cli/docs.md @@ -1,6 +1,6 @@ --- title: stack docs -sidebar_position: 57 +sidebar_position: 58 --- # `stack docs` diff --git a/docs/docs/cli/llm-tag.md b/docs/docs/cli/llm-tag.md new file mode 100644 index 0000000..e08c067 --- /dev/null +++ b/docs/docs/cli/llm-tag.md @@ -0,0 +1,38 @@ +--- +title: stack llm tag +sidebar_position: 54 +--- + +# `stack llm tag` + +``` +Usage: stack llm tag [OPTIONS] + + Closed-vocabulary theme tagging of one docket's comments (P35): shortlist by + similarity to the theme cards, judge each candidate yes/no on the largest live + host with the scoring chunk as evidence, record state in the item's + extra_json, and replace its llm: tags. Resumable — unchanged comments are + skipped unless --force. + +╭─ Options ────────────────────────────────────────────────────────────────────╮ +│ * --docket TEXT regulations.gov docket id, e.g. │ +│ CMS-2026-2377. │ +│ [required] │ +│ --limit INTEGER Stop after N comments (newest first; 0 = │ +│ all). │ +│ [default: 0] │ +│ --force Re-tag comments whose stored state is │ +│ current. │ +│ --dry-run Compute and print, write nothing. │ +│ --top INTEGER Themes shortlisted per comment before │ +│ judging. │ +│ [default: 8] │ +│ --max-tags INTEGER Most themes written per comment. │ +│ [default: 4] │ +│ --min-confidence FLOAT Shortlist similarity below which a │ +│ judged yes is not written. │ +│ [default: 0.0] │ +│ --verbose Print every comment's tags. │ +│ --help Show this message and exit. │ +╰──────────────────────────────────────────────────────────────────────────────╯ +``` diff --git a/docs/docs/cli/llm.md b/docs/docs/cli/llm.md index 4128367..a04923d 100644 --- a/docs/docs/cli/llm.md +++ b/docs/docs/cli/llm.md @@ -23,5 +23,10 @@ Usage: stack llm [OPTIONS] COMMAND [ARGS]... │ and list its slugs, or show one theme's definition, synonyms and │ │ the FR │ │ section stems it was seeded from. │ +│ tag Closed-vocabulary theme tagging of one docket's comments (P35): │ +│ shortlist by similarity to the theme cards, judge each candidate │ +│ yes/no on the largest live host with the scoring chunk as evidence, │ +│ record state in the item's extra_json, and replace its llm: tags. │ +│ Resumable — unchanged comments are skipped unless --force. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/docs/docs/cli/mail-attach-smarthost.md b/docs/docs/cli/mail-attach-smarthost.md index f2b38f0..7758fbd 100644 --- a/docs/docs/cli/mail-attach-smarthost.md +++ b/docs/docs/cli/mail-attach-smarthost.md @@ -1,6 +1,6 @@ --- title: stack mail attach-smarthost -sidebar_position: 103 +sidebar_position: 104 --- # `stack mail attach-smarthost` diff --git a/docs/docs/cli/mail-dkim-export.md b/docs/docs/cli/mail-dkim-export.md index a97e00b..9c67c22 100644 --- a/docs/docs/cli/mail-dkim-export.md +++ b/docs/docs/cli/mail-dkim-export.md @@ -1,6 +1,6 @@ --- title: stack mail dkim-export -sidebar_position: 102 +sidebar_position: 103 --- # `stack mail dkim-export` diff --git a/docs/docs/cli/mail-dns.md b/docs/docs/cli/mail-dns.md index 595e177..e1730e5 100644 --- a/docs/docs/cli/mail-dns.md +++ b/docs/docs/cli/mail-dns.md @@ -1,6 +1,6 @@ --- title: stack mail dns -sidebar_position: 101 +sidebar_position: 102 --- # `stack mail dns` diff --git a/docs/docs/cli/mail-down.md b/docs/docs/cli/mail-down.md index e23a231..2deb87e 100644 --- a/docs/docs/cli/mail-down.md +++ b/docs/docs/cli/mail-down.md @@ -1,6 +1,6 @@ --- title: stack mail down -sidebar_position: 99 +sidebar_position: 100 --- # `stack mail down` diff --git a/docs/docs/cli/mail-provision.md b/docs/docs/cli/mail-provision.md index cceec5d..f6fe5b0 100644 --- a/docs/docs/cli/mail-provision.md +++ b/docs/docs/cli/mail-provision.md @@ -1,6 +1,6 @@ --- title: stack mail provision -sidebar_position: 97 +sidebar_position: 98 --- # `stack mail provision` diff --git a/docs/docs/cli/mail-rotate-creds.md b/docs/docs/cli/mail-rotate-creds.md index 0d98cd4..1f80c5f 100644 --- a/docs/docs/cli/mail-rotate-creds.md +++ b/docs/docs/cli/mail-rotate-creds.md @@ -1,6 +1,6 @@ --- title: stack mail rotate-creds -sidebar_position: 104 +sidebar_position: 105 --- # `stack mail rotate-creds` diff --git a/docs/docs/cli/mail-seed-mailboxes.md b/docs/docs/cli/mail-seed-mailboxes.md index ffe2741..3dada3d 100644 --- a/docs/docs/cli/mail-seed-mailboxes.md +++ b/docs/docs/cli/mail-seed-mailboxes.md @@ -1,6 +1,6 @@ --- title: stack mail seed-mailboxes -sidebar_position: 105 +sidebar_position: 106 --- # `stack mail seed-mailboxes` diff --git a/docs/docs/cli/mail-status.md b/docs/docs/cli/mail-status.md index b0a565a..9ed0514 100644 --- a/docs/docs/cli/mail-status.md +++ b/docs/docs/cli/mail-status.md @@ -1,6 +1,6 @@ --- title: stack mail status -sidebar_position: 100 +sidebar_position: 101 --- # `stack mail status` diff --git a/docs/docs/cli/mail-up.md b/docs/docs/cli/mail-up.md index 1bda771..c39cb51 100644 --- a/docs/docs/cli/mail-up.md +++ b/docs/docs/cli/mail-up.md @@ -1,6 +1,6 @@ --- title: stack mail up -sidebar_position: 98 +sidebar_position: 99 --- # `stack mail up` diff --git a/docs/docs/cli/mail-wire-git.md b/docs/docs/cli/mail-wire-git.md index 21a087e..df87669 100644 --- a/docs/docs/cli/mail-wire-git.md +++ b/docs/docs/cli/mail-wire-git.md @@ -1,6 +1,6 @@ --- title: stack mail wire-git -sidebar_position: 106 +sidebar_position: 107 --- # `stack mail wire-git` diff --git a/docs/docs/cli/mail.md b/docs/docs/cli/mail.md index 99f002e..b7ef94b 100644 --- a/docs/docs/cli/mail.md +++ b/docs/docs/cli/mail.md @@ -1,6 +1,6 @@ --- title: stack mail -sidebar_position: 96 +sidebar_position: 97 --- # `stack mail` diff --git a/docs/docs/cli/perf-show.md b/docs/docs/cli/perf-show.md index 7f0bec3..f1e0c71 100644 --- a/docs/docs/cli/perf-show.md +++ b/docs/docs/cli/perf-show.md @@ -1,6 +1,6 @@ --- title: stack perf show -sidebar_position: 64 +sidebar_position: 65 --- # `stack perf show` diff --git a/docs/docs/cli/perf.md b/docs/docs/cli/perf.md index a07728d..04402e7 100644 --- a/docs/docs/cli/perf.md +++ b/docs/docs/cli/perf.md @@ -1,6 +1,6 @@ --- title: stack perf -sidebar_position: 63 +sidebar_position: 64 --- # `stack perf` diff --git a/docs/docs/cli/pfs-cpt-ingest.md b/docs/docs/cli/pfs-cpt-ingest.md index b4d6a02..c3cbd82 100644 --- a/docs/docs/cli/pfs-cpt-ingest.md +++ b/docs/docs/cli/pfs-cpt-ingest.md @@ -1,6 +1,6 @@ --- title: stack pfs cpt-ingest -sidebar_position: 78 +sidebar_position: 79 --- # `stack pfs cpt-ingest` diff --git a/docs/docs/cli/pfs-elements.md b/docs/docs/cli/pfs-elements.md index cffc9c9..b30a493 100644 --- a/docs/docs/cli/pfs-elements.md +++ b/docs/docs/cli/pfs-elements.md @@ -1,6 +1,6 @@ --- title: stack pfs elements -sidebar_position: 70 +sidebar_position: 71 --- # `stack pfs elements` diff --git a/docs/docs/cli/pfs-exposure.md b/docs/docs/cli/pfs-exposure.md index 3b443d3..32c8775 100644 --- a/docs/docs/cli/pfs-exposure.md +++ b/docs/docs/cli/pfs-exposure.md @@ -1,6 +1,6 @@ --- title: stack pfs exposure -sidebar_position: 75 +sidebar_position: 76 --- # `stack pfs exposure` diff --git a/docs/docs/cli/pfs-families.md b/docs/docs/cli/pfs-families.md index 87c34cd..f120337 100644 --- a/docs/docs/cli/pfs-families.md +++ b/docs/docs/cli/pfs-families.md @@ -1,6 +1,6 @@ --- title: stack pfs families -sidebar_position: 72 +sidebar_position: 73 --- # `stack pfs families` diff --git a/docs/docs/cli/pfs-guidance.md b/docs/docs/cli/pfs-guidance.md index e5837e5..8f4ff52 100644 --- a/docs/docs/cli/pfs-guidance.md +++ b/docs/docs/cli/pfs-guidance.md @@ -1,6 +1,6 @@ --- title: stack pfs guidance -sidebar_position: 73 +sidebar_position: 74 --- # `stack pfs guidance` diff --git a/docs/docs/cli/pfs-lineage.md b/docs/docs/cli/pfs-lineage.md index a12df77..6dcc13c 100644 --- a/docs/docs/cli/pfs-lineage.md +++ b/docs/docs/cli/pfs-lineage.md @@ -1,6 +1,6 @@ --- title: stack pfs lineage -sidebar_position: 71 +sidebar_position: 72 --- # `stack pfs lineage` diff --git a/docs/docs/cli/pfs-reaction.md b/docs/docs/cli/pfs-reaction.md index ebd436a..529be24 100644 --- a/docs/docs/cli/pfs-reaction.md +++ b/docs/docs/cli/pfs-reaction.md @@ -1,6 +1,6 @@ --- title: stack pfs reaction -sidebar_position: 74 +sidebar_position: 75 --- # `stack pfs reaction` diff --git a/docs/docs/cli/pfs-review.md b/docs/docs/cli/pfs-review.md index e8836a8..0bf7a12 100644 --- a/docs/docs/cli/pfs-review.md +++ b/docs/docs/cli/pfs-review.md @@ -1,6 +1,6 @@ --- title: stack pfs review -sidebar_position: 77 +sidebar_position: 78 --- # `stack pfs review` diff --git a/docs/docs/cli/pfs-utilization.md b/docs/docs/cli/pfs-utilization.md index 9c2a8d9..f98974b 100644 --- a/docs/docs/cli/pfs-utilization.md +++ b/docs/docs/cli/pfs-utilization.md @@ -1,6 +1,6 @@ --- title: stack pfs utilization -sidebar_position: 76 +sidebar_position: 77 --- # `stack pfs utilization` diff --git a/docs/docs/cli/pfs.md b/docs/docs/cli/pfs.md index 7578636..7fb6804 100644 --- a/docs/docs/cli/pfs.md +++ b/docs/docs/cli/pfs.md @@ -1,6 +1,6 @@ --- title: stack pfs -sidebar_position: 69 +sidebar_position: 70 --- # `stack pfs` diff --git a/docs/docs/cli/prisma-eligible.md b/docs/docs/cli/prisma-eligible.md index 08610fb..62fff2e 100644 --- a/docs/docs/cli/prisma-eligible.md +++ b/docs/docs/cli/prisma-eligible.md @@ -1,6 +1,6 @@ --- title: stack prisma eligible -sidebar_position: 90 +sidebar_position: 91 --- # `stack prisma eligible` diff --git a/docs/docs/cli/prisma-export.md b/docs/docs/cli/prisma-export.md index 26a5bcc..849b725 100644 --- a/docs/docs/cli/prisma-export.md +++ b/docs/docs/cli/prisma-export.md @@ -1,6 +1,6 @@ --- title: stack prisma export -sidebar_position: 87 +sidebar_position: 88 --- # `stack prisma export` diff --git a/docs/docs/cli/prisma-extract.md b/docs/docs/cli/prisma-extract.md index 08cbf09..fe02a9b 100644 --- a/docs/docs/cli/prisma-extract.md +++ b/docs/docs/cli/prisma-extract.md @@ -1,6 +1,6 @@ --- title: stack prisma extract -sidebar_position: 91 +sidebar_position: 92 --- # `stack prisma extract` diff --git a/docs/docs/cli/prisma-fetch.md b/docs/docs/cli/prisma-fetch.md index 33bf0eb..8c7ceda 100644 --- a/docs/docs/cli/prisma-fetch.md +++ b/docs/docs/cli/prisma-fetch.md @@ -1,6 +1,6 @@ --- title: stack prisma fetch -sidebar_position: 93 +sidebar_position: 94 --- # `stack prisma fetch` diff --git a/docs/docs/cli/prisma-flow.md b/docs/docs/cli/prisma-flow.md index 44a0781..97a7eab 100644 --- a/docs/docs/cli/prisma-flow.md +++ b/docs/docs/cli/prisma-flow.md @@ -1,6 +1,6 @@ --- title: stack prisma flow -sidebar_position: 92 +sidebar_position: 93 --- # `stack prisma flow` diff --git a/docs/docs/cli/prisma-init.md b/docs/docs/cli/prisma-init.md index f93b2fb..6ed98a4 100644 --- a/docs/docs/cli/prisma-init.md +++ b/docs/docs/cli/prisma-init.md @@ -1,6 +1,6 @@ --- title: stack prisma init -sidebar_position: 86 +sidebar_position: 87 --- # `stack prisma init` diff --git a/docs/docs/cli/prisma-ping-llm.md b/docs/docs/cli/prisma-ping-llm.md index 9d161f3..195b9ad 100644 --- a/docs/docs/cli/prisma-ping-llm.md +++ b/docs/docs/cli/prisma-ping-llm.md @@ -1,6 +1,6 @@ --- title: stack prisma ping-llm -sidebar_position: 88 +sidebar_position: 89 --- # `stack prisma ping-llm` diff --git a/docs/docs/cli/prisma-run.md b/docs/docs/cli/prisma-run.md index 404bd45..00db0a0 100644 --- a/docs/docs/cli/prisma-run.md +++ b/docs/docs/cli/prisma-run.md @@ -1,6 +1,6 @@ --- title: stack prisma run -sidebar_position: 94 +sidebar_position: 95 --- # `stack prisma run` diff --git a/docs/docs/cli/prisma-screen.md b/docs/docs/cli/prisma-screen.md index 152cc1d..a0e9da0 100644 --- a/docs/docs/cli/prisma-screen.md +++ b/docs/docs/cli/prisma-screen.md @@ -1,6 +1,6 @@ --- title: stack prisma screen -sidebar_position: 89 +sidebar_position: 90 --- # `stack prisma screen` diff --git a/docs/docs/cli/prisma-vpn.md b/docs/docs/cli/prisma-vpn.md index 6b63da7..d9f9625 100644 --- a/docs/docs/cli/prisma-vpn.md +++ b/docs/docs/cli/prisma-vpn.md @@ -1,6 +1,6 @@ --- title: stack prisma vpn -sidebar_position: 95 +sidebar_position: 96 --- # `stack prisma vpn` diff --git a/docs/docs/cli/prisma.md b/docs/docs/cli/prisma.md index 42d5a5e..457471a 100644 --- a/docs/docs/cli/prisma.md +++ b/docs/docs/cli/prisma.md @@ -1,6 +1,6 @@ --- title: stack prisma -sidebar_position: 85 +sidebar_position: 86 --- # `stack prisma` diff --git a/docs/docs/cli/rec-list.md b/docs/docs/cli/rec-list.md index 455d9f4..313cf27 100644 --- a/docs/docs/cli/rec-list.md +++ b/docs/docs/cli/rec-list.md @@ -1,6 +1,6 @@ --- title: stack rec list -sidebar_position: 66 +sidebar_position: 67 --- # `stack rec list` diff --git a/docs/docs/cli/rec-opps.md b/docs/docs/cli/rec-opps.md index 073871c..da62a2c 100644 --- a/docs/docs/cli/rec-opps.md +++ b/docs/docs/cli/rec-opps.md @@ -1,6 +1,6 @@ --- title: stack rec opps -sidebar_position: 68 +sidebar_position: 69 --- # `stack rec opps` diff --git a/docs/docs/cli/rec-pfs.md b/docs/docs/cli/rec-pfs.md index f2d5ec9..3ba7b6f 100644 --- a/docs/docs/cli/rec-pfs.md +++ b/docs/docs/cli/rec-pfs.md @@ -1,6 +1,6 @@ --- title: stack rec pfs -sidebar_position: 67 +sidebar_position: 68 --- # `stack rec pfs` diff --git a/docs/docs/cli/rec.md b/docs/docs/cli/rec.md index 9efb9db..14ae46c 100644 --- a/docs/docs/cli/rec.md +++ b/docs/docs/cli/rec.md @@ -1,6 +1,6 @@ --- title: stack rec -sidebar_position: 65 +sidebar_position: 66 --- # `stack rec` diff --git a/docs/docs/cli/zot-dump-schema.md b/docs/docs/cli/zot-dump-schema.md index cc95b01..af6928d 100644 --- a/docs/docs/cli/zot-dump-schema.md +++ b/docs/docs/cli/zot-dump-schema.md @@ -1,6 +1,6 @@ --- title: stack zot dump-schema -sidebar_position: 80 +sidebar_position: 81 --- # `stack zot dump-schema` diff --git a/docs/docs/cli/zot-fix-dates.md b/docs/docs/cli/zot-fix-dates.md index 545efe9..5211d99 100644 --- a/docs/docs/cli/zot-fix-dates.md +++ b/docs/docs/cli/zot-fix-dates.md @@ -1,6 +1,6 @@ --- title: stack zot fix-dates -sidebar_position: 81 +sidebar_position: 82 --- # `stack zot fix-dates` diff --git a/docs/docs/cli/zot-fix-fields.md b/docs/docs/cli/zot-fix-fields.md index aba4715..396cef0 100644 --- a/docs/docs/cli/zot-fix-fields.md +++ b/docs/docs/cli/zot-fix-fields.md @@ -1,6 +1,6 @@ --- title: stack zot fix-fields -sidebar_position: 83 +sidebar_position: 84 --- # `stack zot fix-fields` diff --git a/docs/docs/cli/zot-fix-keys.md b/docs/docs/cli/zot-fix-keys.md index 67ae1af..120dbd0 100644 --- a/docs/docs/cli/zot-fix-keys.md +++ b/docs/docs/cli/zot-fix-keys.md @@ -1,6 +1,6 @@ --- title: stack zot fix-keys -sidebar_position: 82 +sidebar_position: 83 --- # `stack zot fix-keys` diff --git a/docs/docs/cli/zot-verify-parity.md b/docs/docs/cli/zot-verify-parity.md index 4fd4338..8637bcb 100644 --- a/docs/docs/cli/zot-verify-parity.md +++ b/docs/docs/cli/zot-verify-parity.md @@ -1,6 +1,6 @@ --- title: stack zot verify-parity -sidebar_position: 84 +sidebar_position: 85 --- # `stack zot verify-parity` diff --git a/docs/docs/cli/zot.md b/docs/docs/cli/zot.md index 190db74..87587ae 100644 --- a/docs/docs/cli/zot.md +++ b/docs/docs/cli/zot.md @@ -1,6 +1,6 @@ --- title: stack zot -sidebar_position: 79 +sidebar_position: 80 --- # `stack zot` diff --git a/src/bib/store.py b/src/bib/store.py index c73fa52..ee527db 100644 --- a/src/bib/store.py +++ b/src/bib/store.py @@ -613,6 +613,24 @@ class Store: con.commit() return added + def merge_extra(self, item_key: str, updates: dict[str, Any]) -> None: + """Merge *updates* into the item's ``extra_json`` (top-level keys + replace; everything else is kept). A missing item is a no-op.""" + con = self._con() + row = con.execute( + "SELECT extra_json FROM items WHERE key = ?", (item_key,) + ).fetchone() + if row is None: + return + current = json.loads(row[0] or "{}") if row[0] else {} + current.update(updates) + con.execute( + "UPDATE items SET extra_json = ?, " + "updated_at = strftime('%Y-%m-%dT%H:%M:%SZ','now') WHERE key = ?", + (json.dumps(current), item_key), + ) + con.commit() + def remove_tag(self, item_key: str, tag: str) -> None: """Remove a tag from an item.""" con = self._con() diff --git a/src/cli/llm.py b/src/cli/llm.py index 14fa0c9..5ebe3e4 100644 --- a/src/cli/llm.py +++ b/src/cli/llm.py @@ -1,5 +1,7 @@ """stack llm — local RAG over the library (index + fleet + chat serve).""" +from typing import Any + import typer app = typer.Typer(no_args_is_help=True) @@ -236,3 +238,103 @@ def vocab( typer.echo( f"vocabulary v{v.version}: {len(v)} themes, {len(v.retired)} retired ({v.source})" ) + + +def _tagging_runtime(cfg: Any) -> dict[str, Any]: + """Everything `stack llm tag` needs from the live stack — one place + the tests monkeypatch: the bib store, the pgvector chunk loader, the + theme-card vectors (embedded once per run through the pool), the + yes/no judge on the largest live host, and the model name recorded + in each item's state.""" + from conf.connect import bib + from llm.index import _engine + from llm.pool import HostPool, PoolEmbeddings, pick_model + from llm.tagger import card_vectors, make_judge, pg_chunks + from llm.vocab import load as load_vocab + + vocab = load_vocab() + pool = HostPool.from_config(cfg) + pool.check(cfg.embed_model) + cards = card_vectors(vocab, PoolEmbeddings(pool, cfg.embed_model).embed_documents) + judge = make_judge(cfg, pool) + with pool.acquire_generation() as host: + model = pick_model(cfg, pool, host) + return { + "store": bib(), + "vocab": vocab, + "load_chunks": pg_chunks(_engine(cfg)), + "card_vecs": cards, + "judge": judge, + "model": model, + } + + +@app.command() +def tag( + docket: str = typer.Option( + ..., "--docket", help="regulations.gov docket id, e.g. CMS-2026-2377." + ), + limit: int = typer.Option( + 0, "--limit", help="Stop after N comments (newest first; 0 = all)." + ), + force: bool = typer.Option( + False, "--force", help="Re-tag comments whose stored state is current." + ), + dry_run: bool = typer.Option( + False, "--dry-run", help="Compute and print, write nothing." + ), + top: int = typer.Option( + 8, "--top", help="Themes shortlisted per comment before judging." + ), + max_tags: int = typer.Option( + 4, "--max-tags", help="Most themes written per comment." + ), + min_confidence: float = typer.Option( + 0.0, + "--min-confidence", + help="Shortlist similarity below which a judged yes is not written.", + ), + verbose: bool = typer.Option( + False, "--verbose", help="Print every comment's tags." + ), +) -> None: + """Closed-vocabulary theme tagging of one docket's comments (P35): + shortlist by similarity to the theme cards, judge each candidate + yes/no on the largest live host with the scoring chunk as evidence, + record state in the item's extra_json, and replace its llm: tags. + Resumable — unchanged comments are skipped unless --force.""" + from llm import config as llm_config + from llm.tagger import run as run_tagging + + cfg = llm_config.load() + rt = _tagging_runtime(cfg) + + def progress(key: str, result: Any, why: str) -> None: + if result is None: + if verbose: + typer.echo(f"{key} {why}") + return + if verbose or dry_run: + typer.echo(f"{key} {', '.join(result.slugs) or '(none)'}") + + stats = run_tagging( + rt["store"], + vocab=rt["vocab"], + docket=docket, + load_chunks=rt["load_chunks"], + card_vecs=rt["card_vecs"], + judge=rt["judge"], + model=rt["model"], + limit=limit, + force=force, + dry_run=dry_run, + top=top, + max_tags=max_tags, + min_confidence=min_confidence, + progress=progress, + ) + typer.echo( + f"tag {docket} (vocab v{rt['vocab'].version}, {rt['model']}{', dry run' if dry_run else ''}): " + f"seen={stats['seen']} tagged={stats['tagged']} unchanged={stats['skipped_state']} " + f"no_chunks={stats['no_chunks']} tags_written={stats['tags_written']}" + ) diff --git a/src/llm/tagger.py b/src/llm/tagger.py new file mode 100644 index 0000000..1bfa3bf --- /dev/null +++ b/src/llm/tagger.py @@ -0,0 +1,369 @@ +"""Closed-vocabulary theme tagging of rulemaking comments (P35, #575/#576). + +For one comment the chain is: its already-indexed chunks (text + vector, +``langchain_pg_embedding`` ``comments`` collection) → cosine similarity +against every theme card (``llm.vocab.Theme.card`` embedded once per run) +→ the ``top`` themes shortlisted, each with the chunk that scored it → +one yes/no judgement per candidate on the largest live host, with that +chunk as the evidence → up to ``max_tags`` accepted themes, confidence = +the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``. + +State lives on the bib item (``extra_json["llm_tags"]``: vocab version, +model, content hash, the accepted tags with confidence and evidence +chunk, and the shortlist with every verdict) so a run is resumable and a +re-run skips comments whose version + model + content are unchanged. +Write-back replaces exactly the item's ``llm:*`` tags and never touches +hand tags. + +Every collaborator is injected (``embed``, ``judge``, ``load_chunks``) so +the pure pieces run in tests without Ollama or pgvector; the CLI wires +the real ones (``make_judge``, ``card_vectors``, ``pg_chunks``). +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import math +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from typing import Any, Callable, Iterable, Mapping, Sequence + +from llm.vocab import Theme, Vocab, split_tags + +log = logging.getLogger(__name__) + +Vector = Sequence[float] +Embed = Callable[[Sequence[str]], list[list[float]]] +Judge = Callable[[Theme, str], bool | None] +ChunkLoader = Callable[[str], list[tuple[int, str, list[float]]]] + +STATE_KEY = "llm_tags" + + +@dataclass(frozen=True) +class Candidate: + slug: str + score: float + chunk_idx: int + + +@dataclass(frozen=True) +class TagInfo: + confidence: float + evidence_chunk: int + + +@dataclass(frozen=True) +class TagResult: + key: str + tags: dict[str, TagInfo] + shortlisted: tuple[Candidate, ...] + judged: dict[str, bool | None] + content_hash: str + + @property + def slugs(self) -> list[str]: + return sorted(self.tags, key=lambda s: -self.tags[s].confidence) + + +# ── pure pieces ─────────────────────────────────────────────────────── + + +def cosine(a: Vector, b: Vector) -> float: + dot = sum(x * y for x, y in zip(a, b)) + na = math.sqrt(sum(x * x for x in a)) + nb = math.sqrt(sum(y * y for y in b)) + return dot / (na * nb) if na and nb else 0.0 + + +def shortlist( + chunk_vecs: Sequence[Vector], card_vecs: Mapping[str, Vector], *, top: int = 8 +) -> list[Candidate]: + """The ``top`` themes by their best cosine similarity to any chunk, + each with the index of the chunk that scored it (the evidence).""" + if not chunk_vecs: + return [] + out: list[Candidate] = [] + for slug, cv in card_vecs.items(): + best_i, best = 0, -1.0 + for i, v in enumerate(chunk_vecs): + s = cosine(v, cv) + if s > best: + best_i, best = i, s + out.append(Candidate(slug, round(best, 4), best_i)) + out.sort(key=lambda c: (-c.score, c.slug)) + return out[:top] + + +_SYSTEM = ( + "You decide whether a public comment letter on a Medicare physician fee " + "schedule rule addresses a given theme. Answer with exactly one word: yes " + "or no. Say yes only when the excerpt actually discusses the theme, not " + "when it merely mentions a related word in passing." +) + + +def judge_prompt(theme: Theme, evidence: str) -> list[dict]: + return [ + {"role": "system", "content": _SYSTEM}, + { + "role": "user", + "content": ( + f"Theme: {theme.label}\nDefinition: {theme.definition}\n" + f"Phrases letters use: {', '.join(theme.synonyms) or '—'}\n\n" + f"Excerpt from the comment:\n{evidence.strip()}\n\n" + "Does this comment address the theme? Answer yes or no." + ), + }, + ] + + +def parse_yes_no(reply: str) -> bool | None: + word = (reply or "").strip().strip(".!\"'").lower().split() + if not word: + return None + if word[0] in ("yes", "y"): + return True + if word[0] in ("no", "n"): + return False + return None + + +def content_hash(chunks: Sequence[str]) -> str: + h = hashlib.sha1() # noqa: S324 — change detection, not security + for c in chunks: + h.update(c.encode("utf-8", "replace")) + h.update(b"\x00") + return h.hexdigest()[:16] + + +def tag_comment( + key: str, + chunks: Sequence[str], + chunk_vecs: Sequence[Vector], + *, + vocab: Vocab, + card_vecs: Mapping[str, Vector], + judge: Judge, + top: int = 8, + max_tags: int = 4, + min_confidence: float = 0.0, +) -> TagResult: + cands = shortlist(chunk_vecs, card_vecs, top=top) + judged: dict[str, bool | None] = {} + tags: dict[str, TagInfo] = {} + for c in cands: + if c.slug not in vocab: + continue + verdict = judge(vocab.get(c.slug), chunks[c.chunk_idx]) + judged[c.slug] = verdict + if verdict and c.score >= min_confidence and len(tags) < max_tags: + tags[c.slug] = TagInfo(confidence=c.score, evidence_chunk=c.chunk_idx) + return TagResult(key, tags, tuple(cands), judged, content_hash(chunks)) + + +def state_matches( + extra: Mapping[str, Any], *, version: int, model: str, digest: str +) -> bool: + st = extra.get(STATE_KEY) or {} + return ( + st.get("version") == version + and st.get("model") == model + and st.get("content_hash") == digest + ) + + +def state_payload(result: TagResult, *, version: int, model: str) -> dict: + return { + "version": version, + "model": model, + "content_hash": result.content_hash, + "tagged_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "tags": {s: asdict(i) for s, i in result.tags.items()}, + "shortlist": [ + { + "slug": c.slug, + "score": c.score, + "chunk": c.chunk_idx, + "verdict": result.judged.get(c.slug), + } + for c in result.shortlisted + ], + } + + +# ── bib side ────────────────────────────────────────────────────────── + + +def apply_tags( + store: Any, key: str, slugs: Iterable[str], *, vocab: Vocab +) -> tuple[int, int]: + """Make the item's ``llm:*`` tags exactly ``slugs``; hand tags untouched. + Returns ``(added, removed)``.""" + want = {vocab.tag(s) for s in slugs} + have, _hand = split_tags(store.get(key).tags) + removed = 0 + for t in have: + if t not in want: + store.remove_tag(key, t) + removed += 1 + added = store.add_tags(key, sorted(want - set(have))) if want - set(have) else 0 + return added, removed + + +def _extra(store: Any, key: str) -> dict: + row = ( + store._con() + .execute( # noqa: SLF001 + "SELECT extra_json FROM items WHERE key = ?", (key,) + ) + .fetchone() + ) + return json.loads(row[0] or "{}") if row and row[0] else {} + + +def docket_keys(store: Any, docket: str, *, limit: int = 0) -> list[str]: + """Comment item keys in *docket*, newest posted first.""" + con = store._con() # noqa: SLF001 + sql = ( + "SELECT i.key FROM items i " + "WHERE i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN " + "(SELECT id FROM tags WHERE name = ?)) " + "AND i.url LIKE 'https://www.regulations.gov/comment/%' " + "ORDER BY i.date_published DESC, i.id DESC" + ) + if limit: + sql += f" LIMIT {int(limit)}" + return [r[0] for r in con.execute(sql, (f"reg-docket:{docket}",)).fetchall()] + + +def run( + store: Any, + *, + vocab: Vocab, + docket: str, + load_chunks: ChunkLoader, + card_vecs: Mapping[str, Vector], + judge: Judge, + model: str, + limit: int = 0, + force: bool = False, + dry_run: bool = False, + top: int = 8, + max_tags: int = 4, + min_confidence: float = 0.0, + progress: Callable[[str, TagResult | None, str], None] | None = None, +) -> dict[str, int]: + """Tag every comment in *docket* (newest first); resumable via the + item's stored state; ``dry_run`` computes and reports but writes + nothing.""" + stats = { + "seen": 0, + "tagged": 0, + "skipped_state": 0, + "no_chunks": 0, + "tags_written": 0, + } + for key in docket_keys(store, docket, limit=limit): + stats["seen"] += 1 + extra = _extra(store, key) + rows = load_chunks(key) + if not rows: + stats["no_chunks"] += 1 + if progress: + progress(key, None, "no-chunks") + continue + rows = sorted(rows, key=lambda r: r[0]) + texts = [t for _, t, _ in rows] + digest = content_hash(texts) + if not force and state_matches( + extra, version=vocab.version, model=model, digest=digest + ): + stats["skipped_state"] += 1 + if progress: + progress(key, None, "unchanged") + continue + result = tag_comment( + key, + texts, + [v for _, _, v in rows], + vocab=vocab, + card_vecs=card_vecs, + judge=judge, + top=top, + max_tags=max_tags, + min_confidence=min_confidence, + ) + stats["tagged"] += 1 + if not dry_run: + store.merge_extra( + key, + {STATE_KEY: state_payload(result, version=vocab.version, model=model)}, + ) + added, _removed = apply_tags(store, key, result.slugs, vocab=vocab) + stats["tags_written"] += added + if progress: + progress(key, result, "tagged") + return stats + + +# ── real collaborators (the CLI wires these) ────────────────────────── + + +def card_vectors(vocab: Vocab, embed: Embed) -> dict[str, list[float]]: + slugs = list(vocab.cards) + vecs = embed([vocab.cards[s] for s in slugs]) + return dict(zip(slugs, vecs)) + + +def make_judge( + cfg: Any, pool: Any, *, post: Callable[..., dict] | None = None +) -> Judge: + """A yes/no judge on the largest live host, ``temperature 0`` — the + same route ``llm.classify.closed_vocab_classifier`` takes.""" + from llm.classify import _default_post + from llm.pool import pick_model + + send = post or _default_post + pool.check(cfg.instruct_model) + + def judge(theme: Theme, evidence: str) -> bool | None: + with pool.acquire_generation() as host: + model = pick_model(cfg, pool, host) + data = send( + f"{host}/api/chat", + json={ + "model": model, + "messages": judge_prompt(theme, evidence), + "stream": False, + "options": {"num_ctx": cfg.chat_num_ctx, "temperature": 0}, + }, + ) + reply = data.get("message", {}).get("content", "") + verdict = parse_yes_no(reply) + if verdict is None: + log.info("judge: unparseable reply %r for %s", reply[:60], theme.slug) + return verdict + + return judge + + +def pg_chunks(engine: Any, collection: str = "comments") -> ChunkLoader: + """``load_chunks(key)`` over ``langchain_pg_embedding`` — the text and + vector of every chunk indexed for the item, in ``seq`` order.""" + from sqlalchemy import text as _text + + sql = _text( + "SELECT (e.cmetadata->>'seq')::int AS seq, e.document, e.embedding::text " + "FROM langchain_pg_embedding e JOIN langchain_pg_collection c ON c.uuid = e.collection_id " + "WHERE c.name = :collection AND e.cmetadata->>'item_key' = :key ORDER BY seq" + ) + + def load(key: str) -> list[tuple[int, str, list[float]]]: + with engine.begin() as con: + rows = con.execute(sql, {"collection": collection, "key": key}).fetchall() + return [(int(seq or 0), doc or "", json.loads(vec)) for seq, doc, vec in rows] + + return load diff --git a/tests/cli/test_llm_tag_cli.py b/tests/cli/test_llm_tag_cli.py new file mode 100644 index 0000000..14193ce --- /dev/null +++ b/tests/cli/test_llm_tag_cli.py @@ -0,0 +1,103 @@ +"""stack llm tag / vocab (#574, #575).""" + +from __future__ import annotations + +import pytest +from typer.testing import CliRunner + +import cli.llm as llm_cli +from bib.item import Source +from bib.store import Store +from cli import app +from llm.vocab import parse + +runner = CliRunner() + +VOCAB = parse( + "version: 1\nthemes:\n - slug: telehealth\n definition: Telehealth.\n synonyms: [telehealth]\n" + " - slug: drugs\n definition: Drugs.\n synonyms: [ASP]\n" +) + + +@pytest.fixture +def rt(monkeypatch, tmp_path): + store = Store(":memory:", storage_dir=tmp_path / "st") + key = store.upsert( + Source( + title="c1", + url="https://www.regulations.gov/comment/CMS-2026-2377-1", + date_published="2026-09-01", + ), + tags=["reg-docket:CMS-2026-2377", "source:regulations-gov"], + ) + judged = [] + + def judge(theme, evidence): + judged.append(theme.slug) + return theme.slug == "telehealth" + + runtime = { + "store": store, + "vocab": VOCAB, + "load_chunks": lambda k: ( + [(0, "telehealth text", [1.0, 0.0])] if k == key else [] + ), + "card_vecs": {"telehealth": [1.0, 0.0], "drugs": [0.0, 1.0]}, + "judge": judge, + "model": "m", + } + monkeypatch.setattr(llm_cli, "_tagging_runtime", lambda cfg: runtime) + monkeypatch.setattr("llm.config.load", lambda: object()) + try: + yield store, key, judged + finally: + store.close() + + +class TestTag: + def test_tags_and_reports(self, rt): + store, key, judged = rt + res = runner.invoke( + app, ["llm", "tag", "--docket", "CMS-2026-2377", "--verbose"] + ) + assert res.exit_code == 0, res.output + assert f"{key} telehealth" in res.output + assert "seen=1 tagged=1 unchanged=0 no_chunks=0 tags_written=1" in res.output + assert ( + "llm:telehealth" in store.get(key).tags + and "llm:drugs" not in store.get(key).tags + ) + assert judged == ["telehealth", "drugs"] + res = runner.invoke(app, ["llm", "tag", "--docket", "CMS-2026-2377"]) + assert "unchanged=1" in res.output + + def test_dry_run_prints_but_writes_nothing(self, rt): + store, key, _ = rt + res = runner.invoke( + app, ["llm", "tag", "--docket", "CMS-2026-2377", "--dry-run"] + ) + assert res.exit_code == 0, res.output + assert "dry run" in res.output and f"{key} telehealth" in res.output + assert not any(t.startswith("llm:") for t in store.get(key).tags) + + def test_docket_is_required(self): + assert runner.invoke(app, ["llm", "tag"]).exit_code != 0 + + +class TestVocab: + def test_lists_and_shows(self): + res = runner.invoke(app, ["llm", "vocab"]) + assert ( + res.exit_code == 0 + and "vocabulary v" in res.output + and "telehealth" in res.output + ) + res = runner.invoke(app, ["llm", "vocab", "--slug", "telehealth"]) + assert res.exit_code == 0 and "llm:telehealth" in res.output + assert runner.invoke(app, ["llm", "vocab", "--slug", "nope"]).exit_code != 0 + + def test_invalid_file(self, tmp_path): + p = tmp_path / "bad.yaml" + p.write_text("version: 1\nthemes: []\n") + res = runner.invoke(app, ["llm", "vocab", "--path", str(p)]) + assert res.exit_code == 1 and "invalid vocabulary" in res.output diff --git a/tests/llm/test_tagger.py b/tests/llm/test_tagger.py new file mode 100644 index 0000000..31962fb --- /dev/null +++ b/tests/llm/test_tagger.py @@ -0,0 +1,390 @@ +"""llm.tagger — shortlist, judge, state, write-back and the docket runner (#575/#576).""" + +from __future__ import annotations + +import json + +import pytest + +from bib.item import Source +from bib.store import Store +from llm.tagger import ( + Candidate, + TagResult, + apply_tags, + card_vectors, + content_hash, + cosine, + docket_keys, + judge_prompt, + make_judge, + parse_yes_no, + run, + shortlist, + state_matches, + state_payload, + tag_comment, +) +from llm.vocab import parse + +VOCAB = parse( + """ +version: 2 +themes: + - slug: telehealth + label: Telehealth + definition: Medicare telehealth services. + synonyms: [telehealth] + - slug: care-management + label: Care management + definition: Monthly care management codes. + synonyms: [CCM] + - slug: drugs + label: Part B drugs + definition: ASP drug payment. + synonyms: [ASP] +""" +) +CARDS = { + "telehealth": [1.0, 0.0, 0.0], + "care-management": [0.0, 1.0, 0.0], + "drugs": [0.0, 0.0, 1.0], +} + + +class TestPure: + def test_cosine(self): + assert cosine([1, 0], [1, 0]) == pytest.approx(1.0) + assert cosine([1, 0], [0, 1]) == pytest.approx(0.0) + assert cosine([0, 0], [1, 1]) == 0.0 + + def test_shortlist_best_chunk_per_theme(self): + chunks = [[0.9, 0.1, 0.0], [0.1, 0.9, 0.0]] + out = shortlist(chunks, CARDS, top=2) + assert [c.slug for c in out] == ["care-management", "telehealth"] or [ + c.slug for c in out + ] == ["telehealth", "care-management"] + by = {c.slug: c for c in out} + assert by["telehealth"].chunk_idx == 0 and by["care-management"].chunk_idx == 1 + assert all(isinstance(c, Candidate) and 0 < c.score <= 1 for c in out) + assert shortlist([], CARDS) == [] + + def test_judge_prompt_and_parse(self): + msgs = judge_prompt( + VOCAB.get("telehealth"), "We support audio-only telehealth." + ) + assert msgs[0]["role"] == "system" and "Telehealth" in msgs[1]["content"] + assert "audio-only telehealth" in msgs[1]["content"] + assert parse_yes_no("Yes.") is True and parse_yes_no(" no\n") is False + assert parse_yes_no("Maybe yes") is None and parse_yes_no("") is None + + def test_content_hash_is_order_sensitive(self): + assert content_hash(["a", "b"]) != content_hash(["b", "a"]) + assert len(content_hash(["a"])) == 16 + + def test_tag_comment_accepts_judged_yes_up_to_max(self): + chunks = ["telehealth text", "ccm text", "asp text"] + vecs = [[1.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]] + seen = [] + + def judge(theme, evidence): + seen.append((theme.slug, evidence)) + return theme.slug != "drugs" + + r = tag_comment( + "K", + chunks, + vecs, + vocab=VOCAB, + card_vecs=CARDS, + judge=judge, + top=3, + max_tags=1, + ) + assert isinstance(r, TagResult) and len(r.tags) == 1 + assert r.judged["drugs"] is False and any(v for v in r.judged.values()) + assert ( + "telehealth", + "telehealth text", + ) in seen # evidence is the scoring chunk + assert r.content_hash == content_hash(chunks) + + def test_min_confidence_filters_weak_matches(self): + chunks = ["x"] + vecs = [[0.6, 0.8, 0.0]] # cosine 0.6 with telehealth, 0.8 with care-management + r = tag_comment( + "K", + chunks, + vecs, + vocab=VOCAB, + card_vecs=CARDS, + judge=lambda t, e: True, + min_confidence=0.7, + ) + assert r.slugs == ["care-management"] + + def test_state(self): + r = tag_comment( + "K", + ["t"], + [[1.0, 0, 0]], + vocab=VOCAB, + card_vecs=CARDS, + judge=lambda t, e: True, + max_tags=1, + ) + payload = state_payload(r, version=2, model="m") + assert payload["tags"] == { + "telehealth": {"confidence": 1.0, "evidence_chunk": 0} + } + assert payload["shortlist"][0]["verdict"] is True and "tagged_at" in payload + assert state_matches( + {"llm_tags": payload}, version=2, model="m", digest=r.content_hash + ) + assert not state_matches( + {"llm_tags": payload}, version=3, model="m", digest=r.content_hash + ) + assert not state_matches({}, version=2, model="m", digest=r.content_hash) + + def test_card_vectors_embeds_every_card_once(self): + calls = [] + + def embed(texts): + calls.append(list(texts)) + return [[float(i)] for i in range(len(texts))] + + cv = card_vectors(VOCAB, embed) + assert set(cv) == {"telehealth", "care-management", "drugs"} and len(calls) == 1 + assert "Medicare telehealth services" in calls[0][0] + + +class TestJudge: + def test_make_judge_posts_to_largest_host_at_temperature_zero(self): + class Pool: + def check(self, model): + return ["http://h"] + + def vram(self, host): + return 0.0 + + def serves(self, host, model): + return False + + class _cm: + def __enter__(self): + return "http://h" + + def __exit__(self, *a): + return False + + def acquire_generation(self): + return self._cm() + + class Cfg: + instruct_model = "m" + instruct_model_large = "" + large_min_vram_gb = 20.0 + chat_num_ctx = 4096 + + sent = {} + + def post(url, json=None): + sent["url"], sent["json"] = url, json + return {"message": {"content": "Yes"}} + + judge = make_judge(Cfg(), Pool(), post=post) + assert judge(VOCAB.get("drugs"), "ASP add-on") is True + assert ( + sent["url"] == "http://h/api/chat" + and sent["json"]["options"]["temperature"] == 0 + ) + assert sent["json"]["model"] == "m" + + +def _store(tmp_path): + s = Store(":memory:", storage_dir=tmp_path / "st") + keys = {} + for i, (cid, date) in enumerate( + [ + ("CMS-2026-2377-1", "2026-09-01"), + ("CMS-2026-2377-2", "2026-09-05"), + ("CMS-2025-0304-9", "2025-09-01"), + ] + ): + docket = cid.rsplit("-", 1)[0] + keys[cid] = s.upsert( + Source( + title=cid, + url=f"https://www.regulations.gov/comment/{cid}", + date_published=date, + ), + tags=[f"reg-docket:{docket}", "source:regulations-gov", "topic:palliative"], + ) + return s, keys + + +class TestBibSide: + def test_docket_keys_newest_first(self, tmp_path): + s, keys = _store(tmp_path) + assert docket_keys(s, "CMS-2026-2377") == [ + keys["CMS-2026-2377-2"], + keys["CMS-2026-2377-1"], + ] + assert docket_keys(s, "CMS-2026-2377", limit=1) == [keys["CMS-2026-2377-2"]] + assert docket_keys(s, "NOPE") == [] + s.close() + + def test_apply_tags_replaces_only_llm_tags(self, tmp_path): + s, keys = _store(tmp_path) + k = keys["CMS-2026-2377-1"] + s.add_tags(k, ["llm:drugs"]) + added, removed = apply_tags( + s, k, ["telehealth", "care-management"], vocab=VOCAB + ) + assert (added, removed) == (2, 1) + tags = set(s.get(k).tags) + assert { + "llm:telehealth", + "llm:care-management", + "topic:palliative", + "reg-docket:CMS-2026-2377", + } <= tags + assert "llm:drugs" not in tags + assert apply_tags(s, k, ["telehealth", "care-management"], vocab=VOCAB) == ( + 0, + 0, + ) # idempotent + assert ( + apply_tags(s, k, [], vocab=VOCAB) == (0, 2) + and "topic:palliative" in s.get(k).tags + ) + s.close() + + def test_merge_extra(self, tmp_path): + s, keys = _store(tmp_path) + k = keys["CMS-2026-2377-1"] + s.merge_extra(k, {"a": 1}) + s.merge_extra(k, {"b": {"x": 2}}) + s.merge_extra("ZZZZZZZZ", {"c": 3}) # unknown key: no-op, no error + extra = json.loads( + s._con() + .execute("SELECT extra_json FROM items WHERE key=?", (k,)) + .fetchone()[0] + ) # noqa: SLF001 + assert extra["a"] == 1 and extra["b"] == {"x": 2} + s.close() + + +class TestRun: + def _chunks(self, keys): + def load(key): + if key == keys["CMS-2026-2377-1"]: + return [ + (1, "ccm text", [0.0, 1.0, 0.0]), + (0, "telehealth text", [1.0, 0.0, 0.0]), + ] + if key == keys["CMS-2026-2377-2"]: + return [] # not indexed yet + return [(0, "asp", [0.0, 0.0, 1.0])] + + return load + + def test_tags_persist_state_and_resume(self, tmp_path): + s, keys = _store(tmp_path) + events = [] + judge_calls = [] + + def judge(theme, evidence): + judge_calls.append(theme.slug) + return True + + stats = run( + s, + vocab=VOCAB, + docket="CMS-2026-2377", + load_chunks=self._chunks(keys), + card_vecs=CARDS, + judge=judge, + model="m", + top=2, + max_tags=2, + progress=lambda k, r, why: events.append((k, why)), + ) + assert stats == { + "seen": 2, + "tagged": 1, + "skipped_state": 0, + "no_chunks": 1, + "tags_written": 2, + } + k = keys["CMS-2026-2377-1"] + assert {"llm:telehealth", "llm:care-management"} <= set(s.get(k).tags) + extra = json.loads( + s._con() + .execute("SELECT extra_json FROM items WHERE key=?", (k,)) + .fetchone()[0] + ) # noqa: SLF001 + assert extra["llm_tags"]["version"] == 2 and extra["llm_tags"]["model"] == "m" + assert set(extra["llm_tags"]["tags"]) == {"telehealth", "care-management"} + assert (keys["CMS-2026-2377-2"], "no-chunks") in events + # second run: unchanged → skipped without judging + judge_calls.clear() + stats2 = run( + s, + vocab=VOCAB, + docket="CMS-2026-2377", + load_chunks=self._chunks(keys), + card_vecs=CARDS, + judge=judge, + model="m", + top=2, + ) + assert ( + stats2["skipped_state"] == 1 and stats2["tagged"] == 0 and judge_calls == [] + ) + # a new model re-tags; --force re-tags + assert ( + run( + s, + vocab=VOCAB, + docket="CMS-2026-2377", + load_chunks=self._chunks(keys), + card_vecs=CARDS, + judge=judge, + model="m2", + top=2, + )["tagged"] + == 1 + ) + assert ( + run( + s, + vocab=VOCAB, + docket="CMS-2026-2377", + load_chunks=self._chunks(keys), + card_vecs=CARDS, + judge=judge, + model="m2", + top=2, + force=True, + )["tagged"] + == 1 + ) + s.close() + + def test_dry_run_writes_nothing(self, tmp_path): + s, keys = _store(tmp_path) + stats = run( + s, + vocab=VOCAB, + docket="CMS-2026-2377", + load_chunks=self._chunks(keys), + card_vecs=CARDS, + judge=lambda t, e: True, + model="m", + dry_run=True, + ) + assert stats["tagged"] == 1 and stats["tags_written"] == 0 + k = keys["CMS-2026-2377-1"] + assert not any(t.startswith("llm:") for t in s.get(k).tags) + s.close()