merge: #574/#575 — P35 theme vocabulary + tagging chain, runner and write-back (refs #574, #575, #576)
All checks were successful
CI / lint (push) Successful in 33s
CI / notebooks-smoke (push) Successful in 1m48s
CI / test (push) Successful in 2m26s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 16s
Infra CI / notebooks (push) Successful in 51s
Infra CI / api (push) Successful in 1m10s
Infra CI / docs (push) Successful in 1m43s
Infra CI / mc (push) Successful in 16s
Deploy / report (push) Successful in 13s
Infra CI / llm (push) Successful in 44s

This commit is contained in:
kert
2026-09-22 16:54:59 -04:00
66 changed files with 1804 additions and 53 deletions

View File

@@ -0,0 +1,81 @@
"""Seed a theme-vocabulary revision from Federal Register section headings (#574).
Walks ``fr_anchors`` for heading-like paragraphs ("B. Determination of
Practice Expense…") in every PFS rule whose rule year is at or after
``--since`` (default 2018 — the regulations.gov comment dockets start in
2017), stems the titles, and prints the stems ordered by how many rule
years carry them, with an example title each. The output is raw material
for a human editing ``src/llm/vocab/themes.yaml`` — it is not the vocab.
Usage:
uv run python dev/scripts/llm_vocab_candidates.py [--since 2018] [--top 200]
"""
from __future__ import annotations
import argparse
import re
from collections import defaultdict
_HEADING = re.compile(r"^(?:[IVX]{1,4}|[A-Z]|\d{1,2}|[a-z])\. +([A-Z][^.]{6,120})$")
_STOP = re.compile(
r"\b(cy|20\d\d|proposed|final|rule|for|the|of|and|to|in|a|an|under|pfs|physician fee schedule)\b"
)
_SKIP = ("authority citation", "on page", "column", "paragraph")
def stem(title: str) -> str:
norm = re.sub(r"[^a-z0-9 ]", " ", title.lower())
norm = _STOP.sub(" ", norm)
return re.sub(r"\s+", " ", norm).strip()
def candidates(store, *, since: int) -> list[tuple[int, str, str]]:
"""``(n_rule_years, stem, example title)`` sorted by n desc."""
from pfs.descriptors import rule_year_of
con = store._con() # noqa: SLF001
years: dict[str, int] = {}
for key, title, published in con.execute(
"SELECT key, title, date_published FROM items WHERE item_type = 'rule'"
).fetchall():
# rule_year_of reads the payment year from the title, else falls
# back to publication year + 1 (PFS rules publish July–December).
years[key] = int(rule_year_of(title or "", published or ""))
recent = {k for k, y in years.items() if y and y >= since}
by: dict[str, set[int]] = defaultdict(set)
example: dict[str, str] = {}
for key, text in con.execute(
"SELECT item_key, text FROM fr_anchors WHERE length(text) < 140"
).fetchall():
if key not in recent:
continue
m = _HEADING.match((text or "").strip())
if not m:
continue
title = m.group(1).strip()
s = stem(title)
if len(s) < 5 or any(s.startswith(x) or x in s for x in _SKIP):
continue
by[s].add(years[key])
example.setdefault(s, title)
return sorted(
((len(v), k, example[k]) for k, v in by.items()), key=lambda r: (-r[0], r[1])
)
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
ap.add_argument("--since", type=int, default=2018)
ap.add_argument("--top", type=int, default=200)
args = ap.parse_args()
from conf import connect
store = connect.bib()
for n, s, ex in candidates(store, since=args.since)[: args.top]:
print(f"{n:3d} {ex}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -1,6 +1,6 @@
---
title: stack api serve
sidebar_position: 61
sidebar_position: 63
---
# `stack api serve`

View File

@@ -1,6 +1,6 @@
---
title: stack api
sidebar_position: 60
sidebar_position: 62
---
# `stack api`

View File

@@ -1,6 +1,6 @@
---
title: stack db comment
sidebar_position: 54
sidebar_position: 56
---
# `stack db comment`

View File

@@ -1,6 +1,6 @@
---
title: stack db inspect
sidebar_position: 55
sidebar_position: 57
---
# `stack db inspect`

View File

@@ -1,6 +1,6 @@
---
title: stack db
sidebar_position: 53
sidebar_position: 55
---
# `stack db`

View File

@@ -1,6 +1,6 @@
---
title: stack docs build
sidebar_position: 57
sidebar_position: 59
---
# `stack docs build`

View File

@@ -1,6 +1,6 @@
---
title: stack docs generate
sidebar_position: 59
sidebar_position: 61
---
# `stack docs generate`

View File

@@ -1,6 +1,6 @@
---
title: stack docs serve
sidebar_position: 58
sidebar_position: 60
---
# `stack docs serve`

View File

@@ -1,6 +1,6 @@
---
title: stack docs
sidebar_position: 56
sidebar_position: 58
---
# `stack docs`

38
docs/docs/cli/llm-tag.md Normal file
View File

@@ -0,0 +1,38 @@
---
title: stack llm tag
sidebar_position: 54
---
# `stack llm tag`
```
Usage: stack llm tag [OPTIONS]
Closed-vocabulary theme tagging of one docket's comments (P35): shortlist by
similarity to the theme cards, judge each candidate yes/no on the largest live
host with the scoring chunk as evidence, record state in the item's
extra_json, and replace its llm: tags. Resumable — unchanged comments are
skipped unless --force.
╭─ Options ────────────────────────────────────────────────────────────────────╮
│ * --docket TEXT regulations.gov docket id, e.g. │
│ CMS-2026-2377. │
│ [required] │
│ --limit INTEGER Stop after N comments (newest first; 0 = │
│ all). │
│ [default: 0] │
│ --force Re-tag comments whose stored state is │
│ current. │
│ --dry-run Compute and print, write nothing. │
│ --top INTEGER Themes shortlisted per comment before │
│ judging. │
│ [default: 8] │
│ --max-tags INTEGER Most themes written per comment. │
│ [default: 4] │
│ --min-confidence FLOAT Shortlist similarity below which a │
│ judged yes is not written. │
│ [default: 0.0] │
│ --verbose Print every comment's tags. │
│ --help Show this message and exit. │
╰──────────────────────────────────────────────────────────────────────────────╯
```

View File

@@ -0,0 +1,21 @@
---
title: stack llm vocab
sidebar_position: 53
---
# `stack llm vocab`
```
Usage: stack llm vocab [OPTIONS]
The closed theme vocabulary for comment tagging (#574): validate it and list
its slugs, or show one theme's definition, synonyms and the FR section stems
it was seeded from.
╭─ Options ────────────────────────────────────────────────────────────────────╮
│ --path TEXT A vocabulary file to validate instead of the packaged │
│ one. │
│ --slug TEXT Show one theme in full. │
│ --help Show this message and exit. │
╰──────────────────────────────────────────────────────────────────────────────╯
```

View File

@@ -19,5 +19,14 @@ Usage: stack llm [OPTIONS] COMMAND [ARGS]...
│ hosts Show the Ollama fleet: declared VRAM, liveness, models, and which │
│ host + model would answer a chat right now. │
│ serve Serve the SSO-guarded chat UI (llm.api:app). │
│ vocab The closed theme vocabulary for comment tagging (#574): validate it │
│ and list its slugs, or show one theme's definition, synonyms and │
│ the FR │
│ section stems it was seeded from. │
│ tag Closed-vocabulary theme tagging of one docket's comments (P35): │
│ shortlist by similarity to the theme cards, judge each candidate │
│ yes/no on the largest live host with the scoring chunk as evidence, │
│ record state in the item's extra_json, and replace its llm: tags. │
│ Resumable — unchanged comments are skipped unless --force. │
╰──────────────────────────────────────────────────────────────────────────────╯
```

View File

@@ -1,6 +1,6 @@
---
title: stack mail attach-smarthost
sidebar_position: 102
sidebar_position: 104
---
# `stack mail attach-smarthost`

View File

@@ -1,6 +1,6 @@
---
title: stack mail dkim-export
sidebar_position: 101
sidebar_position: 103
---
# `stack mail dkim-export`

View File

@@ -1,6 +1,6 @@
---
title: stack mail dns
sidebar_position: 100
sidebar_position: 102
---
# `stack mail dns`

View File

@@ -1,6 +1,6 @@
---
title: stack mail down
sidebar_position: 98
sidebar_position: 100
---
# `stack mail down`

View File

@@ -1,6 +1,6 @@
---
title: stack mail provision
sidebar_position: 96
sidebar_position: 98
---
# `stack mail provision`

View File

@@ -1,6 +1,6 @@
---
title: stack mail rotate-creds
sidebar_position: 103
sidebar_position: 105
---
# `stack mail rotate-creds`

View File

@@ -1,6 +1,6 @@
---
title: stack mail seed-mailboxes
sidebar_position: 104
sidebar_position: 106
---
# `stack mail seed-mailboxes`

View File

@@ -1,6 +1,6 @@
---
title: stack mail status
sidebar_position: 99
sidebar_position: 101
---
# `stack mail status`

View File

@@ -1,6 +1,6 @@
---
title: stack mail up
sidebar_position: 97
sidebar_position: 99
---
# `stack mail up`

View File

@@ -1,6 +1,6 @@
---
title: stack mail wire-git
sidebar_position: 105
sidebar_position: 107
---
# `stack mail wire-git`

View File

@@ -1,6 +1,6 @@
---
title: stack mail
sidebar_position: 95
sidebar_position: 97
---
# `stack mail`

View File

@@ -1,6 +1,6 @@
---
title: stack perf show
sidebar_position: 63
sidebar_position: 65
---
# `stack perf show`

View File

@@ -1,6 +1,6 @@
---
title: stack perf
sidebar_position: 62
sidebar_position: 64
---
# `stack perf`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs cpt-ingest
sidebar_position: 77
sidebar_position: 79
---
# `stack pfs cpt-ingest`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs elements
sidebar_position: 69
sidebar_position: 71
---
# `stack pfs elements`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs exposure
sidebar_position: 74
sidebar_position: 76
---
# `stack pfs exposure`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs families
sidebar_position: 71
sidebar_position: 73
---
# `stack pfs families`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs guidance
sidebar_position: 72
sidebar_position: 74
---
# `stack pfs guidance`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs lineage
sidebar_position: 70
sidebar_position: 72
---
# `stack pfs lineage`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs reaction
sidebar_position: 73
sidebar_position: 75
---
# `stack pfs reaction`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs review
sidebar_position: 76
sidebar_position: 78
---
# `stack pfs review`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs utilization
sidebar_position: 75
sidebar_position: 77
---
# `stack pfs utilization`

View File

@@ -1,6 +1,6 @@
---
title: stack pfs
sidebar_position: 68
sidebar_position: 70
---
# `stack pfs`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma eligible
sidebar_position: 89
sidebar_position: 91
---
# `stack prisma eligible`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma export
sidebar_position: 86
sidebar_position: 88
---
# `stack prisma export`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma extract
sidebar_position: 90
sidebar_position: 92
---
# `stack prisma extract`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma fetch
sidebar_position: 92
sidebar_position: 94
---
# `stack prisma fetch`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma flow
sidebar_position: 91
sidebar_position: 93
---
# `stack prisma flow`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma init
sidebar_position: 85
sidebar_position: 87
---
# `stack prisma init`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma ping-llm
sidebar_position: 87
sidebar_position: 89
---
# `stack prisma ping-llm`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma run
sidebar_position: 93
sidebar_position: 95
---
# `stack prisma run`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma screen
sidebar_position: 88
sidebar_position: 90
---
# `stack prisma screen`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma vpn
sidebar_position: 94
sidebar_position: 96
---
# `stack prisma vpn`

View File

@@ -1,6 +1,6 @@
---
title: stack prisma
sidebar_position: 84
sidebar_position: 86
---
# `stack prisma`

View File

@@ -1,6 +1,6 @@
---
title: stack rec list
sidebar_position: 65
sidebar_position: 67
---
# `stack rec list`

View File

@@ -1,6 +1,6 @@
---
title: stack rec opps
sidebar_position: 67
sidebar_position: 69
---
# `stack rec opps`

View File

@@ -1,6 +1,6 @@
---
title: stack rec pfs
sidebar_position: 66
sidebar_position: 68
---
# `stack rec pfs`

View File

@@ -1,6 +1,6 @@
---
title: stack rec
sidebar_position: 64
sidebar_position: 66
---
# `stack rec`

View File

@@ -1,6 +1,6 @@
---
title: stack zot dump-schema
sidebar_position: 79
sidebar_position: 81
---
# `stack zot dump-schema`

View File

@@ -1,6 +1,6 @@
---
title: stack zot fix-dates
sidebar_position: 80
sidebar_position: 82
---
# `stack zot fix-dates`

View File

@@ -1,6 +1,6 @@
---
title: stack zot fix-fields
sidebar_position: 82
sidebar_position: 84
---
# `stack zot fix-fields`

View File

@@ -1,6 +1,6 @@
---
title: stack zot fix-keys
sidebar_position: 81
sidebar_position: 83
---
# `stack zot fix-keys`

View File

@@ -1,6 +1,6 @@
---
title: stack zot verify-parity
sidebar_position: 83
sidebar_position: 85
---
# `stack zot verify-parity`

View File

@@ -1,6 +1,6 @@
---
title: stack zot
sidebar_position: 78
sidebar_position: 80
---
# `stack zot`

View File

@@ -0,0 +1,46 @@
# P35 — closed-vocabulary RAG tagging of rulemaking comments into bib/Zotero
Tracker: milestone P35 (#574 vocabulary, #575 chain + batch runner, #576 write-back + Zotero sync, #577 golden-set gate). Parent spec: `2026-07-16-llm-module-design.md` §P35. Builds on the comment pipeline (P47 dockets/seals, `.state/comments/<docket>/<id>/combined.md` bodies), the pgvector index (`comments` collection, chunk metadata `docket`/`comment_id`/`item_key`/`date`), `llm.classify.closed_vocab_classifier` (one slug or `none` from the largest live host), `llm.pool` (dead-host drop, #796), and the bib tag conventions (`bib.tag` namespaces, `Store.add_tags`, `remove_tag`, nightly `zotero-sync --tag`).
## Goal
Every regulations.gov comment on a PFS docket carries a small set of **theme tags** from a closed vocabulary — which parts of the rule the letter is about — written as `llm:<slug>` tags on its bib item so they reach Zotero with the nightly sync, are filterable in `stack bib query` and the search UI, and give #254–#256 (position classification, stakeholder segmentation, provision mapping) a stable thematic axis. Self-hosted inference only (the llm module never calls a cloud API).
## Decisions
1. **The vocabulary is a versioned YAML file in the repo**, `src/llm/vocab/themes.yaml`, hand-curated. It is *seeded* from the Federal Register section headings of the CY2018–CY2027 PFS rules (mined from `fr_anchors`, `dev/scripts/llm_vocab_candidates.py`) and from the hand code families, but the file is the artifact; a run records the `version` it was tagged with. Slugs are kebab-case, ~60 themes, each with `label`, `definition` (one sentence the model sees), `synonyms` (phrases that appear in letters), and `sections` (the FR heading stems it maps to). Mechanical bib namespaces (`docket:`, `year:`, `rule:`, `reg-docket:`, `enriched:`, `code:`, `family:`) are excluded by construction — they are provenance, not themes.
2. **Multi-label, evidence-first.** A comment gets zero to `max_tags` (default 4) themes. The chain does not ask the model to pick from 60 slugs cold: it first shortlists candidates by embedding similarity between the comment's chunks and each theme's definition+synonyms (the theme "cards" are embedded once per vocab version), then asks the model a yes/no per candidate with the comment's most relevant chunk as evidence, on the largest live host, `temperature 0`. Anything outside `{yes, no}` counts as `no`. This is the closed-vocab discipline of `llm.classify` applied per theme, and the evidence chunk is stored so a human can audit any tag.
3. **Resumable, docket-scoped, idempotent.** State lives in bib: `extra_json["llm_tags"] = {"version": v, "model": m, "tags": {slug: {"confidence": c, "evidence_chunk": seq}}, "tagged_at": iso}`. A comment is skipped when its stored `version`+`model`+`content_hash` match; `--force` redoes it. `stack llm tag --docket D [--limit N] [--force]` walks a docket newest-first through the pool (`HostPool` fan-out; a dead host is dropped, #796).
4. **Write-back replaces only `llm:` tags.** `apply_tags(store, key, slugs)` removes every existing `llm:*` tag on the item and adds the new set — hand tags (`topic:`, `stance:`, …) are never touched. Confidence below `min_confidence` (default 0.6) is recorded in `extra_json` but not written as a tag. The nightly `zotero-sync` already carries tags; `--tag llm:` scopes a manual push.
5. **Golden set gates fan-out.** `tests/llm/golden_themes.yaml` holds ~200 hand-labelled comments across ≥3 dockets (comment id, expected slugs). `stack llm eval-tags` computes per-slug precision/recall and the abstain rate; CI runs it mocked (the harness only), the nightly `llm-golden` job runs it live. Corpus-wide tagging beyond the pilot docket is blocked until micro-F1 ≥ 0.70 and no theme with support ≥ 5 has precision < 0.5.
6. **Observability** reuses `llm.metrics` (`dispatch`, a `tagged` counter) — the P34 Grafana panel gains a tagging row later, not in this milestone.
## Data flow
```
themes.yaml ──embed cards──▶ pgvector "themes" collection (version-stamped)
comment chunks (pgvector) ──similarity vs cards──▶ shortlist (top 8 themes)
shortlist × evidence chunk ──yes/no on largest host──▶ tags + confidence
└──▶ bib extra_json.llm_tags (state)
└──▶ bib tags llm:<slug> (write-back)
└──▶ nightly zotero-sync
golden_themes.yaml ──▶ stack llm eval-tags ──▶ precision/recall report (gate)
```
## Modules
- `src/llm/vocab.py` — `Theme` dataclass, `load()` (validates unique slugs, kebab-case, non-empty definitions), `version()` (the file's `version` field), `cards()` (text per theme for embedding).
- `src/llm/tagger.py` — `shortlist()`, `judge()`, `tag_comment()`, `run()` (batch), `apply_tags()`; pure pieces take injected callables so tests never touch Ollama or pgvector.
- `src/cli/llm.py` — `vocab` (list/validate), `tag` (batch), `eval-tags` (gate).
- `dev/scripts/llm_vocab_candidates.py` — the heading miner that seeds a vocab revision (not run in CI).
## Slices
1. **#574** vocabulary YAML + `llm.vocab` + `stack llm vocab` (this slice).
2. **#575** theme cards in pgvector, shortlist + judge chain, batch runner with state.
3. **#576** write-back + sync verification on a pilot docket (CMS-2026-2377).
4. **#577** golden set + `eval-tags` + the gate; then corpus-wide run.
## Out of scope
Position/stance classification (#254's other half — reuse `stance:` from P43 tooling later), stakeholder segmentation (#255), cloud LLMs, tagging rules or the Zotero corpus (comments only).

View File

@@ -613,6 +613,24 @@ class Store:
con.commit()
return added
def merge_extra(self, item_key: str, updates: dict[str, Any]) -> None:
"""Merge *updates* into the item's ``extra_json`` (top-level keys
replace; everything else is kept). A missing item is a no-op."""
con = self._con()
row = con.execute(
"SELECT extra_json FROM items WHERE key = ?", (item_key,)
).fetchone()
if row is None:
return
current = json.loads(row[0] or "{}") if row[0] else {}
current.update(updates)
con.execute(
"UPDATE items SET extra_json = ?, "
"updated_at = strftime('%Y-%m-%dT%H:%M:%SZ','now') WHERE key = ?",
(json.dumps(current), item_key),
)
con.commit()
def remove_tag(self, item_key: str, tag: str) -> None:
"""Remove a tag from an item."""
con = self._con()

View File

@@ -1,5 +1,7 @@
"""stack llm — local RAG over the library (index + fleet + chat serve)."""
from typing import Any
import typer
app = typer.Typer(no_args_is_help=True)
@@ -203,3 +205,136 @@ def serve(
import uvicorn
uvicorn.run("llm.api:app", host=host, port=port, log_level="info")
@app.command()
def vocab(
path: str = typer.Option(
"", "--path", help="A vocabulary file to validate instead of the packaged one."
),
slug: str = typer.Option("", "--slug", help="Show one theme in full."),
) -> None:
"""The closed theme vocabulary for comment tagging (#574): validate it
and list its slugs, or show one theme's definition, synonyms and the FR
section stems it was seeded from."""
from llm.vocab import VocabError, load
try:
v = load(path or None)
except VocabError as exc:
typer.echo(f"invalid vocabulary: {exc}")
raise typer.Exit(1) from exc
if slug:
if slug not in v:
raise typer.BadParameter(f"unknown slug {slug!r}")
t = v.get(slug)
typer.echo(f"{t.tag} {t.label}")
typer.echo(f" {t.definition}")
typer.echo(f" synonyms: {', '.join(t.synonyms) or '—'}")
typer.echo(f" sections: {' | '.join(t.sections) or '—'}")
return
for t in v.themes:
typer.echo(f"{t.slug:<34} {t.label}")
typer.echo(
f"vocabulary v{v.version}: {len(v)} themes, {len(v.retired)} retired ({v.source})"
)
def _tagging_runtime(cfg: Any) -> dict[str, Any]:
"""Everything `stack llm tag` needs from the live stack — one place
the tests monkeypatch: the bib store, the pgvector chunk loader, the
theme-card vectors (embedded once per run through the pool), the
yes/no judge on the largest live host, and the model name recorded
in each item's state."""
from conf.connect import bib
from llm.index import _engine
from llm.pool import HostPool, PoolEmbeddings, pick_model
from llm.tagger import card_vectors, make_judge, pg_chunks
from llm.vocab import load as load_vocab
vocab = load_vocab()
pool = HostPool.from_config(cfg)
pool.check(cfg.embed_model)
cards = card_vectors(vocab, PoolEmbeddings(pool, cfg.embed_model).embed_documents)
judge = make_judge(cfg, pool)
with pool.acquire_generation() as host:
model = pick_model(cfg, pool, host)
return {
"store": bib(),
"vocab": vocab,
"load_chunks": pg_chunks(_engine(cfg)),
"card_vecs": cards,
"judge": judge,
"model": model,
}
@app.command()
def tag(
docket: str = typer.Option(
..., "--docket", help="regulations.gov docket id, e.g. CMS-2026-2377."
),
limit: int = typer.Option(
0, "--limit", help="Stop after N comments (newest first; 0 = all)."
),
force: bool = typer.Option(
False, "--force", help="Re-tag comments whose stored state is current."
),
dry_run: bool = typer.Option(
False, "--dry-run", help="Compute and print, write nothing."
),
top: int = typer.Option(
8, "--top", help="Themes shortlisted per comment before judging."
),
max_tags: int = typer.Option(
4, "--max-tags", help="Most themes written per comment."
),
min_confidence: float = typer.Option(
0.0,
"--min-confidence",
help="Shortlist similarity below which a judged yes is not written.",
),
verbose: bool = typer.Option(
False, "--verbose", help="Print every comment's tags."
),
) -> None:
"""Closed-vocabulary theme tagging of one docket's comments (P35):
shortlist by similarity to the theme cards, judge each candidate
yes/no on the largest live host with the scoring chunk as evidence,
record state in the item's extra_json, and replace its llm: tags.
Resumable — unchanged comments are skipped unless --force."""
from llm import config as llm_config
from llm.tagger import run as run_tagging
cfg = llm_config.load()
rt = _tagging_runtime(cfg)
def progress(key: str, result: Any, why: str) -> None:
if result is None:
if verbose:
typer.echo(f"{key} {why}")
return
if verbose or dry_run:
typer.echo(f"{key} {', '.join(result.slugs) or '(none)'}")
stats = run_tagging(
rt["store"],
vocab=rt["vocab"],
docket=docket,
load_chunks=rt["load_chunks"],
card_vecs=rt["card_vecs"],
judge=rt["judge"],
model=rt["model"],
limit=limit,
force=force,
dry_run=dry_run,
top=top,
max_tags=max_tags,
min_confidence=min_confidence,
progress=progress,
)
typer.echo(
f"tag {docket} (vocab v{rt['vocab'].version}, {rt['model']}{', dry run' if dry_run else ''}): "
f"seen={stats['seen']} tagged={stats['tagged']} unchanged={stats['skipped_state']} "
f"no_chunks={stats['no_chunks']} tags_written={stats['tags_written']}"
)

369
src/llm/tagger.py Normal file
View File

@@ -0,0 +1,369 @@
"""Closed-vocabulary theme tagging of rulemaking comments (P35, #575/#576).
For one comment the chain is: its already-indexed chunks (text + vector,
``langchain_pg_embedding`` ``comments`` collection) → cosine similarity
against every theme card (``llm.vocab.Theme.card`` embedded once per run)
→ the ``top`` themes shortlisted, each with the chunk that scored it →
one yes/no judgement per candidate on the largest live host, with that
chunk as the evidence → up to ``max_tags`` accepted themes, confidence =
the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``.
State lives on the bib item (``extra_json["llm_tags"]``: vocab version,
model, content hash, the accepted tags with confidence and evidence
chunk, and the shortlist with every verdict) so a run is resumable and a
re-run skips comments whose version + model + content are unchanged.
Write-back replaces exactly the item's ``llm:*`` tags and never touches
hand tags.
Every collaborator is injected (``embed``, ``judge``, ``load_chunks``) so
the pure pieces run in tests without Ollama or pgvector; the CLI wires
the real ones (``make_judge``, ``card_vectors``, ``pg_chunks``).
"""
from __future__ import annotations
import hashlib
import json
import logging
import math
from dataclasses import asdict, dataclass
from datetime import datetime, timezone
from typing import Any, Callable, Iterable, Mapping, Sequence
from llm.vocab import Theme, Vocab, split_tags
log = logging.getLogger(__name__)
Vector = Sequence[float]
Embed = Callable[[Sequence[str]], list[list[float]]]
Judge = Callable[[Theme, str], bool | None]
ChunkLoader = Callable[[str], list[tuple[int, str, list[float]]]]
STATE_KEY = "llm_tags"
@dataclass(frozen=True)
class Candidate:
slug: str
score: float
chunk_idx: int
@dataclass(frozen=True)
class TagInfo:
confidence: float
evidence_chunk: int
@dataclass(frozen=True)
class TagResult:
key: str
tags: dict[str, TagInfo]
shortlisted: tuple[Candidate, ...]
judged: dict[str, bool | None]
content_hash: str
@property
def slugs(self) -> list[str]:
return sorted(self.tags, key=lambda s: -self.tags[s].confidence)
# ── pure pieces ───────────────────────────────────────────────────────
def cosine(a: Vector, b: Vector) -> float:
dot = sum(x * y for x, y in zip(a, b))
na = math.sqrt(sum(x * x for x in a))
nb = math.sqrt(sum(y * y for y in b))
return dot / (na * nb) if na and nb else 0.0
def shortlist(
chunk_vecs: Sequence[Vector], card_vecs: Mapping[str, Vector], *, top: int = 8
) -> list[Candidate]:
"""The ``top`` themes by their best cosine similarity to any chunk,
each with the index of the chunk that scored it (the evidence)."""
if not chunk_vecs:
return []
out: list[Candidate] = []
for slug, cv in card_vecs.items():
best_i, best = 0, -1.0
for i, v in enumerate(chunk_vecs):
s = cosine(v, cv)
if s > best:
best_i, best = i, s
out.append(Candidate(slug, round(best, 4), best_i))
out.sort(key=lambda c: (-c.score, c.slug))
return out[:top]
_SYSTEM = (
"You decide whether a public comment letter on a Medicare physician fee "
"schedule rule addresses a given theme. Answer with exactly one word: yes "
"or no. Say yes only when the excerpt actually discusses the theme, not "
"when it merely mentions a related word in passing."
)
def judge_prompt(theme: Theme, evidence: str) -> list[dict]:
return [
{"role": "system", "content": _SYSTEM},
{
"role": "user",
"content": (
f"Theme: {theme.label}\nDefinition: {theme.definition}\n"
f"Phrases letters use: {', '.join(theme.synonyms) or '—'}\n\n"
f"Excerpt from the comment:\n{evidence.strip()}\n\n"
"Does this comment address the theme? Answer yes or no."
),
},
]
def parse_yes_no(reply: str) -> bool | None:
word = (reply or "").strip().strip(".!\"'").lower().split()
if not word:
return None
if word[0] in ("yes", "y"):
return True
if word[0] in ("no", "n"):
return False
return None
def content_hash(chunks: Sequence[str]) -> str:
h = hashlib.sha1() # noqa: S324 — change detection, not security
for c in chunks:
h.update(c.encode("utf-8", "replace"))
h.update(b"\x00")
return h.hexdigest()[:16]
def tag_comment(
key: str,
chunks: Sequence[str],
chunk_vecs: Sequence[Vector],
*,
vocab: Vocab,
card_vecs: Mapping[str, Vector],
judge: Judge,
top: int = 8,
max_tags: int = 4,
min_confidence: float = 0.0,
) -> TagResult:
cands = shortlist(chunk_vecs, card_vecs, top=top)
judged: dict[str, bool | None] = {}
tags: dict[str, TagInfo] = {}
for c in cands:
if c.slug not in vocab:
continue
verdict = judge(vocab.get(c.slug), chunks[c.chunk_idx])
judged[c.slug] = verdict
if verdict and c.score >= min_confidence and len(tags) < max_tags:
tags[c.slug] = TagInfo(confidence=c.score, evidence_chunk=c.chunk_idx)
return TagResult(key, tags, tuple(cands), judged, content_hash(chunks))
def state_matches(
extra: Mapping[str, Any], *, version: int, model: str, digest: str
) -> bool:
st = extra.get(STATE_KEY) or {}
return (
st.get("version") == version
and st.get("model") == model
and st.get("content_hash") == digest
)
def state_payload(result: TagResult, *, version: int, model: str) -> dict:
return {
"version": version,
"model": model,
"content_hash": result.content_hash,
"tagged_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
"tags": {s: asdict(i) for s, i in result.tags.items()},
"shortlist": [
{
"slug": c.slug,
"score": c.score,
"chunk": c.chunk_idx,
"verdict": result.judged.get(c.slug),
}
for c in result.shortlisted
],
}
# ── bib side ──────────────────────────────────────────────────────────
def apply_tags(
store: Any, key: str, slugs: Iterable[str], *, vocab: Vocab
) -> tuple[int, int]:
"""Make the item's ``llm:*`` tags exactly ``slugs``; hand tags untouched.
Returns ``(added, removed)``."""
want = {vocab.tag(s) for s in slugs}
have, _hand = split_tags(store.get(key).tags)
removed = 0
for t in have:
if t not in want:
store.remove_tag(key, t)
removed += 1
added = store.add_tags(key, sorted(want - set(have))) if want - set(have) else 0
return added, removed
def _extra(store: Any, key: str) -> dict:
row = (
store._con()
.execute( # noqa: SLF001
"SELECT extra_json FROM items WHERE key = ?", (key,)
)
.fetchone()
)
return json.loads(row[0] or "{}") if row and row[0] else {}
def docket_keys(store: Any, docket: str, *, limit: int = 0) -> list[str]:
"""Comment item keys in *docket*, newest posted first."""
con = store._con() # noqa: SLF001
sql = (
"SELECT i.key FROM items i "
"WHERE i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
"(SELECT id FROM tags WHERE name = ?)) "
"AND i.url LIKE 'https://www.regulations.gov/comment/%' "
"ORDER BY i.date_published DESC, i.id DESC"
)
if limit:
sql += f" LIMIT {int(limit)}"
return [r[0] for r in con.execute(sql, (f"reg-docket:{docket}",)).fetchall()]
def run(
store: Any,
*,
vocab: Vocab,
docket: str,
load_chunks: ChunkLoader,
card_vecs: Mapping[str, Vector],
judge: Judge,
model: str,
limit: int = 0,
force: bool = False,
dry_run: bool = False,
top: int = 8,
max_tags: int = 4,
min_confidence: float = 0.0,
progress: Callable[[str, TagResult | None, str], None] | None = None,
) -> dict[str, int]:
"""Tag every comment in *docket* (newest first); resumable via the
item's stored state; ``dry_run`` computes and reports but writes
nothing."""
stats = {
"seen": 0,
"tagged": 0,
"skipped_state": 0,
"no_chunks": 0,
"tags_written": 0,
}
for key in docket_keys(store, docket, limit=limit):
stats["seen"] += 1
extra = _extra(store, key)
rows = load_chunks(key)
if not rows:
stats["no_chunks"] += 1
if progress:
progress(key, None, "no-chunks")
continue
rows = sorted(rows, key=lambda r: r[0])
texts = [t for _, t, _ in rows]
digest = content_hash(texts)
if not force and state_matches(
extra, version=vocab.version, model=model, digest=digest
):
stats["skipped_state"] += 1
if progress:
progress(key, None, "unchanged")
continue
result = tag_comment(
key,
texts,
[v for _, _, v in rows],
vocab=vocab,
card_vecs=card_vecs,
judge=judge,
top=top,
max_tags=max_tags,
min_confidence=min_confidence,
)
stats["tagged"] += 1
if not dry_run:
store.merge_extra(
key,
{STATE_KEY: state_payload(result, version=vocab.version, model=model)},
)
added, _removed = apply_tags(store, key, result.slugs, vocab=vocab)
stats["tags_written"] += added
if progress:
progress(key, result, "tagged")
return stats
# ── real collaborators (the CLI wires these) ──────────────────────────
def card_vectors(vocab: Vocab, embed: Embed) -> dict[str, list[float]]:
slugs = list(vocab.cards)
vecs = embed([vocab.cards[s] for s in slugs])
return dict(zip(slugs, vecs))
def make_judge(
cfg: Any, pool: Any, *, post: Callable[..., dict] | None = None
) -> Judge:
"""A yes/no judge on the largest live host, ``temperature 0`` — the
same route ``llm.classify.closed_vocab_classifier`` takes."""
from llm.classify import _default_post
from llm.pool import pick_model
send = post or _default_post
pool.check(cfg.instruct_model)
def judge(theme: Theme, evidence: str) -> bool | None:
with pool.acquire_generation() as host:
model = pick_model(cfg, pool, host)
data = send(
f"{host}/api/chat",
json={
"model": model,
"messages": judge_prompt(theme, evidence),
"stream": False,
"options": {"num_ctx": cfg.chat_num_ctx, "temperature": 0},
},
)
reply = data.get("message", {}).get("content", "")
verdict = parse_yes_no(reply)
if verdict is None:
log.info("judge: unparseable reply %r for %s", reply[:60], theme.slug)
return verdict
return judge
def pg_chunks(engine: Any, collection: str = "comments") -> ChunkLoader:
"""``load_chunks(key)`` over ``langchain_pg_embedding`` — the text and
vector of every chunk indexed for the item, in ``seq`` order."""
from sqlalchemy import text as _text
sql = _text(
"SELECT (e.cmetadata->>'seq')::int AS seq, e.document, e.embedding::text "
"FROM langchain_pg_embedding e JOIN langchain_pg_collection c ON c.uuid = e.collection_id "
"WHERE c.name = :collection AND e.cmetadata->>'item_key' = :key ORDER BY seq"
)
def load(key: str) -> list[tuple[int, str, list[float]]]:
with engine.begin() as con:
rows = con.execute(sql, {"collection": collection, "key": key}).fetchall()
return [(int(seq or 0), doc or "", json.loads(vec)) for seq, doc, vec in rows]
return load

151
src/llm/vocab/__init__.py Normal file
View File

@@ -0,0 +1,151 @@
"""The closed theme vocabulary for comment tagging (P35, #574).
``themes.yaml`` next to this file is the artifact: hand-curated, versioned
(``version:``), seeded from the CY2018+ PFS rules' section headings
(``dev/scripts/llm_vocab_candidates.py``). Every tagging run records the
version it used, so a vocabulary change never silently re-labels old
work — bump ``version`` when a slug, definition or synonym changes and
retire slugs into ``retired:`` instead of deleting them.
load() → Vocab (validated: unique kebab-case slugs, definitions)
Vocab.cards → {slug: text} — what gets embedded per theme (#575)
Vocab.tag(slug) → "llm:<slug>", the bib tag label
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from importlib import resources
from pathlib import Path
from typing import Any, Sequence
import yaml
TAG_NAMESPACE = "llm"
_SLUG = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
@dataclass(frozen=True)
class Theme:
slug: str
label: str
definition: str
synonyms: tuple[str, ...] = ()
sections: tuple[str, ...] = ()
@property
def tag(self) -> str:
return f"{TAG_NAMESPACE}:{self.slug}"
@property
def card(self) -> str:
"""The text embedded for shortlisting: label, definition and the
phrases letters actually use."""
syn = ", ".join(self.synonyms)
return f"{self.label}. {self.definition}" + (f" Also: {syn}." if syn else "")
@dataclass(frozen=True)
class Vocab:
version: int
themes: tuple[Theme, ...]
retired: tuple[str, ...] = ()
source: str = ""
_by_slug: dict[str, Theme] = field(default_factory=dict, repr=False, compare=False)
def __post_init__(self) -> None:
object.__setattr__(self, "_by_slug", {t.slug: t for t in self.themes})
@property
def slugs(self) -> tuple[str, ...]:
return tuple(t.slug for t in self.themes)
def get(self, slug: str) -> Theme:
return self._by_slug[slug]
def __contains__(self, slug: object) -> bool:
return slug in self._by_slug
def __len__(self) -> int:
return len(self.themes)
@property
def cards(self) -> dict[str, str]:
return {t.slug: t.card for t in self.themes}
def tag(self, slug: str) -> str:
return self.get(slug).tag
class VocabError(ValueError):
"""The YAML is not a valid vocabulary — say exactly what is wrong."""
def _theme(raw: dict[str, Any]) -> Theme:
slug = str(raw.get("slug") or "").strip()
if not _SLUG.match(slug):
raise VocabError(f"slug {slug!r} is not kebab-case")
definition = str(raw.get("definition") or "").strip()
if not definition:
raise VocabError(f"{slug}: definition is empty")
label = str(raw.get("label") or "").strip() or slug
return Theme(
slug=slug,
label=label,
definition=definition,
synonyms=tuple(
str(s).strip() for s in (raw.get("synonyms") or []) if str(s).strip()
),
sections=tuple(
str(s).strip() for s in (raw.get("sections") or []) if str(s).strip()
),
)
def parse(text: str, *, source: str = "") -> Vocab:
data = yaml.safe_load(text) or {}
if not isinstance(data, dict) or "themes" not in data:
raise VocabError(
f"{source or 'vocabulary'}: expected a mapping with a `themes` list"
)
try:
version = int(data.get("version", 0))
except (TypeError, ValueError) as exc:
raise VocabError(f"{source}: version must be an integer") from exc
if version < 1:
raise VocabError(f"{source}: version must be >= 1")
themes = tuple(_theme(t) for t in data.get("themes") or [])
if not themes:
raise VocabError(f"{source}: no themes")
seen: set[str] = set()
for t in themes:
if t.slug in seen:
raise VocabError(f"duplicate slug {t.slug!r}")
seen.add(t.slug)
retired = tuple(
str(s).strip() for s in (data.get("retired") or []) if str(s).strip()
)
clash = seen & set(retired)
if clash:
raise VocabError(f"retired slugs still active: {sorted(clash)}")
return Vocab(version=version, themes=themes, retired=retired, source=source)
def load(path: Path | str | None = None) -> Vocab:
"""The packaged ``themes.yaml`` (default) or a file on disk."""
if path is None:
res = resources.files(__name__).joinpath("themes.yaml")
return parse(res.read_text(encoding="utf-8"), source=str(res))
p = Path(path)
return parse(p.read_text(encoding="utf-8"), source=str(p))
def is_llm_tag(label: str) -> bool:
return label.startswith(f"{TAG_NAMESPACE}:")
def split_tags(labels: Sequence[str]) -> tuple[list[str], list[str]]:
"""``(llm tags, everything else)`` — the write-back only ever touches the first."""
ours = [t for t in labels if is_llm_tag(t)]
return ours, [t for t in labels if not is_llm_tag(t)]

276
src/llm/vocab/themes.yaml Normal file
View File

@@ -0,0 +1,276 @@
# Closed theme vocabulary for tagging PFS rulemaking comments (P35, #574).
#
# Seeded from the section headings of the CY2018–CY2027 PFS proposed and
# final rules (dev/scripts/llm_vocab_candidates.py over fr_anchors) and the
# hand code families; curated by hand. Slugs are stable once published —
# retire a theme by moving it to `retired:` rather than deleting it, and
# bump `version` whenever a slug, definition or synonym list changes, since
# every tag run records the version it used.
version: 1
themes:
- slug: conversion-factor
label: Conversion factor and payment update
definition: The annual PFS conversion factor, statutory update percentage, budget-neutrality adjustment, or the overall payment cut or increase.
synonyms: [conversion factor, CF, payment update, budget neutrality, statutory update, payment cut, physician payment cut]
sections: [Conversion Factor, Physician Fee Schedule Update, Budget Neutrality]
- slug: practice-expense
label: Practice expense RVUs
definition: Practice expense methodology, direct and indirect PE inputs, clinical labor pricing, supply and equipment pricing, or the PE/HR survey data.
synonyms: [practice expense, PE RVU, clinical labor, supply pricing, equipment pricing, indirect practice expense, PE/HR, AMA PPIS survey]
sections: [Determination of Practice Expense (PE) Relative Value Units (RVUs), Practice Expense Methodology]
- slug: work-rvu
label: Work RVUs and code valuation
definition: Physician work RVUs, RUC recommendations, time assumptions, or the valuation of specific CPT/HCPCS codes.
synonyms: [work RVU, RUC recommendation, valuation of specific codes, intraservice time, survey data, code valuation]
sections: [Valuation of Specific Codes, Work RVUs]
- slug: malpractice-rvu
label: Malpractice RVUs
definition: Malpractice (MP) RVU methodology, premium data, or specialty risk factors.
synonyms: [malpractice RVU, MP RVU, professional liability insurance premium, risk factor]
sections: [Determination of Malpractice Relative Value Units (RVUs)]
- slug: misvalued-codes
label: Potentially misvalued codes
definition: Identification, review or revaluation of potentially misvalued services, including the statutory misvalued-code target.
synonyms: [misvalued, potentially misvalued codes, misvalued code target, revaluation]
sections: [Potentially Misvalued Services Under the PFS]
- slug: gpci-localities
label: GPCIs and payment localities
definition: Geographic practice cost indices, locality definitions, or geographic adjustment of payment.
synonyms: [GPCI, geographic practice cost index, payment locality, locality, geographic adjustment, work GPCI floor]
sections: [Geographic Practice Cost Indices (GPCIs), Payment Localities]
- slug: telehealth
label: Medicare telehealth services
definition: The Medicare telehealth services list, originating-site and geographic restrictions, audio-only, virtual direct supervision, or telehealth flexibilities after the public health emergency.
synonyms: [telehealth, telemedicine, audio-only, originating site, distant site, virtual supervision, telehealth list, home as originating site]
sections: [Medicare Telehealth Services, Payment for Medicare Telehealth Services Under Section 1834(m) of the Act]
- slug: communication-technology-services
label: Communication technology-based services
definition: Virtual check-ins, e-visits, remote evaluation of images, interprofessional consultations and other communication technology-based services that are not telehealth.
synonyms: [virtual check-in, e-visit, online digital E/M, interprofessional consultation, remote evaluation, CTBS, G2012, 99421]
sections: [Communication Technology-Based Services (CTBS)]
- slug: remote-monitoring
label: Remote physiologic and therapeutic monitoring
definition: Remote physiologic monitoring (RPM) or remote therapeutic monitoring (RTM) codes, device supply days, or monitoring time requirements.
synonyms: [remote physiologic monitoring, RPM, remote therapeutic monitoring, RTM, 99454, 99457, 98977, 16 days]
sections: [Remote Physiologic Monitoring]
- slug: evaluation-management
label: Evaluation and management visits
definition: Office/outpatient, hospital, nursing facility or home E/M visit coding, documentation guidelines, level selection, prolonged services, or split/shared visits.
synonyms: [E/M, evaluation and management, office visit, 99213, 99214, prolonged services, split/shared, level selection, documentation guidelines]
sections: [Evaluation & Management (E/M) Visits, Evaluation & Management (E/M) Guidelines and Care Management Services]
- slug: visit-complexity-add-on
label: Visit complexity add-on (G2211)
definition: The office/outpatient visit complexity add-on code G2211, its use with modifier 25, or its budget impact.
synonyms: [G2211, visit complexity, complexity add-on, longitudinal relationship, modifier 25]
sections: [Visit Complexity]
- slug: care-management
label: Care management services
definition: Chronic care management, principal care management, transitional care management, advanced primary care management, or other monthly care-management codes and their consent, staffing and time requirements.
synonyms: [chronic care management, CCM, principal care management, PCM, transitional care management, TCM, advanced primary care management, APCM, 99490, 99439, G0556, care management]
sections: [Care Management Services, Advanced Primary Care Management]
- slug: behavioral-health
label: Behavioral health integration and services
definition: Behavioral health integration, collaborative care, psychotherapy, marriage and family therapists and mental health counselors, or crisis and safety-planning services.
synonyms: [behavioral health, mental health, collaborative care, BHI, psychotherapy, MFT, MHC, safety planning, crisis psychotherapy, community health integration]
sections: [Behavioral Health Services]
- slug: opioid-treatment
label: Opioid treatment programs and substance use disorder
definition: Opioid treatment program bundled payments, medications for opioid use disorder, or substance use disorder treatment via telehealth.
synonyms: [opioid treatment program, OTP, medication for opioid use disorder, MOUD, buprenorphine, methadone, substance use disorder, SUD]
sections: [Requirements for Opioid Treatment Programs (OTP)]
- slug: caregiver-training
label: Caregiver training services
definition: Caregiver training services furnished to a patient's caregivers, with or without the patient present.
synonyms: [caregiver training, CTS, 97550, 96202]
sections: [Caregiver Training Services]
- slug: community-health-social-needs
label: Community health integration and social needs
definition: Community health integration, principal illness navigation, social determinants of health risk assessment, or community health worker services.
synonyms: [community health integration, CHI, principal illness navigation, PIN, social determinants of health, SDOH risk assessment, community health worker, G0019, G0136]
sections: [Community Health Integration, Principal Illness Navigation]
- slug: palliative-care
label: Palliative and serious-illness care
definition: Community-based palliative care, advance care planning, hospice interaction, or serious-illness care management.
synonyms: [palliative care, advance care planning, ACP, serious illness, hospice, end-of-life]
sections: [Request for Information on Community-Based Palliative Care]
- slug: supervision
label: Supervision requirements
definition: Direct, general or virtual supervision of auxiliary personnel, incident-to services, or supervision of diagnostic tests and residents.
synonyms: [direct supervision, general supervision, incident to, incident-to, supervision of residents, virtual presence, teaching physician]
sections: [Direct Supervision by Interactive Telecommunications Technology, Teaching Physician Documentation Requirements]
- slug: non-physician-practitioners
label: Non-physician practitioners and scope
definition: Nurse practitioners, physician assistants, clinical nurse specialists, therapists or other non-physician practitioners' billing, supervision or scope-of-practice policies.
synonyms: [nurse practitioner, physician assistant, PA, NP, advanced practice, scope of practice, CRNA, non-physician practitioner, NPP]
sections: [Physician Assistants, Nurse Practitioners]
- slug: therapy-services
label: Physical, occupational and speech therapy
definition: Therapy services, therapy assistants and the CQ/CO modifiers, the KX modifier threshold, plan-of-care certification, or therapy supervision.
synonyms: [physical therapy, occupational therapy, speech-language pathology, therapy assistant, CQ modifier, CO modifier, KX modifier, therapy threshold, plan of care]
sections: [Therapy Services, Therapy Caps]
- slug: drugs-biologicals
label: Part B drugs and biologicals
definition: Average sales price methodology, drug add-on percentages, biosimilars, discarded-drug refunds, or drug payment under Part B.
synonyms: [average sales price, ASP, WAC, biosimilar, discarded drug, JW modifier, JZ modifier, Part B drug, drug payment, 340B]
sections: [Part B Drug Payment, Payment for Biosimilar Biological Products]
- slug: vaccines-preventive
label: Vaccines and preventive services
definition: Vaccine administration payment, preventive services coverage, screening services, or the Medicare Diabetes Prevention Program.
synonyms: [vaccine administration, preventive services, screening, colorectal cancer screening, Medicare Diabetes Prevention Program, MDPP, hepatitis B vaccine, HIV PrEP]
sections: [Medicare Diabetes Prevention Program, Preventive Services]
- slug: skin-substitutes
label: Skin substitutes and wound care
definition: Payment or coding for skin substitute products, cellular and tissue-based products, or wound care services.
synonyms: [skin substitute, cellular and tissue-based product, CTP, wound care, Q4 code, amniotic, dermal substitute]
sections: [Skin Substitutes]
- slug: dental-services
label: Dental services
definition: Medicare payment for dental services inextricably linked to covered medical services.
synonyms: [dental, oral health, dental services, inextricably linked]
sections: [Dental Services]
- slug: global-surgery
label: Global surgical packages
definition: Global surgery periods, post-operative visit reporting, or the transfer-of-care modifier for global packages.
synonyms: [global surgery, global period, 010-day, 090-day, post-operative visit, modifier 54, modifier 55, transfer of care, G0559]
sections: [Global Surgery]
- slug: anesthesia
label: Anesthesia services
definition: The anesthesia conversion factor, anesthesia base units, or anesthesia billing and supervision.
synonyms: [anesthesia, anesthesia conversion factor, base units, CRNA, medical direction]
sections: [Anesthesia Services, Anesthesia Conversion Factor]
- slug: imaging-radiology
label: Imaging and radiology
definition: Diagnostic imaging payment, appropriate use criteria for advanced imaging, or radiology assistants and supervision.
synonyms: [imaging, radiology, appropriate use criteria, AUC, clinical decision support, CT, MRI, radiologist assistant]
sections: [Appropriate Use Criteria for Advanced Diagnostic Imaging Services]
- slug: laboratory
label: Clinical laboratory fee schedule
definition: The clinical laboratory fee schedule, private payer rate reporting, specimen collection, or laboratory test payment.
synonyms: [clinical laboratory fee schedule, CLFS, private payor rate, PAMA, laboratory test, specimen collection]
sections: [Clinical Laboratory Fee Schedule]
- slug: ambulance
label: Ambulance services
definition: The ambulance fee schedule, ground or air ambulance payment, or ambulance cost data collection.
synonyms: [ambulance, ambulance fee schedule, ground ambulance, air ambulance, ambulance cost collection]
sections: [Ambulance Fee Schedule]
- slug: dme-supplies
label: Durable medical equipment and supplies
definition: DME infusion drugs, DMEPOS competitive bidding, or supplies furnished with physician services.
synonyms: [durable medical equipment, DME, DMEPOS, infusion drugs, competitive bidding]
sections: [Payment for DME Infusion Drugs]
- slug: esrd-dialysis
label: ESRD and dialysis services
definition: Monthly capitation payment for ESRD-related services, home dialysis, or kidney disease education.
synonyms: [ESRD, dialysis, monthly capitation payment, MCP, home dialysis, kidney disease education]
sections: [Monthly Capitation Payment (MCP) for ESRD-Related Services]
- slug: rhc-fqhc
label: Rural health clinics and FQHCs
definition: Payment or care-coordination policies for rural health clinics and federally qualified health centers.
synonyms: [rural health clinic, RHC, federally qualified health center, FQHC, G0511, all-inclusive rate]
sections: [Rural Health Clinics (RHCs) and Federally Qualified Health Centers (FQHCs)]
- slug: rural-access
label: Rural and underserved access
definition: Access to care in rural or underserved areas, workforce shortages, or health equity concerns raised about a policy's effect on access.
synonyms: [rural, underserved, health equity, access to care, workforce shortage, health professional shortage area, HPSA]
sections: [Health Equity]
- slug: shared-savings-program
label: Medicare Shared Savings Program
definition: Shared Savings Program ACO participation, benchmarking, risk tracks, quality reporting, beneficiary assignment, or advance investment payments.
synonyms: [Shared Savings Program, MSSP, accountable care organization, ACO, benchmark, ENHANCED track, BASIC track, beneficiary assignment, advance investment payment]
sections: [Medicare Shared Savings Program]
- slug: quality-payment-program
label: Quality Payment Program and MIPS
definition: MIPS categories, MIPS Value Pathways, scoring, performance thresholds, or the QPP reporting requirements.
synonyms: [Quality Payment Program, QPP, MIPS, MIPS Value Pathway, MVP, performance threshold, promoting interoperability, cost category, improvement activities]
sections: [Updates to the Quality Payment Program]
- slug: advanced-apm
label: Advanced APMs and APM incentives
definition: Advanced alternative payment model participation, qualifying participant thresholds, the APM incentive payment, or the APM conversion factor.
synonyms: [advanced APM, alternative payment model, qualifying participant, QP threshold, APM incentive, APM conversion factor]
sections: [Advanced Alternative Payment Models]
- slug: quality-measures
label: Quality measures and reporting
definition: Specific quality measures, measure specifications, electronic clinical quality measures, or measure reporting burden.
synonyms: [quality measure, eCQM, measure specification, reporting burden, measure set, benchmarking of measures]
sections: [Quality Measures]
- slug: innovation-models
label: Innovation Center models
definition: CMS Innovation Center models and demonstrations, including primary care, kidney, oncology or radiation models.
synonyms: [Innovation Center, CMMI, model, demonstration, Primary Care First, ACO REACH, oncology care model, radiation oncology model]
sections: [Innovation Center Models]
- slug: self-referral
label: Physician self-referral law
definition: Stark law exceptions, the annual code list update, or self-referral compliance.
synonyms: [self-referral, Stark, physician self-referral law, designated health services]
sections: [Physician Self-Referral Law]
- slug: enrollment-program-integrity
label: Enrollment and program integrity
definition: Provider enrollment requirements, revocations, prior authorization, fraud, waste and abuse controls, or medical review.
synonyms: [enrollment, revocation, program integrity, fraud waste and abuse, prior authorization, medical review, overpayment, audit]
sections: [Medicare Provider Enrollment, Program Integrity]
- slug: coverage-policy
label: Coverage and benefit categories
definition: Whether a service is a covered Medicare benefit, benefit-category determinations, or national coverage policy raised in the rule.
synonyms: [coverage, benefit category, covered service, national coverage determination, NCD, reasonable and necessary]
sections: [Coverage]
- slug: beneficiary-cost-sharing
label: Beneficiary cost sharing
definition: Coinsurance, deductibles, or beneficiary out-of-pocket effects of a payment policy.
synonyms: [cost sharing, coinsurance, deductible, out-of-pocket, beneficiary liability, copay]
sections: [Telehealth Modalities and Cost-sharing]
- slug: administrative-burden
label: Administrative burden and documentation
definition: Documentation, paperwork, reporting or compliance burden imposed on practices, including collection-of-information estimates.
synonyms: [administrative burden, documentation burden, paperwork, compliance burden, collection of information, patients over paperwork, prior authorization burden]
sections: [Collection of Information Requirements]
- slug: regulatory-impact
label: Regulatory impact and specialty impacts
definition: The regulatory impact analysis, specialty-level payment impact tables, or estimated effects on practices and beneficiaries.
synonyms: [regulatory impact analysis, impact table, specialty impact, Table 118, estimated impact, small entities]
sections: [Regulatory Impact Analysis]
- slug: medicare-economic-index
label: Medicare Economic Index and cost weights
definition: The Medicare Economic Index, its rebasing or revised cost-share weights, and their effect on RVU allocation.
synonyms: [Medicare Economic Index, MEI, cost share weights, rebasing]
sections: [The Percentage Change in the Medicare Economic Index (MEI)]
- slug: efficiency-adjustment
label: Efficiency adjustment to work RVUs
definition: An across-the-board efficiency adjustment to work RVUs or intraservice time for non-time-based services.
synonyms: [efficiency adjustment, efficiency gains, across-the-board reduction, non-time-based services]
sections: [Efficiency Adjustment]
- slug: site-of-service
label: Site of service and facility payment
definition: Facility versus non-facility payment, site-neutral policies, hospital outpatient department comparisons, or indirect PE by setting.
synonyms: [site of service, site neutral, facility rate, non-facility rate, hospital outpatient department, HOPD, office-based]
sections: [Site of Service]
- slug: hospice-home-health
label: Hospice and home health interaction
definition: Hospice face-to-face encounters, homebound status, home health certification, or physician services interacting with hospice and home health benefits.
synonyms: [hospice, home health, face-to-face encounter, homebound, certification of home health]
sections: [Clarification of Homebound Status Under the Medicare Home Health Benefit, Telehealth and the Medicare Hospice Face-to-Face Encounter Requirement]
- slug: interoperability-health-it
label: Health IT and interoperability
definition: Certified EHR technology, promoting interoperability requirements, data exchange, or the EHR incentive programs.
synonyms: [electronic health record, EHR, certified EHR technology, CEHRT, interoperability, health information exchange, promoting interoperability]
sections: [Medicare EHR Incentive Program, Medicaid Promoting Interoperability Program Requirements]
- slug: price-transparency
label: Price transparency
definition: Public disclosure of prices, charges or payment rates to beneficiaries.
synonyms: [price transparency, charge transparency, disclosure of prices]
sections: [Request for Information on Price Transparency]
- slug: covid-phe
label: COVID-19 public health emergency flexibilities
definition: Policies tied to the COVID-19 public health emergency and their extension or expiration.
synonyms: [public health emergency, PHE, COVID-19, pandemic, flexibilities, waiver]
sections: [Counting of Resident Time During the PHE for the COVID-19 Pandemic]
- slug: data-requests-rfi
label: Requests for information
definition: Responses to a request for information or comment solicitation rather than a specific proposal.
synonyms: [request for information, RFI, comment solicitation, seeking comment, solicit feedback]
sections: [Requests for Information]
- slug: coding-descriptors
label: Coding and code descriptors
definition: Creation, deletion or crosswalk of CPT/HCPCS codes, code descriptors, or bundling of services into other codes.
synonyms: [new code, deleted code, HCPCS code, G code, crosswalk, bundled, code descriptor, CPT Editorial Panel, unbundle]
sections: [Proposed Valuation of Specific Codes]
retired: []

View File

@@ -0,0 +1,103 @@
"""stack llm tag / vocab (#574, #575)."""
from __future__ import annotations
import pytest
from typer.testing import CliRunner
import cli.llm as llm_cli
from bib.item import Source
from bib.store import Store
from cli import app
from llm.vocab import parse
runner = CliRunner()
VOCAB = parse(
"version: 1\nthemes:\n - slug: telehealth\n definition: Telehealth.\n synonyms: [telehealth]\n"
" - slug: drugs\n definition: Drugs.\n synonyms: [ASP]\n"
)
@pytest.fixture
def rt(monkeypatch, tmp_path):
store = Store(":memory:", storage_dir=tmp_path / "st")
key = store.upsert(
Source(
title="c1",
url="https://www.regulations.gov/comment/CMS-2026-2377-1",
date_published="2026-09-01",
),
tags=["reg-docket:CMS-2026-2377", "source:regulations-gov"],
)
judged = []
def judge(theme, evidence):
judged.append(theme.slug)
return theme.slug == "telehealth"
runtime = {
"store": store,
"vocab": VOCAB,
"load_chunks": lambda k: (
[(0, "telehealth text", [1.0, 0.0])] if k == key else []
),
"card_vecs": {"telehealth": [1.0, 0.0], "drugs": [0.0, 1.0]},
"judge": judge,
"model": "m",
}
monkeypatch.setattr(llm_cli, "_tagging_runtime", lambda cfg: runtime)
monkeypatch.setattr("llm.config.load", lambda: object())
try:
yield store, key, judged
finally:
store.close()
class TestTag:
def test_tags_and_reports(self, rt):
store, key, judged = rt
res = runner.invoke(
app, ["llm", "tag", "--docket", "CMS-2026-2377", "--verbose"]
)
assert res.exit_code == 0, res.output
assert f"{key} telehealth" in res.output
assert "seen=1 tagged=1 unchanged=0 no_chunks=0 tags_written=1" in res.output
assert (
"llm:telehealth" in store.get(key).tags
and "llm:drugs" not in store.get(key).tags
)
assert judged == ["telehealth", "drugs"]
res = runner.invoke(app, ["llm", "tag", "--docket", "CMS-2026-2377"])
assert "unchanged=1" in res.output
def test_dry_run_prints_but_writes_nothing(self, rt):
store, key, _ = rt
res = runner.invoke(
app, ["llm", "tag", "--docket", "CMS-2026-2377", "--dry-run"]
)
assert res.exit_code == 0, res.output
assert "dry run" in res.output and f"{key} telehealth" in res.output
assert not any(t.startswith("llm:") for t in store.get(key).tags)
def test_docket_is_required(self):
assert runner.invoke(app, ["llm", "tag"]).exit_code != 0
class TestVocab:
def test_lists_and_shows(self):
res = runner.invoke(app, ["llm", "vocab"])
assert (
res.exit_code == 0
and "vocabulary v" in res.output
and "telehealth" in res.output
)
res = runner.invoke(app, ["llm", "vocab", "--slug", "telehealth"])
assert res.exit_code == 0 and "llm:telehealth" in res.output
assert runner.invoke(app, ["llm", "vocab", "--slug", "nope"]).exit_code != 0
def test_invalid_file(self, tmp_path):
p = tmp_path / "bad.yaml"
p.write_text("version: 1\nthemes: []\n")
res = runner.invoke(app, ["llm", "vocab", "--path", str(p)])
assert res.exit_code == 1 and "invalid vocabulary" in res.output

390
tests/llm/test_tagger.py Normal file
View File

@@ -0,0 +1,390 @@
"""llm.tagger — shortlist, judge, state, write-back and the docket runner (#575/#576)."""
from __future__ import annotations
import json
import pytest
from bib.item import Source
from bib.store import Store
from llm.tagger import (
Candidate,
TagResult,
apply_tags,
card_vectors,
content_hash,
cosine,
docket_keys,
judge_prompt,
make_judge,
parse_yes_no,
run,
shortlist,
state_matches,
state_payload,
tag_comment,
)
from llm.vocab import parse
VOCAB = parse(
"""
version: 2
themes:
- slug: telehealth
label: Telehealth
definition: Medicare telehealth services.
synonyms: [telehealth]
- slug: care-management
label: Care management
definition: Monthly care management codes.
synonyms: [CCM]
- slug: drugs
label: Part B drugs
definition: ASP drug payment.
synonyms: [ASP]
"""
)
CARDS = {
"telehealth": [1.0, 0.0, 0.0],
"care-management": [0.0, 1.0, 0.0],
"drugs": [0.0, 0.0, 1.0],
}
class TestPure:
def test_cosine(self):
assert cosine([1, 0], [1, 0]) == pytest.approx(1.0)
assert cosine([1, 0], [0, 1]) == pytest.approx(0.0)
assert cosine([0, 0], [1, 1]) == 0.0
def test_shortlist_best_chunk_per_theme(self):
chunks = [[0.9, 0.1, 0.0], [0.1, 0.9, 0.0]]
out = shortlist(chunks, CARDS, top=2)
assert [c.slug for c in out] == ["care-management", "telehealth"] or [
c.slug for c in out
] == ["telehealth", "care-management"]
by = {c.slug: c for c in out}
assert by["telehealth"].chunk_idx == 0 and by["care-management"].chunk_idx == 1
assert all(isinstance(c, Candidate) and 0 < c.score <= 1 for c in out)
assert shortlist([], CARDS) == []
def test_judge_prompt_and_parse(self):
msgs = judge_prompt(
VOCAB.get("telehealth"), "We support audio-only telehealth."
)
assert msgs[0]["role"] == "system" and "Telehealth" in msgs[1]["content"]
assert "audio-only telehealth" in msgs[1]["content"]
assert parse_yes_no("Yes.") is True and parse_yes_no(" no\n") is False
assert parse_yes_no("Maybe yes") is None and parse_yes_no("") is None
def test_content_hash_is_order_sensitive(self):
assert content_hash(["a", "b"]) != content_hash(["b", "a"])
assert len(content_hash(["a"])) == 16
def test_tag_comment_accepts_judged_yes_up_to_max(self):
chunks = ["telehealth text", "ccm text", "asp text"]
vecs = [[1.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]]
seen = []
def judge(theme, evidence):
seen.append((theme.slug, evidence))
return theme.slug != "drugs"
r = tag_comment(
"K",
chunks,
vecs,
vocab=VOCAB,
card_vecs=CARDS,
judge=judge,
top=3,
max_tags=1,
)
assert isinstance(r, TagResult) and len(r.tags) == 1
assert r.judged["drugs"] is False and any(v for v in r.judged.values())
assert (
"telehealth",
"telehealth text",
) in seen # evidence is the scoring chunk
assert r.content_hash == content_hash(chunks)
def test_min_confidence_filters_weak_matches(self):
chunks = ["x"]
vecs = [[0.6, 0.8, 0.0]] # cosine 0.6 with telehealth, 0.8 with care-management
r = tag_comment(
"K",
chunks,
vecs,
vocab=VOCAB,
card_vecs=CARDS,
judge=lambda t, e: True,
min_confidence=0.7,
)
assert r.slugs == ["care-management"]
def test_state(self):
r = tag_comment(
"K",
["t"],
[[1.0, 0, 0]],
vocab=VOCAB,
card_vecs=CARDS,
judge=lambda t, e: True,
max_tags=1,
)
payload = state_payload(r, version=2, model="m")
assert payload["tags"] == {
"telehealth": {"confidence": 1.0, "evidence_chunk": 0}
}
assert payload["shortlist"][0]["verdict"] is True and "tagged_at" in payload
assert state_matches(
{"llm_tags": payload}, version=2, model="m", digest=r.content_hash
)
assert not state_matches(
{"llm_tags": payload}, version=3, model="m", digest=r.content_hash
)
assert not state_matches({}, version=2, model="m", digest=r.content_hash)
def test_card_vectors_embeds_every_card_once(self):
calls = []
def embed(texts):
calls.append(list(texts))
return [[float(i)] for i in range(len(texts))]
cv = card_vectors(VOCAB, embed)
assert set(cv) == {"telehealth", "care-management", "drugs"} and len(calls) == 1
assert "Medicare telehealth services" in calls[0][0]
class TestJudge:
def test_make_judge_posts_to_largest_host_at_temperature_zero(self):
class Pool:
def check(self, model):
return ["http://h"]
def vram(self, host):
return 0.0
def serves(self, host, model):
return False
class _cm:
def __enter__(self):
return "http://h"
def __exit__(self, *a):
return False
def acquire_generation(self):
return self._cm()
class Cfg:
instruct_model = "m"
instruct_model_large = ""
large_min_vram_gb = 20.0
chat_num_ctx = 4096
sent = {}
def post(url, json=None):
sent["url"], sent["json"] = url, json
return {"message": {"content": "Yes"}}
judge = make_judge(Cfg(), Pool(), post=post)
assert judge(VOCAB.get("drugs"), "ASP add-on") is True
assert (
sent["url"] == "http://h/api/chat"
and sent["json"]["options"]["temperature"] == 0
)
assert sent["json"]["model"] == "m"
def _store(tmp_path):
s = Store(":memory:", storage_dir=tmp_path / "st")
keys = {}
for i, (cid, date) in enumerate(
[
("CMS-2026-2377-1", "2026-09-01"),
("CMS-2026-2377-2", "2026-09-05"),
("CMS-2025-0304-9", "2025-09-01"),
]
):
docket = cid.rsplit("-", 1)[0]
keys[cid] = s.upsert(
Source(
title=cid,
url=f"https://www.regulations.gov/comment/{cid}",
date_published=date,
),
tags=[f"reg-docket:{docket}", "source:regulations-gov", "topic:palliative"],
)
return s, keys
class TestBibSide:
def test_docket_keys_newest_first(self, tmp_path):
s, keys = _store(tmp_path)
assert docket_keys(s, "CMS-2026-2377") == [
keys["CMS-2026-2377-2"],
keys["CMS-2026-2377-1"],
]
assert docket_keys(s, "CMS-2026-2377", limit=1) == [keys["CMS-2026-2377-2"]]
assert docket_keys(s, "NOPE") == []
s.close()
def test_apply_tags_replaces_only_llm_tags(self, tmp_path):
s, keys = _store(tmp_path)
k = keys["CMS-2026-2377-1"]
s.add_tags(k, ["llm:drugs"])
added, removed = apply_tags(
s, k, ["telehealth", "care-management"], vocab=VOCAB
)
assert (added, removed) == (2, 1)
tags = set(s.get(k).tags)
assert {
"llm:telehealth",
"llm:care-management",
"topic:palliative",
"reg-docket:CMS-2026-2377",
} <= tags
assert "llm:drugs" not in tags
assert apply_tags(s, k, ["telehealth", "care-management"], vocab=VOCAB) == (
0,
0,
) # idempotent
assert (
apply_tags(s, k, [], vocab=VOCAB) == (0, 2)
and "topic:palliative" in s.get(k).tags
)
s.close()
def test_merge_extra(self, tmp_path):
s, keys = _store(tmp_path)
k = keys["CMS-2026-2377-1"]
s.merge_extra(k, {"a": 1})
s.merge_extra(k, {"b": {"x": 2}})
s.merge_extra("ZZZZZZZZ", {"c": 3}) # unknown key: no-op, no error
extra = json.loads(
s._con()
.execute("SELECT extra_json FROM items WHERE key=?", (k,))
.fetchone()[0]
) # noqa: SLF001
assert extra["a"] == 1 and extra["b"] == {"x": 2}
s.close()
class TestRun:
def _chunks(self, keys):
def load(key):
if key == keys["CMS-2026-2377-1"]:
return [
(1, "ccm text", [0.0, 1.0, 0.0]),
(0, "telehealth text", [1.0, 0.0, 0.0]),
]
if key == keys["CMS-2026-2377-2"]:
return [] # not indexed yet
return [(0, "asp", [0.0, 0.0, 1.0])]
return load
def test_tags_persist_state_and_resume(self, tmp_path):
s, keys = _store(tmp_path)
events = []
judge_calls = []
def judge(theme, evidence):
judge_calls.append(theme.slug)
return True
stats = run(
s,
vocab=VOCAB,
docket="CMS-2026-2377",
load_chunks=self._chunks(keys),
card_vecs=CARDS,
judge=judge,
model="m",
top=2,
max_tags=2,
progress=lambda k, r, why: events.append((k, why)),
)
assert stats == {
"seen": 2,
"tagged": 1,
"skipped_state": 0,
"no_chunks": 1,
"tags_written": 2,
}
k = keys["CMS-2026-2377-1"]
assert {"llm:telehealth", "llm:care-management"} <= set(s.get(k).tags)
extra = json.loads(
s._con()
.execute("SELECT extra_json FROM items WHERE key=?", (k,))
.fetchone()[0]
) # noqa: SLF001
assert extra["llm_tags"]["version"] == 2 and extra["llm_tags"]["model"] == "m"
assert set(extra["llm_tags"]["tags"]) == {"telehealth", "care-management"}
assert (keys["CMS-2026-2377-2"], "no-chunks") in events
# second run: unchanged → skipped without judging
judge_calls.clear()
stats2 = run(
s,
vocab=VOCAB,
docket="CMS-2026-2377",
load_chunks=self._chunks(keys),
card_vecs=CARDS,
judge=judge,
model="m",
top=2,
)
assert (
stats2["skipped_state"] == 1 and stats2["tagged"] == 0 and judge_calls == []
)
# a new model re-tags; --force re-tags
assert (
run(
s,
vocab=VOCAB,
docket="CMS-2026-2377",
load_chunks=self._chunks(keys),
card_vecs=CARDS,
judge=judge,
model="m2",
top=2,
)["tagged"]
== 1
)
assert (
run(
s,
vocab=VOCAB,
docket="CMS-2026-2377",
load_chunks=self._chunks(keys),
card_vecs=CARDS,
judge=judge,
model="m2",
top=2,
force=True,
)["tagged"]
== 1
)
s.close()
def test_dry_run_writes_nothing(self, tmp_path):
s, keys = _store(tmp_path)
stats = run(
s,
vocab=VOCAB,
docket="CMS-2026-2377",
load_chunks=self._chunks(keys),
card_vecs=CARDS,
judge=lambda t, e: True,
model="m",
dry_run=True,
)
assert stats["tagged"] == 1 and stats["tags_written"] == 0
k = keys["CMS-2026-2377-1"]
assert not any(t.startswith("llm:") for t in s.get(k).tags)
s.close()

114
tests/llm/test_vocab.py Normal file
View File

@@ -0,0 +1,114 @@
"""llm.vocab — the closed theme vocabulary (#574)."""
from __future__ import annotations
import pytest
from llm.vocab import Theme, Vocab, VocabError, is_llm_tag, load, parse, split_tags
GOOD = """
version: 3
themes:
- slug: telehealth
label: Telehealth
definition: Medicare telehealth services.
synonyms: [telehealth, audio-only]
sections: [Medicare Telehealth Services]
- slug: care-management
definition: Monthly care management codes.
retired: [old-theme]
"""
class TestParse:
def test_reads_themes_and_version(self):
v = parse(GOOD, source="t")
assert isinstance(v, Vocab) and v.version == 3 and len(v) == 2
assert v.slugs == ("telehealth", "care-management")
t = v.get("telehealth")
assert isinstance(t, Theme) and t.synonyms == ("telehealth", "audio-only")
assert (
v.get("care-management").label == "care-management"
) # label defaults to slug
assert v.retired == ("old-theme",) and "telehealth" in v and "nope" not in v
def test_cards_and_tags(self):
v = parse(GOOD)
assert (
v.cards["telehealth"]
== "Telehealth. Medicare telehealth services. Also: telehealth, audio-only."
)
assert (
v.cards["care-management"]
== "care-management. Monthly care management codes."
)
assert (
v.tag("telehealth") == "llm:telehealth"
and v.get("telehealth").tag == "llm:telehealth"
)
@pytest.mark.parametrize(
"text, msg",
[
(
"version: 1\nthemes:\n - slug: Bad Slug\n definition: x\n",
"kebab-case",
),
(
"version: 1\nthemes:\n - slug: a\n definition: ''\n",
"definition is empty",
),
(
"version: 1\nthemes:\n - slug: a\n definition: x\n - slug: a\n definition: y\n",
"duplicate slug",
),
(
"version: 0\nthemes:\n - slug: a\n definition: x\n",
"version must be >= 1",
),
("version: x\nthemes: []\n", "version must be an integer"),
("version: 1\nthemes: []\n", "no themes"),
("- a\n- b\n", "expected a mapping"),
(
"version: 1\nthemes:\n - slug: a\n definition: x\nretired: [a]\n",
"retired slugs still active",
),
],
)
def test_rejects_bad_files(self, text, msg):
with pytest.raises(VocabError, match=msg):
parse(text, source="t")
class TestPackagedVocab:
def test_loads_and_is_sane(self):
v = load()
assert v.version >= 1 and 40 <= len(v) <= 150
assert all(t.definition.endswith(".") for t in v.themes)
assert all(len(t.synonyms) >= 1 for t in v.themes)
for must in (
"telehealth",
"care-management",
"conversion-factor",
"practice-expense",
"palliative-care",
"skin-substitutes",
):
assert must in v
def test_load_from_path(self, tmp_path):
p = tmp_path / "v.yaml"
p.write_text(GOOD)
assert load(p).source == str(p)
class TestTagHelpers:
def test_split(self):
ours, rest = split_tags(
["llm:telehealth", "topic:palliative", "llm:care-management", "docket:x"]
)
assert ours == ["llm:telehealth", "llm:care-management"] and rest == [
"topic:palliative",
"docket:x",
]
assert is_llm_tag("llm:x") and not is_llm_tag("llmx:y")