merge: #574/#575 — P35 theme vocabulary + tagging chain, runner and write-back (refs #574, #575, #576)
All checks were successful
CI / lint (push) Successful in 33s
CI / notebooks-smoke (push) Successful in 1m48s
CI / test (push) Successful in 2m26s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 16s
Infra CI / notebooks (push) Successful in 51s
Infra CI / api (push) Successful in 1m10s
Infra CI / docs (push) Successful in 1m43s
Infra CI / mc (push) Successful in 16s
Deploy / report (push) Successful in 13s
Infra CI / llm (push) Successful in 44s
All checks were successful
CI / lint (push) Successful in 33s
CI / notebooks-smoke (push) Successful in 1m48s
CI / test (push) Successful in 2m26s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 16s
Infra CI / notebooks (push) Successful in 51s
Infra CI / api (push) Successful in 1m10s
Infra CI / docs (push) Successful in 1m43s
Infra CI / mc (push) Successful in 16s
Deploy / report (push) Successful in 13s
Infra CI / llm (push) Successful in 44s
This commit is contained in:
81
dev/scripts/llm_vocab_candidates.py
Normal file
81
dev/scripts/llm_vocab_candidates.py
Normal file
@@ -0,0 +1,81 @@
|
||||
"""Seed a theme-vocabulary revision from Federal Register section headings (#574).
|
||||
|
||||
Walks ``fr_anchors`` for heading-like paragraphs ("B. Determination of
|
||||
Practice Expense…") in every PFS rule whose rule year is at or after
|
||||
``--since`` (default 2018 — the regulations.gov comment dockets start in
|
||||
2017), stems the titles, and prints the stems ordered by how many rule
|
||||
years carry them, with an example title each. The output is raw material
|
||||
for a human editing ``src/llm/vocab/themes.yaml`` — it is not the vocab.
|
||||
|
||||
Usage:
|
||||
uv run python dev/scripts/llm_vocab_candidates.py [--since 2018] [--top 200]
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
from collections import defaultdict
|
||||
|
||||
_HEADING = re.compile(r"^(?:[IVX]{1,4}|[A-Z]|\d{1,2}|[a-z])\. +([A-Z][^.]{6,120})$")
|
||||
_STOP = re.compile(
|
||||
r"\b(cy|20\d\d|proposed|final|rule|for|the|of|and|to|in|a|an|under|pfs|physician fee schedule)\b"
|
||||
)
|
||||
_SKIP = ("authority citation", "on page", "column", "paragraph")
|
||||
|
||||
|
||||
def stem(title: str) -> str:
|
||||
norm = re.sub(r"[^a-z0-9 ]", " ", title.lower())
|
||||
norm = _STOP.sub(" ", norm)
|
||||
return re.sub(r"\s+", " ", norm).strip()
|
||||
|
||||
|
||||
def candidates(store, *, since: int) -> list[tuple[int, str, str]]:
|
||||
"""``(n_rule_years, stem, example title)`` sorted by n desc."""
|
||||
from pfs.descriptors import rule_year_of
|
||||
|
||||
con = store._con() # noqa: SLF001
|
||||
years: dict[str, int] = {}
|
||||
for key, title, published in con.execute(
|
||||
"SELECT key, title, date_published FROM items WHERE item_type = 'rule'"
|
||||
).fetchall():
|
||||
# rule_year_of reads the payment year from the title, else falls
|
||||
# back to publication year + 1 (PFS rules publish July–December).
|
||||
years[key] = int(rule_year_of(title or "", published or ""))
|
||||
recent = {k for k, y in years.items() if y and y >= since}
|
||||
by: dict[str, set[int]] = defaultdict(set)
|
||||
example: dict[str, str] = {}
|
||||
for key, text in con.execute(
|
||||
"SELECT item_key, text FROM fr_anchors WHERE length(text) < 140"
|
||||
).fetchall():
|
||||
if key not in recent:
|
||||
continue
|
||||
m = _HEADING.match((text or "").strip())
|
||||
if not m:
|
||||
continue
|
||||
title = m.group(1).strip()
|
||||
s = stem(title)
|
||||
if len(s) < 5 or any(s.startswith(x) or x in s for x in _SKIP):
|
||||
continue
|
||||
by[s].add(years[key])
|
||||
example.setdefault(s, title)
|
||||
return sorted(
|
||||
((len(v), k, example[k]) for k, v in by.items()), key=lambda r: (-r[0], r[1])
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
ap.add_argument("--since", type=int, default=2018)
|
||||
ap.add_argument("--top", type=int, default=200)
|
||||
args = ap.parse_args()
|
||||
from conf import connect
|
||||
|
||||
store = connect.bib()
|
||||
for n, s, ex in candidates(store, since=args.since)[: args.top]:
|
||||
print(f"{n:3d} {ex}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack api serve
|
||||
sidebar_position: 61
|
||||
sidebar_position: 63
|
||||
---
|
||||
|
||||
# `stack api serve`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack api
|
||||
sidebar_position: 60
|
||||
sidebar_position: 62
|
||||
---
|
||||
|
||||
# `stack api`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack db comment
|
||||
sidebar_position: 54
|
||||
sidebar_position: 56
|
||||
---
|
||||
|
||||
# `stack db comment`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack db inspect
|
||||
sidebar_position: 55
|
||||
sidebar_position: 57
|
||||
---
|
||||
|
||||
# `stack db inspect`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack db
|
||||
sidebar_position: 53
|
||||
sidebar_position: 55
|
||||
---
|
||||
|
||||
# `stack db`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack docs build
|
||||
sidebar_position: 57
|
||||
sidebar_position: 59
|
||||
---
|
||||
|
||||
# `stack docs build`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack docs generate
|
||||
sidebar_position: 59
|
||||
sidebar_position: 61
|
||||
---
|
||||
|
||||
# `stack docs generate`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack docs serve
|
||||
sidebar_position: 58
|
||||
sidebar_position: 60
|
||||
---
|
||||
|
||||
# `stack docs serve`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack docs
|
||||
sidebar_position: 56
|
||||
sidebar_position: 58
|
||||
---
|
||||
|
||||
# `stack docs`
|
||||
|
||||
38
docs/docs/cli/llm-tag.md
Normal file
38
docs/docs/cli/llm-tag.md
Normal file
@@ -0,0 +1,38 @@
|
||||
---
|
||||
title: stack llm tag
|
||||
sidebar_position: 54
|
||||
---
|
||||
|
||||
# `stack llm tag`
|
||||
|
||||
```
|
||||
Usage: stack llm tag [OPTIONS]
|
||||
|
||||
Closed-vocabulary theme tagging of one docket's comments (P35): shortlist by
|
||||
similarity to the theme cards, judge each candidate yes/no on the largest live
|
||||
host with the scoring chunk as evidence, record state in the item's
|
||||
extra_json, and replace its llm: tags. Resumable — unchanged comments are
|
||||
skipped unless --force.
|
||||
|
||||
╭─ Options ────────────────────────────────────────────────────────────────────╮
|
||||
│ * --docket TEXT regulations.gov docket id, e.g. │
|
||||
│ CMS-2026-2377. │
|
||||
│ [required] │
|
||||
│ --limit INTEGER Stop after N comments (newest first; 0 = │
|
||||
│ all). │
|
||||
│ [default: 0] │
|
||||
│ --force Re-tag comments whose stored state is │
|
||||
│ current. │
|
||||
│ --dry-run Compute and print, write nothing. │
|
||||
│ --top INTEGER Themes shortlisted per comment before │
|
||||
│ judging. │
|
||||
│ [default: 8] │
|
||||
│ --max-tags INTEGER Most themes written per comment. │
|
||||
│ [default: 4] │
|
||||
│ --min-confidence FLOAT Shortlist similarity below which a │
|
||||
│ judged yes is not written. │
|
||||
│ [default: 0.0] │
|
||||
│ --verbose Print every comment's tags. │
|
||||
│ --help Show this message and exit. │
|
||||
╰──────────────────────────────────────────────────────────────────────────────╯
|
||||
```
|
||||
21
docs/docs/cli/llm-vocab.md
Normal file
21
docs/docs/cli/llm-vocab.md
Normal file
@@ -0,0 +1,21 @@
|
||||
---
|
||||
title: stack llm vocab
|
||||
sidebar_position: 53
|
||||
---
|
||||
|
||||
# `stack llm vocab`
|
||||
|
||||
```
|
||||
Usage: stack llm vocab [OPTIONS]
|
||||
|
||||
The closed theme vocabulary for comment tagging (#574): validate it and list
|
||||
its slugs, or show one theme's definition, synonyms and the FR section stems
|
||||
it was seeded from.
|
||||
|
||||
╭─ Options ────────────────────────────────────────────────────────────────────╮
|
||||
│ --path TEXT A vocabulary file to validate instead of the packaged │
|
||||
│ one. │
|
||||
│ --slug TEXT Show one theme in full. │
|
||||
│ --help Show this message and exit. │
|
||||
╰──────────────────────────────────────────────────────────────────────────────╯
|
||||
```
|
||||
@@ -19,5 +19,14 @@ Usage: stack llm [OPTIONS] COMMAND [ARGS]...
|
||||
│ hosts Show the Ollama fleet: declared VRAM, liveness, models, and which │
|
||||
│ host + model would answer a chat right now. │
|
||||
│ serve Serve the SSO-guarded chat UI (llm.api:app). │
|
||||
│ vocab The closed theme vocabulary for comment tagging (#574): validate it │
|
||||
│ and list its slugs, or show one theme's definition, synonyms and │
|
||||
│ the FR │
|
||||
│ section stems it was seeded from. │
|
||||
│ tag Closed-vocabulary theme tagging of one docket's comments (P35): │
|
||||
│ shortlist by similarity to the theme cards, judge each candidate │
|
||||
│ yes/no on the largest live host with the scoring chunk as evidence, │
|
||||
│ record state in the item's extra_json, and replace its llm: tags. │
|
||||
│ Resumable — unchanged comments are skipped unless --force. │
|
||||
╰──────────────────────────────────────────────────────────────────────────────╯
|
||||
```
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail attach-smarthost
|
||||
sidebar_position: 102
|
||||
sidebar_position: 104
|
||||
---
|
||||
|
||||
# `stack mail attach-smarthost`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail dkim-export
|
||||
sidebar_position: 101
|
||||
sidebar_position: 103
|
||||
---
|
||||
|
||||
# `stack mail dkim-export`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail dns
|
||||
sidebar_position: 100
|
||||
sidebar_position: 102
|
||||
---
|
||||
|
||||
# `stack mail dns`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail down
|
||||
sidebar_position: 98
|
||||
sidebar_position: 100
|
||||
---
|
||||
|
||||
# `stack mail down`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail provision
|
||||
sidebar_position: 96
|
||||
sidebar_position: 98
|
||||
---
|
||||
|
||||
# `stack mail provision`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail rotate-creds
|
||||
sidebar_position: 103
|
||||
sidebar_position: 105
|
||||
---
|
||||
|
||||
# `stack mail rotate-creds`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail seed-mailboxes
|
||||
sidebar_position: 104
|
||||
sidebar_position: 106
|
||||
---
|
||||
|
||||
# `stack mail seed-mailboxes`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail status
|
||||
sidebar_position: 99
|
||||
sidebar_position: 101
|
||||
---
|
||||
|
||||
# `stack mail status`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail up
|
||||
sidebar_position: 97
|
||||
sidebar_position: 99
|
||||
---
|
||||
|
||||
# `stack mail up`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail wire-git
|
||||
sidebar_position: 105
|
||||
sidebar_position: 107
|
||||
---
|
||||
|
||||
# `stack mail wire-git`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack mail
|
||||
sidebar_position: 95
|
||||
sidebar_position: 97
|
||||
---
|
||||
|
||||
# `stack mail`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack perf show
|
||||
sidebar_position: 63
|
||||
sidebar_position: 65
|
||||
---
|
||||
|
||||
# `stack perf show`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack perf
|
||||
sidebar_position: 62
|
||||
sidebar_position: 64
|
||||
---
|
||||
|
||||
# `stack perf`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs cpt-ingest
|
||||
sidebar_position: 77
|
||||
sidebar_position: 79
|
||||
---
|
||||
|
||||
# `stack pfs cpt-ingest`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs elements
|
||||
sidebar_position: 69
|
||||
sidebar_position: 71
|
||||
---
|
||||
|
||||
# `stack pfs elements`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs exposure
|
||||
sidebar_position: 74
|
||||
sidebar_position: 76
|
||||
---
|
||||
|
||||
# `stack pfs exposure`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs families
|
||||
sidebar_position: 71
|
||||
sidebar_position: 73
|
||||
---
|
||||
|
||||
# `stack pfs families`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs guidance
|
||||
sidebar_position: 72
|
||||
sidebar_position: 74
|
||||
---
|
||||
|
||||
# `stack pfs guidance`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs lineage
|
||||
sidebar_position: 70
|
||||
sidebar_position: 72
|
||||
---
|
||||
|
||||
# `stack pfs lineage`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs reaction
|
||||
sidebar_position: 73
|
||||
sidebar_position: 75
|
||||
---
|
||||
|
||||
# `stack pfs reaction`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs review
|
||||
sidebar_position: 76
|
||||
sidebar_position: 78
|
||||
---
|
||||
|
||||
# `stack pfs review`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs utilization
|
||||
sidebar_position: 75
|
||||
sidebar_position: 77
|
||||
---
|
||||
|
||||
# `stack pfs utilization`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack pfs
|
||||
sidebar_position: 68
|
||||
sidebar_position: 70
|
||||
---
|
||||
|
||||
# `stack pfs`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma eligible
|
||||
sidebar_position: 89
|
||||
sidebar_position: 91
|
||||
---
|
||||
|
||||
# `stack prisma eligible`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma export
|
||||
sidebar_position: 86
|
||||
sidebar_position: 88
|
||||
---
|
||||
|
||||
# `stack prisma export`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma extract
|
||||
sidebar_position: 90
|
||||
sidebar_position: 92
|
||||
---
|
||||
|
||||
# `stack prisma extract`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma fetch
|
||||
sidebar_position: 92
|
||||
sidebar_position: 94
|
||||
---
|
||||
|
||||
# `stack prisma fetch`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma flow
|
||||
sidebar_position: 91
|
||||
sidebar_position: 93
|
||||
---
|
||||
|
||||
# `stack prisma flow`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma init
|
||||
sidebar_position: 85
|
||||
sidebar_position: 87
|
||||
---
|
||||
|
||||
# `stack prisma init`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma ping-llm
|
||||
sidebar_position: 87
|
||||
sidebar_position: 89
|
||||
---
|
||||
|
||||
# `stack prisma ping-llm`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma run
|
||||
sidebar_position: 93
|
||||
sidebar_position: 95
|
||||
---
|
||||
|
||||
# `stack prisma run`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma screen
|
||||
sidebar_position: 88
|
||||
sidebar_position: 90
|
||||
---
|
||||
|
||||
# `stack prisma screen`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma vpn
|
||||
sidebar_position: 94
|
||||
sidebar_position: 96
|
||||
---
|
||||
|
||||
# `stack prisma vpn`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack prisma
|
||||
sidebar_position: 84
|
||||
sidebar_position: 86
|
||||
---
|
||||
|
||||
# `stack prisma`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack rec list
|
||||
sidebar_position: 65
|
||||
sidebar_position: 67
|
||||
---
|
||||
|
||||
# `stack rec list`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack rec opps
|
||||
sidebar_position: 67
|
||||
sidebar_position: 69
|
||||
---
|
||||
|
||||
# `stack rec opps`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack rec pfs
|
||||
sidebar_position: 66
|
||||
sidebar_position: 68
|
||||
---
|
||||
|
||||
# `stack rec pfs`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack rec
|
||||
sidebar_position: 64
|
||||
sidebar_position: 66
|
||||
---
|
||||
|
||||
# `stack rec`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot dump-schema
|
||||
sidebar_position: 79
|
||||
sidebar_position: 81
|
||||
---
|
||||
|
||||
# `stack zot dump-schema`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot fix-dates
|
||||
sidebar_position: 80
|
||||
sidebar_position: 82
|
||||
---
|
||||
|
||||
# `stack zot fix-dates`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot fix-fields
|
||||
sidebar_position: 82
|
||||
sidebar_position: 84
|
||||
---
|
||||
|
||||
# `stack zot fix-fields`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot fix-keys
|
||||
sidebar_position: 81
|
||||
sidebar_position: 83
|
||||
---
|
||||
|
||||
# `stack zot fix-keys`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot verify-parity
|
||||
sidebar_position: 83
|
||||
sidebar_position: 85
|
||||
---
|
||||
|
||||
# `stack zot verify-parity`
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: stack zot
|
||||
sidebar_position: 78
|
||||
sidebar_position: 80
|
||||
---
|
||||
|
||||
# `stack zot`
|
||||
|
||||
46
docs/superpowers/specs/2026-09-22-llm-tagging-design.md
Normal file
46
docs/superpowers/specs/2026-09-22-llm-tagging-design.md
Normal file
@@ -0,0 +1,46 @@
|
||||
# P35 — closed-vocabulary RAG tagging of rulemaking comments into bib/Zotero
|
||||
|
||||
Tracker: milestone P35 (#574 vocabulary, #575 chain + batch runner, #576 write-back + Zotero sync, #577 golden-set gate). Parent spec: `2026-07-16-llm-module-design.md` §P35. Builds on the comment pipeline (P47 dockets/seals, `.state/comments/<docket>/<id>/combined.md` bodies), the pgvector index (`comments` collection, chunk metadata `docket`/`comment_id`/`item_key`/`date`), `llm.classify.closed_vocab_classifier` (one slug or `none` from the largest live host), `llm.pool` (dead-host drop, #796), and the bib tag conventions (`bib.tag` namespaces, `Store.add_tags`, `remove_tag`, nightly `zotero-sync --tag`).
|
||||
|
||||
## Goal
|
||||
|
||||
Every regulations.gov comment on a PFS docket carries a small set of **theme tags** from a closed vocabulary — which parts of the rule the letter is about — written as `llm:<slug>` tags on its bib item so they reach Zotero with the nightly sync, are filterable in `stack bib query` and the search UI, and give #254–#256 (position classification, stakeholder segmentation, provision mapping) a stable thematic axis. Self-hosted inference only (the llm module never calls a cloud API).
|
||||
|
||||
## Decisions
|
||||
|
||||
1. **The vocabulary is a versioned YAML file in the repo**, `src/llm/vocab/themes.yaml`, hand-curated. It is *seeded* from the Federal Register section headings of the CY2018–CY2027 PFS rules (mined from `fr_anchors`, `dev/scripts/llm_vocab_candidates.py`) and from the hand code families, but the file is the artifact; a run records the `version` it was tagged with. Slugs are kebab-case, ~60 themes, each with `label`, `definition` (one sentence the model sees), `synonyms` (phrases that appear in letters), and `sections` (the FR heading stems it maps to). Mechanical bib namespaces (`docket:`, `year:`, `rule:`, `reg-docket:`, `enriched:`, `code:`, `family:`) are excluded by construction — they are provenance, not themes.
|
||||
2. **Multi-label, evidence-first.** A comment gets zero to `max_tags` (default 4) themes. The chain does not ask the model to pick from 60 slugs cold: it first shortlists candidates by embedding similarity between the comment's chunks and each theme's definition+synonyms (the theme "cards" are embedded once per vocab version), then asks the model a yes/no per candidate with the comment's most relevant chunk as evidence, on the largest live host, `temperature 0`. Anything outside `{yes, no}` counts as `no`. This is the closed-vocab discipline of `llm.classify` applied per theme, and the evidence chunk is stored so a human can audit any tag.
|
||||
3. **Resumable, docket-scoped, idempotent.** State lives in bib: `extra_json["llm_tags"] = {"version": v, "model": m, "tags": {slug: {"confidence": c, "evidence_chunk": seq}}, "tagged_at": iso}`. A comment is skipped when its stored `version`+`model`+`content_hash` match; `--force` redoes it. `stack llm tag --docket D [--limit N] [--force]` walks a docket newest-first through the pool (`HostPool` fan-out; a dead host is dropped, #796).
|
||||
4. **Write-back replaces only `llm:` tags.** `apply_tags(store, key, slugs)` removes every existing `llm:*` tag on the item and adds the new set — hand tags (`topic:`, `stance:`, …) are never touched. Confidence below `min_confidence` (default 0.6) is recorded in `extra_json` but not written as a tag. The nightly `zotero-sync` already carries tags; `--tag llm:` scopes a manual push.
|
||||
5. **Golden set gates fan-out.** `tests/llm/golden_themes.yaml` holds ~200 hand-labelled comments across ≥3 dockets (comment id, expected slugs). `stack llm eval-tags` computes per-slug precision/recall and the abstain rate; CI runs it mocked (the harness only), the nightly `llm-golden` job runs it live. Corpus-wide tagging beyond the pilot docket is blocked until micro-F1 ≥ 0.70 and no theme with support ≥ 5 has precision < 0.5.
|
||||
6. **Observability** reuses `llm.metrics` (`dispatch`, a `tagged` counter) — the P34 Grafana panel gains a tagging row later, not in this milestone.
|
||||
|
||||
## Data flow
|
||||
|
||||
```
|
||||
themes.yaml ──embed cards──▶ pgvector "themes" collection (version-stamped)
|
||||
comment chunks (pgvector) ──similarity vs cards──▶ shortlist (top 8 themes)
|
||||
shortlist × evidence chunk ──yes/no on largest host──▶ tags + confidence
|
||||
└──▶ bib extra_json.llm_tags (state)
|
||||
└──▶ bib tags llm:<slug> (write-back)
|
||||
└──▶ nightly zotero-sync
|
||||
golden_themes.yaml ──▶ stack llm eval-tags ──▶ precision/recall report (gate)
|
||||
```
|
||||
|
||||
## Modules
|
||||
|
||||
- `src/llm/vocab.py` — `Theme` dataclass, `load()` (validates unique slugs, kebab-case, non-empty definitions), `version()` (the file's `version` field), `cards()` (text per theme for embedding).
|
||||
- `src/llm/tagger.py` — `shortlist()`, `judge()`, `tag_comment()`, `run()` (batch), `apply_tags()`; pure pieces take injected callables so tests never touch Ollama or pgvector.
|
||||
- `src/cli/llm.py` — `vocab` (list/validate), `tag` (batch), `eval-tags` (gate).
|
||||
- `dev/scripts/llm_vocab_candidates.py` — the heading miner that seeds a vocab revision (not run in CI).
|
||||
|
||||
## Slices
|
||||
|
||||
1. **#574** vocabulary YAML + `llm.vocab` + `stack llm vocab` (this slice).
|
||||
2. **#575** theme cards in pgvector, shortlist + judge chain, batch runner with state.
|
||||
3. **#576** write-back + sync verification on a pilot docket (CMS-2026-2377).
|
||||
4. **#577** golden set + `eval-tags` + the gate; then corpus-wide run.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Position/stance classification (#254's other half — reuse `stance:` from P43 tooling later), stakeholder segmentation (#255), cloud LLMs, tagging rules or the Zotero corpus (comments only).
|
||||
@@ -613,6 +613,24 @@ class Store:
|
||||
con.commit()
|
||||
return added
|
||||
|
||||
def merge_extra(self, item_key: str, updates: dict[str, Any]) -> None:
|
||||
"""Merge *updates* into the item's ``extra_json`` (top-level keys
|
||||
replace; everything else is kept). A missing item is a no-op."""
|
||||
con = self._con()
|
||||
row = con.execute(
|
||||
"SELECT extra_json FROM items WHERE key = ?", (item_key,)
|
||||
).fetchone()
|
||||
if row is None:
|
||||
return
|
||||
current = json.loads(row[0] or "{}") if row[0] else {}
|
||||
current.update(updates)
|
||||
con.execute(
|
||||
"UPDATE items SET extra_json = ?, "
|
||||
"updated_at = strftime('%Y-%m-%dT%H:%M:%SZ','now') WHERE key = ?",
|
||||
(json.dumps(current), item_key),
|
||||
)
|
||||
con.commit()
|
||||
|
||||
def remove_tag(self, item_key: str, tag: str) -> None:
|
||||
"""Remove a tag from an item."""
|
||||
con = self._con()
|
||||
|
||||
135
src/cli/llm.py
135
src/cli/llm.py
@@ -1,5 +1,7 @@
|
||||
"""stack llm — local RAG over the library (index + fleet + chat serve)."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
import typer
|
||||
|
||||
app = typer.Typer(no_args_is_help=True)
|
||||
@@ -203,3 +205,136 @@ def serve(
|
||||
import uvicorn
|
||||
|
||||
uvicorn.run("llm.api:app", host=host, port=port, log_level="info")
|
||||
|
||||
|
||||
@app.command()
|
||||
def vocab(
|
||||
path: str = typer.Option(
|
||||
"", "--path", help="A vocabulary file to validate instead of the packaged one."
|
||||
),
|
||||
slug: str = typer.Option("", "--slug", help="Show one theme in full."),
|
||||
) -> None:
|
||||
"""The closed theme vocabulary for comment tagging (#574): validate it
|
||||
and list its slugs, or show one theme's definition, synonyms and the FR
|
||||
section stems it was seeded from."""
|
||||
from llm.vocab import VocabError, load
|
||||
|
||||
try:
|
||||
v = load(path or None)
|
||||
except VocabError as exc:
|
||||
typer.echo(f"invalid vocabulary: {exc}")
|
||||
raise typer.Exit(1) from exc
|
||||
if slug:
|
||||
if slug not in v:
|
||||
raise typer.BadParameter(f"unknown slug {slug!r}")
|
||||
t = v.get(slug)
|
||||
typer.echo(f"{t.tag} {t.label}")
|
||||
typer.echo(f" {t.definition}")
|
||||
typer.echo(f" synonyms: {', '.join(t.synonyms) or '—'}")
|
||||
typer.echo(f" sections: {' | '.join(t.sections) or '—'}")
|
||||
return
|
||||
for t in v.themes:
|
||||
typer.echo(f"{t.slug:<34} {t.label}")
|
||||
typer.echo(
|
||||
f"vocabulary v{v.version}: {len(v)} themes, {len(v.retired)} retired ({v.source})"
|
||||
)
|
||||
|
||||
|
||||
def _tagging_runtime(cfg: Any) -> dict[str, Any]:
|
||||
"""Everything `stack llm tag` needs from the live stack — one place
|
||||
the tests monkeypatch: the bib store, the pgvector chunk loader, the
|
||||
theme-card vectors (embedded once per run through the pool), the
|
||||
yes/no judge on the largest live host, and the model name recorded
|
||||
in each item's state."""
|
||||
from conf.connect import bib
|
||||
from llm.index import _engine
|
||||
from llm.pool import HostPool, PoolEmbeddings, pick_model
|
||||
from llm.tagger import card_vectors, make_judge, pg_chunks
|
||||
from llm.vocab import load as load_vocab
|
||||
|
||||
vocab = load_vocab()
|
||||
pool = HostPool.from_config(cfg)
|
||||
pool.check(cfg.embed_model)
|
||||
cards = card_vectors(vocab, PoolEmbeddings(pool, cfg.embed_model).embed_documents)
|
||||
judge = make_judge(cfg, pool)
|
||||
with pool.acquire_generation() as host:
|
||||
model = pick_model(cfg, pool, host)
|
||||
return {
|
||||
"store": bib(),
|
||||
"vocab": vocab,
|
||||
"load_chunks": pg_chunks(_engine(cfg)),
|
||||
"card_vecs": cards,
|
||||
"judge": judge,
|
||||
"model": model,
|
||||
}
|
||||
|
||||
|
||||
@app.command()
|
||||
def tag(
|
||||
docket: str = typer.Option(
|
||||
..., "--docket", help="regulations.gov docket id, e.g. CMS-2026-2377."
|
||||
),
|
||||
limit: int = typer.Option(
|
||||
0, "--limit", help="Stop after N comments (newest first; 0 = all)."
|
||||
),
|
||||
force: bool = typer.Option(
|
||||
False, "--force", help="Re-tag comments whose stored state is current."
|
||||
),
|
||||
dry_run: bool = typer.Option(
|
||||
False, "--dry-run", help="Compute and print, write nothing."
|
||||
),
|
||||
top: int = typer.Option(
|
||||
8, "--top", help="Themes shortlisted per comment before judging."
|
||||
),
|
||||
max_tags: int = typer.Option(
|
||||
4, "--max-tags", help="Most themes written per comment."
|
||||
),
|
||||
min_confidence: float = typer.Option(
|
||||
0.0,
|
||||
"--min-confidence",
|
||||
help="Shortlist similarity below which a judged yes is not written.",
|
||||
),
|
||||
verbose: bool = typer.Option(
|
||||
False, "--verbose", help="Print every comment's tags."
|
||||
),
|
||||
) -> None:
|
||||
"""Closed-vocabulary theme tagging of one docket's comments (P35):
|
||||
shortlist by similarity to the theme cards, judge each candidate
|
||||
yes/no on the largest live host with the scoring chunk as evidence,
|
||||
record state in the item's extra_json, and replace its llm: tags.
|
||||
Resumable — unchanged comments are skipped unless --force."""
|
||||
from llm import config as llm_config
|
||||
from llm.tagger import run as run_tagging
|
||||
|
||||
cfg = llm_config.load()
|
||||
rt = _tagging_runtime(cfg)
|
||||
|
||||
def progress(key: str, result: Any, why: str) -> None:
|
||||
if result is None:
|
||||
if verbose:
|
||||
typer.echo(f"{key} {why}")
|
||||
return
|
||||
if verbose or dry_run:
|
||||
typer.echo(f"{key} {', '.join(result.slugs) or '(none)'}")
|
||||
|
||||
stats = run_tagging(
|
||||
rt["store"],
|
||||
vocab=rt["vocab"],
|
||||
docket=docket,
|
||||
load_chunks=rt["load_chunks"],
|
||||
card_vecs=rt["card_vecs"],
|
||||
judge=rt["judge"],
|
||||
model=rt["model"],
|
||||
limit=limit,
|
||||
force=force,
|
||||
dry_run=dry_run,
|
||||
top=top,
|
||||
max_tags=max_tags,
|
||||
min_confidence=min_confidence,
|
||||
progress=progress,
|
||||
)
|
||||
typer.echo(
|
||||
f"tag {docket} (vocab v{rt['vocab'].version}, {rt['model']}{', dry run' if dry_run else ''}): "
|
||||
f"seen={stats['seen']} tagged={stats['tagged']} unchanged={stats['skipped_state']} "
|
||||
f"no_chunks={stats['no_chunks']} tags_written={stats['tags_written']}"
|
||||
)
|
||||
|
||||
369
src/llm/tagger.py
Normal file
369
src/llm/tagger.py
Normal file
@@ -0,0 +1,369 @@
|
||||
"""Closed-vocabulary theme tagging of rulemaking comments (P35, #575/#576).
|
||||
|
||||
For one comment the chain is: its already-indexed chunks (text + vector,
|
||||
``langchain_pg_embedding`` ``comments`` collection) → cosine similarity
|
||||
against every theme card (``llm.vocab.Theme.card`` embedded once per run)
|
||||
→ the ``top`` themes shortlisted, each with the chunk that scored it →
|
||||
one yes/no judgement per candidate on the largest live host, with that
|
||||
chunk as the evidence → up to ``max_tags`` accepted themes, confidence =
|
||||
the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``.
|
||||
|
||||
State lives on the bib item (``extra_json["llm_tags"]``: vocab version,
|
||||
model, content hash, the accepted tags with confidence and evidence
|
||||
chunk, and the shortlist with every verdict) so a run is resumable and a
|
||||
re-run skips comments whose version + model + content are unchanged.
|
||||
Write-back replaces exactly the item's ``llm:*`` tags and never touches
|
||||
hand tags.
|
||||
|
||||
Every collaborator is injected (``embed``, ``judge``, ``load_chunks``) so
|
||||
the pure pieces run in tests without Ollama or pgvector; the CLI wires
|
||||
the real ones (``make_judge``, ``card_vectors``, ``pg_chunks``).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
from dataclasses import asdict, dataclass
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Iterable, Mapping, Sequence
|
||||
|
||||
from llm.vocab import Theme, Vocab, split_tags
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
Vector = Sequence[float]
|
||||
Embed = Callable[[Sequence[str]], list[list[float]]]
|
||||
Judge = Callable[[Theme, str], bool | None]
|
||||
ChunkLoader = Callable[[str], list[tuple[int, str, list[float]]]]
|
||||
|
||||
STATE_KEY = "llm_tags"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Candidate:
|
||||
slug: str
|
||||
score: float
|
||||
chunk_idx: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TagInfo:
|
||||
confidence: float
|
||||
evidence_chunk: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TagResult:
|
||||
key: str
|
||||
tags: dict[str, TagInfo]
|
||||
shortlisted: tuple[Candidate, ...]
|
||||
judged: dict[str, bool | None]
|
||||
content_hash: str
|
||||
|
||||
@property
|
||||
def slugs(self) -> list[str]:
|
||||
return sorted(self.tags, key=lambda s: -self.tags[s].confidence)
|
||||
|
||||
|
||||
# ── pure pieces ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def cosine(a: Vector, b: Vector) -> float:
|
||||
dot = sum(x * y for x, y in zip(a, b))
|
||||
na = math.sqrt(sum(x * x for x in a))
|
||||
nb = math.sqrt(sum(y * y for y in b))
|
||||
return dot / (na * nb) if na and nb else 0.0
|
||||
|
||||
|
||||
def shortlist(
|
||||
chunk_vecs: Sequence[Vector], card_vecs: Mapping[str, Vector], *, top: int = 8
|
||||
) -> list[Candidate]:
|
||||
"""The ``top`` themes by their best cosine similarity to any chunk,
|
||||
each with the index of the chunk that scored it (the evidence)."""
|
||||
if not chunk_vecs:
|
||||
return []
|
||||
out: list[Candidate] = []
|
||||
for slug, cv in card_vecs.items():
|
||||
best_i, best = 0, -1.0
|
||||
for i, v in enumerate(chunk_vecs):
|
||||
s = cosine(v, cv)
|
||||
if s > best:
|
||||
best_i, best = i, s
|
||||
out.append(Candidate(slug, round(best, 4), best_i))
|
||||
out.sort(key=lambda c: (-c.score, c.slug))
|
||||
return out[:top]
|
||||
|
||||
|
||||
_SYSTEM = (
|
||||
"You decide whether a public comment letter on a Medicare physician fee "
|
||||
"schedule rule addresses a given theme. Answer with exactly one word: yes "
|
||||
"or no. Say yes only when the excerpt actually discusses the theme, not "
|
||||
"when it merely mentions a related word in passing."
|
||||
)
|
||||
|
||||
|
||||
def judge_prompt(theme: Theme, evidence: str) -> list[dict]:
|
||||
return [
|
||||
{"role": "system", "content": _SYSTEM},
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
f"Theme: {theme.label}\nDefinition: {theme.definition}\n"
|
||||
f"Phrases letters use: {', '.join(theme.synonyms) or '—'}\n\n"
|
||||
f"Excerpt from the comment:\n{evidence.strip()}\n\n"
|
||||
"Does this comment address the theme? Answer yes or no."
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def parse_yes_no(reply: str) -> bool | None:
|
||||
word = (reply or "").strip().strip(".!\"'").lower().split()
|
||||
if not word:
|
||||
return None
|
||||
if word[0] in ("yes", "y"):
|
||||
return True
|
||||
if word[0] in ("no", "n"):
|
||||
return False
|
||||
return None
|
||||
|
||||
|
||||
def content_hash(chunks: Sequence[str]) -> str:
|
||||
h = hashlib.sha1() # noqa: S324 — change detection, not security
|
||||
for c in chunks:
|
||||
h.update(c.encode("utf-8", "replace"))
|
||||
h.update(b"\x00")
|
||||
return h.hexdigest()[:16]
|
||||
|
||||
|
||||
def tag_comment(
|
||||
key: str,
|
||||
chunks: Sequence[str],
|
||||
chunk_vecs: Sequence[Vector],
|
||||
*,
|
||||
vocab: Vocab,
|
||||
card_vecs: Mapping[str, Vector],
|
||||
judge: Judge,
|
||||
top: int = 8,
|
||||
max_tags: int = 4,
|
||||
min_confidence: float = 0.0,
|
||||
) -> TagResult:
|
||||
cands = shortlist(chunk_vecs, card_vecs, top=top)
|
||||
judged: dict[str, bool | None] = {}
|
||||
tags: dict[str, TagInfo] = {}
|
||||
for c in cands:
|
||||
if c.slug not in vocab:
|
||||
continue
|
||||
verdict = judge(vocab.get(c.slug), chunks[c.chunk_idx])
|
||||
judged[c.slug] = verdict
|
||||
if verdict and c.score >= min_confidence and len(tags) < max_tags:
|
||||
tags[c.slug] = TagInfo(confidence=c.score, evidence_chunk=c.chunk_idx)
|
||||
return TagResult(key, tags, tuple(cands), judged, content_hash(chunks))
|
||||
|
||||
|
||||
def state_matches(
|
||||
extra: Mapping[str, Any], *, version: int, model: str, digest: str
|
||||
) -> bool:
|
||||
st = extra.get(STATE_KEY) or {}
|
||||
return (
|
||||
st.get("version") == version
|
||||
and st.get("model") == model
|
||||
and st.get("content_hash") == digest
|
||||
)
|
||||
|
||||
|
||||
def state_payload(result: TagResult, *, version: int, model: str) -> dict:
|
||||
return {
|
||||
"version": version,
|
||||
"model": model,
|
||||
"content_hash": result.content_hash,
|
||||
"tagged_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
"tags": {s: asdict(i) for s, i in result.tags.items()},
|
||||
"shortlist": [
|
||||
{
|
||||
"slug": c.slug,
|
||||
"score": c.score,
|
||||
"chunk": c.chunk_idx,
|
||||
"verdict": result.judged.get(c.slug),
|
||||
}
|
||||
for c in result.shortlisted
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
# ── bib side ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def apply_tags(
|
||||
store: Any, key: str, slugs: Iterable[str], *, vocab: Vocab
|
||||
) -> tuple[int, int]:
|
||||
"""Make the item's ``llm:*`` tags exactly ``slugs``; hand tags untouched.
|
||||
Returns ``(added, removed)``."""
|
||||
want = {vocab.tag(s) for s in slugs}
|
||||
have, _hand = split_tags(store.get(key).tags)
|
||||
removed = 0
|
||||
for t in have:
|
||||
if t not in want:
|
||||
store.remove_tag(key, t)
|
||||
removed += 1
|
||||
added = store.add_tags(key, sorted(want - set(have))) if want - set(have) else 0
|
||||
return added, removed
|
||||
|
||||
|
||||
def _extra(store: Any, key: str) -> dict:
|
||||
row = (
|
||||
store._con()
|
||||
.execute( # noqa: SLF001
|
||||
"SELECT extra_json FROM items WHERE key = ?", (key,)
|
||||
)
|
||||
.fetchone()
|
||||
)
|
||||
return json.loads(row[0] or "{}") if row and row[0] else {}
|
||||
|
||||
|
||||
def docket_keys(store: Any, docket: str, *, limit: int = 0) -> list[str]:
|
||||
"""Comment item keys in *docket*, newest posted first."""
|
||||
con = store._con() # noqa: SLF001
|
||||
sql = (
|
||||
"SELECT i.key FROM items i "
|
||||
"WHERE i.id IN (SELECT item_id FROM item_tags WHERE tag_id IN "
|
||||
"(SELECT id FROM tags WHERE name = ?)) "
|
||||
"AND i.url LIKE 'https://www.regulations.gov/comment/%' "
|
||||
"ORDER BY i.date_published DESC, i.id DESC"
|
||||
)
|
||||
if limit:
|
||||
sql += f" LIMIT {int(limit)}"
|
||||
return [r[0] for r in con.execute(sql, (f"reg-docket:{docket}",)).fetchall()]
|
||||
|
||||
|
||||
def run(
|
||||
store: Any,
|
||||
*,
|
||||
vocab: Vocab,
|
||||
docket: str,
|
||||
load_chunks: ChunkLoader,
|
||||
card_vecs: Mapping[str, Vector],
|
||||
judge: Judge,
|
||||
model: str,
|
||||
limit: int = 0,
|
||||
force: bool = False,
|
||||
dry_run: bool = False,
|
||||
top: int = 8,
|
||||
max_tags: int = 4,
|
||||
min_confidence: float = 0.0,
|
||||
progress: Callable[[str, TagResult | None, str], None] | None = None,
|
||||
) -> dict[str, int]:
|
||||
"""Tag every comment in *docket* (newest first); resumable via the
|
||||
item's stored state; ``dry_run`` computes and reports but writes
|
||||
nothing."""
|
||||
stats = {
|
||||
"seen": 0,
|
||||
"tagged": 0,
|
||||
"skipped_state": 0,
|
||||
"no_chunks": 0,
|
||||
"tags_written": 0,
|
||||
}
|
||||
for key in docket_keys(store, docket, limit=limit):
|
||||
stats["seen"] += 1
|
||||
extra = _extra(store, key)
|
||||
rows = load_chunks(key)
|
||||
if not rows:
|
||||
stats["no_chunks"] += 1
|
||||
if progress:
|
||||
progress(key, None, "no-chunks")
|
||||
continue
|
||||
rows = sorted(rows, key=lambda r: r[0])
|
||||
texts = [t for _, t, _ in rows]
|
||||
digest = content_hash(texts)
|
||||
if not force and state_matches(
|
||||
extra, version=vocab.version, model=model, digest=digest
|
||||
):
|
||||
stats["skipped_state"] += 1
|
||||
if progress:
|
||||
progress(key, None, "unchanged")
|
||||
continue
|
||||
result = tag_comment(
|
||||
key,
|
||||
texts,
|
||||
[v for _, _, v in rows],
|
||||
vocab=vocab,
|
||||
card_vecs=card_vecs,
|
||||
judge=judge,
|
||||
top=top,
|
||||
max_tags=max_tags,
|
||||
min_confidence=min_confidence,
|
||||
)
|
||||
stats["tagged"] += 1
|
||||
if not dry_run:
|
||||
store.merge_extra(
|
||||
key,
|
||||
{STATE_KEY: state_payload(result, version=vocab.version, model=model)},
|
||||
)
|
||||
added, _removed = apply_tags(store, key, result.slugs, vocab=vocab)
|
||||
stats["tags_written"] += added
|
||||
if progress:
|
||||
progress(key, result, "tagged")
|
||||
return stats
|
||||
|
||||
|
||||
# ── real collaborators (the CLI wires these) ──────────────────────────
|
||||
|
||||
|
||||
def card_vectors(vocab: Vocab, embed: Embed) -> dict[str, list[float]]:
|
||||
slugs = list(vocab.cards)
|
||||
vecs = embed([vocab.cards[s] for s in slugs])
|
||||
return dict(zip(slugs, vecs))
|
||||
|
||||
|
||||
def make_judge(
|
||||
cfg: Any, pool: Any, *, post: Callable[..., dict] | None = None
|
||||
) -> Judge:
|
||||
"""A yes/no judge on the largest live host, ``temperature 0`` — the
|
||||
same route ``llm.classify.closed_vocab_classifier`` takes."""
|
||||
from llm.classify import _default_post
|
||||
from llm.pool import pick_model
|
||||
|
||||
send = post or _default_post
|
||||
pool.check(cfg.instruct_model)
|
||||
|
||||
def judge(theme: Theme, evidence: str) -> bool | None:
|
||||
with pool.acquire_generation() as host:
|
||||
model = pick_model(cfg, pool, host)
|
||||
data = send(
|
||||
f"{host}/api/chat",
|
||||
json={
|
||||
"model": model,
|
||||
"messages": judge_prompt(theme, evidence),
|
||||
"stream": False,
|
||||
"options": {"num_ctx": cfg.chat_num_ctx, "temperature": 0},
|
||||
},
|
||||
)
|
||||
reply = data.get("message", {}).get("content", "")
|
||||
verdict = parse_yes_no(reply)
|
||||
if verdict is None:
|
||||
log.info("judge: unparseable reply %r for %s", reply[:60], theme.slug)
|
||||
return verdict
|
||||
|
||||
return judge
|
||||
|
||||
|
||||
def pg_chunks(engine: Any, collection: str = "comments") -> ChunkLoader:
|
||||
"""``load_chunks(key)`` over ``langchain_pg_embedding`` — the text and
|
||||
vector of every chunk indexed for the item, in ``seq`` order."""
|
||||
from sqlalchemy import text as _text
|
||||
|
||||
sql = _text(
|
||||
"SELECT (e.cmetadata->>'seq')::int AS seq, e.document, e.embedding::text "
|
||||
"FROM langchain_pg_embedding e JOIN langchain_pg_collection c ON c.uuid = e.collection_id "
|
||||
"WHERE c.name = :collection AND e.cmetadata->>'item_key' = :key ORDER BY seq"
|
||||
)
|
||||
|
||||
def load(key: str) -> list[tuple[int, str, list[float]]]:
|
||||
with engine.begin() as con:
|
||||
rows = con.execute(sql, {"collection": collection, "key": key}).fetchall()
|
||||
return [(int(seq or 0), doc or "", json.loads(vec)) for seq, doc, vec in rows]
|
||||
|
||||
return load
|
||||
151
src/llm/vocab/__init__.py
Normal file
151
src/llm/vocab/__init__.py
Normal file
@@ -0,0 +1,151 @@
|
||||
"""The closed theme vocabulary for comment tagging (P35, #574).
|
||||
|
||||
``themes.yaml`` next to this file is the artifact: hand-curated, versioned
|
||||
(``version:``), seeded from the CY2018+ PFS rules' section headings
|
||||
(``dev/scripts/llm_vocab_candidates.py``). Every tagging run records the
|
||||
version it used, so a vocabulary change never silently re-labels old
|
||||
work — bump ``version`` when a slug, definition or synonym changes and
|
||||
retire slugs into ``retired:`` instead of deleting them.
|
||||
|
||||
load() → Vocab (validated: unique kebab-case slugs, definitions)
|
||||
Vocab.cards → {slug: text} — what gets embedded per theme (#575)
|
||||
Vocab.tag(slug) → "llm:<slug>", the bib tag label
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from importlib import resources
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
|
||||
import yaml
|
||||
|
||||
TAG_NAMESPACE = "llm"
|
||||
_SLUG = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Theme:
|
||||
slug: str
|
||||
label: str
|
||||
definition: str
|
||||
synonyms: tuple[str, ...] = ()
|
||||
sections: tuple[str, ...] = ()
|
||||
|
||||
@property
|
||||
def tag(self) -> str:
|
||||
return f"{TAG_NAMESPACE}:{self.slug}"
|
||||
|
||||
@property
|
||||
def card(self) -> str:
|
||||
"""The text embedded for shortlisting: label, definition and the
|
||||
phrases letters actually use."""
|
||||
syn = ", ".join(self.synonyms)
|
||||
return f"{self.label}. {self.definition}" + (f" Also: {syn}." if syn else "")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Vocab:
|
||||
version: int
|
||||
themes: tuple[Theme, ...]
|
||||
retired: tuple[str, ...] = ()
|
||||
source: str = ""
|
||||
_by_slug: dict[str, Theme] = field(default_factory=dict, repr=False, compare=False)
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
object.__setattr__(self, "_by_slug", {t.slug: t for t in self.themes})
|
||||
|
||||
@property
|
||||
def slugs(self) -> tuple[str, ...]:
|
||||
return tuple(t.slug for t in self.themes)
|
||||
|
||||
def get(self, slug: str) -> Theme:
|
||||
return self._by_slug[slug]
|
||||
|
||||
def __contains__(self, slug: object) -> bool:
|
||||
return slug in self._by_slug
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self.themes)
|
||||
|
||||
@property
|
||||
def cards(self) -> dict[str, str]:
|
||||
return {t.slug: t.card for t in self.themes}
|
||||
|
||||
def tag(self, slug: str) -> str:
|
||||
return self.get(slug).tag
|
||||
|
||||
|
||||
class VocabError(ValueError):
|
||||
"""The YAML is not a valid vocabulary — say exactly what is wrong."""
|
||||
|
||||
|
||||
def _theme(raw: dict[str, Any]) -> Theme:
|
||||
slug = str(raw.get("slug") or "").strip()
|
||||
if not _SLUG.match(slug):
|
||||
raise VocabError(f"slug {slug!r} is not kebab-case")
|
||||
definition = str(raw.get("definition") or "").strip()
|
||||
if not definition:
|
||||
raise VocabError(f"{slug}: definition is empty")
|
||||
label = str(raw.get("label") or "").strip() or slug
|
||||
return Theme(
|
||||
slug=slug,
|
||||
label=label,
|
||||
definition=definition,
|
||||
synonyms=tuple(
|
||||
str(s).strip() for s in (raw.get("synonyms") or []) if str(s).strip()
|
||||
),
|
||||
sections=tuple(
|
||||
str(s).strip() for s in (raw.get("sections") or []) if str(s).strip()
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def parse(text: str, *, source: str = "") -> Vocab:
|
||||
data = yaml.safe_load(text) or {}
|
||||
if not isinstance(data, dict) or "themes" not in data:
|
||||
raise VocabError(
|
||||
f"{source or 'vocabulary'}: expected a mapping with a `themes` list"
|
||||
)
|
||||
try:
|
||||
version = int(data.get("version", 0))
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise VocabError(f"{source}: version must be an integer") from exc
|
||||
if version < 1:
|
||||
raise VocabError(f"{source}: version must be >= 1")
|
||||
themes = tuple(_theme(t) for t in data.get("themes") or [])
|
||||
if not themes:
|
||||
raise VocabError(f"{source}: no themes")
|
||||
seen: set[str] = set()
|
||||
for t in themes:
|
||||
if t.slug in seen:
|
||||
raise VocabError(f"duplicate slug {t.slug!r}")
|
||||
seen.add(t.slug)
|
||||
retired = tuple(
|
||||
str(s).strip() for s in (data.get("retired") or []) if str(s).strip()
|
||||
)
|
||||
clash = seen & set(retired)
|
||||
if clash:
|
||||
raise VocabError(f"retired slugs still active: {sorted(clash)}")
|
||||
return Vocab(version=version, themes=themes, retired=retired, source=source)
|
||||
|
||||
|
||||
def load(path: Path | str | None = None) -> Vocab:
|
||||
"""The packaged ``themes.yaml`` (default) or a file on disk."""
|
||||
if path is None:
|
||||
res = resources.files(__name__).joinpath("themes.yaml")
|
||||
return parse(res.read_text(encoding="utf-8"), source=str(res))
|
||||
p = Path(path)
|
||||
return parse(p.read_text(encoding="utf-8"), source=str(p))
|
||||
|
||||
|
||||
def is_llm_tag(label: str) -> bool:
|
||||
return label.startswith(f"{TAG_NAMESPACE}:")
|
||||
|
||||
|
||||
def split_tags(labels: Sequence[str]) -> tuple[list[str], list[str]]:
|
||||
"""``(llm tags, everything else)`` — the write-back only ever touches the first."""
|
||||
ours = [t for t in labels if is_llm_tag(t)]
|
||||
return ours, [t for t in labels if not is_llm_tag(t)]
|
||||
276
src/llm/vocab/themes.yaml
Normal file
276
src/llm/vocab/themes.yaml
Normal file
@@ -0,0 +1,276 @@
|
||||
# Closed theme vocabulary for tagging PFS rulemaking comments (P35, #574).
|
||||
#
|
||||
# Seeded from the section headings of the CY2018–CY2027 PFS proposed and
|
||||
# final rules (dev/scripts/llm_vocab_candidates.py over fr_anchors) and the
|
||||
# hand code families; curated by hand. Slugs are stable once published —
|
||||
# retire a theme by moving it to `retired:` rather than deleting it, and
|
||||
# bump `version` whenever a slug, definition or synonym list changes, since
|
||||
# every tag run records the version it used.
|
||||
version: 1
|
||||
themes:
|
||||
- slug: conversion-factor
|
||||
label: Conversion factor and payment update
|
||||
definition: The annual PFS conversion factor, statutory update percentage, budget-neutrality adjustment, or the overall payment cut or increase.
|
||||
synonyms: [conversion factor, CF, payment update, budget neutrality, statutory update, payment cut, physician payment cut]
|
||||
sections: [Conversion Factor, Physician Fee Schedule Update, Budget Neutrality]
|
||||
- slug: practice-expense
|
||||
label: Practice expense RVUs
|
||||
definition: Practice expense methodology, direct and indirect PE inputs, clinical labor pricing, supply and equipment pricing, or the PE/HR survey data.
|
||||
synonyms: [practice expense, PE RVU, clinical labor, supply pricing, equipment pricing, indirect practice expense, PE/HR, AMA PPIS survey]
|
||||
sections: [Determination of Practice Expense (PE) Relative Value Units (RVUs), Practice Expense Methodology]
|
||||
- slug: work-rvu
|
||||
label: Work RVUs and code valuation
|
||||
definition: Physician work RVUs, RUC recommendations, time assumptions, or the valuation of specific CPT/HCPCS codes.
|
||||
synonyms: [work RVU, RUC recommendation, valuation of specific codes, intraservice time, survey data, code valuation]
|
||||
sections: [Valuation of Specific Codes, Work RVUs]
|
||||
- slug: malpractice-rvu
|
||||
label: Malpractice RVUs
|
||||
definition: Malpractice (MP) RVU methodology, premium data, or specialty risk factors.
|
||||
synonyms: [malpractice RVU, MP RVU, professional liability insurance premium, risk factor]
|
||||
sections: [Determination of Malpractice Relative Value Units (RVUs)]
|
||||
- slug: misvalued-codes
|
||||
label: Potentially misvalued codes
|
||||
definition: Identification, review or revaluation of potentially misvalued services, including the statutory misvalued-code target.
|
||||
synonyms: [misvalued, potentially misvalued codes, misvalued code target, revaluation]
|
||||
sections: [Potentially Misvalued Services Under the PFS]
|
||||
- slug: gpci-localities
|
||||
label: GPCIs and payment localities
|
||||
definition: Geographic practice cost indices, locality definitions, or geographic adjustment of payment.
|
||||
synonyms: [GPCI, geographic practice cost index, payment locality, locality, geographic adjustment, work GPCI floor]
|
||||
sections: [Geographic Practice Cost Indices (GPCIs), Payment Localities]
|
||||
- slug: telehealth
|
||||
label: Medicare telehealth services
|
||||
definition: The Medicare telehealth services list, originating-site and geographic restrictions, audio-only, virtual direct supervision, or telehealth flexibilities after the public health emergency.
|
||||
synonyms: [telehealth, telemedicine, audio-only, originating site, distant site, virtual supervision, telehealth list, home as originating site]
|
||||
sections: [Medicare Telehealth Services, Payment for Medicare Telehealth Services Under Section 1834(m) of the Act]
|
||||
- slug: communication-technology-services
|
||||
label: Communication technology-based services
|
||||
definition: Virtual check-ins, e-visits, remote evaluation of images, interprofessional consultations and other communication technology-based services that are not telehealth.
|
||||
synonyms: [virtual check-in, e-visit, online digital E/M, interprofessional consultation, remote evaluation, CTBS, G2012, 99421]
|
||||
sections: [Communication Technology-Based Services (CTBS)]
|
||||
- slug: remote-monitoring
|
||||
label: Remote physiologic and therapeutic monitoring
|
||||
definition: Remote physiologic monitoring (RPM) or remote therapeutic monitoring (RTM) codes, device supply days, or monitoring time requirements.
|
||||
synonyms: [remote physiologic monitoring, RPM, remote therapeutic monitoring, RTM, 99454, 99457, 98977, 16 days]
|
||||
sections: [Remote Physiologic Monitoring]
|
||||
- slug: evaluation-management
|
||||
label: Evaluation and management visits
|
||||
definition: Office/outpatient, hospital, nursing facility or home E/M visit coding, documentation guidelines, level selection, prolonged services, or split/shared visits.
|
||||
synonyms: [E/M, evaluation and management, office visit, 99213, 99214, prolonged services, split/shared, level selection, documentation guidelines]
|
||||
sections: [Evaluation & Management (E/M) Visits, Evaluation & Management (E/M) Guidelines and Care Management Services]
|
||||
- slug: visit-complexity-add-on
|
||||
label: Visit complexity add-on (G2211)
|
||||
definition: The office/outpatient visit complexity add-on code G2211, its use with modifier 25, or its budget impact.
|
||||
synonyms: [G2211, visit complexity, complexity add-on, longitudinal relationship, modifier 25]
|
||||
sections: [Visit Complexity]
|
||||
- slug: care-management
|
||||
label: Care management services
|
||||
definition: Chronic care management, principal care management, transitional care management, advanced primary care management, or other monthly care-management codes and their consent, staffing and time requirements.
|
||||
synonyms: [chronic care management, CCM, principal care management, PCM, transitional care management, TCM, advanced primary care management, APCM, 99490, 99439, G0556, care management]
|
||||
sections: [Care Management Services, Advanced Primary Care Management]
|
||||
- slug: behavioral-health
|
||||
label: Behavioral health integration and services
|
||||
definition: Behavioral health integration, collaborative care, psychotherapy, marriage and family therapists and mental health counselors, or crisis and safety-planning services.
|
||||
synonyms: [behavioral health, mental health, collaborative care, BHI, psychotherapy, MFT, MHC, safety planning, crisis psychotherapy, community health integration]
|
||||
sections: [Behavioral Health Services]
|
||||
- slug: opioid-treatment
|
||||
label: Opioid treatment programs and substance use disorder
|
||||
definition: Opioid treatment program bundled payments, medications for opioid use disorder, or substance use disorder treatment via telehealth.
|
||||
synonyms: [opioid treatment program, OTP, medication for opioid use disorder, MOUD, buprenorphine, methadone, substance use disorder, SUD]
|
||||
sections: [Requirements for Opioid Treatment Programs (OTP)]
|
||||
- slug: caregiver-training
|
||||
label: Caregiver training services
|
||||
definition: Caregiver training services furnished to a patient's caregivers, with or without the patient present.
|
||||
synonyms: [caregiver training, CTS, 97550, 96202]
|
||||
sections: [Caregiver Training Services]
|
||||
- slug: community-health-social-needs
|
||||
label: Community health integration and social needs
|
||||
definition: Community health integration, principal illness navigation, social determinants of health risk assessment, or community health worker services.
|
||||
synonyms: [community health integration, CHI, principal illness navigation, PIN, social determinants of health, SDOH risk assessment, community health worker, G0019, G0136]
|
||||
sections: [Community Health Integration, Principal Illness Navigation]
|
||||
- slug: palliative-care
|
||||
label: Palliative and serious-illness care
|
||||
definition: Community-based palliative care, advance care planning, hospice interaction, or serious-illness care management.
|
||||
synonyms: [palliative care, advance care planning, ACP, serious illness, hospice, end-of-life]
|
||||
sections: [Request for Information on Community-Based Palliative Care]
|
||||
- slug: supervision
|
||||
label: Supervision requirements
|
||||
definition: Direct, general or virtual supervision of auxiliary personnel, incident-to services, or supervision of diagnostic tests and residents.
|
||||
synonyms: [direct supervision, general supervision, incident to, incident-to, supervision of residents, virtual presence, teaching physician]
|
||||
sections: [Direct Supervision by Interactive Telecommunications Technology, Teaching Physician Documentation Requirements]
|
||||
- slug: non-physician-practitioners
|
||||
label: Non-physician practitioners and scope
|
||||
definition: Nurse practitioners, physician assistants, clinical nurse specialists, therapists or other non-physician practitioners' billing, supervision or scope-of-practice policies.
|
||||
synonyms: [nurse practitioner, physician assistant, PA, NP, advanced practice, scope of practice, CRNA, non-physician practitioner, NPP]
|
||||
sections: [Physician Assistants, Nurse Practitioners]
|
||||
- slug: therapy-services
|
||||
label: Physical, occupational and speech therapy
|
||||
definition: Therapy services, therapy assistants and the CQ/CO modifiers, the KX modifier threshold, plan-of-care certification, or therapy supervision.
|
||||
synonyms: [physical therapy, occupational therapy, speech-language pathology, therapy assistant, CQ modifier, CO modifier, KX modifier, therapy threshold, plan of care]
|
||||
sections: [Therapy Services, Therapy Caps]
|
||||
- slug: drugs-biologicals
|
||||
label: Part B drugs and biologicals
|
||||
definition: Average sales price methodology, drug add-on percentages, biosimilars, discarded-drug refunds, or drug payment under Part B.
|
||||
synonyms: [average sales price, ASP, WAC, biosimilar, discarded drug, JW modifier, JZ modifier, Part B drug, drug payment, 340B]
|
||||
sections: [Part B Drug Payment, Payment for Biosimilar Biological Products]
|
||||
- slug: vaccines-preventive
|
||||
label: Vaccines and preventive services
|
||||
definition: Vaccine administration payment, preventive services coverage, screening services, or the Medicare Diabetes Prevention Program.
|
||||
synonyms: [vaccine administration, preventive services, screening, colorectal cancer screening, Medicare Diabetes Prevention Program, MDPP, hepatitis B vaccine, HIV PrEP]
|
||||
sections: [Medicare Diabetes Prevention Program, Preventive Services]
|
||||
- slug: skin-substitutes
|
||||
label: Skin substitutes and wound care
|
||||
definition: Payment or coding for skin substitute products, cellular and tissue-based products, or wound care services.
|
||||
synonyms: [skin substitute, cellular and tissue-based product, CTP, wound care, Q4 code, amniotic, dermal substitute]
|
||||
sections: [Skin Substitutes]
|
||||
- slug: dental-services
|
||||
label: Dental services
|
||||
definition: Medicare payment for dental services inextricably linked to covered medical services.
|
||||
synonyms: [dental, oral health, dental services, inextricably linked]
|
||||
sections: [Dental Services]
|
||||
- slug: global-surgery
|
||||
label: Global surgical packages
|
||||
definition: Global surgery periods, post-operative visit reporting, or the transfer-of-care modifier for global packages.
|
||||
synonyms: [global surgery, global period, 010-day, 090-day, post-operative visit, modifier 54, modifier 55, transfer of care, G0559]
|
||||
sections: [Global Surgery]
|
||||
- slug: anesthesia
|
||||
label: Anesthesia services
|
||||
definition: The anesthesia conversion factor, anesthesia base units, or anesthesia billing and supervision.
|
||||
synonyms: [anesthesia, anesthesia conversion factor, base units, CRNA, medical direction]
|
||||
sections: [Anesthesia Services, Anesthesia Conversion Factor]
|
||||
- slug: imaging-radiology
|
||||
label: Imaging and radiology
|
||||
definition: Diagnostic imaging payment, appropriate use criteria for advanced imaging, or radiology assistants and supervision.
|
||||
synonyms: [imaging, radiology, appropriate use criteria, AUC, clinical decision support, CT, MRI, radiologist assistant]
|
||||
sections: [Appropriate Use Criteria for Advanced Diagnostic Imaging Services]
|
||||
- slug: laboratory
|
||||
label: Clinical laboratory fee schedule
|
||||
definition: The clinical laboratory fee schedule, private payer rate reporting, specimen collection, or laboratory test payment.
|
||||
synonyms: [clinical laboratory fee schedule, CLFS, private payor rate, PAMA, laboratory test, specimen collection]
|
||||
sections: [Clinical Laboratory Fee Schedule]
|
||||
- slug: ambulance
|
||||
label: Ambulance services
|
||||
definition: The ambulance fee schedule, ground or air ambulance payment, or ambulance cost data collection.
|
||||
synonyms: [ambulance, ambulance fee schedule, ground ambulance, air ambulance, ambulance cost collection]
|
||||
sections: [Ambulance Fee Schedule]
|
||||
- slug: dme-supplies
|
||||
label: Durable medical equipment and supplies
|
||||
definition: DME infusion drugs, DMEPOS competitive bidding, or supplies furnished with physician services.
|
||||
synonyms: [durable medical equipment, DME, DMEPOS, infusion drugs, competitive bidding]
|
||||
sections: [Payment for DME Infusion Drugs]
|
||||
- slug: esrd-dialysis
|
||||
label: ESRD and dialysis services
|
||||
definition: Monthly capitation payment for ESRD-related services, home dialysis, or kidney disease education.
|
||||
synonyms: [ESRD, dialysis, monthly capitation payment, MCP, home dialysis, kidney disease education]
|
||||
sections: [Monthly Capitation Payment (MCP) for ESRD-Related Services]
|
||||
- slug: rhc-fqhc
|
||||
label: Rural health clinics and FQHCs
|
||||
definition: Payment or care-coordination policies for rural health clinics and federally qualified health centers.
|
||||
synonyms: [rural health clinic, RHC, federally qualified health center, FQHC, G0511, all-inclusive rate]
|
||||
sections: [Rural Health Clinics (RHCs) and Federally Qualified Health Centers (FQHCs)]
|
||||
- slug: rural-access
|
||||
label: Rural and underserved access
|
||||
definition: Access to care in rural or underserved areas, workforce shortages, or health equity concerns raised about a policy's effect on access.
|
||||
synonyms: [rural, underserved, health equity, access to care, workforce shortage, health professional shortage area, HPSA]
|
||||
sections: [Health Equity]
|
||||
- slug: shared-savings-program
|
||||
label: Medicare Shared Savings Program
|
||||
definition: Shared Savings Program ACO participation, benchmarking, risk tracks, quality reporting, beneficiary assignment, or advance investment payments.
|
||||
synonyms: [Shared Savings Program, MSSP, accountable care organization, ACO, benchmark, ENHANCED track, BASIC track, beneficiary assignment, advance investment payment]
|
||||
sections: [Medicare Shared Savings Program]
|
||||
- slug: quality-payment-program
|
||||
label: Quality Payment Program and MIPS
|
||||
definition: MIPS categories, MIPS Value Pathways, scoring, performance thresholds, or the QPP reporting requirements.
|
||||
synonyms: [Quality Payment Program, QPP, MIPS, MIPS Value Pathway, MVP, performance threshold, promoting interoperability, cost category, improvement activities]
|
||||
sections: [Updates to the Quality Payment Program]
|
||||
- slug: advanced-apm
|
||||
label: Advanced APMs and APM incentives
|
||||
definition: Advanced alternative payment model participation, qualifying participant thresholds, the APM incentive payment, or the APM conversion factor.
|
||||
synonyms: [advanced APM, alternative payment model, qualifying participant, QP threshold, APM incentive, APM conversion factor]
|
||||
sections: [Advanced Alternative Payment Models]
|
||||
- slug: quality-measures
|
||||
label: Quality measures and reporting
|
||||
definition: Specific quality measures, measure specifications, electronic clinical quality measures, or measure reporting burden.
|
||||
synonyms: [quality measure, eCQM, measure specification, reporting burden, measure set, benchmarking of measures]
|
||||
sections: [Quality Measures]
|
||||
- slug: innovation-models
|
||||
label: Innovation Center models
|
||||
definition: CMS Innovation Center models and demonstrations, including primary care, kidney, oncology or radiation models.
|
||||
synonyms: [Innovation Center, CMMI, model, demonstration, Primary Care First, ACO REACH, oncology care model, radiation oncology model]
|
||||
sections: [Innovation Center Models]
|
||||
- slug: self-referral
|
||||
label: Physician self-referral law
|
||||
definition: Stark law exceptions, the annual code list update, or self-referral compliance.
|
||||
synonyms: [self-referral, Stark, physician self-referral law, designated health services]
|
||||
sections: [Physician Self-Referral Law]
|
||||
- slug: enrollment-program-integrity
|
||||
label: Enrollment and program integrity
|
||||
definition: Provider enrollment requirements, revocations, prior authorization, fraud, waste and abuse controls, or medical review.
|
||||
synonyms: [enrollment, revocation, program integrity, fraud waste and abuse, prior authorization, medical review, overpayment, audit]
|
||||
sections: [Medicare Provider Enrollment, Program Integrity]
|
||||
- slug: coverage-policy
|
||||
label: Coverage and benefit categories
|
||||
definition: Whether a service is a covered Medicare benefit, benefit-category determinations, or national coverage policy raised in the rule.
|
||||
synonyms: [coverage, benefit category, covered service, national coverage determination, NCD, reasonable and necessary]
|
||||
sections: [Coverage]
|
||||
- slug: beneficiary-cost-sharing
|
||||
label: Beneficiary cost sharing
|
||||
definition: Coinsurance, deductibles, or beneficiary out-of-pocket effects of a payment policy.
|
||||
synonyms: [cost sharing, coinsurance, deductible, out-of-pocket, beneficiary liability, copay]
|
||||
sections: [Telehealth Modalities and Cost-sharing]
|
||||
- slug: administrative-burden
|
||||
label: Administrative burden and documentation
|
||||
definition: Documentation, paperwork, reporting or compliance burden imposed on practices, including collection-of-information estimates.
|
||||
synonyms: [administrative burden, documentation burden, paperwork, compliance burden, collection of information, patients over paperwork, prior authorization burden]
|
||||
sections: [Collection of Information Requirements]
|
||||
- slug: regulatory-impact
|
||||
label: Regulatory impact and specialty impacts
|
||||
definition: The regulatory impact analysis, specialty-level payment impact tables, or estimated effects on practices and beneficiaries.
|
||||
synonyms: [regulatory impact analysis, impact table, specialty impact, Table 118, estimated impact, small entities]
|
||||
sections: [Regulatory Impact Analysis]
|
||||
- slug: medicare-economic-index
|
||||
label: Medicare Economic Index and cost weights
|
||||
definition: The Medicare Economic Index, its rebasing or revised cost-share weights, and their effect on RVU allocation.
|
||||
synonyms: [Medicare Economic Index, MEI, cost share weights, rebasing]
|
||||
sections: [The Percentage Change in the Medicare Economic Index (MEI)]
|
||||
- slug: efficiency-adjustment
|
||||
label: Efficiency adjustment to work RVUs
|
||||
definition: An across-the-board efficiency adjustment to work RVUs or intraservice time for non-time-based services.
|
||||
synonyms: [efficiency adjustment, efficiency gains, across-the-board reduction, non-time-based services]
|
||||
sections: [Efficiency Adjustment]
|
||||
- slug: site-of-service
|
||||
label: Site of service and facility payment
|
||||
definition: Facility versus non-facility payment, site-neutral policies, hospital outpatient department comparisons, or indirect PE by setting.
|
||||
synonyms: [site of service, site neutral, facility rate, non-facility rate, hospital outpatient department, HOPD, office-based]
|
||||
sections: [Site of Service]
|
||||
- slug: hospice-home-health
|
||||
label: Hospice and home health interaction
|
||||
definition: Hospice face-to-face encounters, homebound status, home health certification, or physician services interacting with hospice and home health benefits.
|
||||
synonyms: [hospice, home health, face-to-face encounter, homebound, certification of home health]
|
||||
sections: [Clarification of Homebound Status Under the Medicare Home Health Benefit, Telehealth and the Medicare Hospice Face-to-Face Encounter Requirement]
|
||||
- slug: interoperability-health-it
|
||||
label: Health IT and interoperability
|
||||
definition: Certified EHR technology, promoting interoperability requirements, data exchange, or the EHR incentive programs.
|
||||
synonyms: [electronic health record, EHR, certified EHR technology, CEHRT, interoperability, health information exchange, promoting interoperability]
|
||||
sections: [Medicare EHR Incentive Program, Medicaid Promoting Interoperability Program Requirements]
|
||||
- slug: price-transparency
|
||||
label: Price transparency
|
||||
definition: Public disclosure of prices, charges or payment rates to beneficiaries.
|
||||
synonyms: [price transparency, charge transparency, disclosure of prices]
|
||||
sections: [Request for Information on Price Transparency]
|
||||
- slug: covid-phe
|
||||
label: COVID-19 public health emergency flexibilities
|
||||
definition: Policies tied to the COVID-19 public health emergency and their extension or expiration.
|
||||
synonyms: [public health emergency, PHE, COVID-19, pandemic, flexibilities, waiver]
|
||||
sections: [Counting of Resident Time During the PHE for the COVID-19 Pandemic]
|
||||
- slug: data-requests-rfi
|
||||
label: Requests for information
|
||||
definition: Responses to a request for information or comment solicitation rather than a specific proposal.
|
||||
synonyms: [request for information, RFI, comment solicitation, seeking comment, solicit feedback]
|
||||
sections: [Requests for Information]
|
||||
- slug: coding-descriptors
|
||||
label: Coding and code descriptors
|
||||
definition: Creation, deletion or crosswalk of CPT/HCPCS codes, code descriptors, or bundling of services into other codes.
|
||||
synonyms: [new code, deleted code, HCPCS code, G code, crosswalk, bundled, code descriptor, CPT Editorial Panel, unbundle]
|
||||
sections: [Proposed Valuation of Specific Codes]
|
||||
retired: []
|
||||
103
tests/cli/test_llm_tag_cli.py
Normal file
103
tests/cli/test_llm_tag_cli.py
Normal file
@@ -0,0 +1,103 @@
|
||||
"""stack llm tag / vocab (#574, #575)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from typer.testing import CliRunner
|
||||
|
||||
import cli.llm as llm_cli
|
||||
from bib.item import Source
|
||||
from bib.store import Store
|
||||
from cli import app
|
||||
from llm.vocab import parse
|
||||
|
||||
runner = CliRunner()
|
||||
|
||||
VOCAB = parse(
|
||||
"version: 1\nthemes:\n - slug: telehealth\n definition: Telehealth.\n synonyms: [telehealth]\n"
|
||||
" - slug: drugs\n definition: Drugs.\n synonyms: [ASP]\n"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def rt(monkeypatch, tmp_path):
|
||||
store = Store(":memory:", storage_dir=tmp_path / "st")
|
||||
key = store.upsert(
|
||||
Source(
|
||||
title="c1",
|
||||
url="https://www.regulations.gov/comment/CMS-2026-2377-1",
|
||||
date_published="2026-09-01",
|
||||
),
|
||||
tags=["reg-docket:CMS-2026-2377", "source:regulations-gov"],
|
||||
)
|
||||
judged = []
|
||||
|
||||
def judge(theme, evidence):
|
||||
judged.append(theme.slug)
|
||||
return theme.slug == "telehealth"
|
||||
|
||||
runtime = {
|
||||
"store": store,
|
||||
"vocab": VOCAB,
|
||||
"load_chunks": lambda k: (
|
||||
[(0, "telehealth text", [1.0, 0.0])] if k == key else []
|
||||
),
|
||||
"card_vecs": {"telehealth": [1.0, 0.0], "drugs": [0.0, 1.0]},
|
||||
"judge": judge,
|
||||
"model": "m",
|
||||
}
|
||||
monkeypatch.setattr(llm_cli, "_tagging_runtime", lambda cfg: runtime)
|
||||
monkeypatch.setattr("llm.config.load", lambda: object())
|
||||
try:
|
||||
yield store, key, judged
|
||||
finally:
|
||||
store.close()
|
||||
|
||||
|
||||
class TestTag:
|
||||
def test_tags_and_reports(self, rt):
|
||||
store, key, judged = rt
|
||||
res = runner.invoke(
|
||||
app, ["llm", "tag", "--docket", "CMS-2026-2377", "--verbose"]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
assert f"{key} telehealth" in res.output
|
||||
assert "seen=1 tagged=1 unchanged=0 no_chunks=0 tags_written=1" in res.output
|
||||
assert (
|
||||
"llm:telehealth" in store.get(key).tags
|
||||
and "llm:drugs" not in store.get(key).tags
|
||||
)
|
||||
assert judged == ["telehealth", "drugs"]
|
||||
res = runner.invoke(app, ["llm", "tag", "--docket", "CMS-2026-2377"])
|
||||
assert "unchanged=1" in res.output
|
||||
|
||||
def test_dry_run_prints_but_writes_nothing(self, rt):
|
||||
store, key, _ = rt
|
||||
res = runner.invoke(
|
||||
app, ["llm", "tag", "--docket", "CMS-2026-2377", "--dry-run"]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
assert "dry run" in res.output and f"{key} telehealth" in res.output
|
||||
assert not any(t.startswith("llm:") for t in store.get(key).tags)
|
||||
|
||||
def test_docket_is_required(self):
|
||||
assert runner.invoke(app, ["llm", "tag"]).exit_code != 0
|
||||
|
||||
|
||||
class TestVocab:
|
||||
def test_lists_and_shows(self):
|
||||
res = runner.invoke(app, ["llm", "vocab"])
|
||||
assert (
|
||||
res.exit_code == 0
|
||||
and "vocabulary v" in res.output
|
||||
and "telehealth" in res.output
|
||||
)
|
||||
res = runner.invoke(app, ["llm", "vocab", "--slug", "telehealth"])
|
||||
assert res.exit_code == 0 and "llm:telehealth" in res.output
|
||||
assert runner.invoke(app, ["llm", "vocab", "--slug", "nope"]).exit_code != 0
|
||||
|
||||
def test_invalid_file(self, tmp_path):
|
||||
p = tmp_path / "bad.yaml"
|
||||
p.write_text("version: 1\nthemes: []\n")
|
||||
res = runner.invoke(app, ["llm", "vocab", "--path", str(p)])
|
||||
assert res.exit_code == 1 and "invalid vocabulary" in res.output
|
||||
390
tests/llm/test_tagger.py
Normal file
390
tests/llm/test_tagger.py
Normal file
@@ -0,0 +1,390 @@
|
||||
"""llm.tagger — shortlist, judge, state, write-back and the docket runner (#575/#576)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from bib.item import Source
|
||||
from bib.store import Store
|
||||
from llm.tagger import (
|
||||
Candidate,
|
||||
TagResult,
|
||||
apply_tags,
|
||||
card_vectors,
|
||||
content_hash,
|
||||
cosine,
|
||||
docket_keys,
|
||||
judge_prompt,
|
||||
make_judge,
|
||||
parse_yes_no,
|
||||
run,
|
||||
shortlist,
|
||||
state_matches,
|
||||
state_payload,
|
||||
tag_comment,
|
||||
)
|
||||
from llm.vocab import parse
|
||||
|
||||
VOCAB = parse(
|
||||
"""
|
||||
version: 2
|
||||
themes:
|
||||
- slug: telehealth
|
||||
label: Telehealth
|
||||
definition: Medicare telehealth services.
|
||||
synonyms: [telehealth]
|
||||
- slug: care-management
|
||||
label: Care management
|
||||
definition: Monthly care management codes.
|
||||
synonyms: [CCM]
|
||||
- slug: drugs
|
||||
label: Part B drugs
|
||||
definition: ASP drug payment.
|
||||
synonyms: [ASP]
|
||||
"""
|
||||
)
|
||||
CARDS = {
|
||||
"telehealth": [1.0, 0.0, 0.0],
|
||||
"care-management": [0.0, 1.0, 0.0],
|
||||
"drugs": [0.0, 0.0, 1.0],
|
||||
}
|
||||
|
||||
|
||||
class TestPure:
|
||||
def test_cosine(self):
|
||||
assert cosine([1, 0], [1, 0]) == pytest.approx(1.0)
|
||||
assert cosine([1, 0], [0, 1]) == pytest.approx(0.0)
|
||||
assert cosine([0, 0], [1, 1]) == 0.0
|
||||
|
||||
def test_shortlist_best_chunk_per_theme(self):
|
||||
chunks = [[0.9, 0.1, 0.0], [0.1, 0.9, 0.0]]
|
||||
out = shortlist(chunks, CARDS, top=2)
|
||||
assert [c.slug for c in out] == ["care-management", "telehealth"] or [
|
||||
c.slug for c in out
|
||||
] == ["telehealth", "care-management"]
|
||||
by = {c.slug: c for c in out}
|
||||
assert by["telehealth"].chunk_idx == 0 and by["care-management"].chunk_idx == 1
|
||||
assert all(isinstance(c, Candidate) and 0 < c.score <= 1 for c in out)
|
||||
assert shortlist([], CARDS) == []
|
||||
|
||||
def test_judge_prompt_and_parse(self):
|
||||
msgs = judge_prompt(
|
||||
VOCAB.get("telehealth"), "We support audio-only telehealth."
|
||||
)
|
||||
assert msgs[0]["role"] == "system" and "Telehealth" in msgs[1]["content"]
|
||||
assert "audio-only telehealth" in msgs[1]["content"]
|
||||
assert parse_yes_no("Yes.") is True and parse_yes_no(" no\n") is False
|
||||
assert parse_yes_no("Maybe yes") is None and parse_yes_no("") is None
|
||||
|
||||
def test_content_hash_is_order_sensitive(self):
|
||||
assert content_hash(["a", "b"]) != content_hash(["b", "a"])
|
||||
assert len(content_hash(["a"])) == 16
|
||||
|
||||
def test_tag_comment_accepts_judged_yes_up_to_max(self):
|
||||
chunks = ["telehealth text", "ccm text", "asp text"]
|
||||
vecs = [[1.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]]
|
||||
seen = []
|
||||
|
||||
def judge(theme, evidence):
|
||||
seen.append((theme.slug, evidence))
|
||||
return theme.slug != "drugs"
|
||||
|
||||
r = tag_comment(
|
||||
"K",
|
||||
chunks,
|
||||
vecs,
|
||||
vocab=VOCAB,
|
||||
card_vecs=CARDS,
|
||||
judge=judge,
|
||||
top=3,
|
||||
max_tags=1,
|
||||
)
|
||||
assert isinstance(r, TagResult) and len(r.tags) == 1
|
||||
assert r.judged["drugs"] is False and any(v for v in r.judged.values())
|
||||
assert (
|
||||
"telehealth",
|
||||
"telehealth text",
|
||||
) in seen # evidence is the scoring chunk
|
||||
assert r.content_hash == content_hash(chunks)
|
||||
|
||||
def test_min_confidence_filters_weak_matches(self):
|
||||
chunks = ["x"]
|
||||
vecs = [[0.6, 0.8, 0.0]] # cosine 0.6 with telehealth, 0.8 with care-management
|
||||
r = tag_comment(
|
||||
"K",
|
||||
chunks,
|
||||
vecs,
|
||||
vocab=VOCAB,
|
||||
card_vecs=CARDS,
|
||||
judge=lambda t, e: True,
|
||||
min_confidence=0.7,
|
||||
)
|
||||
assert r.slugs == ["care-management"]
|
||||
|
||||
def test_state(self):
|
||||
r = tag_comment(
|
||||
"K",
|
||||
["t"],
|
||||
[[1.0, 0, 0]],
|
||||
vocab=VOCAB,
|
||||
card_vecs=CARDS,
|
||||
judge=lambda t, e: True,
|
||||
max_tags=1,
|
||||
)
|
||||
payload = state_payload(r, version=2, model="m")
|
||||
assert payload["tags"] == {
|
||||
"telehealth": {"confidence": 1.0, "evidence_chunk": 0}
|
||||
}
|
||||
assert payload["shortlist"][0]["verdict"] is True and "tagged_at" in payload
|
||||
assert state_matches(
|
||||
{"llm_tags": payload}, version=2, model="m", digest=r.content_hash
|
||||
)
|
||||
assert not state_matches(
|
||||
{"llm_tags": payload}, version=3, model="m", digest=r.content_hash
|
||||
)
|
||||
assert not state_matches({}, version=2, model="m", digest=r.content_hash)
|
||||
|
||||
def test_card_vectors_embeds_every_card_once(self):
|
||||
calls = []
|
||||
|
||||
def embed(texts):
|
||||
calls.append(list(texts))
|
||||
return [[float(i)] for i in range(len(texts))]
|
||||
|
||||
cv = card_vectors(VOCAB, embed)
|
||||
assert set(cv) == {"telehealth", "care-management", "drugs"} and len(calls) == 1
|
||||
assert "Medicare telehealth services" in calls[0][0]
|
||||
|
||||
|
||||
class TestJudge:
|
||||
def test_make_judge_posts_to_largest_host_at_temperature_zero(self):
|
||||
class Pool:
|
||||
def check(self, model):
|
||||
return ["http://h"]
|
||||
|
||||
def vram(self, host):
|
||||
return 0.0
|
||||
|
||||
def serves(self, host, model):
|
||||
return False
|
||||
|
||||
class _cm:
|
||||
def __enter__(self):
|
||||
return "http://h"
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
def acquire_generation(self):
|
||||
return self._cm()
|
||||
|
||||
class Cfg:
|
||||
instruct_model = "m"
|
||||
instruct_model_large = ""
|
||||
large_min_vram_gb = 20.0
|
||||
chat_num_ctx = 4096
|
||||
|
||||
sent = {}
|
||||
|
||||
def post(url, json=None):
|
||||
sent["url"], sent["json"] = url, json
|
||||
return {"message": {"content": "Yes"}}
|
||||
|
||||
judge = make_judge(Cfg(), Pool(), post=post)
|
||||
assert judge(VOCAB.get("drugs"), "ASP add-on") is True
|
||||
assert (
|
||||
sent["url"] == "http://h/api/chat"
|
||||
and sent["json"]["options"]["temperature"] == 0
|
||||
)
|
||||
assert sent["json"]["model"] == "m"
|
||||
|
||||
|
||||
def _store(tmp_path):
|
||||
s = Store(":memory:", storage_dir=tmp_path / "st")
|
||||
keys = {}
|
||||
for i, (cid, date) in enumerate(
|
||||
[
|
||||
("CMS-2026-2377-1", "2026-09-01"),
|
||||
("CMS-2026-2377-2", "2026-09-05"),
|
||||
("CMS-2025-0304-9", "2025-09-01"),
|
||||
]
|
||||
):
|
||||
docket = cid.rsplit("-", 1)[0]
|
||||
keys[cid] = s.upsert(
|
||||
Source(
|
||||
title=cid,
|
||||
url=f"https://www.regulations.gov/comment/{cid}",
|
||||
date_published=date,
|
||||
),
|
||||
tags=[f"reg-docket:{docket}", "source:regulations-gov", "topic:palliative"],
|
||||
)
|
||||
return s, keys
|
||||
|
||||
|
||||
class TestBibSide:
|
||||
def test_docket_keys_newest_first(self, tmp_path):
|
||||
s, keys = _store(tmp_path)
|
||||
assert docket_keys(s, "CMS-2026-2377") == [
|
||||
keys["CMS-2026-2377-2"],
|
||||
keys["CMS-2026-2377-1"],
|
||||
]
|
||||
assert docket_keys(s, "CMS-2026-2377", limit=1) == [keys["CMS-2026-2377-2"]]
|
||||
assert docket_keys(s, "NOPE") == []
|
||||
s.close()
|
||||
|
||||
def test_apply_tags_replaces_only_llm_tags(self, tmp_path):
|
||||
s, keys = _store(tmp_path)
|
||||
k = keys["CMS-2026-2377-1"]
|
||||
s.add_tags(k, ["llm:drugs"])
|
||||
added, removed = apply_tags(
|
||||
s, k, ["telehealth", "care-management"], vocab=VOCAB
|
||||
)
|
||||
assert (added, removed) == (2, 1)
|
||||
tags = set(s.get(k).tags)
|
||||
assert {
|
||||
"llm:telehealth",
|
||||
"llm:care-management",
|
||||
"topic:palliative",
|
||||
"reg-docket:CMS-2026-2377",
|
||||
} <= tags
|
||||
assert "llm:drugs" not in tags
|
||||
assert apply_tags(s, k, ["telehealth", "care-management"], vocab=VOCAB) == (
|
||||
0,
|
||||
0,
|
||||
) # idempotent
|
||||
assert (
|
||||
apply_tags(s, k, [], vocab=VOCAB) == (0, 2)
|
||||
and "topic:palliative" in s.get(k).tags
|
||||
)
|
||||
s.close()
|
||||
|
||||
def test_merge_extra(self, tmp_path):
|
||||
s, keys = _store(tmp_path)
|
||||
k = keys["CMS-2026-2377-1"]
|
||||
s.merge_extra(k, {"a": 1})
|
||||
s.merge_extra(k, {"b": {"x": 2}})
|
||||
s.merge_extra("ZZZZZZZZ", {"c": 3}) # unknown key: no-op, no error
|
||||
extra = json.loads(
|
||||
s._con()
|
||||
.execute("SELECT extra_json FROM items WHERE key=?", (k,))
|
||||
.fetchone()[0]
|
||||
) # noqa: SLF001
|
||||
assert extra["a"] == 1 and extra["b"] == {"x": 2}
|
||||
s.close()
|
||||
|
||||
|
||||
class TestRun:
|
||||
def _chunks(self, keys):
|
||||
def load(key):
|
||||
if key == keys["CMS-2026-2377-1"]:
|
||||
return [
|
||||
(1, "ccm text", [0.0, 1.0, 0.0]),
|
||||
(0, "telehealth text", [1.0, 0.0, 0.0]),
|
||||
]
|
||||
if key == keys["CMS-2026-2377-2"]:
|
||||
return [] # not indexed yet
|
||||
return [(0, "asp", [0.0, 0.0, 1.0])]
|
||||
|
||||
return load
|
||||
|
||||
def test_tags_persist_state_and_resume(self, tmp_path):
|
||||
s, keys = _store(tmp_path)
|
||||
events = []
|
||||
judge_calls = []
|
||||
|
||||
def judge(theme, evidence):
|
||||
judge_calls.append(theme.slug)
|
||||
return True
|
||||
|
||||
stats = run(
|
||||
s,
|
||||
vocab=VOCAB,
|
||||
docket="CMS-2026-2377",
|
||||
load_chunks=self._chunks(keys),
|
||||
card_vecs=CARDS,
|
||||
judge=judge,
|
||||
model="m",
|
||||
top=2,
|
||||
max_tags=2,
|
||||
progress=lambda k, r, why: events.append((k, why)),
|
||||
)
|
||||
assert stats == {
|
||||
"seen": 2,
|
||||
"tagged": 1,
|
||||
"skipped_state": 0,
|
||||
"no_chunks": 1,
|
||||
"tags_written": 2,
|
||||
}
|
||||
k = keys["CMS-2026-2377-1"]
|
||||
assert {"llm:telehealth", "llm:care-management"} <= set(s.get(k).tags)
|
||||
extra = json.loads(
|
||||
s._con()
|
||||
.execute("SELECT extra_json FROM items WHERE key=?", (k,))
|
||||
.fetchone()[0]
|
||||
) # noqa: SLF001
|
||||
assert extra["llm_tags"]["version"] == 2 and extra["llm_tags"]["model"] == "m"
|
||||
assert set(extra["llm_tags"]["tags"]) == {"telehealth", "care-management"}
|
||||
assert (keys["CMS-2026-2377-2"], "no-chunks") in events
|
||||
# second run: unchanged → skipped without judging
|
||||
judge_calls.clear()
|
||||
stats2 = run(
|
||||
s,
|
||||
vocab=VOCAB,
|
||||
docket="CMS-2026-2377",
|
||||
load_chunks=self._chunks(keys),
|
||||
card_vecs=CARDS,
|
||||
judge=judge,
|
||||
model="m",
|
||||
top=2,
|
||||
)
|
||||
assert (
|
||||
stats2["skipped_state"] == 1 and stats2["tagged"] == 0 and judge_calls == []
|
||||
)
|
||||
# a new model re-tags; --force re-tags
|
||||
assert (
|
||||
run(
|
||||
s,
|
||||
vocab=VOCAB,
|
||||
docket="CMS-2026-2377",
|
||||
load_chunks=self._chunks(keys),
|
||||
card_vecs=CARDS,
|
||||
judge=judge,
|
||||
model="m2",
|
||||
top=2,
|
||||
)["tagged"]
|
||||
== 1
|
||||
)
|
||||
assert (
|
||||
run(
|
||||
s,
|
||||
vocab=VOCAB,
|
||||
docket="CMS-2026-2377",
|
||||
load_chunks=self._chunks(keys),
|
||||
card_vecs=CARDS,
|
||||
judge=judge,
|
||||
model="m2",
|
||||
top=2,
|
||||
force=True,
|
||||
)["tagged"]
|
||||
== 1
|
||||
)
|
||||
s.close()
|
||||
|
||||
def test_dry_run_writes_nothing(self, tmp_path):
|
||||
s, keys = _store(tmp_path)
|
||||
stats = run(
|
||||
s,
|
||||
vocab=VOCAB,
|
||||
docket="CMS-2026-2377",
|
||||
load_chunks=self._chunks(keys),
|
||||
card_vecs=CARDS,
|
||||
judge=lambda t, e: True,
|
||||
model="m",
|
||||
dry_run=True,
|
||||
)
|
||||
assert stats["tagged"] == 1 and stats["tags_written"] == 0
|
||||
k = keys["CMS-2026-2377-1"]
|
||||
assert not any(t.startswith("llm:") for t in s.get(k).tags)
|
||||
s.close()
|
||||
114
tests/llm/test_vocab.py
Normal file
114
tests/llm/test_vocab.py
Normal file
@@ -0,0 +1,114 @@
|
||||
"""llm.vocab — the closed theme vocabulary (#574)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from llm.vocab import Theme, Vocab, VocabError, is_llm_tag, load, parse, split_tags
|
||||
|
||||
GOOD = """
|
||||
version: 3
|
||||
themes:
|
||||
- slug: telehealth
|
||||
label: Telehealth
|
||||
definition: Medicare telehealth services.
|
||||
synonyms: [telehealth, audio-only]
|
||||
sections: [Medicare Telehealth Services]
|
||||
- slug: care-management
|
||||
definition: Monthly care management codes.
|
||||
retired: [old-theme]
|
||||
"""
|
||||
|
||||
|
||||
class TestParse:
|
||||
def test_reads_themes_and_version(self):
|
||||
v = parse(GOOD, source="t")
|
||||
assert isinstance(v, Vocab) and v.version == 3 and len(v) == 2
|
||||
assert v.slugs == ("telehealth", "care-management")
|
||||
t = v.get("telehealth")
|
||||
assert isinstance(t, Theme) and t.synonyms == ("telehealth", "audio-only")
|
||||
assert (
|
||||
v.get("care-management").label == "care-management"
|
||||
) # label defaults to slug
|
||||
assert v.retired == ("old-theme",) and "telehealth" in v and "nope" not in v
|
||||
|
||||
def test_cards_and_tags(self):
|
||||
v = parse(GOOD)
|
||||
assert (
|
||||
v.cards["telehealth"]
|
||||
== "Telehealth. Medicare telehealth services. Also: telehealth, audio-only."
|
||||
)
|
||||
assert (
|
||||
v.cards["care-management"]
|
||||
== "care-management. Monthly care management codes."
|
||||
)
|
||||
assert (
|
||||
v.tag("telehealth") == "llm:telehealth"
|
||||
and v.get("telehealth").tag == "llm:telehealth"
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"text, msg",
|
||||
[
|
||||
(
|
||||
"version: 1\nthemes:\n - slug: Bad Slug\n definition: x\n",
|
||||
"kebab-case",
|
||||
),
|
||||
(
|
||||
"version: 1\nthemes:\n - slug: a\n definition: ''\n",
|
||||
"definition is empty",
|
||||
),
|
||||
(
|
||||
"version: 1\nthemes:\n - slug: a\n definition: x\n - slug: a\n definition: y\n",
|
||||
"duplicate slug",
|
||||
),
|
||||
(
|
||||
"version: 0\nthemes:\n - slug: a\n definition: x\n",
|
||||
"version must be >= 1",
|
||||
),
|
||||
("version: x\nthemes: []\n", "version must be an integer"),
|
||||
("version: 1\nthemes: []\n", "no themes"),
|
||||
("- a\n- b\n", "expected a mapping"),
|
||||
(
|
||||
"version: 1\nthemes:\n - slug: a\n definition: x\nretired: [a]\n",
|
||||
"retired slugs still active",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_rejects_bad_files(self, text, msg):
|
||||
with pytest.raises(VocabError, match=msg):
|
||||
parse(text, source="t")
|
||||
|
||||
|
||||
class TestPackagedVocab:
|
||||
def test_loads_and_is_sane(self):
|
||||
v = load()
|
||||
assert v.version >= 1 and 40 <= len(v) <= 150
|
||||
assert all(t.definition.endswith(".") for t in v.themes)
|
||||
assert all(len(t.synonyms) >= 1 for t in v.themes)
|
||||
for must in (
|
||||
"telehealth",
|
||||
"care-management",
|
||||
"conversion-factor",
|
||||
"practice-expense",
|
||||
"palliative-care",
|
||||
"skin-substitutes",
|
||||
):
|
||||
assert must in v
|
||||
|
||||
def test_load_from_path(self, tmp_path):
|
||||
p = tmp_path / "v.yaml"
|
||||
p.write_text(GOOD)
|
||||
assert load(p).source == str(p)
|
||||
|
||||
|
||||
class TestTagHelpers:
|
||||
def test_split(self):
|
||||
ours, rest = split_tags(
|
||||
["llm:telehealth", "topic:palliative", "llm:care-management", "docket:x"]
|
||||
)
|
||||
assert ours == ["llm:telehealth", "llm:care-management"] and rest == [
|
||||
"topic:palliative",
|
||||
"docket:x",
|
||||
]
|
||||
assert is_llm_tag("llm:x") and not is_llm_tag("llmx:y")
|
||||
Reference in New Issue
Block a user