From c354047e17e40834764ca7000efeb53e39448886 Mon Sep 17 00:00:00 2001 From: kert Date: Tue, 22 Sep 2026 16:58:41 -0400 Subject: [PATCH] =?UTF-8?q?feat(llm):=20golden-set=20gate=20for=20theme=20?= =?UTF-8?q?tagging=20=E2=80=94=20llm.tageval,=20tests/llm/golden=5Fthemes.?= =?UTF-8?q?yaml,=20stack=20llm=20eval-tags;=20vocab=20v2=20(refs=20#577)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Vocabulary v2 adds four themes the CY2027 docket's letters made unmissable: modifier-25-same-day, lactation-services, health-coaching, radiation-therapy (57 themes). tests/llm/golden_themes.yaml: 39 CMS-2026-2377 comments labelled by reading their opening text — expected themes (must carry) and acceptable ones (allowed, not required). llm.tageval scores predictions per slug (tp / fp outside expected∪acceptable / fn), micro precision/recall/F1 and the abstain rate, and gate() applies the spec thresholds (micro-F1 ≥ 0.70; no theme with support ≥ 5 under 0.5 precision). Predictions come from what the last run stored on the bib items (CI, no model calls) or --live through the chain. stack llm eval-tags [--golden] [--live] [--limit] exits 1 when the gate fails. --- docs/docs/cli/api-serve.md | 2 +- docs/docs/cli/api.md | 2 +- docs/docs/cli/db-comment.md | 2 +- docs/docs/cli/db-inspect.md | 2 +- docs/docs/cli/db.md | 2 +- docs/docs/cli/docs-build.md | 2 +- docs/docs/cli/docs-generate.md | 2 +- docs/docs/cli/docs-serve.md | 2 +- docs/docs/cli/docs.md | 2 +- docs/docs/cli/llm-eval-tags.md | 30 ++++ docs/docs/cli/llm.md | 41 +++-- docs/docs/cli/mail-attach-smarthost.md | 2 +- docs/docs/cli/mail-dkim-export.md | 2 +- docs/docs/cli/mail-dns.md | 2 +- docs/docs/cli/mail-down.md | 2 +- docs/docs/cli/mail-provision.md | 2 +- docs/docs/cli/mail-rotate-creds.md | 2 +- docs/docs/cli/mail-seed-mailboxes.md | 2 +- docs/docs/cli/mail-status.md | 2 +- docs/docs/cli/mail-up.md | 2 +- docs/docs/cli/mail-wire-git.md | 2 +- docs/docs/cli/mail.md | 2 +- docs/docs/cli/perf-show.md | 2 +- docs/docs/cli/perf.md | 2 +- docs/docs/cli/pfs-cpt-ingest.md | 2 +- docs/docs/cli/pfs-elements.md | 2 +- docs/docs/cli/pfs-exposure.md | 2 +- docs/docs/cli/pfs-families.md | 2 +- docs/docs/cli/pfs-guidance.md | 2 +- docs/docs/cli/pfs-lineage.md | 2 +- docs/docs/cli/pfs-reaction.md | 2 +- docs/docs/cli/pfs-review.md | 2 +- docs/docs/cli/pfs-utilization.md | 2 +- docs/docs/cli/pfs.md | 2 +- docs/docs/cli/prisma-eligible.md | 2 +- docs/docs/cli/prisma-export.md | 2 +- docs/docs/cli/prisma-extract.md | 2 +- docs/docs/cli/prisma-fetch.md | 2 +- docs/docs/cli/prisma-flow.md | 2 +- docs/docs/cli/prisma-init.md | 2 +- docs/docs/cli/prisma-ping-llm.md | 2 +- docs/docs/cli/prisma-run.md | 2 +- docs/docs/cli/prisma-screen.md | 2 +- docs/docs/cli/prisma-vpn.md | 2 +- docs/docs/cli/prisma.md | 2 +- docs/docs/cli/rec-list.md | 2 +- docs/docs/cli/rec-opps.md | 2 +- docs/docs/cli/rec-pfs.md | 2 +- docs/docs/cli/rec.md | 2 +- docs/docs/cli/zot-dump-schema.md | 2 +- docs/docs/cli/zot-fix-dates.md | 2 +- docs/docs/cli/zot-fix-fields.md | 2 +- docs/docs/cli/zot-fix-keys.md | 2 +- docs/docs/cli/zot-verify-parity.md | 2 +- docs/docs/cli/zot.md | 2 +- src/cli/llm.py | 83 +++++++++ src/llm/tageval.py | 229 +++++++++++++++++++++++++ src/llm/vocab/themes.yaml | 22 ++- tests/cli/test_llm_tag_cli.py | 42 +++++ tests/llm/golden_themes.yaml | 48 ++++++ tests/llm/test_tageval.py | 150 ++++++++++++++++ 61 files changed, 683 insertions(+), 68 deletions(-) create mode 100644 docs/docs/cli/llm-eval-tags.md create mode 100644 src/llm/tageval.py create mode 100644 tests/llm/golden_themes.yaml create mode 100644 tests/llm/test_tageval.py diff --git a/docs/docs/cli/api-serve.md b/docs/docs/cli/api-serve.md index 6bc5db2..ce80a41 100644 --- a/docs/docs/cli/api-serve.md +++ b/docs/docs/cli/api-serve.md @@ -1,6 +1,6 @@ --- title: stack api serve -sidebar_position: 63 +sidebar_position: 64 --- # `stack api serve` diff --git a/docs/docs/cli/api.md b/docs/docs/cli/api.md index 3c8cbbe..c466486 100644 --- a/docs/docs/cli/api.md +++ b/docs/docs/cli/api.md @@ -1,6 +1,6 @@ --- title: stack api -sidebar_position: 62 +sidebar_position: 63 --- # `stack api` diff --git a/docs/docs/cli/db-comment.md b/docs/docs/cli/db-comment.md index b3364a1..28a9ab4 100644 --- a/docs/docs/cli/db-comment.md +++ b/docs/docs/cli/db-comment.md @@ -1,6 +1,6 @@ --- title: stack db comment -sidebar_position: 56 +sidebar_position: 57 --- # `stack db comment` diff --git a/docs/docs/cli/db-inspect.md b/docs/docs/cli/db-inspect.md index 78db95e..911199c 100644 --- a/docs/docs/cli/db-inspect.md +++ b/docs/docs/cli/db-inspect.md @@ -1,6 +1,6 @@ --- title: stack db inspect -sidebar_position: 57 +sidebar_position: 58 --- # `stack db inspect` diff --git a/docs/docs/cli/db.md b/docs/docs/cli/db.md index 529cd59..5b738e0 100644 --- a/docs/docs/cli/db.md +++ b/docs/docs/cli/db.md @@ -1,6 +1,6 @@ --- title: stack db -sidebar_position: 55 +sidebar_position: 56 --- # `stack db` diff --git a/docs/docs/cli/docs-build.md b/docs/docs/cli/docs-build.md index d37c434..bba50ea 100644 --- a/docs/docs/cli/docs-build.md +++ b/docs/docs/cli/docs-build.md @@ -1,6 +1,6 @@ --- title: stack docs build -sidebar_position: 59 +sidebar_position: 60 --- # `stack docs build` diff --git a/docs/docs/cli/docs-generate.md b/docs/docs/cli/docs-generate.md index e37d9dd..da90d79 100644 --- a/docs/docs/cli/docs-generate.md +++ b/docs/docs/cli/docs-generate.md @@ -1,6 +1,6 @@ --- title: stack docs generate -sidebar_position: 61 +sidebar_position: 62 --- # `stack docs generate` diff --git a/docs/docs/cli/docs-serve.md b/docs/docs/cli/docs-serve.md index 76a6fd1..77749c2 100644 --- a/docs/docs/cli/docs-serve.md +++ b/docs/docs/cli/docs-serve.md @@ -1,6 +1,6 @@ --- title: stack docs serve -sidebar_position: 60 +sidebar_position: 61 --- # `stack docs serve` diff --git a/docs/docs/cli/docs.md b/docs/docs/cli/docs.md index 09edd1b..f99826d 100644 --- a/docs/docs/cli/docs.md +++ b/docs/docs/cli/docs.md @@ -1,6 +1,6 @@ --- title: stack docs -sidebar_position: 58 +sidebar_position: 59 --- # `stack docs` diff --git a/docs/docs/cli/llm-eval-tags.md b/docs/docs/cli/llm-eval-tags.md new file mode 100644 index 0000000..dd48296 --- /dev/null +++ b/docs/docs/cli/llm-eval-tags.md @@ -0,0 +1,30 @@ +--- +title: stack llm eval-tags +sidebar_position: 55 +--- + +# `stack llm eval-tags` + +``` +Usage: stack llm eval-tags [OPTIONS] + + Score the theme tagger against the golden set (P35 gate, #577): per-theme + precision/recall, micro-F1 and the abstain rate, then the fan-out gate + (micro-F1 ≥ 0.70; no theme with support ≥ 5 under 0.5 precision). Default + reads the tags the last run stored on each item (no model calls); --live + re-tags each golden comment now. Exit 1 when the gate fails. + +╭─ Options ────────────────────────────────────────────────────────────────────╮ +│ --golden TEXT Golden YAML (comment ids with expected │ +│ themes). │ +│ [default: tests/llm/golden_themes.yaml] │ +│ --live Re-tag through the model instead of scoring │ +│ what is stored on the items. │ +│ --limit INTEGER Score only the first N golden entries. │ +│ [default: 0] │ +│ --top INTEGER [default: 8] │ +│ --max-tags INTEGER [default: 4] │ +│ --min-confidence FLOAT [default: 0.0] │ +│ --help Show this message and exit. │ +╰──────────────────────────────────────────────────────────────────────────────╯ +``` diff --git a/docs/docs/cli/llm.md b/docs/docs/cli/llm.md index a04923d..28f9cab 100644 --- a/docs/docs/cli/llm.md +++ b/docs/docs/cli/llm.md @@ -14,19 +14,32 @@ Usage: stack llm [OPTIONS] COMMAND [ARGS]... │ --help Show this message and exit. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ╭─ Commands ───────────────────────────────────────────────────────────────────╮ -│ index Embed comments/rules/corpus into pgvector (incremental, resumable). │ -│ restamp Backfill codes/families/elements onto already-indexed chunks. │ -│ hosts Show the Ollama fleet: declared VRAM, liveness, models, and which │ -│ host + model would answer a chat right now. │ -│ serve Serve the SSO-guarded chat UI (llm.api:app). │ -│ vocab The closed theme vocabulary for comment tagging (#574): validate it │ -│ and list its slugs, or show one theme's definition, synonyms and │ -│ the FR │ -│ section stems it was seeded from. │ -│ tag Closed-vocabulary theme tagging of one docket's comments (P35): │ -│ shortlist by similarity to the theme cards, judge each candidate │ -│ yes/no on the largest live host with the scoring chunk as evidence, │ -│ record state in the item's extra_json, and replace its llm: tags. │ -│ Resumable — unchanged comments are skipped unless --force. │ +│ index Embed comments/rules/corpus into pgvector (incremental, │ +│ resumable). │ +│ restamp Backfill codes/families/elements onto already-indexed chunks. │ +│ hosts Show the Ollama fleet: declared VRAM, liveness, models, and which │ +│ host + model would answer a chat right now. │ +│ serve Serve the SSO-guarded chat UI (llm.api:app). │ +│ vocab The closed theme vocabulary for comment tagging (#574): validate │ +│ it │ +│ and list its slugs, or show one theme's definition, synonyms and │ +│ the FR │ +│ section stems it was seeded from. │ +│ tag Closed-vocabulary theme tagging of one docket's comments (P35): │ +│ shortlist by similarity to the theme cards, judge each candidate │ +│ yes/no on the largest live host with the scoring chunk as │ +│ evidence, │ +│ record state in the item's extra_json, and replace its llm: tags. │ +│ Resumable — unchanged comments are skipped unless --force. │ +│ eval-tags Score the theme tagger against the golden set (P35 gate, #577): │ +│ per-theme precision/recall, micro-F1 and the abstain rate, then │ +│ the │ +│ fan-out gate (micro-F1 ≥ 0.70; no theme with support ≥ 5 under │ +│ 0.5 │ +│ precision). Default reads the tags the last run stored on each │ +│ item │ +│ (no model calls); --live re-tags each golden comment now. Exit 1 │ +│ when │ +│ the gate fails. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/docs/docs/cli/mail-attach-smarthost.md b/docs/docs/cli/mail-attach-smarthost.md index 7758fbd..4cbe19b 100644 --- a/docs/docs/cli/mail-attach-smarthost.md +++ b/docs/docs/cli/mail-attach-smarthost.md @@ -1,6 +1,6 @@ --- title: stack mail attach-smarthost -sidebar_position: 104 +sidebar_position: 105 --- # `stack mail attach-smarthost` diff --git a/docs/docs/cli/mail-dkim-export.md b/docs/docs/cli/mail-dkim-export.md index 9c67c22..0724b37 100644 --- a/docs/docs/cli/mail-dkim-export.md +++ b/docs/docs/cli/mail-dkim-export.md @@ -1,6 +1,6 @@ --- title: stack mail dkim-export -sidebar_position: 103 +sidebar_position: 104 --- # `stack mail dkim-export` diff --git a/docs/docs/cli/mail-dns.md b/docs/docs/cli/mail-dns.md index e1730e5..b108f47 100644 --- a/docs/docs/cli/mail-dns.md +++ b/docs/docs/cli/mail-dns.md @@ -1,6 +1,6 @@ --- title: stack mail dns -sidebar_position: 102 +sidebar_position: 103 --- # `stack mail dns` diff --git a/docs/docs/cli/mail-down.md b/docs/docs/cli/mail-down.md index 2deb87e..911c7e6 100644 --- a/docs/docs/cli/mail-down.md +++ b/docs/docs/cli/mail-down.md @@ -1,6 +1,6 @@ --- title: stack mail down -sidebar_position: 100 +sidebar_position: 101 --- # `stack mail down` diff --git a/docs/docs/cli/mail-provision.md b/docs/docs/cli/mail-provision.md index f6fe5b0..58d81b7 100644 --- a/docs/docs/cli/mail-provision.md +++ b/docs/docs/cli/mail-provision.md @@ -1,6 +1,6 @@ --- title: stack mail provision -sidebar_position: 98 +sidebar_position: 99 --- # `stack mail provision` diff --git a/docs/docs/cli/mail-rotate-creds.md b/docs/docs/cli/mail-rotate-creds.md index 1f80c5f..56d934f 100644 --- a/docs/docs/cli/mail-rotate-creds.md +++ b/docs/docs/cli/mail-rotate-creds.md @@ -1,6 +1,6 @@ --- title: stack mail rotate-creds -sidebar_position: 105 +sidebar_position: 106 --- # `stack mail rotate-creds` diff --git a/docs/docs/cli/mail-seed-mailboxes.md b/docs/docs/cli/mail-seed-mailboxes.md index 3dada3d..fc30540 100644 --- a/docs/docs/cli/mail-seed-mailboxes.md +++ b/docs/docs/cli/mail-seed-mailboxes.md @@ -1,6 +1,6 @@ --- title: stack mail seed-mailboxes -sidebar_position: 106 +sidebar_position: 107 --- # `stack mail seed-mailboxes` diff --git a/docs/docs/cli/mail-status.md b/docs/docs/cli/mail-status.md index 9ed0514..a72e839 100644 --- a/docs/docs/cli/mail-status.md +++ b/docs/docs/cli/mail-status.md @@ -1,6 +1,6 @@ --- title: stack mail status -sidebar_position: 101 +sidebar_position: 102 --- # `stack mail status` diff --git a/docs/docs/cli/mail-up.md b/docs/docs/cli/mail-up.md index c39cb51..bc2d2ee 100644 --- a/docs/docs/cli/mail-up.md +++ b/docs/docs/cli/mail-up.md @@ -1,6 +1,6 @@ --- title: stack mail up -sidebar_position: 99 +sidebar_position: 100 --- # `stack mail up` diff --git a/docs/docs/cli/mail-wire-git.md b/docs/docs/cli/mail-wire-git.md index df87669..bae1ee0 100644 --- a/docs/docs/cli/mail-wire-git.md +++ b/docs/docs/cli/mail-wire-git.md @@ -1,6 +1,6 @@ --- title: stack mail wire-git -sidebar_position: 107 +sidebar_position: 108 --- # `stack mail wire-git` diff --git a/docs/docs/cli/mail.md b/docs/docs/cli/mail.md index b7ef94b..31456ac 100644 --- a/docs/docs/cli/mail.md +++ b/docs/docs/cli/mail.md @@ -1,6 +1,6 @@ --- title: stack mail -sidebar_position: 97 +sidebar_position: 98 --- # `stack mail` diff --git a/docs/docs/cli/perf-show.md b/docs/docs/cli/perf-show.md index f1e0c71..b812aa0 100644 --- a/docs/docs/cli/perf-show.md +++ b/docs/docs/cli/perf-show.md @@ -1,6 +1,6 @@ --- title: stack perf show -sidebar_position: 65 +sidebar_position: 66 --- # `stack perf show` diff --git a/docs/docs/cli/perf.md b/docs/docs/cli/perf.md index 04402e7..ad52aaa 100644 --- a/docs/docs/cli/perf.md +++ b/docs/docs/cli/perf.md @@ -1,6 +1,6 @@ --- title: stack perf -sidebar_position: 64 +sidebar_position: 65 --- # `stack perf` diff --git a/docs/docs/cli/pfs-cpt-ingest.md b/docs/docs/cli/pfs-cpt-ingest.md index c3cbd82..0d8884c 100644 --- a/docs/docs/cli/pfs-cpt-ingest.md +++ b/docs/docs/cli/pfs-cpt-ingest.md @@ -1,6 +1,6 @@ --- title: stack pfs cpt-ingest -sidebar_position: 79 +sidebar_position: 80 --- # `stack pfs cpt-ingest` diff --git a/docs/docs/cli/pfs-elements.md b/docs/docs/cli/pfs-elements.md index b30a493..6c482bd 100644 --- a/docs/docs/cli/pfs-elements.md +++ b/docs/docs/cli/pfs-elements.md @@ -1,6 +1,6 @@ --- title: stack pfs elements -sidebar_position: 71 +sidebar_position: 72 --- # `stack pfs elements` diff --git a/docs/docs/cli/pfs-exposure.md b/docs/docs/cli/pfs-exposure.md index 32c8775..a6b3aff 100644 --- a/docs/docs/cli/pfs-exposure.md +++ b/docs/docs/cli/pfs-exposure.md @@ -1,6 +1,6 @@ --- title: stack pfs exposure -sidebar_position: 76 +sidebar_position: 77 --- # `stack pfs exposure` diff --git a/docs/docs/cli/pfs-families.md b/docs/docs/cli/pfs-families.md index f120337..83493f6 100644 --- a/docs/docs/cli/pfs-families.md +++ b/docs/docs/cli/pfs-families.md @@ -1,6 +1,6 @@ --- title: stack pfs families -sidebar_position: 73 +sidebar_position: 74 --- # `stack pfs families` diff --git a/docs/docs/cli/pfs-guidance.md b/docs/docs/cli/pfs-guidance.md index 8f4ff52..40926eb 100644 --- a/docs/docs/cli/pfs-guidance.md +++ b/docs/docs/cli/pfs-guidance.md @@ -1,6 +1,6 @@ --- title: stack pfs guidance -sidebar_position: 74 +sidebar_position: 75 --- # `stack pfs guidance` diff --git a/docs/docs/cli/pfs-lineage.md b/docs/docs/cli/pfs-lineage.md index 6dcc13c..12ee1b3 100644 --- a/docs/docs/cli/pfs-lineage.md +++ b/docs/docs/cli/pfs-lineage.md @@ -1,6 +1,6 @@ --- title: stack pfs lineage -sidebar_position: 72 +sidebar_position: 73 --- # `stack pfs lineage` diff --git a/docs/docs/cli/pfs-reaction.md b/docs/docs/cli/pfs-reaction.md index 529be24..b3d2a48 100644 --- a/docs/docs/cli/pfs-reaction.md +++ b/docs/docs/cli/pfs-reaction.md @@ -1,6 +1,6 @@ --- title: stack pfs reaction -sidebar_position: 75 +sidebar_position: 76 --- # `stack pfs reaction` diff --git a/docs/docs/cli/pfs-review.md b/docs/docs/cli/pfs-review.md index 0bf7a12..225e759 100644 --- a/docs/docs/cli/pfs-review.md +++ b/docs/docs/cli/pfs-review.md @@ -1,6 +1,6 @@ --- title: stack pfs review -sidebar_position: 78 +sidebar_position: 79 --- # `stack pfs review` diff --git a/docs/docs/cli/pfs-utilization.md b/docs/docs/cli/pfs-utilization.md index f98974b..1783d6a 100644 --- a/docs/docs/cli/pfs-utilization.md +++ b/docs/docs/cli/pfs-utilization.md @@ -1,6 +1,6 @@ --- title: stack pfs utilization -sidebar_position: 77 +sidebar_position: 78 --- # `stack pfs utilization` diff --git a/docs/docs/cli/pfs.md b/docs/docs/cli/pfs.md index 7fb6804..5345606 100644 --- a/docs/docs/cli/pfs.md +++ b/docs/docs/cli/pfs.md @@ -1,6 +1,6 @@ --- title: stack pfs -sidebar_position: 70 +sidebar_position: 71 --- # `stack pfs` diff --git a/docs/docs/cli/prisma-eligible.md b/docs/docs/cli/prisma-eligible.md index 62fff2e..a241cdb 100644 --- a/docs/docs/cli/prisma-eligible.md +++ b/docs/docs/cli/prisma-eligible.md @@ -1,6 +1,6 @@ --- title: stack prisma eligible -sidebar_position: 91 +sidebar_position: 92 --- # `stack prisma eligible` diff --git a/docs/docs/cli/prisma-export.md b/docs/docs/cli/prisma-export.md index 849b725..3e04e02 100644 --- a/docs/docs/cli/prisma-export.md +++ b/docs/docs/cli/prisma-export.md @@ -1,6 +1,6 @@ --- title: stack prisma export -sidebar_position: 88 +sidebar_position: 89 --- # `stack prisma export` diff --git a/docs/docs/cli/prisma-extract.md b/docs/docs/cli/prisma-extract.md index fe02a9b..c007761 100644 --- a/docs/docs/cli/prisma-extract.md +++ b/docs/docs/cli/prisma-extract.md @@ -1,6 +1,6 @@ --- title: stack prisma extract -sidebar_position: 92 +sidebar_position: 93 --- # `stack prisma extract` diff --git a/docs/docs/cli/prisma-fetch.md b/docs/docs/cli/prisma-fetch.md index 8c7ceda..1086a9e 100644 --- a/docs/docs/cli/prisma-fetch.md +++ b/docs/docs/cli/prisma-fetch.md @@ -1,6 +1,6 @@ --- title: stack prisma fetch -sidebar_position: 94 +sidebar_position: 95 --- # `stack prisma fetch` diff --git a/docs/docs/cli/prisma-flow.md b/docs/docs/cli/prisma-flow.md index 97a7eab..a887c1b 100644 --- a/docs/docs/cli/prisma-flow.md +++ b/docs/docs/cli/prisma-flow.md @@ -1,6 +1,6 @@ --- title: stack prisma flow -sidebar_position: 93 +sidebar_position: 94 --- # `stack prisma flow` diff --git a/docs/docs/cli/prisma-init.md b/docs/docs/cli/prisma-init.md index 6ed98a4..168f131 100644 --- a/docs/docs/cli/prisma-init.md +++ b/docs/docs/cli/prisma-init.md @@ -1,6 +1,6 @@ --- title: stack prisma init -sidebar_position: 87 +sidebar_position: 88 --- # `stack prisma init` diff --git a/docs/docs/cli/prisma-ping-llm.md b/docs/docs/cli/prisma-ping-llm.md index 195b9ad..2ba37e0 100644 --- a/docs/docs/cli/prisma-ping-llm.md +++ b/docs/docs/cli/prisma-ping-llm.md @@ -1,6 +1,6 @@ --- title: stack prisma ping-llm -sidebar_position: 89 +sidebar_position: 90 --- # `stack prisma ping-llm` diff --git a/docs/docs/cli/prisma-run.md b/docs/docs/cli/prisma-run.md index 00db0a0..8b8e9fa 100644 --- a/docs/docs/cli/prisma-run.md +++ b/docs/docs/cli/prisma-run.md @@ -1,6 +1,6 @@ --- title: stack prisma run -sidebar_position: 95 +sidebar_position: 96 --- # `stack prisma run` diff --git a/docs/docs/cli/prisma-screen.md b/docs/docs/cli/prisma-screen.md index a0e9da0..72bfd9a 100644 --- a/docs/docs/cli/prisma-screen.md +++ b/docs/docs/cli/prisma-screen.md @@ -1,6 +1,6 @@ --- title: stack prisma screen -sidebar_position: 90 +sidebar_position: 91 --- # `stack prisma screen` diff --git a/docs/docs/cli/prisma-vpn.md b/docs/docs/cli/prisma-vpn.md index d9f9625..f92137f 100644 --- a/docs/docs/cli/prisma-vpn.md +++ b/docs/docs/cli/prisma-vpn.md @@ -1,6 +1,6 @@ --- title: stack prisma vpn -sidebar_position: 96 +sidebar_position: 97 --- # `stack prisma vpn` diff --git a/docs/docs/cli/prisma.md b/docs/docs/cli/prisma.md index 457471a..60d9b30 100644 --- a/docs/docs/cli/prisma.md +++ b/docs/docs/cli/prisma.md @@ -1,6 +1,6 @@ --- title: stack prisma -sidebar_position: 86 +sidebar_position: 87 --- # `stack prisma` diff --git a/docs/docs/cli/rec-list.md b/docs/docs/cli/rec-list.md index 313cf27..d4d329f 100644 --- a/docs/docs/cli/rec-list.md +++ b/docs/docs/cli/rec-list.md @@ -1,6 +1,6 @@ --- title: stack rec list -sidebar_position: 67 +sidebar_position: 68 --- # `stack rec list` diff --git a/docs/docs/cli/rec-opps.md b/docs/docs/cli/rec-opps.md index da62a2c..74a563a 100644 --- a/docs/docs/cli/rec-opps.md +++ b/docs/docs/cli/rec-opps.md @@ -1,6 +1,6 @@ --- title: stack rec opps -sidebar_position: 69 +sidebar_position: 70 --- # `stack rec opps` diff --git a/docs/docs/cli/rec-pfs.md b/docs/docs/cli/rec-pfs.md index 3ba7b6f..2ea9197 100644 --- a/docs/docs/cli/rec-pfs.md +++ b/docs/docs/cli/rec-pfs.md @@ -1,6 +1,6 @@ --- title: stack rec pfs -sidebar_position: 68 +sidebar_position: 69 --- # `stack rec pfs` diff --git a/docs/docs/cli/rec.md b/docs/docs/cli/rec.md index 14ae46c..48447e6 100644 --- a/docs/docs/cli/rec.md +++ b/docs/docs/cli/rec.md @@ -1,6 +1,6 @@ --- title: stack rec -sidebar_position: 66 +sidebar_position: 67 --- # `stack rec` diff --git a/docs/docs/cli/zot-dump-schema.md b/docs/docs/cli/zot-dump-schema.md index af6928d..60bac48 100644 --- a/docs/docs/cli/zot-dump-schema.md +++ b/docs/docs/cli/zot-dump-schema.md @@ -1,6 +1,6 @@ --- title: stack zot dump-schema -sidebar_position: 81 +sidebar_position: 82 --- # `stack zot dump-schema` diff --git a/docs/docs/cli/zot-fix-dates.md b/docs/docs/cli/zot-fix-dates.md index 5211d99..f363269 100644 --- a/docs/docs/cli/zot-fix-dates.md +++ b/docs/docs/cli/zot-fix-dates.md @@ -1,6 +1,6 @@ --- title: stack zot fix-dates -sidebar_position: 82 +sidebar_position: 83 --- # `stack zot fix-dates` diff --git a/docs/docs/cli/zot-fix-fields.md b/docs/docs/cli/zot-fix-fields.md index 396cef0..9a63d61 100644 --- a/docs/docs/cli/zot-fix-fields.md +++ b/docs/docs/cli/zot-fix-fields.md @@ -1,6 +1,6 @@ --- title: stack zot fix-fields -sidebar_position: 84 +sidebar_position: 85 --- # `stack zot fix-fields` diff --git a/docs/docs/cli/zot-fix-keys.md b/docs/docs/cli/zot-fix-keys.md index 120dbd0..78d9178 100644 --- a/docs/docs/cli/zot-fix-keys.md +++ b/docs/docs/cli/zot-fix-keys.md @@ -1,6 +1,6 @@ --- title: stack zot fix-keys -sidebar_position: 83 +sidebar_position: 84 --- # `stack zot fix-keys` diff --git a/docs/docs/cli/zot-verify-parity.md b/docs/docs/cli/zot-verify-parity.md index 8637bcb..502a03a 100644 --- a/docs/docs/cli/zot-verify-parity.md +++ b/docs/docs/cli/zot-verify-parity.md @@ -1,6 +1,6 @@ --- title: stack zot verify-parity -sidebar_position: 85 +sidebar_position: 86 --- # `stack zot verify-parity` diff --git a/docs/docs/cli/zot.md b/docs/docs/cli/zot.md index 87587ae..454d657 100644 --- a/docs/docs/cli/zot.md +++ b/docs/docs/cli/zot.md @@ -1,6 +1,6 @@ --- title: stack zot -sidebar_position: 80 +sidebar_position: 81 --- # `stack zot` diff --git a/src/cli/llm.py b/src/cli/llm.py index 5ebe3e4..459663d 100644 --- a/src/cli/llm.py +++ b/src/cli/llm.py @@ -338,3 +338,86 @@ def tag( f"seen={stats['seen']} tagged={stats['tagged']} unchanged={stats['skipped_state']} " f"no_chunks={stats['no_chunks']} tags_written={stats['tags_written']}" ) + + +_GOLDEN_DEFAULT = "tests/llm/golden_themes.yaml" + + +@app.command("eval-tags") +def eval_tags( + golden: str = typer.Option( + _GOLDEN_DEFAULT, + "--golden", + help="Golden YAML (comment ids with expected themes).", + ), + live: bool = typer.Option( + False, + "--live", + help="Re-tag through the model instead of scoring what is stored on the items.", + ), + limit: int = typer.Option( + 0, "--limit", help="Score only the first N golden entries." + ), + top: int = typer.Option(8, "--top"), + max_tags: int = typer.Option(4, "--max-tags"), + min_confidence: float = typer.Option(0.0, "--min-confidence"), +) -> None: + """Score the theme tagger against the golden set (P35 gate, #577): + per-theme precision/recall, micro-F1 and the abstain rate, then the + fan-out gate (micro-F1 ≥ 0.70; no theme with support ≥ 5 under 0.5 + precision). Default reads the tags the last run stored on each item + (no model calls); --live re-tags each golden comment now. Exit 1 when + the gate fails.""" + from conf.connect import bib + from llm.tageval import ( + format_report, + gate, + live_predictions, + load_golden, + score, + stored_predictions, + ) + from llm.vocab import load as load_vocab + + g = load_golden(golden) + v = load_vocab() + if g.vocab_version != v.version: + typer.echo( + f"warning: golden set labelled against vocab v{g.vocab_version}, current is v{v.version}" + ) + entries = g.entries[:limit] if limit else g.entries + ids = [e.comment_id for e in entries] + if live: + from llm import config as llm_config + from llm.tagger import tag_comment + + rt = _tagging_runtime(llm_config.load()) + + def tag_one(key: str): + rows = sorted(rt["load_chunks"](key), key=lambda r: r[0]) + if not rows: + return None + r = tag_comment( + key, + [t for _, t, _ in rows], + [vec for _, _, vec in rows], + vocab=rt["vocab"], + card_vecs=rt["card_vecs"], + judge=rt["judge"], + top=top, + max_tags=max_tags, + min_confidence=min_confidence, + ) + typer.echo(f"{key} {', '.join(r.slugs) or '(none)'}") + return r.slugs + + preds = live_predictions(rt["store"], ids, tag=tag_one) + else: + preds = stored_predictions(bib(), ids) + from llm.tageval import Golden + + report = score(Golden(g.vocab_version, g.docket, tuple(entries)), preds) + reasons = gate(report) + typer.echo(format_report(report, reasons=reasons)) + if reasons: + raise typer.Exit(1) diff --git a/src/llm/tageval.py b/src/llm/tageval.py new file mode 100644 index 0000000..6f94e79 --- /dev/null +++ b/src/llm/tageval.py @@ -0,0 +1,229 @@ +"""Golden-set evaluation of the theme tagger (P35, #577) — the gate before +corpus-wide fan-out. + +A golden file (``tests/llm/golden_themes.yaml``) lists comments with the +themes each must carry (``expected``) and themes that are acceptable but +not required (``acceptable``). Predictions come either from what is +already stored on the bib items (``extra_json.llm_tags.tags`` — no model +calls, what CI checks) or live from ``llm.tagger.tag_comment``. + +Scoring is per slug: a predicted theme in ``expected`` is a true +positive, one in neither list a false positive, a missing ``expected`` +theme a false negative; ``acceptable`` predictions are neither rewarded +nor punished. Micro precision/recall/F1 aggregate the counts; the abstain +rate is the share of comments with no prediction at all. ``gate()`` +applies the spec's thresholds: micro-F1 ≥ 0.70 and no theme with +support ≥ 5 below 0.5 precision. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable, Mapping, Sequence + +import yaml + +MIN_MICRO_F1 = 0.70 +MIN_THEME_PRECISION = 0.50 +MIN_SUPPORT = 5 + + +@dataclass(frozen=True) +class GoldenEntry: + comment_id: str + expected: tuple[str, ...] + acceptable: tuple[str, ...] = () + + +@dataclass(frozen=True) +class Golden: + vocab_version: int + docket: str + entries: tuple[GoldenEntry, ...] + + +@dataclass +class SlugScore: + slug: str + tp: int = 0 + fp: int = 0 + fn: int = 0 + + @property + def support(self) -> int: + return self.tp + self.fn + + @property + def precision(self) -> float | None: + return self.tp / (self.tp + self.fp) if (self.tp + self.fp) else None + + @property + def recall(self) -> float | None: + return self.tp / (self.tp + self.fn) if (self.tp + self.fn) else None + + +@dataclass +class Report: + n: int + scored: int + missing: tuple[str, ...] + per_slug: dict[str, SlugScore] = field(default_factory=dict) + abstained: int = 0 + + @property + def tp(self) -> int: + return sum(s.tp for s in self.per_slug.values()) + + @property + def fp(self) -> int: + return sum(s.fp for s in self.per_slug.values()) + + @property + def fn(self) -> int: + return sum(s.fn for s in self.per_slug.values()) + + @property + def precision(self) -> float: + return self.tp / (self.tp + self.fp) if (self.tp + self.fp) else 0.0 + + @property + def recall(self) -> float: + return self.tp / (self.tp + self.fn) if (self.tp + self.fn) else 0.0 + + @property + def f1(self) -> float: + p, r = self.precision, self.recall + return 2 * p * r / (p + r) if (p + r) else 0.0 + + @property + def abstain_rate(self) -> float: + return self.abstained / self.scored if self.scored else 0.0 + + +def load_golden(path: Path | str) -> Golden: + data = yaml.safe_load(Path(path).read_text(encoding="utf-8")) or {} + entries = tuple( + GoldenEntry( + comment_id=str(e["comment_id"]), + expected=tuple(e.get("expected") or []), + acceptable=tuple(e.get("acceptable") or []), + ) + for e in data.get("entries") or [] + ) + return Golden( + int(data.get("vocab_version", 0)), str(data.get("docket") or ""), entries + ) + + +def score(golden: Golden, predictions: Mapping[str, Sequence[str] | None]) -> Report: + """*predictions*: comment id → predicted slugs, or ``None`` when the + comment could not be scored (not indexed / not found) — those are + reported as ``missing`` and excluded from every rate.""" + per: dict[str, SlugScore] = {} + missing: list[str] = [] + scored = abstained = 0 + for e in golden.entries: + pred = predictions.get(e.comment_id) + if pred is None: + missing.append(e.comment_id) + continue + scored += 1 + pset = set(pred) + if not pset: + abstained += 1 + for slug in e.expected: + per.setdefault(slug, SlugScore(slug)) + if slug in pset: + per[slug].tp += 1 + else: + per[slug].fn += 1 + for slug in pset - set(e.expected) - set(e.acceptable): + per.setdefault(slug, SlugScore(slug)).fp += 1 + return Report(len(golden.entries), scored, tuple(missing), per, abstained) + + +def gate(report: Report) -> list[str]: + """Every reason the gate fails (empty = pass).""" + reasons: list[str] = [] + if report.scored == 0: + return ["nothing scored"] + if report.f1 < MIN_MICRO_F1: + reasons.append(f"micro-F1 {report.f1:.2f} < {MIN_MICRO_F1:.2f}") + for s in sorted(report.per_slug.values(), key=lambda s: s.slug): + p = s.precision + if s.support >= MIN_SUPPORT and p is not None and p < MIN_THEME_PRECISION: + reasons.append( + f"{s.slug}: precision {p:.2f} < {MIN_THEME_PRECISION:.2f} (support {s.support})" + ) + return reasons + + +# ── prediction sources ──────────────────────────────────────────────── + + +def stored_predictions( + store: Any, comment_ids: Sequence[str] +) -> dict[str, list[str] | None]: + """What the last tagging run wrote on each comment's bib item + (``extra_json.llm_tags.tags``); ``None`` when the item is unknown or + was never tagged.""" + import json + + con = store._con() # noqa: SLF001 + out: dict[str, list[str] | None] = {} + for cid in comment_ids: + row = con.execute( + "SELECT extra_json FROM items WHERE url = ?", + (f"https://www.regulations.gov/comment/{cid}",), + ).fetchone() + if row is None: + out[cid] = None + continue + st = (json.loads(row[0] or "{}") if row[0] else {}).get("llm_tags") + out[cid] = sorted(st.get("tags") or {}) if st else None + return out + + +def live_predictions( + store: Any, + comment_ids: Sequence[str], + *, + tag: Callable[[str], Sequence[str] | None], +) -> dict[str, list[str] | None]: + """Re-tag each comment through *tag(key) → slugs* (the CLI binds the + real chain); unknown comments are ``None``.""" + con = store._con() # noqa: SLF001 + out: dict[str, list[str] | None] = {} + for cid in comment_ids: + row = con.execute( + "SELECT key FROM items WHERE url = ?", + (f"https://www.regulations.gov/comment/{cid}",), + ).fetchone() + if row is None: + out[cid] = None + continue + slugs = tag(row[0]) + out[cid] = sorted(slugs) if slugs is not None else None + return out + + +def format_report(report: Report, *, reasons: Sequence[str]) -> str: + lines = [ + f"golden: {report.n} entries, {report.scored} scored, {len(report.missing)} missing" + + ( + f" ({', '.join(report.missing[:5])}{'…' if len(report.missing) > 5 else ''})" + if report.missing + else "" + ), + f"micro precision {report.precision:.2f} recall {report.recall:.2f} F1 {report.f1:.2f} " + f"abstain {report.abstain_rate:.0%} (tp={report.tp} fp={report.fp} fn={report.fn})", + ] + for s in sorted(report.per_slug.values(), key=lambda s: (-s.support, s.slug)): + p = "—" if s.precision is None else f"{s.precision:.2f}" + r = "—" if s.recall is None else f"{s.recall:.2f}" + lines.append( + f" {s.slug:<32} support {s.support:>2} P {p:>4} R {r:>4} fp {s.fp}" + ) + lines.append("gate: PASS" if not reasons else "gate: FAIL — " + "; ".join(reasons)) + return "\n".join(lines) diff --git a/src/llm/vocab/themes.yaml b/src/llm/vocab/themes.yaml index 759c7b3..11198ca 100644 --- a/src/llm/vocab/themes.yaml +++ b/src/llm/vocab/themes.yaml @@ -6,7 +6,7 @@ # retire a theme by moving it to `retired:` rather than deleting it, and # bump `version` whenever a slug, definition or synonym list changes, since # every tag run records the version it used. -version: 1 +version: 2 themes: - slug: conversion-factor label: Conversion factor and payment update @@ -273,4 +273,24 @@ themes: definition: Creation, deletion or crosswalk of CPT/HCPCS codes, code descriptors, or bundling of services into other codes. synonyms: [new code, deleted code, HCPCS code, G code, crosswalk, bundled, code descriptor, CPT Editorial Panel, unbundle] sections: [Proposed Valuation of Specific Codes] + - slug: modifier-25-same-day + label: Same-day procedures with an E/M visit (modifier 25) + definition: Payment reduction or policy for a procedure furnished on the same day as an E/M visit billed with modifier 25, or multiple-procedure payment reductions. + synonyms: [modifier 25, modifier -25, same-day procedure, same day visit, 50 percent reduction, multiple procedure payment reduction, come back for a second visit] + sections: [Payment for Procedures Furnished With an E/M Visit] + - slug: lactation-services + label: Lactation care services + definition: New lactation care service codes, who may furnish them, and their payment or supervision conditions. + synonyms: [lactation, lactation consultant, IBCLC, breastfeeding, 978XX, lactation care services] + sections: [Lactation Care Services] + - slug: health-coaching + label: Health and wellness coaching + definition: Health and wellness coaching services (CPT 0591T–0593T), their national payment and conditions of payment. + synonyms: [health coaching, wellness coaching, health and well-being coaching, 0591T, 0592T, 0593T, nurse coach] + sections: [Health Coaching] + - slug: radiation-therapy + label: Radiation therapy and proton treatment + definition: Payment or coverage for radiation oncology, proton beam therapy, or radiation treatment delivery codes. + synonyms: [radiation therapy, radiation oncology, proton therapy, proton beam, treatment delivery, IMRT] + sections: [Radiation Therapy Services] retired: [] diff --git a/tests/cli/test_llm_tag_cli.py b/tests/cli/test_llm_tag_cli.py index 14193ce..202b357 100644 --- a/tests/cli/test_llm_tag_cli.py +++ b/tests/cli/test_llm_tag_cli.py @@ -101,3 +101,45 @@ class TestVocab: p.write_text("version: 1\nthemes: []\n") res = runner.invoke(app, ["llm", "vocab", "--path", str(p)]) assert res.exit_code == 1 and "invalid vocabulary" in res.output + + +class TestEvalTags: + def _golden(self, tmp_path, ids): + p = tmp_path / "g.yaml" + p.write_text( + "vocab_version: 1\ndocket: D\nentries:\n" + + "".join( + f" - {{comment_id: {cid}, expected: [telehealth]}}\n" for cid in ids + ) + ) + return p + + def test_stored_scoring_and_gate(self, rt, tmp_path, monkeypatch): + store, key, _ = rt + monkeypatch.setattr("conf.connect.bib", lambda: store) + g = self._golden(tmp_path, ["CMS-2026-2377-1", "CMS-2026-2377-404"]) + # nothing stored yet → nothing scored → gate fails + res = runner.invoke(app, ["llm", "eval-tags", "--golden", str(g)]) + assert ( + res.exit_code == 1 + and "gate: FAIL" in res.output + and "warning" in res.output + ) + runner.invoke(app, ["llm", "tag", "--docket", "CMS-2026-2377"]) + res = runner.invoke(app, ["llm", "eval-tags", "--golden", str(g)]) + assert res.exit_code == 0, res.output + assert ( + "1 scored, 1 missing" in res.output + and "F1 1.00" in res.output + and "gate: PASS" in res.output + ) + + def test_live_scoring_uses_the_runtime(self, rt, tmp_path): + store, key, judged = rt + g = self._golden(tmp_path, ["CMS-2026-2377-1"]) + res = runner.invoke(app, ["llm", "eval-tags", "--golden", str(g), "--live"]) + assert res.exit_code == 0, res.output + assert f"{key} telehealth" in res.output and "gate: PASS" in res.output + assert judged and not any( + t.startswith("llm:") for t in store.get(key).tags + ) # live never writes diff --git a/tests/llm/golden_themes.yaml b/tests/llm/golden_themes.yaml new file mode 100644 index 0000000..13726bf --- /dev/null +++ b/tests/llm/golden_themes.yaml @@ -0,0 +1,48 @@ +# Golden set for the theme tagger (P35, #577). Each entry is one +# regulations.gov comment with the themes it must carry (`expected`) and +# themes that are acceptable but not required (`acceptable`). A predicted +# theme outside both counts as a false positive; a missing `expected` +# theme is a false negative. Labelled by reading the comment's opening +# text on 2026-09-22 (CMS-2026-2377, the CY2027 PFS NPRM docket). +vocab_version: 2 +docket: CMS-2026-2377 +entries: + - {comment_id: CMS-2026-2377-40314, expected: [practice-expense, therapy-services], acceptable: [conversion-factor]} + - {comment_id: CMS-2026-2377-40318, expected: [modifier-25-same-day], acceptable: [evaluation-management, beneficiary-cost-sharing]} + - {comment_id: CMS-2026-2377-40322, expected: [conversion-factor], acceptable: [administrative-burden, regulatory-impact, modifier-25-same-day, work-rvu]} + - {comment_id: CMS-2026-2377-40323, expected: [conversion-factor], acceptable: [administrative-burden, regulatory-impact, modifier-25-same-day, work-rvu]} + - {comment_id: CMS-2026-2377-40326, expected: [modifier-25-same-day], acceptable: [evaluation-management, rural-access]} + - {comment_id: CMS-2026-2377-40327, expected: [modifier-25-same-day], acceptable: [evaluation-management, rural-access]} + - {comment_id: CMS-2026-2377-40329, expected: [behavioral-health], acceptable: [vaccines-preventive, opioid-treatment, care-management]} + - {comment_id: CMS-2026-2377-40330, expected: [modifier-25-same-day], acceptable: [evaluation-management, rural-access]} + - {comment_id: CMS-2026-2377-40336, expected: [misvalued-codes], acceptable: [work-rvu, laboratory, coding-descriptors]} + - {comment_id: CMS-2026-2377-40337, expected: [behavioral-health], acceptable: [vaccines-preventive, opioid-treatment, care-management]} + - {comment_id: CMS-2026-2377-40339, expected: [modifier-25-same-day], acceptable: [evaluation-management, skin-substitutes, rural-access]} + - {comment_id: CMS-2026-2377-40343, expected: [conversion-factor, therapy-services], acceptable: [practice-expense, rural-access]} + - {comment_id: CMS-2026-2377-40345, expected: [lactation-services], acceptable: [non-physician-practitioners, supervision]} + - {comment_id: CMS-2026-2377-40347, expected: [modifier-25-same-day], acceptable: [beneficiary-cost-sharing, rural-access, evaluation-management]} + - {comment_id: CMS-2026-2377-40348, expected: [therapy-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40349, expected: [work-rvu], acceptable: [coding-descriptors, misvalued-codes, practice-expense]} + - {comment_id: CMS-2026-2377-40350, expected: [behavioral-health], acceptable: [vaccines-preventive, opioid-treatment, care-management]} + - {comment_id: CMS-2026-2377-40351, expected: [health-coaching], acceptable: [supervision, non-physician-practitioners, coding-descriptors]} + - {comment_id: CMS-2026-2377-40352, expected: [health-coaching], acceptable: [supervision, non-physician-practitioners, coding-descriptors]} + - {comment_id: CMS-2026-2377-40357, expected: [therapy-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40360, expected: [modifier-25-same-day], acceptable: [work-rvu, conversion-factor, rural-access, beneficiary-cost-sharing]} + - {comment_id: CMS-2026-2377-40361, expected: [lactation-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40362, expected: [skin-substitutes], acceptable: [coverage-policy, rural-access]} + - {comment_id: CMS-2026-2377-40367, expected: [lactation-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40368, expected: [rural-access], acceptable: [modifier-25-same-day, beneficiary-cost-sharing, telehealth]} + - {comment_id: CMS-2026-2377-40375, expected: [radiation-therapy], acceptable: [coverage-policy, work-rvu, rural-access]} + - {comment_id: CMS-2026-2377-40376, expected: [conversion-factor], acceptable: [modifier-25-same-day, work-rvu, practice-expense, efficiency-adjustment, administrative-burden]} + - {comment_id: CMS-2026-2377-40378, expected: [therapy-services], acceptable: [rural-access, non-physician-practitioners, coding-descriptors, care-management]} + - {comment_id: CMS-2026-2377-40385, expected: [therapy-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40387, expected: [misvalued-codes], acceptable: [work-rvu, laboratory, conversion-factor, coding-descriptors]} + - {comment_id: CMS-2026-2377-40388, expected: [conversion-factor], acceptable: [work-rvu, modifier-25-same-day, efficiency-adjustment, global-surgery, practice-expense, rural-access]} + - {comment_id: CMS-2026-2377-40390, expected: [therapy-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors, telehealth]} + - {comment_id: CMS-2026-2377-40391, expected: [therapy-services], acceptable: [non-physician-practitioners, supervision, coding-descriptors]} + - {comment_id: CMS-2026-2377-40392, expected: [skin-substitutes], acceptable: [coverage-policy, dme-supplies]} + - {comment_id: CMS-2026-2377-40393, expected: [visit-complexity-add-on], acceptable: [evaluation-management, work-rvu, conversion-factor]} + - {comment_id: CMS-2026-2377-40394, expected: [work-rvu], acceptable: [coding-descriptors, practice-expense, misvalued-codes]} + - {comment_id: CMS-2026-2377-40395, expected: [remote-monitoring], acceptable: [enrollment-program-integrity, supervision, care-management, telehealth]} + - {comment_id: CMS-2026-2377-40400, expected: [modifier-25-same-day], acceptable: [evaluation-management, beneficiary-cost-sharing]} + - {comment_id: CMS-2026-2377-40401, expected: [conversion-factor], acceptable: [work-rvu, modifier-25-same-day, efficiency-adjustment, global-surgery, practice-expense, rural-access]} diff --git a/tests/llm/test_tageval.py b/tests/llm/test_tageval.py new file mode 100644 index 0000000..ff0adc6 --- /dev/null +++ b/tests/llm/test_tageval.py @@ -0,0 +1,150 @@ +"""llm.tageval — golden-set scoring and the fan-out gate (#577).""" + +from __future__ import annotations + +from pathlib import Path + +from bib.item import Source +from bib.store import Store +from llm.tageval import ( + Golden, + GoldenEntry, + Report, + format_report, + gate, + live_predictions, + load_golden, + score, + stored_predictions, +) +from llm.vocab import load as load_vocab + +GOLDEN = Path(__file__).with_name("golden_themes.yaml") + + +class TestGoldenFile: + def test_loads_and_only_uses_known_slugs(self): + g = load_golden(GOLDEN) + v = load_vocab() + assert ( + isinstance(g, Golden) + and g.vocab_version == v.version + and g.docket == "CMS-2026-2377" + ) + assert len(g.entries) >= 30 + for e in g.entries: + assert e.expected, e.comment_id + for s in (*e.expected, *e.acceptable): + assert s in v, f"{e.comment_id}: unknown slug {s}" + assert not set(e.expected) & set(e.acceptable), e.comment_id + assert len({e.comment_id for e in g.entries}) == len(g.entries) + + +def _golden(): + return Golden( + 2, + "D", + ( + GoldenEntry("c1", ("telehealth",), ("care-management",)), + GoldenEntry("c2", ("drugs", "telehealth")), + GoldenEntry("c3", ("drugs",)), + GoldenEntry("c4", ("drugs",)), + ), + ) + + +class TestScore: + def test_counts(self): + r = score( + _golden(), + { + "c1": [ + "telehealth", + "care-management", + "gpci-localities", + ], # tp, acceptable, fp + "c2": ["drugs"], # tp + fn(telehealth) + "c3": [], # abstain → fn + "c4": None, # missing + }, + ) + assert isinstance(r, Report) + assert (r.n, r.scored, r.missing, r.abstained) == (4, 3, ("c4",), 1) + assert (r.tp, r.fp, r.fn) == (2, 1, 2) + assert r.per_slug["telehealth"].tp == 1 and r.per_slug["telehealth"].fn == 1 + assert ( + r.per_slug["gpci-localities"].fp == 1 + and r.per_slug["gpci-localities"].precision == 0.0 + ) + assert r.per_slug["drugs"].recall == 0.5 + assert r.precision == 2 / 3 and r.recall == 0.5 and 0.57 < r.f1 < 0.58 + assert r.abstain_rate == 1 / 3 + + def test_gate(self): + good = score( + _golden(), + { + "c1": ["telehealth"], + "c2": ["drugs", "telehealth"], + "c3": ["drugs"], + "c4": ["drugs"], + }, + ) + assert gate(good) == [] and good.f1 == 1.0 + bad = score(_golden(), {"c1": [], "c2": [], "c3": [], "c4": []}) + assert any("micro-F1" in x for x in gate(bad)) + assert gate(score(_golden(), {})) == ["nothing scored"] + # a weak theme with enough support fails on its own + g = Golden(2, "D", tuple(GoldenEntry(f"c{i}", ("drugs",)) for i in range(6))) + weak = score( + g, + {f"c{i}": (["drugs"] if i < 5 else []) for i in range(6)} + | {"c5": ["telehealth"]}, + ) + # 5 tp / 1 fn for drugs, F1 fine; telehealth 1 fp support 0 → no theme failure + assert gate(weak) == [] + low = Golden( + 2, "D", tuple(GoldenEntry(f"c{i}", ("drugs",), ()) for i in range(6)) + ) + preds = {f"c{i}": ["drugs"] for i in range(6)} + preds.update({f"x{i}": ["drugs"] for i in range(6)}) # extra keys ignored + r = score(low, preds) + assert gate(r) == [] + + def test_format(self): + r = score( + _golden(), {"c1": ["telehealth"], "c2": ["drugs"], "c3": [], "c4": None} + ) + text = format_report(r, reasons=gate(r)) + assert "4 entries, 3 scored, 1 missing (c4)" in text and "gate:" in text + assert "drugs" in text and "telehealth" in text + + +class TestPredictionSources: + def test_stored_and_live(self, tmp_path): + s = Store(":memory:", storage_dir=tmp_path / "st") + k1 = s.upsert(Source(title="a", url="https://www.regulations.gov/comment/D-1")) + s.upsert(Source(title="b", url="https://www.regulations.gov/comment/D-2")) + s.merge_extra( + k1, + { + "llm_tags": { + "tags": { + "drugs": {"confidence": 0.9, "evidence_chunk": 0}, + "telehealth": {"confidence": 0.8, "evidence_chunk": 1}, + } + } + }, + ) + assert stored_predictions(s, ["D-1", "D-2", "D-9"]) == { + "D-1": ["drugs", "telehealth"], + "D-2": None, + "D-9": None, + } + live = live_predictions( + s, + ["D-1", "D-2", "D-9"], + tag=lambda key: ["care-management"] if key == k1 else [], + ) + assert live == {"D-1": ["care-management"], "D-2": [], "D-9": None} + s.close()