feat(llm): tagger defaults from the golden sweep — 3 tags, min confidence 0.60 (refs #577)

Offline sweep over the live golden dump (39 CMS-2026-2377 comments):
with the judge's verdicts fixed, max_tags 4 / no threshold scores
micro-F1 0.67 (P 0.60, R 0.76); 3 tags at 0.60 scores 0.73 (P 0.70,
R 0.76) and passes the gate; 2 tags at 0.60 scores 0.79 but the golden
set skews single-theme. Recall is capped by the judge (11 misses at any
threshold). Tuned on the same 39 comments it is scored on — a held-out
set from the pilot docket is the next check.
This commit is contained in:
kert
2026-09-22 17:06:38 -04:00
parent 7000d2779f
commit c5e060affd
3 changed files with 14 additions and 10 deletions

View File

@@ -23,8 +23,11 @@ Usage: stack llm eval-tags [OPTIONS]
│ --limit INTEGER Score only the first N golden entries. │
│ [default: 0] │
│ --top INTEGER [default: 8] │
│ --max-tags INTEGER [default: 4] │
│ --min-confidence FLOAT [default: 0.0] │
│ --max-tags INTEGER [default: 3] │
│ --min-confidence FLOAT [default: 0.6] │
│ --dump TEXT With --live: write every comment's │
│ shortlist, verdicts and confidences to this │
│ JSON file. │
│ --help Show this message and exit. │
╰──────────────────────────────────────────────────────────────────────────────╯
```

View File

@@ -359,8 +359,8 @@ def eval_tags(
0, "--limit", help="Score only the first N golden entries."
),
top: int = typer.Option(8, "--top"),
max_tags: int = typer.Option(4, "--max-tags"),
min_confidence: float = typer.Option(0.0, "--min-confidence"),
max_tags: int = typer.Option(3, "--max-tags"),
min_confidence: float = typer.Option(0.60, "--min-confidence"),
dump: str = typer.Option(
"",
"--dump",

View File

@@ -5,8 +5,9 @@ For one comment the chain is: its already-indexed chunks (text + vector,
against every theme card (``llm.vocab.Theme.card`` embedded once per run)
→ the ``top`` themes shortlisted, each with the chunk that scored it →
one yes/no judgement per candidate on the largest live host, with that
chunk as the evidence → up to ``max_tags`` accepted themes, confidence =
the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``.
chunk as the evidence → up to ``max_tags`` accepted themes whose shortlist similarity clears
``min_confidence`` (defaults 3 and 0.60 — the settings that pass the
golden gate, #577), confidence = that similarity. Nothing outside ``{yes, no}`` counts as ``no``.
State lives on the bib item (``extra_json["llm_tags"]``: vocab version,
model, content hash, the accepted tags with confidence and evidence
@@ -148,8 +149,8 @@ def tag_comment(
card_vecs: Mapping[str, Vector],
judge: Judge,
top: int = 8,
max_tags: int = 4,
min_confidence: float = 0.0,
max_tags: int = 3,
min_confidence: float = 0.60,
) -> TagResult:
cands = shortlist(chunk_vecs, card_vecs, top=top)
judged: dict[str, bool | None] = {}
@@ -252,8 +253,8 @@ def run(
force: bool = False,
dry_run: bool = False,
top: int = 8,
max_tags: int = 4,
min_confidence: float = 0.0,
max_tags: int = 3,
min_confidence: float = 0.60,
progress: Callable[[str, TagResult | None, str], None] | None = None,
) -> dict[str, int]:
"""Tag every comment in *docket* (newest first); resumable via the