feat(llm): tagger defaults from the golden sweep — 3 tags, min confidence 0.60 (refs #577)
Offline sweep over the live golden dump (39 CMS-2026-2377 comments): with the judge's verdicts fixed, max_tags 4 / no threshold scores micro-F1 0.67 (P 0.60, R 0.76); 3 tags at 0.60 scores 0.73 (P 0.70, R 0.76) and passes the gate; 2 tags at 0.60 scores 0.79 but the golden set skews single-theme. Recall is capped by the judge (11 misses at any threshold). Tuned on the same 39 comments it is scored on — a held-out set from the pilot docket is the next check.
This commit is contained in:
@@ -23,8 +23,11 @@ Usage: stack llm eval-tags [OPTIONS]
|
||||
│ --limit INTEGER Score only the first N golden entries. │
|
||||
│ [default: 0] │
|
||||
│ --top INTEGER [default: 8] │
|
||||
│ --max-tags INTEGER [default: 4] │
|
||||
│ --min-confidence FLOAT [default: 0.0] │
|
||||
│ --max-tags INTEGER [default: 3] │
|
||||
│ --min-confidence FLOAT [default: 0.6] │
|
||||
│ --dump TEXT With --live: write every comment's │
|
||||
│ shortlist, verdicts and confidences to this │
|
||||
│ JSON file. │
|
||||
│ --help Show this message and exit. │
|
||||
╰──────────────────────────────────────────────────────────────────────────────╯
|
||||
```
|
||||
|
||||
@@ -359,8 +359,8 @@ def eval_tags(
|
||||
0, "--limit", help="Score only the first N golden entries."
|
||||
),
|
||||
top: int = typer.Option(8, "--top"),
|
||||
max_tags: int = typer.Option(4, "--max-tags"),
|
||||
min_confidence: float = typer.Option(0.0, "--min-confidence"),
|
||||
max_tags: int = typer.Option(3, "--max-tags"),
|
||||
min_confidence: float = typer.Option(0.60, "--min-confidence"),
|
||||
dump: str = typer.Option(
|
||||
"",
|
||||
"--dump",
|
||||
|
||||
@@ -5,8 +5,9 @@ For one comment the chain is: its already-indexed chunks (text + vector,
|
||||
against every theme card (``llm.vocab.Theme.card`` embedded once per run)
|
||||
→ the ``top`` themes shortlisted, each with the chunk that scored it →
|
||||
one yes/no judgement per candidate on the largest live host, with that
|
||||
chunk as the evidence → up to ``max_tags`` accepted themes, confidence =
|
||||
the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``.
|
||||
chunk as the evidence → up to ``max_tags`` accepted themes whose shortlist similarity clears
|
||||
``min_confidence`` (defaults 3 and 0.60 — the settings that pass the
|
||||
golden gate, #577), confidence = that similarity. Nothing outside ``{yes, no}`` counts as ``no``.
|
||||
|
||||
State lives on the bib item (``extra_json["llm_tags"]``: vocab version,
|
||||
model, content hash, the accepted tags with confidence and evidence
|
||||
@@ -148,8 +149,8 @@ def tag_comment(
|
||||
card_vecs: Mapping[str, Vector],
|
||||
judge: Judge,
|
||||
top: int = 8,
|
||||
max_tags: int = 4,
|
||||
min_confidence: float = 0.0,
|
||||
max_tags: int = 3,
|
||||
min_confidence: float = 0.60,
|
||||
) -> TagResult:
|
||||
cands = shortlist(chunk_vecs, card_vecs, top=top)
|
||||
judged: dict[str, bool | None] = {}
|
||||
@@ -252,8 +253,8 @@ def run(
|
||||
force: bool = False,
|
||||
dry_run: bool = False,
|
||||
top: int = 8,
|
||||
max_tags: int = 4,
|
||||
min_confidence: float = 0.0,
|
||||
max_tags: int = 3,
|
||||
min_confidence: float = 0.60,
|
||||
progress: Callable[[str, TagResult | None, str], None] | None = None,
|
||||
) -> dict[str, int]:
|
||||
"""Tag every comment in *docket* (newest first); resumable via the
|
||||
|
||||
Reference in New Issue
Block a user