From c5e060affd052a1d5b86da1dfdd74019ef605a08 Mon Sep 17 00:00:00 2001 From: kert Date: Tue, 22 Sep 2026 17:06:38 -0400 Subject: [PATCH] =?UTF-8?q?feat(llm):=20tagger=20defaults=20from=20the=20g?= =?UTF-8?q?olden=20sweep=20=E2=80=94=203=20tags,=20min=20confidence=200.60?= =?UTF-8?q?=20(refs=20#577)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Offline sweep over the live golden dump (39 CMS-2026-2377 comments): with the judge's verdicts fixed, max_tags 4 / no threshold scores micro-F1 0.67 (P 0.60, R 0.76); 3 tags at 0.60 scores 0.73 (P 0.70, R 0.76) and passes the gate; 2 tags at 0.60 scores 0.79 but the golden set skews single-theme. Recall is capped by the judge (11 misses at any threshold). Tuned on the same 39 comments it is scored on — a held-out set from the pilot docket is the next check. --- docs/docs/cli/llm-eval-tags.md | 7 +++++-- src/cli/llm.py | 4 ++-- src/llm/tagger.py | 13 +++++++------ 3 files changed, 14 insertions(+), 10 deletions(-) diff --git a/docs/docs/cli/llm-eval-tags.md b/docs/docs/cli/llm-eval-tags.md index dd48296..9192922 100644 --- a/docs/docs/cli/llm-eval-tags.md +++ b/docs/docs/cli/llm-eval-tags.md @@ -23,8 +23,11 @@ Usage: stack llm eval-tags [OPTIONS] │ --limit INTEGER Score only the first N golden entries. │ │ [default: 0] │ │ --top INTEGER [default: 8] │ -│ --max-tags INTEGER [default: 4] │ -│ --min-confidence FLOAT [default: 0.0] │ +│ --max-tags INTEGER [default: 3] │ +│ --min-confidence FLOAT [default: 0.6] │ +│ --dump TEXT With --live: write every comment's │ +│ shortlist, verdicts and confidences to this │ +│ JSON file. │ │ --help Show this message and exit. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/src/cli/llm.py b/src/cli/llm.py index fd5ebfa..eea5ccc 100644 --- a/src/cli/llm.py +++ b/src/cli/llm.py @@ -359,8 +359,8 @@ def eval_tags( 0, "--limit", help="Score only the first N golden entries." ), top: int = typer.Option(8, "--top"), - max_tags: int = typer.Option(4, "--max-tags"), - min_confidence: float = typer.Option(0.0, "--min-confidence"), + max_tags: int = typer.Option(3, "--max-tags"), + min_confidence: float = typer.Option(0.60, "--min-confidence"), dump: str = typer.Option( "", "--dump", diff --git a/src/llm/tagger.py b/src/llm/tagger.py index 1bfa3bf..f19bc96 100644 --- a/src/llm/tagger.py +++ b/src/llm/tagger.py @@ -5,8 +5,9 @@ For one comment the chain is: its already-indexed chunks (text + vector, against every theme card (``llm.vocab.Theme.card`` embedded once per run) → the ``top`` themes shortlisted, each with the chunk that scored it → one yes/no judgement per candidate on the largest live host, with that -chunk as the evidence → up to ``max_tags`` accepted themes, confidence = -the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``. +chunk as the evidence → up to ``max_tags`` accepted themes whose shortlist similarity clears +``min_confidence`` (defaults 3 and 0.60 — the settings that pass the +golden gate, #577), confidence = that similarity. Nothing outside ``{yes, no}`` counts as ``no``. State lives on the bib item (``extra_json["llm_tags"]``: vocab version, model, content hash, the accepted tags with confidence and evidence @@ -148,8 +149,8 @@ def tag_comment( card_vecs: Mapping[str, Vector], judge: Judge, top: int = 8, - max_tags: int = 4, - min_confidence: float = 0.0, + max_tags: int = 3, + min_confidence: float = 0.60, ) -> TagResult: cands = shortlist(chunk_vecs, card_vecs, top=top) judged: dict[str, bool | None] = {} @@ -252,8 +253,8 @@ def run( force: bool = False, dry_run: bool = False, top: int = 8, - max_tags: int = 4, - min_confidence: float = 0.0, + max_tags: int = 3, + min_confidence: float = 0.60, progress: Callable[[str, TagResult | None, str], None] | None = None, ) -> dict[str, int]: """Tag every comment in *docket* (newest first); resumable via the