diff --git a/docs/docs/cli/llm-eval-tags.md b/docs/docs/cli/llm-eval-tags.md index dd48296..9192922 100644 --- a/docs/docs/cli/llm-eval-tags.md +++ b/docs/docs/cli/llm-eval-tags.md @@ -23,8 +23,11 @@ Usage: stack llm eval-tags [OPTIONS] │ --limit INTEGER Score only the first N golden entries. │ │ [default: 0] │ │ --top INTEGER [default: 8] │ -│ --max-tags INTEGER [default: 4] │ -│ --min-confidence FLOAT [default: 0.0] │ +│ --max-tags INTEGER [default: 3] │ +│ --min-confidence FLOAT [default: 0.6] │ +│ --dump TEXT With --live: write every comment's │ +│ shortlist, verdicts and confidences to this │ +│ JSON file. │ │ --help Show this message and exit. │ ╰──────────────────────────────────────────────────────────────────────────────╯ ``` diff --git a/src/cli/llm.py b/src/cli/llm.py index fd5ebfa..eea5ccc 100644 --- a/src/cli/llm.py +++ b/src/cli/llm.py @@ -359,8 +359,8 @@ def eval_tags( 0, "--limit", help="Score only the first N golden entries." ), top: int = typer.Option(8, "--top"), - max_tags: int = typer.Option(4, "--max-tags"), - min_confidence: float = typer.Option(0.0, "--min-confidence"), + max_tags: int = typer.Option(3, "--max-tags"), + min_confidence: float = typer.Option(0.60, "--min-confidence"), dump: str = typer.Option( "", "--dump", diff --git a/src/llm/tagger.py b/src/llm/tagger.py index 1bfa3bf..f19bc96 100644 --- a/src/llm/tagger.py +++ b/src/llm/tagger.py @@ -5,8 +5,9 @@ For one comment the chain is: its already-indexed chunks (text + vector, against every theme card (``llm.vocab.Theme.card`` embedded once per run) → the ``top`` themes shortlisted, each with the chunk that scored it → one yes/no judgement per candidate on the largest live host, with that -chunk as the evidence → up to ``max_tags`` accepted themes, confidence = -the shortlist similarity. Nothing outside ``{yes, no}`` counts as ``no``. +chunk as the evidence → up to ``max_tags`` accepted themes whose shortlist similarity clears +``min_confidence`` (defaults 3 and 0.60 — the settings that pass the +golden gate, #577), confidence = that similarity. Nothing outside ``{yes, no}`` counts as ``no``. State lives on the bib item (``extra_json["llm_tags"]``: vocab version, model, content hash, the accepted tags with confidence and evidence @@ -148,8 +149,8 @@ def tag_comment( card_vecs: Mapping[str, Vector], judge: Judge, top: int = 8, - max_tags: int = 4, - min_confidence: float = 0.0, + max_tags: int = 3, + min_confidence: float = 0.60, ) -> TagResult: cands = shortlist(chunk_vecs, card_vecs, top=top) judged: dict[str, bool | None] = {} @@ -252,8 +253,8 @@ def run( force: bool = False, dry_run: bool = False, top: int = 8, - max_tags: int = 4, - min_confidence: float = 0.0, + max_tags: int = 3, + min_confidence: float = 0.60, progress: Callable[[str, TagResult | None, str], None] | None = None, ) -> dict[str, int]: """Tag every comment in *docket* (newest first); resumable via the