feat(llm): golden longitudinal evaluation — yaml set, runner with anchor/label/forbidden checks, nightly workflow + issue filer (refs #692)
This commit is contained in:
55
.gitea/workflows/llm-golden.yml
Normal file
55
.gitea/workflows/llm-golden.yml
Normal file
@@ -0,0 +1,55 @@
|
|||||||
|
# DO NOT EDIT — generated by gen_config.py from stack.toml
|
||||||
|
# Re-generate: uv run python dev/scripts/gen_config.py
|
||||||
|
|
||||||
|
name: LLM Golden
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
schedule:
|
||||||
|
- cron: "40 3 * * *"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
llm-golden:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: https://github.com/actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up uv
|
||||||
|
run: curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
env:
|
||||||
|
UV_INSTALL_DIR: /usr/local/bin
|
||||||
|
|
||||||
|
- name: Run golden longitudinal evaluation against the live chat
|
||||||
|
env:
|
||||||
|
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
docker exec llm mkdir -p /tmp/golden
|
||||||
|
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
|
||||||
|
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
|
||||||
|
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
|
||||||
|
docker exec \
|
||||||
|
-e GITEA_TOKEN \
|
||||||
|
-e GITEA_API_BASE=http://git:3000/api/v1 \
|
||||||
|
-e NB_ISSUE_LABEL=llm \
|
||||||
|
llm \
|
||||||
|
uv run --project /app python /tmp/golden/llm_golden.py run \
|
||||||
|
--url http://localhost:8000 \
|
||||||
|
--set /tmp/golden/golden_lineage.yaml \
|
||||||
|
--report /tmp/golden/report.json \
|
||||||
|
--file-issues --source nightly-llm-golden
|
||||||
|
docker cp llm:/tmp/golden/report.json llm-golden-report.json
|
||||||
|
cat llm-golden-report.json
|
||||||
|
|
||||||
|
- name: File failure issue
|
||||||
|
if: failure()
|
||||||
|
env:
|
||||||
|
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
||||||
|
run: |
|
||||||
|
uv sync --no-dev --quiet 2>/dev/null || true
|
||||||
|
uv run python -m api.diag.ci \
|
||||||
|
--workflow "LLM Golden" --job "llm-golden" \
|
||||||
|
--run "${{ github.run_number }}" \
|
||||||
|
--sha "${{ github.sha }}" \
|
||||||
|
--ref "${{ github.ref }}" || true
|
||||||
@@ -643,6 +643,62 @@ jobs:
|
|||||||
return (".gitea/workflows/zotero-sync.yml", content)
|
return (".gitea/workflows/zotero-sync.yml", content)
|
||||||
|
|
||||||
|
|
||||||
|
def _gen_llm_golden(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]:
|
||||||
|
"""Nightly golden longitudinal evaluation of the chat (P49 Task 7).
|
||||||
|
|
||||||
|
Mirrors ``_gen_notebooks_integration``: the golden set needs a live
|
||||||
|
chat over the real pgvector/DuckDB/Ollama stack, which only the
|
||||||
|
``llm`` container (``uvicorn llm.api:app``, port 8000, no published
|
||||||
|
port) has — so the runner and the golden set are docker-cp'd in and
|
||||||
|
run there against its own ``http://localhost:8000``. Regressions
|
||||||
|
are filed through ``nb_issue_filer`` (dedup + auto-close) under the
|
||||||
|
``llm`` label, not ``notebooks``. 03:40 sits after the 03:30
|
||||||
|
notebooks-integration run and before the 04:15 zotero-sync run.
|
||||||
|
"""
|
||||||
|
content = f"""\
|
||||||
|
{_HEADER}
|
||||||
|
name: LLM Golden
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
schedule:
|
||||||
|
- cron: "40 3 * * *"
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
llm-golden:
|
||||||
|
runs-on: {runner}
|
||||||
|
steps:
|
||||||
|
{_checkout_step()}
|
||||||
|
|
||||||
|
{_setup_uv_step(uv_version)}
|
||||||
|
|
||||||
|
- name: Run golden longitudinal evaluation against the live chat
|
||||||
|
env:
|
||||||
|
GITEA_TOKEN: ${{{{ secrets.DEPLOY_TOKEN }}}}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
docker exec llm mkdir -p /tmp/golden
|
||||||
|
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
|
||||||
|
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
|
||||||
|
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
|
||||||
|
docker exec \\
|
||||||
|
-e GITEA_TOKEN \\
|
||||||
|
-e GITEA_API_BASE=http://git:3000/api/v1 \\
|
||||||
|
-e NB_ISSUE_LABEL=llm \\
|
||||||
|
llm \\
|
||||||
|
uv run --project /app python /tmp/golden/llm_golden.py run \\
|
||||||
|
--url http://localhost:8000 \\
|
||||||
|
--set /tmp/golden/golden_lineage.yaml \\
|
||||||
|
--report /tmp/golden/report.json \\
|
||||||
|
--file-issues --source nightly-llm-golden
|
||||||
|
docker cp llm:/tmp/golden/report.json llm-golden-report.json
|
||||||
|
cat llm-golden-report.json
|
||||||
|
|
||||||
|
{_failure_step("LLM Golden", "llm-golden")}
|
||||||
|
"""
|
||||||
|
return (".gitea/workflows/llm-golden.yml", content)
|
||||||
|
|
||||||
|
|
||||||
def _gen_release(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]:
|
def _gen_release(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]:
|
||||||
content = f"""\
|
content = f"""\
|
||||||
{_HEADER}
|
{_HEADER}
|
||||||
@@ -710,6 +766,7 @@ def emit(
|
|||||||
_gen_infra_ci,
|
_gen_infra_ci,
|
||||||
_gen_notebooks_integration,
|
_gen_notebooks_integration,
|
||||||
_gen_zotero_sync,
|
_gen_zotero_sync,
|
||||||
|
_gen_llm_golden,
|
||||||
_gen_release,
|
_gen_release,
|
||||||
):
|
):
|
||||||
path, content = gen_fn(**common) # type: ignore[arg-type]
|
path, content = gen_fn(**common) # type: ignore[arg-type]
|
||||||
|
|||||||
389
dev/scripts/llm_golden.py
Normal file
389
dev/scripts/llm_golden.py
Normal file
@@ -0,0 +1,389 @@
|
|||||||
|
"""Golden longitudinal evaluation for the chat (P49 Task 7).
|
||||||
|
|
||||||
|
Streams the live ``/chat`` SSE endpoint for a fixed set of longitudinal
|
||||||
|
questions (``tests/llm/golden_lineage.yaml``) and checks each answer
|
||||||
|
against verified expectations: FR paragraph anchors that must appear in
|
||||||
|
``sources``, citation labels that must appear in the answer prose,
|
||||||
|
forbidden patterns (e.g. a bare dollar amount not backed by a valuation
|
||||||
|
row), lineage timeline events, comment dockets, and the number of
|
||||||
|
distinct rule eras the sources span. Regressions are filed/swept through
|
||||||
|
``dev/scripts/nb_issue_filer.py`` (copied beside this script in CI, and
|
||||||
|
imported by path — never re-implemented here).
|
||||||
|
|
||||||
|
Stdlib + httpx + PyYAML only — this script is docker-cp'd into the
|
||||||
|
``llm`` container and run there against its own ``http://localhost:8000``
|
||||||
|
(see ``dev/scripts/backends/gitea.py::_gen_llm_golden``); it must not
|
||||||
|
import ``llm.*``/``pfs.*``.
|
||||||
|
|
||||||
|
Usage::
|
||||||
|
|
||||||
|
uv run python dev/scripts/llm_golden.py run \\
|
||||||
|
--url http://localhost:8000 --set tests/llm/golden_lineage.yaml \\
|
||||||
|
--report report.json [--file-issues --source nightly-llm-golden] \\
|
||||||
|
[--only g2058-replacement] [--timeout 180]
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import dataclasses
|
||||||
|
import importlib.util
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
_HERE = Path(__file__).resolve().parent
|
||||||
|
|
||||||
|
_EXPECTATION_KEYS = (
|
||||||
|
"expect_anchors",
|
||||||
|
"expect_labels",
|
||||||
|
"forbid",
|
||||||
|
"expect_events",
|
||||||
|
"expect_dockets",
|
||||||
|
"min_eras",
|
||||||
|
)
|
||||||
|
|
||||||
|
_SENT_SPLIT = re.compile(r"(?<=[.!?])\s+")
|
||||||
|
_BRACKET_RE = re.compile(r"\[([^\]]+)\]")
|
||||||
|
_DOCKET_YEAR_RE = re.compile(r"(19|20)\d{2}")
|
||||||
|
_DATE_YEAR_RE = re.compile(r"^(\d{4})")
|
||||||
|
|
||||||
|
|
||||||
|
# ── golden set loading ───────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def load_set(path: Path) -> list[dict]:
|
||||||
|
"""Parse and validate the golden YAML set: a top-level list (or
|
||||||
|
``{questions: [...]}``) of entries, each with a unique ``id``, a
|
||||||
|
``question``, and at least one expectation key."""
|
||||||
|
raw = yaml.safe_load(path.read_text())
|
||||||
|
if isinstance(raw, dict) and "questions" in raw:
|
||||||
|
entries = raw["questions"]
|
||||||
|
elif isinstance(raw, list):
|
||||||
|
entries = raw
|
||||||
|
else:
|
||||||
|
raise ValueError(f"{path}: expected a list or {{questions: [...]}}")
|
||||||
|
|
||||||
|
ids: set[str] = set()
|
||||||
|
for e in entries:
|
||||||
|
if not isinstance(e, dict) or "id" not in e or "question" not in e:
|
||||||
|
raise ValueError(f"{path}: entry missing id/question: {e!r}")
|
||||||
|
if e["id"] in ids:
|
||||||
|
raise ValueError(f"{path}: duplicate id {e['id']!r}")
|
||||||
|
ids.add(e["id"])
|
||||||
|
if not any(k in e for k in _EXPECTATION_KEYS):
|
||||||
|
raise ValueError(f"{path}: entry {e['id']!r} has no expectations")
|
||||||
|
return entries
|
||||||
|
|
||||||
|
|
||||||
|
# ── transcript ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Transcript:
|
||||||
|
events: list[dict] = field(default_factory=list)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def answer_text(self) -> str:
|
||||||
|
return "".join(
|
||||||
|
e.get("text", "") for e in self.events if e.get("type") == "token"
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def sources(self) -> list[dict]:
|
||||||
|
for e in self.events:
|
||||||
|
if e.get("type") == "sources":
|
||||||
|
return e.get("sources", [])
|
||||||
|
return []
|
||||||
|
|
||||||
|
@property
|
||||||
|
def lineage_events(self) -> list[dict]:
|
||||||
|
for e in self.events:
|
||||||
|
if e.get("type") == "lineage":
|
||||||
|
return e.get("events", [])
|
||||||
|
return []
|
||||||
|
|
||||||
|
@property
|
||||||
|
def error(self) -> str | None:
|
||||||
|
for e in self.events:
|
||||||
|
if e.get("type") == "error":
|
||||||
|
return e.get("message", "stream error")
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CheckResult:
|
||||||
|
name: str
|
||||||
|
passed: bool
|
||||||
|
detail: str
|
||||||
|
|
||||||
|
|
||||||
|
# ── checks (pure functions over a Transcript) ───────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_pid_range(spec: Any) -> tuple[int, int]:
|
||||||
|
if isinstance(spec, (list, tuple)):
|
||||||
|
return int(spec[0]), int(spec[1])
|
||||||
|
return int(spec), int(spec)
|
||||||
|
|
||||||
|
|
||||||
|
def check_anchors(transcript: Transcript, expect_anchors: list[dict]) -> CheckResult:
|
||||||
|
sources = transcript.sources
|
||||||
|
missing = []
|
||||||
|
for a in expect_anchors:
|
||||||
|
lo, hi = _parse_pid_range(a["p_id"])
|
||||||
|
found = False
|
||||||
|
for s in sources:
|
||||||
|
if s.get("item_key") != a["item_key"]:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
pid = int(s.get("p_id") or "")
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
continue
|
||||||
|
if lo <= pid <= hi:
|
||||||
|
found = True
|
||||||
|
break
|
||||||
|
if not found:
|
||||||
|
label = (
|
||||||
|
f"{a['item_key']} p{lo}" if lo == hi else f"{a['item_key']} p{lo}-{hi}"
|
||||||
|
)
|
||||||
|
missing.append(label)
|
||||||
|
passed = not missing
|
||||||
|
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||||
|
return CheckResult("anchors", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def check_labels(transcript: Transcript, expect_labels: list[str]) -> CheckResult:
|
||||||
|
text = transcript.answer_text
|
||||||
|
missing = [pat for pat in expect_labels if not re.search(pat, text)]
|
||||||
|
passed = not missing
|
||||||
|
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||||
|
return CheckResult("labels", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def check_forbidden(transcript: Transcript, forbid: list[dict]) -> CheckResult:
|
||||||
|
sentences = _SENT_SPLIT.split(transcript.answer_text)
|
||||||
|
violations = []
|
||||||
|
for rule in forbid:
|
||||||
|
pat = re.compile(rule["pattern"])
|
||||||
|
unless = rule.get("unless_label")
|
||||||
|
unless_re = re.compile(unless) if unless else None
|
||||||
|
for sent in sentences:
|
||||||
|
if not pat.search(sent):
|
||||||
|
continue
|
||||||
|
labels = _BRACKET_RE.findall(sent)
|
||||||
|
ok = unless_re is not None and any(unless_re.search(l) for l in labels)
|
||||||
|
if not ok:
|
||||||
|
violations.append(f"{rule['pattern']!r} in {sent.strip()[:120]!r}")
|
||||||
|
passed = not violations
|
||||||
|
detail = "ok" if passed else "; ".join(violations)
|
||||||
|
return CheckResult("forbidden", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def check_events(transcript: Transcript, expect_events: list[dict]) -> CheckResult:
|
||||||
|
events = transcript.lineage_events
|
||||||
|
missing = []
|
||||||
|
for exp in expect_events:
|
||||||
|
kinds = set(str(exp["kind"]).split("|"))
|
||||||
|
code = str(exp["code"])
|
||||||
|
year = int(exp["year"])
|
||||||
|
found = any(
|
||||||
|
e.get("code") == code
|
||||||
|
and e.get("kind") in kinds
|
||||||
|
and int(e.get("year", -1)) == year
|
||||||
|
for e in events
|
||||||
|
)
|
||||||
|
if not found:
|
||||||
|
missing.append(f"{code} {exp['kind']} {year}")
|
||||||
|
passed = not missing
|
||||||
|
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||||
|
return CheckResult("events", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def check_dockets(transcript: Transcript, expect_dockets: list[str]) -> CheckResult:
|
||||||
|
have = {
|
||||||
|
s.get("docket", "")
|
||||||
|
for s in transcript.sources
|
||||||
|
if s.get("kind") == "comment" and s.get("docket")
|
||||||
|
}
|
||||||
|
missing = [d for d in expect_dockets if d not in have]
|
||||||
|
passed = not missing
|
||||||
|
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||||
|
return CheckResult("dockets", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def _rule_year(source: dict) -> int:
|
||||||
|
"""The rule year a source belongs to: the docket id's embedded year
|
||||||
|
for a comment, else the source's own date year — no imports, so this
|
||||||
|
is a simplification of ``llm.rag.era_of`` (which resolves a
|
||||||
|
comment's docket to its actual PFS rule year via the bib store)."""
|
||||||
|
if source.get("kind") == "comment":
|
||||||
|
m = _DOCKET_YEAR_RE.search(source.get("docket", "") or "")
|
||||||
|
return int(m.group(0)) if m else 0
|
||||||
|
m = _DATE_YEAR_RE.match(source.get("date", "") or "")
|
||||||
|
return int(m.group(1)) if m else 0
|
||||||
|
|
||||||
|
|
||||||
|
def check_eras(transcript: Transcript, min_eras: int) -> CheckResult:
|
||||||
|
eras = {_rule_year(s) for s in transcript.sources}
|
||||||
|
eras.discard(0)
|
||||||
|
passed = len(eras) >= int(min_eras)
|
||||||
|
detail = f"eras={sorted(eras)}"
|
||||||
|
return CheckResult("eras", passed, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def evaluate(entry: dict, transcript: Transcript) -> list[CheckResult]:
|
||||||
|
if transcript.error is not None:
|
||||||
|
return [CheckResult("stream", False, transcript.error)]
|
||||||
|
results: list[CheckResult] = []
|
||||||
|
if "expect_anchors" in entry:
|
||||||
|
results.append(check_anchors(transcript, entry["expect_anchors"]))
|
||||||
|
if "expect_labels" in entry:
|
||||||
|
results.append(check_labels(transcript, entry["expect_labels"]))
|
||||||
|
if "forbid" in entry:
|
||||||
|
results.append(check_forbidden(transcript, entry["forbid"]))
|
||||||
|
if "expect_events" in entry:
|
||||||
|
results.append(check_events(transcript, entry["expect_events"]))
|
||||||
|
if "expect_dockets" in entry:
|
||||||
|
results.append(check_dockets(transcript, entry["expect_dockets"]))
|
||||||
|
if "min_eras" in entry:
|
||||||
|
results.append(check_eras(transcript, entry["min_eras"]))
|
||||||
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
# ── streaming the live /chat endpoint ───────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def stream_chat(url: str, question: str, mode: str, timeout: float) -> list[dict]:
|
||||||
|
"""POST ``{question, mode}`` to ``{url}/chat`` and parse the
|
||||||
|
``data: {json}\\n\\n`` SSE lines into a list of event dicts."""
|
||||||
|
events: list[dict] = []
|
||||||
|
with httpx.Client(timeout=timeout) as client:
|
||||||
|
with client.stream(
|
||||||
|
"POST", f"{url.rstrip('/')}/chat", json={"question": question, "mode": mode}
|
||||||
|
) as resp:
|
||||||
|
resp.raise_for_status()
|
||||||
|
for line in resp.iter_lines():
|
||||||
|
if not line or not line.startswith("data:"):
|
||||||
|
continue
|
||||||
|
payload = line[len("data:") :].strip()
|
||||||
|
if payload:
|
||||||
|
events.append(json.loads(payload))
|
||||||
|
return events
|
||||||
|
|
||||||
|
|
||||||
|
# ── issue filer (imported by path — nb_issue_filer.py sits beside
|
||||||
|
# this script both in-repo and when docker-cp'd into the container) ──
|
||||||
|
|
||||||
|
|
||||||
|
def _load_filer():
|
||||||
|
spec = importlib.util.spec_from_file_location(
|
||||||
|
"nb_issue_filer", _HERE / "nb_issue_filer.py"
|
||||||
|
)
|
||||||
|
assert spec and spec.loader
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
sys.modules["nb_issue_filer"] = mod
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
return mod
|
||||||
|
|
||||||
|
|
||||||
|
# ── runner ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def cmd_run(args: argparse.Namespace) -> int:
|
||||||
|
entries = load_set(Path(args.set_path))
|
||||||
|
if args.only:
|
||||||
|
entries = [e for e in entries if e["id"] == args.only]
|
||||||
|
if not entries:
|
||||||
|
print(f"no entry with id {args.only!r} in {args.set_path}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
rows: list[dict] = []
|
||||||
|
findings: list[dict] = []
|
||||||
|
any_fail = False
|
||||||
|
|
||||||
|
for entry in entries:
|
||||||
|
qid = entry["id"]
|
||||||
|
mode = entry.get("mode", "auto")
|
||||||
|
try:
|
||||||
|
events = stream_chat(args.url, entry["question"], mode, args.timeout)
|
||||||
|
results = evaluate(entry, Transcript(events))
|
||||||
|
except Exception as e: # noqa: BLE001 — a request failure is a finding, not a crash
|
||||||
|
results = [CheckResult("request", False, str(e))]
|
||||||
|
|
||||||
|
failing = [r for r in results if not r.passed]
|
||||||
|
if failing:
|
||||||
|
any_fail = True
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"id": qid,
|
||||||
|
"passed": not failing,
|
||||||
|
"checks": [dataclasses.asdict(r) for r in results],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
for r in failing:
|
||||||
|
findings.append(
|
||||||
|
{
|
||||||
|
"notebook": qid,
|
||||||
|
"ename": r.name,
|
||||||
|
"evalue": r.detail,
|
||||||
|
"detail": r.detail,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
for row in rows:
|
||||||
|
status = "PASS" if row["passed"] else "FAIL"
|
||||||
|
failing_names = ",".join(c["name"] for c in row["checks"] if not c["passed"])
|
||||||
|
print(f"{row['id']:32} {status:4} {failing_names}")
|
||||||
|
|
||||||
|
report = {
|
||||||
|
"rows": rows,
|
||||||
|
"pass": sum(r["passed"] for r in rows),
|
||||||
|
"fail": sum(not r["passed"] for r in rows),
|
||||||
|
}
|
||||||
|
if args.report:
|
||||||
|
Path(args.report).write_text(json.dumps(report, indent=2))
|
||||||
|
|
||||||
|
if args.file_issues:
|
||||||
|
filer = _load_filer()
|
||||||
|
filer.cmd_report(findings, args.source)
|
||||||
|
active = {
|
||||||
|
filer.signature(f["notebook"], f["ename"], f["evalue"]) for f in findings
|
||||||
|
}
|
||||||
|
filer.cmd_sweep(active, args.source)
|
||||||
|
|
||||||
|
return 1 if any_fail else 0
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
sub = parser.add_subparsers(dest="cmd", required=True)
|
||||||
|
|
||||||
|
p_run = sub.add_parser("run", help="stream the golden set against a live /chat")
|
||||||
|
p_run.add_argument("--url", required=True, help="base URL of the chat service")
|
||||||
|
p_run.add_argument("--set", required=True, dest="set_path", help="golden YAML path")
|
||||||
|
p_run.add_argument("--report", help="write a JSON report here")
|
||||||
|
p_run.add_argument(
|
||||||
|
"--file-issues",
|
||||||
|
action="store_true",
|
||||||
|
help="file/sweep regressions via nb_issue_filer",
|
||||||
|
)
|
||||||
|
p_run.add_argument("--source", default="llm-golden", help="filer 'source' label")
|
||||||
|
p_run.add_argument("--only", help="run only this entry id")
|
||||||
|
p_run.add_argument("--timeout", type=float, default=180.0)
|
||||||
|
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
if args.cmd == "run":
|
||||||
|
return cmd_run(args)
|
||||||
|
parser.error(f"unknown command {args.cmd!r}")
|
||||||
|
return 2 # pragma: no cover — argparse exits before this
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
61
tests/dev/test_gen_config_llm_golden.py
Normal file
61
tests/dev/test_gen_config_llm_golden.py
Normal file
@@ -0,0 +1,61 @@
|
|||||||
|
"""Snapshot tests for the generated LLM Golden nightly workflow (P49
|
||||||
|
Task 7, refs #692).
|
||||||
|
|
||||||
|
``dev/scripts/backends/gitea.py::_gen_llm_golden`` is modelled on
|
||||||
|
``_gen_notebooks_integration``: the golden set needs a live chat over
|
||||||
|
the real pgvector/DuckDB/Ollama stack, which only the ``llm`` container
|
||||||
|
has, so the runner is docker-cp'd in and run there. These tests guard
|
||||||
|
the shape of the generated workflow (docker exec against the ``llm``
|
||||||
|
container, the nightly cron, the failing-run issue label) without
|
||||||
|
requiring a live Gitea Actions run.
|
||||||
|
|
||||||
|
``backends`` is a plain (non-src) package under ``dev/scripts/`` — not
|
||||||
|
importable by its dotted name without that directory on ``sys.path``,
|
||||||
|
the same way ``dev/scripts/gen_config.py`` relies on being executed
|
||||||
|
from that directory.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
_DEV_SCRIPTS = Path(__file__).resolve().parents[2] / "dev" / "scripts"
|
||||||
|
if str(_DEV_SCRIPTS) not in sys.path:
|
||||||
|
sys.path.insert(0, str(_DEV_SCRIPTS))
|
||||||
|
|
||||||
|
from backends import gitea # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
def test_gen_llm_golden_path_and_shape():
|
||||||
|
path, content = gitea._gen_llm_golden("ubuntu-latest", "latest")
|
||||||
|
assert path == ".gitea/workflows/llm-golden.yml"
|
||||||
|
assert 'cron: "40 3 * * *"' in content
|
||||||
|
assert "docker exec llm mkdir -p /tmp/golden" in content
|
||||||
|
assert "docker cp dev/scripts/llm_golden.py llm:/tmp/golden/" in content
|
||||||
|
assert "docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/" in content
|
||||||
|
assert "docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/" in content
|
||||||
|
assert "uv run --project /app python /tmp/golden/llm_golden.py run" in content
|
||||||
|
assert "--url http://localhost:8000" in content
|
||||||
|
assert "--file-issues --source nightly-llm-golden" in content
|
||||||
|
assert "NB_ISSUE_LABEL=llm" in content
|
||||||
|
assert "GITEA_API_BASE=http://git:3000/api/v1" in content
|
||||||
|
assert "secrets.DEPLOY_TOKEN" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_gen_llm_golden_registered_in_emit():
|
||||||
|
files = gitea.emit(
|
||||||
|
[],
|
||||||
|
[],
|
||||||
|
{
|
||||||
|
"registry": "reg.example",
|
||||||
|
"org": "homelab",
|
||||||
|
"image_prefix": "stack",
|
||||||
|
"domain": "example.test",
|
||||||
|
"host_ip": "127.0.0.1",
|
||||||
|
"repo": "homelab/stack",
|
||||||
|
},
|
||||||
|
{},
|
||||||
|
)
|
||||||
|
assert ".gitea/workflows/llm-golden.yml" in files
|
||||||
|
assert "docker exec llm" in files[".gitea/workflows/llm-golden.yml"]
|
||||||
104
tests/llm/golden_lineage.yaml
Normal file
104
tests/llm/golden_lineage.yaml
Normal file
@@ -0,0 +1,104 @@
|
|||||||
|
# Golden longitudinal evaluation set (P49 Task 7, refs #692).
|
||||||
|
#
|
||||||
|
# Anchors verified against the corpus in #692. Run with:
|
||||||
|
# uv run python dev/scripts/llm_golden.py run --url <chat-url> \
|
||||||
|
# --set tests/llm/golden_lineage.yaml --report report.json
|
||||||
|
#
|
||||||
|
# Schema per entry:
|
||||||
|
# id unique string
|
||||||
|
# question the question sent to POST /chat
|
||||||
|
# mode "auto" (default), "timeline" or "recent"
|
||||||
|
# expect_anchors [{item_key, p_id}] p_id may be [lo, hi] for a range;
|
||||||
|
# each anchor must appear among the streamed `sources`
|
||||||
|
# expect_labels [regex] each must be matched somewhere in the
|
||||||
|
# concatenated `token` text (the cited answer prose)
|
||||||
|
# forbid [{pattern, unless_label}] `pattern` may appear only
|
||||||
|
# in a sentence that also cites a bracketed label
|
||||||
|
# matching `unless_label`
|
||||||
|
# expect_events [{code, kind, year}] kind may be "a|b" alternatives;
|
||||||
|
# must appear in the `lineage` event payload
|
||||||
|
# expect_dockets [docket_id] must appear among comment-kind sources
|
||||||
|
# min_eras distinct rule years among sources (rule/corpus: the
|
||||||
|
# source's date year; comment: its docket id's year)
|
||||||
|
|
||||||
|
- id: ccm-history
|
||||||
|
question: What is the history of CCM coding and payment?
|
||||||
|
mode: timeline
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: DE2VH9PD
|
||||||
|
p_id: [1249, 1251]
|
||||||
|
- item_key: YBM4IZUS
|
||||||
|
p_id: 1578
|
||||||
|
- item_key: JJ6AM5HJ
|
||||||
|
p_id: 1163
|
||||||
|
expect_events:
|
||||||
|
- code: "99490"
|
||||||
|
kind: created
|
||||||
|
year: 2015
|
||||||
|
- code: G2058
|
||||||
|
kind: replaced_by
|
||||||
|
year: 2021
|
||||||
|
- code: G0556
|
||||||
|
kind: created
|
||||||
|
year: 2025
|
||||||
|
min_eras: 4
|
||||||
|
|
||||||
|
- id: g2058-replacement
|
||||||
|
question: What replaced G2058, and why?
|
||||||
|
mode: timeline
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: YBM4IZUS
|
||||||
|
p_id: 1578
|
||||||
|
- item_key: YBM4IZUS
|
||||||
|
p_id: 2369
|
||||||
|
expect_events:
|
||||||
|
- code: G2058
|
||||||
|
kind: replaced_by
|
||||||
|
year: 2021
|
||||||
|
|
||||||
|
- id: apcm-vs-ccm-elements
|
||||||
|
question: How do APCM's elements differ from CCM's?
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: JJ6AM5HJ
|
||||||
|
p_id: [1164, 1185]
|
||||||
|
- item_key: DE2VH9PD
|
||||||
|
p_id: [1245, 1247]
|
||||||
|
|
||||||
|
- id: audio-only-em-99441
|
||||||
|
question: When were audio-only E/M codes payable, and why did 99441-99443 end?
|
||||||
|
mode: timeline
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: XFGGRBDH
|
||||||
|
p_id: 489
|
||||||
|
expect_events:
|
||||||
|
- code: "99441"
|
||||||
|
kind: deleted|disappeared|cpt_deleted
|
||||||
|
year: 2025
|
||||||
|
|
||||||
|
- id: g2211-commenters-2023-vs-2025
|
||||||
|
question: What did commenters say about G2211 in 2023 versus 2025?
|
||||||
|
mode: timeline
|
||||||
|
expect_dockets:
|
||||||
|
- CMS-2023-0121
|
||||||
|
- CMS-2025-0304
|
||||||
|
|
||||||
|
- id: 99490-telehealth-steps
|
||||||
|
question: Does 99490 pass telehealth Steps 1 through 3?
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: 2KVJ2HKX
|
||||||
|
p_id: 394
|
||||||
|
- item_key: 2KVJ2HKX
|
||||||
|
p_id: 396
|
||||||
|
- item_key: 2KVJ2HKX
|
||||||
|
p_id: 398
|
||||||
|
|
||||||
|
- id: g2064-g2065-to-99424-99426-rate
|
||||||
|
question: How did G2064/G2065 becoming 99424/99426 change the RHC/FQHC G0511 rate?
|
||||||
|
mode: timeline
|
||||||
|
expect_anchors:
|
||||||
|
- item_key: JE7KYBW3
|
||||||
|
p_id: 1100
|
||||||
|
- item_key: JE7KYBW3
|
||||||
|
p_id: 1111
|
||||||
|
- item_key: MZ24MX5S
|
||||||
|
p_id: 1305
|
||||||
269
tests/llm/test_golden.py
Normal file
269
tests/llm/test_golden.py
Normal file
@@ -0,0 +1,269 @@
|
|||||||
|
"""Tests for dev/scripts/llm_golden.py — the golden longitudinal chat
|
||||||
|
evaluation (P49 Task 7, refs #692).
|
||||||
|
|
||||||
|
Loaded by path like the other dev/scripts tests (tests/dev/test_nb_
|
||||||
|
issue_filer.py) since dev/scripts/ is not a package. Checkers are
|
||||||
|
exercised against canned SSE transcripts (pass and fail cases); the
|
||||||
|
YAML set is validated for shape; a live end-to-end test runs entry
|
||||||
|
"g2058-replacement" against a real /chat and is skipped unless
|
||||||
|
LLM_CHAT_URL is set.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import importlib.util
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
_SCRIPT = Path(__file__).resolve().parents[2] / "dev" / "scripts" / "llm_golden.py"
|
||||||
|
_spec = importlib.util.spec_from_file_location("_llm_golden", _SCRIPT)
|
||||||
|
assert _spec and _spec.loader
|
||||||
|
golden = importlib.util.module_from_spec(_spec)
|
||||||
|
sys.modules["_llm_golden"] = golden
|
||||||
|
_spec.loader.exec_module(golden)
|
||||||
|
|
||||||
|
Transcript = golden.Transcript
|
||||||
|
GOLDEN_SET = Path(__file__).resolve().parent / "golden_lineage.yaml"
|
||||||
|
|
||||||
|
|
||||||
|
def _sources(*rows: dict) -> dict:
|
||||||
|
return {"type": "sources", "sources": list(rows)}
|
||||||
|
|
||||||
|
|
||||||
|
def _tokens(*texts: str) -> list[dict]:
|
||||||
|
return [{"type": "token", "text": t} for t in texts]
|
||||||
|
|
||||||
|
|
||||||
|
def _lineage(*events: dict) -> dict:
|
||||||
|
return {"type": "lineage", "events": list(events)}
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_anchors ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_anchors_pass_exact_and_range():
|
||||||
|
t = Transcript(
|
||||||
|
[
|
||||||
|
_sources(
|
||||||
|
{"item_key": "YBM4IZUS", "p_id": "1578"},
|
||||||
|
{"item_key": "DE2VH9PD", "p_id": "1250"},
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
result = golden.check_anchors(
|
||||||
|
t,
|
||||||
|
[
|
||||||
|
{"item_key": "YBM4IZUS", "p_id": 1578},
|
||||||
|
{"item_key": "DE2VH9PD", "p_id": [1249, 1251]},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_anchors_fail_missing():
|
||||||
|
t = Transcript([_sources({"item_key": "YBM4IZUS", "p_id": "1578"})])
|
||||||
|
result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": 1250}])
|
||||||
|
assert not result.passed
|
||||||
|
assert "DE2VH9PD" in result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_anchors_fail_pid_outside_range():
|
||||||
|
t = Transcript([_sources({"item_key": "DE2VH9PD", "p_id": "1300"})])
|
||||||
|
result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": [1249, 1251]}])
|
||||||
|
assert not result.passed
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_labels ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_labels_pass():
|
||||||
|
t = Transcript(
|
||||||
|
_tokens("As discussed in ", "[CY2021 PFS final ¶1578], G2058 was...")
|
||||||
|
)
|
||||||
|
result = golden.check_labels(t, [r"CY2021 PFS final ¶1578"])
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_labels_fail():
|
||||||
|
t = Transcript(_tokens("No citations here."))
|
||||||
|
result = golden.check_labels(t, [r"CY2021 PFS final"])
|
||||||
|
assert not result.passed
|
||||||
|
assert "CY2021 PFS final" in result.detail
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_forbidden ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_forbidden_pass_when_labeled():
|
||||||
|
t = Transcript(
|
||||||
|
_tokens("The payment is $34.85 [PFS CY2026 Addendum B]. ", "That is all.")
|
||||||
|
)
|
||||||
|
result = golden.check_forbidden(
|
||||||
|
t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}]
|
||||||
|
)
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_forbidden_fail_unlabeled_dollar():
|
||||||
|
t = Transcript(_tokens("The payment is roughly $34.85 based on recent rules."))
|
||||||
|
result = golden.check_forbidden(
|
||||||
|
t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}]
|
||||||
|
)
|
||||||
|
assert not result.passed
|
||||||
|
assert "$" in result.detail or "\\$" in result.detail
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_events ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_events_pass_alternatives():
|
||||||
|
t = Transcript([_lineage({"code": "99441", "kind": "disappeared", "year": 2025})])
|
||||||
|
result = golden.check_events(
|
||||||
|
t, [{"code": "99441", "kind": "deleted|disappeared|cpt_deleted", "year": 2025}]
|
||||||
|
)
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_events_fail_wrong_year():
|
||||||
|
t = Transcript([_lineage({"code": "G2058", "kind": "replaced_by", "year": 2020})])
|
||||||
|
result = golden.check_events(
|
||||||
|
t, [{"code": "G2058", "kind": "replaced_by", "year": 2021}]
|
||||||
|
)
|
||||||
|
assert not result.passed
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_dockets ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_dockets_pass():
|
||||||
|
t = Transcript(
|
||||||
|
[
|
||||||
|
_sources(
|
||||||
|
{"kind": "comment", "docket": "CMS-2023-0121"},
|
||||||
|
{"kind": "comment", "docket": "CMS-2025-0304"},
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"])
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_dockets_fail_missing_one():
|
||||||
|
t = Transcript([_sources({"kind": "comment", "docket": "CMS-2023-0121"})])
|
||||||
|
result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"])
|
||||||
|
assert not result.passed
|
||||||
|
assert "CMS-2025-0304" in result.detail
|
||||||
|
|
||||||
|
|
||||||
|
# ── check_eras ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_eras_pass():
|
||||||
|
t = Transcript(
|
||||||
|
[
|
||||||
|
_sources(
|
||||||
|
{"kind": "rule", "date": "2020-11-01"},
|
||||||
|
{"kind": "rule", "date": "2015-11-01"},
|
||||||
|
{"kind": "comment", "docket": "CMS-2023-0121"},
|
||||||
|
{"kind": "corpus", "date": "2025-01-01"},
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
result = golden.check_eras(t, 4)
|
||||||
|
assert result.passed, result.detail
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_eras_fail_too_few():
|
||||||
|
t = Transcript(
|
||||||
|
[
|
||||||
|
_sources(
|
||||||
|
{"kind": "rule", "date": "2020-11-01"},
|
||||||
|
{"kind": "rule", "date": "2020-12-01"},
|
||||||
|
)
|
||||||
|
]
|
||||||
|
)
|
||||||
|
result = golden.check_eras(t, 2)
|
||||||
|
assert not result.passed
|
||||||
|
|
||||||
|
|
||||||
|
def test_check_eras_ignores_undated():
|
||||||
|
t = Transcript([_sources({"kind": "corpus", "date": ""})])
|
||||||
|
result = golden.check_eras(t, 1)
|
||||||
|
assert not result.passed
|
||||||
|
|
||||||
|
|
||||||
|
# ── evaluate: stream error short-circuits ───────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_evaluate_short_circuits_on_error_event():
|
||||||
|
t = Transcript([{"type": "error", "message": "boom"}])
|
||||||
|
results = golden.evaluate({"expect_labels": ["x"]}, t)
|
||||||
|
assert len(results) == 1
|
||||||
|
assert results[0].name == "stream"
|
||||||
|
assert not results[0].passed
|
||||||
|
assert "boom" in results[0].detail
|
||||||
|
|
||||||
|
|
||||||
|
# ── YAML shape ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_golden_set_loads_and_is_well_formed():
|
||||||
|
entries = golden.load_set(GOLDEN_SET)
|
||||||
|
assert len(entries) == 7
|
||||||
|
ids = [e["id"] for e in entries]
|
||||||
|
assert len(ids) == len(set(ids)), "duplicate ids"
|
||||||
|
for e in entries:
|
||||||
|
assert "id" in e and "question" in e
|
||||||
|
assert any(k in e for k in golden._EXPECTATION_KEYS), e["id"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_golden_set_expected_ids_present():
|
||||||
|
entries = {e["id"] for e in golden.load_set(GOLDEN_SET)}
|
||||||
|
assert entries == {
|
||||||
|
"ccm-history",
|
||||||
|
"g2058-replacement",
|
||||||
|
"apcm-vs-ccm-elements",
|
||||||
|
"audio-only-em-99441",
|
||||||
|
"g2211-commenters-2023-vs-2025",
|
||||||
|
"99490-telehealth-steps",
|
||||||
|
"g2064-g2065-to-99424-99426-rate",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_set_rejects_duplicate_ids(tmp_path):
|
||||||
|
bad = tmp_path / "bad.yaml"
|
||||||
|
bad.write_text(
|
||||||
|
"- id: a\n question: q1\n min_eras: 1\n"
|
||||||
|
"- id: a\n question: q2\n min_eras: 1\n"
|
||||||
|
)
|
||||||
|
with pytest.raises(ValueError, match="duplicate id"):
|
||||||
|
golden.load_set(bad)
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_set_rejects_entry_without_expectations(tmp_path):
|
||||||
|
bad = tmp_path / "bad.yaml"
|
||||||
|
bad.write_text("- id: a\n question: q1\n")
|
||||||
|
with pytest.raises(ValueError, match="no expectations"):
|
||||||
|
golden.load_set(bad)
|
||||||
|
|
||||||
|
|
||||||
|
# ── live (opt-in) ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
not os.environ.get("LLM_CHAT_URL"),
|
||||||
|
reason="LLM_CHAT_URL not set — skipping live chat test",
|
||||||
|
)
|
||||||
|
def test_live_g2058_replacement():
|
||||||
|
url = os.environ["LLM_CHAT_URL"]
|
||||||
|
entries = {e["id"]: e for e in golden.load_set(GOLDEN_SET)}
|
||||||
|
entry = entries["g2058-replacement"]
|
||||||
|
events = golden.stream_chat(
|
||||||
|
url, entry["question"], entry.get("mode", "auto"), 180.0
|
||||||
|
)
|
||||||
|
results = golden.evaluate(entry, Transcript(events))
|
||||||
|
failing = [r for r in results if not r.passed]
|
||||||
|
assert not failing, [(r.name, r.detail) for r in failing]
|
||||||
Reference in New Issue
Block a user