feat(llm): golden longitudinal evaluation — yaml set, runner with anchor/label/forbidden checks, nightly workflow + issue filer (refs #692)
This commit is contained in:
55
.gitea/workflows/llm-golden.yml
Normal file
55
.gitea/workflows/llm-golden.yml
Normal file
@@ -0,0 +1,55 @@
|
||||
# DO NOT EDIT — generated by gen_config.py from stack.toml
|
||||
# Re-generate: uv run python dev/scripts/gen_config.py
|
||||
|
||||
name: LLM Golden
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "40 3 * * *"
|
||||
|
||||
jobs:
|
||||
llm-golden:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: https://github.com/actions/checkout@v4
|
||||
|
||||
- name: Set up uv
|
||||
run: curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
env:
|
||||
UV_INSTALL_DIR: /usr/local/bin
|
||||
|
||||
- name: Run golden longitudinal evaluation against the live chat
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker exec llm mkdir -p /tmp/golden
|
||||
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
|
||||
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
|
||||
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
|
||||
docker exec \
|
||||
-e GITEA_TOKEN \
|
||||
-e GITEA_API_BASE=http://git:3000/api/v1 \
|
||||
-e NB_ISSUE_LABEL=llm \
|
||||
llm \
|
||||
uv run --project /app python /tmp/golden/llm_golden.py run \
|
||||
--url http://localhost:8000 \
|
||||
--set /tmp/golden/golden_lineage.yaml \
|
||||
--report /tmp/golden/report.json \
|
||||
--file-issues --source nightly-llm-golden
|
||||
docker cp llm:/tmp/golden/report.json llm-golden-report.json
|
||||
cat llm-golden-report.json
|
||||
|
||||
- name: File failure issue
|
||||
if: failure()
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
||||
run: |
|
||||
uv sync --no-dev --quiet 2>/dev/null || true
|
||||
uv run python -m api.diag.ci \
|
||||
--workflow "LLM Golden" --job "llm-golden" \
|
||||
--run "${{ github.run_number }}" \
|
||||
--sha "${{ github.sha }}" \
|
||||
--ref "${{ github.ref }}" || true
|
||||
@@ -643,6 +643,62 @@ jobs:
|
||||
return (".gitea/workflows/zotero-sync.yml", content)
|
||||
|
||||
|
||||
def _gen_llm_golden(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]:
|
||||
"""Nightly golden longitudinal evaluation of the chat (P49 Task 7).
|
||||
|
||||
Mirrors ``_gen_notebooks_integration``: the golden set needs a live
|
||||
chat over the real pgvector/DuckDB/Ollama stack, which only the
|
||||
``llm`` container (``uvicorn llm.api:app``, port 8000, no published
|
||||
port) has — so the runner and the golden set are docker-cp'd in and
|
||||
run there against its own ``http://localhost:8000``. Regressions
|
||||
are filed through ``nb_issue_filer`` (dedup + auto-close) under the
|
||||
``llm`` label, not ``notebooks``. 03:40 sits after the 03:30
|
||||
notebooks-integration run and before the 04:15 zotero-sync run.
|
||||
"""
|
||||
content = f"""\
|
||||
{_HEADER}
|
||||
name: LLM Golden
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "40 3 * * *"
|
||||
|
||||
jobs:
|
||||
llm-golden:
|
||||
runs-on: {runner}
|
||||
steps:
|
||||
{_checkout_step()}
|
||||
|
||||
{_setup_uv_step(uv_version)}
|
||||
|
||||
- name: Run golden longitudinal evaluation against the live chat
|
||||
env:
|
||||
GITEA_TOKEN: ${{{{ secrets.DEPLOY_TOKEN }}}}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker exec llm mkdir -p /tmp/golden
|
||||
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
|
||||
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
|
||||
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
|
||||
docker exec \\
|
||||
-e GITEA_TOKEN \\
|
||||
-e GITEA_API_BASE=http://git:3000/api/v1 \\
|
||||
-e NB_ISSUE_LABEL=llm \\
|
||||
llm \\
|
||||
uv run --project /app python /tmp/golden/llm_golden.py run \\
|
||||
--url http://localhost:8000 \\
|
||||
--set /tmp/golden/golden_lineage.yaml \\
|
||||
--report /tmp/golden/report.json \\
|
||||
--file-issues --source nightly-llm-golden
|
||||
docker cp llm:/tmp/golden/report.json llm-golden-report.json
|
||||
cat llm-golden-report.json
|
||||
|
||||
{_failure_step("LLM Golden", "llm-golden")}
|
||||
"""
|
||||
return (".gitea/workflows/llm-golden.yml", content)
|
||||
|
||||
|
||||
def _gen_release(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]:
|
||||
content = f"""\
|
||||
{_HEADER}
|
||||
@@ -710,6 +766,7 @@ def emit(
|
||||
_gen_infra_ci,
|
||||
_gen_notebooks_integration,
|
||||
_gen_zotero_sync,
|
||||
_gen_llm_golden,
|
||||
_gen_release,
|
||||
):
|
||||
path, content = gen_fn(**common) # type: ignore[arg-type]
|
||||
|
||||
389
dev/scripts/llm_golden.py
Normal file
389
dev/scripts/llm_golden.py
Normal file
@@ -0,0 +1,389 @@
|
||||
"""Golden longitudinal evaluation for the chat (P49 Task 7).
|
||||
|
||||
Streams the live ``/chat`` SSE endpoint for a fixed set of longitudinal
|
||||
questions (``tests/llm/golden_lineage.yaml``) and checks each answer
|
||||
against verified expectations: FR paragraph anchors that must appear in
|
||||
``sources``, citation labels that must appear in the answer prose,
|
||||
forbidden patterns (e.g. a bare dollar amount not backed by a valuation
|
||||
row), lineage timeline events, comment dockets, and the number of
|
||||
distinct rule eras the sources span. Regressions are filed/swept through
|
||||
``dev/scripts/nb_issue_filer.py`` (copied beside this script in CI, and
|
||||
imported by path — never re-implemented here).
|
||||
|
||||
Stdlib + httpx + PyYAML only — this script is docker-cp'd into the
|
||||
``llm`` container and run there against its own ``http://localhost:8000``
|
||||
(see ``dev/scripts/backends/gitea.py::_gen_llm_golden``); it must not
|
||||
import ``llm.*``/``pfs.*``.
|
||||
|
||||
Usage::
|
||||
|
||||
uv run python dev/scripts/llm_golden.py run \\
|
||||
--url http://localhost:8000 --set tests/llm/golden_lineage.yaml \\
|
||||
--report report.json [--file-issues --source nightly-llm-golden] \\
|
||||
[--only g2058-replacement] [--timeout 180]
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import dataclasses
|
||||
import importlib.util
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
import yaml
|
||||
|
||||
_HERE = Path(__file__).resolve().parent
|
||||
|
||||
_EXPECTATION_KEYS = (
|
||||
"expect_anchors",
|
||||
"expect_labels",
|
||||
"forbid",
|
||||
"expect_events",
|
||||
"expect_dockets",
|
||||
"min_eras",
|
||||
)
|
||||
|
||||
_SENT_SPLIT = re.compile(r"(?<=[.!?])\s+")
|
||||
_BRACKET_RE = re.compile(r"\[([^\]]+)\]")
|
||||
_DOCKET_YEAR_RE = re.compile(r"(19|20)\d{2}")
|
||||
_DATE_YEAR_RE = re.compile(r"^(\d{4})")
|
||||
|
||||
|
||||
# ── golden set loading ───────────────────────────────────────────
|
||||
|
||||
|
||||
def load_set(path: Path) -> list[dict]:
|
||||
"""Parse and validate the golden YAML set: a top-level list (or
|
||||
``{questions: [...]}``) of entries, each with a unique ``id``, a
|
||||
``question``, and at least one expectation key."""
|
||||
raw = yaml.safe_load(path.read_text())
|
||||
if isinstance(raw, dict) and "questions" in raw:
|
||||
entries = raw["questions"]
|
||||
elif isinstance(raw, list):
|
||||
entries = raw
|
||||
else:
|
||||
raise ValueError(f"{path}: expected a list or {{questions: [...]}}")
|
||||
|
||||
ids: set[str] = set()
|
||||
for e in entries:
|
||||
if not isinstance(e, dict) or "id" not in e or "question" not in e:
|
||||
raise ValueError(f"{path}: entry missing id/question: {e!r}")
|
||||
if e["id"] in ids:
|
||||
raise ValueError(f"{path}: duplicate id {e['id']!r}")
|
||||
ids.add(e["id"])
|
||||
if not any(k in e for k in _EXPECTATION_KEYS):
|
||||
raise ValueError(f"{path}: entry {e['id']!r} has no expectations")
|
||||
return entries
|
||||
|
||||
|
||||
# ── transcript ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
@dataclass
|
||||
class Transcript:
|
||||
events: list[dict] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def answer_text(self) -> str:
|
||||
return "".join(
|
||||
e.get("text", "") for e in self.events if e.get("type") == "token"
|
||||
)
|
||||
|
||||
@property
|
||||
def sources(self) -> list[dict]:
|
||||
for e in self.events:
|
||||
if e.get("type") == "sources":
|
||||
return e.get("sources", [])
|
||||
return []
|
||||
|
||||
@property
|
||||
def lineage_events(self) -> list[dict]:
|
||||
for e in self.events:
|
||||
if e.get("type") == "lineage":
|
||||
return e.get("events", [])
|
||||
return []
|
||||
|
||||
@property
|
||||
def error(self) -> str | None:
|
||||
for e in self.events:
|
||||
if e.get("type") == "error":
|
||||
return e.get("message", "stream error")
|
||||
return None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CheckResult:
|
||||
name: str
|
||||
passed: bool
|
||||
detail: str
|
||||
|
||||
|
||||
# ── checks (pure functions over a Transcript) ───────────────────
|
||||
|
||||
|
||||
def _parse_pid_range(spec: Any) -> tuple[int, int]:
|
||||
if isinstance(spec, (list, tuple)):
|
||||
return int(spec[0]), int(spec[1])
|
||||
return int(spec), int(spec)
|
||||
|
||||
|
||||
def check_anchors(transcript: Transcript, expect_anchors: list[dict]) -> CheckResult:
|
||||
sources = transcript.sources
|
||||
missing = []
|
||||
for a in expect_anchors:
|
||||
lo, hi = _parse_pid_range(a["p_id"])
|
||||
found = False
|
||||
for s in sources:
|
||||
if s.get("item_key") != a["item_key"]:
|
||||
continue
|
||||
try:
|
||||
pid = int(s.get("p_id") or "")
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if lo <= pid <= hi:
|
||||
found = True
|
||||
break
|
||||
if not found:
|
||||
label = (
|
||||
f"{a['item_key']} p{lo}" if lo == hi else f"{a['item_key']} p{lo}-{hi}"
|
||||
)
|
||||
missing.append(label)
|
||||
passed = not missing
|
||||
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||
return CheckResult("anchors", passed, detail)
|
||||
|
||||
|
||||
def check_labels(transcript: Transcript, expect_labels: list[str]) -> CheckResult:
|
||||
text = transcript.answer_text
|
||||
missing = [pat for pat in expect_labels if not re.search(pat, text)]
|
||||
passed = not missing
|
||||
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||
return CheckResult("labels", passed, detail)
|
||||
|
||||
|
||||
def check_forbidden(transcript: Transcript, forbid: list[dict]) -> CheckResult:
|
||||
sentences = _SENT_SPLIT.split(transcript.answer_text)
|
||||
violations = []
|
||||
for rule in forbid:
|
||||
pat = re.compile(rule["pattern"])
|
||||
unless = rule.get("unless_label")
|
||||
unless_re = re.compile(unless) if unless else None
|
||||
for sent in sentences:
|
||||
if not pat.search(sent):
|
||||
continue
|
||||
labels = _BRACKET_RE.findall(sent)
|
||||
ok = unless_re is not None and any(unless_re.search(l) for l in labels)
|
||||
if not ok:
|
||||
violations.append(f"{rule['pattern']!r} in {sent.strip()[:120]!r}")
|
||||
passed = not violations
|
||||
detail = "ok" if passed else "; ".join(violations)
|
||||
return CheckResult("forbidden", passed, detail)
|
||||
|
||||
|
||||
def check_events(transcript: Transcript, expect_events: list[dict]) -> CheckResult:
|
||||
events = transcript.lineage_events
|
||||
missing = []
|
||||
for exp in expect_events:
|
||||
kinds = set(str(exp["kind"]).split("|"))
|
||||
code = str(exp["code"])
|
||||
year = int(exp["year"])
|
||||
found = any(
|
||||
e.get("code") == code
|
||||
and e.get("kind") in kinds
|
||||
and int(e.get("year", -1)) == year
|
||||
for e in events
|
||||
)
|
||||
if not found:
|
||||
missing.append(f"{code} {exp['kind']} {year}")
|
||||
passed = not missing
|
||||
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||
return CheckResult("events", passed, detail)
|
||||
|
||||
|
||||
def check_dockets(transcript: Transcript, expect_dockets: list[str]) -> CheckResult:
|
||||
have = {
|
||||
s.get("docket", "")
|
||||
for s in transcript.sources
|
||||
if s.get("kind") == "comment" and s.get("docket")
|
||||
}
|
||||
missing = [d for d in expect_dockets if d not in have]
|
||||
passed = not missing
|
||||
detail = "ok" if passed else "missing: " + ", ".join(missing)
|
||||
return CheckResult("dockets", passed, detail)
|
||||
|
||||
|
||||
def _rule_year(source: dict) -> int:
|
||||
"""The rule year a source belongs to: the docket id's embedded year
|
||||
for a comment, else the source's own date year — no imports, so this
|
||||
is a simplification of ``llm.rag.era_of`` (which resolves a
|
||||
comment's docket to its actual PFS rule year via the bib store)."""
|
||||
if source.get("kind") == "comment":
|
||||
m = _DOCKET_YEAR_RE.search(source.get("docket", "") or "")
|
||||
return int(m.group(0)) if m else 0
|
||||
m = _DATE_YEAR_RE.match(source.get("date", "") or "")
|
||||
return int(m.group(1)) if m else 0
|
||||
|
||||
|
||||
def check_eras(transcript: Transcript, min_eras: int) -> CheckResult:
|
||||
eras = {_rule_year(s) for s in transcript.sources}
|
||||
eras.discard(0)
|
||||
passed = len(eras) >= int(min_eras)
|
||||
detail = f"eras={sorted(eras)}"
|
||||
return CheckResult("eras", passed, detail)
|
||||
|
||||
|
||||
def evaluate(entry: dict, transcript: Transcript) -> list[CheckResult]:
|
||||
if transcript.error is not None:
|
||||
return [CheckResult("stream", False, transcript.error)]
|
||||
results: list[CheckResult] = []
|
||||
if "expect_anchors" in entry:
|
||||
results.append(check_anchors(transcript, entry["expect_anchors"]))
|
||||
if "expect_labels" in entry:
|
||||
results.append(check_labels(transcript, entry["expect_labels"]))
|
||||
if "forbid" in entry:
|
||||
results.append(check_forbidden(transcript, entry["forbid"]))
|
||||
if "expect_events" in entry:
|
||||
results.append(check_events(transcript, entry["expect_events"]))
|
||||
if "expect_dockets" in entry:
|
||||
results.append(check_dockets(transcript, entry["expect_dockets"]))
|
||||
if "min_eras" in entry:
|
||||
results.append(check_eras(transcript, entry["min_eras"]))
|
||||
return results
|
||||
|
||||
|
||||
# ── streaming the live /chat endpoint ───────────────────────────
|
||||
|
||||
|
||||
def stream_chat(url: str, question: str, mode: str, timeout: float) -> list[dict]:
|
||||
"""POST ``{question, mode}`` to ``{url}/chat`` and parse the
|
||||
``data: {json}\\n\\n`` SSE lines into a list of event dicts."""
|
||||
events: list[dict] = []
|
||||
with httpx.Client(timeout=timeout) as client:
|
||||
with client.stream(
|
||||
"POST", f"{url.rstrip('/')}/chat", json={"question": question, "mode": mode}
|
||||
) as resp:
|
||||
resp.raise_for_status()
|
||||
for line in resp.iter_lines():
|
||||
if not line or not line.startswith("data:"):
|
||||
continue
|
||||
payload = line[len("data:") :].strip()
|
||||
if payload:
|
||||
events.append(json.loads(payload))
|
||||
return events
|
||||
|
||||
|
||||
# ── issue filer (imported by path — nb_issue_filer.py sits beside
|
||||
# this script both in-repo and when docker-cp'd into the container) ──
|
||||
|
||||
|
||||
def _load_filer():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"nb_issue_filer", _HERE / "nb_issue_filer.py"
|
||||
)
|
||||
assert spec and spec.loader
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
sys.modules["nb_issue_filer"] = mod
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
|
||||
|
||||
# ── runner ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def cmd_run(args: argparse.Namespace) -> int:
|
||||
entries = load_set(Path(args.set_path))
|
||||
if args.only:
|
||||
entries = [e for e in entries if e["id"] == args.only]
|
||||
if not entries:
|
||||
print(f"no entry with id {args.only!r} in {args.set_path}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
rows: list[dict] = []
|
||||
findings: list[dict] = []
|
||||
any_fail = False
|
||||
|
||||
for entry in entries:
|
||||
qid = entry["id"]
|
||||
mode = entry.get("mode", "auto")
|
||||
try:
|
||||
events = stream_chat(args.url, entry["question"], mode, args.timeout)
|
||||
results = evaluate(entry, Transcript(events))
|
||||
except Exception as e: # noqa: BLE001 — a request failure is a finding, not a crash
|
||||
results = [CheckResult("request", False, str(e))]
|
||||
|
||||
failing = [r for r in results if not r.passed]
|
||||
if failing:
|
||||
any_fail = True
|
||||
rows.append(
|
||||
{
|
||||
"id": qid,
|
||||
"passed": not failing,
|
||||
"checks": [dataclasses.asdict(r) for r in results],
|
||||
}
|
||||
)
|
||||
for r in failing:
|
||||
findings.append(
|
||||
{
|
||||
"notebook": qid,
|
||||
"ename": r.name,
|
||||
"evalue": r.detail,
|
||||
"detail": r.detail,
|
||||
}
|
||||
)
|
||||
|
||||
for row in rows:
|
||||
status = "PASS" if row["passed"] else "FAIL"
|
||||
failing_names = ",".join(c["name"] for c in row["checks"] if not c["passed"])
|
||||
print(f"{row['id']:32} {status:4} {failing_names}")
|
||||
|
||||
report = {
|
||||
"rows": rows,
|
||||
"pass": sum(r["passed"] for r in rows),
|
||||
"fail": sum(not r["passed"] for r in rows),
|
||||
}
|
||||
if args.report:
|
||||
Path(args.report).write_text(json.dumps(report, indent=2))
|
||||
|
||||
if args.file_issues:
|
||||
filer = _load_filer()
|
||||
filer.cmd_report(findings, args.source)
|
||||
active = {
|
||||
filer.signature(f["notebook"], f["ename"], f["evalue"]) for f in findings
|
||||
}
|
||||
filer.cmd_sweep(active, args.source)
|
||||
|
||||
return 1 if any_fail else 0
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
sub = parser.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
p_run = sub.add_parser("run", help="stream the golden set against a live /chat")
|
||||
p_run.add_argument("--url", required=True, help="base URL of the chat service")
|
||||
p_run.add_argument("--set", required=True, dest="set_path", help="golden YAML path")
|
||||
p_run.add_argument("--report", help="write a JSON report here")
|
||||
p_run.add_argument(
|
||||
"--file-issues",
|
||||
action="store_true",
|
||||
help="file/sweep regressions via nb_issue_filer",
|
||||
)
|
||||
p_run.add_argument("--source", default="llm-golden", help="filer 'source' label")
|
||||
p_run.add_argument("--only", help="run only this entry id")
|
||||
p_run.add_argument("--timeout", type=float, default=180.0)
|
||||
|
||||
args = parser.parse_args(argv)
|
||||
if args.cmd == "run":
|
||||
return cmd_run(args)
|
||||
parser.error(f"unknown command {args.cmd!r}")
|
||||
return 2 # pragma: no cover — argparse exits before this
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
61
tests/dev/test_gen_config_llm_golden.py
Normal file
61
tests/dev/test_gen_config_llm_golden.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Snapshot tests for the generated LLM Golden nightly workflow (P49
|
||||
Task 7, refs #692).
|
||||
|
||||
``dev/scripts/backends/gitea.py::_gen_llm_golden`` is modelled on
|
||||
``_gen_notebooks_integration``: the golden set needs a live chat over
|
||||
the real pgvector/DuckDB/Ollama stack, which only the ``llm`` container
|
||||
has, so the runner is docker-cp'd in and run there. These tests guard
|
||||
the shape of the generated workflow (docker exec against the ``llm``
|
||||
container, the nightly cron, the failing-run issue label) without
|
||||
requiring a live Gitea Actions run.
|
||||
|
||||
``backends`` is a plain (non-src) package under ``dev/scripts/`` — not
|
||||
importable by its dotted name without that directory on ``sys.path``,
|
||||
the same way ``dev/scripts/gen_config.py`` relies on being executed
|
||||
from that directory.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
_DEV_SCRIPTS = Path(__file__).resolve().parents[2] / "dev" / "scripts"
|
||||
if str(_DEV_SCRIPTS) not in sys.path:
|
||||
sys.path.insert(0, str(_DEV_SCRIPTS))
|
||||
|
||||
from backends import gitea # noqa: E402
|
||||
|
||||
|
||||
def test_gen_llm_golden_path_and_shape():
|
||||
path, content = gitea._gen_llm_golden("ubuntu-latest", "latest")
|
||||
assert path == ".gitea/workflows/llm-golden.yml"
|
||||
assert 'cron: "40 3 * * *"' in content
|
||||
assert "docker exec llm mkdir -p /tmp/golden" in content
|
||||
assert "docker cp dev/scripts/llm_golden.py llm:/tmp/golden/" in content
|
||||
assert "docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/" in content
|
||||
assert "docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/" in content
|
||||
assert "uv run --project /app python /tmp/golden/llm_golden.py run" in content
|
||||
assert "--url http://localhost:8000" in content
|
||||
assert "--file-issues --source nightly-llm-golden" in content
|
||||
assert "NB_ISSUE_LABEL=llm" in content
|
||||
assert "GITEA_API_BASE=http://git:3000/api/v1" in content
|
||||
assert "secrets.DEPLOY_TOKEN" in content
|
||||
|
||||
|
||||
def test_gen_llm_golden_registered_in_emit():
|
||||
files = gitea.emit(
|
||||
[],
|
||||
[],
|
||||
{
|
||||
"registry": "reg.example",
|
||||
"org": "homelab",
|
||||
"image_prefix": "stack",
|
||||
"domain": "example.test",
|
||||
"host_ip": "127.0.0.1",
|
||||
"repo": "homelab/stack",
|
||||
},
|
||||
{},
|
||||
)
|
||||
assert ".gitea/workflows/llm-golden.yml" in files
|
||||
assert "docker exec llm" in files[".gitea/workflows/llm-golden.yml"]
|
||||
104
tests/llm/golden_lineage.yaml
Normal file
104
tests/llm/golden_lineage.yaml
Normal file
@@ -0,0 +1,104 @@
|
||||
# Golden longitudinal evaluation set (P49 Task 7, refs #692).
|
||||
#
|
||||
# Anchors verified against the corpus in #692. Run with:
|
||||
# uv run python dev/scripts/llm_golden.py run --url <chat-url> \
|
||||
# --set tests/llm/golden_lineage.yaml --report report.json
|
||||
#
|
||||
# Schema per entry:
|
||||
# id unique string
|
||||
# question the question sent to POST /chat
|
||||
# mode "auto" (default), "timeline" or "recent"
|
||||
# expect_anchors [{item_key, p_id}] p_id may be [lo, hi] for a range;
|
||||
# each anchor must appear among the streamed `sources`
|
||||
# expect_labels [regex] each must be matched somewhere in the
|
||||
# concatenated `token` text (the cited answer prose)
|
||||
# forbid [{pattern, unless_label}] `pattern` may appear only
|
||||
# in a sentence that also cites a bracketed label
|
||||
# matching `unless_label`
|
||||
# expect_events [{code, kind, year}] kind may be "a|b" alternatives;
|
||||
# must appear in the `lineage` event payload
|
||||
# expect_dockets [docket_id] must appear among comment-kind sources
|
||||
# min_eras distinct rule years among sources (rule/corpus: the
|
||||
# source's date year; comment: its docket id's year)
|
||||
|
||||
- id: ccm-history
|
||||
question: What is the history of CCM coding and payment?
|
||||
mode: timeline
|
||||
expect_anchors:
|
||||
- item_key: DE2VH9PD
|
||||
p_id: [1249, 1251]
|
||||
- item_key: YBM4IZUS
|
||||
p_id: 1578
|
||||
- item_key: JJ6AM5HJ
|
||||
p_id: 1163
|
||||
expect_events:
|
||||
- code: "99490"
|
||||
kind: created
|
||||
year: 2015
|
||||
- code: G2058
|
||||
kind: replaced_by
|
||||
year: 2021
|
||||
- code: G0556
|
||||
kind: created
|
||||
year: 2025
|
||||
min_eras: 4
|
||||
|
||||
- id: g2058-replacement
|
||||
question: What replaced G2058, and why?
|
||||
mode: timeline
|
||||
expect_anchors:
|
||||
- item_key: YBM4IZUS
|
||||
p_id: 1578
|
||||
- item_key: YBM4IZUS
|
||||
p_id: 2369
|
||||
expect_events:
|
||||
- code: G2058
|
||||
kind: replaced_by
|
||||
year: 2021
|
||||
|
||||
- id: apcm-vs-ccm-elements
|
||||
question: How do APCM's elements differ from CCM's?
|
||||
expect_anchors:
|
||||
- item_key: JJ6AM5HJ
|
||||
p_id: [1164, 1185]
|
||||
- item_key: DE2VH9PD
|
||||
p_id: [1245, 1247]
|
||||
|
||||
- id: audio-only-em-99441
|
||||
question: When were audio-only E/M codes payable, and why did 99441-99443 end?
|
||||
mode: timeline
|
||||
expect_anchors:
|
||||
- item_key: XFGGRBDH
|
||||
p_id: 489
|
||||
expect_events:
|
||||
- code: "99441"
|
||||
kind: deleted|disappeared|cpt_deleted
|
||||
year: 2025
|
||||
|
||||
- id: g2211-commenters-2023-vs-2025
|
||||
question: What did commenters say about G2211 in 2023 versus 2025?
|
||||
mode: timeline
|
||||
expect_dockets:
|
||||
- CMS-2023-0121
|
||||
- CMS-2025-0304
|
||||
|
||||
- id: 99490-telehealth-steps
|
||||
question: Does 99490 pass telehealth Steps 1 through 3?
|
||||
expect_anchors:
|
||||
- item_key: 2KVJ2HKX
|
||||
p_id: 394
|
||||
- item_key: 2KVJ2HKX
|
||||
p_id: 396
|
||||
- item_key: 2KVJ2HKX
|
||||
p_id: 398
|
||||
|
||||
- id: g2064-g2065-to-99424-99426-rate
|
||||
question: How did G2064/G2065 becoming 99424/99426 change the RHC/FQHC G0511 rate?
|
||||
mode: timeline
|
||||
expect_anchors:
|
||||
- item_key: JE7KYBW3
|
||||
p_id: 1100
|
||||
- item_key: JE7KYBW3
|
||||
p_id: 1111
|
||||
- item_key: MZ24MX5S
|
||||
p_id: 1305
|
||||
269
tests/llm/test_golden.py
Normal file
269
tests/llm/test_golden.py
Normal file
@@ -0,0 +1,269 @@
|
||||
"""Tests for dev/scripts/llm_golden.py — the golden longitudinal chat
|
||||
evaluation (P49 Task 7, refs #692).
|
||||
|
||||
Loaded by path like the other dev/scripts tests (tests/dev/test_nb_
|
||||
issue_filer.py) since dev/scripts/ is not a package. Checkers are
|
||||
exercised against canned SSE transcripts (pass and fail cases); the
|
||||
YAML set is validated for shape; a live end-to-end test runs entry
|
||||
"g2058-replacement" against a real /chat and is skipped unless
|
||||
LLM_CHAT_URL is set.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
_SCRIPT = Path(__file__).resolve().parents[2] / "dev" / "scripts" / "llm_golden.py"
|
||||
_spec = importlib.util.spec_from_file_location("_llm_golden", _SCRIPT)
|
||||
assert _spec and _spec.loader
|
||||
golden = importlib.util.module_from_spec(_spec)
|
||||
sys.modules["_llm_golden"] = golden
|
||||
_spec.loader.exec_module(golden)
|
||||
|
||||
Transcript = golden.Transcript
|
||||
GOLDEN_SET = Path(__file__).resolve().parent / "golden_lineage.yaml"
|
||||
|
||||
|
||||
def _sources(*rows: dict) -> dict:
|
||||
return {"type": "sources", "sources": list(rows)}
|
||||
|
||||
|
||||
def _tokens(*texts: str) -> list[dict]:
|
||||
return [{"type": "token", "text": t} for t in texts]
|
||||
|
||||
|
||||
def _lineage(*events: dict) -> dict:
|
||||
return {"type": "lineage", "events": list(events)}
|
||||
|
||||
|
||||
# ── check_anchors ────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_anchors_pass_exact_and_range():
|
||||
t = Transcript(
|
||||
[
|
||||
_sources(
|
||||
{"item_key": "YBM4IZUS", "p_id": "1578"},
|
||||
{"item_key": "DE2VH9PD", "p_id": "1250"},
|
||||
)
|
||||
]
|
||||
)
|
||||
result = golden.check_anchors(
|
||||
t,
|
||||
[
|
||||
{"item_key": "YBM4IZUS", "p_id": 1578},
|
||||
{"item_key": "DE2VH9PD", "p_id": [1249, 1251]},
|
||||
],
|
||||
)
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_anchors_fail_missing():
|
||||
t = Transcript([_sources({"item_key": "YBM4IZUS", "p_id": "1578"})])
|
||||
result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": 1250}])
|
||||
assert not result.passed
|
||||
assert "DE2VH9PD" in result.detail
|
||||
|
||||
|
||||
def test_check_anchors_fail_pid_outside_range():
|
||||
t = Transcript([_sources({"item_key": "DE2VH9PD", "p_id": "1300"})])
|
||||
result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": [1249, 1251]}])
|
||||
assert not result.passed
|
||||
|
||||
|
||||
# ── check_labels ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_labels_pass():
|
||||
t = Transcript(
|
||||
_tokens("As discussed in ", "[CY2021 PFS final ¶1578], G2058 was...")
|
||||
)
|
||||
result = golden.check_labels(t, [r"CY2021 PFS final ¶1578"])
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_labels_fail():
|
||||
t = Transcript(_tokens("No citations here."))
|
||||
result = golden.check_labels(t, [r"CY2021 PFS final"])
|
||||
assert not result.passed
|
||||
assert "CY2021 PFS final" in result.detail
|
||||
|
||||
|
||||
# ── check_forbidden ──────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_forbidden_pass_when_labeled():
|
||||
t = Transcript(
|
||||
_tokens("The payment is $34.85 [PFS CY2026 Addendum B]. ", "That is all.")
|
||||
)
|
||||
result = golden.check_forbidden(
|
||||
t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}]
|
||||
)
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_forbidden_fail_unlabeled_dollar():
|
||||
t = Transcript(_tokens("The payment is roughly $34.85 based on recent rules."))
|
||||
result = golden.check_forbidden(
|
||||
t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}]
|
||||
)
|
||||
assert not result.passed
|
||||
assert "$" in result.detail or "\\$" in result.detail
|
||||
|
||||
|
||||
# ── check_events ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_events_pass_alternatives():
|
||||
t = Transcript([_lineage({"code": "99441", "kind": "disappeared", "year": 2025})])
|
||||
result = golden.check_events(
|
||||
t, [{"code": "99441", "kind": "deleted|disappeared|cpt_deleted", "year": 2025}]
|
||||
)
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_events_fail_wrong_year():
|
||||
t = Transcript([_lineage({"code": "G2058", "kind": "replaced_by", "year": 2020})])
|
||||
result = golden.check_events(
|
||||
t, [{"code": "G2058", "kind": "replaced_by", "year": 2021}]
|
||||
)
|
||||
assert not result.passed
|
||||
|
||||
|
||||
# ── check_dockets ────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_dockets_pass():
|
||||
t = Transcript(
|
||||
[
|
||||
_sources(
|
||||
{"kind": "comment", "docket": "CMS-2023-0121"},
|
||||
{"kind": "comment", "docket": "CMS-2025-0304"},
|
||||
)
|
||||
]
|
||||
)
|
||||
result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"])
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_dockets_fail_missing_one():
|
||||
t = Transcript([_sources({"kind": "comment", "docket": "CMS-2023-0121"})])
|
||||
result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"])
|
||||
assert not result.passed
|
||||
assert "CMS-2025-0304" in result.detail
|
||||
|
||||
|
||||
# ── check_eras ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_check_eras_pass():
|
||||
t = Transcript(
|
||||
[
|
||||
_sources(
|
||||
{"kind": "rule", "date": "2020-11-01"},
|
||||
{"kind": "rule", "date": "2015-11-01"},
|
||||
{"kind": "comment", "docket": "CMS-2023-0121"},
|
||||
{"kind": "corpus", "date": "2025-01-01"},
|
||||
)
|
||||
]
|
||||
)
|
||||
result = golden.check_eras(t, 4)
|
||||
assert result.passed, result.detail
|
||||
|
||||
|
||||
def test_check_eras_fail_too_few():
|
||||
t = Transcript(
|
||||
[
|
||||
_sources(
|
||||
{"kind": "rule", "date": "2020-11-01"},
|
||||
{"kind": "rule", "date": "2020-12-01"},
|
||||
)
|
||||
]
|
||||
)
|
||||
result = golden.check_eras(t, 2)
|
||||
assert not result.passed
|
||||
|
||||
|
||||
def test_check_eras_ignores_undated():
|
||||
t = Transcript([_sources({"kind": "corpus", "date": ""})])
|
||||
result = golden.check_eras(t, 1)
|
||||
assert not result.passed
|
||||
|
||||
|
||||
# ── evaluate: stream error short-circuits ───────────────────────
|
||||
|
||||
|
||||
def test_evaluate_short_circuits_on_error_event():
|
||||
t = Transcript([{"type": "error", "message": "boom"}])
|
||||
results = golden.evaluate({"expect_labels": ["x"]}, t)
|
||||
assert len(results) == 1
|
||||
assert results[0].name == "stream"
|
||||
assert not results[0].passed
|
||||
assert "boom" in results[0].detail
|
||||
|
||||
|
||||
# ── YAML shape ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_golden_set_loads_and_is_well_formed():
|
||||
entries = golden.load_set(GOLDEN_SET)
|
||||
assert len(entries) == 7
|
||||
ids = [e["id"] for e in entries]
|
||||
assert len(ids) == len(set(ids)), "duplicate ids"
|
||||
for e in entries:
|
||||
assert "id" in e and "question" in e
|
||||
assert any(k in e for k in golden._EXPECTATION_KEYS), e["id"]
|
||||
|
||||
|
||||
def test_golden_set_expected_ids_present():
|
||||
entries = {e["id"] for e in golden.load_set(GOLDEN_SET)}
|
||||
assert entries == {
|
||||
"ccm-history",
|
||||
"g2058-replacement",
|
||||
"apcm-vs-ccm-elements",
|
||||
"audio-only-em-99441",
|
||||
"g2211-commenters-2023-vs-2025",
|
||||
"99490-telehealth-steps",
|
||||
"g2064-g2065-to-99424-99426-rate",
|
||||
}
|
||||
|
||||
|
||||
def test_load_set_rejects_duplicate_ids(tmp_path):
|
||||
bad = tmp_path / "bad.yaml"
|
||||
bad.write_text(
|
||||
"- id: a\n question: q1\n min_eras: 1\n"
|
||||
"- id: a\n question: q2\n min_eras: 1\n"
|
||||
)
|
||||
with pytest.raises(ValueError, match="duplicate id"):
|
||||
golden.load_set(bad)
|
||||
|
||||
|
||||
def test_load_set_rejects_entry_without_expectations(tmp_path):
|
||||
bad = tmp_path / "bad.yaml"
|
||||
bad.write_text("- id: a\n question: q1\n")
|
||||
with pytest.raises(ValueError, match="no expectations"):
|
||||
golden.load_set(bad)
|
||||
|
||||
|
||||
# ── live (opt-in) ────────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not os.environ.get("LLM_CHAT_URL"),
|
||||
reason="LLM_CHAT_URL not set — skipping live chat test",
|
||||
)
|
||||
def test_live_g2058_replacement():
|
||||
url = os.environ["LLM_CHAT_URL"]
|
||||
entries = {e["id"]: e for e in golden.load_set(GOLDEN_SET)}
|
||||
entry = entries["g2058-replacement"]
|
||||
events = golden.stream_chat(
|
||||
url, entry["question"], entry.get("mode", "auto"), 180.0
|
||||
)
|
||||
results = golden.evaluate(entry, Transcript(events))
|
||||
failing = [r for r in results if not r.passed]
|
||||
assert not failing, [(r.name, r.detail) for r in failing]
|
||||
Reference in New Issue
Block a user