diff --git a/.gitea/workflows/llm-golden.yml b/.gitea/workflows/llm-golden.yml new file mode 100644 index 0000000..adc0f6c --- /dev/null +++ b/.gitea/workflows/llm-golden.yml @@ -0,0 +1,55 @@ +# DO NOT EDIT — generated by gen_config.py from stack.toml +# Re-generate: uv run python dev/scripts/gen_config.py + +name: LLM Golden + +on: + workflow_dispatch: + schedule: + - cron: "40 3 * * *" + +jobs: + llm-golden: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: https://github.com/actions/checkout@v4 + + - name: Set up uv + run: curl -LsSf https://astral.sh/uv/install.sh | sh + env: + UV_INSTALL_DIR: /usr/local/bin + + - name: Run golden longitudinal evaluation against the live chat + env: + GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }} + run: | + set -euo pipefail + docker exec llm mkdir -p /tmp/golden + docker cp dev/scripts/llm_golden.py llm:/tmp/golden/ + docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/ + docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/ + docker exec \ + -e GITEA_TOKEN \ + -e GITEA_API_BASE=http://git:3000/api/v1 \ + -e NB_ISSUE_LABEL=llm \ + llm \ + uv run --project /app python /tmp/golden/llm_golden.py run \ + --url http://localhost:8000 \ + --set /tmp/golden/golden_lineage.yaml \ + --report /tmp/golden/report.json \ + --file-issues --source nightly-llm-golden + docker cp llm:/tmp/golden/report.json llm-golden-report.json + cat llm-golden-report.json + + - name: File failure issue + if: failure() + env: + GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }} + run: | + uv sync --no-dev --quiet 2>/dev/null || true + uv run python -m api.diag.ci \ + --workflow "LLM Golden" --job "llm-golden" \ + --run "${{ github.run_number }}" \ + --sha "${{ github.sha }}" \ + --ref "${{ github.ref }}" || true diff --git a/dev/scripts/backends/gitea.py b/dev/scripts/backends/gitea.py index e34644a..b645d0d 100644 --- a/dev/scripts/backends/gitea.py +++ b/dev/scripts/backends/gitea.py @@ -643,6 +643,62 @@ jobs: return (".gitea/workflows/zotero-sync.yml", content) +def _gen_llm_golden(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]: + """Nightly golden longitudinal evaluation of the chat (P49 Task 7). + + Mirrors ``_gen_notebooks_integration``: the golden set needs a live + chat over the real pgvector/DuckDB/Ollama stack, which only the + ``llm`` container (``uvicorn llm.api:app``, port 8000, no published + port) has — so the runner and the golden set are docker-cp'd in and + run there against its own ``http://localhost:8000``. Regressions + are filed through ``nb_issue_filer`` (dedup + auto-close) under the + ``llm`` label, not ``notebooks``. 03:40 sits after the 03:30 + notebooks-integration run and before the 04:15 zotero-sync run. + """ + content = f"""\ +{_HEADER} +name: LLM Golden + +on: + workflow_dispatch: + schedule: + - cron: "40 3 * * *" + +jobs: + llm-golden: + runs-on: {runner} + steps: +{_checkout_step()} + +{_setup_uv_step(uv_version)} + + - name: Run golden longitudinal evaluation against the live chat + env: + GITEA_TOKEN: ${{{{ secrets.DEPLOY_TOKEN }}}} + run: | + set -euo pipefail + docker exec llm mkdir -p /tmp/golden + docker cp dev/scripts/llm_golden.py llm:/tmp/golden/ + docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/ + docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/ + docker exec \\ + -e GITEA_TOKEN \\ + -e GITEA_API_BASE=http://git:3000/api/v1 \\ + -e NB_ISSUE_LABEL=llm \\ + llm \\ + uv run --project /app python /tmp/golden/llm_golden.py run \\ + --url http://localhost:8000 \\ + --set /tmp/golden/golden_lineage.yaml \\ + --report /tmp/golden/report.json \\ + --file-issues --source nightly-llm-golden + docker cp llm:/tmp/golden/report.json llm-golden-report.json + cat llm-golden-report.json + +{_failure_step("LLM Golden", "llm-golden")} +""" + return (".gitea/workflows/llm-golden.yml", content) + + def _gen_release(runner: str, uv_version: str, **_kw: object) -> tuple[str, str]: content = f"""\ {_HEADER} @@ -710,6 +766,7 @@ def emit( _gen_infra_ci, _gen_notebooks_integration, _gen_zotero_sync, + _gen_llm_golden, _gen_release, ): path, content = gen_fn(**common) # type: ignore[arg-type] diff --git a/dev/scripts/llm_golden.py b/dev/scripts/llm_golden.py new file mode 100644 index 0000000..899d231 --- /dev/null +++ b/dev/scripts/llm_golden.py @@ -0,0 +1,389 @@ +"""Golden longitudinal evaluation for the chat (P49 Task 7). + +Streams the live ``/chat`` SSE endpoint for a fixed set of longitudinal +questions (``tests/llm/golden_lineage.yaml``) and checks each answer +against verified expectations: FR paragraph anchors that must appear in +``sources``, citation labels that must appear in the answer prose, +forbidden patterns (e.g. a bare dollar amount not backed by a valuation +row), lineage timeline events, comment dockets, and the number of +distinct rule eras the sources span. Regressions are filed/swept through +``dev/scripts/nb_issue_filer.py`` (copied beside this script in CI, and +imported by path — never re-implemented here). + +Stdlib + httpx + PyYAML only — this script is docker-cp'd into the +``llm`` container and run there against its own ``http://localhost:8000`` +(see ``dev/scripts/backends/gitea.py::_gen_llm_golden``); it must not +import ``llm.*``/``pfs.*``. + +Usage:: + + uv run python dev/scripts/llm_golden.py run \\ + --url http://localhost:8000 --set tests/llm/golden_lineage.yaml \\ + --report report.json [--file-issues --source nightly-llm-golden] \\ + [--only g2058-replacement] [--timeout 180] +""" + +from __future__ import annotations + +import argparse +import dataclasses +import importlib.util +import json +import re +import sys +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +import httpx +import yaml + +_HERE = Path(__file__).resolve().parent + +_EXPECTATION_KEYS = ( + "expect_anchors", + "expect_labels", + "forbid", + "expect_events", + "expect_dockets", + "min_eras", +) + +_SENT_SPLIT = re.compile(r"(?<=[.!?])\s+") +_BRACKET_RE = re.compile(r"\[([^\]]+)\]") +_DOCKET_YEAR_RE = re.compile(r"(19|20)\d{2}") +_DATE_YEAR_RE = re.compile(r"^(\d{4})") + + +# ── golden set loading ─────────────────────────────────────────── + + +def load_set(path: Path) -> list[dict]: + """Parse and validate the golden YAML set: a top-level list (or + ``{questions: [...]}``) of entries, each with a unique ``id``, a + ``question``, and at least one expectation key.""" + raw = yaml.safe_load(path.read_text()) + if isinstance(raw, dict) and "questions" in raw: + entries = raw["questions"] + elif isinstance(raw, list): + entries = raw + else: + raise ValueError(f"{path}: expected a list or {{questions: [...]}}") + + ids: set[str] = set() + for e in entries: + if not isinstance(e, dict) or "id" not in e or "question" not in e: + raise ValueError(f"{path}: entry missing id/question: {e!r}") + if e["id"] in ids: + raise ValueError(f"{path}: duplicate id {e['id']!r}") + ids.add(e["id"]) + if not any(k in e for k in _EXPECTATION_KEYS): + raise ValueError(f"{path}: entry {e['id']!r} has no expectations") + return entries + + +# ── transcript ─────────────────────────────────────────────────── + + +@dataclass +class Transcript: + events: list[dict] = field(default_factory=list) + + @property + def answer_text(self) -> str: + return "".join( + e.get("text", "") for e in self.events if e.get("type") == "token" + ) + + @property + def sources(self) -> list[dict]: + for e in self.events: + if e.get("type") == "sources": + return e.get("sources", []) + return [] + + @property + def lineage_events(self) -> list[dict]: + for e in self.events: + if e.get("type") == "lineage": + return e.get("events", []) + return [] + + @property + def error(self) -> str | None: + for e in self.events: + if e.get("type") == "error": + return e.get("message", "stream error") + return None + + +@dataclass(frozen=True) +class CheckResult: + name: str + passed: bool + detail: str + + +# ── checks (pure functions over a Transcript) ─────────────────── + + +def _parse_pid_range(spec: Any) -> tuple[int, int]: + if isinstance(spec, (list, tuple)): + return int(spec[0]), int(spec[1]) + return int(spec), int(spec) + + +def check_anchors(transcript: Transcript, expect_anchors: list[dict]) -> CheckResult: + sources = transcript.sources + missing = [] + for a in expect_anchors: + lo, hi = _parse_pid_range(a["p_id"]) + found = False + for s in sources: + if s.get("item_key") != a["item_key"]: + continue + try: + pid = int(s.get("p_id") or "") + except (TypeError, ValueError): + continue + if lo <= pid <= hi: + found = True + break + if not found: + label = ( + f"{a['item_key']} p{lo}" if lo == hi else f"{a['item_key']} p{lo}-{hi}" + ) + missing.append(label) + passed = not missing + detail = "ok" if passed else "missing: " + ", ".join(missing) + return CheckResult("anchors", passed, detail) + + +def check_labels(transcript: Transcript, expect_labels: list[str]) -> CheckResult: + text = transcript.answer_text + missing = [pat for pat in expect_labels if not re.search(pat, text)] + passed = not missing + detail = "ok" if passed else "missing: " + ", ".join(missing) + return CheckResult("labels", passed, detail) + + +def check_forbidden(transcript: Transcript, forbid: list[dict]) -> CheckResult: + sentences = _SENT_SPLIT.split(transcript.answer_text) + violations = [] + for rule in forbid: + pat = re.compile(rule["pattern"]) + unless = rule.get("unless_label") + unless_re = re.compile(unless) if unless else None + for sent in sentences: + if not pat.search(sent): + continue + labels = _BRACKET_RE.findall(sent) + ok = unless_re is not None and any(unless_re.search(l) for l in labels) + if not ok: + violations.append(f"{rule['pattern']!r} in {sent.strip()[:120]!r}") + passed = not violations + detail = "ok" if passed else "; ".join(violations) + return CheckResult("forbidden", passed, detail) + + +def check_events(transcript: Transcript, expect_events: list[dict]) -> CheckResult: + events = transcript.lineage_events + missing = [] + for exp in expect_events: + kinds = set(str(exp["kind"]).split("|")) + code = str(exp["code"]) + year = int(exp["year"]) + found = any( + e.get("code") == code + and e.get("kind") in kinds + and int(e.get("year", -1)) == year + for e in events + ) + if not found: + missing.append(f"{code} {exp['kind']} {year}") + passed = not missing + detail = "ok" if passed else "missing: " + ", ".join(missing) + return CheckResult("events", passed, detail) + + +def check_dockets(transcript: Transcript, expect_dockets: list[str]) -> CheckResult: + have = { + s.get("docket", "") + for s in transcript.sources + if s.get("kind") == "comment" and s.get("docket") + } + missing = [d for d in expect_dockets if d not in have] + passed = not missing + detail = "ok" if passed else "missing: " + ", ".join(missing) + return CheckResult("dockets", passed, detail) + + +def _rule_year(source: dict) -> int: + """The rule year a source belongs to: the docket id's embedded year + for a comment, else the source's own date year — no imports, so this + is a simplification of ``llm.rag.era_of`` (which resolves a + comment's docket to its actual PFS rule year via the bib store).""" + if source.get("kind") == "comment": + m = _DOCKET_YEAR_RE.search(source.get("docket", "") or "") + return int(m.group(0)) if m else 0 + m = _DATE_YEAR_RE.match(source.get("date", "") or "") + return int(m.group(1)) if m else 0 + + +def check_eras(transcript: Transcript, min_eras: int) -> CheckResult: + eras = {_rule_year(s) for s in transcript.sources} + eras.discard(0) + passed = len(eras) >= int(min_eras) + detail = f"eras={sorted(eras)}" + return CheckResult("eras", passed, detail) + + +def evaluate(entry: dict, transcript: Transcript) -> list[CheckResult]: + if transcript.error is not None: + return [CheckResult("stream", False, transcript.error)] + results: list[CheckResult] = [] + if "expect_anchors" in entry: + results.append(check_anchors(transcript, entry["expect_anchors"])) + if "expect_labels" in entry: + results.append(check_labels(transcript, entry["expect_labels"])) + if "forbid" in entry: + results.append(check_forbidden(transcript, entry["forbid"])) + if "expect_events" in entry: + results.append(check_events(transcript, entry["expect_events"])) + if "expect_dockets" in entry: + results.append(check_dockets(transcript, entry["expect_dockets"])) + if "min_eras" in entry: + results.append(check_eras(transcript, entry["min_eras"])) + return results + + +# ── streaming the live /chat endpoint ─────────────────────────── + + +def stream_chat(url: str, question: str, mode: str, timeout: float) -> list[dict]: + """POST ``{question, mode}`` to ``{url}/chat`` and parse the + ``data: {json}\\n\\n`` SSE lines into a list of event dicts.""" + events: list[dict] = [] + with httpx.Client(timeout=timeout) as client: + with client.stream( + "POST", f"{url.rstrip('/')}/chat", json={"question": question, "mode": mode} + ) as resp: + resp.raise_for_status() + for line in resp.iter_lines(): + if not line or not line.startswith("data:"): + continue + payload = line[len("data:") :].strip() + if payload: + events.append(json.loads(payload)) + return events + + +# ── issue filer (imported by path — nb_issue_filer.py sits beside +# this script both in-repo and when docker-cp'd into the container) ── + + +def _load_filer(): + spec = importlib.util.spec_from_file_location( + "nb_issue_filer", _HERE / "nb_issue_filer.py" + ) + assert spec and spec.loader + mod = importlib.util.module_from_spec(spec) + sys.modules["nb_issue_filer"] = mod + spec.loader.exec_module(mod) + return mod + + +# ── runner ─────────────────────────────────────────────────────── + + +def cmd_run(args: argparse.Namespace) -> int: + entries = load_set(Path(args.set_path)) + if args.only: + entries = [e for e in entries if e["id"] == args.only] + if not entries: + print(f"no entry with id {args.only!r} in {args.set_path}", file=sys.stderr) + return 2 + + rows: list[dict] = [] + findings: list[dict] = [] + any_fail = False + + for entry in entries: + qid = entry["id"] + mode = entry.get("mode", "auto") + try: + events = stream_chat(args.url, entry["question"], mode, args.timeout) + results = evaluate(entry, Transcript(events)) + except Exception as e: # noqa: BLE001 — a request failure is a finding, not a crash + results = [CheckResult("request", False, str(e))] + + failing = [r for r in results if not r.passed] + if failing: + any_fail = True + rows.append( + { + "id": qid, + "passed": not failing, + "checks": [dataclasses.asdict(r) for r in results], + } + ) + for r in failing: + findings.append( + { + "notebook": qid, + "ename": r.name, + "evalue": r.detail, + "detail": r.detail, + } + ) + + for row in rows: + status = "PASS" if row["passed"] else "FAIL" + failing_names = ",".join(c["name"] for c in row["checks"] if not c["passed"]) + print(f"{row['id']:32} {status:4} {failing_names}") + + report = { + "rows": rows, + "pass": sum(r["passed"] for r in rows), + "fail": sum(not r["passed"] for r in rows), + } + if args.report: + Path(args.report).write_text(json.dumps(report, indent=2)) + + if args.file_issues: + filer = _load_filer() + filer.cmd_report(findings, args.source) + active = { + filer.signature(f["notebook"], f["ename"], f["evalue"]) for f in findings + } + filer.cmd_sweep(active, args.source) + + return 1 if any_fail else 0 + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="cmd", required=True) + + p_run = sub.add_parser("run", help="stream the golden set against a live /chat") + p_run.add_argument("--url", required=True, help="base URL of the chat service") + p_run.add_argument("--set", required=True, dest="set_path", help="golden YAML path") + p_run.add_argument("--report", help="write a JSON report here") + p_run.add_argument( + "--file-issues", + action="store_true", + help="file/sweep regressions via nb_issue_filer", + ) + p_run.add_argument("--source", default="llm-golden", help="filer 'source' label") + p_run.add_argument("--only", help="run only this entry id") + p_run.add_argument("--timeout", type=float, default=180.0) + + args = parser.parse_args(argv) + if args.cmd == "run": + return cmd_run(args) + parser.error(f"unknown command {args.cmd!r}") + return 2 # pragma: no cover — argparse exits before this + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/dev/test_gen_config_llm_golden.py b/tests/dev/test_gen_config_llm_golden.py new file mode 100644 index 0000000..29c9bb0 --- /dev/null +++ b/tests/dev/test_gen_config_llm_golden.py @@ -0,0 +1,61 @@ +"""Snapshot tests for the generated LLM Golden nightly workflow (P49 +Task 7, refs #692). + +``dev/scripts/backends/gitea.py::_gen_llm_golden`` is modelled on +``_gen_notebooks_integration``: the golden set needs a live chat over +the real pgvector/DuckDB/Ollama stack, which only the ``llm`` container +has, so the runner is docker-cp'd in and run there. These tests guard +the shape of the generated workflow (docker exec against the ``llm`` +container, the nightly cron, the failing-run issue label) without +requiring a live Gitea Actions run. + +``backends`` is a plain (non-src) package under ``dev/scripts/`` — not +importable by its dotted name without that directory on ``sys.path``, +the same way ``dev/scripts/gen_config.py`` relies on being executed +from that directory. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +_DEV_SCRIPTS = Path(__file__).resolve().parents[2] / "dev" / "scripts" +if str(_DEV_SCRIPTS) not in sys.path: + sys.path.insert(0, str(_DEV_SCRIPTS)) + +from backends import gitea # noqa: E402 + + +def test_gen_llm_golden_path_and_shape(): + path, content = gitea._gen_llm_golden("ubuntu-latest", "latest") + assert path == ".gitea/workflows/llm-golden.yml" + assert 'cron: "40 3 * * *"' in content + assert "docker exec llm mkdir -p /tmp/golden" in content + assert "docker cp dev/scripts/llm_golden.py llm:/tmp/golden/" in content + assert "docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/" in content + assert "docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/" in content + assert "uv run --project /app python /tmp/golden/llm_golden.py run" in content + assert "--url http://localhost:8000" in content + assert "--file-issues --source nightly-llm-golden" in content + assert "NB_ISSUE_LABEL=llm" in content + assert "GITEA_API_BASE=http://git:3000/api/v1" in content + assert "secrets.DEPLOY_TOKEN" in content + + +def test_gen_llm_golden_registered_in_emit(): + files = gitea.emit( + [], + [], + { + "registry": "reg.example", + "org": "homelab", + "image_prefix": "stack", + "domain": "example.test", + "host_ip": "127.0.0.1", + "repo": "homelab/stack", + }, + {}, + ) + assert ".gitea/workflows/llm-golden.yml" in files + assert "docker exec llm" in files[".gitea/workflows/llm-golden.yml"] diff --git a/tests/llm/golden_lineage.yaml b/tests/llm/golden_lineage.yaml new file mode 100644 index 0000000..9d499d8 --- /dev/null +++ b/tests/llm/golden_lineage.yaml @@ -0,0 +1,104 @@ +# Golden longitudinal evaluation set (P49 Task 7, refs #692). +# +# Anchors verified against the corpus in #692. Run with: +# uv run python dev/scripts/llm_golden.py run --url \ +# --set tests/llm/golden_lineage.yaml --report report.json +# +# Schema per entry: +# id unique string +# question the question sent to POST /chat +# mode "auto" (default), "timeline" or "recent" +# expect_anchors [{item_key, p_id}] p_id may be [lo, hi] for a range; +# each anchor must appear among the streamed `sources` +# expect_labels [regex] each must be matched somewhere in the +# concatenated `token` text (the cited answer prose) +# forbid [{pattern, unless_label}] `pattern` may appear only +# in a sentence that also cites a bracketed label +# matching `unless_label` +# expect_events [{code, kind, year}] kind may be "a|b" alternatives; +# must appear in the `lineage` event payload +# expect_dockets [docket_id] must appear among comment-kind sources +# min_eras distinct rule years among sources (rule/corpus: the +# source's date year; comment: its docket id's year) + +- id: ccm-history + question: What is the history of CCM coding and payment? + mode: timeline + expect_anchors: + - item_key: DE2VH9PD + p_id: [1249, 1251] + - item_key: YBM4IZUS + p_id: 1578 + - item_key: JJ6AM5HJ + p_id: 1163 + expect_events: + - code: "99490" + kind: created + year: 2015 + - code: G2058 + kind: replaced_by + year: 2021 + - code: G0556 + kind: created + year: 2025 + min_eras: 4 + +- id: g2058-replacement + question: What replaced G2058, and why? + mode: timeline + expect_anchors: + - item_key: YBM4IZUS + p_id: 1578 + - item_key: YBM4IZUS + p_id: 2369 + expect_events: + - code: G2058 + kind: replaced_by + year: 2021 + +- id: apcm-vs-ccm-elements + question: How do APCM's elements differ from CCM's? + expect_anchors: + - item_key: JJ6AM5HJ + p_id: [1164, 1185] + - item_key: DE2VH9PD + p_id: [1245, 1247] + +- id: audio-only-em-99441 + question: When were audio-only E/M codes payable, and why did 99441-99443 end? + mode: timeline + expect_anchors: + - item_key: XFGGRBDH + p_id: 489 + expect_events: + - code: "99441" + kind: deleted|disappeared|cpt_deleted + year: 2025 + +- id: g2211-commenters-2023-vs-2025 + question: What did commenters say about G2211 in 2023 versus 2025? + mode: timeline + expect_dockets: + - CMS-2023-0121 + - CMS-2025-0304 + +- id: 99490-telehealth-steps + question: Does 99490 pass telehealth Steps 1 through 3? + expect_anchors: + - item_key: 2KVJ2HKX + p_id: 394 + - item_key: 2KVJ2HKX + p_id: 396 + - item_key: 2KVJ2HKX + p_id: 398 + +- id: g2064-g2065-to-99424-99426-rate + question: How did G2064/G2065 becoming 99424/99426 change the RHC/FQHC G0511 rate? + mode: timeline + expect_anchors: + - item_key: JE7KYBW3 + p_id: 1100 + - item_key: JE7KYBW3 + p_id: 1111 + - item_key: MZ24MX5S + p_id: 1305 diff --git a/tests/llm/test_golden.py b/tests/llm/test_golden.py new file mode 100644 index 0000000..e006b02 --- /dev/null +++ b/tests/llm/test_golden.py @@ -0,0 +1,269 @@ +"""Tests for dev/scripts/llm_golden.py — the golden longitudinal chat +evaluation (P49 Task 7, refs #692). + +Loaded by path like the other dev/scripts tests (tests/dev/test_nb_ +issue_filer.py) since dev/scripts/ is not a package. Checkers are +exercised against canned SSE transcripts (pass and fail cases); the +YAML set is validated for shape; a live end-to-end test runs entry +"g2058-replacement" against a real /chat and is skipped unless +LLM_CHAT_URL is set. +""" + +from __future__ import annotations + +import importlib.util +import os +import sys +from pathlib import Path + +import pytest + +_SCRIPT = Path(__file__).resolve().parents[2] / "dev" / "scripts" / "llm_golden.py" +_spec = importlib.util.spec_from_file_location("_llm_golden", _SCRIPT) +assert _spec and _spec.loader +golden = importlib.util.module_from_spec(_spec) +sys.modules["_llm_golden"] = golden +_spec.loader.exec_module(golden) + +Transcript = golden.Transcript +GOLDEN_SET = Path(__file__).resolve().parent / "golden_lineage.yaml" + + +def _sources(*rows: dict) -> dict: + return {"type": "sources", "sources": list(rows)} + + +def _tokens(*texts: str) -> list[dict]: + return [{"type": "token", "text": t} for t in texts] + + +def _lineage(*events: dict) -> dict: + return {"type": "lineage", "events": list(events)} + + +# ── check_anchors ──────────────────────────────────────────────── + + +def test_check_anchors_pass_exact_and_range(): + t = Transcript( + [ + _sources( + {"item_key": "YBM4IZUS", "p_id": "1578"}, + {"item_key": "DE2VH9PD", "p_id": "1250"}, + ) + ] + ) + result = golden.check_anchors( + t, + [ + {"item_key": "YBM4IZUS", "p_id": 1578}, + {"item_key": "DE2VH9PD", "p_id": [1249, 1251]}, + ], + ) + assert result.passed, result.detail + + +def test_check_anchors_fail_missing(): + t = Transcript([_sources({"item_key": "YBM4IZUS", "p_id": "1578"})]) + result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": 1250}]) + assert not result.passed + assert "DE2VH9PD" in result.detail + + +def test_check_anchors_fail_pid_outside_range(): + t = Transcript([_sources({"item_key": "DE2VH9PD", "p_id": "1300"})]) + result = golden.check_anchors(t, [{"item_key": "DE2VH9PD", "p_id": [1249, 1251]}]) + assert not result.passed + + +# ── check_labels ───────────────────────────────────────────────── + + +def test_check_labels_pass(): + t = Transcript( + _tokens("As discussed in ", "[CY2021 PFS final ¶1578], G2058 was...") + ) + result = golden.check_labels(t, [r"CY2021 PFS final ¶1578"]) + assert result.passed, result.detail + + +def test_check_labels_fail(): + t = Transcript(_tokens("No citations here.")) + result = golden.check_labels(t, [r"CY2021 PFS final"]) + assert not result.passed + assert "CY2021 PFS final" in result.detail + + +# ── check_forbidden ────────────────────────────────────────────── + + +def test_check_forbidden_pass_when_labeled(): + t = Transcript( + _tokens("The payment is $34.85 [PFS CY2026 Addendum B]. ", "That is all.") + ) + result = golden.check_forbidden( + t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}] + ) + assert result.passed, result.detail + + +def test_check_forbidden_fail_unlabeled_dollar(): + t = Transcript(_tokens("The payment is roughly $34.85 based on recent rules.")) + result = golden.check_forbidden( + t, [{"pattern": r"\$\d", "unless_label": "Addendum B"}] + ) + assert not result.passed + assert "$" in result.detail or "\\$" in result.detail + + +# ── check_events ───────────────────────────────────────────────── + + +def test_check_events_pass_alternatives(): + t = Transcript([_lineage({"code": "99441", "kind": "disappeared", "year": 2025})]) + result = golden.check_events( + t, [{"code": "99441", "kind": "deleted|disappeared|cpt_deleted", "year": 2025}] + ) + assert result.passed, result.detail + + +def test_check_events_fail_wrong_year(): + t = Transcript([_lineage({"code": "G2058", "kind": "replaced_by", "year": 2020})]) + result = golden.check_events( + t, [{"code": "G2058", "kind": "replaced_by", "year": 2021}] + ) + assert not result.passed + + +# ── check_dockets ──────────────────────────────────────────────── + + +def test_check_dockets_pass(): + t = Transcript( + [ + _sources( + {"kind": "comment", "docket": "CMS-2023-0121"}, + {"kind": "comment", "docket": "CMS-2025-0304"}, + ) + ] + ) + result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"]) + assert result.passed, result.detail + + +def test_check_dockets_fail_missing_one(): + t = Transcript([_sources({"kind": "comment", "docket": "CMS-2023-0121"})]) + result = golden.check_dockets(t, ["CMS-2023-0121", "CMS-2025-0304"]) + assert not result.passed + assert "CMS-2025-0304" in result.detail + + +# ── check_eras ─────────────────────────────────────────────────── + + +def test_check_eras_pass(): + t = Transcript( + [ + _sources( + {"kind": "rule", "date": "2020-11-01"}, + {"kind": "rule", "date": "2015-11-01"}, + {"kind": "comment", "docket": "CMS-2023-0121"}, + {"kind": "corpus", "date": "2025-01-01"}, + ) + ] + ) + result = golden.check_eras(t, 4) + assert result.passed, result.detail + + +def test_check_eras_fail_too_few(): + t = Transcript( + [ + _sources( + {"kind": "rule", "date": "2020-11-01"}, + {"kind": "rule", "date": "2020-12-01"}, + ) + ] + ) + result = golden.check_eras(t, 2) + assert not result.passed + + +def test_check_eras_ignores_undated(): + t = Transcript([_sources({"kind": "corpus", "date": ""})]) + result = golden.check_eras(t, 1) + assert not result.passed + + +# ── evaluate: stream error short-circuits ─────────────────────── + + +def test_evaluate_short_circuits_on_error_event(): + t = Transcript([{"type": "error", "message": "boom"}]) + results = golden.evaluate({"expect_labels": ["x"]}, t) + assert len(results) == 1 + assert results[0].name == "stream" + assert not results[0].passed + assert "boom" in results[0].detail + + +# ── YAML shape ─────────────────────────────────────────────────── + + +def test_golden_set_loads_and_is_well_formed(): + entries = golden.load_set(GOLDEN_SET) + assert len(entries) == 7 + ids = [e["id"] for e in entries] + assert len(ids) == len(set(ids)), "duplicate ids" + for e in entries: + assert "id" in e and "question" in e + assert any(k in e for k in golden._EXPECTATION_KEYS), e["id"] + + +def test_golden_set_expected_ids_present(): + entries = {e["id"] for e in golden.load_set(GOLDEN_SET)} + assert entries == { + "ccm-history", + "g2058-replacement", + "apcm-vs-ccm-elements", + "audio-only-em-99441", + "g2211-commenters-2023-vs-2025", + "99490-telehealth-steps", + "g2064-g2065-to-99424-99426-rate", + } + + +def test_load_set_rejects_duplicate_ids(tmp_path): + bad = tmp_path / "bad.yaml" + bad.write_text( + "- id: a\n question: q1\n min_eras: 1\n" + "- id: a\n question: q2\n min_eras: 1\n" + ) + with pytest.raises(ValueError, match="duplicate id"): + golden.load_set(bad) + + +def test_load_set_rejects_entry_without_expectations(tmp_path): + bad = tmp_path / "bad.yaml" + bad.write_text("- id: a\n question: q1\n") + with pytest.raises(ValueError, match="no expectations"): + golden.load_set(bad) + + +# ── live (opt-in) ──────────────────────────────────────────────── + + +@pytest.mark.skipif( + not os.environ.get("LLM_CHAT_URL"), + reason="LLM_CHAT_URL not set — skipping live chat test", +) +def test_live_g2058_replacement(): + url = os.environ["LLM_CHAT_URL"] + entries = {e["id"]: e for e in golden.load_set(GOLDEN_SET)} + entry = entries["g2058-replacement"] + events = golden.stream_chat( + url, entry["question"], entry.get("mode", "auto"), 180.0 + ) + results = golden.evaluate(entry, Transcript(events)) + failing = [r for r in results if not r.passed] + assert not failing, [(r.name, r.detail) for r in failing]