Files
stack/.gitea/workflows/llm-golden.yml

56 lines
1.8 KiB
YAML

# DO NOT EDIT — generated by gen_config.py from stack.toml
# Re-generate: uv run python dev/scripts/gen_config.py
name: LLM Golden
on:
workflow_dispatch:
schedule:
- cron: "40 3 * * *"
jobs:
llm-golden:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: https://github.com/actions/checkout@v4
- name: Set up uv
run: curl -LsSf https://astral.sh/uv/install.sh | sh
env:
UV_INSTALL_DIR: /usr/local/bin
- name: Run golden longitudinal evaluation against the live chat
env:
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
run: |
set -euo pipefail
docker exec llm mkdir -p /tmp/golden
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
docker exec \
-e GITEA_TOKEN \
-e GITEA_API_BASE=http://git:3000/api/v1 \
-e NB_ISSUE_LABEL=llm \
llm \
uv run --project /app python /tmp/golden/llm_golden.py run \
--url http://localhost:8000 \
--set /tmp/golden/golden_lineage.yaml \
--report /tmp/golden/report.json \
--file-issues --source nightly-llm-golden
docker cp llm:/tmp/golden/report.json llm-golden-report.json
cat llm-golden-report.json
- name: File failure issue
if: failure()
env:
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
run: |
uv sync --no-dev --quiet 2>/dev/null || true
uv run python -m api.diag.ci \
--workflow "LLM Golden" --job "llm-golden" \
--run "${{ github.run_number }}" \
--sha "${{ github.sha }}" \
--ref "${{ github.ref }}" || true