56 lines
1.8 KiB
YAML
56 lines
1.8 KiB
YAML
# DO NOT EDIT — generated by gen_config.py from stack.toml
|
|
# Re-generate: uv run python dev/scripts/gen_config.py
|
|
|
|
name: LLM Golden
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
schedule:
|
|
- cron: "40 3 * * *"
|
|
|
|
jobs:
|
|
llm-golden:
|
|
runs-on: ubuntu-latest
|
|
steps:
|
|
- name: Checkout
|
|
uses: https://github.com/actions/checkout@v4
|
|
|
|
- name: Set up uv
|
|
run: curl -LsSf https://astral.sh/uv/install.sh | sh
|
|
env:
|
|
UV_INSTALL_DIR: /usr/local/bin
|
|
|
|
- name: Run golden longitudinal evaluation against the live chat
|
|
env:
|
|
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
|
run: |
|
|
set -euo pipefail
|
|
docker exec llm mkdir -p /tmp/golden
|
|
docker cp dev/scripts/llm_golden.py llm:/tmp/golden/
|
|
docker cp dev/scripts/nb_issue_filer.py llm:/tmp/golden/
|
|
docker cp tests/llm/golden_lineage.yaml llm:/tmp/golden/
|
|
docker exec \
|
|
-e GITEA_TOKEN \
|
|
-e GITEA_API_BASE=http://git:3000/api/v1 \
|
|
-e NB_ISSUE_LABEL=llm \
|
|
llm \
|
|
uv run --project /app python /tmp/golden/llm_golden.py run \
|
|
--url http://localhost:8000 \
|
|
--set /tmp/golden/golden_lineage.yaml \
|
|
--report /tmp/golden/report.json \
|
|
--file-issues --source nightly-llm-golden
|
|
docker cp llm:/tmp/golden/report.json llm-golden-report.json
|
|
cat llm-golden-report.json
|
|
|
|
- name: File failure issue
|
|
if: failure()
|
|
env:
|
|
GITEA_TOKEN: ${{ secrets.DEPLOY_TOKEN }}
|
|
run: |
|
|
uv sync --no-dev --quiet 2>/dev/null || true
|
|
uv run python -m api.diag.ci \
|
|
--workflow "LLM Golden" --job "llm-golden" \
|
|
--run "${{ github.run_number }}" \
|
|
--sha "${{ github.sha }}" \
|
|
--ref "${{ github.ref }}" || true
|