EVOKORE// WORKFLOWS / benchmarks-weekly.yml
Agent-33workflow · 206 lines
.github/workflows/benchmarks-weekly.yml
CI / automation
View on GitHub →# REFERENCE-ONLY: GitHub Actions secrets referenced below are illustrative.
# Do not inject real credentials. See your CI/CD provider for proper secret management.
# Weekly SkillsBench full-run evaluation (POST-2.1b)
# - Runs agent33 bench run against the full 86-task SkillsBench suite
# - Detects >5pp regression vs committed baseline; auto-opens GitHub issue
# - Commits results to the benchmarks branch (NOT main)
# Node.js 24 migration: actions/checkout@v6, actions/setup-python@v6,
# actions/upload-artifact@v7, actions/github-script@v9
name: Benchmarks Weekly
on:
schedule:
- cron: "0 0 * * 0" # Every Sunday at 00:00 UTC
workflow_dispatch:
inputs:
model:
description: "LLM model to use for evaluation"
required: false
default: "llama3.2"
skillsbench_ref:
description: "SkillsBench repo ref to checkout (tag or SHA)"
required: false
default: "main"
permissions:
contents: write # push to benchmarks branch
issues: write # open regression issues
jobs:
bench-full-run:
runs-on: ubuntu-latest
timeout-minutes: 180
steps:
- name: Checkout main
uses: actions/checkout@v6
with:
fetch-depth: 0 # needed to push to benchmarks branch
token: ${{ secrets.GITHUB_TOKEN }}
- uses: actions/setup-python@v6
with:
python-version: "3.11"
- name: Install agent33
working-directory: engine
env:
AGENT33_SKIP_FRONTEND_BUILD: "1"
run: pip install -e ".[dev]"
- name: Checkout SkillsBench
uses: actions/checkout@v6
with:
repository: benchflow-ai/skillsbench
ref: ${{ inputs.skillsbench_ref || 'main' }}
path: skillsbench
- name: Fetch current baseline
run: |
git fetch origin benchmarks:refs/remotes/origin/benchmarks || true
mkdir -p bench-results
if git show origin/benchmarks:baselines/ctrf-baseline-latest.json > bench-results/ctrf-baseline.json 2>/dev/null; then
echo "HAVE_BASELINE=true" >> "$GITHUB_ENV"
echo "Baseline fetched from benchmarks branch"
else
echo "HAVE_BASELINE=false" >> "$GITHUB_ENV"
echo "No existing baseline -- this run will create it"
fi
- name: Run full SkillsBench suite
env:
AGENT33_SKIP_FRONTEND_BUILD: "1"
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
OLLAMA_BASE_URL: ${{ secrets.OLLAMA_BASE_URL }}
working-directory: engine
run: |
DATE=$(date +%Y-%m-%d)
MODEL="${{ inputs.model || 'llama3.2' }}"
agent33 bench run \
--skillsbench-root ../skillsbench \
--output ../bench-results/ctrf-run-${DATE}.json \
--model "${MODEL}"
cp ../bench-results/ctrf-run-${DATE}.json ../bench-results/ctrf-run-latest.json
- name: Summarize SkillsBench report
if: always()
working-directory: engine
run: |
if [ -f "../bench-results/ctrf-run-latest.json" ]; then
if [ "${HAVE_BASELINE}" = "true" ]; then
agent33 bench report \
../bench-results/ctrf-run-latest.json \
--baseline ../bench-results/ctrf-baseline.json \
--github-step-summary
else
agent33 bench report \
../bench-results/ctrf-run-latest.json \
--github-step-summary
fi
fi
- name: Detect regression vs baseline
id: regression
if: env.HAVE_BASELINE == 'true'
run: |
python3 - <<'PYEOF'
import json, os, sys
def pass_rate(d):
s = d.get("results", {}).get("summary", {})
total = s.get("tests", 0)
passed = s.get("passed", 0)
return (passed / total * 100) if total > 0 else 0.0
with open("bench-results/ctrf-run-latest.json") as f:
current = json.load(f)
with open("bench-results/ctrf-baseline.json") as f:
baseline = json.load(f)
cur_rate = pass_rate(current)
base_rate = pass_rate(baseline)
drop = base_rate - cur_rate
print(f"Current pass rate: {cur_rate:.1f}%")
print(f"Baseline pass rate: {base_rate:.1f}%")
print(f"Drop: {drop:.1f}pp")
with open(os.environ["GITHUB_OUTPUT"], "a") as fh:
if drop > 5.0:
print(f"REGRESSION DETECTED: {drop:.1f}pp drop exceeds 5pp threshold")
fh.write(f"regression=true\n")
fh.write(f"drop={drop:.1f}\n")
fh.write(f"current_rate={cur_rate:.1f}\n")
fh.write(f"baseline_rate={base_rate:.1f}\n")
else:
fh.write(f"regression=false\n")
fh.write(f"drop={drop:.1f}\n")
PYEOF
- name: Open regression issue
if: steps.regression.outputs.regression == 'true'
uses: actions/github-script@v9
with:
script: |
const drop = '${{ steps.regression.outputs.drop }}';
const current = '${{ steps.regression.outputs.current_rate }}';
const baseline = '${{ steps.regression.outputs.baseline_rate }}';
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`;
await github.rest.issues.create({
owner: context.repo.owner,
repo: context.repo.repo,
title: `[SkillsBench Regression] Pass rate dropped ${drop}pp vs baseline`,
labels: ['benchmark-regression', 'automated'],
body: [
'## SkillsBench Regression Detected',
'',
`| Metric | Value |`,
`|---|---|`,
`| Current pass rate | ${current}% |`,
`| Baseline pass rate | ${baseline}% |`,
`| Drop | **${drop}pp** (threshold: 5pp) |`,
`| Workflow run | [View run](${runUrl}) |`,
'',
'## Next Steps',
'1. Review the full CTRF report artifact attached to the workflow run',
'2. Identify which task categories regressed',
'3. Open a fix PR or update the baseline if the regression is expected',
'',
'_Opened automatically by the Benchmarks Weekly workflow._',
].join('\n'),
});
- name: Commit results to benchmarks branch
run: |
DATE=$(date +%Y-%m-%d)
RUN_SHA=$(git rev-parse --short HEAD)
# Configure git identity for bot commit
git config user.name "agent33-bot"
git config user.email "agent33-bot@users.noreply.github.com"
# Checkout benchmarks branch (orphan-safe)
git fetch origin benchmarks
git checkout benchmarks
# Copy results
mkdir -p baselines
cp bench-results/ctrf-run-latest.json baselines/ctrf-baseline-latest.json
cp bench-results/ctrf-run-latest.json baselines/ctrf-baseline-${DATE}.json
# Commit and push (skip if nothing changed)
git add baselines/
git diff --cached --quiet || git commit \
-m "benchmark(weekly): SkillsBench run ${DATE} from ${RUN_SHA}" \
--no-verify
git push origin benchmarks
- name: Upload artifacts
if: always()
uses: actions/upload-artifact@v7
with:
name: skillsbench-weekly-results
path: bench-results/
retention-days: 90