← Files ConductorARCHIVED FILE
.github/workflows/evals.yml
5.55 KB · Oct 3, 2026 · 06:32 UTC
name: Skill Evals
# Cost-aware triggers:
# - workflow_dispatch — on-demand, pick any model
# - schedule — weekly regression check against the default model
# - pull_request — only when labeled `run-evals` (not on every PR)
# - push to main — only when skill content or evals themselves change
#
# Required repository secrets:
# ANTHROPIC_API_KEY — agent + default judge (required)
# OPENAI_API_KEY — only needed when running against gpt-* models
# GEMINI_API_KEY — only needed when running against gemini-* models
on:
workflow_dispatch:
inputs:
model:
description: 'Agent model (e.g. claude-sonnet-4-6, gpt-5.4, gemini-3-flash-preview)'
required: false
default: 'claude-sonnet-4-6'
judge:
description: 'Judge model'
required: false
default: 'claude-sonnet-5'
schedule:
# Sundays 08:00 UTC — weekly regression check
- cron: '0 8 * * 0'
pull_request:
types: [labeled]
push:
branches: [main]
paths:
- 'skills/**'
- 'commands/**'
- 'evaluations/**'
- 'scripts/run_evals.py'
- 'scripts/render_evals_html.py'
- '.github/workflows/evals.yml'
jobs:
evals:
# PR runs are opt-in via the `run-evals` label. Other triggers always run.
if: github.event_name != 'pull_request' || github.event.label.name == 'run-evals'
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.11'
- name: Resolve model + judge
id: cfg
run: |
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
echo "model=${{ inputs.model }}" >> "$GITHUB_OUTPUT"
echo "judge=${{ inputs.judge }}" >> "$GITHUB_OUTPUT"
else
echo "model=claude-sonnet-4-6" >> "$GITHUB_OUTPUT"
echo "judge=claude-sonnet-5" >> "$GITHUB_OUTPUT"
fi
- name: Validate plugin (must pass before spending API tokens)
run: python3 scripts/validate_plugin.py
- name: Run evals
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
run: |
mkdir -p eval-output
python3 scripts/run_evals.py \
--model "${{ steps.cfg.outputs.model }}" \
--judge-model "${{ steps.cfg.outputs.judge }}" \
--json \
--output eval-output/report.json
- name: Render HTML report
if: always()
run: |
if [ -f eval-output/report.json ]; then
python3 scripts/render_evals_html.py \
eval-output/report.json \
-o eval-output/report.html \
--title "Conductor Skill Evals — ${{ steps.cfg.outputs.model }}"
fi
- name: Upload report artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: eval-report-${{ steps.cfg.outputs.model }}
path: eval-output/
retention-days: 30
- name: Comment on PR with summary
if: github.event_name == 'pull_request' && always()
uses: actions/github-script@v7
with:
script: |
const fs = require('fs');
if (!fs.existsSync('eval-output/report.json')) {
await github.rest.issues.createComment({
issue_number: context.issue.number,
owner: context.repo.owner,
repo: context.repo.repo,
body: '## Eval Results\n\n:x: The eval run failed before producing a report. Check the workflow logs.'
});
return;
}
const report = JSON.parse(fs.readFileSync('eval-output/report.json'));
const results = report.results || [];
const passed = results.filter(r => r.overall_pass).length;
const total = results.length;
const critPass = results.reduce((s, r) => s + (r.passed || 0), 0);
const critTotal = results.reduce((s, r) => s + (r.total || 0), 0);
const pct = critTotal ? (critPass / critTotal * 100).toFixed(1) : '0.0';
const allPassed = passed === total;
let body = `## Eval Results\n\n`;
body += `**Model:** \`${report.model}\` · **Judge:** \`${report.judge_model}\`\n\n`;
body += `- Evals: **${passed}/${total}** ${allPassed ? ':white_check_mark:' : ':warning:'}\n`;
body += `- Criteria: **${critPass}/${critTotal}** (${pct}%)\n\n`;
const fails = results.filter(r => !r.overall_pass);
if (fails.length) {
body += `### Failed scenarios\n\n`;
for (const f of fails) {
body += `- \`${f.name}\` — ${f.passed}/${f.total}\n`;
}
}
const partials = results.filter(r => r.overall_pass && (r.passed || 0) < (r.total || 0));
if (partials.length) {
body += `### Partial-pass deltas\n\n`;
for (const p of partials) {
body += `- \`${p.name}\` — ${p.passed}/${p.total}\n`;
}
}
body += `\nDownload \`eval-report-${report.model}\` artifact for the full HTML report.\n`;
await github.rest.issues.createComment({
issue_number: context.issue.number,
owner: context.repo.owner,
repo: context.repo.repo,
body
});
SHA-256: 1b84d9fb43217d2f29bc67d3999c3f725212430f842ea321bbcf0ed8cb6ab326