← Files ConductorARCHIVED FILE

.github/workflows/evals.yml

5.55 KB · Oct 3, 2026 · 06:32 UTC

↓ Download file

name: Skill Evals

# Cost-aware triggers:
#   - workflow_dispatch — on-demand, pick any model
#   - schedule — weekly regression check against the default model
#   - pull_request — only when labeled `run-evals` (not on every PR)
#   - push to main — only when skill content or evals themselves change
#
# Required repository secrets:
#   ANTHROPIC_API_KEY  — agent + default judge (required)
#   OPENAI_API_KEY     — only needed when running against gpt-* models
#   GEMINI_API_KEY     — only needed when running against gemini-* models

on:
  workflow_dispatch:
    inputs:
      model:
        description: 'Agent model (e.g. claude-sonnet-4-6, gpt-5.4, gemini-3-flash-preview)'
        required: false
        default: 'claude-sonnet-4-6'
      judge:
        description: 'Judge model'
        required: false
        default: 'claude-sonnet-5'

  schedule:
    # Sundays 08:00 UTC — weekly regression check
    - cron: '0 8 * * 0'

  pull_request:
    types: [labeled]

  push:
    branches: [main]
    paths:
      - 'skills/**'
      - 'commands/**'
      - 'evaluations/**'
      - 'scripts/run_evals.py'
      - 'scripts/render_evals_html.py'
      - '.github/workflows/evals.yml'

jobs:
  evals:
    # PR runs are opt-in via the `run-evals` label. Other triggers always run.
    if: github.event_name != 'pull_request' || github.event.label.name == 'run-evals'
    runs-on: ubuntu-latest
    timeout-minutes: 30

    steps:
      - uses: actions/checkout@v4

      - name: Set up Python
        uses: actions/setup-python@v5
        with:
          python-version: '3.11'

      - name: Resolve model + judge
        id: cfg
        run: |
          if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
            echo "model=${{ inputs.model }}"   >> "$GITHUB_OUTPUT"
            echo "judge=${{ inputs.judge }}"   >> "$GITHUB_OUTPUT"
          else
            echo "model=claude-sonnet-4-6"               >> "$GITHUB_OUTPUT"
            echo "judge=claude-sonnet-5"        >> "$GITHUB_OUTPUT"
          fi

      - name: Validate plugin (must pass before spending API tokens)
        run: python3 scripts/validate_plugin.py

      - name: Run evals
        env:
          ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
          OPENAI_API_KEY:    ${{ secrets.OPENAI_API_KEY }}
          GEMINI_API_KEY:    ${{ secrets.GEMINI_API_KEY }}
        run: |
          mkdir -p eval-output
          python3 scripts/run_evals.py \
            --model "${{ steps.cfg.outputs.model }}" \
            --judge-model "${{ steps.cfg.outputs.judge }}" \
            --json \
            --output eval-output/report.json

      - name: Render HTML report
        if: always()
        run: |
          if [ -f eval-output/report.json ]; then
            python3 scripts/render_evals_html.py \
              eval-output/report.json \
              -o eval-output/report.html \
              --title "Conductor Skill Evals — ${{ steps.cfg.outputs.model }}"
          fi

      - name: Upload report artifacts
        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: eval-report-${{ steps.cfg.outputs.model }}
          path: eval-output/
          retention-days: 30

      - name: Comment on PR with summary
        if: github.event_name == 'pull_request' && always()
        uses: actions/github-script@v7
        with:
          script: |
            const fs = require('fs');
            if (!fs.existsSync('eval-output/report.json')) {
              await github.rest.issues.createComment({
                issue_number: context.issue.number,
                owner: context.repo.owner,
                repo: context.repo.repo,
                body: '## Eval Results\n\n:x: The eval run failed before producing a report. Check the workflow logs.'
              });
              return;
            }
            const report = JSON.parse(fs.readFileSync('eval-output/report.json'));
            const results = report.results || [];
            const passed = results.filter(r => r.overall_pass).length;
            const total = results.length;
            const critPass = results.reduce((s, r) => s + (r.passed || 0), 0);
            const critTotal = results.reduce((s, r) => s + (r.total || 0), 0);
            const pct = critTotal ? (critPass / critTotal * 100).toFixed(1) : '0.0';
            const allPassed = passed === total;

            let body = `## Eval Results\n\n`;
            body += `**Model:** \`${report.model}\` · **Judge:** \`${report.judge_model}\`\n\n`;
            body += `- Evals: **${passed}/${total}** ${allPassed ? ':white_check_mark:' : ':warning:'}\n`;
            body += `- Criteria: **${critPass}/${critTotal}** (${pct}%)\n\n`;

            const fails = results.filter(r => !r.overall_pass);
            if (fails.length) {
              body += `### Failed scenarios\n\n`;
              for (const f of fails) {
                body += `- \`${f.name}\` — ${f.passed}/${f.total}\n`;
              }
            }

            const partials = results.filter(r => r.overall_pass && (r.passed || 0) < (r.total || 0));
            if (partials.length) {
              body += `### Partial-pass deltas\n\n`;
              for (const p of partials) {
                body += `- \`${p.name}\` — ${p.passed}/${p.total}\n`;
              }
            }

            body += `\nDownload \`eval-report-${report.model}\` artifact for the full HTML report.\n`;

            await github.rest.issues.createComment({
              issue_number: context.issue.number,
              owner: context.repo.owner,
              repo: context.repo.repo,
              body
            });

SHA-256: 1b84d9fb43217d2f29bc67d3999c3f725212430f842ea321bbcf0ed8cb6ab326