← Files ConductorARCHIVED FILE

.github/workflows/eval-compare.yml

8.26 KB · Oct 3, 2026 · 06:32 UTC

↓ Download file

name: Skill Evals — Multi-Model Compare

# Runs the eval suite against three models in parallel (matrix) and produces
# a single side-by-side HTML comparison report. Heavier and pricier than
# evals.yml — opt-in only.
#
# Cost per matrix run (3 models × 19 scenarios): roughly $6–7.
#
# Triggers:
#   - workflow_dispatch — pick which models, override defaults
#   - pull_request labeled `run-eval-compare`
#
# Required repository secrets:
#   ANTHROPIC_API_KEY  — agent + judge (always)
#   OPENAI_API_KEY     — agent (for gpt-* models)
#   GEMINI_API_KEY     — agent (for gemini-* models)

on:
  workflow_dispatch:
    inputs:
      models:
        description: 'Comma-separated agent models (default: claude-sonnet-4-6, gpt-5.4, gemini-3-flash-preview)'
        required: false
        default: 'claude-sonnet-4-6,gpt-5.4,gemini-3-flash-preview'
      judge:
        description: 'Judge model'
        required: false
        default: 'claude-sonnet-5'

  pull_request:
    types: [labeled]

jobs:
  # --------------------------------------------------------------------
  # Resolve the model list once so both matrix + aggregator agree.
  # --------------------------------------------------------------------
  plan:
    if: github.event_name != 'pull_request' || github.event.label.name == 'run-eval-compare'
    runs-on: ubuntu-latest
    outputs:
      models_json: ${{ steps.parse.outputs.models_json }}
      judge: ${{ steps.parse.outputs.judge }}
    steps:
      - name: Parse model list
        id: parse
        run: |
          if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
            MODELS="${{ inputs.models }}"
            JUDGE="${{ inputs.judge }}"
          else
            MODELS="claude-sonnet-4-6,gpt-5.4,gemini-3-flash-preview"
            JUDGE="claude-sonnet-5"
          fi
          # Convert "a,b,c" → '["a","b","c"]' for matrix consumption
          JSON=$(echo "$MODELS" | python3 -c "import sys,json; print(json.dumps([m.strip() for m in sys.stdin.read().split(',') if m.strip()]))")
          echo "models_json=$JSON" >> "$GITHUB_OUTPUT"
          echo "judge=$JUDGE" >> "$GITHUB_OUTPUT"
          echo "Will run: $JSON"
          echo "Judge: $JUDGE"

  # --------------------------------------------------------------------
  # Matrix: one job per model, fail-isolated so one model's API issue
  # doesn't kill the others.
  # --------------------------------------------------------------------
  evals:
    needs: plan
    runs-on: ubuntu-latest
    timeout-minutes: 30
    strategy:
      fail-fast: false
      matrix:
        model: ${{ fromJson(needs.plan.outputs.models_json) }}
    steps:
      - uses: actions/checkout@v4

      - name: Set up Python
        uses: actions/setup-python@v5
        with:
          python-version: '3.11'

      - name: Validate plugin (cheap gate before spending API tokens)
        run: python3 scripts/validate_plugin.py

      - name: Run evals — ${{ matrix.model }}
        env:
          ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
          OPENAI_API_KEY:    ${{ secrets.OPENAI_API_KEY }}
          GEMINI_API_KEY:    ${{ secrets.GEMINI_API_KEY }}
        run: |
          mkdir -p out
          # Sanitize model name for filenames
          SAFE=$(echo "${{ matrix.model }}" | tr '/' '-' | tr ':' '-')
          python3 scripts/run_evals.py \
            --model "${{ matrix.model }}" \
            --judge-model "${{ needs.plan.outputs.judge }}" \
            --json \
            --output "out/report-${SAFE}.json"

      - name: Upload per-model report
        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: eval-json-${{ matrix.model }}
          path: out/
          retention-days: 30

  # --------------------------------------------------------------------
  # Aggregate: pulls every matrix artifact and renders one HTML page
  # with N model columns. Always runs (even if one matrix job failed)
  # so we still get a partial comparison.
  # --------------------------------------------------------------------
  compare:
    needs: evals
    if: always() && needs.plan.result == 'success'
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@v4

      - name: Set up Python
        uses: actions/setup-python@v5
        with:
          python-version: '3.11'

      - name: Download all matrix reports
        uses: actions/download-artifact@v4
        with:
          pattern: eval-json-*
          path: artifacts/
          merge-multiple: true

      - name: List downloaded reports
        run: ls -la artifacts/ || true

      - name: Render comparison HTML
        run: |
          shopt -s nullglob
          JSONS=(artifacts/*.json)
          if [ ${#JSONS[@]} -eq 0 ]; then
            echo "No JSON reports downloaded; skipping render"
            exit 0
          fi
          python3 scripts/render_evals_html.py \
            "${JSONS[@]}" \
            -o comparison.html \
            --title "Conductor Skill — Multi-Model Comparison"
          ls -la comparison.html

      - name: Upload comparison HTML
        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: eval-comparison
          path: |
            comparison.html
            artifacts/
          retention-days: 30

      - name: Comment on PR with combined summary
        if: github.event_name == 'pull_request' && always()
        uses: actions/github-script@v7
        with:
          script: |
            const fs = require('fs');
            const path = require('path');
            const dir = 'artifacts';
            if (!fs.existsSync(dir)) {
              await github.rest.issues.createComment({
                issue_number: context.issue.number,
                owner: context.repo.owner,
                repo: context.repo.repo,
                body: '## Multi-model eval results\n\n:x: No reports produced — check matrix logs.'
              });
              return;
            }
            const files = fs.readdirSync(dir).filter(f => f.endsWith('.json'));
            const rows = [];
            for (const f of files) {
              try {
                const r = JSON.parse(fs.readFileSync(path.join(dir, f)));
                const results = r.results || [];
                const passed = results.filter(x => x.overall_pass).length;
                const total = results.length;
                const critPass = results.reduce((s,x) => s + (x.passed||0), 0);
                const critTotal = results.reduce((s,x) => s + (x.total||0), 0);
                const pct = critTotal ? (critPass/critTotal*100).toFixed(1) : '0.0';
                const allPassed = passed === total;
                const fails = results.filter(x => !x.overall_pass).map(x => x.name);
                rows.push({
                  model: r.model || f,
                  passed, total,
                  critPass, critTotal,
                  pct, allPassed,
                  fails
                });
              } catch (e) {
                rows.push({ model: f, error: e.message });
              }
            }
            rows.sort((a, b) => (a.model || '').localeCompare(b.model || ''));

            let body = `## Multi-model eval results\n\n`;
            body += `| Model | Evals | Criteria | Status |\n|---|---|---|---|\n`;
            for (const r of rows) {
              if (r.error) {
                body += `| \`${r.model}\` | — | — | :x: ${r.error} |\n`;
                continue;
              }
              const icon = r.allPassed ? ':white_check_mark:' : ':warning:';
              body += `| \`${r.model}\` | ${r.passed}/${r.total} | ${r.critPass}/${r.critTotal} (${r.pct}%) | ${icon} |\n`;
            }
            const anyFails = rows.some(r => r.fails && r.fails.length);
            if (anyFails) {
              body += `\n### Failed scenarios\n\n`;
              for (const r of rows) {
                if (r.fails && r.fails.length) {
                  body += `**\`${r.model}\`:** ${r.fails.map(n => `\`${n}\``).join(', ')}\n\n`;
                }
              }
            }
            body += `\nDownload the \`eval-comparison\` artifact for the side-by-side HTML report.\n`;

            await github.rest.issues.createComment({
              issue_number: context.issue.number,
              owner: context.repo.owner,
              repo: context.repo.repo,
              body
            });

SHA-256: f5b66bc857f18436d532dc2ea4eabe910135e1c53d1f572860bc0d93d4a06255