← Files Compound EngineeringARCHIVED FILE

skills/ce-optimize/references/example-expensive-benchmark-spec.yaml

2.08 KB · Oct 4, 2026 · 12:33 UTC

↓ Download file

# Expensive-benchmark template (test-suite wall time, CI critical path, runner-minutes).
# Use this shape when each evaluation costs minutes and "better" is more than one hard target.
# Existing single-primary specs stay valid; this file is an opt-in example, not a new default.

name: reduce-test-suite-wall-time
description: >
  Reduce local full-suite wall time without raising the CI critical path
  or aggregate runner-minutes. A change that helps only CI is eligible
  if it does not regress the other required targets.

metric:
  primary:
    type: hard
    name: local_wall_seconds
    direction: minimize
    target: 300
  objectives:
    - name: local_wall_seconds
      direction: minimize
      role: required
      target: 300
    - name: ci_critical_path_seconds
      direction: minimize
      role: required
      target: 90
    - name: runner_minutes
      direction: minimize
      role: required
      target: 40
  degenerate_gates:
    - name: suite_passed
      check: "== 1"
      description: The full suite must stay green
  diagnostics:
    - name: python_group_seconds
    - name: go_group_seconds

measurement:
  command: "python tools/eval/measure_suite.py"
  timeout_seconds: 1200
  working_directory: "."
  stability:
    mode: ladder
    repeat_count: 5
    aggregation: median
    noise_threshold: 10
    comparison:
      method: relative
      relative_threshold: 0.05
    ladder:
      smoke_command: "python tools/eval/measure_suite.py --smoke"
      exploratory_pairs: 1
      confirmation_repeats: 5
      futility:
        worse_factor: 1.2

scope:
  mutable:
    - "Makefile"
    - "scripts/test/"
    - ".github/workflows/"
  immutable:
    - "tools/eval/measure_suite.py"
    - "tests/fixtures/"

execution:
  mode: serial
  backend: worktree
  max_concurrent: 1

parallel:
  port_strategy: none
  shared_files: []

dependencies:
  approved: []

constraints:
  - "Do not skip required tests to win a timing metric"
  - "Do not change the measurement harness"

stopping:
  max_iterations: 12
  max_hours: 8
  plateau_iterations: 6
  target_reached: true

max_runner_up_merges_per_batch: 0

SHA-256: 826bc6a89fd72da1c129bcbff7039e05ca16cc8d1942ec9afd86e9f74b62e806