# Example benchmark config — minimal but complete. # Copy and edit for real runs. Pre-registration is REQUIRED; the path must # exist and be committed to git before the run starts. # # Run with: # uv run python -m tests.benchmarks._framework.cli validate tests/benchmarks/cloudopsbench/configs/cloudopsbench_smoke.yml # uv run python -m tests.benchmarks._framework.cli run tests/benchmarks/cloudopsbench/configs/cloudopsbench_smoke.yml --dev # # Remove --dev for production runs (enables IntegrityGuard). benchmark: cloudopsbench modes: - opensre+llm llms: - claude-default # use whatever opensre is configured for; llm_dispatch.py # will let this be a specific provider once it ships # Pinned model versions — refused at runtime if a model resolves differently. # For real runs, replace with real provider snapshots: # claude-4-sonnet: claude-sonnet-4-5-20250929 # gpt-5: gpt-5-2025-08-07 # deepseek-v3.2: deepseek-chat-v3.2 model_versions: claude-default: "(opensre-default)" # Replication — required ≥3 for stochastic LLMs (Box-Hunter-Hunter Ch 3.4) runs_per_case: 3 # Parallelism — sized for laptop; bump on AWS workers: 4 # Hard cost cap — framework aborts cleanly when exceeded cost_budget_usd: 50.0 # Seeded random case selection — Mechanism 6 (no cherry-picking) seed: 42 # Where artifacts (report.json + per-case JSON) land output_dir: .bench-results/example/ # Pre-registration — Phase 0 integrity gate. Path must exist + be non-empty # AND be committed to git before the run. See: # ~/DevBox/opensre-notes/opensre-benchmark-framework.md § 0 pre_registration_path: tests/benchmarks/cloudopsbench/configs/preregistrations/cloudopsbench_smoke.yml filters: # Filter to a small slice for first runs. Remove these for the full grid. limit: 5 seen_shape: [true] # Leave empty to include all systems / categories systems: [] fault_categories: [] report_formats: - json - markdown