# Cloud-OpsBench v1 — Anthropic slice (claude-4-sonnet). # # Provider slice of cloudopsbench_v1.yml. Use this when running just the # Anthropic model (no OPENAI_API_KEY / DEEPSEEK_API_KEY required). The # full grid (all 4 paper models in one run) is still # cloudopsbench_v1.yml — keep that for the publication-grade comparison # once all provider credits are available. # # Sibling slices: # - cloudopsbench_v1_openai.yml (gpt-4o + gpt-5) # - cloudopsbench_v1_deepseek.yml (deepseek-v3.2) # # Required env at run time: ANTHROPIC_API_KEY. # # Run with --dev first to verify the chain, then drop --dev for production: # uv run python -m tests.benchmarks._framework.cli run \ # tests/benchmarks/cloudopsbench/configs/cloudopsbench_v1_anthropic.yml --dev benchmark: cloudopsbench modes: - opensre+llm llms: - claude-4-sonnet model_versions: claude-4-sonnet: claude-sonnet-4-5-20250929 runs_per_case: 3 workers: 4 cost_budget_usd: 500.0 seed: 42 output_dir: .bench-results/cloudopsbench_v1_anthropic/ pre_registration_path: tests/benchmarks/cloudopsbench/configs/preregistrations/cloudopsbench_v1.yml filters: limit: 30 seen_shape: [true] systems: [] fault_categories: [] report_formats: - json - markdown - html