# Cloud-OpsBench v1 — DeepSeek slice (deepseek-v3.2). # # Provider slice of cloudopsbench_v1.yml. Use this when running just the # DeepSeek model (no ANTHROPIC_API_KEY / OPENAI_API_KEY required). The # full grid (all 4 paper models in one run) is still # cloudopsbench_v1.yml — keep that for the publication-grade comparison # once all provider credits are available. # # Sibling slices: # - cloudopsbench_v1_anthropic.yml (claude-4-sonnet) # - cloudopsbench_v1_openai.yml (gpt-4o + gpt-5) # # Required env at run time: DEEPSEEK_API_KEY. The llm_dispatch LLMSpec # routes DeepSeek through opensre's OpenAI-compatible client with the # DeepSeek base_url; no separate provider integration required. # # Run with --dev first to verify the chain, then drop --dev for production: # uv run python -m tests.benchmarks._framework.cli run \ # tests/benchmarks/cloudopsbench/configs/cloudopsbench_v1_deepseek.yml --dev benchmark: cloudopsbench modes: - opensre+llm llms: - deepseek-v3.2 model_versions: deepseek-v3.2: deepseek-chat-v3.2 runs_per_case: 3 workers: 4 cost_budget_usd: 500.0 seed: 42 output_dir: .bench-results/cloudopsbench_v1_deepseek/ pre_registration_path: tests/benchmarks/cloudopsbench/configs/preregistrations/cloudopsbench_v1.yml filters: limit: 30 seen_shape: [true] systems: [] fault_categories: [] report_formats: - json - markdown - html